From daa09f6cd2cd18902ccc6dff3bb615e44bfeb1ea Mon Sep 17 00:00:00 2001 From: Alexandre Mutel Date: Thu, 25 Feb 2016 22:10:52 +0900 Subject: [PATCH] Rewrite BlockParser, simplfy code. --- src/Testamina.Markdig.Benchmarks/Program.cs | 18 +- src/Textamina.Markdig.Tests/TestParser.cs | 6 +- .../Formatters/HtmlFormatter.cs | 1 + src/Textamina.Markdig/Helpers/HtmlHelper.cs | 1 - src/Textamina.Markdig/Helpers/LinkHelper.cs | 22 +- src/Textamina.Markdig/Parsers/BlockParser.cs | 36 ++ .../Parsers/BlockParserState.cs | 483 +++++++++++++++ src/Textamina.Markdig/Parsers/BlockState.cs | 40 ++ .../Parsers/CodeBlockParser.cs | 35 ++ .../Parsers/FencedCodeBlockParser.cs | 143 +++++ .../Parsers/HeadingBlockParser.cs | 108 ++++ .../Parsers/HtmlBlockParser.cs | 286 +++++++++ src/Textamina.Markdig/Parsers/IBlockParser.cs | 20 + src/Textamina.Markdig/Parsers/InlineParser.cs | 9 + .../{Parsing => Parsers}/InlineParserState.cs | 4 +- .../Parsers/Inlines/AutolineInlineParser.cs | 37 ++ .../Parsers/Inlines/CodeInlineParser.cs | 98 +++ .../Parsers/Inlines/EmphasisInlineParser.cs | 111 ++++ .../Parsers/Inlines/EscapeInlineParser.cs | 31 + .../Inlines/HardlineBreakInlineParser.cs | 46 ++ .../Parsers/Inlines/LinkInlineParser.cs | 254 ++++++++ .../Parsers/Inlines/LiteralInlineParser.cs | 32 + .../Parsers/ListBlockParser.cs | 398 +++++++++++++ .../Parsers/MarkdownParser.cs | 229 ++++++++ .../Parsers/ParagraphBlockParser.cs | 166 ++++++ src/Textamina.Markdig/Parsers/ParserList.cs | 72 +++ .../Parsers/QuoteBlockParser.cs | 60 ++ .../Parsers/ThematicBreakParser.cs | 77 +++ src/Textamina.Markdig/Parsing/BlockParser.cs | 24 - .../Parsing/BlockParserState.cs | 176 ------ src/Textamina.Markdig/Parsing/InlineParser.cs | 9 - .../Parsing/MarkdownParser.cs | 556 ------------------ .../Parsing/MatchLineResult.cs | 17 - .../Parsing/NewBlockParserState.cs | 119 ---- .../Syntax/BlankLineBlock.cs | 4 +- src/Textamina.Markdig/Syntax/Block.cs | 5 +- src/Textamina.Markdig/Syntax/CodeBlock.cs | 27 +- .../Syntax/ContainerBlock.cs | 5 +- .../Syntax/FencedCodeBlock.cs | 170 +----- src/Textamina.Markdig/Syntax/HeadingBlock.cs | 98 +-- src/Textamina.Markdig/Syntax/HtmlBlock.cs | 298 +--------- .../Syntax/Inlines/AutoLinkInline.cs | 40 +- .../Syntax/Inlines/CodeInline.cs | 101 +--- .../Syntax/Inlines/ContainerInline.cs | 3 +- .../Syntax/Inlines/DelimiterInline.cs | 5 +- .../Syntax/Inlines/DelimiterType.cs | 2 +- .../Syntax/Inlines/EmphasisDelimiterInline.cs | 5 +- .../Syntax/Inlines/EmphasisInline.cs | 116 +--- .../Syntax/Inlines/EscapeInline.cs | 40 -- .../Syntax/Inlines/HardlineBreakInline.cs | 46 +- .../Syntax/Inlines/HtmlInline.cs | 2 +- .../Syntax/Inlines/Inline.cs | 6 +- .../Syntax/Inlines/InlineStack.cs | 319 ---------- .../Syntax/Inlines/LeafInline.cs | 4 +- .../Syntax/Inlines/LineBreakInline.cs | 2 +- .../Syntax/Inlines/LinkDelimiterInline.cs | 4 +- .../Syntax/Inlines/LinkInline.cs | 256 +------- .../Syntax/Inlines/LiteralInline.cs | 30 +- .../Syntax/Inlines/RawHtmlInline.cs | 3 +- src/Textamina.Markdig/Syntax/LeafBlock.cs | 18 +- .../Syntax/LinkReferenceDefinitionBlock.cs | 3 +- src/Textamina.Markdig/Syntax/ListBlock.cs | 363 +----------- src/Textamina.Markdig/Syntax/ListItemBlock.cs | 2 +- .../Syntax/ParagraphBlock.cs | 191 +----- src/Textamina.Markdig/Syntax/QuoteBlock.cs | 43 +- src/Textamina.Markdig/Syntax/StringSlice.cs | 91 +-- .../Syntax/StringSliceList.cs | 12 +- .../Syntax/ThematicBreakBlock.cs | 79 +-- .../Textamina.Markdig.csproj | 33 +- .../Textamina.Markdig.csproj.DotSettings | 2 +- 70 files changed, 2955 insertions(+), 3197 deletions(-) create mode 100644 src/Textamina.Markdig/Parsers/BlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/BlockParserState.cs create mode 100644 src/Textamina.Markdig/Parsers/BlockState.cs create mode 100644 src/Textamina.Markdig/Parsers/CodeBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/FencedCodeBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/HeadingBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/HtmlBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/IBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/InlineParser.cs rename src/Textamina.Markdig/{Parsing => Parsers}/InlineParserState.cs (89%) create mode 100644 src/Textamina.Markdig/Parsers/Inlines/AutolineInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/CodeInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/EmphasisInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/EscapeInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/HardlineBreakInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/LinkInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/Inlines/LiteralInlineParser.cs create mode 100644 src/Textamina.Markdig/Parsers/ListBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/MarkdownParser.cs create mode 100644 src/Textamina.Markdig/Parsers/ParagraphBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/ParserList.cs create mode 100644 src/Textamina.Markdig/Parsers/QuoteBlockParser.cs create mode 100644 src/Textamina.Markdig/Parsers/ThematicBreakParser.cs delete mode 100644 src/Textamina.Markdig/Parsing/BlockParser.cs delete mode 100644 src/Textamina.Markdig/Parsing/BlockParserState.cs delete mode 100644 src/Textamina.Markdig/Parsing/InlineParser.cs delete mode 100644 src/Textamina.Markdig/Parsing/MarkdownParser.cs delete mode 100644 src/Textamina.Markdig/Parsing/MatchLineResult.cs delete mode 100644 src/Textamina.Markdig/Parsing/NewBlockParserState.cs delete mode 100644 src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs delete mode 100644 src/Textamina.Markdig/Syntax/Inlines/InlineStack.cs diff --git a/src/Testamina.Markdig.Benchmarks/Program.cs b/src/Testamina.Markdig.Benchmarks/Program.cs index 2f214b37..c2d0cbee 100644 --- a/src/Testamina.Markdig.Benchmarks/Program.cs +++ b/src/Testamina.Markdig.Benchmarks/Program.cs @@ -8,7 +8,7 @@ using System.Threading.Tasks; using BenchmarkDotNet.Attributes; using BenchmarkDotNet.Running; using Textamina.Markdig.Formatters; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Testamina.Markdig.Benchmarks { @@ -46,15 +46,15 @@ namespace Testamina.Markdig.Benchmarks static void Main(string[] args) { - var clock = Stopwatch.StartNew(); + //var clock = Stopwatch.StartNew(); var program = new Program(); - for (int i = 0; i < 200; i++) - { - //program.TestMarkdig(); - program.TestCommonMark(); - } - Console.WriteLine($"time: {clock.ElapsedMilliseconds}ms"); - DumpGC(); + //for (int i = 0; i < 200; i++) + //{ + program.TestMarkdig(); + // //program.TestCommonMark(); + //} + //Console.WriteLine($"time: {clock.ElapsedMilliseconds}ms"); + //DumpGC(); //BenchmarkRunner.Run(); } diff --git a/src/Textamina.Markdig.Tests/TestParser.cs b/src/Textamina.Markdig.Tests/TestParser.cs index 7380cee8..975974d1 100644 --- a/src/Textamina.Markdig.Tests/TestParser.cs +++ b/src/Textamina.Markdig.Tests/TestParser.cs @@ -9,7 +9,7 @@ using System.Text; using System.Text.RegularExpressions; using NUnit.Framework; using Textamina.Markdig.Formatters; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Tests { @@ -24,7 +24,9 @@ namespace Textamina.Markdig.Tests [Test] public void TestSimple() { - var reader = new StringReader(@"foo"); + var reader = new StringReader(@">> foo +test + "); // var reader = new StringReader(@"> > toto tata //> titi toto //"); diff --git a/src/Textamina.Markdig/Formatters/HtmlFormatter.cs b/src/Textamina.Markdig/Formatters/HtmlFormatter.cs index 88d7c916..fe5fdebd 100644 --- a/src/Textamina.Markdig/Formatters/HtmlFormatter.cs +++ b/src/Textamina.Markdig/Formatters/HtmlFormatter.cs @@ -4,6 +4,7 @@ using System.Globalization; using System.IO; using Textamina.Markdig.Helpers; using Textamina.Markdig.Syntax; +using Textamina.Markdig.Syntax.Inlines; namespace Textamina.Markdig.Formatters { diff --git a/src/Textamina.Markdig/Helpers/HtmlHelper.cs b/src/Textamina.Markdig/Helpers/HtmlHelper.cs index 84d079e9..55239212 100644 --- a/src/Textamina.Markdig/Helpers/HtmlHelper.cs +++ b/src/Textamina.Markdig/Helpers/HtmlHelper.cs @@ -1,5 +1,4 @@ using System; -using System.Collections.Generic; using System.Text; using Textamina.Markdig.Formatters; using Textamina.Markdig.Syntax; diff --git a/src/Textamina.Markdig/Helpers/LinkHelper.cs b/src/Textamina.Markdig/Helpers/LinkHelper.cs index c879b772..d9a35058 100644 --- a/src/Textamina.Markdig/Helpers/LinkHelper.cs +++ b/src/Textamina.Markdig/Helpers/LinkHelper.cs @@ -274,12 +274,12 @@ namespace Textamina.Markdig.Helpers return isValid; } - public static bool TryParseTitle(StringSlice text, out string title) + public static bool TryParseTitle(T text, out string title) where T : ICharIterator { return TryParseTitle(ref text, out title); } - public static bool TryParseTitle(ref StringSlice text, out string title) + public static bool TryParseTitle(ref T text, out string title) where T : ICharIterator { bool isValid = false; var buffer = StringBuilderCache.Local(); @@ -344,12 +344,12 @@ namespace Textamina.Markdig.Helpers return isValid; } - public static bool TryParseUrl(StringSlice text, out string link) + public static bool TryParseUrl(T text, out string link) where T : ICharIterator { return TryParseUrl(ref text, out link); } - public static bool TryParseUrl(ref StringSlice text, out string link) + public static bool TryParseUrl(ref T text, out string link) where T : ICharIterator { bool isValid = false; var buffer = StringBuilderCache.Local(); @@ -467,14 +467,14 @@ namespace Textamina.Markdig.Helpers return isValid; } - public static bool TryParseLinkReferenceDefinition(StringSlice text, out string label, out string url, - out string title) + public static bool TryParseLinkReferenceDefinition(T text, out string label, out string url, + out string title) where T : ICharIterator { return TryParseLinkReferenceDefinition(ref text, out label, out url, out title); } - public static bool TryParseLinkReferenceDefinition(ref StringSlice text, out string label, out string url, - out string title) + public static bool TryParseLinkReferenceDefinition(ref T text, out string label, out string url, + out string title) where T : ICharIterator { url = null; title = null; @@ -551,17 +551,17 @@ namespace Textamina.Markdig.Helpers return true; } - public static bool TryParseLabel(StringSlice lines, out string label) + public static bool TryParseLabel(T lines, out string label) where T : ICharIterator { return TryParseLabel(ref lines, false, out label); } - public static bool TryParseLabel(ref StringSlice lines, out string label) + public static bool TryParseLabel(ref T lines, out string label) where T : ICharIterator { return TryParseLabel(ref lines, false, out label); } - public static bool TryParseLabel(ref StringSlice lines, bool allowEmpty, out string label) + public static bool TryParseLabel(ref T lines, bool allowEmpty, out string label) where T : ICharIterator { label = null; char c = lines.CurrentChar; diff --git a/src/Textamina.Markdig/Parsers/BlockParser.cs b/src/Textamina.Markdig/Parsers/BlockParser.cs new file mode 100644 index 00000000..8f6ea2c7 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/BlockParser.cs @@ -0,0 +1,36 @@ +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public abstract class BlockParser : IBlockParser + { + protected BlockParser() + { + } + + public char[] OpeningCharacters { get; protected set; } + + public virtual bool CanInterrupt(BlockParserState state, Block block) + { + // By default, all blocks can interrupt a paragraph except: + // - setext heading + // - indented code block + // - a special HTML blocks + return true; + } + + public abstract BlockState TryOpen(BlockParserState state); + + public virtual BlockState TryContinue(BlockParserState state, Block block) + { + // By default we don't expect any newline + return BlockState.None; + } + + public virtual bool Close(BlockParserState state, Block block) + { + // By default keep the block + return true; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/BlockParserState.cs b/src/Textamina.Markdig/Parsers/BlockParserState.cs new file mode 100644 index 00000000..7a9283cf --- /dev/null +++ b/src/Textamina.Markdig/Parsers/BlockParserState.cs @@ -0,0 +1,483 @@ +using System; +using System.Collections.Generic; +using System.Diagnostics; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class BlockParserState + { + private readonly ParserList blockParsers; + private int currentStackIndex; + + public BlockParserState(StringBuilderCache stringBuilders, Document root) + { + if (stringBuilders == null) throw new ArgumentNullException(nameof(stringBuilders)); + if (root == null) throw new ArgumentNullException(nameof(root)); + StringBuilders = stringBuilders; + Root = root; + NewBlocks = new Stack(); + root.IsOpen = true; + Stack = new List {root}; + + blockParsers = new ParserList() + { + new ThematicBreakParser(), + new HeadingBlockParser(), + new QuoteBlockParser(), + //ListBlock.Parser, + + new HtmlBlockParser(), + new CodeBlockParser(), + new FencedCodeBlockParser(), + new ParagraphBlockParser(), + }; + blockParsers.Initialize(); + } + + public List Stack { get; } + + public Stack NewBlocks { get; } + + public ContainerBlock CurrentContainer { get; private set; } + + public Block LastBlock { get; private set; } + + public Block NextContinue => currentStackIndex + 1 < Stack.Count ? Stack[currentStackIndex + 1] : null; + + public Document Root { get; } + + public bool ContinueProcessingLine { get; set; } + + public StringSlice Line; + + public int LineIndex { get; private set; } + + public bool IsBlankLine => CurrentChar == '\0'; + + public bool IsEndOfLine => Line.IsEndOfSlice; + + public char CurrentChar => Line.CurrentChar; + + public char NextChar() + { + var c = Line.CurrentChar; + if (c == '\t') + { + Column = ((Column + 3) >> 2) << 2; + } + else + { + Column++; + } + return Line.NextChar(); + } + + public char CharAt(int index) => Line[index]; + + public int Start => Line.Start; + + public int EndOffset => Line.End; + + public int Indent => Column - ColumnBegin; + + public bool IsCodeIndent => Indent >= 4; + + public int ColumnBegin { get; private set; } + + public int Column { get; set; } + + public StringBuilderCache StringBuilders { get; } + + public char PeekChar(int offset) + { + return Line.PeekChar(offset); + } + + public void ParseIndent() + { + var c = CurrentChar; + ColumnBegin = Column; + while (c !='\0') + { + if (c == ' ') + { + Column++; + } + else if (c == '\t') + { + Column = ((Column + 3) >> 2) << 2; + } + else + { + break; + } + c = NextChar(); + } + } + + public void Close(Block block) + { + // If we close a block, we close all blocks above + for (int i = Stack.Count - 1; i >= 1; i--) + { + if (Stack[i] == block) + { + for (int j = Stack.Count - 1; j >= i; j--) + { + Close(j); + } + break; + } + } + } + + public void Discard(Block block) + { + for (int i = Stack.Count - 1; i >= 1; i--) + { + if (Stack[i] == block) + { + block.Parent.Children.Remove(block); + Stack.RemoveAt(i); + break; + } + } + } + + public void Close(int index) + { + var block = Stack[index]; + // If the pending object is removed, we need to remove it from the parent container + if (!block.Parser.Close(this, block)) + { + block.Parent?.Children.Remove(block); + } + Stack.RemoveAt(index); + } + + public void CloseAll(bool force) + { + // Close any previous blocks not opened + for (int i = Stack.Count - 1; i >= 1; i--) + { + var block = Stack[i]; + + // Stop on the first open block + if (!force && block.IsOpen) + { + break; + } + Close(i); + } + } + + public void ProcessLine(string newLine) + { + ContinueProcessingLine = true; + + Line = new StringSlice(newLine); + ParseIndent(); + + LineIndex++; + + TryContinueBlocks(); + + // If we have already reached eol and the last block was a paragraph + // we close it + if (Line.IsEndOfSlice) + { + int index = Stack.Count - 1; + if (Stack[index] is ParagraphBlock) + { + Close(index); + return; + } + } + + // If the line was not entirely processed by pending blocks, try to process it with any new block + TryOpenBlocks(); + + // Close blocks that are no longer opened + CloseAll(false); + } + + private void OpenAll() + { + for (int i = 1; i < Stack.Count; i++) + { + Stack[i].IsOpen = true; + } + } + + internal void UpdateLast(int stackIndex) + { + currentStackIndex = stackIndex < 0 ? Stack.Count - 1 : stackIndex; + LastBlock = null; + for (int i = Stack.Count - 1; i >= 0; i--) + { + var block = Stack[i]; + if (LastBlock == null) + { + LastBlock = block; + } + + var container = block as ContainerBlock; + if (container != null) + { + CurrentContainer = container; + break; + } + } + } + + private void TryContinueBlocks() + { + // Set all blocks non opened. + // They will be marked as open in the following loop + for (int i = 1; i < Stack.Count; i++) + { + Stack[i].IsOpen = false; + } + + // Process any current block potentially opened + for (int i = 1; i < Stack.Count; i++) + { + var block = Stack[i]; + + // If we have a paragraph block, we want to try to match other blocks before trying the Paragraph + if (block is ParagraphBlock) + { + break; + } + + // Else tries to match the Parser with the current line + var parser = block.Parser; + + // If we have a discard, we can remove it from the current state + UpdateLast(i); + var result = parser.TryContinue(this, block); + if (result == BlockState.Skip) + { + continue; + } + + if (result == BlockState.None) + { + break; + } + + // In case the BlockParser has modified the blockParserState we are iterating on + if (i >= Stack.Count) + { + i = Stack.Count - 1; + } + + // If a parser is adding a block, it must be the last of the list + if ((i + 1) < Stack.Count && NewBlocks.Count > 0) + { + throw new InvalidOperationException("A pending parser cannot add a new block when it is not the last pending block"); + } + + // If we have a leaf block + var leaf = block as LeafBlock; + if (leaf != null && NewBlocks.Count == 0) + { + ContinueProcessingLine = false; + if (!result.IsDiscard()) + { + leaf.AppendLine(ref Line); + } + + if (NewBlocks.Count > 0) + { + throw new InvalidOperationException( + "The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed"); + } + } + + // A block is open only if it has a Continue state. + // otherwise it is a Break state, and we don't keep it opened + block.IsOpen = result == BlockState.Continue || result == BlockState.ContinueDiscard; + + if (result == BlockState.BreakDiscard) + { + ContinueProcessingLine = false; + break; + } + + bool isLast = i == Stack.Count - 1; + if (ContinueProcessingLine) + { + ProcessNewBlocks(result, false); + } + if (isLast || !ContinueProcessingLine) + { + break; + } + } + } + + private void TryOpenBlocks() + { + while (ContinueProcessingLine) + { + // Eat indent spaces before checking the character + ParseIndent(); + + var parsers = blockParsers.GetParsersForOpeningCharacter(CurrentChar); + var globalParsers = blockParsers.GlobalParsers; + + if (parsers != null) + { + if (TryOpenBlocks(parsers)) + { + continue; + } + } + + if (globalParsers != null && ContinueProcessingLine) + { + if (TryOpenBlocks(globalParsers)) + { + continue; + } + } + + break; + } + } + + private bool TryOpenBlocks(BlockParser[] parsers) + { + for (int j = 0; j < parsers.Length; j++) + { + var blockParser = parsers[j]; + if (Line.IsEndOfSlice) + { + ContinueProcessingLine = false; + break; + } + + // UpdateLast the state of LastBlock and LastContainer + UpdateLast(-1); + + // If a block parser cannot interrupt a paragraph, and the last block is a paragraph + // we can skip this parser + + var lastBlock = LastBlock; + if (!blockParser.CanInterrupt(this, lastBlock)) + { + continue; + } + + bool isLazyParagraph = blockParser is ParagraphBlockParser && lastBlock is ParagraphBlock; + + var result = isLazyParagraph + ? blockParser.TryContinue(this, lastBlock) + : blockParser.TryOpen(this); + + if (result == BlockState.None) + { + // If we have reached a blank line after trying to parse a paragraph + // we can ignore it + if (isLazyParagraph && IsBlankLine) + { + ContinueProcessingLine = false; + break; + } + continue; + } + + // Special case for paragraph + UpdateLast(-1); + + var paragraph = LastBlock as ParagraphBlock; + if (isLazyParagraph && paragraph != null) + { + Debug.Assert(NewBlocks.Count == 0); + + if (!result.IsDiscard()) + { + paragraph.AppendLine(ref Line); + } + + // We have just found a lazy continuation for a paragraph, early exit + // Mark all block opened after a lazy continuation + OpenAll(); + + ContinueProcessingLine = false; + break; + } + + // Nothing found but the BlockParser may instruct to break, so early exit + if (NewBlocks.Count == 0 && result == BlockState.BreakDiscard) + { + ContinueProcessingLine = false; + break; + } + + // If we have a container, we can retry to match against all types of block. + ProcessNewBlocks(result, true); + return ContinueProcessingLine; + + // We have a leaf node, we can stop + } + return false; + } + + private void ProcessNewBlocks(BlockState result, bool allowClosing) + { + var newBlocks = NewBlocks; + while (newBlocks.Count > 0) + { + var block = newBlocks.Pop(); + + block.Line = LineIndex; + + // If we have a leaf block + var leaf = block as LeafBlock; + if (leaf != null) + { + if (!result.IsDiscard()) + { + leaf.AppendLine(ref Line); + } + + if (newBlocks.Count > 0) + { + throw new InvalidOperationException( + "The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed"); + } + } + + if (allowClosing) + { + // Close any previous blocks not opened + CloseAll(false); + } + + // If previous block is a container, add the new block as a children of the previous block + if (block.Parent == null) + { + CurrentContainer.Children.Add(block); + block.Parent = CurrentContainer; + } + + block.IsOpen = result.IsContinue(); + + // Add a block blockParserState to the stack (and leave it opened) + Stack.Add(block); + + if (leaf != null) + { + ContinueProcessingLine = false; + return; + } + } + ContinueProcessingLine = true; + } + + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/BlockState.cs b/src/Textamina.Markdig/Parsers/BlockState.cs new file mode 100644 index 00000000..ad5a71e9 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/BlockState.cs @@ -0,0 +1,40 @@ +using System.Runtime.CompilerServices; + +namespace Textamina.Markdig.Parsers +{ + public enum BlockState + { + None, + + Skip, + + Continue, + + ContinueDiscard, + + Break, + + BreakDiscard + } + + public static class BlockStateExtensions + { + [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] + public static bool IsDiscard(this BlockState blockState) + { + return blockState == BlockState.ContinueDiscard || blockState == BlockState.BreakDiscard; + } + + [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] + public static bool IsContinue(this BlockState blockState) + { + return blockState == BlockState.Continue || blockState == BlockState.ContinueDiscard; + } + + [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] + public static bool IsBreak(this BlockState blockState) + { + return blockState == BlockState.Break || blockState == BlockState.BreakDiscard; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/CodeBlockParser.cs b/src/Textamina.Markdig/Parsers/CodeBlockParser.cs new file mode 100644 index 00000000..cc202deb --- /dev/null +++ b/src/Textamina.Markdig/Parsers/CodeBlockParser.cs @@ -0,0 +1,35 @@ +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class CodeBlockParser : BlockParser + { + public CodeBlockParser() + { + } + + public override bool CanInterrupt(BlockParserState state, Block block) + { + return !(block is ParagraphBlock); + } + + public override BlockState TryOpen(BlockParserState state) + { + var result = TryContinue(state, null); + if (result == BlockState.Continue) + { + state.NewBlocks.Push(new CodeBlock(this) { Column = state.Column }); + } + return result; + } + + public override BlockState TryContinue(BlockParserState state, Block block) + { + if (!state.IsCodeIndent || state.IsBlankLine) + { + return state.IsBlankLine && block != null ? BlockState.BreakDiscard : BlockState.None; + } + return BlockState.Continue; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/FencedCodeBlockParser.cs b/src/Textamina.Markdig/Parsers/FencedCodeBlockParser.cs new file mode 100644 index 00000000..1d07962a --- /dev/null +++ b/src/Textamina.Markdig/Parsers/FencedCodeBlockParser.cs @@ -0,0 +1,143 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class FencedCodeBlockParser : BlockParser + { + public FencedCodeBlockParser() + { + OpeningCharacters = new[] {'`', '~'}; + } + + public override BlockState TryOpen(BlockParserState state) + { + // Else if the we have an indent, it is not valid + if (state.IsCodeIndent) + { + return BlockState.None; + } + + int count = 0; + var line = state.Line; + char c = line.CurrentChar; + var matchChar = c; + while (c != '\0') + { + if (c != matchChar) + { + break; + } + count++; + c = line.NextChar(); + } + + if (count < 3) + { + return BlockState.None; + } + + // TODO: We need to count the number of leading space to remove them on each line + var column = state.Column; + + // specs spaces: Is space and tabs? or only spaces? Use space and tab for this case + while (c.IsSpaceOrTab()) + { + c = line.NextChar(); + } + string infoString; + string argString = null; + + // An info string cannot contain any backsticks + int firstSpace = -1; + for (int i = line.Start; i <= line.End; i++) + { + c = line.Text[i]; + if (c == '`') + { + return BlockState.None; + } + + if (firstSpace < 0 && c.IsSpaceOrTab()) + { + firstSpace = i; + } + } + + if (firstSpace > 0) + { + infoString = line.Text.Substring(line.Start, firstSpace - line.Start); + + // Skip any spaces after info string + firstSpace++; + while (true) + { + c = line[firstSpace]; + if (c.IsSpaceOrTab()) + { + firstSpace++; + } + else + { + break; + } + } + + argString = line.Text.Substring(firstSpace, line.End - firstSpace + 1); + } + else + { + infoString = line.ToString(); + } + + // Store the number of matched string into the context + state.NewBlocks.Push(new FencedCodeBlock(this) + { + Column = column, + FencedChar = matchChar, + FencedCharCount = count, + IndentCount = state.Indent, + Language = HtmlHelper.Unescape(infoString), + Arguments = HtmlHelper.Unescape(argString), + }); + + // Discard the current line as it is already parsed + return BlockState.ContinueDiscard; + } + + public override BlockState TryContinue(BlockParserState state, Block block) + { + var fence = (FencedCodeBlock)block; + var count = fence.FencedCharCount; + var matchChar = fence.FencedChar; + var c = state.CurrentChar; + + // Work on a copy of StringSlice + var line = state.Line; + while (c == matchChar) + { + c = line.NextChar(); + count--; + } + + if (count <=0 && line.TrimEnd()) + { + // Don't keep the last line + return BlockState.BreakDiscard; + } + + // Remove any indent spaces + c = state.CurrentChar; + var indentCount = fence.IndentCount; + while (indentCount > 0 && c.IsSpace()) + { + indentCount--; + c = state.NextChar(); + } + + // TODO: It is unclear how to handle this correctly + // Break only if Eof + return BlockState.Continue; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/HeadingBlockParser.cs b/src/Textamina.Markdig/Parsers/HeadingBlockParser.cs new file mode 100644 index 00000000..e4ae7ffa --- /dev/null +++ b/src/Textamina.Markdig/Parsers/HeadingBlockParser.cs @@ -0,0 +1,108 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class HeadingBlockParser : BlockParser + { + public HeadingBlockParser() + { + OpeningCharacters = new[] {'#'}; + } + + public override BlockState TryOpen(BlockParserState state) + { + // If we are in a CodeIndent, early exit + if (state.IsCodeIndent) + { + return BlockState.None; + } + + // 4.2 ATX headings + // An ATX heading consists of a string of characters, parsed as inline content, + // between an opening sequence of 1–6 unescaped # characters and an optional + // closing sequence of any number of unescaped # characters. The opening sequence + // of # characters must be followed by a space or by the end of line. The optional + // closing sequence of #s must be preceded by a space and may be followed by spaces + // only. The opening # character may be indented 0-3 spaces. The raw contents of + // the heading are stripped of leading and trailing spaces before being parsed as + // inline content. The heading level is equal to the number of # characters in the + // opening sequence. + var column = state.Column; + var line = state.Line; + var c = line.CurrentChar; + var matchingChar = c; + + int leadingCount = 0; + while (c != '\0' && leadingCount <= 6) + { + if (c != matchingChar) + { + break; + } + c = line.NextChar(); + leadingCount++; + } + + // closing # will be handled later, because anyway we have matched + + // A space is required after leading # + if (leadingCount > 0 && leadingCount <= 6 && (c.IsSpace() || c == '\0')) + { + // Move to the content + state.Line.Start = leadingCount + 1; + state.NewBlocks.Push(new HeadingBlock(this) {Level = leadingCount, Column = column }); + + // The optional closing sequence of #s must be preceded by a space and may be followed by spaces only. + int endState = 0; + int countClosingTags = 0; + for (int i = state.Line.End; i >= state.Line.Start - 1; i--) // Go up to Start - 1 in order to match the space after the first ### + { + c = state.Line.Text[i]; + if (endState == 0) + { + if (c.IsSpace()) // TODO: Not clear if it is a space or space+tab in the specs + { + continue; + } + endState = 1; + } + if (endState == 1) + { + if (c == '#') + { + countClosingTags++; + continue; + } + + if (countClosingTags > 0) + { + if (c.IsSpace()) + { + state.Line.End = i - 1; + } + break; + } + else + { + break; + } + } + } + + // We expect a single line, so don't continue + return BlockState.Break; + } + + // Else we don't have an header + return BlockState.None; + } + + //public override bool Close(BlockParserState state, Block block) + //{ + // var heading = (HeadingBlock) block; + // heading.Lines.Trim(); + // return true; + //} + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/HtmlBlockParser.cs b/src/Textamina.Markdig/Parsers/HtmlBlockParser.cs new file mode 100644 index 00000000..f3912117 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/HtmlBlockParser.cs @@ -0,0 +1,286 @@ +using System; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class HtmlBlockParser : BlockParser + { + public HtmlBlockParser() + { + OpeningCharacters = new[] {'<'}; + } + + private static readonly string[] HtmlTags = + { + "address", // 0 + "article", // 1 + "aside", // 2 + "base", // 3 + "basefont", // 4 + "blockquote", // 5 + "body", // 6 + "caption", // 7 + "center", // 8 + "col", // 9 + "colgroup", // 10 + "dd", // 11 + "details", // 12 + "dialog", // 13 + "dir", // 14 + "div", // 15 + "dl", // 16 + "dt", // 17 + "fieldset", // 18 + "figcaption", // 19 + "figure", // 20 + "footer", // 21 + "form", // 22 + "frame", // 23 + "frameset", // 24 + "h1", // 25 + "head", // 26 + "header", // 27 + "hr", // 28 + "html", // 29 + "iframe", // 30 + "legend", // 31 + "li", // 32 + "link", // 33 + "main", // 34 + "menu", // 35 + "menuitem", // 36 + "meta", // 37 + "nav", // 38 + "noframes", // 39 + "ol", // 40 + "optgroup", // 41 + "option", // 42 + "p", // 43 + "param", // 44 + "pre", // 45 <- special group 1 + "script", // 46 <- special group 1 + "section", // 47 + "source", // 48 + "style", // 49 <- special group 1 + "summary", // 50 + "table", // 51 + "tbody", // 52 + "td", // 53 + "tfoot", // 54 + "th", // 55 + "thead", // 56 + "title", // 57 + "tr", // 58 + "track", // 59 + "ul", // 60 + }; + + public override BlockState TryOpen(BlockParserState state) + { + var result = MatchStart(state); + // An end-tag can occur on the same line + if (result == BlockState.Continue) + { + result = MatchEnd(state, (HtmlBlock) state.NewBlocks.Peek()); + } + return result; + } + + public virtual BlockState TryContinue(BlockParserState state, Block block) + { + var htmlBlock = (HtmlBlock) block; + return MatchEnd(state, htmlBlock); + } + + private BlockState MatchStart(BlockParserState state) + { + if (state.IsCodeIndent) + { + return BlockState.None; + } + + var result = TryParseTagType16(state, state.Line, state.ColumnBegin); + + // HTML blocks of type 7 cannot interrupt a paragraph: + if (result == BlockState.None && !(state.LastBlock is ParagraphBlock)) + { + result = TryParseTagType7(state, state.Line, state.ColumnBegin); + } + return result; + } + + private BlockState TryParseTagType7(BlockParserState state, StringSlice line, int startColumn) + { + var builder = StringBuilderCache.Local(); + var c = line.CurrentChar; + var result = BlockState.None; + if ((c == '/' && HtmlHelper.TryParseHtmlCloseTag(line, builder)) || HtmlHelper.TryParseHtmlTagOpenTag(line, builder)) + { + // Must be followed by whitespace only + bool hasOnlySpaces = true; + c = line.CurrentChar; + while (true) + { + if (c == '\0') + { + break; + } + if (!c.IsWhitespace()) + { + hasOnlySpaces = false; + break; + } + c = line.NextChar(); + } + + if (hasOnlySpaces) + { + result = CreateHtmlBlock(state, HtmlBlockType.NonInterruptingBlock, startColumn); + } + } + + builder.Clear(); + return result; + } + + private BlockState TryParseTagType16(BlockParserState state, StringSlice line, int startColumn) + { + char c; + c = line.CurrentChar; + if (c == '!') + { + c = line.PeekChar(1); + if (c == '-' && line.PeekChar(2) == '-') + { + return CreateHtmlBlock(state, HtmlBlockType.Comment, startColumn); // group 2 + } + if (c.IsAlphaUpper()) + { + return CreateHtmlBlock(state, HtmlBlockType.DocumentType, startColumn); // group 4 + } + if (c == '[' && line.Match("CDATA[", 3)) + { + return CreateHtmlBlock(state, HtmlBlockType.CData, startColumn); // group 5 + } + + return BlockState.None; + } + + if (c == '?') + { + return CreateHtmlBlock(state, HtmlBlockType.ProcessingInstruction, startColumn); // group 3 + } + + var hasLeadingClose = c == '/'; + if (hasLeadingClose) + { + line.NextChar(); + } + + var tag = new char[10]; + var count = 0; + for (; count < tag.Length; count++) + { + c = line.NextChar(); + if (!c.IsAlphaNumeric()) + { + break; + } + tag[count] = Char.ToLowerInvariant(c); + } + + if ( + !(c == '>' || (!hasLeadingClose && c == '/' && line.PeekChar(1) == '>') || c.IsWhitespace() || + c == '\0')) + { + return BlockState.None; + } + + if (count == 0) + { + return BlockState.None; + } + + var tagName = new string(tag, 0, count); + var tagIndex = Array.BinarySearch(HtmlTags, tagName, StringComparer.Ordinal); + if (tagIndex < 0) + { + return BlockState.None; + } + + // Cannot start with ")) + { + return BlockState.Break; + } + break; + case HtmlBlockType.CData: + if (line.Search("]]>")) + { + return BlockState.Break; + } + break; + case HtmlBlockType.ProcessingInstruction: + if (line.Search("?>")) + { + return BlockState.Break; + } + break; + case HtmlBlockType.DocumentType: + if (line.Search(">")) + { + return BlockState.Break; + } + break; + case HtmlBlockType.ScriptPreOrStyle: + // TODO: could be optimized with a dedicated parser + if (line.SearchLowercase("") || line.SearchLowercase("") || line.SearchLowercase("")) + { + return BlockState.Break; + } + break; + case HtmlBlockType.InterruptingBlock: + if (state.IsBlankLine) + { + return BlockState.BreakDiscard; + } + break; + case HtmlBlockType.NonInterruptingBlock: + if (state.IsBlankLine) + { + return BlockState.BreakDiscard; + } + break; + } + + return BlockState.Continue; + } + + private BlockState CreateHtmlBlock(BlockParserState state, HtmlBlockType type, int startColumn) + { + state.NewBlocks.Push(new HtmlBlock(this) {Column = startColumn, Type = type}); + return BlockState.Continue; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/IBlockParser.cs b/src/Textamina.Markdig/Parsers/IBlockParser.cs new file mode 100644 index 00000000..685f23c7 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/IBlockParser.cs @@ -0,0 +1,20 @@ +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public interface ICharacterParser + { + char[] OpeningCharacters { get; } + } + + public interface IBlockParser : ICharacterParser + { + bool CanInterrupt(BlockParserState state, Block block); + + BlockState TryOpen(BlockParserState state); + + BlockState TryContinue(BlockParserState state, Block block); + + bool Close(BlockParserState state, Block block); + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/InlineParser.cs b/src/Textamina.Markdig/Parsers/InlineParser.cs new file mode 100644 index 00000000..23df4b56 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/InlineParser.cs @@ -0,0 +1,9 @@ +namespace Textamina.Markdig.Parsers +{ + public abstract class InlineParser : ICharacterParser + { + public char[] OpeningCharacters { get; protected set; } + + public abstract bool Match(InlineParserState state); + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/InlineParserState.cs b/src/Textamina.Markdig/Parsers/InlineParserState.cs similarity index 89% rename from src/Textamina.Markdig/Parsing/InlineParserState.cs rename to src/Textamina.Markdig/Parsers/InlineParserState.cs index d04d382e..d773f9d2 100644 --- a/src/Textamina.Markdig/Parsing/InlineParserState.cs +++ b/src/Textamina.Markdig/Parsers/InlineParserState.cs @@ -1,9 +1,9 @@ using System.Collections.Generic; -using System.Text; using Textamina.Markdig.Helpers; using Textamina.Markdig.Syntax; +using Textamina.Markdig.Syntax.Inlines; -namespace Textamina.Markdig.Parsing +namespace Textamina.Markdig.Parsers { public class InlineParserState { diff --git a/src/Textamina.Markdig/Parsers/Inlines/AutolineInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/AutolineInlineParser.cs new file mode 100644 index 00000000..4c01d94b --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/AutolineInlineParser.cs @@ -0,0 +1,37 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class AutolineInlineParser : InlineParser + { + public AutolineInlineParser() + { + OpeningCharacters = new[] {'<'}; + } + + public override bool Match(InlineParserState state) + { + string link; + bool isEmail; + var saved = state.Text; + if (LinkHelper.TryParseAutolink(ref state.Text, out link, out isEmail)) + { + state.Inline = new AutolinkInline() {IsEmail = isEmail, Url = link}; + } + else + { + state.Text = saved; + string htmlTag; + if (!HtmlHelper.TryParseHtmlTag(ref state.Text, out htmlTag)) + { + return false; + } + + state.Inline = new HtmlInline() { Tag = htmlTag }; + } + + return true; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/CodeInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/CodeInlineParser.cs new file mode 100644 index 00000000..e2e11bb6 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/CodeInlineParser.cs @@ -0,0 +1,98 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class CodeInlineParser : InlineParser + { + public CodeInlineParser() + { + OpeningCharacters = new[] { '`' }; + } + + public override bool Match(InlineParserState state) + { + var text = state.Text; + + int openSticks = 0; + if (text.PeekChar(-1) == '`') + { + return false; + } + + while (text.CurrentChar == '`') + { + openSticks++; + text.NextChar(); + } + + bool isMatching = false; + + var builder = state.StringBuilders.Get(); + int closeSticks = 0; + var c = text.CurrentChar; + + // A backtick string is a string of one or more backtick characters (`) that is neither preceded nor followed by a backtick. + // A code span begins with a backtick string and ends with a backtick string of equal length. + // The contents of the code span are the characters between the two backtick strings, with leading and trailing spaces and line endings removed, and whitespace collapsed to single spaces. + var pc = ' '; + + while (c != '\0') + { + // Transform '\n' into a single space + if (c == '\n') + { + c = ' '; + } + + if (c != '`' && (c != ' ' || pc != ' ')) + { + builder.Append(c); + } + else + { + while (c == '`') + { + closeSticks++; + pc = c; + c = text.NextChar(); + } + + if (openSticks == closeSticks) + { + break; + } + } + + if (closeSticks > 0) + { + builder.Append('`', closeSticks); + closeSticks = 0; + } + else + { + pc = c; + c = text.NextChar(); + } + } + + if (closeSticks == openSticks) + { + // Remove trailing space + if (builder.Length > 0) + { + if (builder[builder.Length - 1].IsWhitespace()) + { + builder.Length--; + } + } + state.Inline = new CodeInline() { Content = builder.ToString() }; + isMatching = true; + } + + // Release the builder if not used + state.StringBuilders.Release(builder); + return isMatching; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/EmphasisInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/EmphasisInlineParser.cs new file mode 100644 index 00000000..43c4b6ab --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/EmphasisInlineParser.cs @@ -0,0 +1,111 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class EmphasisInlineParser : InlineParser + { + public EmphasisInlineParser() + { + OpeningCharacters = new[] { '*', '_' }; + } + + public override bool Match(InlineParserState state) + { + // First, some definitions. A delimiter run is either a sequence of one or more * characters that + // is not preceded or followed by a * character, or a sequence of one or more _ characters that + // is not preceded or followed by a _ character. + + var delimiterChar = state.Text.CurrentChar; + var pc = state.Text.PeekChar(-1); + if (delimiterChar == pc && state.Text.PeekChar(-2) != '\\') + { + return false; + } + + int delimiterCount = 0; + char c; + do + { + delimiterCount++; + c = state.Text.NextChar(); + } while (c == delimiterChar); + + + // A left-flanking delimiter run is a delimiter run that is + // (a) not followed by Unicode whitespace, and + // (b) either not followed by a punctuation character, or preceded by Unicode whitespace + // or a punctuation character. + // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. + bool nextIsPunctuation; + bool nextIsWhiteSpace; + bool prevIsPunctuation; + bool prevIsWhiteSpace; + pc.CheckUnicodeCategory(out prevIsWhiteSpace, out prevIsPunctuation); + c.CheckUnicodeCategory(out nextIsWhiteSpace, out nextIsPunctuation); + + bool canOpen = !nextIsWhiteSpace && + (!nextIsPunctuation || prevIsWhiteSpace || prevIsPunctuation); + + + // A right-flanking delimiter run is a delimiter run that is + // (a) not preceded by Unicode whitespace, and + // (b) either not preceded by a punctuation character, or followed by Unicode whitespace + // or a punctuation character. + // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. + bool canClose = !prevIsWhiteSpace && + (!prevIsPunctuation || nextIsWhiteSpace || nextIsPunctuation); + + if (delimiterChar == '_') + { + var temp = canOpen; + // A single _ character can open emphasis iff it is part of a left-flanking delimiter run and either + // (a) not part of a right-flanking delimiter run or + // (b) part of a right-flanking delimiter run preceded by punctuation. + canOpen = canOpen && (!canClose || prevIsPunctuation); + + // A single _ character can close emphasis iff it is part of a right-flanking delimiter run and either + // (a) not part of a left-flanking delimiter run or + // (b) part of a left-flanking delimiter run followed by punctuation. + canClose = canClose && (!temp || nextIsPunctuation); + } + + //// If we can close, try to find a matching open + //if (canClose && state.Inline != null) + //{ + // var matching = DelimiterInline.FindMatchingOpen(state.Inline, 0, delimiterRun, delimiterCount); + + // // transform matching into + + // return true; + //} + + // We have potentially an open or close emphasis + if (canOpen || canClose) + { + var delimiterType = DelimiterType.None; + if (canOpen) + { + delimiterType |= DelimiterType.Open; + } + if (canClose) + { + delimiterType |= DelimiterType.Close; + } + + var delimiter = new EmphasisDelimiterInline(this) + { + DelimiterChar = delimiterChar, + DelimiterCount = delimiterCount, + Type = delimiterType, + }; + + state.Inline = delimiter; + return true; + } + + // We don't have an emphasis + return false; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/EscapeInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/EscapeInlineParser.cs new file mode 100644 index 00000000..7ab632f1 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/EscapeInlineParser.cs @@ -0,0 +1,31 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class EscapeInlineParser : InlineParser + { + public static readonly EscapeInlineParser Default = new EscapeInlineParser(); + + public EscapeInlineParser() + { + OpeningCharacters = new[] {'\\'}; + } + + public override bool Match(InlineParserState state) + { + // Go to escape character + var c = state.Text.PeekChar(1); + if (c.IsAsciiPunctuation()) + { + var literal = state.Inline as LiteralInline ?? + new LiteralInline() {ContentBuilder = state.StringBuilders.Get()}; + literal.ContentBuilder.Append(c); + state.Inline = literal; + state.Text.NextChar(); + return true; + } + return false; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/HardlineBreakInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/HardlineBreakInlineParser.cs new file mode 100644 index 00000000..5882fbbc --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/HardlineBreakInlineParser.cs @@ -0,0 +1,46 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class HardlineBreakInlineParser : InlineParser + { + public static readonly HardlineBreakInlineParser Default = new HardlineBreakInlineParser(); + + public HardlineBreakInlineParser() + { + OpeningCharacters = new[] {'\n'}; + } + + public override bool Match(InlineParserState state) + { + // Hard line breaks are for separating inline content within a block. Neither syntax for hard line breaks works at the end of a paragraph or other block element: + if (!(state.Block is ParagraphBlock) || state.Text.Column == 0 || !state.Text.PeekChar(-1).IsSpace() || !state.Text.PeekChar(-2).IsSpace()) + { + return false; + } + + //// A line break (not in a code span or HTML tag) that is preceded by two or more spaces + //// and does not occur at the end of a block is parsed as a hard line break (rendered in HTML as a
tag) + //var text = state.Lines; + //int spaceCount = 0; + //var c = text.CurrentChar; + //while (c.IsSpaceOrTab()) + //{ + // c = text.NextChar(); + // spaceCount++; + //} + //if (c == '\\') + //{ + // c = text.NextChar(); + // spaceCount = 2; + //} + //if (c != '\n' || spaceCount < 2) + //{ + // return false; + //} + + return false; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/LinkInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/LinkInlineParser.cs new file mode 100644 index 00000000..c1aed4d8 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/LinkInlineParser.cs @@ -0,0 +1,254 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class LinkInlineParser : InlineParser + { + public static readonly InlineParser Default = new LinkInlineParser(); + + public LinkInlineParser() + { + OpeningCharacters = new[] {'[', ']', '!'}; + } + + public override bool Match(InlineParserState state) + { + var c = state.Text.CurrentChar; + + bool isImage = false; + if (c == '!') + { + isImage = true; + c = state.Text.NextChar(); + if (c != '[') + { + return false; + } + } + + switch (c) + { + case '[': + // If this is not an image, we may have a reference link shortcut + // so we try to resolve it here + var saved = state.Text; + string label; + + // If the label is followed by either a ( or a [, this is not a shortcut + if (LinkHelper.TryParseLabel(ref state.Text, out label)) + { + if (!state.Document.LinkReferenceDefinitions.ContainsKey(label)) + { + label = null; + } + } + state.Text = saved; + + // Else we insert a LinkDelimiter + state.Text.NextChar(); + state.Inline = new LinkDelimiterInline(this) + { + Type = DelimiterType.Open, + Label = label, + IsImage = isImage + }; + return true; + + case ']': + state.Text.NextChar(); + if (state.Inline != null) + { + if (TryProcessLinkOrImage(state, ref state.Text)) + { + return true; + } + } + + // If we don’t find one, we return a literal text node ]. + // (Done after by the LiteralInline parser) + return false; + } + + // We don't have an emphasis + return false; + } + + private bool ProcessLinkReference(InlineParserState state, string label, bool isImage, Inline child = null) + { + bool isValidLink = false; + LinkReferenceDefinitionBlock linkRef; + if (state.Document.LinkReferenceDefinitions.TryGetValue(label, out linkRef)) + { + // Inline Link + var link = new LinkInline() + { + Url = HtmlHelper.Unescape(linkRef.Url), + Title = HtmlHelper.Unescape(linkRef.Title), + IsImage = isImage, + }; + + if (child == null) + { + child = new LiteralInline() + { + Content = label, + IsClosed = true + }; + link.AppendChild(child); + } + else + { + // Insert all child into the link + while (child != null) + { + var next = child.NextSibling; + child.Remove(); + link.AppendChild(child); + child = next; + } + } + link.IsClosed = true; + + EmphasisInline.ProcessEmphasis(link); + + state.Inline = link; + isValidLink = true; + } + //else + //{ + // // Else output a literal, leave it opened as we may have literals after + // // that could be append to this one + // var literal = new LiteralInline() + // { + // ContentBuilder = state.StringBuilders.Get().Append('[').Append(label).Append(']') + // }; + // state.Inline = literal; + //} + return isValidLink; + } + + private bool TryProcessLinkOrImage(InlineParserState inlineState, ref StringSlice text) + { + LinkDelimiterInline openParent = null; + foreach (var parent in inlineState.Inline.FindParentOfType()) + { + openParent = parent; + break; + } + + // This will be matched as a literal + if (openParent != null) + { + var parentDelimiter = openParent.Parent; + switch (text.CurrentChar) + { + case '(': + string url; + string title; + if (LinkHelper.TryParseInlineLink(ref text, out url, out title)) + { + // Inline Link + var link = new LinkInline() + { + Url = HtmlHelper.Unescape(url), + Title = HtmlHelper.Unescape(title), + IsImage = openParent.IsImage, + }; + + openParent.ReplaceBy(link); + inlineState.Inline = link; + + EmphasisInline.ProcessEmphasis(link); + + ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter); + + link.IsClosed = true; + + return true; + } + break; + default: + + string label = null; + // Handle Collapsed links + if (text.CurrentChar == '[') + { + if (text.PeekChar(1) == ']') + { + label = openParent.Label; + text.NextChar(); // Skip [ + text.NextChar(); // Skip ] + } + } + else + { + label = openParent.Label; + } + + if (label != null || LinkHelper.TryParseLabel(ref text, true, out label)) + { + if (ProcessLinkReference(inlineState, label, openParent.IsImage, + openParent.FirstChild)) + { + // Remove the open parent + openParent.Remove(); + ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter); + } + else + { + return false; + } + return true; + } + break; + } + + // We have a nested [ ] + // firstParent.Remove(); + // The opening [ will be transformed to a literal followed by all the childrens of the [ + + var literal = new LiteralInline() + { + ContentBuilder = inlineState.StringBuilders.Get().Append(openParent.IsImage ? "![" : "[") + }; + + inlineState.InlinesToClose.Add(literal); + inlineState.Inline = openParent.ReplaceBy(literal); + return false; + } + + return false; + } + + private void ReplaceParentIfNotImage(bool isImage, Inline inline) + { + if (isImage || inline == null) + { + return; + } + + foreach (var parent in inline.FindParentOfType()) + { + if (parent.IsImage) + { + break; + } + + var literal = new LiteralInline() + { + Content = "[", + IsClosed = true + }; + + parent.ReplaceBy(literal); + } + } + + private bool TryParseLinkTitle(InlineParserState state) + { + return false; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/Inlines/LiteralInlineParser.cs b/src/Textamina.Markdig/Parsers/Inlines/LiteralInlineParser.cs new file mode 100644 index 00000000..271a9c76 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/Inlines/LiteralInlineParser.cs @@ -0,0 +1,32 @@ +using System.Text; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers.Inlines +{ + public class LiteralInlineParser : InlineParser + { + public static readonly LiteralInlineParser Default = new LiteralInlineParser(); + + public override bool Match(InlineParserState state) + { + // A literal will always match + var literal = state.Inline as LiteralInline; + StringBuilder builder; + if (literal == null) + { + builder = state.StringBuilders.Get(); + literal = new LiteralInline {ContentBuilder = builder}; + state.Inline = literal; + } + else + { + builder = literal.ContentBuilder; + } + + var text = state.Text; + builder.Append(text.CurrentChar); + text.NextChar(); + return true; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/ListBlockParser.cs b/src/Textamina.Markdig/Parsers/ListBlockParser.cs new file mode 100644 index 00000000..69ec8bdc --- /dev/null +++ b/src/Textamina.Markdig/Parsers/ListBlockParser.cs @@ -0,0 +1,398 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class ListBlockParser : BlockParser + { + public ListBlockParser() + { + OpeningCharacters = new[] {'-', '+', '*'}; + } + + public override BlockState TryOpen(BlockParserState state) + { + // When both a thematic break and a list item are possible + // interpretations of a line, the thematic break takes precedence + if (ThematicBreakParser.Default.TryOpen(state) == BlockState.Break) + { + // Remove the ThematicBreakBlock as we will let the ThematicBreakBlock to catch it later + return BlockState.Break; + } + + // 5.2 List items + // TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space + + int preIndent = 0; + for (int i = state.Line.Start - 1; i >= 0; i--) + { + if (state.Line[i].IsSpaceOrTab()) + { + preIndent++; + } + else + { + break; + } + } + + return TryParseListItem(state, null, preIndent); + } + + public override BlockState TryContinue(BlockParserState state, Block block) + { + if (block is ListBlock && state.NextContinue is ListItemBlock) + { + // We try to match only on item block if the ListBlock + return BlockState.Skip; + } + + // When both a thematic break and a list item are possible + // interpretations of a line, the thematic break takes precedence + var save = state.Line; + if (ThematicBreakBlock.Parser.TryOpen(state) == BlockState.Break) + { + return BlockState.Break; + } + state.Line = save; + state.ParseIndent(); + + // 5.2 List items + // TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space + + int preIndent = 0; + for (int i = state.Line.Start - 1; i >= 0; i--) + { + if (state.Line[i].IsSpaceOrTab()) + { + preIndent++; + } + else + { + break; + } + } + + var saveLiner = state.Line; + + // If we have already a ListItemBlock, we are going to try to append to it + var listItem = block as ListItemBlock; + if (listItem != null) + { + var list = (ListBlock)listItem.Parent; + + // Allow all blanks lines if the last block is a fenced code block + // Allow 1 blank line inside a list + // If > 1 blank line, terminate this list + var isBlankLine = state.IsBlankLine; + //if (isBlankLine && !(state.LastBlock is FencedCodeBlock)) // TODO: Handle this case + var isInFencedBlock = state.LastBlock is FencedCodeBlock; + if (isBlankLine) + { + // TODO: Check with a generic way (allow a block to have multiple empty lines) + if (!isInFencedBlock) + { + if (!(state.NextContinue is ListBlock)) + { + list.CountAllBlankLines++; + listItem.Children.Add(BlankLineBlock.Instance); + } + list.CountBlankLinesReset++; + } + + if (list.CountBlankLinesReset > 1) + { + // TODO: Close all lists and not only this one + return BlockState.BreakDiscard; + } + + if (list.CountBlankLinesReset == 1 && listItem.NumberOfSpaces < 0) + { + state.Close(listItem); + + // Leave the list open + list.IsOpen = true; + return BlockState.Continue; + } + + return BlockState.Continue; + } + + list.CountBlankLinesReset = 0; + + var c = state.Line.CurrentChar; + var startPosition = state.Line.Start; + + // List Item starting with a blank line (-1) + if (listItem.NumberOfSpaces < 0) + { + int expectedCount = -listItem.NumberOfSpaces; + int countSpaces = 0; + var saved = new StringSlice(); + while (c.IsSpaceOrTab()) + { + c = state.Line.NextChar(); + countSpaces = preIndent + state.Line.Column - startPosition; + if (countSpaces == expectedCount) + { + saved = state.Line; + } + else if (countSpaces >= 4) + { + state.Line = saved; + state.ParseIndent(); + countSpaces = expectedCount; + break; + } + } + + if (countSpaces == expectedCount) + { + listItem.NumberOfSpaces = countSpaces; + return BlockState.Continue; + } + } + else + { + while (c.IsSpaceOrTab()) + { + c = state.Line.NextChar(); + var countSpaces = preIndent + state.Line.Column - startPosition; + if (countSpaces >= listItem.NumberOfSpaces) + { + return BlockState.Continue; + } + } + } + state.Line = saveLiner; + state.ParseIndent(); + } + + return TryParseListItem(state, block, preIndent); + } + + + private BlockState TryParseListItem(BlockParserState state, Block block, int preIndent) + { + var isInList = block is ListItemBlock; + + var preStartPosition = state.Line.Start; + + var c = state.Line.CurrentChar; + if (isInList) + { + while (c.IsSpaceOrTab()) + { + c = state.Line.NextChar(); + } + } + else + { + // TODO + //state.Line.SkipLeadingSpaces3(); + c = state.Line.CurrentChar; + } + preIndent = preIndent + state.Line.Start - preStartPosition; + + var isOrdered = false; + var bulletChar = (char) 0; + int orderedStart = 0; + var orderedDelimiter = (char) 0; + + var column = state.Line.Start; + + if (c.IsBulletListMarker()) + { + bulletChar = c; + preIndent++; + } + else if (c.IsDigit()) + { + int countDigit = 0; + while (c.IsDigit()) + { + orderedStart = orderedStart*10 + c - '0'; + c = state.Line.NextChar(); + preIndent++; + countDigit++; + } + + // Note that ordered list start numbers must be nine digits or less: + if (countDigit > 9) + { + return BlockState.None; + } + + // We don't have an ordered list + if (c != '.' && c != ')') + { + return BlockState.None; + } + preIndent++; + isOrdered = true; + orderedDelimiter = c; + } + else + { + return BlockState.None; + } + + // Skip Bullet or '.' + state.Line.NextChar(); + + // Item starting with a blank line + int numberOfSpaces; + if (state.IsBlankLine) + { + // Use a negative number to store the number of expected chars + numberOfSpaces = -(preIndent + 1); + } + else + { + var startPosition = -1; + int countSpaceAfterBullet = 0; + var saved = new StringSlice(); + for (int i = 0; i <= 4; i++) + { + c = state.Line.CurrentChar; + if (!c.IsSpaceOrTab()) + { + break; + } + if (i == 0) + { + startPosition = state.Line.Column; + } + + var endPosition = state.Line.Column; + countSpaceAfterBullet = endPosition - startPosition; + + if (countSpaceAfterBullet == 1) + { + saved = state.Line; + } + else if (countSpaceAfterBullet >= 4) + { + //state.Line.SpaceHeaderCount = countSpaceAfterBullet - 4; + countSpaceAfterBullet = 0; + state.Line = saved; + state.ParseIndent(); + break; + } + state.Line.NextChar(); + } + + // If we haven't matched any spaces, early exit + if (startPosition < 0) + { + return BlockState.None; + } + // Number of spaces required for the following content to be part of this list item + numberOfSpaces = preIndent + countSpaceAfterBullet + 1; + } + + var newListItem = new ListItemBlock(this) + { + Column = column, + NumberOfSpaces = numberOfSpaces + }; + state.NewBlocks.Push(newListItem); + + var currentListItem = block as ListItemBlock; + var currentParent = block as ListBlock ?? (ListBlock)currentListItem?.Parent; + + if (currentParent != null) + { + // If we have a new list item, close the previous one + if (currentListItem != null) + { + state.Close(currentListItem); + } + + // Reset the list if it is a new list or a new type of bullet + if (currentParent.IsOrdered != isOrdered || + (isOrdered && currentParent.OrderedDelimiter != orderedDelimiter) || + (!isOrdered && currentParent.BulletChar != bulletChar) + //(numberOfSpaces < ((ListItemBlock) currentParent.LastChild).NumberOfSpaces) + ) + { + state.Close(currentParent); + currentParent = null; + } + } + + if (currentParent == null) + { + var newList = new ListBlock(this) + { + Column = column, + IsOrdered = isOrdered, + BulletChar = bulletChar, + OrderedDelimiter = orderedDelimiter, + OrderedStart = orderedStart, + }; + state.NewBlocks.Push(newList); + } + + return BlockState.Continue; + } + + public override bool Close(BlockParserState state, Block blockToClose) + { + var listBlock = blockToClose as ListBlock; + + // Process only if we have blank lines + if (listBlock == null || listBlock.CountAllBlankLines <= 0) + { + return true; + } + + // TODO: This code is UGLY and WAY TOO LONG, simplify! + bool isLastListItem = true; + for (int listIndex = listBlock.Children.Count - 1; listIndex >= 0; listIndex--) + { + var block = listBlock.Children[listIndex]; + var listItem = (ListItemBlock) block; + var children = listItem.Children; + bool isLastElement = true; + for (int i = children.Count - 1; i >= 0; i--) + { + var item = children[i]; + if (item is BlankLineBlock) + { + if ((isLastElement && listIndex < listBlock.Children.Count - 1) || (children.Count > 2 && (i > 0 && i < (children.Count - 1)))) + { + listBlock.IsLoose = true; + } + + if (isLastElement && isLastListItem) + { + // Inform the outer list that we have a blank line + var parentListItemBlock = listBlock.Parent as ListItemBlock; + if (parentListItemBlock != null) + { + var parentList = (ListBlock) parentListItemBlock.Parent; + + parentList.CountAllBlankLines++; + parentListItemBlock.Children.Add(BlankLineBlock.Instance); + } + } + + children.RemoveAt(i); + + // If we have remove all blank lines, we can exit + listBlock.CountAllBlankLines--; + if (listBlock.CountAllBlankLines == 0) + { + return false; + } + } + isLastElement = false; + } + isLastListItem = false; + } + + return true; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/MarkdownParser.cs b/src/Textamina.Markdig/Parsers/MarkdownParser.cs new file mode 100644 index 00000000..41c02e4c --- /dev/null +++ b/src/Textamina.Markdig/Parsers/MarkdownParser.cs @@ -0,0 +1,229 @@ +using System.Collections.Generic; +using System.IO; +using System.Threading.Tasks; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; +using Textamina.Markdig.Syntax.Inlines; + +namespace Textamina.Markdig.Parsers +{ + public class MarkdownParser + { + public static TextWriter Log; + private readonly ParserList inlineParsers; + private readonly Document document; + private readonly BlockParserState blockParserState; + private readonly StringBuilderCache stringBuilderCache; + + public MarkdownParser(TextReader reader) + { + document = new Document(); + Reader = reader; + stringBuilderCache = new StringBuilderCache(); + blockParserState = new BlockParserState(stringBuilderCache, document); + + inlineParsers = new ParserList() + { + LinkInline.Parser, + EmphasisInline.Parser, + EscapeInline.Parser, + CodeInline.Parser, + AutolinkInline.Parser, + HardlineBreakInline.Parser, + LiteralInline.Parser, + }; + inlineParsers.Initialize(); + } + + public TextReader Reader { get; } + + public Document Parse() + { + ParseLines(); + //ProcessInlines(document); + return document; + } + + private void ParseLines() + { + while (true) + { + var lineText = Reader.ReadLine(); + + // If this is the end of file and the last line is empty + if (lineText == null) + { + break; + } + blockParserState.ProcessLine(lineText); + } + blockParserState.CloseAll(true); + } + + private void ProcessInlines(ContainerBlock container) + { + var list = new Stack(); + list.Push(container); + var leafs = new List(); + + while (list.Count > 0) + { + container = list.Pop(); + foreach (var block in container.Children) + { + var leafBlock = block as LeafBlock; + if (leafBlock != null) + { + if (leafBlock.ProcessInlines) + { + var task = new Task(() => ProcessInlineLeaf(leafBlock)); + task.Start(); + leafs.Add(task); + //ProcessInlineLeaf(leafBlock); + } + } + else + { + list.Push((ContainerBlock)block); + } + } + } + + Task.WaitAll(leafs.ToArray()); + } + + private void ProcessInlineLeaf(LeafBlock leafBlock) + { + var lines = leafBlock.Lines; + + leafBlock.Inline = new ContainerInline() {IsClosed = false}; + var inlineState = new InlineParserState(stringBuilderCache, document) + { + Text = new StringSlice(leafBlock.Lines.ToString()), + Inline = leafBlock.Inline, + Block = leafBlock + }; + + while (!inlineState.Text.IsEndOfSlice) + { + var saveLine = inlineState.Text; + + var c = saveLine.CurrentChar; + + var parsers = inlineParsers.GetParsersForOpeningCharacter(c); + bool match = false; + if (parsers != null) + { + for (int i = 0; i < parsers.Length; i++) + { + if (parsers[i].Match(inlineState)) + { + match = true; + break; + } + } + } + parsers = inlineParsers.GlobalParsers; + if (!match && parsers != null) + { + for (int i = 0; i < parsers.Length; i++) + { + if (parsers[i].Match(inlineState)) + { + break; + } + } + } + + var nextInline = inlineState.Inline; + + if (nextInline != null) + { + if (nextInline.Parent == null) + { + // Get deepest container + var container = (ContainerInline)leafBlock.Inline; + while (true) + { + var nextContainer = container.LastChild as ContainerInline; + if (nextContainer != null && !nextContainer.IsClosed) + { + container = nextContainer; + } + else + { + break; + } + } + + container.AppendChild(nextInline); + } + + if (nextInline.IsClosable && !nextInline.IsClosed) + { + var inlinesToClose = inlineState.InlinesToClose; + var last = inlinesToClose.Count > 0 + ? inlineState.InlinesToClose[inlinesToClose.Count - 1] + : null; + if (last != nextInline) + { + inlineState.InlinesToClose.Add(nextInline); + } + } + } + else + { + // Get deepest container + var container = (ContainerInline)leafBlock.Inline; + while (true) + { + var nextContainer = container.LastChild as ContainerInline; + if (nextContainer != null && !nextContainer.IsClosed) + { + container = nextContainer; + } + else + { + break; + } + } + + inlineState.Inline = container.LastChild is LeafInline ? container.LastChild : container; + } + + if (Log != null) + { + Log.WriteLine($"** Dump: char '{c}"); + leafBlock.Inline.DumpTo(Log); + } + } + + // Close all inlines not closed + inlineState.Inline = null; + foreach (var inline in inlineState.InlinesToClose) + { + inline.CloseInternal(inlineState); + } + inlineState.InlinesToClose.Clear(); + + if (Log != null) + { + Log.WriteLine("** Dump before Emphasis:"); + leafBlock.Inline.DumpTo(Log); + EmphasisInline.ProcessEmphasis(leafBlock.Inline); + + Log.WriteLine(); + Log.WriteLine("** Dump after Emphasis:"); + leafBlock.Inline.DumpTo(Log); + } + // TODO: Close opened inlines + + // Close last inline + //while (inlineStack.Count > 0) + //{ + // var inlineState = inlineStack.Pop(); + // inlineState.Parser.Close(state, inlineState.Inline); + //} + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/ParagraphBlockParser.cs b/src/Textamina.Markdig/Parsers/ParagraphBlockParser.cs new file mode 100644 index 00000000..4068adf7 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/ParagraphBlockParser.cs @@ -0,0 +1,166 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class ParagraphBlockParser : BlockParser + { + public override BlockState TryOpen(BlockParserState state) + { + if (state.IsBlankLine) + { + return BlockState.None; + } + + // We continue trying to match by default + state.NewBlocks.Push(new ParagraphBlock(this) {Column = state.Column}); + return BlockState.Continue; + } + + public override BlockState TryContinue(BlockParserState state, Block block) + { + return state.IsBlankLine ? BlockState.BreakDiscard : BlockState.Continue; + } + + public override bool Close(BlockParserState state, Block block) + { + var paragraph = block as ParagraphBlock; + var heading = block as HeadingBlock; + if (paragraph != null) + { + var lines = paragraph.Lines; + + TryMatchLinkReferenceDefinition(lines, state); + + // If Paragraph is empty, we can discard it + if (lines.Count == 0) + { + return false; + } + + var lineCount = lines.Count; + for (int i = 0; i < lineCount; i++) + { + lines.Slices[i].Trim(); + } + } + else // if (heading?.Lines.Count > 1) + { + //heading.Lines.RemoveAt(heading.Lines.Count - 1); + } + + return true; + } + + private BlockState TryParseSetexHeading(BlockParserState state, Block block) + { + var paragraph = (ParagraphBlock) block; + var headingChar = (char)0; + bool checkForSpaces = false; + for (int i = state.Start; i <= state.EndOffset; i++) + { + var c = state.Line[i]; + if (headingChar == 0) + { + if (c == '=' || c == '-') + { + headingChar = c; + continue; + } + break; + } + + if (checkForSpaces) + { + if (!c.IsSpaceOrTab()) + { + headingChar = (char)0; + break; + } + } + else if (c != headingChar) + { + if (c.IsSpaceOrTab()) + { + checkForSpaces = true; + } + else + { + headingChar = (char)0; + break; + } + } + } + + if (headingChar != 0) + { + state.Discard(paragraph); + + // If we matched a LinkReferenceDefinition before matching the heading, and the remaining + // lines are empty, we can early exit and remove the paragraph + if (!TryMatchLinkReferenceDefinition(paragraph.Lines, state) && paragraph.Lines.Count == 0) + { + var level = headingChar == '=' ? 1 : 2; + + var heading = new HeadingBlock(this) + { + Column = paragraph.Column, + Level = level, + Lines = paragraph.Lines, + }; + heading.Lines.Trim(); + + // Remove the paragraph as a pending block + state.NewBlocks.Push(heading); + } + return BlockState.BreakDiscard; + } + + return BlockState.Continue; + } + + private bool TryMatchLinkReferenceDefinition(StringSliceList localLineGroup, BlockParserState state) + { + bool atLeastOneFound = false; + + //var saved = new StringSliceList.State(); + //while (true) + //{ + // // If we have found a LinkReferenceDefinition, we can discard the previous paragraph + // localLineGroup.Save(ref saved); + // LinkReferenceDefinitionBlock linkReferenceDefinition; + // if (LinkReferenceDefinitionBlock.TryParse(localLineGroup, out linkReferenceDefinition)) + // { + // if (!state.Root.LinkReferenceDefinitions.ContainsKey(linkReferenceDefinition.Label)) + // { + // state.Root.LinkReferenceDefinitions[linkReferenceDefinition.Label] = linkReferenceDefinition; + // } + // atLeastOneFound = true; + + // // Remove lines that have been matched + // if (localLineGroup.LinePosition == localLineGroup.Count) + // { + // localLineGroup.Clear(); + // } + // else + // { + // for (int i = localLineGroup.LinePosition - 1; i >= 0; i--) + // { + // localLineGroup.RemoveAt(i); + // } + // } + // } + // else + // { + // if (!atLeastOneFound) + // { + // localLineGroup.Restore(ref saved); + // } + // break; + // } + //} + + return atLeastOneFound; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/ParserList.cs b/src/Textamina.Markdig/Parsers/ParserList.cs new file mode 100644 index 00000000..dd1ddb48 --- /dev/null +++ b/src/Textamina.Markdig/Parsers/ParserList.cs @@ -0,0 +1,72 @@ +using System.Collections.Generic; + +namespace Textamina.Markdig.Parsers +{ + public class ParserList : List where T : class, ICharacterParser + { + private T[][] parsersWithOpeningCharacters; + private T[] globalParsers; + + public T[] GlobalParsers => globalParsers; + + public T[] GetParsersForOpeningCharacter(char openingChar) + { + return openingChar < parsersWithOpeningCharacters.Length ? parsersWithOpeningCharacters[openingChar] : null; + } + + public void Initialize() + { + var charCounter = new Dictionary(); + int globalCounter = 0; + int maxChar = 0; + foreach (var parser in this) + { + if (parser.OpeningCharacters != null && parser.OpeningCharacters.Length != 0) + { + foreach (var openingChar in parser.OpeningCharacters) + { + if (!charCounter.ContainsKey(openingChar)) + { + charCounter[openingChar] = 0; + } + charCounter[openingChar]++; + if (openingChar > maxChar) + { + maxChar = openingChar; + } + } + } + else + { + globalCounter++; + } + } + + globalParsers = new T[globalCounter]; + parsersWithOpeningCharacters = new T[maxChar+1][]; + + foreach (var parser in this) + { + if (parser.OpeningCharacters != null && parser.OpeningCharacters.Length != 0) + { + foreach (var openingChar in parser.OpeningCharacters) + { + if (parsersWithOpeningCharacters[openingChar] == null) + { + parsersWithOpeningCharacters[openingChar] = new T[charCounter[openingChar]]; + } + var list = parsersWithOpeningCharacters[openingChar]; + var index = list.Length - charCounter[openingChar]; + list[index] = parser; + charCounter[openingChar]--; + } + } + else + { + globalParsers[globalParsers.Length - globalCounter] = parser; + globalCounter--; + } + } + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/QuoteBlockParser.cs b/src/Textamina.Markdig/Parsers/QuoteBlockParser.cs new file mode 100644 index 00000000..ff17b48c --- /dev/null +++ b/src/Textamina.Markdig/Parsers/QuoteBlockParser.cs @@ -0,0 +1,60 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class QuoteBlockParser : BlockParser + { + public QuoteBlockParser() + { + OpeningCharacters = new[] {'>'}; + } + + public override BlockState TryOpen(BlockParserState state) + { + if (state.IsCodeIndent) + { + return BlockState.None; + } + + var column = state.Column; + + // 5.1 Block quotes + // A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space. + var quoteChar = state.CurrentChar; + var c = state.NextChar(); + if (c.IsSpace()) + { + state.NextChar(); + } + state.NewBlocks.Push(new QuoteBlock(this) {QuoteChar = quoteChar, Column = column}); + return BlockState.Continue; + } + + public override BlockState TryContinue(BlockParserState state, Block block) + { + if (state.IsCodeIndent) + { + return BlockState.None; + } + + var quote = (QuoteBlock) block; + + // 5.1 Block quotes + // A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space. + var c = state.CurrentChar; + if (c != quote.QuoteChar) + { + return state.IsBlankLine ? BlockState.BreakDiscard : BlockState.None; + } + + c = state.NextChar(); // Skip opening char + if (c.IsSpace()) + { + state.NextChar(); // Skip following space + } + + return BlockState.Continue; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsers/ThematicBreakParser.cs b/src/Textamina.Markdig/Parsers/ThematicBreakParser.cs new file mode 100644 index 00000000..825dcb8a --- /dev/null +++ b/src/Textamina.Markdig/Parsers/ThematicBreakParser.cs @@ -0,0 +1,77 @@ +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Syntax; + +namespace Textamina.Markdig.Parsers +{ + public class ThematicBreakParser : BlockParser + { + public static readonly ThematicBreakParser Default = new ThematicBreakParser(); + + public ThematicBreakParser() + { + OpeningCharacters = new[] {'-', '_', '*'}; + } + + public override BlockState TryOpen(BlockParserState state) + { + if (state.IsCodeIndent) + { + return BlockState.None; + } + + // 4.1 Thematic breaks + // A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces + int breakCharCount = 1; + var breakChar = state.CurrentChar; + bool hasSpacesSinceLastMatch = false; + bool hasInnerSpaces = false; + int offset = 0; + var c = state.PeekChar(offset++); + while (c != '\0') + { + if (c == breakChar) + { + if (hasSpacesSinceLastMatch) + { + hasInnerSpaces = true; + } + + breakCharCount++; + } + else if (c.IsSpace()) + { + hasSpacesSinceLastMatch = true; + } + else + { + return BlockState.None; + } + + c = state.PeekChar(offset); + offset++; + } + + // If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it + var previousParagraph = state.LastBlock as ParagraphBlock; + + var isSetexHeading = previousParagraph != null && breakChar == '-' && !hasInnerSpaces; + if (isSetexHeading) + { + var parent = previousParagraph.Parent; + if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != state.Column)) + { + isSetexHeading = false; + } + } + + if (breakCharCount < 3 || isSetexHeading) + { + return BlockState.None; + } + + // Push a new block + state.NewBlocks.Push(new ThematicBreakBlock(this) { Column = state.Column }); + return BlockState.BreakDiscard; + } + } +} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/BlockParser.cs b/src/Textamina.Markdig/Parsing/BlockParser.cs deleted file mode 100644 index 8b0abd38..00000000 --- a/src/Textamina.Markdig/Parsing/BlockParser.cs +++ /dev/null @@ -1,24 +0,0 @@ -using Textamina.Markdig.Syntax; - -namespace Textamina.Markdig.Parsing -{ - public abstract class BlockParser - { - protected BlockParser() - { - // By default, all blocks can interrupt a paragraph except: - // - setext heading - // - indented code block - // - a special HTML blocks - CanInterruptParagraph = true; - } - - public bool CanInterruptParagraph { get; protected set; } - - public abstract MatchLineResult Match(BlockParserState state); - - public virtual void Close(BlockParserState state) - { - } - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/BlockParserState.cs b/src/Textamina.Markdig/Parsing/BlockParserState.cs deleted file mode 100644 index ebf445f2..00000000 --- a/src/Textamina.Markdig/Parsing/BlockParserState.cs +++ /dev/null @@ -1,176 +0,0 @@ - - - -using System; -using System.Collections.Generic; -using System.Runtime.CompilerServices; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Syntax; - -namespace Textamina.Markdig.Parsing -{ - public class BlockParserState : List - { - public BlockParserState(StringBuilderCache stringBuilders, Document root) - { - if (stringBuilders == null) throw new ArgumentNullException(nameof(stringBuilders)); - if (root == null) throw new ArgumentNullException(nameof(root)); - StringBuilders = stringBuilders; - Root = root; - NewBlocks = new Stack(); - Add(root); - } - - - public StringSlice Line; - - public int LineIndex; - - public bool IsBlankLine => CurrentChar == '\0'; - - public bool IsEndOfLine => Line.IsEndOfSlice; - - public char CurrentChar => Line.CurrentChar; - - public char NextChar() => Line.NextChar(); - - public char CharAt(int index) => Line[index]; - - public int Start => Line.Start; - - public int EndOffset => Line.End; - - public int Indent => Column - ColumnBegin; - - public bool IsCodeIndent => Indent >= 4; - - public int ColumnBegin { get; private set; } - - public int Column { get; set; } - - public Block Pending { get; set; } - - public int PendingIndex { get; internal set; } - - public readonly Stack NewBlocks; - - public ContainerBlock CurrentContainer; - - public Block LastBlock; - - public readonly Document Root; - - public StringBuilderCache StringBuilders { get; } - - public char PeekChar(int offset) - { - return Line.PeekChar(offset); - } - - [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public void SetCurrentLine(ref StringSlice line) - { - Line = line; - EatSpaces(); - } - - public void EatSpaces() - { - var c = CurrentChar; - ColumnBegin = Column; - while (c !='\0') - { - if (c == ' ') - { - Column++; - } - else if (c == '\t') - { - Column = ((Column + 3) >> 2) << 2; - } - else - { - break; - } - c = NextChar(); - } - } - - public Block NextPending - { - get { return PendingIndex + 1 < Count ? this[PendingIndex + 1] : null; } - } - - public void Close(Block block) - { - // If we close a block, we close all blocks above - for (int i = Count - 1; i >= 1; i--) - { - if (this[i] == block) - { - for (int j = Count - 1; j >= i; j--) - { - Close(j); - } - break; - } - } - } - - public void Discard(Block block) - { - for (int i = Count - 1; i >= 1; i--) - { - if (this[i] == block) - { - if (Pending == block) - { - Pending = null; - } - block.Parent.Children.Remove(block); - RemoveAt(i); - break; - } - } - } - - public void Close(int index) - { - var block = this[index]; - - var saveBlock = Pending; - - Pending = block; - block.Parser.Close(this); - - // If the pending object is removed, we need to remove it from the parent container - if (Pending == null) - { - var parent = block.Parent as ContainerBlock; - if (parent != null) - { - parent.Children.Remove(block); - } - } - - RemoveAt(index); - Pending = saveBlock; - } - - public void CloseAll(bool force) - { - // Close any previous blocks not opened - for (int i = Count - 1; i >= 1; i--) - { - var block = this[i]; - - // Stop on the first open block - if (!force && block.IsOpen) - { - break; - } - Close(i); - } - } - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/InlineParser.cs b/src/Textamina.Markdig/Parsing/InlineParser.cs deleted file mode 100644 index 8e00355f..00000000 --- a/src/Textamina.Markdig/Parsing/InlineParser.cs +++ /dev/null @@ -1,9 +0,0 @@ -namespace Textamina.Markdig.Parsing -{ - public abstract class InlineParser - { - public char[] FirstChars { get; protected set; } - - public abstract bool Match(InlineParserState state); - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/MarkdownParser.cs b/src/Textamina.Markdig/Parsing/MarkdownParser.cs deleted file mode 100644 index 98737213..00000000 --- a/src/Textamina.Markdig/Parsing/MarkdownParser.cs +++ /dev/null @@ -1,556 +0,0 @@ -using System; -using System.Collections.Generic; -using System.Diagnostics; -using System.IO; -using System.Threading.Tasks; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Syntax; - -namespace Textamina.Markdig.Parsing -{ - public class MarkdownParser - { - public static TextWriter Log; - private readonly List blockParsers; - private readonly List inlineParsers; - private readonly List regularInlineParsers; - private readonly InlineParser[] inlineWithFirstCharParsers; - private readonly Document document; - private readonly BlockParserState blockParserState; - private readonly StringBuilderCache stringBuilderCache; - - public MarkdownParser(TextReader reader) - { - document = new Document(); - Reader = reader; - blockParsers = new List(); - inlineParsers = new List(); - inlineWithFirstCharParsers = new InlineParser[128]; - regularInlineParsers = new List(); - stringBuilderCache = new StringBuilderCache(); - blockParserState = new BlockParserState(stringBuilderCache, document); - blockParsers = new List() - { - ThematicBreakBlock.Parser, - HeadingBlock.Parser, - QuoteBlock.Parser, - ListBlock.Parser, - - HtmlBlock.Parser, - CodeBlock.Parser, - FencedCodeBlock.Parser, - ParagraphBlock.Parser, - }; - - inlineParsers = new List() - { - LinkInline.Parser, - EmphasisInline.Parser, - EscapeInline.Parser, - CodeInline.Parser, - AutolinkInline.Parser, - HardlineBreakInline.Parser, - LiteralInline.Parser, - }; - InitializeInlineParsers(); - } - - private void InitializeInlineParsers() - { - foreach (var inlineParser in inlineParsers) - { - if (inlineParser.FirstChars != null && inlineParser.FirstChars.Length > 0) - { - foreach (var firstChar in inlineParser.FirstChars) - { - if (firstChar >= 128) - { - throw new InvalidOperationException($"Invalid character '{firstChar}'. Support only ASCII < 128 chars"); - } - inlineWithFirstCharParsers[firstChar] = inlineParser; - } - } - else - { - regularInlineParsers.Add(inlineParser); - } - } - } - - public TextReader Reader { get; } - - private Block LastBlock - { - get - { - var count = blockParserState.Count; - return count > 0 ? blockParserState[count - 1] : null; - } - } - - private ContainerBlock LastContainer - { - get - { - for (int i = blockParserState.Count - 1; i >= 0; i--) - { - var container = blockParserState[i] as ContainerBlock; - if (container != null) - { - return container; - } - } - return null; - } - } - - public Document Parse() - { - ParseLines(); - //ProcessInlines(document); - return document; - } - - private void ParseLines() - { - while (true) - { - var lineText = Reader.ReadLine(); - - // If this is the end of file and the last line is empty - if (lineText == null) - { - break; - } - var line = new StringSlice(lineText); - blockParserState.SetCurrentLine(ref line); - blockParserState.LineIndex++; - - bool continueProcessLiner = ProcessPendingBlocks(); - - // If we have already reached eol and the last block was a paragraph - // we close it - if (blockParserState.Line.IsEndOfSlice) - { - int index = blockParserState.Count - 1; - if (blockParserState[index] is ParagraphBlock) - { - blockParserState.Close(index); - continue; - } - } - - // If the line was not entirely processed by pending blocks, try to process it with any new block - while (continueProcessLiner) - { - ParseNewBlocks(ref continueProcessLiner); - } - - // Close blocks that are no longer opened - blockParserState.CloseAll(false); - } - - blockParserState.CloseAll(true); - // Close opened blocks - //ProcessPendingBlocks(true); - } - - private void ProcessInlines(ContainerBlock container) - { - var list = new Stack(); - list.Push(container); - var leafs = new List(); - - while (list.Count > 0) - { - container = list.Pop(); - foreach (var block in container.Children) - { - var leafBlock = block as LeafBlock; - if (leafBlock != null) - { - if (!leafBlock.NoInline) - { - var task = new Task(() => ProcessInlineLeaf(leafBlock)); - task.Start(); - leafs.Add(task); - //ProcessInlineLeaf(leafBlock); - } - } - else - { - list.Push((ContainerBlock)block); - } - } - } - - Task.WaitAll(leafs.ToArray()); - } - - private bool ProcessPendingBlocks() - { - bool processLiner = true; - - // Set all blocks non opened. - // They will be marked as open in the following loop - for (int i = 1; i < blockParserState.Count; i++) - { - blockParserState[i].IsOpen = false; - } - - // Create the line state that will be used by all parser - blockParserState.Pending = null; - - // Process any current block potentially opened - for (int i = 1; i < blockParserState.Count; i++) - { - var block = blockParserState[i]; - - // Else tries to match the Parser with the current line - var parser = block.Parser; - blockParserState.Pending = block; - - // If we have a paragraph block, we want to try to match over blocks before trying the Paragraph - if (blockParserState.Pending is ParagraphBlock) - { - break; - } - - var saveLiner = blockParserState.Line; - - // If we have a discard, we can remove it from the current state - blockParserState.CurrentContainer = LastContainer; - blockParserState.PendingIndex = i; - blockParserState.LastBlock = LastBlock; - var result = parser.Match(blockParserState); - if (result == MatchLineResult.Skip) - { - continue; - } - - if (result == MatchLineResult.None) - { - // Restore the Line where it was - blockParserState.SetCurrentLine(ref saveLiner); - break; - } - - // In case the BlockParser has modified the blockParserState we are iterating on - if (i >= blockParserState.Count) - { - i = blockParserState.Count - 1; - } - - // If a parser is adding a block, it must be the last of the list - if ((i + 1) < blockParserState.Count && blockParserState.NewBlocks.Count > 0) - { - throw new InvalidOperationException("A pending parser cannot add a new block when it is not the last pending block"); - } - - // If we have a leaf block - var leaf = blockParserState.Pending as LeafBlock; - if (leaf != null && blockParserState.NewBlocks.Count == 0) - { - processLiner = false; - if (result != MatchLineResult.LastDiscard && result != MatchLineResult.ContinueDiscard) - { - leaf.Lines.Append(ref blockParserState.Line); - } - - if (blockParserState.NewBlocks.Count > 0) - { - throw new InvalidOperationException( - "The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed"); - } - } - - // A block is open only if it has a Continue state. - // otherwise it is a Last state, and we don't keep it opened - block.IsOpen = result == MatchLineResult.Continue || result == MatchLineResult.ContinueDiscard; - - if (result == MatchLineResult.LastDiscard) - { - processLiner = false; - break; - } - - bool isLast = i == blockParserState.Count - 1; - if (processLiner) - { - processLiner = ProcessNewBlocks(result, false); - } - if (isLast || !processLiner) - { - break; - } - } - - return processLiner; - } - - private void ParseNewBlocks(ref bool continueProcessLiner) - { - blockParserState.Pending = null; - - for (int j = 0; j < blockParsers.Count; j++) - { - var blockParser = blockParsers[j]; - if (blockParserState.Line.IsEndOfSlice) - { - continueProcessLiner = false; - break; - } - - // If a block parser cannot interrupt a paragraph, and the last block is a paragraph - // we can skip this parser - var lastBlock = LastBlock; - var paragraph = lastBlock as ParagraphBlock; - if (paragraph != null && !blockParser.CanInterruptParagraph) - { - continue; - } - - bool isParsingParagraph = blockParser == ParagraphBlock.Parser; - blockParserState.Pending = isParsingParagraph ? paragraph : null; - blockParserState.CurrentContainer = LastContainer; - blockParserState.LastBlock = lastBlock; - - var saveLiner = blockParserState.Line; - var result = blockParser.Match(blockParserState); - if (result == MatchLineResult.None) - { - // If we have reached a blank line after trying to parse a paragraph - // we can ignore it - if (isParsingParagraph && blockParserState.IsBlankLine) - { - continueProcessLiner = false; - break; - } - - blockParserState.SetCurrentLine(ref saveLiner); - continue; - } - - // Special case for paragraph - paragraph = LastBlock as ParagraphBlock; - if (isParsingParagraph && paragraph != null) - { - Debug.Assert(blockParserState.NewBlocks.Count == 0); - - continueProcessLiner = false; - paragraph.Lines.Append(ref blockParserState.Line); - - // We have just found a lazy continuation for a paragraph, early exit - // Mark all block opened after a lazy continuation - for (int i = 0; i < blockParserState.Count; i++) - { - blockParserState[i].IsOpen = true; - } - break; - } - - // Nothing found but the BlockParser may instruct to break, so early exit - if (blockParserState.NewBlocks.Count == 0 && result == MatchLineResult.LastDiscard) - { - continueProcessLiner = false; - break; - } - - continueProcessLiner = ProcessNewBlocks(result, true); - - // If we have a container, we can retry to match against all types of block. - if (continueProcessLiner) - { - // rewind to the first parser - j = -1; - } - else - { - // We have a leaf node, we can stop - break; - } - } - } - - private bool ProcessNewBlocks(MatchLineResult result, bool allowClosing) - { - var newBlocks = blockParserState.NewBlocks; - while (newBlocks.Count > 0) - { - var block = newBlocks.Pop(); - - block.Line = blockParserState.LineIndex; - - // If we have a leaf block - var leaf = block as LeafBlock; - if (leaf != null) - { - if (result != MatchLineResult.LastDiscard && result != MatchLineResult.ContinueDiscard) - { - leaf.Lines.Append(ref blockParserState.Line); - } - - if (newBlocks.Count > 0) - { - throw new InvalidOperationException( - "The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed"); - } - } - - if (allowClosing) - { - // Close any previous blocks not opened - blockParserState.CloseAll(false); - } - - // If previous block is a container, add the new block as a children of the previous block - if (block.Parent == null) - { - var container = LastContainer; - LastContainer.Children.Add(block); - block.Parent = container; - } - - block.IsOpen = result == MatchLineResult.Continue || result == MatchLineResult.ContinueDiscard; - - // Add a block blockParserState to the stack (and leave it opened) - blockParserState.Add(block); - - if (leaf != null) - { - return false; - } - } - return true; - } - - private void ProcessInlineLeaf(LeafBlock leafBlock) - { - var lines = leafBlock.Lines; - - leafBlock.Inline = new ContainerInline() {IsClosed = false}; - var inlineState = new InlineParserState(stringBuilderCache, document) - { - Text = new StringSlice(leafBlock.Lines.ToString()), - Inline = leafBlock.Inline, - Block = leafBlock - }; - - while (!inlineState.Text.IsEndOfSlice) - { - var saveLine = inlineState.Text; - - var c = saveLine.CurrentChar; - var inlineParser = c < 128 ? inlineWithFirstCharParsers[c] : null; - if (inlineParser == null || !inlineParser.Match(inlineState)) - { - for (int i = 0; i < regularInlineParsers.Count; i++) - { - inlineState.Text = saveLine; - inlineParser = regularInlineParsers[i]; - if (inlineParser.Match(inlineState)) - { - break; - } - - inlineParser = null; - } - - if (inlineParser == null) - { - inlineState.Text = saveLine; - } - } - - var nextInline = inlineState.Inline; - - if (nextInline != null) - { - if (nextInline.Parent == null) - { - // Get deepest container - var container = (ContainerInline)leafBlock.Inline; - while (true) - { - var nextContainer = container.LastChild as ContainerInline; - if (nextContainer != null && !nextContainer.IsClosed) - { - container = nextContainer; - } - else - { - break; - } - } - - container.AppendChild(nextInline); - } - - if (nextInline.IsClosable && !nextInline.IsClosed) - { - var inlinesToClose = inlineState.InlinesToClose; - var last = inlinesToClose.Count > 0 - ? inlineState.InlinesToClose[inlinesToClose.Count - 1] - : null; - if (last != nextInline) - { - inlineState.InlinesToClose.Add(nextInline); - } - } - } - else - { - // Get deepest container - var container = (ContainerInline)leafBlock.Inline; - while (true) - { - var nextContainer = container.LastChild as ContainerInline; - if (nextContainer != null && !nextContainer.IsClosed) - { - container = nextContainer; - } - else - { - break; - } - } - - inlineState.Inline = container.LastChild is LeafInline ? container.LastChild : container; - } - - if (Log != null) - { - Log.WriteLine($"** Dump: char '{c}"); - leafBlock.Inline.DumpTo(Log); - } - } - - // Close all inlines not closed - inlineState.Inline = null; - foreach (var inline in inlineState.InlinesToClose) - { - inline.CloseInternal(inlineState); - } - inlineState.InlinesToClose.Clear(); - - if (Log != null) - { - Log.WriteLine("** Dump before Emphasis:"); - leafBlock.Inline.DumpTo(Log); - EmphasisInline.ProcessEmphasis(leafBlock.Inline); - - Log.WriteLine(); - Log.WriteLine("** Dump after Emphasis:"); - leafBlock.Inline.DumpTo(Log); - } - // TODO: Close opened inlines - - // Close last inline - //while (inlineStack.Count > 0) - //{ - // var inlineState = inlineStack.Pop(); - // inlineState.Parser.Close(state, inlineState.Inline); - //} - } - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/MatchLineResult.cs b/src/Textamina.Markdig/Parsing/MatchLineResult.cs deleted file mode 100644 index dfe38270..00000000 --- a/src/Textamina.Markdig/Parsing/MatchLineResult.cs +++ /dev/null @@ -1,17 +0,0 @@ -namespace Textamina.Markdig.Parsing -{ - public enum MatchLineResult - { - None, - - Skip, - - Continue, - - ContinueDiscard, - - Last, - - LastDiscard - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/NewBlockParserState.cs b/src/Textamina.Markdig/Parsing/NewBlockParserState.cs deleted file mode 100644 index 17ee6370..00000000 --- a/src/Textamina.Markdig/Parsing/NewBlockParserState.cs +++ /dev/null @@ -1,119 +0,0 @@ - - - -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Syntax; - -namespace Textamina.Markdig.Parsing -{ - public interface IBlockParser - { - char[] OpeningCharacters { get; } - - BlockState TryOpen(BlockParserState state); - - BlockState TryContinue(BlockParserState state); - - bool Close(BlockParserState state, Block block); - } - - public abstract class NewBlockParser : IBlockParser - { - public char[] OpeningCharacters { get; protected set; } - - public abstract BlockState TryOpen(BlockParserState state); - - public virtual BlockState TryContinue(BlockParserState state) - { - // By default we don't expect any newline - return BlockState.None; - } - - public virtual bool Close(BlockParserState state, Block block) - { - // By default keep the block - return true; - } - } - - - public class ThematicBreakBlockParser : NewBlockParser - { - public ThematicBreakBlockParser() - { - OpeningCharacters = new [] {'-', '_', '*'}; - } - - public override BlockState TryOpen(BlockParserState state) - { - var liner = state.Line; - // 4.1 Thematic breaks - // A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces - var c = liner.Current; - - var matchChar = liner.Current; - var count = 1; - c = liner.NextChar(); - bool hasSpacesSinceLastMatch = false; - bool hasInnerSpaces = false; - while (c != '\0') - { - if (c == matchChar) - { - if (hasSpacesSinceLastMatch) - { - hasInnerSpaces = true; - } - - count++; - } - else if (!c.IsSpace()) - { - return BlockState.None; - } - else if (c.IsSpace()) - { - hasSpacesSinceLastMatch = true; - } - c = liner.NextChar(); - } - - // If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it - var previousParagraph = state.LastBlock as ParagraphBlock; - - var isSetexHeading = previousParagraph != null && matchChar == '-' && !hasInnerSpaces; - if (isSetexHeading) - { - var parent = previousParagraph.Parent; - if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != column)) - { - isSetexHeading = false; - } - } - - if (count < 3 || isSetexHeading) - { - return BlockState.None; - } - - state.NewBlocks.Push(new BreakBlock(this) {Column = column}); - return BlockState.LastDiscard; - } - } - - - public enum BlockState - { - None, - - Skip, - - Continue, - - ContinueDiscard, - - Last, - - LastDiscard - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/BlankLineBlock.cs b/src/Textamina.Markdig/Syntax/BlankLineBlock.cs index 48be6693..6b685f66 100644 --- a/src/Textamina.Markdig/Syntax/BlankLineBlock.cs +++ b/src/Textamina.Markdig/Syntax/BlankLineBlock.cs @@ -1,6 +1,4 @@ -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax { public sealed class BlankLineBlock : Block { diff --git a/src/Textamina.Markdig/Syntax/Block.cs b/src/Textamina.Markdig/Syntax/Block.cs index 16f481ba..db1468a5 100644 --- a/src/Textamina.Markdig/Syntax/Block.cs +++ b/src/Textamina.Markdig/Syntax/Block.cs @@ -1,7 +1,4 @@ -using System; -using System.Collections.Generic; -using System.Text; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { diff --git a/src/Textamina.Markdig/Syntax/CodeBlock.cs b/src/Textamina.Markdig/Syntax/CodeBlock.cs index 3cac7a8b..e74153fb 100644 --- a/src/Textamina.Markdig/Syntax/CodeBlock.cs +++ b/src/Textamina.Markdig/Syntax/CodeBlock.cs @@ -1,4 +1,4 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { @@ -10,32 +10,9 @@ namespace Textamina.Markdig.Syntax /// public class CodeBlock : LeafBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public CodeBlock(BlockParser parser) : base(parser) { - NoInline = true; - } - - private class ParserInternal : BlockParser - { - public ParserInternal() - { - CanInterruptParagraph = false; - } - - public override MatchLineResult Match(BlockParserState state) - { - if (!state.IsCodeIndent || state.IsBlankLine) - { - return state.IsBlankLine && state.Pending != null ? MatchLineResult.LastDiscard : MatchLineResult.None; - } - if (state.Pending == null) - { - state.NewBlocks.Push(new CodeBlock(this) { Column = state.Column }); - } - return MatchLineResult.Continue; - } + ProcessInlines = false; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/ContainerBlock.cs b/src/Textamina.Markdig/Syntax/ContainerBlock.cs index 047a5a71..6d611bf6 100644 --- a/src/Textamina.Markdig/Syntax/ContainerBlock.cs +++ b/src/Textamina.Markdig/Syntax/ContainerBlock.cs @@ -1,10 +1,10 @@ using System.Collections.Generic; using System.Diagnostics; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { - [DebuggerDisplay("Container: {GetType().Name} Count = {Children.Count}")] + [DebuggerDisplay("{GetType().Name} Count = {Children.Count}")] public abstract class ContainerBlock : Block { protected ContainerBlock(BlockParser parser) : base(parser) @@ -12,6 +12,7 @@ namespace Textamina.Markdig.Syntax Children = new List(); } + // TODO: Remove Children and use only inner list public List Children { get; } public Block LastChild => Children.Count > 0 ? Children[Children.Count - 1] : null; diff --git a/src/Textamina.Markdig/Syntax/FencedCodeBlock.cs b/src/Textamina.Markdig/Syntax/FencedCodeBlock.cs index 1e7c3100..00e30789 100644 --- a/src/Textamina.Markdig/Syntax/FencedCodeBlock.cs +++ b/src/Textamina.Markdig/Syntax/FencedCodeBlock.cs @@ -1,8 +1,4 @@ - - - -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { @@ -14,8 +10,6 @@ namespace Textamina.Markdig.Syntax /// public class FencedCodeBlock : CodeBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public FencedCodeBlock(BlockParser parser) : base(parser) { } @@ -24,166 +18,10 @@ namespace Textamina.Markdig.Syntax public string Arguments { get; set; } - private int fencedCharCount; + public int FencedCharCount { get; set; } - private char fencedChar; + public char FencedChar { get; set; } - private int indentCount; - - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - int count; - char matchChar; - char c = state.CurrentChar; - int offset = 0; - - var currentFenced = state.Pending as FencedCodeBlock; - if (currentFenced != null) - { - count = currentFenced.fencedCharCount; - matchChar = currentFenced.fencedChar; - while (c == matchChar) - { - offset++; - c = state.Line.PeekChar(offset); - } - - if (offset >= count) - { - state.Line.TrimEnd(true); - if (state.CurrentChar == matchChar) - { - return MatchLineResult.LastDiscard; - } - } - - // TODO: It is unclear how to handle this correctly - // Break only if Eof - return MatchLineResult.Continue; - } - - // Else if the we have an indent, it is not valid - if (state.IsCodeIndent) - { - return MatchLineResult.None; - } - - count = 0; - matchChar = (char) 0; - while (c != '\0') - { - if (count == 0 && (c == '`' || c == '~')) - { - matchChar = c; - } - else if (c != matchChar) - { - break; - } - count++; - c = state.PeekChar(count); - } - - if (count >= 3) - { - return MatchLineResult.None; - } - - // TODO: We need to count the number of leading space to remove them on each line - var column = state.Column; - - // specs spaces: Is space and tabs? or only spaces? Use space and tab for this case - while (c.IsSpaceOrTab()) - { - offset++; - c = state.PeekChar(offset); - } - var start = state.Start + count + offset; - - string infoString; - string argString = null; - - // An info string cannot contain any backsticks - int firstSpace = -1; - for (int i = start; i <= state.EndOffset; i++) - { - c = state.Line[i]; - if (c == '`') - { - return MatchLineResult.None; - } - - if (firstSpace < 0 && c.IsSpaceOrTab()) - { - firstSpace = i; - } - } - - if (firstSpace > 0) - { - infoString = state.Line.Text.Substring(start, firstSpace - start); - - // Skip any spaces after info string - firstSpace++; - while (true) - { - c = state.Line[firstSpace]; - if (c.IsSpaceOrTab()) - { - firstSpace++; - } - else - { - break; - } - } - - argString = state.Line.Text.Substring(firstSpace, state.Line.End - firstSpace + 1); - } - else - { - infoString = state.Line.Text.Substring(start, state.EndOffset - start + 1); - } - - // Store the number of matched string into the context - state.NewBlocks.Push(new FencedCodeBlock(this) - { - Column = column, - fencedChar = matchChar, - fencedCharCount = count, - indentCount = state.Indent, - Language = HtmlHelper.Unescape(infoString), - Arguments = HtmlHelper.Unescape(argString), - }); - - // Discard the current line - return MatchLineResult.ContinueDiscard; - } - - public override void Close(BlockParserState state) - { - var fenced = ((FencedCodeBlock) state.Pending); - var lines = fenced.Lines; - for (int i = 0; i < lines.Count; i++) - { - // Fences can be indented. If the opening fence is indented, - // content lines will have equivalent opening indentation removed, if present: - for (int j = 0; j < fenced.indentCount; j++) - { - var start = lines.Slices[i].Start; - if (start < lines.Slices[i].End && lines.Slices[i][start].IsSpace()) - { - lines.Slices[i].Start++; - } - else - { - break; - } - } - } - } - } + public int IndentCount { get; set; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/HeadingBlock.cs b/src/Textamina.Markdig/Syntax/HeadingBlock.cs index f046d711..1bf42133 100644 --- a/src/Textamina.Markdig/Syntax/HeadingBlock.cs +++ b/src/Textamina.Markdig/Syntax/HeadingBlock.cs @@ -1,110 +1,18 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using System.Diagnostics; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { /// /// Repressents a thematic break. /// + [DebuggerDisplay("{GetType().Name} Line: {Line}, {Lines} Level: {Level}")] public class HeadingBlock : LeafBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public HeadingBlock(BlockParser parser) : base(parser) { } public int Level { get; set; } - - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - if (state.IsCodeIndent) - { - return MatchLineResult.None; - } - - // 4.2 ATX headings - // An ATX heading consists of a string of characters, parsed as inline content, - // between an opening sequence of 1–6 unescaped # characters and an optional - // closing sequence of any number of unescaped # characters. The opening sequence - // of # characters must be followed by a space or by the end of line. The optional - // closing sequence of #s must be preceded by a space and may be followed by spaces - // only. The opening # character may be indented 0-3 spaces. The raw contents of - // the heading are stripped of leading and trailing spaces before being parsed as - // inline content. The heading level is equal to the number of # characters in the - // opening sequence. - var column = state.Column; - var c = state.CurrentChar; - - int leadingCount = 0; - for (; !state.Line.IsEndOfSlice && leadingCount <= 6; leadingCount++) - { - if (c != '#') - { - break; - } - - c = state.PeekChar(leadingCount); - } - - // closing # will be handled later, because anyway we have matched - - // A space is required after leading # - if (leadingCount > 0 && leadingCount <=6 && (c.IsSpace() || state.Line.IsEndOfSlice)) - { - state.Line.Start = leadingCount + 1; - state.NewBlocks.Push(new HeadingBlock(this) {Level = leadingCount, Column = column }); - - // The optional closing sequence of #s must be preceded by a space and may be followed by spaces only. - int endState = 0; - int countClosingTags = 0; - for (int i = state.Line.End; i >= state.Line.Start - 1; i--) // Go up to Start - 1 in order to match the space after the first ### - { - c = state.Line[i]; - if (endState == 0) - { - if (c.IsSpace()) // TODO: Not clear if it is a space or space+tab in the specs - { - continue; - } - endState = 1; - } - if (endState == 1) - { - if (c == '#') - { - countClosingTags++; - continue; - } - - if (countClosingTags > 0) - { - if (c.IsSpace()) - { - state.Line.End = i - 1; - } - break; - } - else - { - break; - } - } - } - - return MatchLineResult.Last; - } - - return MatchLineResult.None; - } - - public override void Close(BlockParserState state) - { - var heading = (HeadingBlock) state.Pending; - heading.Lines.Trim(); - } - } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/HtmlBlock.cs b/src/Textamina.Markdig/Syntax/HtmlBlock.cs index 6becd456..ac270140 100644 --- a/src/Textamina.Markdig/Syntax/HtmlBlock.cs +++ b/src/Textamina.Markdig/Syntax/HtmlBlock.cs @@ -1,309 +1,17 @@ -using System; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { public class HtmlBlock : LeafBlock { - public static readonly BlockParser Parser = new ParserInternal(); + public static readonly BlockParser Parser = new HtmlBlockParser(); public HtmlBlock(BlockParser parser) : base(parser) { // We don't process inline of an html block, as we will copy the content as-is - NoInline = true; + ProcessInlines = false; } public HtmlBlockType Type { get; set; } - - private class ParserInternal : BlockParser - { - private static readonly string[] HtmlTags = - { - "address", // 0 - "article", // 1 - "aside", // 2 - "base", // 3 - "basefont", // 4 - "blockquote", // 5 - "body", // 6 - "caption", // 7 - "center", // 8 - "col", // 9 - "colgroup", // 10 - "dd", // 11 - "details", // 12 - "dialog", // 13 - "dir", // 14 - "div", // 15 - "dl", // 16 - "dt", // 17 - "fieldset", // 18 - "figcaption", // 19 - "figure", // 20 - "footer", // 21 - "form", // 22 - "frame", // 23 - "frameset", // 24 - "h1", // 25 - "head", // 26 - "header", // 27 - "hr", // 28 - "html", // 29 - "iframe", // 30 - "legend", // 31 - "li", // 32 - "link", // 33 - "main", // 34 - "menu", // 35 - "menuitem", // 36 - "meta", // 37 - "nav", // 38 - "noframes", // 39 - "ol", // 40 - "optgroup", // 41 - "option", // 42 - "p", // 43 - "param", // 44 - "pre", // 45 <- special group 1 - "script", // 46 <- special group 1 - "section", // 47 - "source", // 48 - "style", // 49 <- special group 1 - "summary", // 50 - "table", // 51 - "tbody", // 52 - "td", // 53 - "tfoot", // 54 - "th", // 55 - "thead", // 56 - "title", // 57 - "tr", // 58 - "track", // 59 - "ul", // 60 - }; - - public override MatchLineResult Match(BlockParserState state) - { - var htmlBlock = state.Pending as HtmlBlock; - if (htmlBlock == null) - { - var result = MatchStart(state); - // An end-tag can occur on the same line - if (result == MatchLineResult.Continue) - { - return MatchEnd(state, (HtmlBlock) state.NewBlocks.Peek()); - } - return result; - } - - return MatchEnd(state, htmlBlock); - } - - private MatchLineResult MatchStart(BlockParserState state) - { - int index = 0; - - for (int i = 0; i < 3; i++) - { - if (!state.Line.PeekChar(index).IsSpace()) - { - break; - } - index++; - } - - // Early exit if it is not starting by an HTML tag - var column = index; - var c = state.Line.PeekChar(index++); - if (c != '<') - { - return MatchLineResult.None; - } - - var result = TryParseTagType16(state, ref state.Line, index, column); - - // HTML blocks of type 7 cannot interrupt a paragraph: - if (result == MatchLineResult.None && !(state.LastBlock is ParagraphBlock)) - { - result = TryParseTagType7(state, ref state.Line, index, column); - } - - return result; - } - - private MatchLineResult TryParseTagType7(BlockParserState state, ref StringSlice liner, int index, int startColumn) - { - var builder = StringBuilderCache.Local(); - liner.Start = index; - var c = liner.CurrentChar; - var result = MatchLineResult.None; - if ((c == '/' && HtmlHelper.TryParseHtmlCloseTag(liner, builder)) || HtmlHelper.TryParseHtmlTagOpenTag(liner, builder)) - { - // Must be followed by whitespace only - bool hasOnlySpaces = true; - c = liner.CurrentChar; - while (true) - { - if (c == '\0') - { - break; - } - if (!c.IsWhitespace()) - { - hasOnlySpaces = false; - break; - } - c = liner.NextChar(); - } - - if (hasOnlySpaces) - { - result = CreateHtmlBlock(state, HtmlBlockType.NonInterruptingBlock, startColumn); - } - } - - builder.Clear(); - return result; - } - - private MatchLineResult TryParseTagType16(BlockParserState state, ref StringSlice liner, int index, int startColumn) - { - char c; - c = liner.PeekChar(index); - if (c == '!') - { - c = liner.PeekChar(index + 1); - if (c == '-' && liner.PeekChar(index + 2) == '-') - { - return CreateHtmlBlock(state, HtmlBlockType.Comment, startColumn); // group 2 - } - if (c.IsAlphaUpper()) - { - return CreateHtmlBlock(state, HtmlBlockType.DocumentType, startColumn); // group 4 - } - if (c == '[' && liner.Match("CDATA[", 3)) - { - return CreateHtmlBlock(state, HtmlBlockType.CData, startColumn); // group 5 - } - - return MatchLineResult.None; - } - - if (c == '?') - { - return CreateHtmlBlock(state, HtmlBlockType.ProcessingInstruction, startColumn); // group 3 - } - - var hasLeadingClose = c == '/'; - if (hasLeadingClose) - { - index++; - } - - var tag = new char[10]; - var count = 0; - for (; count < tag.Length; index++, count++) - { - c = liner.PeekChar(index); - if (!c.IsAlphaNumeric()) - { - break; - } - tag[count] = char.ToLowerInvariant(c); - } - - if ( - !(c == '>' || (!hasLeadingClose && c == '/' && liner.PeekChar(index + 1) == '>') || c.IsWhitespace() || - c == '\0')) - { - return MatchLineResult.None; - } - - if (count == 0) - { - return MatchLineResult.None; - } - - var tagName = new string(tag, 0, count); - var tagIndex = Array.BinarySearch(HtmlTags, tagName, StringComparer.Ordinal); - if (tagIndex < 0) - { - return MatchLineResult.None; - } - - // Cannot start with ")) - { - return MatchLineResult.Last; - } - break; - case HtmlBlockType.CData: - if (state.Line.Search("]]>")) - { - return MatchLineResult.Last; - } - break; - case HtmlBlockType.ProcessingInstruction: - if (state.Line.Search("?>")) - { - return MatchLineResult.Last; - } - break; - case HtmlBlockType.DocumentType: - if (state.Line.Search(">")) - { - return MatchLineResult.Last; - } - break; - case HtmlBlockType.ScriptPreOrStyle: - // TODO: could be optimized with a dedicated parser - if (state.Line.SearchLowercase("") || state.Line.SearchLowercase("") || state.Line.SearchLowercase("")) - { - return MatchLineResult.Last; - } - break; - case HtmlBlockType.InterruptingBlock: - if (state.IsBlankLine) - { - return MatchLineResult.LastDiscard; - } - break; - case HtmlBlockType.NonInterruptingBlock: - if (state.IsBlankLine) - { - return MatchLineResult.LastDiscard; - } - break; - } - - return MatchLineResult.Continue; - } - - private MatchLineResult CreateHtmlBlock(BlockParserState state, HtmlBlockType type, int startColumn) - { - state.NewBlocks.Push(new HtmlBlock(this) {Column = startColumn, Type = type}); - return MatchLineResult.Continue; - } - } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/Inlines/AutoLinkInline.cs b/src/Textamina.Markdig/Syntax/Inlines/AutoLinkInline.cs index 784acf5a..6c782e58 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/AutoLinkInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/AutoLinkInline.cs @@ -1,48 +1,16 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; +using Textamina.Markdig.Parsers.Inlines; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class AutolinkInline : LeafInline { - public static readonly InlineParser Parser = new ParserInternal(); + public static readonly InlineParser Parser = new AutolineInlineParser(); public bool IsEmail { get; set; } public string Url { get; set; } - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] {'<'}; - } - - public override bool Match(InlineParserState state) - { - string link; - bool isEmail; - var saved = state.Text; - if (LinkHelper.TryParseAutolink(ref state.Text, out link, out isEmail)) - { - state.Inline = new AutolinkInline() {IsEmail = isEmail, Url = link}; - } - else - { - state.Text = saved; - string htmlTag; - if (!HtmlHelper.TryParseHtmlTag(ref state.Text, out htmlTag)) - { - return false; - } - - state.Inline = new HtmlInline() { Tag = htmlTag }; - } - - return true; - } - } - public override string ToString() { return Url; diff --git a/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs b/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs index d05cd209..bcfcb746 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs @@ -1,105 +1,12 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; +using Textamina.Markdig.Parsers.Inlines; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class CodeInline : LeafInline { - public static readonly InlineParser Parser = new ParserInternal(); + public static readonly InlineParser Parser = new CodeInlineParser(); public string Content { get; set; } - - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] { '`' }; - } - - public override bool Match(InlineParserState state) - { - var text = state.Text; - - int openSticks = 0; - if (text.PeekChar(-1) == '`') - { - return false; - } - - while (text.CurrentChar == '`') - { - openSticks++; - text.NextChar(); - } - - bool isMatching = false; - - var builder = state.StringBuilders.Get(); - int closeSticks = 0; - var c = text.CurrentChar; - - // A backtick string is a string of one or more backtick characters (`) that is neither preceded nor followed by a backtick. - // A code span begins with a backtick string and ends with a backtick string of equal length. - // The contents of the code span are the characters between the two backtick strings, with leading and trailing spaces and line endings removed, and whitespace collapsed to single spaces. - var pc = ' '; - - while (c != '\0') - { - // Transform '\n' into a single space - if (c == '\n') - { - c = ' '; - } - - if (c != '`' && (c != ' ' || pc != ' ')) - { - builder.Append(c); - } - else - { - while (c == '`') - { - closeSticks++; - pc = c; - c = text.NextChar(); - } - - if (openSticks == closeSticks) - { - break; - } - } - - if (closeSticks > 0) - { - builder.Append('`', closeSticks); - closeSticks = 0; - } - else - { - pc = c; - c = text.NextChar(); - } - } - - if (closeSticks == openSticks) - { - // Remove trailing space - if (builder.Length > 0) - { - if (builder[builder.Length - 1].IsWhitespace()) - { - builder.Length--; - } - } - state.Inline = new CodeInline() { Content = builder.ToString() }; - isMatching = true; - } - - // Release the builder if not used - state.StringBuilders.Release(builder); - return isMatching; - } - } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/Inlines/ContainerInline.cs b/src/Textamina.Markdig/Syntax/Inlines/ContainerInline.cs index 540ced5e..98f319ce 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/ContainerInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/ContainerInline.cs @@ -1,9 +1,8 @@ using System; using System.Collections.Generic; using System.IO; -using Textamina.Markdig.Parsing; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class ContainerInline : Inline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/DelimiterInline.cs b/src/Textamina.Markdig/Syntax/Inlines/DelimiterInline.cs index 8464c55b..04484a87 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/DelimiterInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/DelimiterInline.cs @@ -1,8 +1,7 @@ using System; -using System.Collections.Generic; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public abstract class DelimiterInline : ContainerInline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/DelimiterType.cs b/src/Textamina.Markdig/Syntax/Inlines/DelimiterType.cs index c1d98348..b43f1f51 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/DelimiterType.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/DelimiterType.cs @@ -1,6 +1,6 @@ using System; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { [Flags] public enum DelimiterType diff --git a/src/Textamina.Markdig/Syntax/Inlines/EmphasisDelimiterInline.cs b/src/Textamina.Markdig/Syntax/Inlines/EmphasisDelimiterInline.cs index e45d3ced..27cfc9ac 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/EmphasisDelimiterInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/EmphasisDelimiterInline.cs @@ -1,7 +1,6 @@ -using System.Collections.Generic; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class EmphasisDelimiterInline : DelimiterInline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs b/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs index 063e604b..da57bc48 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs @@ -1,14 +1,12 @@ -using System; using System.Collections.Generic; -using System.Linq; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; +using Textamina.Markdig.Parsers.Inlines; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class EmphasisInline : ContainerInline { - public static readonly InlineParser Parser = new ParserInternal(); + public static readonly InlineParser Parser = new EmphasisInlineParser(); public char DelimiterChar { get; set; } @@ -205,111 +203,5 @@ namespace Textamina.Markdig.Syntax } delimiters.Clear(); } - - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] { '*', '_' }; - } - - public override bool Match(InlineParserState state) - { - // First, some definitions. A delimiter run is either a sequence of one or more * characters that - // is not preceded or followed by a * character, or a sequence of one or more _ characters that - // is not preceded or followed by a _ character. - - var delimiterChar = state.Text.CurrentChar; - var pc = state.Text.PeekChar(-1); - if (delimiterChar == pc && state.Text.PeekChar(-2) != '\\') - { - return false; - } - - int delimiterCount = 0; - char c; - do - { - delimiterCount++; - c = state.Text.NextChar(); - } while (c == delimiterChar); - - - // A left-flanking delimiter run is a delimiter run that is - // (a) not followed by Unicode whitespace, and - // (b) either not followed by a punctuation character, or preceded by Unicode whitespace - // or a punctuation character. - // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. - bool nextIsPunctuation; - bool nextIsWhiteSpace; - bool prevIsPunctuation; - bool prevIsWhiteSpace; - pc.CheckUnicodeCategory(out prevIsWhiteSpace, out prevIsPunctuation); - c.CheckUnicodeCategory(out nextIsWhiteSpace, out nextIsPunctuation); - - bool canOpen = !nextIsWhiteSpace && - (!nextIsPunctuation || prevIsWhiteSpace || prevIsPunctuation); - - - // A right-flanking delimiter run is a delimiter run that is - // (a) not preceded by Unicode whitespace, and - // (b) either not preceded by a punctuation character, or followed by Unicode whitespace - // or a punctuation character. - // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. - bool canClose = !prevIsWhiteSpace && - (!prevIsPunctuation || nextIsWhiteSpace || nextIsPunctuation); - - if (delimiterChar == '_') - { - var temp = canOpen; - // A single _ character can open emphasis iff it is part of a left-flanking delimiter run and either - // (a) not part of a right-flanking delimiter run or - // (b) part of a right-flanking delimiter run preceded by punctuation. - canOpen = canOpen && (!canClose || prevIsPunctuation); - - // A single _ character can close emphasis iff it is part of a right-flanking delimiter run and either - // (a) not part of a left-flanking delimiter run or - // (b) part of a left-flanking delimiter run followed by punctuation. - canClose = canClose && (!temp || nextIsPunctuation); - } - - //// If we can close, try to find a matching open - //if (canClose && state.Inline != null) - //{ - // var matching = DelimiterInline.FindMatchingOpen(state.Inline, 0, delimiterRun, delimiterCount); - - // // transform matching into - - // return true; - //} - - // We have potentially an open or close emphasis - if (canOpen || canClose) - { - var delimiterType = DelimiterType.None; - if (canOpen) - { - delimiterType |= DelimiterType.Open; - } - if (canClose) - { - delimiterType |= DelimiterType.Close; - } - - var delimiter = new EmphasisDelimiterInline(this) - { - DelimiterChar = delimiterChar, - DelimiterCount = delimiterCount, - Type = delimiterType, - }; - - state.Inline = delimiter; - return true; - } - - // We don't have an emphasis - return false; - } - } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs b/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs deleted file mode 100644 index 0869d214..00000000 --- a/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs +++ /dev/null @@ -1,40 +0,0 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Syntax -{ - /// - /// There is actually no EscapeInline inheriting from Inline, as - /// the parser will transform it to a LiteralInline - /// - public static class EscapeInline - { - public static readonly InlineParser Parser = new ParserInternal(); - - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] {'\\'}; - } - - public override bool Match(InlineParserState state) - { - var lines = state.Text; - - // Go to escape character - lines.NextChar(); - if (lines.CurrentChar.IsAsciiPunctuation()) - { - var literal = state.Inline as LiteralInline ?? - new LiteralInline() {ContentBuilder = state.StringBuilders.Get()}; - literal.ContentBuilder.Append(lines.CurrentChar); - state.Inline = literal; - lines.NextChar(); - return true; - } - return false; - } - } - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/Inlines/HardlineBreakInline.cs b/src/Textamina.Markdig/Syntax/Inlines/HardlineBreakInline.cs index 322605fc..5bfdf530 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/HardlineBreakInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/HardlineBreakInline.cs @@ -1,51 +1,7 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class HardlineBreakInline : LeafInline { - public static readonly InlineParser Parser = new ParserInternal(); - - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] {'\n'}; - } - - public override bool Match(InlineParserState state) - { - // Hard line breaks are for separating inline content within a block. Neither syntax for hard line breaks works at the end of a paragraph or other block element: - if (!(state.Block is ParagraphBlock) || state.Text.Column == 0 || !state.Text.PeekChar(-1).IsSpace() || !state.Text.PeekChar(-2).IsSpace()) - { - return false; - } - - //// A line break (not in a code span or HTML tag) that is preceded by two or more spaces - //// and does not occur at the end of a block is parsed as a hard line break (rendered in HTML as a
tag) - //var text = state.Lines; - //int spaceCount = 0; - //var c = text.CurrentChar; - //while (c.IsSpaceOrTab()) - //{ - // c = text.NextChar(); - // spaceCount++; - //} - //if (c == '\\') - //{ - // c = text.NextChar(); - // spaceCount = 2; - //} - //if (c != '\n' || spaceCount < 2) - //{ - // return false; - //} - - return false; - } - } - public override string ToString() { return "
"; diff --git a/src/Textamina.Markdig/Syntax/Inlines/HtmlInline.cs b/src/Textamina.Markdig/Syntax/Inlines/HtmlInline.cs index 33f8f369..9b51cb93 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/HtmlInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/HtmlInline.cs @@ -1,4 +1,4 @@ -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class HtmlInline : LeafInline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/Inline.cs b/src/Textamina.Markdig/Syntax/Inlines/Inline.cs index b5e5aa97..7d578675 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/Inline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/Inline.cs @@ -1,9 +1,9 @@ using System; using System.Collections.Generic; using System.IO; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public abstract class Inline { @@ -111,7 +111,7 @@ namespace Textamina.Markdig.Syntax } else if (parent != null) { - ((ContainerInline) parent).AppendChild(inline); + parent.AppendChild(inline); } var container = this as ContainerInline; diff --git a/src/Textamina.Markdig/Syntax/Inlines/InlineStack.cs b/src/Textamina.Markdig/Syntax/Inlines/InlineStack.cs deleted file mode 100644 index 1dae5988..00000000 --- a/src/Textamina.Markdig/Syntax/Inlines/InlineStack.cs +++ /dev/null @@ -1,319 +0,0 @@ -/* -using System; -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Syntax -{ - /// - /// Describes an element in a stack of possible inline openers. - /// - internal sealed class InlineStack - { - /// - /// The parser priority if this stack entry. - /// - public InlineStackPriority Priority; - - /// - /// Previous entry in the stack. null if this is the last one. - /// - public InlineStack Previous; - - /// - /// Next entry in the stack. null if this is the last one. - /// - public InlineStack Next; - - /// - /// The at-the-moment text inline that could be transformed into the opener. - /// - public Inline StartingInline; - - /// - /// The number of delimiter characters found for this opener. - /// - public int DelimiterCount; - - /// - /// The character that was used in the opener. - /// - public char Delimiter; - - /// - /// The position in the where this inline element was found. - /// Used only if the specific parser requires this information. - /// - public StringLineGroup.BlockState StartPosition; - - /// - /// The flags set for this stack entry. - /// - public InlineStackFlags Flags; - - [Flags] - public enum InlineStackFlags : byte - { - None = 0, - Opener = 1, - Closer = 2, - ImageLink = 4 - } - - public enum InlineStackPriority : byte - { - Emphasis = 0, - Links = 1, - Maximum = Links - } - - public InlineStack FindMatchingOpener(InlineStackPriority priority, - char delimiter, out bool canClose) - { - canClose = true; - var istack = this; - while (true) - { - if (istack == null) - { - // this cannot be a closer since there is no opener available. - canClose = false; - return null; - } - - if (istack.Priority > priority || - (istack.Delimiter == delimiter && 0 != (istack.Flags & InlineStackFlags.Closer))) - { - // there might be a closer further back but we cannot go there yet because a higher priority element is blocking - // the other option is that the stack entry could be a closer for the same char - this means - // that any opener we might find would first have to be matched against this closer. - return null; - } - - if (istack.Delimiter == delimiter) - return istack; - - istack = istack.Previous; - } - } - - public void AppendStackEntry(InlineParserState subj) - { - if (subj.LastPendingInline != null) - { - Previous = subj.LastPendingInline; - subj.LastPendingInline.Next = this; - } - - if (subj.FirstPendingInline == null) - subj.FirstPendingInline = this; - - subj.LastPendingInline = this; - } - - /// - /// Removes a subset of the stack. - /// - /// The subject associated with this stack. Can be null if the pointers in the subject should not be updated. - /// The last entry to be removed. Can be null if everything starting from has to be removed. - public void RemoveStackEntry(InlineParserState subj, InlineStack last) - { - var first = this; - var curPriority = first.Priority; - - if (last == null) - { - if (first.Previous != null) - first.Previous.Next = null; - else if (subj != null) - subj.FirstPendingInline = null; - - if (subj != null) - { - last = subj.LastPendingInline; - subj.LastPendingInline = first.Previous; - } - - first = first.Next; - } - else - { - if (first.Previous != null) - first.Previous.Next = last.Next; - else if (subj != null) - subj.FirstPendingInline = last.Next; - - if (last.Next != null) - last.Next.Previous = first.Previous; - else if (subj != null) - subj.LastPendingInline = first.Previous; - - if (first == last) - return; - - first = first.Next; - last = last.Previous; - } - - if (last == null || first == null) - return; - - first.Previous = null; - last.Next = null; - - // handle case like [*b*] (the whole [..] is being removed but the inner *..* must still be matched). - // this is not done automatically because the initial * is recognized as a potential closer (assuming - // potential scenario '*[*' ). - if (curPriority > 0) - PostProcessInlineStack(null, first, last, curPriority); - } - - public static void PostProcessInlineStack(InlineParserState subj, InlineStack first, InlineStack last, - InlineStackPriority ignorePriority) - { - while (ignorePriority > 0) - { - var istack = first; - while (istack != null) - { - if (istack.Priority >= ignorePriority) - { - istack.RemoveStackEntry(subj, istack); - } - else if (0 != (istack.Flags & InlineStackFlags.Closer)) - { - bool canClose; - var iopener = FindMatchingOpener(istack.Previous, istack.Priority, istack.Delimiter, - out canClose); - if (iopener != null) - { - bool retry = false; - if (iopener.Delimiter == '~') - { - iopener.MatchInlineStack(subj, istack.DelimiterCount, istack, true); - if (istack.DelimiterCount > 1) - retry = true; - } - else - { - iopener.MatchInlineStack(subj, istack.DelimiterCount, istack, false); - if (istack.DelimiterCount > 0) - retry = true; - } - - if (retry) - { - // remove everything between opened and closer (not inclusive). - if (istack.Previous != null && iopener.Next != istack.Previous) - iopener.Next.RemoveStackEntry(subj, istack.Previous); - - continue; - } - else - { - // remove opener, everything in between, and the closer - iopener.RemoveStackEntry(subj, istack); - } - } - else if (!canClose) - { - // this case means that a matching opener does not exist - // remove the Closer flag so that a future Opener can be matched against it. - istack.Flags &= ~InlineStackFlags.Closer; - } - } - - if (istack == last) - break; - - istack = istack.Next; - } - - ignorePriority--; - } - } - - - public int MatchInlineStack(InlineParserState subj, int closingDelimiterCount, InlineStack closer, bool onlySingleCharTag) - { - // calculate the actual number of delimiters used from this closer - int useDelims; - var openerDelims = this.DelimiterCount; - - if (closingDelimiterCount < 3 || openerDelims < 3) - { - useDelims = closingDelimiterCount <= openerDelims ? closingDelimiterCount : openerDelims; - if (useDelims == 1 && onlySingleCharTag) - return 0; - } - else if (onlySingleCharTag) - useDelims = 2; - else - useDelims = closingDelimiterCount % 2 == 0 ? 2 : 1; - - Inline inl = this.StartingInline; - InlineTag tag = useDelims == 1 ? singleCharTag : doubleCharTag; - if (openerDelims == useDelims) - { - // the opener is completely used up - remove the stack entry and reuse the inline element - inl.Tag = tag; - inl.LiteralContent = null; - inl.FirstChild = inl.NextSibling; - inl.NextSibling = null; - - RemoveStackEntry(subj, closer?.Previous); - } - else - { - // the opener will only partially be used - stack entry remains (truncated) and a new inline is added. - this.DelimiterCount -= useDelims; - inl.LiteralContent = inl.LiteralContent.Substring(0, this.DelimiterCount); - inl.SourceLastPosition -= useDelims; - - inl.NextSibling = new Inline(tag, inl.NextSibling); - inl = inl.NextSibling; - - inl.SourcePosition = this.StartingInline.SourcePosition + this.DelimiterCount; - } - - // there are two callers for this method, distinguished by the `closer` argument. - // if closer == null it means the method is called during the initial subject parsing and the closer - // characters are at the current position in the subject. The main benefit is that there is nothing - // parsed that is located after the matched inline element. - // if closer != null it means the method is called when the second pass for previously unmatched - // stack elements is done. The drawback is that there can be other elements after the closer. - if (closer != null) - { - var clInl = closer.StartingInline; - if ((closer.DelimiterCount -= useDelims) > 0) - { - // a new inline element must be created because the old one has to be the one that - // finalizes the children of the emphasis - var newCloserInline = new Inline(clInl.LiteralContent.Substring(useDelims)); - newCloserInline.SourcePosition = inl.SourceLastPosition = clInl.SourcePosition + useDelims; - newCloserInline.SourceLength = closer.DelimiterCount; - newCloserInline.NextSibling = clInl.NextSibling; - - clInl.LiteralContent = null; - clInl.NextSibling = null; - inl.NextSibling = closer.StartingInline = newCloserInline; - } - else - { - inl.SourceLastPosition = clInl.SourceLastPosition; - - clInl.LiteralContent = null; - inl.NextSibling = clInl.NextSibling; - clInl.NextSibling = null; - } - } - else if (subj != null) - { - inl.SourceLastPosition = subj.Position - closingDelimiterCount + useDelims; - subj.LastInline = inl; - } - - return useDelims; - } - } -} -*/ \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/Inlines/LeafInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LeafInline.cs index 0f54ccd2..bdbaa795 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LeafInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LeafInline.cs @@ -1,6 +1,4 @@ -using System.Text; - -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public abstract class LeafInline : Inline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/LineBreakInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LineBreakInline.cs index 5114a860..874f887c 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LineBreakInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LineBreakInline.cs @@ -1,4 +1,4 @@ -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class LineBreakInline : LeafInline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/LinkDelimiterInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LinkDelimiterInline.cs index df530f6a..78940a3b 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LinkDelimiterInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LinkDelimiterInline.cs @@ -1,6 +1,6 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class LinkDelimiterInline : DelimiterInline { diff --git a/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs index 8a001cc4..99cdad53 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs @@ -1,15 +1,7 @@ -using System; -using System.Text; -using Textamina.Markdig.Formatters; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class LinkInline : ContainerInline { - public static readonly InlineParser Parser = new ParserInternal(); - public string Url { get; set; } public string Title { get; set; } @@ -20,251 +12,5 @@ namespace Textamina.Markdig.Syntax { return (IsImage ? ""; } - - private class ParserInternal : InlineParser - { - public ParserInternal() - { - FirstChars = new[] {'[', ']', '!'}; - } - - public override bool Match(InlineParserState state) - { - var c = state.Text.CurrentChar; - - bool isImage = false; - if (c == '!') - { - isImage = true; - c = state.Text.NextChar(); - if (c != '[') - { - return false; - } - } - - switch (c) - { - case '[': - // If this is not an image, we may have a reference link shortcut - // so we try to resolve it here - var saved = state.Text; - string label; - - // If the label is followed by either a ( or a [, this is not a shortcut - if (LinkHelper.TryParseLabel(ref state.Text, out label)) - { - if (!state.Document.LinkReferenceDefinitions.ContainsKey(label)) - { - label = null; - } - } - state.Text = saved; - - // Else we insert a LinkDelimiter - state.Text.NextChar(); - state.Inline = new LinkDelimiterInline(this) - { - Type = DelimiterType.Open, - Label = label, - IsImage = isImage - }; - return true; - - case ']': - state.Text.NextChar(); - if (state.Inline != null) - { - if (TryProcessLinkOrImage(state, ref state.Text)) - { - return true; - } - } - - // If we don’t find one, we return a literal text node ]. - // (Done after by the LiteralInline parser) - return false; - } - - // We don't have an emphasis - return false; - } - - private bool ProcessLinkReference(InlineParserState state, string label, bool isImage, Inline child = null) - { - bool isValidLink = false; - LinkReferenceDefinitionBlock linkRef; - if (state.Document.LinkReferenceDefinitions.TryGetValue(label, out linkRef)) - { - // Inline Link - var link = new LinkInline() - { - Url = HtmlHelper.Unescape(linkRef.Url), - Title = HtmlHelper.Unescape(linkRef.Title), - IsImage = isImage, - }; - - if (child == null) - { - child = new LiteralInline() - { - Content = label, - IsClosed = true - }; - link.AppendChild(child); - } - else - { - // Insert all child into the link - while (child != null) - { - var next = child.NextSibling; - child.Remove(); - link.AppendChild(child); - child = next; - } - } - link.IsClosed = true; - - EmphasisInline.ProcessEmphasis(link); - - state.Inline = link; - isValidLink = true; - } - //else - //{ - // // Else output a literal, leave it opened as we may have literals after - // // that could be append to this one - // var literal = new LiteralInline() - // { - // ContentBuilder = state.StringBuilders.Get().Append('[').Append(label).Append(']') - // }; - // state.Inline = literal; - //} - return isValidLink; - } - - private bool TryProcessLinkOrImage(InlineParserState inlineState, ref StringSlice text) - { - LinkDelimiterInline openParent = null; - foreach (var parent in inlineState.Inline.FindParentOfType()) - { - openParent = parent; - break; - } - - // This will be matched as a literal - if (openParent != null) - { - var parentDelimiter = openParent.Parent; - switch (text.CurrentChar) - { - case '(': - string url; - string title; - if (LinkHelper.TryParseInlineLink(ref text, out url, out title)) - { - // Inline Link - var link = new LinkInline() - { - Url = HtmlHelper.Unescape(url), - Title = HtmlHelper.Unescape(title), - IsImage = openParent.IsImage, - }; - - openParent.ReplaceBy(link); - inlineState.Inline = link; - - EmphasisInline.ProcessEmphasis(link); - - ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter); - - link.IsClosed = true; - - return true; - } - break; - default: - - string label = null; - // Handle Collapsed links - if (text.CurrentChar == '[') - { - if (text.PeekChar(1) == ']') - { - label = openParent.Label; - text.NextChar(); // Skip [ - text.NextChar(); // Skip ] - } - } - else - { - label = openParent.Label; - } - - if (label != null || LinkHelper.TryParseLabel(ref text, true, out label)) - { - if (ProcessLinkReference(inlineState, label, openParent.IsImage, - openParent.FirstChild)) - { - // Remove the open parent - openParent.Remove(); - ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter); - } - else - { - return false; - } - return true; - } - break; - } - - // We have a nested [ ] - // firstParent.Remove(); - // The opening [ will be transformed to a literal followed by all the childrens of the [ - - var literal = new LiteralInline() - { - ContentBuilder = inlineState.StringBuilders.Get().Append(openParent.IsImage ? "![" : "[") - }; - - inlineState.InlinesToClose.Add(literal); - inlineState.Inline = openParent.ReplaceBy(literal); - return false; - } - - return false; - } - - private void ReplaceParentIfNotImage(bool isImage, Inline inline) - { - if (isImage || inline == null) - { - return; - } - - foreach (var parent in inline.FindParentOfType()) - { - if (parent.IsImage) - { - break; - } - - var literal = new LiteralInline() - { - Content = "[", - IsClosed = true - }; - - parent.ReplaceBy(literal); - } - } - - private bool TryParseLinkTitle(InlineParserState state) - { - return false; - } - } } } diff --git a/src/Textamina.Markdig/Syntax/Inlines/LiteralInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LiteralInline.cs index fbd63cb6..06c984c9 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LiteralInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LiteralInline.cs @@ -1,12 +1,11 @@ using System.Text; using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class LiteralInline : LeafInline { - public static readonly InlineParser Parser = new ParserInternal(); public LiteralInline() { IsClosable = true; @@ -23,31 +22,6 @@ namespace Textamina.Markdig.Syntax ContentBuilder = null; } - private class ParserInternal : InlineParser - { - public override bool Match(InlineParserState state) - { - // A literal will always match - var literal = state.Inline as LiteralInline; - StringBuilder builder; - if (literal == null) - { - builder = state.StringBuilders.Get(); - literal = new LiteralInline {ContentBuilder = builder}; - state.Inline = literal; - } - else - { - builder = literal.ContentBuilder; - } - - var text = state.Text; - builder.Append(text.CurrentChar); - text.NextChar(); - return true; - } - } - public override string ToString() { return Content ?? (ContentBuilder != null ? ContentBuilder.ToString() : string.Empty); diff --git a/src/Textamina.Markdig/Syntax/Inlines/RawHtmlInline.cs b/src/Textamina.Markdig/Syntax/Inlines/RawHtmlInline.cs index a1b5c911..fc797838 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/RawHtmlInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/RawHtmlInline.cs @@ -1,7 +1,6 @@ -namespace Textamina.Markdig.Syntax +namespace Textamina.Markdig.Syntax.Inlines { public class RawHtmlInline : LeafInline { - } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/LeafBlock.cs b/src/Textamina.Markdig/Syntax/LeafBlock.cs index 1d6cb9b7..1fe514aa 100644 --- a/src/Textamina.Markdig/Syntax/LeafBlock.cs +++ b/src/Textamina.Markdig/Syntax/LeafBlock.cs @@ -1,18 +1,30 @@ -using Textamina.Markdig.Parsing; +using System.Diagnostics; +using Textamina.Markdig.Parsers; +using Textamina.Markdig.Syntax.Inlines; namespace Textamina.Markdig.Syntax { + [DebuggerDisplay("{GetType().Name} Line: {Line}, {Lines}")] public abstract class LeafBlock : Block { protected LeafBlock(BlockParser parser) : base(parser) { - Lines = new StringSliceList(); + ProcessInlines = false; } public StringSliceList Lines { get; set; } public Inline Inline { get; set; } - public bool NoInline { get; set; } + public bool ProcessInlines { get; set; } + + public void AppendLine(ref StringSlice line) + { + if (Lines == null) + { + Lines = new StringSliceList(); + } + Lines.Append(ref line); + } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/LinkReferenceDefinitionBlock.cs b/src/Textamina.Markdig/Syntax/LinkReferenceDefinitionBlock.cs index bba37aae..2bbb55c9 100644 --- a/src/Textamina.Markdig/Syntax/LinkReferenceDefinitionBlock.cs +++ b/src/Textamina.Markdig/Syntax/LinkReferenceDefinitionBlock.cs @@ -1,5 +1,4 @@ using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax { @@ -23,7 +22,7 @@ namespace Textamina.Markdig.Syntax public string Title { get; set; } - public static bool TryParse(ref StringSlice text, out LinkReferenceDefinitionBlock block) + public static bool TryParse(ref ICharIterator text, out LinkReferenceDefinitionBlock block) where T : ICharIterator { block = null; string label; diff --git a/src/Textamina.Markdig/Syntax/ListBlock.cs b/src/Textamina.Markdig/Syntax/ListBlock.cs index 6037e20b..c0ec0ab6 100644 --- a/src/Textamina.Markdig/Syntax/ListBlock.cs +++ b/src/Textamina.Markdig/Syntax/ListBlock.cs @@ -1,15 +1,9 @@ - - - -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { public class ListBlock : ContainerBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public ListBlock(BlockParser parser) : base(parser) { } @@ -24,359 +18,8 @@ namespace Textamina.Markdig.Syntax public bool IsLoose { get; set; } - private int CountAllBlankLines { get; set; } + internal int CountAllBlankLines { get; set; } - private int CountBlankLinesReset { get; set; } - - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - if (state.Pending is ListBlock && state.NextPending is ListItemBlock) - { - // We try to match only on item block if the ListBlock - return MatchLineResult.Skip; - } - - // When both a thematic break and a list item are possible - // interpretations of a line, the thematic break takes precedence - var save = state.Line; - if (ThematicBreakBlock.Parser.Match(state) == MatchLineResult.Last) - { - // Remove the ThematicBreakBlock as we will let the ThematicBreakBlock to catch it later - state.NewBlocks.Pop(); - return MatchLineResult.None; - } - state.SetCurrentLine(ref save); - - // 5.2 List items - // TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space - - int preIndent = 0; - for (int i = state.Line.Start - 1; i >= 0; i--) - { - if (state.Line[i].IsSpaceOrTab()) - { - preIndent++; - } - else - { - break; - } - } - - var saveLiner = state.Line; - - // If we have already a ListItemBlock, we are going to try to append to it - var listItem = state.Pending as ListItemBlock; - if (listItem != null) - { - var list = (ListBlock) listItem.Parent; - - // Allow all blanks lines if the last block is a fenced code block - // Allow 1 blank line inside a list - // If > 1 blank line, terminate this list - var isBlankLine = state.IsBlankLine; - //if (isBlankLine && !(state.LastBlock is FencedCodeBlock)) // TODO: Handle this case - var isInFencedBlock = state.LastBlock is FencedCodeBlock; - if (isBlankLine) - { - // TODO: Check with a generic way (allow a block to have multiple empty lines) - if (!isInFencedBlock) - { - if (!(state.NextPending is ListBlock)) - { - list.CountAllBlankLines++; - listItem.Children.Add(BlankLineBlock.Instance); - } - list.CountBlankLinesReset++; - } - - if (list.CountBlankLinesReset > 1) - { - // TODO: Close all lists and not only this one - return MatchLineResult.LastDiscard; - } - - if (list.CountBlankLinesReset == 1 && listItem.NumberOfSpaces < 0) - { - state.Close(listItem); - - // Leave the list open - list.IsOpen = true; - return MatchLineResult.Continue; - } - - return MatchLineResult.Continue; - } - - list.CountBlankLinesReset = 0; - - var c = state.Line.CurrentChar; - var startPosition = state.Line.Start; - - // List Item starting with a blank line (-1) - if (listItem.NumberOfSpaces < 0) - { - int expectedCount = -listItem.NumberOfSpaces; - int countSpaces = 0; - var saved = new StringSlice(); - while (c.IsSpaceOrTab()) - { - c = state.Line.NextChar(); - countSpaces = preIndent + state.Line.Column - startPosition; - if (countSpaces == expectedCount) - { - saved = state.Line; - } - else if (countSpaces >= 4) - { - state.SetCurrentLine(ref saved); - countSpaces = expectedCount; - break; - } - } - - if (countSpaces == expectedCount) - { - listItem.NumberOfSpaces = countSpaces; - return MatchLineResult.Continue; - } - } - else - { - while (c.IsSpaceOrTab()) - { - c = state.Line.NextChar(); - var countSpaces = preIndent + state.Line.Column - startPosition; - if (countSpaces >= listItem.NumberOfSpaces) - { - return MatchLineResult.Continue; - } - } - } - state.SetCurrentLine(ref saveLiner); - } - - return TryParseListItem(ref state, preIndent); - } - - private MatchLineResult TryParseListItem(ref BlockParserState state, int preIndent) - { - var isInList = state.Pending is ListItemBlock; - - var preStartPosition = state.Line.Start; - - var c = state.Line.CurrentChar; - if (isInList) - { - while (c.IsSpaceOrTab()) - { - c = state.Line.NextChar(); - } - } - else - { - // TODO - //state.Line.SkipLeadingSpaces3(); - c = state.Line.CurrentChar; - } - preIndent = preIndent + state.Line.Start - preStartPosition; - - var isOrdered = false; - var bulletChar = (char) 0; - int orderedStart = 0; - var orderedDelimiter = (char) 0; - - var column = state.Line.Start; - - if (c.IsBulletListMarker()) - { - bulletChar = c; - preIndent++; - } - else if (c.IsDigit()) - { - int countDigit = 0; - while (c.IsDigit()) - { - orderedStart = orderedStart*10 + c - '0'; - c = state.Line.NextChar(); - preIndent++; - countDigit++; - } - - // Note that ordered list start numbers must be nine digits or less: - if (countDigit > 9) - { - return MatchLineResult.None; - } - - // We don't have an ordered list - if (c != '.' && c != ')') - { - return MatchLineResult.None; - } - preIndent++; - isOrdered = true; - orderedDelimiter = c; - } - else - { - return MatchLineResult.None; - } - - // Skip Bullet or '.' - state.Line.NextChar(); - - // Item starting with a blank line - int numberOfSpaces; - if (state.IsBlankLine) - { - // Use a negative number to store the number of expected chars - numberOfSpaces = -(preIndent + 1); - } - else - { - var startPosition = -1; - int countSpaceAfterBullet = 0; - var saved = new StringSlice(); - for (int i = 0; i <= 4; i++) - { - c = state.Line.CurrentChar; - if (!c.IsSpaceOrTab()) - { - break; - } - if (i == 0) - { - startPosition = state.Line.Column; - } - - var endPosition = state.Line.Column; - countSpaceAfterBullet = endPosition - startPosition; - - if (countSpaceAfterBullet == 1) - { - saved = state.Line; - } - else if (countSpaceAfterBullet >= 4) - { - //state.Line.SpaceHeaderCount = countSpaceAfterBullet - 4; - countSpaceAfterBullet = 0; - state.SetCurrentLine(ref saved); - break; - } - state.Line.NextChar(); - } - - // If we haven't matched any spaces, early exit - if (startPosition < 0) - { - return MatchLineResult.None; - } - // Number of spaces required for the following content to be part of this list item - numberOfSpaces = preIndent + countSpaceAfterBullet + 1; - } - - var newListItem = new ListItemBlock(this) - { - Column = column, - NumberOfSpaces = numberOfSpaces - }; - state.NewBlocks.Push(newListItem); - - var currentListItem = state.Pending as ListItemBlock; - var currentParent = state.Pending as ListBlock ?? (ListBlock)currentListItem?.Parent; - - if (currentParent != null) - { - // If we have a new list item, close the previous one - if (currentListItem != null) - { - state.Close(currentListItem); - } - - // Reset the list if it is a new list or a new type of bullet - if (currentParent.IsOrdered != isOrdered || - (isOrdered && currentParent.OrderedDelimiter != orderedDelimiter) || - (!isOrdered && currentParent.BulletChar != bulletChar) - //(numberOfSpaces < ((ListItemBlock) currentParent.LastChild).NumberOfSpaces) - ) - { - state.Close(currentParent); - currentParent = null; - } - } - - if (currentParent == null) - { - var newList = new ListBlock(this) - { - Column = column, - IsOrdered = isOrdered, - BulletChar = bulletChar, - OrderedDelimiter = orderedDelimiter, - OrderedStart = orderedStart, - }; - state.NewBlocks.Push(newList); - } - - return MatchLineResult.Continue; - } - - public override void Close(BlockParserState state) - { - var listBlock = state.Pending as ListBlock; - - // Process only if we have blank lines - if (listBlock != null && listBlock.CountAllBlankLines > 0) - { - bool isLastListItem = true; - for (int listIndex = listBlock.Children.Count - 1; listIndex >= 0; listIndex--) - { - var block = listBlock.Children[listIndex]; - var listItem = (ListItemBlock) block; - var children = listItem.Children; - bool isLastElement = true; - for (int i = children.Count - 1; i >= 0; i--) - { - var item = children[i]; - if (item is BlankLineBlock) - { - if ((isLastElement && listIndex < listBlock.Children.Count - 1) || (children.Count > 2 && (i > 0 && i < (children.Count - 1)))) - { - listBlock.IsLoose = true; - } - - if (isLastElement && isLastListItem) - { - // Inform the outer list that we have a blank line - var parentListItemBlock = listBlock.Parent as ListItemBlock; - if (parentListItemBlock != null) - { - var parentList = (ListBlock) parentListItemBlock.Parent; - - parentList.CountAllBlankLines++; - parentListItemBlock.Children.Add(BlankLineBlock.Instance); - } - } - - children.RemoveAt(i); - - // If we have remove all blank lines, we can exit - listBlock.CountAllBlankLines--; - if (listBlock.CountAllBlankLines == 0) - { - return; - } - } - isLastElement = false; - } - isLastListItem = false; - } - } - } - } + internal int CountBlankLinesReset { get; set; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/ListItemBlock.cs b/src/Textamina.Markdig/Syntax/ListItemBlock.cs index eb0b21f9..98838853 100644 --- a/src/Textamina.Markdig/Syntax/ListItemBlock.cs +++ b/src/Textamina.Markdig/Syntax/ListItemBlock.cs @@ -1,7 +1,7 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { diff --git a/src/Textamina.Markdig/Syntax/ParagraphBlock.cs b/src/Textamina.Markdig/Syntax/ParagraphBlock.cs index fdc57cd9..8c0172e7 100644 --- a/src/Textamina.Markdig/Syntax/ParagraphBlock.cs +++ b/src/Textamina.Markdig/Syntax/ParagraphBlock.cs @@ -1,9 +1,4 @@ - - - -using System; -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { @@ -15,192 +10,8 @@ namespace Textamina.Markdig.Syntax /// public class ParagraphBlock : LeafBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public ParagraphBlock(BlockParser parser) : base(parser) { } - - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - if (state.IsCodeIndent) - { - return MatchLineResult.None; - } - - var column = state.Column; - - // Else it is a continue, we don't break on blank lines - var isBlankLine = state.IsBlankLine; - - var paragraph = state.Pending as ParagraphBlock; - - if (paragraph == null) - { - if (isBlankLine) - { - return MatchLineResult.None; - } - } - - // We continue trying to match by default - var result = MatchLineResult.Continue; - if (paragraph == null) - { - state.NewBlocks.Push(new ParagraphBlock(this) { Column = column }); - } - else - { - if (isBlankLine) - { - result = MatchLineResult.None; - } - else if (!(paragraph.Parent is QuoteBlock)) - { - var headingChar = (char) 0; - bool checkForSpaces = false; - for (int i = state.Start; i <= state.EndOffset; i++) - { - var c = state.Line[i]; - if (headingChar == 0) - { - if (c == '=' || c == '-') - { - headingChar = c; - continue; - } - break; - } - - if (checkForSpaces) - { - if (!c.IsSpaceOrTab()) - { - headingChar = (char) 0; - break; - } - } - else if (c != headingChar) - { - if (c.IsSpaceOrTab()) - { - checkForSpaces = true; - } - else - { - headingChar = (char)0; - break; - } - } - } - - if (headingChar != 0) - { - // If we matched a LinkReferenceDefinition before matching the heading, and the remaining - // lines are empty, we can early exit and remove the paragraph - if (TryMatchLinkReferenceDefinition(paragraph.Lines, state) && paragraph.Lines.Count == 0) - { - state.Discard(paragraph); - return MatchLineResult.LastDiscard; - } - - var level = headingChar == '=' ? 1 : 2; - - var heading = new HeadingBlock(this) - { - Column = paragraph.Column, - Level = level, - Lines = paragraph.Lines, - }; - heading.Lines.Trim(); - - // Remove the paragraph as a pending block - state.NewBlocks.Push(heading); - state.Discard(paragraph); - - return MatchLineResult.LastDiscard; - } - } - } - - return result; - } - - private bool TryMatchLinkReferenceDefinition(StringSliceList localLineGroup, BlockParserState state) - { - bool atLeastOneFound = false; - - //var saved = new StringSliceList.State(); - //while (true) - //{ - // // If we have found a LinkReferenceDefinition, we can discard the previous paragraph - // localLineGroup.Save(ref saved); - // LinkReferenceDefinitionBlock linkReferenceDefinition; - // if (LinkReferenceDefinitionBlock.TryParse(localLineGroup, out linkReferenceDefinition)) - // { - // if (!state.Root.LinkReferenceDefinitions.ContainsKey(linkReferenceDefinition.Label)) - // { - // state.Root.LinkReferenceDefinitions[linkReferenceDefinition.Label] = linkReferenceDefinition; - // } - // atLeastOneFound = true; - - // // Remove lines that have been matched - // if (localLineGroup.LinePosition == localLineGroup.Count) - // { - // localLineGroup.Clear(); - // } - // else - // { - // for (int i = localLineGroup.LinePosition - 1; i >= 0; i--) - // { - // localLineGroup.RemoveAt(i); - // } - // } - // } - // else - // { - // if (!atLeastOneFound) - // { - // localLineGroup.Restore(ref saved); - // } - // break; - // } - //} - - return atLeastOneFound; - } - - public override void Close(BlockParserState state) - { - var paragraph = state.Pending as ParagraphBlock; - var heading = state.Pending as HeadingBlock; - if (paragraph != null) - { - var lines = paragraph.Lines; - - TryMatchLinkReferenceDefinition(lines, state); - - // If Paragraph is empty, we can discard it - if (lines.Count == 0) - { - state.Pending = null; - return; - } - - var lineCount = lines.Count; - for (int i = 0; i < lineCount; i++) - { - var line = lines.Slices[i]; - line.Trim(); - } - } - else if (heading?.Lines.Count > 1) - { - //heading.Lines.RemoveAt(heading.Lines.Count - 1); - } - } - } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/QuoteBlock.cs b/src/Textamina.Markdig/Syntax/QuoteBlock.cs index 57ecd06f..7fa0baeb 100644 --- a/src/Textamina.Markdig/Syntax/QuoteBlock.cs +++ b/src/Textamina.Markdig/Syntax/QuoteBlock.cs @@ -1,52 +1,13 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { public class QuoteBlock : ContainerBlock { - public new static readonly BlockParser Parser = new ParserInternal(); - public QuoteBlock(BlockParser parser) : base(parser) { } - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - if (state.IsCodeIndent) - { - return MatchLineResult.None; - } - - // 5.1 Block quotes - // A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space. - var c = state.CurrentChar; - var column = state.Column; - if (c != '>') - { - if (state.Pending != null && state.IsBlankLine) - { - state.Close(state.Pending); - return MatchLineResult.None; - } - return MatchLineResult.None; - } - - c = state.NextChar(); - if (c.IsSpace()) - { - state.NextChar(); - } - - if (state.Pending == null) - { - state.NewBlocks.Push(new QuoteBlock(this) { Column = column }); - } - - return MatchLineResult.Continue; - } - } + public char QuoteChar { get; set; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/StringSlice.cs b/src/Textamina.Markdig/Syntax/StringSlice.cs index 5658d86a..7894031e 100644 --- a/src/Textamina.Markdig/Syntax/StringSlice.cs +++ b/src/Textamina.Markdig/Syntax/StringSlice.cs @@ -4,7 +4,22 @@ using Textamina.Markdig.Helpers; namespace Textamina.Markdig.Syntax { - public struct StringSlice + public interface ICharIterator + { + int Start { get; set; } + + char CurrentChar { get; } + + int End { get; set; } + + char NextChar(); + + bool TrimStart(); + + bool TrimStart(out int spaceCount); + } + + public struct StringSlice : ICharIterator { public StringSlice(string text) { @@ -16,9 +31,11 @@ namespace Textamina.Markdig.Syntax public readonly string Text; - public int Start; + public int Start { get; set; } - public int End; + public int End { get; set; } + + public int Length => End - Start + 1; public int Column => Start; @@ -35,8 +52,9 @@ namespace Textamina.Markdig.Syntax if (Start > End) { Start = End + 1; + return '\0'; } - return CurrentChar; + return Text[Start]; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] @@ -117,44 +135,41 @@ namespace Textamina.Markdig.Syntax public bool TrimStart() { // Strip leading spaces - var c = CurrentChar; - var hasWhitespaces = false; - while (c.IsWhitespace()) + for (; Start <= End; Start++) { - c = NextChar(); - hasWhitespaces = true; - } - return hasWhitespaces; - } - - public bool TrimStart(out int newLineCount) - { - bool hasWhitespaces = false; - newLineCount = 0; - var c = CurrentChar; - while (c.IsWhitespace()) - { - if (c == '\n') - { - newLineCount++; - } - c = NextChar(); - hasWhitespaces = true; - } - return hasWhitespaces; - } - - public void TrimEnd(bool includeTabs = false) - { - for (int i = End; i >= Start; i--) - { - End = i; - var c = this[i]; - if (!(includeTabs ? c.IsSpaceOrTab() : c.IsSpace())) + if (!Text[Start].IsWhitespace()) { break; } } + return Start > End; + } + + public bool TrimStart(out int spaceCount) + { + spaceCount = 0; + // Strip leading spaces + for (; Start <= End; Start++) + { + if (!Text[Start].IsWhitespace()) + { + break; + } + spaceCount++; + } + return IsEndOfSlice; + } + + public bool TrimEnd() + { + for (; Start <= End; End--) + { + if (!Text[End].IsWhitespace()) + { + break; + } + } + return IsEndOfSlice; } public void Trim() @@ -165,7 +180,7 @@ namespace Textamina.Markdig.Syntax public override string ToString() { - return Start <= End ? Text.Substring(Start, End - Start + 1) : string.Empty; + return Text != null && Start <= End ? Text.Substring(Start, End - Start + 1) : string.Empty; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Syntax/StringSliceList.cs b/src/Textamina.Markdig/Syntax/StringSliceList.cs index 3a1b54b3..6bcf5aab 100644 --- a/src/Textamina.Markdig/Syntax/StringSliceList.cs +++ b/src/Textamina.Markdig/Syntax/StringSliceList.cs @@ -8,8 +8,6 @@ namespace Textamina.Markdig.Syntax { private static readonly StringSlice[] Empty = new StringSlice[0]; - private StringSlice currentLine; - public StringSliceList() { Slices = Empty; @@ -50,17 +48,17 @@ namespace Textamina.Markdig.Syntax public override string ToString() { - var stringBuilder = StringBuilderCache.Local(); + var builder = StringBuilderCache.Local(); for(int i = 0; i < Count; i++) { if (i > 0) { - stringBuilder.Append('\n'); + builder.Append('\n'); } - stringBuilder.Append(Slices[i]); + builder.Append(Slices[i].Text, Slices[i].Start, Slices[i].Length); } - var str = stringBuilder.ToString(); - stringBuilder.Clear(); + var str = builder.ToString(); + builder.Clear(); return str; } } diff --git a/src/Textamina.Markdig/Syntax/ThematicBreakBlock.cs b/src/Textamina.Markdig/Syntax/ThematicBreakBlock.cs index 4ca96645..7cfb775a 100644 --- a/src/Textamina.Markdig/Syntax/ThematicBreakBlock.cs +++ b/src/Textamina.Markdig/Syntax/ThematicBreakBlock.cs @@ -1,5 +1,4 @@ -using Textamina.Markdig.Helpers; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Parsers; namespace Textamina.Markdig.Syntax { @@ -8,83 +7,11 @@ namespace Textamina.Markdig.Syntax /// public class ThematicBreakBlock : LeafBlock { - public new static readonly BlockParser Parser = new ParserInternal(); + public new static readonly BlockParser Parser = new ThematicBreakParser(); public ThematicBreakBlock(BlockParser parser) : base(parser) { - NoInline = true; - } - - private class ParserInternal : BlockParser - { - public override MatchLineResult Match(BlockParserState state) - { - if (state.IsCodeIndent) - { - return MatchLineResult.None; - } - - var column = state.Column; - - // 4.1 Thematic breaks - // A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces - var c = state.CurrentChar; - - int count = 0; - var matchChar = (char)0; - bool hasSpacesSinceLastMatch = false; - bool hasInnerSpaces = false; - int offset = 0; - while (c != '\0') - { - if (count == 0 && (c == '-' || c == '_' || c == '*')) - { - matchChar = c; - count++; - } - else if (c == matchChar) - { - if (hasSpacesSinceLastMatch) - { - hasInnerSpaces = true; - } - - count++; - } - else if (!c.IsSpace() || count == 0) - { - return MatchLineResult.None; - } - else if (c.IsSpace()) - { - hasSpacesSinceLastMatch = true; - } - - offset++; - c = state.PeekChar(offset); - } - - // If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it - var previousParagraph = state.LastBlock as ParagraphBlock; - - var isSetexHeading = previousParagraph != null && matchChar == '-' && !hasInnerSpaces; - if (isSetexHeading) - { - var parent = previousParagraph.Parent; - if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != column)) - { - isSetexHeading = false; - } - } - - if (count < 3 || isSetexHeading) - { - return MatchLineResult.None; - } - - state.NewBlocks.Push(new ThematicBreakBlock(this) { Column = column }); - return MatchLineResult.LastDiscard; - } + ProcessInlines = false; } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Textamina.Markdig.csproj b/src/Textamina.Markdig/Textamina.Markdig.csproj index 42455093..2f274202 100644 --- a/src/Textamina.Markdig/Textamina.Markdig.csproj +++ b/src/Textamina.Markdig/Textamina.Markdig.csproj @@ -41,39 +41,52 @@ - - - - + + + + + + + + + + + - + + + - + + - - + + + + + @@ -81,17 +94,19 @@ - + + +