Rewrite BlockParser, simplfy code.

This commit is contained in:
Alexandre Mutel
2016-02-25 22:10:52 +09:00
parent 0976c03dcf
commit daa09f6cd2
70 changed files with 2955 additions and 3197 deletions

View File

@@ -8,7 +8,7 @@ using System.Threading.Tasks;
using BenchmarkDotNet.Attributes;
using BenchmarkDotNet.Running;
using Textamina.Markdig.Formatters;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Testamina.Markdig.Benchmarks
{
@@ -46,15 +46,15 @@ namespace Testamina.Markdig.Benchmarks
static void Main(string[] args)
{
var clock = Stopwatch.StartNew();
//var clock = Stopwatch.StartNew();
var program = new Program();
for (int i = 0; i < 200; i++)
{
//program.TestMarkdig();
program.TestCommonMark();
}
Console.WriteLine($"time: {clock.ElapsedMilliseconds}ms");
DumpGC();
//for (int i = 0; i < 200; i++)
//{
program.TestMarkdig();
// //program.TestCommonMark();
//}
//Console.WriteLine($"time: {clock.ElapsedMilliseconds}ms");
//DumpGC();
//BenchmarkRunner.Run<Program>();
}

View File

@@ -9,7 +9,7 @@ using System.Text;
using System.Text.RegularExpressions;
using NUnit.Framework;
using Textamina.Markdig.Formatters;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Tests
{
@@ -24,7 +24,9 @@ namespace Textamina.Markdig.Tests
[Test]
public void TestSimple()
{
var reader = new StringReader(@"foo");
var reader = new StringReader(@">> foo
test
");
// var reader = new StringReader(@"> > toto tata
//> titi toto
//");

View File

@@ -4,6 +4,7 @@ using System.Globalization;
using System.IO;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Formatters
{

View File

@@ -1,5 +1,4 @@
using System;
using System.Collections.Generic;
using System.Text;
using Textamina.Markdig.Formatters;
using Textamina.Markdig.Syntax;

View File

@@ -274,12 +274,12 @@ namespace Textamina.Markdig.Helpers
return isValid;
}
public static bool TryParseTitle(StringSlice text, out string title)
public static bool TryParseTitle<T>(T text, out string title) where T : ICharIterator
{
return TryParseTitle(ref text, out title);
}
public static bool TryParseTitle(ref StringSlice text, out string title)
public static bool TryParseTitle<T>(ref T text, out string title) where T : ICharIterator
{
bool isValid = false;
var buffer = StringBuilderCache.Local();
@@ -344,12 +344,12 @@ namespace Textamina.Markdig.Helpers
return isValid;
}
public static bool TryParseUrl(StringSlice text, out string link)
public static bool TryParseUrl<T>(T text, out string link) where T : ICharIterator
{
return TryParseUrl(ref text, out link);
}
public static bool TryParseUrl(ref StringSlice text, out string link)
public static bool TryParseUrl<T>(ref T text, out string link) where T : ICharIterator
{
bool isValid = false;
var buffer = StringBuilderCache.Local();
@@ -467,14 +467,14 @@ namespace Textamina.Markdig.Helpers
return isValid;
}
public static bool TryParseLinkReferenceDefinition(StringSlice text, out string label, out string url,
out string title)
public static bool TryParseLinkReferenceDefinition<T>(T text, out string label, out string url,
out string title) where T : ICharIterator
{
return TryParseLinkReferenceDefinition(ref text, out label, out url, out title);
}
public static bool TryParseLinkReferenceDefinition(ref StringSlice text, out string label, out string url,
out string title)
public static bool TryParseLinkReferenceDefinition<T>(ref T text, out string label, out string url,
out string title) where T : ICharIterator
{
url = null;
title = null;
@@ -551,17 +551,17 @@ namespace Textamina.Markdig.Helpers
return true;
}
public static bool TryParseLabel(StringSlice lines, out string label)
public static bool TryParseLabel<T>(T lines, out string label) where T : ICharIterator
{
return TryParseLabel(ref lines, false, out label);
}
public static bool TryParseLabel(ref StringSlice lines, out string label)
public static bool TryParseLabel<T>(ref T lines, out string label) where T : ICharIterator
{
return TryParseLabel(ref lines, false, out label);
}
public static bool TryParseLabel(ref StringSlice lines, bool allowEmpty, out string label)
public static bool TryParseLabel<T>(ref T lines, bool allowEmpty, out string label) where T : ICharIterator
{
label = null;
char c = lines.CurrentChar;

View File

@@ -0,0 +1,36 @@
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public abstract class BlockParser : IBlockParser
{
protected BlockParser()
{
}
public char[] OpeningCharacters { get; protected set; }
public virtual bool CanInterrupt(BlockParserState state, Block block)
{
// By default, all blocks can interrupt a paragraph except:
// - setext heading
// - indented code block
// - a special HTML blocks
return true;
}
public abstract BlockState TryOpen(BlockParserState state);
public virtual BlockState TryContinue(BlockParserState state, Block block)
{
// By default we don't expect any newline
return BlockState.None;
}
public virtual bool Close(BlockParserState state, Block block)
{
// By default keep the block
return true;
}
}
}

View File

@@ -0,0 +1,483 @@
using System;
using System.Collections.Generic;
using System.Diagnostics;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class BlockParserState
{
private readonly ParserList<BlockParser> blockParsers;
private int currentStackIndex;
public BlockParserState(StringBuilderCache stringBuilders, Document root)
{
if (stringBuilders == null) throw new ArgumentNullException(nameof(stringBuilders));
if (root == null) throw new ArgumentNullException(nameof(root));
StringBuilders = stringBuilders;
Root = root;
NewBlocks = new Stack<Block>();
root.IsOpen = true;
Stack = new List<Block> {root};
blockParsers = new ParserList<BlockParser>()
{
new ThematicBreakParser(),
new HeadingBlockParser(),
new QuoteBlockParser(),
//ListBlock.Parser,
new HtmlBlockParser(),
new CodeBlockParser(),
new FencedCodeBlockParser(),
new ParagraphBlockParser(),
};
blockParsers.Initialize();
}
public List<Block> Stack { get; }
public Stack<Block> NewBlocks { get; }
public ContainerBlock CurrentContainer { get; private set; }
public Block LastBlock { get; private set; }
public Block NextContinue => currentStackIndex + 1 < Stack.Count ? Stack[currentStackIndex + 1] : null;
public Document Root { get; }
public bool ContinueProcessingLine { get; set; }
public StringSlice Line;
public int LineIndex { get; private set; }
public bool IsBlankLine => CurrentChar == '\0';
public bool IsEndOfLine => Line.IsEndOfSlice;
public char CurrentChar => Line.CurrentChar;
public char NextChar()
{
var c = Line.CurrentChar;
if (c == '\t')
{
Column = ((Column + 3) >> 2) << 2;
}
else
{
Column++;
}
return Line.NextChar();
}
public char CharAt(int index) => Line[index];
public int Start => Line.Start;
public int EndOffset => Line.End;
public int Indent => Column - ColumnBegin;
public bool IsCodeIndent => Indent >= 4;
public int ColumnBegin { get; private set; }
public int Column { get; set; }
public StringBuilderCache StringBuilders { get; }
public char PeekChar(int offset)
{
return Line.PeekChar(offset);
}
public void ParseIndent()
{
var c = CurrentChar;
ColumnBegin = Column;
while (c !='\0')
{
if (c == ' ')
{
Column++;
}
else if (c == '\t')
{
Column = ((Column + 3) >> 2) << 2;
}
else
{
break;
}
c = NextChar();
}
}
public void Close(Block block)
{
// If we close a block, we close all blocks above
for (int i = Stack.Count - 1; i >= 1; i--)
{
if (Stack[i] == block)
{
for (int j = Stack.Count - 1; j >= i; j--)
{
Close(j);
}
break;
}
}
}
public void Discard(Block block)
{
for (int i = Stack.Count - 1; i >= 1; i--)
{
if (Stack[i] == block)
{
block.Parent.Children.Remove(block);
Stack.RemoveAt(i);
break;
}
}
}
public void Close(int index)
{
var block = Stack[index];
// If the pending object is removed, we need to remove it from the parent container
if (!block.Parser.Close(this, block))
{
block.Parent?.Children.Remove(block);
}
Stack.RemoveAt(index);
}
public void CloseAll(bool force)
{
// Close any previous blocks not opened
for (int i = Stack.Count - 1; i >= 1; i--)
{
var block = Stack[i];
// Stop on the first open block
if (!force && block.IsOpen)
{
break;
}
Close(i);
}
}
public void ProcessLine(string newLine)
{
ContinueProcessingLine = true;
Line = new StringSlice(newLine);
ParseIndent();
LineIndex++;
TryContinueBlocks();
// If we have already reached eol and the last block was a paragraph
// we close it
if (Line.IsEndOfSlice)
{
int index = Stack.Count - 1;
if (Stack[index] is ParagraphBlock)
{
Close(index);
return;
}
}
// If the line was not entirely processed by pending blocks, try to process it with any new block
TryOpenBlocks();
// Close blocks that are no longer opened
CloseAll(false);
}
private void OpenAll()
{
for (int i = 1; i < Stack.Count; i++)
{
Stack[i].IsOpen = true;
}
}
internal void UpdateLast(int stackIndex)
{
currentStackIndex = stackIndex < 0 ? Stack.Count - 1 : stackIndex;
LastBlock = null;
for (int i = Stack.Count - 1; i >= 0; i--)
{
var block = Stack[i];
if (LastBlock == null)
{
LastBlock = block;
}
var container = block as ContainerBlock;
if (container != null)
{
CurrentContainer = container;
break;
}
}
}
private void TryContinueBlocks()
{
// Set all blocks non opened.
// They will be marked as open in the following loop
for (int i = 1; i < Stack.Count; i++)
{
Stack[i].IsOpen = false;
}
// Process any current block potentially opened
for (int i = 1; i < Stack.Count; i++)
{
var block = Stack[i];
// If we have a paragraph block, we want to try to match other blocks before trying the Paragraph
if (block is ParagraphBlock)
{
break;
}
// Else tries to match the Parser with the current line
var parser = block.Parser;
// If we have a discard, we can remove it from the current state
UpdateLast(i);
var result = parser.TryContinue(this, block);
if (result == BlockState.Skip)
{
continue;
}
if (result == BlockState.None)
{
break;
}
// In case the BlockParser has modified the blockParserState we are iterating on
if (i >= Stack.Count)
{
i = Stack.Count - 1;
}
// If a parser is adding a block, it must be the last of the list
if ((i + 1) < Stack.Count && NewBlocks.Count > 0)
{
throw new InvalidOperationException("A pending parser cannot add a new block when it is not the last pending block");
}
// If we have a leaf block
var leaf = block as LeafBlock;
if (leaf != null && NewBlocks.Count == 0)
{
ContinueProcessingLine = false;
if (!result.IsDiscard())
{
leaf.AppendLine(ref Line);
}
if (NewBlocks.Count > 0)
{
throw new InvalidOperationException(
"The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed");
}
}
// A block is open only if it has a Continue state.
// otherwise it is a Break state, and we don't keep it opened
block.IsOpen = result == BlockState.Continue || result == BlockState.ContinueDiscard;
if (result == BlockState.BreakDiscard)
{
ContinueProcessingLine = false;
break;
}
bool isLast = i == Stack.Count - 1;
if (ContinueProcessingLine)
{
ProcessNewBlocks(result, false);
}
if (isLast || !ContinueProcessingLine)
{
break;
}
}
}
private void TryOpenBlocks()
{
while (ContinueProcessingLine)
{
// Eat indent spaces before checking the character
ParseIndent();
var parsers = blockParsers.GetParsersForOpeningCharacter(CurrentChar);
var globalParsers = blockParsers.GlobalParsers;
if (parsers != null)
{
if (TryOpenBlocks(parsers))
{
continue;
}
}
if (globalParsers != null && ContinueProcessingLine)
{
if (TryOpenBlocks(globalParsers))
{
continue;
}
}
break;
}
}
private bool TryOpenBlocks(BlockParser[] parsers)
{
for (int j = 0; j < parsers.Length; j++)
{
var blockParser = parsers[j];
if (Line.IsEndOfSlice)
{
ContinueProcessingLine = false;
break;
}
// UpdateLast the state of LastBlock and LastContainer
UpdateLast(-1);
// If a block parser cannot interrupt a paragraph, and the last block is a paragraph
// we can skip this parser
var lastBlock = LastBlock;
if (!blockParser.CanInterrupt(this, lastBlock))
{
continue;
}
bool isLazyParagraph = blockParser is ParagraphBlockParser && lastBlock is ParagraphBlock;
var result = isLazyParagraph
? blockParser.TryContinue(this, lastBlock)
: blockParser.TryOpen(this);
if (result == BlockState.None)
{
// If we have reached a blank line after trying to parse a paragraph
// we can ignore it
if (isLazyParagraph && IsBlankLine)
{
ContinueProcessingLine = false;
break;
}
continue;
}
// Special case for paragraph
UpdateLast(-1);
var paragraph = LastBlock as ParagraphBlock;
if (isLazyParagraph && paragraph != null)
{
Debug.Assert(NewBlocks.Count == 0);
if (!result.IsDiscard())
{
paragraph.AppendLine(ref Line);
}
// We have just found a lazy continuation for a paragraph, early exit
// Mark all block opened after a lazy continuation
OpenAll();
ContinueProcessingLine = false;
break;
}
// Nothing found but the BlockParser may instruct to break, so early exit
if (NewBlocks.Count == 0 && result == BlockState.BreakDiscard)
{
ContinueProcessingLine = false;
break;
}
// If we have a container, we can retry to match against all types of block.
ProcessNewBlocks(result, true);
return ContinueProcessingLine;
// We have a leaf node, we can stop
}
return false;
}
private void ProcessNewBlocks(BlockState result, bool allowClosing)
{
var newBlocks = NewBlocks;
while (newBlocks.Count > 0)
{
var block = newBlocks.Pop();
block.Line = LineIndex;
// If we have a leaf block
var leaf = block as LeafBlock;
if (leaf != null)
{
if (!result.IsDiscard())
{
leaf.AppendLine(ref Line);
}
if (newBlocks.Count > 0)
{
throw new InvalidOperationException(
"The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed");
}
}
if (allowClosing)
{
// Close any previous blocks not opened
CloseAll(false);
}
// If previous block is a container, add the new block as a children of the previous block
if (block.Parent == null)
{
CurrentContainer.Children.Add(block);
block.Parent = CurrentContainer;
}
block.IsOpen = result.IsContinue();
// Add a block blockParserState to the stack (and leave it opened)
Stack.Add(block);
if (leaf != null)
{
ContinueProcessingLine = false;
return;
}
}
ContinueProcessingLine = true;
}
}
}

View File

@@ -0,0 +1,40 @@
using System.Runtime.CompilerServices;
namespace Textamina.Markdig.Parsers
{
public enum BlockState
{
None,
Skip,
Continue,
ContinueDiscard,
Break,
BreakDiscard
}
public static class BlockStateExtensions
{
[MethodImpl(MethodImplOptionPortable.AggressiveInlining)]
public static bool IsDiscard(this BlockState blockState)
{
return blockState == BlockState.ContinueDiscard || blockState == BlockState.BreakDiscard;
}
[MethodImpl(MethodImplOptionPortable.AggressiveInlining)]
public static bool IsContinue(this BlockState blockState)
{
return blockState == BlockState.Continue || blockState == BlockState.ContinueDiscard;
}
[MethodImpl(MethodImplOptionPortable.AggressiveInlining)]
public static bool IsBreak(this BlockState blockState)
{
return blockState == BlockState.Break || blockState == BlockState.BreakDiscard;
}
}
}

View File

@@ -0,0 +1,35 @@
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class CodeBlockParser : BlockParser
{
public CodeBlockParser()
{
}
public override bool CanInterrupt(BlockParserState state, Block block)
{
return !(block is ParagraphBlock);
}
public override BlockState TryOpen(BlockParserState state)
{
var result = TryContinue(state, null);
if (result == BlockState.Continue)
{
state.NewBlocks.Push(new CodeBlock(this) { Column = state.Column });
}
return result;
}
public override BlockState TryContinue(BlockParserState state, Block block)
{
if (!state.IsCodeIndent || state.IsBlankLine)
{
return state.IsBlankLine && block != null ? BlockState.BreakDiscard : BlockState.None;
}
return BlockState.Continue;
}
}
}

View File

@@ -0,0 +1,143 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class FencedCodeBlockParser : BlockParser
{
public FencedCodeBlockParser()
{
OpeningCharacters = new[] {'`', '~'};
}
public override BlockState TryOpen(BlockParserState state)
{
// Else if the we have an indent, it is not valid
if (state.IsCodeIndent)
{
return BlockState.None;
}
int count = 0;
var line = state.Line;
char c = line.CurrentChar;
var matchChar = c;
while (c != '\0')
{
if (c != matchChar)
{
break;
}
count++;
c = line.NextChar();
}
if (count < 3)
{
return BlockState.None;
}
// TODO: We need to count the number of leading space to remove them on each line
var column = state.Column;
// specs spaces: Is space and tabs? or only spaces? Use space and tab for this case
while (c.IsSpaceOrTab())
{
c = line.NextChar();
}
string infoString;
string argString = null;
// An info string cannot contain any backsticks
int firstSpace = -1;
for (int i = line.Start; i <= line.End; i++)
{
c = line.Text[i];
if (c == '`')
{
return BlockState.None;
}
if (firstSpace < 0 && c.IsSpaceOrTab())
{
firstSpace = i;
}
}
if (firstSpace > 0)
{
infoString = line.Text.Substring(line.Start, firstSpace - line.Start);
// Skip any spaces after info string
firstSpace++;
while (true)
{
c = line[firstSpace];
if (c.IsSpaceOrTab())
{
firstSpace++;
}
else
{
break;
}
}
argString = line.Text.Substring(firstSpace, line.End - firstSpace + 1);
}
else
{
infoString = line.ToString();
}
// Store the number of matched string into the context
state.NewBlocks.Push(new FencedCodeBlock(this)
{
Column = column,
FencedChar = matchChar,
FencedCharCount = count,
IndentCount = state.Indent,
Language = HtmlHelper.Unescape(infoString),
Arguments = HtmlHelper.Unescape(argString),
});
// Discard the current line as it is already parsed
return BlockState.ContinueDiscard;
}
public override BlockState TryContinue(BlockParserState state, Block block)
{
var fence = (FencedCodeBlock)block;
var count = fence.FencedCharCount;
var matchChar = fence.FencedChar;
var c = state.CurrentChar;
// Work on a copy of StringSlice
var line = state.Line;
while (c == matchChar)
{
c = line.NextChar();
count--;
}
if (count <=0 && line.TrimEnd())
{
// Don't keep the last line
return BlockState.BreakDiscard;
}
// Remove any indent spaces
c = state.CurrentChar;
var indentCount = fence.IndentCount;
while (indentCount > 0 && c.IsSpace())
{
indentCount--;
c = state.NextChar();
}
// TODO: It is unclear how to handle this correctly
// Break only if Eof
return BlockState.Continue;
}
}
}

View File

@@ -0,0 +1,108 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class HeadingBlockParser : BlockParser
{
public HeadingBlockParser()
{
OpeningCharacters = new[] {'#'};
}
public override BlockState TryOpen(BlockParserState state)
{
// If we are in a CodeIndent, early exit
if (state.IsCodeIndent)
{
return BlockState.None;
}
// 4.2 ATX headings
// An ATX heading consists of a string of characters, parsed as inline content,
// between an opening sequence of 1–6 unescaped # characters and an optional
// closing sequence of any number of unescaped # characters. The opening sequence
// of # characters must be followed by a space or by the end of line. The optional
// closing sequence of #s must be preceded by a space and may be followed by spaces
// only. The opening # character may be indented 0-3 spaces. The raw contents of
// the heading are stripped of leading and trailing spaces before being parsed as
// inline content. The heading level is equal to the number of # characters in the
// opening sequence.
var column = state.Column;
var line = state.Line;
var c = line.CurrentChar;
var matchingChar = c;
int leadingCount = 0;
while (c != '\0' && leadingCount <= 6)
{
if (c != matchingChar)
{
break;
}
c = line.NextChar();
leadingCount++;
}
// closing # will be handled later, because anyway we have matched
// A space is required after leading #
if (leadingCount > 0 && leadingCount <= 6 && (c.IsSpace() || c == '\0'))
{
// Move to the content
state.Line.Start = leadingCount + 1;
state.NewBlocks.Push(new HeadingBlock(this) {Level = leadingCount, Column = column });
// The optional closing sequence of #s must be preceded by a space and may be followed by spaces only.
int endState = 0;
int countClosingTags = 0;
for (int i = state.Line.End; i >= state.Line.Start - 1; i--) // Go up to Start - 1 in order to match the space after the first ###
{
c = state.Line.Text[i];
if (endState == 0)
{
if (c.IsSpace()) // TODO: Not clear if it is a space or space+tab in the specs
{
continue;
}
endState = 1;
}
if (endState == 1)
{
if (c == '#')
{
countClosingTags++;
continue;
}
if (countClosingTags > 0)
{
if (c.IsSpace())
{
state.Line.End = i - 1;
}
break;
}
else
{
break;
}
}
}
// We expect a single line, so don't continue
return BlockState.Break;
}
// Else we don't have an header
return BlockState.None;
}
//public override bool Close(BlockParserState state, Block block)
//{
// var heading = (HeadingBlock) block;
// heading.Lines.Trim();
// return true;
//}
}
}

View File

@@ -0,0 +1,286 @@
using System;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class HtmlBlockParser : BlockParser
{
public HtmlBlockParser()
{
OpeningCharacters = new[] {'<'};
}
private static readonly string[] HtmlTags =
{
"address", // 0
"article", // 1
"aside", // 2
"base", // 3
"basefont", // 4
"blockquote", // 5
"body", // 6
"caption", // 7
"center", // 8
"col", // 9
"colgroup", // 10
"dd", // 11
"details", // 12
"dialog", // 13
"dir", // 14
"div", // 15
"dl", // 16
"dt", // 17
"fieldset", // 18
"figcaption", // 19
"figure", // 20
"footer", // 21
"form", // 22
"frame", // 23
"frameset", // 24
"h1", // 25
"head", // 26
"header", // 27
"hr", // 28
"html", // 29
"iframe", // 30
"legend", // 31
"li", // 32
"link", // 33
"main", // 34
"menu", // 35
"menuitem", // 36
"meta", // 37
"nav", // 38
"noframes", // 39
"ol", // 40
"optgroup", // 41
"option", // 42
"p", // 43
"param", // 44
"pre", // 45 <- special group 1
"script", // 46 <- special group 1
"section", // 47
"source", // 48
"style", // 49 <- special group 1
"summary", // 50
"table", // 51
"tbody", // 52
"td", // 53
"tfoot", // 54
"th", // 55
"thead", // 56
"title", // 57
"tr", // 58
"track", // 59
"ul", // 60
};
public override BlockState TryOpen(BlockParserState state)
{
var result = MatchStart(state);
// An end-tag can occur on the same line
if (result == BlockState.Continue)
{
result = MatchEnd(state, (HtmlBlock) state.NewBlocks.Peek());
}
return result;
}
public virtual BlockState TryContinue(BlockParserState state, Block block)
{
var htmlBlock = (HtmlBlock) block;
return MatchEnd(state, htmlBlock);
}
private BlockState MatchStart(BlockParserState state)
{
if (state.IsCodeIndent)
{
return BlockState.None;
}
var result = TryParseTagType16(state, state.Line, state.ColumnBegin);
// HTML blocks of type 7 cannot interrupt a paragraph:
if (result == BlockState.None && !(state.LastBlock is ParagraphBlock))
{
result = TryParseTagType7(state, state.Line, state.ColumnBegin);
}
return result;
}
private BlockState TryParseTagType7(BlockParserState state, StringSlice line, int startColumn)
{
var builder = StringBuilderCache.Local();
var c = line.CurrentChar;
var result = BlockState.None;
if ((c == '/' && HtmlHelper.TryParseHtmlCloseTag(line, builder)) || HtmlHelper.TryParseHtmlTagOpenTag(line, builder))
{
// Must be followed by whitespace only
bool hasOnlySpaces = true;
c = line.CurrentChar;
while (true)
{
if (c == '\0')
{
break;
}
if (!c.IsWhitespace())
{
hasOnlySpaces = false;
break;
}
c = line.NextChar();
}
if (hasOnlySpaces)
{
result = CreateHtmlBlock(state, HtmlBlockType.NonInterruptingBlock, startColumn);
}
}
builder.Clear();
return result;
}
private BlockState TryParseTagType16(BlockParserState state, StringSlice line, int startColumn)
{
char c;
c = line.CurrentChar;
if (c == '!')
{
c = line.PeekChar(1);
if (c == '-' && line.PeekChar(2) == '-')
{
return CreateHtmlBlock(state, HtmlBlockType.Comment, startColumn); // group 2
}
if (c.IsAlphaUpper())
{
return CreateHtmlBlock(state, HtmlBlockType.DocumentType, startColumn); // group 4
}
if (c == '[' && line.Match("CDATA[", 3))
{
return CreateHtmlBlock(state, HtmlBlockType.CData, startColumn); // group 5
}
return BlockState.None;
}
if (c == '?')
{
return CreateHtmlBlock(state, HtmlBlockType.ProcessingInstruction, startColumn); // group 3
}
var hasLeadingClose = c == '/';
if (hasLeadingClose)
{
line.NextChar();
}
var tag = new char[10];
var count = 0;
for (; count < tag.Length; count++)
{
c = line.NextChar();
if (!c.IsAlphaNumeric())
{
break;
}
tag[count] = Char.ToLowerInvariant(c);
}
if (
!(c == '>' || (!hasLeadingClose && c == '/' && line.PeekChar(1) == '>') || c.IsWhitespace() ||
c == '\0'))
{
return BlockState.None;
}
if (count == 0)
{
return BlockState.None;
}
var tagName = new string(tag, 0, count);
var tagIndex = Array.BinarySearch(HtmlTags, tagName, StringComparer.Ordinal);
if (tagIndex < 0)
{
return BlockState.None;
}
// Cannot start with </script </pre or </style
if ((tagIndex == 45 || tagIndex == 46 || tagIndex == 49))
{
if (c == '/' || hasLeadingClose)
{
return BlockState.None;
}
return CreateHtmlBlock(state, HtmlBlockType.ScriptPreOrStyle, startColumn);
}
return CreateHtmlBlock(state, HtmlBlockType.InterruptingBlock, startColumn);
}
private BlockState MatchEnd(BlockParserState state, HtmlBlock htmlBlock)
{
// Early exit if it is not starting by an HTML tag
var line = state.Line;
var c = line.CurrentChar;
switch (htmlBlock.Type)
{
case HtmlBlockType.Comment:
if (line.Search("-->"))
{
return BlockState.Break;
}
break;
case HtmlBlockType.CData:
if (line.Search("]]>"))
{
return BlockState.Break;
}
break;
case HtmlBlockType.ProcessingInstruction:
if (line.Search("?>"))
{
return BlockState.Break;
}
break;
case HtmlBlockType.DocumentType:
if (line.Search(">"))
{
return BlockState.Break;
}
break;
case HtmlBlockType.ScriptPreOrStyle:
// TODO: could be optimized with a dedicated parser
if (line.SearchLowercase("</script>") || line.SearchLowercase("</pre>") || line.SearchLowercase("</style>"))
{
return BlockState.Break;
}
break;
case HtmlBlockType.InterruptingBlock:
if (state.IsBlankLine)
{
return BlockState.BreakDiscard;
}
break;
case HtmlBlockType.NonInterruptingBlock:
if (state.IsBlankLine)
{
return BlockState.BreakDiscard;
}
break;
}
return BlockState.Continue;
}
private BlockState CreateHtmlBlock(BlockParserState state, HtmlBlockType type, int startColumn)
{
state.NewBlocks.Push(new HtmlBlock(this) {Column = startColumn, Type = type});
return BlockState.Continue;
}
}
}

View File

@@ -0,0 +1,20 @@
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public interface ICharacterParser
{
char[] OpeningCharacters { get; }
}
public interface IBlockParser : ICharacterParser
{
bool CanInterrupt(BlockParserState state, Block block);
BlockState TryOpen(BlockParserState state);
BlockState TryContinue(BlockParserState state, Block block);
bool Close(BlockParserState state, Block block);
}
}

View File

@@ -0,0 +1,9 @@
namespace Textamina.Markdig.Parsers
{
public abstract class InlineParser : ICharacterParser
{
public char[] OpeningCharacters { get; protected set; }
public abstract bool Match(InlineParserState state);
}
}

View File

@@ -1,9 +1,9 @@
using System.Collections.Generic;
using System.Text;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsing
namespace Textamina.Markdig.Parsers
{
public class InlineParserState
{

View File

@@ -0,0 +1,37 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class AutolineInlineParser : InlineParser
{
public AutolineInlineParser()
{
OpeningCharacters = new[] {'<'};
}
public override bool Match(InlineParserState state)
{
string link;
bool isEmail;
var saved = state.Text;
if (LinkHelper.TryParseAutolink(ref state.Text, out link, out isEmail))
{
state.Inline = new AutolinkInline() {IsEmail = isEmail, Url = link};
}
else
{
state.Text = saved;
string htmlTag;
if (!HtmlHelper.TryParseHtmlTag(ref state.Text, out htmlTag))
{
return false;
}
state.Inline = new HtmlInline() { Tag = htmlTag };
}
return true;
}
}
}

View File

@@ -0,0 +1,98 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class CodeInlineParser : InlineParser
{
public CodeInlineParser()
{
OpeningCharacters = new[] { '`' };
}
public override bool Match(InlineParserState state)
{
var text = state.Text;
int openSticks = 0;
if (text.PeekChar(-1) == '`')
{
return false;
}
while (text.CurrentChar == '`')
{
openSticks++;
text.NextChar();
}
bool isMatching = false;
var builder = state.StringBuilders.Get();
int closeSticks = 0;
var c = text.CurrentChar;
// A backtick string is a string of one or more backtick characters (`) that is neither preceded nor followed by a backtick.
// A code span begins with a backtick string and ends with a backtick string of equal length.
// The contents of the code span are the characters between the two backtick strings, with leading and trailing spaces and line endings removed, and whitespace collapsed to single spaces.
var pc = ' ';
while (c != '\0')
{
// Transform '\n' into a single space
if (c == '\n')
{
c = ' ';
}
if (c != '`' && (c != ' ' || pc != ' '))
{
builder.Append(c);
}
else
{
while (c == '`')
{
closeSticks++;
pc = c;
c = text.NextChar();
}
if (openSticks == closeSticks)
{
break;
}
}
if (closeSticks > 0)
{
builder.Append('`', closeSticks);
closeSticks = 0;
}
else
{
pc = c;
c = text.NextChar();
}
}
if (closeSticks == openSticks)
{
// Remove trailing space
if (builder.Length > 0)
{
if (builder[builder.Length - 1].IsWhitespace())
{
builder.Length--;
}
}
state.Inline = new CodeInline() { Content = builder.ToString() };
isMatching = true;
}
// Release the builder if not used
state.StringBuilders.Release(builder);
return isMatching;
}
}
}

View File

@@ -0,0 +1,111 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class EmphasisInlineParser : InlineParser
{
public EmphasisInlineParser()
{
OpeningCharacters = new[] { '*', '_' };
}
public override bool Match(InlineParserState state)
{
// First, some definitions. A delimiter run is either a sequence of one or more * characters that
// is not preceded or followed by a * character, or a sequence of one or more _ characters that
// is not preceded or followed by a _ character.
var delimiterChar = state.Text.CurrentChar;
var pc = state.Text.PeekChar(-1);
if (delimiterChar == pc && state.Text.PeekChar(-2) != '\\')
{
return false;
}
int delimiterCount = 0;
char c;
do
{
delimiterCount++;
c = state.Text.NextChar();
} while (c == delimiterChar);
// A left-flanking delimiter run is a delimiter run that is
// (a) not followed by Unicode whitespace, and
// (b) either not followed by a punctuation character, or preceded by Unicode whitespace
// or a punctuation character.
// For purposes of this definition, the beginning and the end of the line count as Unicode whitespace.
bool nextIsPunctuation;
bool nextIsWhiteSpace;
bool prevIsPunctuation;
bool prevIsWhiteSpace;
pc.CheckUnicodeCategory(out prevIsWhiteSpace, out prevIsPunctuation);
c.CheckUnicodeCategory(out nextIsWhiteSpace, out nextIsPunctuation);
bool canOpen = !nextIsWhiteSpace &&
(!nextIsPunctuation || prevIsWhiteSpace || prevIsPunctuation);
// A right-flanking delimiter run is a delimiter run that is
// (a) not preceded by Unicode whitespace, and
// (b) either not preceded by a punctuation character, or followed by Unicode whitespace
// or a punctuation character.
// For purposes of this definition, the beginning and the end of the line count as Unicode whitespace.
bool canClose = !prevIsWhiteSpace &&
(!prevIsPunctuation || nextIsWhiteSpace || nextIsPunctuation);
if (delimiterChar == '_')
{
var temp = canOpen;
// A single _ character can open emphasis iff it is part of a left-flanking delimiter run and either
// (a) not part of a right-flanking delimiter run or
// (b) part of a right-flanking delimiter run preceded by punctuation.
canOpen = canOpen && (!canClose || prevIsPunctuation);
// A single _ character can close emphasis iff it is part of a right-flanking delimiter run and either
// (a) not part of a left-flanking delimiter run or
// (b) part of a left-flanking delimiter run followed by punctuation.
canClose = canClose && (!temp || nextIsPunctuation);
}
//// If we can close, try to find a matching open
//if (canClose && state.Inline != null)
//{
// var matching = DelimiterInline.FindMatchingOpen(state.Inline, 0, delimiterRun, delimiterCount);
// // transform matching into
// return true;
//}
// We have potentially an open or close emphasis
if (canOpen || canClose)
{
var delimiterType = DelimiterType.None;
if (canOpen)
{
delimiterType |= DelimiterType.Open;
}
if (canClose)
{
delimiterType |= DelimiterType.Close;
}
var delimiter = new EmphasisDelimiterInline(this)
{
DelimiterChar = delimiterChar,
DelimiterCount = delimiterCount,
Type = delimiterType,
};
state.Inline = delimiter;
return true;
}
// We don't have an emphasis
return false;
}
}
}

View File

@@ -0,0 +1,31 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class EscapeInlineParser : InlineParser
{
public static readonly EscapeInlineParser Default = new EscapeInlineParser();
public EscapeInlineParser()
{
OpeningCharacters = new[] {'\\'};
}
public override bool Match(InlineParserState state)
{
// Go to escape character
var c = state.Text.PeekChar(1);
if (c.IsAsciiPunctuation())
{
var literal = state.Inline as LiteralInline ??
new LiteralInline() {ContentBuilder = state.StringBuilders.Get()};
literal.ContentBuilder.Append(c);
state.Inline = literal;
state.Text.NextChar();
return true;
}
return false;
}
}
}

View File

@@ -0,0 +1,46 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers.Inlines
{
public class HardlineBreakInlineParser : InlineParser
{
public static readonly HardlineBreakInlineParser Default = new HardlineBreakInlineParser();
public HardlineBreakInlineParser()
{
OpeningCharacters = new[] {'\n'};
}
public override bool Match(InlineParserState state)
{
// Hard line breaks are for separating inline content within a block. Neither syntax for hard line breaks works at the end of a paragraph or other block element:
if (!(state.Block is ParagraphBlock) || state.Text.Column == 0 || !state.Text.PeekChar(-1).IsSpace() || !state.Text.PeekChar(-2).IsSpace())
{
return false;
}
//// A line break (not in a code span or HTML tag) that is preceded by two or more spaces
//// and does not occur at the end of a block is parsed as a hard line break (rendered in HTML as a <br /> tag)
//var text = state.Lines;
//int spaceCount = 0;
//var c = text.CurrentChar;
//while (c.IsSpaceOrTab())
//{
// c = text.NextChar();
// spaceCount++;
//}
//if (c == '\\')
//{
// c = text.NextChar();
// spaceCount = 2;
//}
//if (c != '\n' || spaceCount < 2)
//{
// return false;
//}
return false;
}
}
}

View File

@@ -0,0 +1,254 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class LinkInlineParser : InlineParser
{
public static readonly InlineParser Default = new LinkInlineParser();
public LinkInlineParser()
{
OpeningCharacters = new[] {'[', ']', '!'};
}
public override bool Match(InlineParserState state)
{
var c = state.Text.CurrentChar;
bool isImage = false;
if (c == '!')
{
isImage = true;
c = state.Text.NextChar();
if (c != '[')
{
return false;
}
}
switch (c)
{
case '[':
// If this is not an image, we may have a reference link shortcut
// so we try to resolve it here
var saved = state.Text;
string label;
// If the label is followed by either a ( or a [, this is not a shortcut
if (LinkHelper.TryParseLabel(ref state.Text, out label))
{
if (!state.Document.LinkReferenceDefinitions.ContainsKey(label))
{
label = null;
}
}
state.Text = saved;
// Else we insert a LinkDelimiter
state.Text.NextChar();
state.Inline = new LinkDelimiterInline(this)
{
Type = DelimiterType.Open,
Label = label,
IsImage = isImage
};
return true;
case ']':
state.Text.NextChar();
if (state.Inline != null)
{
if (TryProcessLinkOrImage(state, ref state.Text))
{
return true;
}
}
// If we don’t find one, we return a literal text node ].
// (Done after by the LiteralInline parser)
return false;
}
// We don't have an emphasis
return false;
}
private bool ProcessLinkReference(InlineParserState state, string label, bool isImage, Inline child = null)
{
bool isValidLink = false;
LinkReferenceDefinitionBlock linkRef;
if (state.Document.LinkReferenceDefinitions.TryGetValue(label, out linkRef))
{
// Inline Link
var link = new LinkInline()
{
Url = HtmlHelper.Unescape(linkRef.Url),
Title = HtmlHelper.Unescape(linkRef.Title),
IsImage = isImage,
};
if (child == null)
{
child = new LiteralInline()
{
Content = label,
IsClosed = true
};
link.AppendChild(child);
}
else
{
// Insert all child into the link
while (child != null)
{
var next = child.NextSibling;
child.Remove();
link.AppendChild(child);
child = next;
}
}
link.IsClosed = true;
EmphasisInline.ProcessEmphasis(link);
state.Inline = link;
isValidLink = true;
}
//else
//{
// // Else output a literal, leave it opened as we may have literals after
// // that could be append to this one
// var literal = new LiteralInline()
// {
// ContentBuilder = state.StringBuilders.Get().Append('[').Append(label).Append(']')
// };
// state.Inline = literal;
//}
return isValidLink;
}
private bool TryProcessLinkOrImage(InlineParserState inlineState, ref StringSlice text)
{
LinkDelimiterInline openParent = null;
foreach (var parent in inlineState.Inline.FindParentOfType<LinkDelimiterInline>())
{
openParent = parent;
break;
}
// This will be matched as a literal
if (openParent != null)
{
var parentDelimiter = openParent.Parent;
switch (text.CurrentChar)
{
case '(':
string url;
string title;
if (LinkHelper.TryParseInlineLink(ref text, out url, out title))
{
// Inline Link
var link = new LinkInline()
{
Url = HtmlHelper.Unescape(url),
Title = HtmlHelper.Unescape(title),
IsImage = openParent.IsImage,
};
openParent.ReplaceBy(link);
inlineState.Inline = link;
EmphasisInline.ProcessEmphasis(link);
ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter);
link.IsClosed = true;
return true;
}
break;
default:
string label = null;
// Handle Collapsed links
if (text.CurrentChar == '[')
{
if (text.PeekChar(1) == ']')
{
label = openParent.Label;
text.NextChar(); // Skip [
text.NextChar(); // Skip ]
}
}
else
{
label = openParent.Label;
}
if (label != null || LinkHelper.TryParseLabel(ref text, true, out label))
{
if (ProcessLinkReference(inlineState, label, openParent.IsImage,
openParent.FirstChild))
{
// Remove the open parent
openParent.Remove();
ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter);
}
else
{
return false;
}
return true;
}
break;
}
// We have a nested [ ]
// firstParent.Remove();
// The opening [ will be transformed to a literal followed by all the childrens of the [
var literal = new LiteralInline()
{
ContentBuilder = inlineState.StringBuilders.Get().Append(openParent.IsImage ? "![" : "[")
};
inlineState.InlinesToClose.Add(literal);
inlineState.Inline = openParent.ReplaceBy(literal);
return false;
}
return false;
}
private void ReplaceParentIfNotImage(bool isImage, Inline inline)
{
if (isImage || inline == null)
{
return;
}
foreach (var parent in inline.FindParentOfType<LinkDelimiterInline>())
{
if (parent.IsImage)
{
break;
}
var literal = new LiteralInline()
{
Content = "[",
IsClosed = true
};
parent.ReplaceBy(literal);
}
}
private bool TryParseLinkTitle(InlineParserState state)
{
return false;
}
}
}

View File

@@ -0,0 +1,32 @@
using System.Text;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers.Inlines
{
public class LiteralInlineParser : InlineParser
{
public static readonly LiteralInlineParser Default = new LiteralInlineParser();
public override bool Match(InlineParserState state)
{
// A literal will always match
var literal = state.Inline as LiteralInline;
StringBuilder builder;
if (literal == null)
{
builder = state.StringBuilders.Get();
literal = new LiteralInline {ContentBuilder = builder};
state.Inline = literal;
}
else
{
builder = literal.ContentBuilder;
}
var text = state.Text;
builder.Append(text.CurrentChar);
text.NextChar();
return true;
}
}
}

View File

@@ -0,0 +1,398 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class ListBlockParser : BlockParser
{
public ListBlockParser()
{
OpeningCharacters = new[] {'-', '+', '*'};
}
public override BlockState TryOpen(BlockParserState state)
{
// When both a thematic break and a list item are possible
// interpretations of a line, the thematic break takes precedence
if (ThematicBreakParser.Default.TryOpen(state) == BlockState.Break)
{
// Remove the ThematicBreakBlock as we will let the ThematicBreakBlock to catch it later
return BlockState.Break;
}
// 5.2 List items
// TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space
int preIndent = 0;
for (int i = state.Line.Start - 1; i >= 0; i--)
{
if (state.Line[i].IsSpaceOrTab())
{
preIndent++;
}
else
{
break;
}
}
return TryParseListItem(state, null, preIndent);
}
public override BlockState TryContinue(BlockParserState state, Block block)
{
if (block is ListBlock && state.NextContinue is ListItemBlock)
{
// We try to match only on item block if the ListBlock
return BlockState.Skip;
}
// When both a thematic break and a list item are possible
// interpretations of a line, the thematic break takes precedence
var save = state.Line;
if (ThematicBreakBlock.Parser.TryOpen(state) == BlockState.Break)
{
return BlockState.Break;
}
state.Line = save;
state.ParseIndent();
// 5.2 List items
// TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space
int preIndent = 0;
for (int i = state.Line.Start - 1; i >= 0; i--)
{
if (state.Line[i].IsSpaceOrTab())
{
preIndent++;
}
else
{
break;
}
}
var saveLiner = state.Line;
// If we have already a ListItemBlock, we are going to try to append to it
var listItem = block as ListItemBlock;
if (listItem != null)
{
var list = (ListBlock)listItem.Parent;
// Allow all blanks lines if the last block is a fenced code block
// Allow 1 blank line inside a list
// If > 1 blank line, terminate this list
var isBlankLine = state.IsBlankLine;
//if (isBlankLine && !(state.LastBlock is FencedCodeBlock)) // TODO: Handle this case
var isInFencedBlock = state.LastBlock is FencedCodeBlock;
if (isBlankLine)
{
// TODO: Check with a generic way (allow a block to have multiple empty lines)
if (!isInFencedBlock)
{
if (!(state.NextContinue is ListBlock))
{
list.CountAllBlankLines++;
listItem.Children.Add(BlankLineBlock.Instance);
}
list.CountBlankLinesReset++;
}
if (list.CountBlankLinesReset > 1)
{
// TODO: Close all lists and not only this one
return BlockState.BreakDiscard;
}
if (list.CountBlankLinesReset == 1 && listItem.NumberOfSpaces < 0)
{
state.Close(listItem);
// Leave the list open
list.IsOpen = true;
return BlockState.Continue;
}
return BlockState.Continue;
}
list.CountBlankLinesReset = 0;
var c = state.Line.CurrentChar;
var startPosition = state.Line.Start;
// List Item starting with a blank line (-1)
if (listItem.NumberOfSpaces < 0)
{
int expectedCount = -listItem.NumberOfSpaces;
int countSpaces = 0;
var saved = new StringSlice();
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
countSpaces = preIndent + state.Line.Column - startPosition;
if (countSpaces == expectedCount)
{
saved = state.Line;
}
else if (countSpaces >= 4)
{
state.Line = saved;
state.ParseIndent();
countSpaces = expectedCount;
break;
}
}
if (countSpaces == expectedCount)
{
listItem.NumberOfSpaces = countSpaces;
return BlockState.Continue;
}
}
else
{
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
var countSpaces = preIndent + state.Line.Column - startPosition;
if (countSpaces >= listItem.NumberOfSpaces)
{
return BlockState.Continue;
}
}
}
state.Line = saveLiner;
state.ParseIndent();
}
return TryParseListItem(state, block, preIndent);
}
private BlockState TryParseListItem(BlockParserState state, Block block, int preIndent)
{
var isInList = block is ListItemBlock;
var preStartPosition = state.Line.Start;
var c = state.Line.CurrentChar;
if (isInList)
{
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
}
}
else
{
// TODO
//state.Line.SkipLeadingSpaces3();
c = state.Line.CurrentChar;
}
preIndent = preIndent + state.Line.Start - preStartPosition;
var isOrdered = false;
var bulletChar = (char) 0;
int orderedStart = 0;
var orderedDelimiter = (char) 0;
var column = state.Line.Start;
if (c.IsBulletListMarker())
{
bulletChar = c;
preIndent++;
}
else if (c.IsDigit())
{
int countDigit = 0;
while (c.IsDigit())
{
orderedStart = orderedStart*10 + c - '0';
c = state.Line.NextChar();
preIndent++;
countDigit++;
}
// Note that ordered list start numbers must be nine digits or less:
if (countDigit > 9)
{
return BlockState.None;
}
// We don't have an ordered list
if (c != '.' && c != ')')
{
return BlockState.None;
}
preIndent++;
isOrdered = true;
orderedDelimiter = c;
}
else
{
return BlockState.None;
}
// Skip Bullet or '.'
state.Line.NextChar();
// Item starting with a blank line
int numberOfSpaces;
if (state.IsBlankLine)
{
// Use a negative number to store the number of expected chars
numberOfSpaces = -(preIndent + 1);
}
else
{
var startPosition = -1;
int countSpaceAfterBullet = 0;
var saved = new StringSlice();
for (int i = 0; i <= 4; i++)
{
c = state.Line.CurrentChar;
if (!c.IsSpaceOrTab())
{
break;
}
if (i == 0)
{
startPosition = state.Line.Column;
}
var endPosition = state.Line.Column;
countSpaceAfterBullet = endPosition - startPosition;
if (countSpaceAfterBullet == 1)
{
saved = state.Line;
}
else if (countSpaceAfterBullet >= 4)
{
//state.Line.SpaceHeaderCount = countSpaceAfterBullet - 4;
countSpaceAfterBullet = 0;
state.Line = saved;
state.ParseIndent();
break;
}
state.Line.NextChar();
}
// If we haven't matched any spaces, early exit
if (startPosition < 0)
{
return BlockState.None;
}
// Number of spaces required for the following content to be part of this list item
numberOfSpaces = preIndent + countSpaceAfterBullet + 1;
}
var newListItem = new ListItemBlock(this)
{
Column = column,
NumberOfSpaces = numberOfSpaces
};
state.NewBlocks.Push(newListItem);
var currentListItem = block as ListItemBlock;
var currentParent = block as ListBlock ?? (ListBlock)currentListItem?.Parent;
if (currentParent != null)
{
// If we have a new list item, close the previous one
if (currentListItem != null)
{
state.Close(currentListItem);
}
// Reset the list if it is a new list or a new type of bullet
if (currentParent.IsOrdered != isOrdered ||
(isOrdered && currentParent.OrderedDelimiter != orderedDelimiter) ||
(!isOrdered && currentParent.BulletChar != bulletChar)
//(numberOfSpaces < ((ListItemBlock) currentParent.LastChild).NumberOfSpaces)
)
{
state.Close(currentParent);
currentParent = null;
}
}
if (currentParent == null)
{
var newList = new ListBlock(this)
{
Column = column,
IsOrdered = isOrdered,
BulletChar = bulletChar,
OrderedDelimiter = orderedDelimiter,
OrderedStart = orderedStart,
};
state.NewBlocks.Push(newList);
}
return BlockState.Continue;
}
public override bool Close(BlockParserState state, Block blockToClose)
{
var listBlock = blockToClose as ListBlock;
// Process only if we have blank lines
if (listBlock == null || listBlock.CountAllBlankLines <= 0)
{
return true;
}
// TODO: This code is UGLY and WAY TOO LONG, simplify!
bool isLastListItem = true;
for (int listIndex = listBlock.Children.Count - 1; listIndex >= 0; listIndex--)
{
var block = listBlock.Children[listIndex];
var listItem = (ListItemBlock) block;
var children = listItem.Children;
bool isLastElement = true;
for (int i = children.Count - 1; i >= 0; i--)
{
var item = children[i];
if (item is BlankLineBlock)
{
if ((isLastElement && listIndex < listBlock.Children.Count - 1) || (children.Count > 2 && (i > 0 && i < (children.Count - 1))))
{
listBlock.IsLoose = true;
}
if (isLastElement && isLastListItem)
{
// Inform the outer list that we have a blank line
var parentListItemBlock = listBlock.Parent as ListItemBlock;
if (parentListItemBlock != null)
{
var parentList = (ListBlock) parentListItemBlock.Parent;
parentList.CountAllBlankLines++;
parentListItemBlock.Children.Add(BlankLineBlock.Instance);
}
}
children.RemoveAt(i);
// If we have remove all blank lines, we can exit
listBlock.CountAllBlankLines--;
if (listBlock.CountAllBlankLines == 0)
{
return false;
}
}
isLastElement = false;
}
isLastListItem = false;
}
return true;
}
}
}

View File

@@ -0,0 +1,229 @@
using System.Collections.Generic;
using System.IO;
using System.Threading.Tasks;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Parsers
{
public class MarkdownParser
{
public static TextWriter Log;
private readonly ParserList<InlineParser> inlineParsers;
private readonly Document document;
private readonly BlockParserState blockParserState;
private readonly StringBuilderCache stringBuilderCache;
public MarkdownParser(TextReader reader)
{
document = new Document();
Reader = reader;
stringBuilderCache = new StringBuilderCache();
blockParserState = new BlockParserState(stringBuilderCache, document);
inlineParsers = new ParserList<InlineParser>()
{
LinkInline.Parser,
EmphasisInline.Parser,
EscapeInline.Parser,
CodeInline.Parser,
AutolinkInline.Parser,
HardlineBreakInline.Parser,
LiteralInline.Parser,
};
inlineParsers.Initialize();
}
public TextReader Reader { get; }
public Document Parse()
{
ParseLines();
//ProcessInlines(document);
return document;
}
private void ParseLines()
{
while (true)
{
var lineText = Reader.ReadLine();
// If this is the end of file and the last line is empty
if (lineText == null)
{
break;
}
blockParserState.ProcessLine(lineText);
}
blockParserState.CloseAll(true);
}
private void ProcessInlines(ContainerBlock container)
{
var list = new Stack<ContainerBlock>();
list.Push(container);
var leafs = new List<Task>();
while (list.Count > 0)
{
container = list.Pop();
foreach (var block in container.Children)
{
var leafBlock = block as LeafBlock;
if (leafBlock != null)
{
if (leafBlock.ProcessInlines)
{
var task = new Task(() => ProcessInlineLeaf(leafBlock));
task.Start();
leafs.Add(task);
//ProcessInlineLeaf(leafBlock);
}
}
else
{
list.Push((ContainerBlock)block);
}
}
}
Task.WaitAll(leafs.ToArray());
}
private void ProcessInlineLeaf(LeafBlock leafBlock)
{
var lines = leafBlock.Lines;
leafBlock.Inline = new ContainerInline() {IsClosed = false};
var inlineState = new InlineParserState(stringBuilderCache, document)
{
Text = new StringSlice(leafBlock.Lines.ToString()),
Inline = leafBlock.Inline,
Block = leafBlock
};
while (!inlineState.Text.IsEndOfSlice)
{
var saveLine = inlineState.Text;
var c = saveLine.CurrentChar;
var parsers = inlineParsers.GetParsersForOpeningCharacter(c);
bool match = false;
if (parsers != null)
{
for (int i = 0; i < parsers.Length; i++)
{
if (parsers[i].Match(inlineState))
{
match = true;
break;
}
}
}
parsers = inlineParsers.GlobalParsers;
if (!match && parsers != null)
{
for (int i = 0; i < parsers.Length; i++)
{
if (parsers[i].Match(inlineState))
{
break;
}
}
}
var nextInline = inlineState.Inline;
if (nextInline != null)
{
if (nextInline.Parent == null)
{
// Get deepest container
var container = (ContainerInline)leafBlock.Inline;
while (true)
{
var nextContainer = container.LastChild as ContainerInline;
if (nextContainer != null && !nextContainer.IsClosed)
{
container = nextContainer;
}
else
{
break;
}
}
container.AppendChild(nextInline);
}
if (nextInline.IsClosable && !nextInline.IsClosed)
{
var inlinesToClose = inlineState.InlinesToClose;
var last = inlinesToClose.Count > 0
? inlineState.InlinesToClose[inlinesToClose.Count - 1]
: null;
if (last != nextInline)
{
inlineState.InlinesToClose.Add(nextInline);
}
}
}
else
{
// Get deepest container
var container = (ContainerInline)leafBlock.Inline;
while (true)
{
var nextContainer = container.LastChild as ContainerInline;
if (nextContainer != null && !nextContainer.IsClosed)
{
container = nextContainer;
}
else
{
break;
}
}
inlineState.Inline = container.LastChild is LeafInline ? container.LastChild : container;
}
if (Log != null)
{
Log.WriteLine($"** Dump: char '{c}");
leafBlock.Inline.DumpTo(Log);
}
}
// Close all inlines not closed
inlineState.Inline = null;
foreach (var inline in inlineState.InlinesToClose)
{
inline.CloseInternal(inlineState);
}
inlineState.InlinesToClose.Clear();
if (Log != null)
{
Log.WriteLine("** Dump before Emphasis:");
leafBlock.Inline.DumpTo(Log);
EmphasisInline.ProcessEmphasis(leafBlock.Inline);
Log.WriteLine();
Log.WriteLine("** Dump after Emphasis:");
leafBlock.Inline.DumpTo(Log);
}
// TODO: Close opened inlines
// Close last inline
//while (inlineStack.Count > 0)
//{
// var inlineState = inlineStack.Pop();
// inlineState.Parser.Close(state, inlineState.Inline);
//}
}
}
}

View File

@@ -0,0 +1,166 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class ParagraphBlockParser : BlockParser
{
public override BlockState TryOpen(BlockParserState state)
{
if (state.IsBlankLine)
{
return BlockState.None;
}
// We continue trying to match by default
state.NewBlocks.Push(new ParagraphBlock(this) {Column = state.Column});
return BlockState.Continue;
}
public override BlockState TryContinue(BlockParserState state, Block block)
{
return state.IsBlankLine ? BlockState.BreakDiscard : BlockState.Continue;
}
public override bool Close(BlockParserState state, Block block)
{
var paragraph = block as ParagraphBlock;
var heading = block as HeadingBlock;
if (paragraph != null)
{
var lines = paragraph.Lines;
TryMatchLinkReferenceDefinition(lines, state);
// If Paragraph is empty, we can discard it
if (lines.Count == 0)
{
return false;
}
var lineCount = lines.Count;
for (int i = 0; i < lineCount; i++)
{
lines.Slices[i].Trim();
}
}
else // if (heading?.Lines.Count > 1)
{
//heading.Lines.RemoveAt(heading.Lines.Count - 1);
}
return true;
}
private BlockState TryParseSetexHeading(BlockParserState state, Block block)
{
var paragraph = (ParagraphBlock) block;
var headingChar = (char)0;
bool checkForSpaces = false;
for (int i = state.Start; i <= state.EndOffset; i++)
{
var c = state.Line[i];
if (headingChar == 0)
{
if (c == '=' || c == '-')
{
headingChar = c;
continue;
}
break;
}
if (checkForSpaces)
{
if (!c.IsSpaceOrTab())
{
headingChar = (char)0;
break;
}
}
else if (c != headingChar)
{
if (c.IsSpaceOrTab())
{
checkForSpaces = true;
}
else
{
headingChar = (char)0;
break;
}
}
}
if (headingChar != 0)
{
state.Discard(paragraph);
// If we matched a LinkReferenceDefinition before matching the heading, and the remaining
// lines are empty, we can early exit and remove the paragraph
if (!TryMatchLinkReferenceDefinition(paragraph.Lines, state) && paragraph.Lines.Count == 0)
{
var level = headingChar == '=' ? 1 : 2;
var heading = new HeadingBlock(this)
{
Column = paragraph.Column,
Level = level,
Lines = paragraph.Lines,
};
heading.Lines.Trim();
// Remove the paragraph as a pending block
state.NewBlocks.Push(heading);
}
return BlockState.BreakDiscard;
}
return BlockState.Continue;
}
private bool TryMatchLinkReferenceDefinition(StringSliceList localLineGroup, BlockParserState state)
{
bool atLeastOneFound = false;
//var saved = new StringSliceList.State();
//while (true)
//{
// // If we have found a LinkReferenceDefinition, we can discard the previous paragraph
// localLineGroup.Save(ref saved);
// LinkReferenceDefinitionBlock linkReferenceDefinition;
// if (LinkReferenceDefinitionBlock.TryParse(localLineGroup, out linkReferenceDefinition))
// {
// if (!state.Root.LinkReferenceDefinitions.ContainsKey(linkReferenceDefinition.Label))
// {
// state.Root.LinkReferenceDefinitions[linkReferenceDefinition.Label] = linkReferenceDefinition;
// }
// atLeastOneFound = true;
// // Remove lines that have been matched
// if (localLineGroup.LinePosition == localLineGroup.Count)
// {
// localLineGroup.Clear();
// }
// else
// {
// for (int i = localLineGroup.LinePosition - 1; i >= 0; i--)
// {
// localLineGroup.RemoveAt(i);
// }
// }
// }
// else
// {
// if (!atLeastOneFound)
// {
// localLineGroup.Restore(ref saved);
// }
// break;
// }
//}
return atLeastOneFound;
}
}
}

View File

@@ -0,0 +1,72 @@
using System.Collections.Generic;
namespace Textamina.Markdig.Parsers
{
public class ParserList<T> : List<T> where T : class, ICharacterParser
{
private T[][] parsersWithOpeningCharacters;
private T[] globalParsers;
public T[] GlobalParsers => globalParsers;
public T[] GetParsersForOpeningCharacter(char openingChar)
{
return openingChar < parsersWithOpeningCharacters.Length ? parsersWithOpeningCharacters[openingChar] : null;
}
public void Initialize()
{
var charCounter = new Dictionary<char, int>();
int globalCounter = 0;
int maxChar = 0;
foreach (var parser in this)
{
if (parser.OpeningCharacters != null && parser.OpeningCharacters.Length != 0)
{
foreach (var openingChar in parser.OpeningCharacters)
{
if (!charCounter.ContainsKey(openingChar))
{
charCounter[openingChar] = 0;
}
charCounter[openingChar]++;
if (openingChar > maxChar)
{
maxChar = openingChar;
}
}
}
else
{
globalCounter++;
}
}
globalParsers = new T[globalCounter];
parsersWithOpeningCharacters = new T[maxChar+1][];
foreach (var parser in this)
{
if (parser.OpeningCharacters != null && parser.OpeningCharacters.Length != 0)
{
foreach (var openingChar in parser.OpeningCharacters)
{
if (parsersWithOpeningCharacters[openingChar] == null)
{
parsersWithOpeningCharacters[openingChar] = new T[charCounter[openingChar]];
}
var list = parsersWithOpeningCharacters[openingChar];
var index = list.Length - charCounter[openingChar];
list[index] = parser;
charCounter[openingChar]--;
}
}
else
{
globalParsers[globalParsers.Length - globalCounter] = parser;
globalCounter--;
}
}
}
}
}

View File

@@ -0,0 +1,60 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class QuoteBlockParser : BlockParser
{
public QuoteBlockParser()
{
OpeningCharacters = new[] {'>'};
}
public override BlockState TryOpen(BlockParserState state)
{
if (state.IsCodeIndent)
{
return BlockState.None;
}
var column = state.Column;
// 5.1 Block quotes
// A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space.
var quoteChar = state.CurrentChar;
var c = state.NextChar();
if (c.IsSpace())
{
state.NextChar();
}
state.NewBlocks.Push(new QuoteBlock(this) {QuoteChar = quoteChar, Column = column});
return BlockState.Continue;
}
public override BlockState TryContinue(BlockParserState state, Block block)
{
if (state.IsCodeIndent)
{
return BlockState.None;
}
var quote = (QuoteBlock) block;
// 5.1 Block quotes
// A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space.
var c = state.CurrentChar;
if (c != quote.QuoteChar)
{
return state.IsBlankLine ? BlockState.BreakDiscard : BlockState.None;
}
c = state.NextChar(); // Skip opening char
if (c.IsSpace())
{
state.NextChar(); // Skip following space
}
return BlockState.Continue;
}
}
}

View File

@@ -0,0 +1,77 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsers
{
public class ThematicBreakParser : BlockParser
{
public static readonly ThematicBreakParser Default = new ThematicBreakParser();
public ThematicBreakParser()
{
OpeningCharacters = new[] {'-', '_', '*'};
}
public override BlockState TryOpen(BlockParserState state)
{
if (state.IsCodeIndent)
{
return BlockState.None;
}
// 4.1 Thematic breaks
// A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces
int breakCharCount = 1;
var breakChar = state.CurrentChar;
bool hasSpacesSinceLastMatch = false;
bool hasInnerSpaces = false;
int offset = 0;
var c = state.PeekChar(offset++);
while (c != '\0')
{
if (c == breakChar)
{
if (hasSpacesSinceLastMatch)
{
hasInnerSpaces = true;
}
breakCharCount++;
}
else if (c.IsSpace())
{
hasSpacesSinceLastMatch = true;
}
else
{
return BlockState.None;
}
c = state.PeekChar(offset);
offset++;
}
// If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it
var previousParagraph = state.LastBlock as ParagraphBlock;
var isSetexHeading = previousParagraph != null && breakChar == '-' && !hasInnerSpaces;
if (isSetexHeading)
{
var parent = previousParagraph.Parent;
if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != state.Column))
{
isSetexHeading = false;
}
}
if (breakCharCount < 3 || isSetexHeading)
{
return BlockState.None;
}
// Push a new block
state.NewBlocks.Push(new ThematicBreakBlock(this) { Column = state.Column });
return BlockState.BreakDiscard;
}
}
}

View File

@@ -1,24 +0,0 @@
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsing
{
public abstract class BlockParser
{
protected BlockParser()
{
// By default, all blocks can interrupt a paragraph except:
// - setext heading
// - indented code block
// - a special HTML blocks
CanInterruptParagraph = true;
}
public bool CanInterruptParagraph { get; protected set; }
public abstract MatchLineResult Match(BlockParserState state);
public virtual void Close(BlockParserState state)
{
}
}
}

View File

@@ -1,176 +0,0 @@
using System;
using System.Collections.Generic;
using System.Runtime.CompilerServices;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsing
{
public class BlockParserState : List<Block>
{
public BlockParserState(StringBuilderCache stringBuilders, Document root)
{
if (stringBuilders == null) throw new ArgumentNullException(nameof(stringBuilders));
if (root == null) throw new ArgumentNullException(nameof(root));
StringBuilders = stringBuilders;
Root = root;
NewBlocks = new Stack<Block>();
Add(root);
}
public StringSlice Line;
public int LineIndex;
public bool IsBlankLine => CurrentChar == '\0';
public bool IsEndOfLine => Line.IsEndOfSlice;
public char CurrentChar => Line.CurrentChar;
public char NextChar() => Line.NextChar();
public char CharAt(int index) => Line[index];
public int Start => Line.Start;
public int EndOffset => Line.End;
public int Indent => Column - ColumnBegin;
public bool IsCodeIndent => Indent >= 4;
public int ColumnBegin { get; private set; }
public int Column { get; set; }
public Block Pending { get; set; }
public int PendingIndex { get; internal set; }
public readonly Stack<Block> NewBlocks;
public ContainerBlock CurrentContainer;
public Block LastBlock;
public readonly Document Root;
public StringBuilderCache StringBuilders { get; }
public char PeekChar(int offset)
{
return Line.PeekChar(offset);
}
[MethodImpl(MethodImplOptionPortable.AggressiveInlining)]
public void SetCurrentLine(ref StringSlice line)
{
Line = line;
EatSpaces();
}
public void EatSpaces()
{
var c = CurrentChar;
ColumnBegin = Column;
while (c !='\0')
{
if (c == ' ')
{
Column++;
}
else if (c == '\t')
{
Column = ((Column + 3) >> 2) << 2;
}
else
{
break;
}
c = NextChar();
}
}
public Block NextPending
{
get { return PendingIndex + 1 < Count ? this[PendingIndex + 1] : null; }
}
public void Close(Block block)
{
// If we close a block, we close all blocks above
for (int i = Count - 1; i >= 1; i--)
{
if (this[i] == block)
{
for (int j = Count - 1; j >= i; j--)
{
Close(j);
}
break;
}
}
}
public void Discard(Block block)
{
for (int i = Count - 1; i >= 1; i--)
{
if (this[i] == block)
{
if (Pending == block)
{
Pending = null;
}
block.Parent.Children.Remove(block);
RemoveAt(i);
break;
}
}
}
public void Close(int index)
{
var block = this[index];
var saveBlock = Pending;
Pending = block;
block.Parser.Close(this);
// If the pending object is removed, we need to remove it from the parent container
if (Pending == null)
{
var parent = block.Parent as ContainerBlock;
if (parent != null)
{
parent.Children.Remove(block);
}
}
RemoveAt(index);
Pending = saveBlock;
}
public void CloseAll(bool force)
{
// Close any previous blocks not opened
for (int i = Count - 1; i >= 1; i--)
{
var block = this[i];
// Stop on the first open block
if (!force && block.IsOpen)
{
break;
}
Close(i);
}
}
}
}

View File

@@ -1,9 +0,0 @@
namespace Textamina.Markdig.Parsing
{
public abstract class InlineParser
{
public char[] FirstChars { get; protected set; }
public abstract bool Match(InlineParserState state);
}
}

View File

@@ -1,556 +0,0 @@
using System;
using System.Collections.Generic;
using System.Diagnostics;
using System.IO;
using System.Threading.Tasks;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsing
{
public class MarkdownParser
{
public static TextWriter Log;
private readonly List<BlockParser> blockParsers;
private readonly List<InlineParser> inlineParsers;
private readonly List<InlineParser> regularInlineParsers;
private readonly InlineParser[] inlineWithFirstCharParsers;
private readonly Document document;
private readonly BlockParserState blockParserState;
private readonly StringBuilderCache stringBuilderCache;
public MarkdownParser(TextReader reader)
{
document = new Document();
Reader = reader;
blockParsers = new List<BlockParser>();
inlineParsers = new List<InlineParser>();
inlineWithFirstCharParsers = new InlineParser[128];
regularInlineParsers = new List<InlineParser>();
stringBuilderCache = new StringBuilderCache();
blockParserState = new BlockParserState(stringBuilderCache, document);
blockParsers = new List<BlockParser>()
{
ThematicBreakBlock.Parser,
HeadingBlock.Parser,
QuoteBlock.Parser,
ListBlock.Parser,
HtmlBlock.Parser,
CodeBlock.Parser,
FencedCodeBlock.Parser,
ParagraphBlock.Parser,
};
inlineParsers = new List<InlineParser>()
{
LinkInline.Parser,
EmphasisInline.Parser,
EscapeInline.Parser,
CodeInline.Parser,
AutolinkInline.Parser,
HardlineBreakInline.Parser,
LiteralInline.Parser,
};
InitializeInlineParsers();
}
private void InitializeInlineParsers()
{
foreach (var inlineParser in inlineParsers)
{
if (inlineParser.FirstChars != null && inlineParser.FirstChars.Length > 0)
{
foreach (var firstChar in inlineParser.FirstChars)
{
if (firstChar >= 128)
{
throw new InvalidOperationException($"Invalid character '{firstChar}'. Support only ASCII < 128 chars");
}
inlineWithFirstCharParsers[firstChar] = inlineParser;
}
}
else
{
regularInlineParsers.Add(inlineParser);
}
}
}
public TextReader Reader { get; }
private Block LastBlock
{
get
{
var count = blockParserState.Count;
return count > 0 ? blockParserState[count - 1] : null;
}
}
private ContainerBlock LastContainer
{
get
{
for (int i = blockParserState.Count - 1; i >= 0; i--)
{
var container = blockParserState[i] as ContainerBlock;
if (container != null)
{
return container;
}
}
return null;
}
}
public Document Parse()
{
ParseLines();
//ProcessInlines(document);
return document;
}
private void ParseLines()
{
while (true)
{
var lineText = Reader.ReadLine();
// If this is the end of file and the last line is empty
if (lineText == null)
{
break;
}
var line = new StringSlice(lineText);
blockParserState.SetCurrentLine(ref line);
blockParserState.LineIndex++;
bool continueProcessLiner = ProcessPendingBlocks();
// If we have already reached eol and the last block was a paragraph
// we close it
if (blockParserState.Line.IsEndOfSlice)
{
int index = blockParserState.Count - 1;
if (blockParserState[index] is ParagraphBlock)
{
blockParserState.Close(index);
continue;
}
}
// If the line was not entirely processed by pending blocks, try to process it with any new block
while (continueProcessLiner)
{
ParseNewBlocks(ref continueProcessLiner);
}
// Close blocks that are no longer opened
blockParserState.CloseAll(false);
}
blockParserState.CloseAll(true);
// Close opened blocks
//ProcessPendingBlocks(true);
}
private void ProcessInlines(ContainerBlock container)
{
var list = new Stack<ContainerBlock>();
list.Push(container);
var leafs = new List<Task>();
while (list.Count > 0)
{
container = list.Pop();
foreach (var block in container.Children)
{
var leafBlock = block as LeafBlock;
if (leafBlock != null)
{
if (!leafBlock.NoInline)
{
var task = new Task(() => ProcessInlineLeaf(leafBlock));
task.Start();
leafs.Add(task);
//ProcessInlineLeaf(leafBlock);
}
}
else
{
list.Push((ContainerBlock)block);
}
}
}
Task.WaitAll(leafs.ToArray());
}
private bool ProcessPendingBlocks()
{
bool processLiner = true;
// Set all blocks non opened.
// They will be marked as open in the following loop
for (int i = 1; i < blockParserState.Count; i++)
{
blockParserState[i].IsOpen = false;
}
// Create the line state that will be used by all parser
blockParserState.Pending = null;
// Process any current block potentially opened
for (int i = 1; i < blockParserState.Count; i++)
{
var block = blockParserState[i];
// Else tries to match the Parser with the current line
var parser = block.Parser;
blockParserState.Pending = block;
// If we have a paragraph block, we want to try to match over blocks before trying the Paragraph
if (blockParserState.Pending is ParagraphBlock)
{
break;
}
var saveLiner = blockParserState.Line;
// If we have a discard, we can remove it from the current state
blockParserState.CurrentContainer = LastContainer;
blockParserState.PendingIndex = i;
blockParserState.LastBlock = LastBlock;
var result = parser.Match(blockParserState);
if (result == MatchLineResult.Skip)
{
continue;
}
if (result == MatchLineResult.None)
{
// Restore the Line where it was
blockParserState.SetCurrentLine(ref saveLiner);
break;
}
// In case the BlockParser has modified the blockParserState we are iterating on
if (i >= blockParserState.Count)
{
i = blockParserState.Count - 1;
}
// If a parser is adding a block, it must be the last of the list
if ((i + 1) < blockParserState.Count && blockParserState.NewBlocks.Count > 0)
{
throw new InvalidOperationException("A pending parser cannot add a new block when it is not the last pending block");
}
// If we have a leaf block
var leaf = blockParserState.Pending as LeafBlock;
if (leaf != null && blockParserState.NewBlocks.Count == 0)
{
processLiner = false;
if (result != MatchLineResult.LastDiscard && result != MatchLineResult.ContinueDiscard)
{
leaf.Lines.Append(ref blockParserState.Line);
}
if (blockParserState.NewBlocks.Count > 0)
{
throw new InvalidOperationException(
"The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed");
}
}
// A block is open only if it has a Continue state.
// otherwise it is a Last state, and we don't keep it opened
block.IsOpen = result == MatchLineResult.Continue || result == MatchLineResult.ContinueDiscard;
if (result == MatchLineResult.LastDiscard)
{
processLiner = false;
break;
}
bool isLast = i == blockParserState.Count - 1;
if (processLiner)
{
processLiner = ProcessNewBlocks(result, false);
}
if (isLast || !processLiner)
{
break;
}
}
return processLiner;
}
private void ParseNewBlocks(ref bool continueProcessLiner)
{
blockParserState.Pending = null;
for (int j = 0; j < blockParsers.Count; j++)
{
var blockParser = blockParsers[j];
if (blockParserState.Line.IsEndOfSlice)
{
continueProcessLiner = false;
break;
}
// If a block parser cannot interrupt a paragraph, and the last block is a paragraph
// we can skip this parser
var lastBlock = LastBlock;
var paragraph = lastBlock as ParagraphBlock;
if (paragraph != null && !blockParser.CanInterruptParagraph)
{
continue;
}
bool isParsingParagraph = blockParser == ParagraphBlock.Parser;
blockParserState.Pending = isParsingParagraph ? paragraph : null;
blockParserState.CurrentContainer = LastContainer;
blockParserState.LastBlock = lastBlock;
var saveLiner = blockParserState.Line;
var result = blockParser.Match(blockParserState);
if (result == MatchLineResult.None)
{
// If we have reached a blank line after trying to parse a paragraph
// we can ignore it
if (isParsingParagraph && blockParserState.IsBlankLine)
{
continueProcessLiner = false;
break;
}
blockParserState.SetCurrentLine(ref saveLiner);
continue;
}
// Special case for paragraph
paragraph = LastBlock as ParagraphBlock;
if (isParsingParagraph && paragraph != null)
{
Debug.Assert(blockParserState.NewBlocks.Count == 0);
continueProcessLiner = false;
paragraph.Lines.Append(ref blockParserState.Line);
// We have just found a lazy continuation for a paragraph, early exit
// Mark all block opened after a lazy continuation
for (int i = 0; i < blockParserState.Count; i++)
{
blockParserState[i].IsOpen = true;
}
break;
}
// Nothing found but the BlockParser may instruct to break, so early exit
if (blockParserState.NewBlocks.Count == 0 && result == MatchLineResult.LastDiscard)
{
continueProcessLiner = false;
break;
}
continueProcessLiner = ProcessNewBlocks(result, true);
// If we have a container, we can retry to match against all types of block.
if (continueProcessLiner)
{
// rewind to the first parser
j = -1;
}
else
{
// We have a leaf node, we can stop
break;
}
}
}
private bool ProcessNewBlocks(MatchLineResult result, bool allowClosing)
{
var newBlocks = blockParserState.NewBlocks;
while (newBlocks.Count > 0)
{
var block = newBlocks.Pop();
block.Line = blockParserState.LineIndex;
// If we have a leaf block
var leaf = block as LeafBlock;
if (leaf != null)
{
if (result != MatchLineResult.LastDiscard && result != MatchLineResult.ContinueDiscard)
{
leaf.Lines.Append(ref blockParserState.Line);
}
if (newBlocks.Count > 0)
{
throw new InvalidOperationException(
"The NewBlocks is not empty. This is happening if a LeafBlock is not the last to be pushed");
}
}
if (allowClosing)
{
// Close any previous blocks not opened
blockParserState.CloseAll(false);
}
// If previous block is a container, add the new block as a children of the previous block
if (block.Parent == null)
{
var container = LastContainer;
LastContainer.Children.Add(block);
block.Parent = container;
}
block.IsOpen = result == MatchLineResult.Continue || result == MatchLineResult.ContinueDiscard;
// Add a block blockParserState to the stack (and leave it opened)
blockParserState.Add(block);
if (leaf != null)
{
return false;
}
}
return true;
}
private void ProcessInlineLeaf(LeafBlock leafBlock)
{
var lines = leafBlock.Lines;
leafBlock.Inline = new ContainerInline() {IsClosed = false};
var inlineState = new InlineParserState(stringBuilderCache, document)
{
Text = new StringSlice(leafBlock.Lines.ToString()),
Inline = leafBlock.Inline,
Block = leafBlock
};
while (!inlineState.Text.IsEndOfSlice)
{
var saveLine = inlineState.Text;
var c = saveLine.CurrentChar;
var inlineParser = c < 128 ? inlineWithFirstCharParsers[c] : null;
if (inlineParser == null || !inlineParser.Match(inlineState))
{
for (int i = 0; i < regularInlineParsers.Count; i++)
{
inlineState.Text = saveLine;
inlineParser = regularInlineParsers[i];
if (inlineParser.Match(inlineState))
{
break;
}
inlineParser = null;
}
if (inlineParser == null)
{
inlineState.Text = saveLine;
}
}
var nextInline = inlineState.Inline;
if (nextInline != null)
{
if (nextInline.Parent == null)
{
// Get deepest container
var container = (ContainerInline)leafBlock.Inline;
while (true)
{
var nextContainer = container.LastChild as ContainerInline;
if (nextContainer != null && !nextContainer.IsClosed)
{
container = nextContainer;
}
else
{
break;
}
}
container.AppendChild(nextInline);
}
if (nextInline.IsClosable && !nextInline.IsClosed)
{
var inlinesToClose = inlineState.InlinesToClose;
var last = inlinesToClose.Count > 0
? inlineState.InlinesToClose[inlinesToClose.Count - 1]
: null;
if (last != nextInline)
{
inlineState.InlinesToClose.Add(nextInline);
}
}
}
else
{
// Get deepest container
var container = (ContainerInline)leafBlock.Inline;
while (true)
{
var nextContainer = container.LastChild as ContainerInline;
if (nextContainer != null && !nextContainer.IsClosed)
{
container = nextContainer;
}
else
{
break;
}
}
inlineState.Inline = container.LastChild is LeafInline ? container.LastChild : container;
}
if (Log != null)
{
Log.WriteLine($"** Dump: char '{c}");
leafBlock.Inline.DumpTo(Log);
}
}
// Close all inlines not closed
inlineState.Inline = null;
foreach (var inline in inlineState.InlinesToClose)
{
inline.CloseInternal(inlineState);
}
inlineState.InlinesToClose.Clear();
if (Log != null)
{
Log.WriteLine("** Dump before Emphasis:");
leafBlock.Inline.DumpTo(Log);
EmphasisInline.ProcessEmphasis(leafBlock.Inline);
Log.WriteLine();
Log.WriteLine("** Dump after Emphasis:");
leafBlock.Inline.DumpTo(Log);
}
// TODO: Close opened inlines
// Close last inline
//while (inlineStack.Count > 0)
//{
// var inlineState = inlineStack.Pop();
// inlineState.Parser.Close(state, inlineState.Inline);
//}
}
}
}

View File

@@ -1,17 +0,0 @@
namespace Textamina.Markdig.Parsing
{
public enum MatchLineResult
{
None,
Skip,
Continue,
ContinueDiscard,
Last,
LastDiscard
}
}

View File

@@ -1,119 +0,0 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax;
namespace Textamina.Markdig.Parsing
{
public interface IBlockParser
{
char[] OpeningCharacters { get; }
BlockState TryOpen(BlockParserState state);
BlockState TryContinue(BlockParserState state);
bool Close(BlockParserState state, Block block);
}
public abstract class NewBlockParser : IBlockParser
{
public char[] OpeningCharacters { get; protected set; }
public abstract BlockState TryOpen(BlockParserState state);
public virtual BlockState TryContinue(BlockParserState state)
{
// By default we don't expect any newline
return BlockState.None;
}
public virtual bool Close(BlockParserState state, Block block)
{
// By default keep the block
return true;
}
}
public class ThematicBreakBlockParser : NewBlockParser
{
public ThematicBreakBlockParser()
{
OpeningCharacters = new [] {'-', '_', '*'};
}
public override BlockState TryOpen(BlockParserState state)
{
var liner = state.Line;
// 4.1 Thematic breaks
// A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces
var c = liner.Current;
var matchChar = liner.Current;
var count = 1;
c = liner.NextChar();
bool hasSpacesSinceLastMatch = false;
bool hasInnerSpaces = false;
while (c != '\0')
{
if (c == matchChar)
{
if (hasSpacesSinceLastMatch)
{
hasInnerSpaces = true;
}
count++;
}
else if (!c.IsSpace())
{
return BlockState.None;
}
else if (c.IsSpace())
{
hasSpacesSinceLastMatch = true;
}
c = liner.NextChar();
}
// If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it
var previousParagraph = state.LastBlock as ParagraphBlock;
var isSetexHeading = previousParagraph != null && matchChar == '-' && !hasInnerSpaces;
if (isSetexHeading)
{
var parent = previousParagraph.Parent;
if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != column))
{
isSetexHeading = false;
}
}
if (count < 3 || isSetexHeading)
{
return BlockState.None;
}
state.NewBlocks.Push(new BreakBlock(this) {Column = column});
return BlockState.LastDiscard;
}
}
public enum BlockState
{
None,
Skip,
Continue,
ContinueDiscard,
Last,
LastDiscard
}
}

View File

@@ -1,6 +1,4 @@
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax
{
public sealed class BlankLineBlock : Block
{

View File

@@ -1,7 +1,4 @@
using System;
using System.Collections.Generic;
using System.Text;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{

View File

@@ -1,4 +1,4 @@
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
@@ -10,32 +10,9 @@ namespace Textamina.Markdig.Syntax
/// </remarks>
public class CodeBlock : LeafBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public CodeBlock(BlockParser parser) : base(parser)
{
NoInline = true;
}
private class ParserInternal : BlockParser
{
public ParserInternal()
{
CanInterruptParagraph = false;
}
public override MatchLineResult Match(BlockParserState state)
{
if (!state.IsCodeIndent || state.IsBlankLine)
{
return state.IsBlankLine && state.Pending != null ? MatchLineResult.LastDiscard : MatchLineResult.None;
}
if (state.Pending == null)
{
state.NewBlocks.Push(new CodeBlock(this) { Column = state.Column });
}
return MatchLineResult.Continue;
}
ProcessInlines = false;
}
}
}

View File

@@ -1,10 +1,10 @@
using System.Collections.Generic;
using System.Diagnostics;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
[DebuggerDisplay("Container: {GetType().Name} Count = {Children.Count}")]
[DebuggerDisplay("{GetType().Name} Count = {Children.Count}")]
public abstract class ContainerBlock : Block
{
protected ContainerBlock(BlockParser parser) : base(parser)
@@ -12,6 +12,7 @@ namespace Textamina.Markdig.Syntax
Children = new List<Block>();
}
// TODO: Remove Children and use only inner list
public List<Block> Children { get; }
public Block LastChild => Children.Count > 0 ? Children[Children.Count - 1] : null;

View File

@@ -1,8 +1,4 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
@@ -14,8 +10,6 @@ namespace Textamina.Markdig.Syntax
/// </remarks>
public class FencedCodeBlock : CodeBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public FencedCodeBlock(BlockParser parser) : base(parser)
{
}
@@ -24,166 +18,10 @@ namespace Textamina.Markdig.Syntax
public string Arguments { get; set; }
private int fencedCharCount;
public int FencedCharCount { get; set; }
private char fencedChar;
public char FencedChar { get; set; }
private int indentCount;
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
int count;
char matchChar;
char c = state.CurrentChar;
int offset = 0;
var currentFenced = state.Pending as FencedCodeBlock;
if (currentFenced != null)
{
count = currentFenced.fencedCharCount;
matchChar = currentFenced.fencedChar;
while (c == matchChar)
{
offset++;
c = state.Line.PeekChar(offset);
}
if (offset >= count)
{
state.Line.TrimEnd(true);
if (state.CurrentChar == matchChar)
{
return MatchLineResult.LastDiscard;
}
}
// TODO: It is unclear how to handle this correctly
// Break only if Eof
return MatchLineResult.Continue;
}
// Else if the we have an indent, it is not valid
if (state.IsCodeIndent)
{
return MatchLineResult.None;
}
count = 0;
matchChar = (char) 0;
while (c != '\0')
{
if (count == 0 && (c == '`' || c == '~'))
{
matchChar = c;
}
else if (c != matchChar)
{
break;
}
count++;
c = state.PeekChar(count);
}
if (count >= 3)
{
return MatchLineResult.None;
}
// TODO: We need to count the number of leading space to remove them on each line
var column = state.Column;
// specs spaces: Is space and tabs? or only spaces? Use space and tab for this case
while (c.IsSpaceOrTab())
{
offset++;
c = state.PeekChar(offset);
}
var start = state.Start + count + offset;
string infoString;
string argString = null;
// An info string cannot contain any backsticks
int firstSpace = -1;
for (int i = start; i <= state.EndOffset; i++)
{
c = state.Line[i];
if (c == '`')
{
return MatchLineResult.None;
}
if (firstSpace < 0 && c.IsSpaceOrTab())
{
firstSpace = i;
}
}
if (firstSpace > 0)
{
infoString = state.Line.Text.Substring(start, firstSpace - start);
// Skip any spaces after info string
firstSpace++;
while (true)
{
c = state.Line[firstSpace];
if (c.IsSpaceOrTab())
{
firstSpace++;
}
else
{
break;
}
}
argString = state.Line.Text.Substring(firstSpace, state.Line.End - firstSpace + 1);
}
else
{
infoString = state.Line.Text.Substring(start, state.EndOffset - start + 1);
}
// Store the number of matched string into the context
state.NewBlocks.Push(new FencedCodeBlock(this)
{
Column = column,
fencedChar = matchChar,
fencedCharCount = count,
indentCount = state.Indent,
Language = HtmlHelper.Unescape(infoString),
Arguments = HtmlHelper.Unescape(argString),
});
// Discard the current line
return MatchLineResult.ContinueDiscard;
}
public override void Close(BlockParserState state)
{
var fenced = ((FencedCodeBlock) state.Pending);
var lines = fenced.Lines;
for (int i = 0; i < lines.Count; i++)
{
// Fences can be indented. If the opening fence is indented,
// content lines will have equivalent opening indentation removed, if present:
for (int j = 0; j < fenced.indentCount; j++)
{
var start = lines.Slices[i].Start;
if (start < lines.Slices[i].End && lines.Slices[i][start].IsSpace())
{
lines.Slices[i].Start++;
}
else
{
break;
}
}
}
}
}
public int IndentCount { get; set; }
}
}

View File

@@ -1,110 +1,18 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using System.Diagnostics;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
/// <summary>
/// Repressents a thematic break.
/// </summary>
[DebuggerDisplay("{GetType().Name} Line: {Line}, {Lines} Level: {Level}")]
public class HeadingBlock : LeafBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public HeadingBlock(BlockParser parser) : base(parser)
{
}
public int Level { get; set; }
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
if (state.IsCodeIndent)
{
return MatchLineResult.None;
}
// 4.2 ATX headings
// An ATX heading consists of a string of characters, parsed as inline content,
// between an opening sequence of 1–6 unescaped # characters and an optional
// closing sequence of any number of unescaped # characters. The opening sequence
// of # characters must be followed by a space or by the end of line. The optional
// closing sequence of #s must be preceded by a space and may be followed by spaces
// only. The opening # character may be indented 0-3 spaces. The raw contents of
// the heading are stripped of leading and trailing spaces before being parsed as
// inline content. The heading level is equal to the number of # characters in the
// opening sequence.
var column = state.Column;
var c = state.CurrentChar;
int leadingCount = 0;
for (; !state.Line.IsEndOfSlice && leadingCount <= 6; leadingCount++)
{
if (c != '#')
{
break;
}
c = state.PeekChar(leadingCount);
}
// closing # will be handled later, because anyway we have matched
// A space is required after leading #
if (leadingCount > 0 && leadingCount <=6 && (c.IsSpace() || state.Line.IsEndOfSlice))
{
state.Line.Start = leadingCount + 1;
state.NewBlocks.Push(new HeadingBlock(this) {Level = leadingCount, Column = column });
// The optional closing sequence of #s must be preceded by a space and may be followed by spaces only.
int endState = 0;
int countClosingTags = 0;
for (int i = state.Line.End; i >= state.Line.Start - 1; i--) // Go up to Start - 1 in order to match the space after the first ###
{
c = state.Line[i];
if (endState == 0)
{
if (c.IsSpace()) // TODO: Not clear if it is a space or space+tab in the specs
{
continue;
}
endState = 1;
}
if (endState == 1)
{
if (c == '#')
{
countClosingTags++;
continue;
}
if (countClosingTags > 0)
{
if (c.IsSpace())
{
state.Line.End = i - 1;
}
break;
}
else
{
break;
}
}
}
return MatchLineResult.Last;
}
return MatchLineResult.None;
}
public override void Close(BlockParserState state)
{
var heading = (HeadingBlock) state.Pending;
heading.Lines.Trim();
}
}
}
}

View File

@@ -1,309 +1,17 @@
using System;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
public class HtmlBlock : LeafBlock
{
public static readonly BlockParser Parser = new ParserInternal();
public static readonly BlockParser Parser = new HtmlBlockParser();
public HtmlBlock(BlockParser parser) : base(parser)
{
// We don't process inline of an html block, as we will copy the content as-is
NoInline = true;
ProcessInlines = false;
}
public HtmlBlockType Type { get; set; }
private class ParserInternal : BlockParser
{
private static readonly string[] HtmlTags =
{
"address", // 0
"article", // 1
"aside", // 2
"base", // 3
"basefont", // 4
"blockquote", // 5
"body", // 6
"caption", // 7
"center", // 8
"col", // 9
"colgroup", // 10
"dd", // 11
"details", // 12
"dialog", // 13
"dir", // 14
"div", // 15
"dl", // 16
"dt", // 17
"fieldset", // 18
"figcaption", // 19
"figure", // 20
"footer", // 21
"form", // 22
"frame", // 23
"frameset", // 24
"h1", // 25
"head", // 26
"header", // 27
"hr", // 28
"html", // 29
"iframe", // 30
"legend", // 31
"li", // 32
"link", // 33
"main", // 34
"menu", // 35
"menuitem", // 36
"meta", // 37
"nav", // 38
"noframes", // 39
"ol", // 40
"optgroup", // 41
"option", // 42
"p", // 43
"param", // 44
"pre", // 45 <- special group 1
"script", // 46 <- special group 1
"section", // 47
"source", // 48
"style", // 49 <- special group 1
"summary", // 50
"table", // 51
"tbody", // 52
"td", // 53
"tfoot", // 54
"th", // 55
"thead", // 56
"title", // 57
"tr", // 58
"track", // 59
"ul", // 60
};
public override MatchLineResult Match(BlockParserState state)
{
var htmlBlock = state.Pending as HtmlBlock;
if (htmlBlock == null)
{
var result = MatchStart(state);
// An end-tag can occur on the same line
if (result == MatchLineResult.Continue)
{
return MatchEnd(state, (HtmlBlock) state.NewBlocks.Peek());
}
return result;
}
return MatchEnd(state, htmlBlock);
}
private MatchLineResult MatchStart(BlockParserState state)
{
int index = 0;
for (int i = 0; i < 3; i++)
{
if (!state.Line.PeekChar(index).IsSpace())
{
break;
}
index++;
}
// Early exit if it is not starting by an HTML tag
var column = index;
var c = state.Line.PeekChar(index++);
if (c != '<')
{
return MatchLineResult.None;
}
var result = TryParseTagType16(state, ref state.Line, index, column);
// HTML blocks of type 7 cannot interrupt a paragraph:
if (result == MatchLineResult.None && !(state.LastBlock is ParagraphBlock))
{
result = TryParseTagType7(state, ref state.Line, index, column);
}
return result;
}
private MatchLineResult TryParseTagType7(BlockParserState state, ref StringSlice liner, int index, int startColumn)
{
var builder = StringBuilderCache.Local();
liner.Start = index;
var c = liner.CurrentChar;
var result = MatchLineResult.None;
if ((c == '/' && HtmlHelper.TryParseHtmlCloseTag(liner, builder)) || HtmlHelper.TryParseHtmlTagOpenTag(liner, builder))
{
// Must be followed by whitespace only
bool hasOnlySpaces = true;
c = liner.CurrentChar;
while (true)
{
if (c == '\0')
{
break;
}
if (!c.IsWhitespace())
{
hasOnlySpaces = false;
break;
}
c = liner.NextChar();
}
if (hasOnlySpaces)
{
result = CreateHtmlBlock(state, HtmlBlockType.NonInterruptingBlock, startColumn);
}
}
builder.Clear();
return result;
}
private MatchLineResult TryParseTagType16(BlockParserState state, ref StringSlice liner, int index, int startColumn)
{
char c;
c = liner.PeekChar(index);
if (c == '!')
{
c = liner.PeekChar(index + 1);
if (c == '-' && liner.PeekChar(index + 2) == '-')
{
return CreateHtmlBlock(state, HtmlBlockType.Comment, startColumn); // group 2
}
if (c.IsAlphaUpper())
{
return CreateHtmlBlock(state, HtmlBlockType.DocumentType, startColumn); // group 4
}
if (c == '[' && liner.Match("CDATA[", 3))
{
return CreateHtmlBlock(state, HtmlBlockType.CData, startColumn); // group 5
}
return MatchLineResult.None;
}
if (c == '?')
{
return CreateHtmlBlock(state, HtmlBlockType.ProcessingInstruction, startColumn); // group 3
}
var hasLeadingClose = c == '/';
if (hasLeadingClose)
{
index++;
}
var tag = new char[10];
var count = 0;
for (; count < tag.Length; index++, count++)
{
c = liner.PeekChar(index);
if (!c.IsAlphaNumeric())
{
break;
}
tag[count] = char.ToLowerInvariant(c);
}
if (
!(c == '>' || (!hasLeadingClose && c == '/' && liner.PeekChar(index + 1) == '>') || c.IsWhitespace() ||
c == '\0'))
{
return MatchLineResult.None;
}
if (count == 0)
{
return MatchLineResult.None;
}
var tagName = new string(tag, 0, count);
var tagIndex = Array.BinarySearch(HtmlTags, tagName, StringComparer.Ordinal);
if (tagIndex < 0)
{
return MatchLineResult.None;
}
// Cannot start with </script </pre or </style
if ((tagIndex == 45 || tagIndex == 46 || tagIndex == 49))
{
if (c == '/' || hasLeadingClose)
{
return MatchLineResult.None;
}
return CreateHtmlBlock(state, HtmlBlockType.ScriptPreOrStyle, startColumn);
}
return CreateHtmlBlock(state, HtmlBlockType.InterruptingBlock, startColumn);
}
private MatchLineResult MatchEnd(BlockParserState state, HtmlBlock htmlBlock)
{
// Early exit if it is not starting by an HTML tag
var c = state.Line.CurrentChar;
switch (htmlBlock.Type)
{
case HtmlBlockType.Comment:
if (state.Line.Search("-->"))
{
return MatchLineResult.Last;
}
break;
case HtmlBlockType.CData:
if (state.Line.Search("]]>"))
{
return MatchLineResult.Last;
}
break;
case HtmlBlockType.ProcessingInstruction:
if (state.Line.Search("?>"))
{
return MatchLineResult.Last;
}
break;
case HtmlBlockType.DocumentType:
if (state.Line.Search(">"))
{
return MatchLineResult.Last;
}
break;
case HtmlBlockType.ScriptPreOrStyle:
// TODO: could be optimized with a dedicated parser
if (state.Line.SearchLowercase("</script>") || state.Line.SearchLowercase("</pre>") || state.Line.SearchLowercase("</style>"))
{
return MatchLineResult.Last;
}
break;
case HtmlBlockType.InterruptingBlock:
if (state.IsBlankLine)
{
return MatchLineResult.LastDiscard;
}
break;
case HtmlBlockType.NonInterruptingBlock:
if (state.IsBlankLine)
{
return MatchLineResult.LastDiscard;
}
break;
}
return MatchLineResult.Continue;
}
private MatchLineResult CreateHtmlBlock(BlockParserState state, HtmlBlockType type, int startColumn)
{
state.NewBlocks.Push(new HtmlBlock(this) {Column = startColumn, Type = type});
return MatchLineResult.Continue;
}
}
}
}

View File

@@ -1,48 +1,16 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
using Textamina.Markdig.Parsers.Inlines;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class AutolinkInline : LeafInline
{
public static readonly InlineParser Parser = new ParserInternal();
public static readonly InlineParser Parser = new AutolineInlineParser();
public bool IsEmail { get; set; }
public string Url { get; set; }
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] {'<'};
}
public override bool Match(InlineParserState state)
{
string link;
bool isEmail;
var saved = state.Text;
if (LinkHelper.TryParseAutolink(ref state.Text, out link, out isEmail))
{
state.Inline = new AutolinkInline() {IsEmail = isEmail, Url = link};
}
else
{
state.Text = saved;
string htmlTag;
if (!HtmlHelper.TryParseHtmlTag(ref state.Text, out htmlTag))
{
return false;
}
state.Inline = new HtmlInline() { Tag = htmlTag };
}
return true;
}
}
public override string ToString()
{
return Url;

View File

@@ -1,105 +1,12 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
using Textamina.Markdig.Parsers.Inlines;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class CodeInline : LeafInline
{
public static readonly InlineParser Parser = new ParserInternal();
public static readonly InlineParser Parser = new CodeInlineParser();
public string Content { get; set; }
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] { '`' };
}
public override bool Match(InlineParserState state)
{
var text = state.Text;
int openSticks = 0;
if (text.PeekChar(-1) == '`')
{
return false;
}
while (text.CurrentChar == '`')
{
openSticks++;
text.NextChar();
}
bool isMatching = false;
var builder = state.StringBuilders.Get();
int closeSticks = 0;
var c = text.CurrentChar;
// A backtick string is a string of one or more backtick characters (`) that is neither preceded nor followed by a backtick.
// A code span begins with a backtick string and ends with a backtick string of equal length.
// The contents of the code span are the characters between the two backtick strings, with leading and trailing spaces and line endings removed, and whitespace collapsed to single spaces.
var pc = ' ';
while (c != '\0')
{
// Transform '\n' into a single space
if (c == '\n')
{
c = ' ';
}
if (c != '`' && (c != ' ' || pc != ' '))
{
builder.Append(c);
}
else
{
while (c == '`')
{
closeSticks++;
pc = c;
c = text.NextChar();
}
if (openSticks == closeSticks)
{
break;
}
}
if (closeSticks > 0)
{
builder.Append('`', closeSticks);
closeSticks = 0;
}
else
{
pc = c;
c = text.NextChar();
}
}
if (closeSticks == openSticks)
{
// Remove trailing space
if (builder.Length > 0)
{
if (builder[builder.Length - 1].IsWhitespace())
{
builder.Length--;
}
}
state.Inline = new CodeInline() { Content = builder.ToString() };
isMatching = true;
}
// Release the builder if not used
state.StringBuilders.Release(builder);
return isMatching;
}
}
}
}

View File

@@ -1,9 +1,8 @@
using System;
using System.Collections.Generic;
using System.IO;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class ContainerInline : Inline
{

View File

@@ -1,8 +1,7 @@
using System;
using System.Collections.Generic;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public abstract class DelimiterInline : ContainerInline
{

View File

@@ -1,6 +1,6 @@
using System;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
[Flags]
public enum DelimiterType

View File

@@ -1,7 +1,6 @@
using System.Collections.Generic;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class EmphasisDelimiterInline : DelimiterInline
{

View File

@@ -1,14 +1,12 @@
using System;
using System.Collections.Generic;
using System.Linq;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
using Textamina.Markdig.Parsers.Inlines;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class EmphasisInline : ContainerInline
{
public static readonly InlineParser Parser = new ParserInternal();
public static readonly InlineParser Parser = new EmphasisInlineParser();
public char DelimiterChar { get; set; }
@@ -205,111 +203,5 @@ namespace Textamina.Markdig.Syntax
}
delimiters.Clear();
}
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] { '*', '_' };
}
public override bool Match(InlineParserState state)
{
// First, some definitions. A delimiter run is either a sequence of one or more * characters that
// is not preceded or followed by a * character, or a sequence of one or more _ characters that
// is not preceded or followed by a _ character.
var delimiterChar = state.Text.CurrentChar;
var pc = state.Text.PeekChar(-1);
if (delimiterChar == pc && state.Text.PeekChar(-2) != '\\')
{
return false;
}
int delimiterCount = 0;
char c;
do
{
delimiterCount++;
c = state.Text.NextChar();
} while (c == delimiterChar);
// A left-flanking delimiter run is a delimiter run that is
// (a) not followed by Unicode whitespace, and
// (b) either not followed by a punctuation character, or preceded by Unicode whitespace
// or a punctuation character.
// For purposes of this definition, the beginning and the end of the line count as Unicode whitespace.
bool nextIsPunctuation;
bool nextIsWhiteSpace;
bool prevIsPunctuation;
bool prevIsWhiteSpace;
pc.CheckUnicodeCategory(out prevIsWhiteSpace, out prevIsPunctuation);
c.CheckUnicodeCategory(out nextIsWhiteSpace, out nextIsPunctuation);
bool canOpen = !nextIsWhiteSpace &&
(!nextIsPunctuation || prevIsWhiteSpace || prevIsPunctuation);
// A right-flanking delimiter run is a delimiter run that is
// (a) not preceded by Unicode whitespace, and
// (b) either not preceded by a punctuation character, or followed by Unicode whitespace
// or a punctuation character.
// For purposes of this definition, the beginning and the end of the line count as Unicode whitespace.
bool canClose = !prevIsWhiteSpace &&
(!prevIsPunctuation || nextIsWhiteSpace || nextIsPunctuation);
if (delimiterChar == '_')
{
var temp = canOpen;
// A single _ character can open emphasis iff it is part of a left-flanking delimiter run and either
// (a) not part of a right-flanking delimiter run or
// (b) part of a right-flanking delimiter run preceded by punctuation.
canOpen = canOpen && (!canClose || prevIsPunctuation);
// A single _ character can close emphasis iff it is part of a right-flanking delimiter run and either
// (a) not part of a left-flanking delimiter run or
// (b) part of a left-flanking delimiter run followed by punctuation.
canClose = canClose && (!temp || nextIsPunctuation);
}
//// If we can close, try to find a matching open
//if (canClose && state.Inline != null)
//{
// var matching = DelimiterInline.FindMatchingOpen(state.Inline, 0, delimiterRun, delimiterCount);
// // transform matching into
// return true;
//}
// We have potentially an open or close emphasis
if (canOpen || canClose)
{
var delimiterType = DelimiterType.None;
if (canOpen)
{
delimiterType |= DelimiterType.Open;
}
if (canClose)
{
delimiterType |= DelimiterType.Close;
}
var delimiter = new EmphasisDelimiterInline(this)
{
DelimiterChar = delimiterChar,
DelimiterCount = delimiterCount,
Type = delimiterType,
};
state.Inline = delimiter;
return true;
}
// We don't have an emphasis
return false;
}
}
}
}

View File

@@ -1,40 +0,0 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
{
/// <summary>
/// There is actually no EscapeInline inheriting from Inline, as
/// the parser will transform it to a LiteralInline
/// </summary>
public static class EscapeInline
{
public static readonly InlineParser Parser = new ParserInternal();
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] {'\\'};
}
public override bool Match(InlineParserState state)
{
var lines = state.Text;
// Go to escape character
lines.NextChar();
if (lines.CurrentChar.IsAsciiPunctuation())
{
var literal = state.Inline as LiteralInline ??
new LiteralInline() {ContentBuilder = state.StringBuilders.Get()};
literal.ContentBuilder.Append(lines.CurrentChar);
state.Inline = literal;
lines.NextChar();
return true;
}
return false;
}
}
}
}

View File

@@ -1,51 +1,7 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class HardlineBreakInline : LeafInline
{
public static readonly InlineParser Parser = new ParserInternal();
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] {'\n'};
}
public override bool Match(InlineParserState state)
{
// Hard line breaks are for separating inline content within a block. Neither syntax for hard line breaks works at the end of a paragraph or other block element:
if (!(state.Block is ParagraphBlock) || state.Text.Column == 0 || !state.Text.PeekChar(-1).IsSpace() || !state.Text.PeekChar(-2).IsSpace())
{
return false;
}
//// A line break (not in a code span or HTML tag) that is preceded by two or more spaces
//// and does not occur at the end of a block is parsed as a hard line break (rendered in HTML as a <br /> tag)
//var text = state.Lines;
//int spaceCount = 0;
//var c = text.CurrentChar;
//while (c.IsSpaceOrTab())
//{
// c = text.NextChar();
// spaceCount++;
//}
//if (c == '\\')
//{
// c = text.NextChar();
// spaceCount = 2;
//}
//if (c != '\n' || spaceCount < 2)
//{
// return false;
//}
return false;
}
}
public override string ToString()
{
return "<br />";

View File

@@ -1,4 +1,4 @@
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class HtmlInline : LeafInline
{

View File

@@ -1,9 +1,9 @@
using System;
using System.Collections.Generic;
using System.IO;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public abstract class Inline
{
@@ -111,7 +111,7 @@ namespace Textamina.Markdig.Syntax
}
else if (parent != null)
{
((ContainerInline) parent).AppendChild(inline);
parent.AppendChild(inline);
}
var container = this as ContainerInline;

View File

@@ -1,319 +0,0 @@
/*
using System;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
{
/// <summary>
/// Describes an element in a stack of possible inline openers.
/// </summary>
internal sealed class InlineStack
{
/// <summary>
/// The parser priority if this stack entry.
/// </summary>
public InlineStackPriority Priority;
/// <summary>
/// Previous entry in the stack. <c>null</c> if this is the last one.
/// </summary>
public InlineStack Previous;
/// <summary>
/// Next entry in the stack. <c>null</c> if this is the last one.
/// </summary>
public InlineStack Next;
/// <summary>
/// The at-the-moment text inline that could be transformed into the opener.
/// </summary>
public Inline StartingInline;
/// <summary>
/// The number of delimiter characters found for this opener.
/// </summary>
public int DelimiterCount;
/// <summary>
/// The character that was used in the opener.
/// </summary>
public char Delimiter;
/// <summary>
/// The position in the <see cref="Buffer"/> where this inline element was found.
/// Used only if the specific parser requires this information.
/// </summary>
public StringLineGroup.BlockState StartPosition;
/// <summary>
/// The flags set for this stack entry.
/// </summary>
public InlineStackFlags Flags;
[Flags]
public enum InlineStackFlags : byte
{
None = 0,
Opener = 1,
Closer = 2,
ImageLink = 4
}
public enum InlineStackPriority : byte
{
Emphasis = 0,
Links = 1,
Maximum = Links
}
public InlineStack FindMatchingOpener(InlineStackPriority priority,
char delimiter, out bool canClose)
{
canClose = true;
var istack = this;
while (true)
{
if (istack == null)
{
// this cannot be a closer since there is no opener available.
canClose = false;
return null;
}
if (istack.Priority > priority ||
(istack.Delimiter == delimiter && 0 != (istack.Flags & InlineStackFlags.Closer)))
{
// there might be a closer further back but we cannot go there yet because a higher priority element is blocking
// the other option is that the stack entry could be a closer for the same char - this means
// that any opener we might find would first have to be matched against this closer.
return null;
}
if (istack.Delimiter == delimiter)
return istack;
istack = istack.Previous;
}
}
public void AppendStackEntry(InlineParserState subj)
{
if (subj.LastPendingInline != null)
{
Previous = subj.LastPendingInline;
subj.LastPendingInline.Next = this;
}
if (subj.FirstPendingInline == null)
subj.FirstPendingInline = this;
subj.LastPendingInline = this;
}
/// <summary>
/// Removes a subset of the stack.
/// </summary>
/// <param name="subj">The subject associated with this stack. Can be <c>null</c> if the pointers in the subject should not be updated.</param>
/// <param name="last">The last entry to be removed. Can be <c>null</c> if everything starting from <paramref name="first" /> has to be removed.</param>
public void RemoveStackEntry(InlineParserState subj, InlineStack last)
{
var first = this;
var curPriority = first.Priority;
if (last == null)
{
if (first.Previous != null)
first.Previous.Next = null;
else if (subj != null)
subj.FirstPendingInline = null;
if (subj != null)
{
last = subj.LastPendingInline;
subj.LastPendingInline = first.Previous;
}
first = first.Next;
}
else
{
if (first.Previous != null)
first.Previous.Next = last.Next;
else if (subj != null)
subj.FirstPendingInline = last.Next;
if (last.Next != null)
last.Next.Previous = first.Previous;
else if (subj != null)
subj.LastPendingInline = first.Previous;
if (first == last)
return;
first = first.Next;
last = last.Previous;
}
if (last == null || first == null)
return;
first.Previous = null;
last.Next = null;
// handle case like [*b*] (the whole [..] is being removed but the inner *..* must still be matched).
// this is not done automatically because the initial * is recognized as a potential closer (assuming
// potential scenario '*[*' ).
if (curPriority > 0)
PostProcessInlineStack(null, first, last, curPriority);
}
public static void PostProcessInlineStack(InlineParserState subj, InlineStack first, InlineStack last,
InlineStackPriority ignorePriority)
{
while (ignorePriority > 0)
{
var istack = first;
while (istack != null)
{
if (istack.Priority >= ignorePriority)
{
istack.RemoveStackEntry(subj, istack);
}
else if (0 != (istack.Flags & InlineStackFlags.Closer))
{
bool canClose;
var iopener = FindMatchingOpener(istack.Previous, istack.Priority, istack.Delimiter,
out canClose);
if (iopener != null)
{
bool retry = false;
if (iopener.Delimiter == '~')
{
iopener.MatchInlineStack(subj, istack.DelimiterCount, istack, true);
if (istack.DelimiterCount > 1)
retry = true;
}
else
{
iopener.MatchInlineStack(subj, istack.DelimiterCount, istack, false);
if (istack.DelimiterCount > 0)
retry = true;
}
if (retry)
{
// remove everything between opened and closer (not inclusive).
if (istack.Previous != null && iopener.Next != istack.Previous)
iopener.Next.RemoveStackEntry(subj, istack.Previous);
continue;
}
else
{
// remove opener, everything in between, and the closer
iopener.RemoveStackEntry(subj, istack);
}
}
else if (!canClose)
{
// this case means that a matching opener does not exist
// remove the Closer flag so that a future Opener can be matched against it.
istack.Flags &= ~InlineStackFlags.Closer;
}
}
if (istack == last)
break;
istack = istack.Next;
}
ignorePriority--;
}
}
public int MatchInlineStack(InlineParserState subj, int closingDelimiterCount, InlineStack closer, bool onlySingleCharTag)
{
// calculate the actual number of delimiters used from this closer
int useDelims;
var openerDelims = this.DelimiterCount;
if (closingDelimiterCount < 3 || openerDelims < 3)
{
useDelims = closingDelimiterCount <= openerDelims ? closingDelimiterCount : openerDelims;
if (useDelims == 1 && onlySingleCharTag)
return 0;
}
else if (onlySingleCharTag)
useDelims = 2;
else
useDelims = closingDelimiterCount % 2 == 0 ? 2 : 1;
Inline inl = this.StartingInline;
InlineTag tag = useDelims == 1 ? singleCharTag : doubleCharTag;
if (openerDelims == useDelims)
{
// the opener is completely used up - remove the stack entry and reuse the inline element
inl.Tag = tag;
inl.LiteralContent = null;
inl.FirstChild = inl.NextSibling;
inl.NextSibling = null;
RemoveStackEntry(subj, closer?.Previous);
}
else
{
// the opener will only partially be used - stack entry remains (truncated) and a new inline is added.
this.DelimiterCount -= useDelims;
inl.LiteralContent = inl.LiteralContent.Substring(0, this.DelimiterCount);
inl.SourceLastPosition -= useDelims;
inl.NextSibling = new Inline(tag, inl.NextSibling);
inl = inl.NextSibling;
inl.SourcePosition = this.StartingInline.SourcePosition + this.DelimiterCount;
}
// there are two callers for this method, distinguished by the `closer` argument.
// if closer == null it means the method is called during the initial subject parsing and the closer
// characters are at the current position in the subject. The main benefit is that there is nothing
// parsed that is located after the matched inline element.
// if closer != null it means the method is called when the second pass for previously unmatched
// stack elements is done. The drawback is that there can be other elements after the closer.
if (closer != null)
{
var clInl = closer.StartingInline;
if ((closer.DelimiterCount -= useDelims) > 0)
{
// a new inline element must be created because the old one has to be the one that
// finalizes the children of the emphasis
var newCloserInline = new Inline(clInl.LiteralContent.Substring(useDelims));
newCloserInline.SourcePosition = inl.SourceLastPosition = clInl.SourcePosition + useDelims;
newCloserInline.SourceLength = closer.DelimiterCount;
newCloserInline.NextSibling = clInl.NextSibling;
clInl.LiteralContent = null;
clInl.NextSibling = null;
inl.NextSibling = closer.StartingInline = newCloserInline;
}
else
{
inl.SourceLastPosition = clInl.SourceLastPosition;
clInl.LiteralContent = null;
inl.NextSibling = clInl.NextSibling;
clInl.NextSibling = null;
}
}
else if (subj != null)
{
inl.SourceLastPosition = subj.Position - closingDelimiterCount + useDelims;
subj.LastInline = inl;
}
return useDelims;
}
}
}
*/

View File

@@ -1,6 +1,4 @@
using System.Text;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public abstract class LeafInline : Inline
{

View File

@@ -1,4 +1,4 @@
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class LineBreakInline : LeafInline
{

View File

@@ -1,6 +1,6 @@
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class LinkDelimiterInline : DelimiterInline
{

View File

@@ -1,15 +1,7 @@
using System;
using System.Text;
using Textamina.Markdig.Formatters;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class LinkInline : ContainerInline
{
public static readonly InlineParser Parser = new ParserInternal();
public string Url { get; set; }
public string Title { get; set; }
@@ -20,251 +12,5 @@ namespace Textamina.Markdig.Syntax
{
return (IsImage ? "<img src=\"" : "<a href=\"") + Url + "\" title=\"" + Title + "\">";
}
private class ParserInternal : InlineParser
{
public ParserInternal()
{
FirstChars = new[] {'[', ']', '!'};
}
public override bool Match(InlineParserState state)
{
var c = state.Text.CurrentChar;
bool isImage = false;
if (c == '!')
{
isImage = true;
c = state.Text.NextChar();
if (c != '[')
{
return false;
}
}
switch (c)
{
case '[':
// If this is not an image, we may have a reference link shortcut
// so we try to resolve it here
var saved = state.Text;
string label;
// If the label is followed by either a ( or a [, this is not a shortcut
if (LinkHelper.TryParseLabel(ref state.Text, out label))
{
if (!state.Document.LinkReferenceDefinitions.ContainsKey(label))
{
label = null;
}
}
state.Text = saved;
// Else we insert a LinkDelimiter
state.Text.NextChar();
state.Inline = new LinkDelimiterInline(this)
{
Type = DelimiterType.Open,
Label = label,
IsImage = isImage
};
return true;
case ']':
state.Text.NextChar();
if (state.Inline != null)
{
if (TryProcessLinkOrImage(state, ref state.Text))
{
return true;
}
}
// If we don’t find one, we return a literal text node ].
// (Done after by the LiteralInline parser)
return false;
}
// We don't have an emphasis
return false;
}
private bool ProcessLinkReference(InlineParserState state, string label, bool isImage, Inline child = null)
{
bool isValidLink = false;
LinkReferenceDefinitionBlock linkRef;
if (state.Document.LinkReferenceDefinitions.TryGetValue(label, out linkRef))
{
// Inline Link
var link = new LinkInline()
{
Url = HtmlHelper.Unescape(linkRef.Url),
Title = HtmlHelper.Unescape(linkRef.Title),
IsImage = isImage,
};
if (child == null)
{
child = new LiteralInline()
{
Content = label,
IsClosed = true
};
link.AppendChild(child);
}
else
{
// Insert all child into the link
while (child != null)
{
var next = child.NextSibling;
child.Remove();
link.AppendChild(child);
child = next;
}
}
link.IsClosed = true;
EmphasisInline.ProcessEmphasis(link);
state.Inline = link;
isValidLink = true;
}
//else
//{
// // Else output a literal, leave it opened as we may have literals after
// // that could be append to this one
// var literal = new LiteralInline()
// {
// ContentBuilder = state.StringBuilders.Get().Append('[').Append(label).Append(']')
// };
// state.Inline = literal;
//}
return isValidLink;
}
private bool TryProcessLinkOrImage(InlineParserState inlineState, ref StringSlice text)
{
LinkDelimiterInline openParent = null;
foreach (var parent in inlineState.Inline.FindParentOfType<LinkDelimiterInline>())
{
openParent = parent;
break;
}
// This will be matched as a literal
if (openParent != null)
{
var parentDelimiter = openParent.Parent;
switch (text.CurrentChar)
{
case '(':
string url;
string title;
if (LinkHelper.TryParseInlineLink(ref text, out url, out title))
{
// Inline Link
var link = new LinkInline()
{
Url = HtmlHelper.Unescape(url),
Title = HtmlHelper.Unescape(title),
IsImage = openParent.IsImage,
};
openParent.ReplaceBy(link);
inlineState.Inline = link;
EmphasisInline.ProcessEmphasis(link);
ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter);
link.IsClosed = true;
return true;
}
break;
default:
string label = null;
// Handle Collapsed links
if (text.CurrentChar == '[')
{
if (text.PeekChar(1) == ']')
{
label = openParent.Label;
text.NextChar(); // Skip [
text.NextChar(); // Skip ]
}
}
else
{
label = openParent.Label;
}
if (label != null || LinkHelper.TryParseLabel(ref text, true, out label))
{
if (ProcessLinkReference(inlineState, label, openParent.IsImage,
openParent.FirstChild))
{
// Remove the open parent
openParent.Remove();
ReplaceParentIfNotImage(openParent.IsImage, parentDelimiter);
}
else
{
return false;
}
return true;
}
break;
}
// We have a nested [ ]
// firstParent.Remove();
// The opening [ will be transformed to a literal followed by all the childrens of the [
var literal = new LiteralInline()
{
ContentBuilder = inlineState.StringBuilders.Get().Append(openParent.IsImage ? "![" : "[")
};
inlineState.InlinesToClose.Add(literal);
inlineState.Inline = openParent.ReplaceBy(literal);
return false;
}
return false;
}
private void ReplaceParentIfNotImage(bool isImage, Inline inline)
{
if (isImage || inline == null)
{
return;
}
foreach (var parent in inline.FindParentOfType<LinkDelimiterInline>())
{
if (parent.IsImage)
{
break;
}
var literal = new LiteralInline()
{
Content = "[",
IsClosed = true
};
parent.ReplaceBy(literal);
}
}
private bool TryParseLinkTitle(InlineParserState state)
{
return false;
}
}
}
}

View File

@@ -1,12 +1,11 @@
using System.Text;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class LiteralInline : LeafInline
{
public static readonly InlineParser Parser = new ParserInternal();
public LiteralInline()
{
IsClosable = true;
@@ -23,31 +22,6 @@ namespace Textamina.Markdig.Syntax
ContentBuilder = null;
}
private class ParserInternal : InlineParser
{
public override bool Match(InlineParserState state)
{
// A literal will always match
var literal = state.Inline as LiteralInline;
StringBuilder builder;
if (literal == null)
{
builder = state.StringBuilders.Get();
literal = new LiteralInline {ContentBuilder = builder};
state.Inline = literal;
}
else
{
builder = literal.ContentBuilder;
}
var text = state.Text;
builder.Append(text.CurrentChar);
text.NextChar();
return true;
}
}
public override string ToString()
{
return Content ?? (ContentBuilder != null ? ContentBuilder.ToString() : string.Empty);

View File

@@ -1,7 +1,6 @@
namespace Textamina.Markdig.Syntax
namespace Textamina.Markdig.Syntax.Inlines
{
public class RawHtmlInline : LeafInline
{
}
}

View File

@@ -1,18 +1,30 @@
using Textamina.Markdig.Parsing;
using System.Diagnostics;
using Textamina.Markdig.Parsers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Syntax
{
[DebuggerDisplay("{GetType().Name} Line: {Line}, {Lines}")]
public abstract class LeafBlock : Block
{
protected LeafBlock(BlockParser parser) : base(parser)
{
Lines = new StringSliceList();
ProcessInlines = false;
}
public StringSliceList Lines { get; set; }
public Inline Inline { get; set; }
public bool NoInline { get; set; }
public bool ProcessInlines { get; set; }
public void AppendLine(ref StringSlice line)
{
if (Lines == null)
{
Lines = new StringSliceList();
}
Lines.Append(ref line);
}
}
}

View File

@@ -1,5 +1,4 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
namespace Textamina.Markdig.Syntax
{
@@ -23,7 +22,7 @@ namespace Textamina.Markdig.Syntax
public string Title { get; set; }
public static bool TryParse(ref StringSlice text, out LinkReferenceDefinitionBlock block)
public static bool TryParse<T>(ref ICharIterator text, out LinkReferenceDefinitionBlock block) where T : ICharIterator
{
block = null;
string label;

View File

@@ -1,15 +1,9 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
public class ListBlock : ContainerBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public ListBlock(BlockParser parser) : base(parser)
{
}
@@ -24,359 +18,8 @@ namespace Textamina.Markdig.Syntax
public bool IsLoose { get; set; }
private int CountAllBlankLines { get; set; }
internal int CountAllBlankLines { get; set; }
private int CountBlankLinesReset { get; set; }
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
if (state.Pending is ListBlock && state.NextPending is ListItemBlock)
{
// We try to match only on item block if the ListBlock
return MatchLineResult.Skip;
}
// When both a thematic break and a list item are possible
// interpretations of a line, the thematic break takes precedence
var save = state.Line;
if (ThematicBreakBlock.Parser.Match(state) == MatchLineResult.Last)
{
// Remove the ThematicBreakBlock as we will let the ThematicBreakBlock to catch it later
state.NewBlocks.Pop();
return MatchLineResult.None;
}
state.SetCurrentLine(ref save);
// 5.2 List items
// TODO: Check with specs, it is not clear that list marker or bullet marker must be followed by at least 1 space
int preIndent = 0;
for (int i = state.Line.Start - 1; i >= 0; i--)
{
if (state.Line[i].IsSpaceOrTab())
{
preIndent++;
}
else
{
break;
}
}
var saveLiner = state.Line;
// If we have already a ListItemBlock, we are going to try to append to it
var listItem = state.Pending as ListItemBlock;
if (listItem != null)
{
var list = (ListBlock) listItem.Parent;
// Allow all blanks lines if the last block is a fenced code block
// Allow 1 blank line inside a list
// If > 1 blank line, terminate this list
var isBlankLine = state.IsBlankLine;
//if (isBlankLine && !(state.LastBlock is FencedCodeBlock)) // TODO: Handle this case
var isInFencedBlock = state.LastBlock is FencedCodeBlock;
if (isBlankLine)
{
// TODO: Check with a generic way (allow a block to have multiple empty lines)
if (!isInFencedBlock)
{
if (!(state.NextPending is ListBlock))
{
list.CountAllBlankLines++;
listItem.Children.Add(BlankLineBlock.Instance);
}
list.CountBlankLinesReset++;
}
if (list.CountBlankLinesReset > 1)
{
// TODO: Close all lists and not only this one
return MatchLineResult.LastDiscard;
}
if (list.CountBlankLinesReset == 1 && listItem.NumberOfSpaces < 0)
{
state.Close(listItem);
// Leave the list open
list.IsOpen = true;
return MatchLineResult.Continue;
}
return MatchLineResult.Continue;
}
list.CountBlankLinesReset = 0;
var c = state.Line.CurrentChar;
var startPosition = state.Line.Start;
// List Item starting with a blank line (-1)
if (listItem.NumberOfSpaces < 0)
{
int expectedCount = -listItem.NumberOfSpaces;
int countSpaces = 0;
var saved = new StringSlice();
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
countSpaces = preIndent + state.Line.Column - startPosition;
if (countSpaces == expectedCount)
{
saved = state.Line;
}
else if (countSpaces >= 4)
{
state.SetCurrentLine(ref saved);
countSpaces = expectedCount;
break;
}
}
if (countSpaces == expectedCount)
{
listItem.NumberOfSpaces = countSpaces;
return MatchLineResult.Continue;
}
}
else
{
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
var countSpaces = preIndent + state.Line.Column - startPosition;
if (countSpaces >= listItem.NumberOfSpaces)
{
return MatchLineResult.Continue;
}
}
}
state.SetCurrentLine(ref saveLiner);
}
return TryParseListItem(ref state, preIndent);
}
private MatchLineResult TryParseListItem(ref BlockParserState state, int preIndent)
{
var isInList = state.Pending is ListItemBlock;
var preStartPosition = state.Line.Start;
var c = state.Line.CurrentChar;
if (isInList)
{
while (c.IsSpaceOrTab())
{
c = state.Line.NextChar();
}
}
else
{
// TODO
//state.Line.SkipLeadingSpaces3();
c = state.Line.CurrentChar;
}
preIndent = preIndent + state.Line.Start - preStartPosition;
var isOrdered = false;
var bulletChar = (char) 0;
int orderedStart = 0;
var orderedDelimiter = (char) 0;
var column = state.Line.Start;
if (c.IsBulletListMarker())
{
bulletChar = c;
preIndent++;
}
else if (c.IsDigit())
{
int countDigit = 0;
while (c.IsDigit())
{
orderedStart = orderedStart*10 + c - '0';
c = state.Line.NextChar();
preIndent++;
countDigit++;
}
// Note that ordered list start numbers must be nine digits or less:
if (countDigit > 9)
{
return MatchLineResult.None;
}
// We don't have an ordered list
if (c != '.' && c != ')')
{
return MatchLineResult.None;
}
preIndent++;
isOrdered = true;
orderedDelimiter = c;
}
else
{
return MatchLineResult.None;
}
// Skip Bullet or '.'
state.Line.NextChar();
// Item starting with a blank line
int numberOfSpaces;
if (state.IsBlankLine)
{
// Use a negative number to store the number of expected chars
numberOfSpaces = -(preIndent + 1);
}
else
{
var startPosition = -1;
int countSpaceAfterBullet = 0;
var saved = new StringSlice();
for (int i = 0; i <= 4; i++)
{
c = state.Line.CurrentChar;
if (!c.IsSpaceOrTab())
{
break;
}
if (i == 0)
{
startPosition = state.Line.Column;
}
var endPosition = state.Line.Column;
countSpaceAfterBullet = endPosition - startPosition;
if (countSpaceAfterBullet == 1)
{
saved = state.Line;
}
else if (countSpaceAfterBullet >= 4)
{
//state.Line.SpaceHeaderCount = countSpaceAfterBullet - 4;
countSpaceAfterBullet = 0;
state.SetCurrentLine(ref saved);
break;
}
state.Line.NextChar();
}
// If we haven't matched any spaces, early exit
if (startPosition < 0)
{
return MatchLineResult.None;
}
// Number of spaces required for the following content to be part of this list item
numberOfSpaces = preIndent + countSpaceAfterBullet + 1;
}
var newListItem = new ListItemBlock(this)
{
Column = column,
NumberOfSpaces = numberOfSpaces
};
state.NewBlocks.Push(newListItem);
var currentListItem = state.Pending as ListItemBlock;
var currentParent = state.Pending as ListBlock ?? (ListBlock)currentListItem?.Parent;
if (currentParent != null)
{
// If we have a new list item, close the previous one
if (currentListItem != null)
{
state.Close(currentListItem);
}
// Reset the list if it is a new list or a new type of bullet
if (currentParent.IsOrdered != isOrdered ||
(isOrdered && currentParent.OrderedDelimiter != orderedDelimiter) ||
(!isOrdered && currentParent.BulletChar != bulletChar)
//(numberOfSpaces < ((ListItemBlock) currentParent.LastChild).NumberOfSpaces)
)
{
state.Close(currentParent);
currentParent = null;
}
}
if (currentParent == null)
{
var newList = new ListBlock(this)
{
Column = column,
IsOrdered = isOrdered,
BulletChar = bulletChar,
OrderedDelimiter = orderedDelimiter,
OrderedStart = orderedStart,
};
state.NewBlocks.Push(newList);
}
return MatchLineResult.Continue;
}
public override void Close(BlockParserState state)
{
var listBlock = state.Pending as ListBlock;
// Process only if we have blank lines
if (listBlock != null && listBlock.CountAllBlankLines > 0)
{
bool isLastListItem = true;
for (int listIndex = listBlock.Children.Count - 1; listIndex >= 0; listIndex--)
{
var block = listBlock.Children[listIndex];
var listItem = (ListItemBlock) block;
var children = listItem.Children;
bool isLastElement = true;
for (int i = children.Count - 1; i >= 0; i--)
{
var item = children[i];
if (item is BlankLineBlock)
{
if ((isLastElement && listIndex < listBlock.Children.Count - 1) || (children.Count > 2 && (i > 0 && i < (children.Count - 1))))
{
listBlock.IsLoose = true;
}
if (isLastElement && isLastListItem)
{
// Inform the outer list that we have a blank line
var parentListItemBlock = listBlock.Parent as ListItemBlock;
if (parentListItemBlock != null)
{
var parentList = (ListBlock) parentListItemBlock.Parent;
parentList.CountAllBlankLines++;
parentListItemBlock.Children.Add(BlankLineBlock.Instance);
}
}
children.RemoveAt(i);
// If we have remove all blank lines, we can exit
listBlock.CountAllBlankLines--;
if (listBlock.CountAllBlankLines == 0)
{
return;
}
}
isLastElement = false;
}
isLastListItem = false;
}
}
}
}
internal int CountBlankLinesReset { get; set; }
}
}

View File

@@ -1,7 +1,7 @@
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{

View File

@@ -1,9 +1,4 @@
using System;
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
@@ -15,192 +10,8 @@ namespace Textamina.Markdig.Syntax
/// </remarks>
public class ParagraphBlock : LeafBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public ParagraphBlock(BlockParser parser) : base(parser)
{
}
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
if (state.IsCodeIndent)
{
return MatchLineResult.None;
}
var column = state.Column;
// Else it is a continue, we don't break on blank lines
var isBlankLine = state.IsBlankLine;
var paragraph = state.Pending as ParagraphBlock;
if (paragraph == null)
{
if (isBlankLine)
{
return MatchLineResult.None;
}
}
// We continue trying to match by default
var result = MatchLineResult.Continue;
if (paragraph == null)
{
state.NewBlocks.Push(new ParagraphBlock(this) { Column = column });
}
else
{
if (isBlankLine)
{
result = MatchLineResult.None;
}
else if (!(paragraph.Parent is QuoteBlock))
{
var headingChar = (char) 0;
bool checkForSpaces = false;
for (int i = state.Start; i <= state.EndOffset; i++)
{
var c = state.Line[i];
if (headingChar == 0)
{
if (c == '=' || c == '-')
{
headingChar = c;
continue;
}
break;
}
if (checkForSpaces)
{
if (!c.IsSpaceOrTab())
{
headingChar = (char) 0;
break;
}
}
else if (c != headingChar)
{
if (c.IsSpaceOrTab())
{
checkForSpaces = true;
}
else
{
headingChar = (char)0;
break;
}
}
}
if (headingChar != 0)
{
// If we matched a LinkReferenceDefinition before matching the heading, and the remaining
// lines are empty, we can early exit and remove the paragraph
if (TryMatchLinkReferenceDefinition(paragraph.Lines, state) && paragraph.Lines.Count == 0)
{
state.Discard(paragraph);
return MatchLineResult.LastDiscard;
}
var level = headingChar == '=' ? 1 : 2;
var heading = new HeadingBlock(this)
{
Column = paragraph.Column,
Level = level,
Lines = paragraph.Lines,
};
heading.Lines.Trim();
// Remove the paragraph as a pending block
state.NewBlocks.Push(heading);
state.Discard(paragraph);
return MatchLineResult.LastDiscard;
}
}
}
return result;
}
private bool TryMatchLinkReferenceDefinition(StringSliceList localLineGroup, BlockParserState state)
{
bool atLeastOneFound = false;
//var saved = new StringSliceList.State();
//while (true)
//{
// // If we have found a LinkReferenceDefinition, we can discard the previous paragraph
// localLineGroup.Save(ref saved);
// LinkReferenceDefinitionBlock linkReferenceDefinition;
// if (LinkReferenceDefinitionBlock.TryParse(localLineGroup, out linkReferenceDefinition))
// {
// if (!state.Root.LinkReferenceDefinitions.ContainsKey(linkReferenceDefinition.Label))
// {
// state.Root.LinkReferenceDefinitions[linkReferenceDefinition.Label] = linkReferenceDefinition;
// }
// atLeastOneFound = true;
// // Remove lines that have been matched
// if (localLineGroup.LinePosition == localLineGroup.Count)
// {
// localLineGroup.Clear();
// }
// else
// {
// for (int i = localLineGroup.LinePosition - 1; i >= 0; i--)
// {
// localLineGroup.RemoveAt(i);
// }
// }
// }
// else
// {
// if (!atLeastOneFound)
// {
// localLineGroup.Restore(ref saved);
// }
// break;
// }
//}
return atLeastOneFound;
}
public override void Close(BlockParserState state)
{
var paragraph = state.Pending as ParagraphBlock;
var heading = state.Pending as HeadingBlock;
if (paragraph != null)
{
var lines = paragraph.Lines;
TryMatchLinkReferenceDefinition(lines, state);
// If Paragraph is empty, we can discard it
if (lines.Count == 0)
{
state.Pending = null;
return;
}
var lineCount = lines.Count;
for (int i = 0; i < lineCount; i++)
{
var line = lines.Slices[i];
line.Trim();
}
}
else if (heading?.Lines.Count > 1)
{
//heading.Lines.RemoveAt(heading.Lines.Count - 1);
}
}
}
}
}

View File

@@ -1,52 +1,13 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
public class QuoteBlock : ContainerBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public QuoteBlock(BlockParser parser) : base(parser)
{
}
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
if (state.IsCodeIndent)
{
return MatchLineResult.None;
}
// 5.1 Block quotes
// A block quote marker consists of 0-3 spaces of initial indent, plus (a) the character > together with a following space, or (b) a single character > not followed by a space.
var c = state.CurrentChar;
var column = state.Column;
if (c != '>')
{
if (state.Pending != null && state.IsBlankLine)
{
state.Close(state.Pending);
return MatchLineResult.None;
}
return MatchLineResult.None;
}
c = state.NextChar();
if (c.IsSpace())
{
state.NextChar();
}
if (state.Pending == null)
{
state.NewBlocks.Push(new QuoteBlock(this) { Column = column });
}
return MatchLineResult.Continue;
}
}
public char QuoteChar { get; set; }
}
}

View File

@@ -4,7 +4,22 @@ using Textamina.Markdig.Helpers;
namespace Textamina.Markdig.Syntax
{
public struct StringSlice
public interface ICharIterator
{
int Start { get; set; }
char CurrentChar { get; }
int End { get; set; }
char NextChar();
bool TrimStart();
bool TrimStart(out int spaceCount);
}
public struct StringSlice : ICharIterator
{
public StringSlice(string text)
{
@@ -16,9 +31,11 @@ namespace Textamina.Markdig.Syntax
public readonly string Text;
public int Start;
public int Start { get; set; }
public int End;
public int End { get; set; }
public int Length => End - Start + 1;
public int Column => Start;
@@ -35,8 +52,9 @@ namespace Textamina.Markdig.Syntax
if (Start > End)
{
Start = End + 1;
return '\0';
}
return CurrentChar;
return Text[Start];
}
[MethodImpl(MethodImplOptionPortable.AggressiveInlining)]
@@ -117,44 +135,41 @@ namespace Textamina.Markdig.Syntax
public bool TrimStart()
{
// Strip leading spaces
var c = CurrentChar;
var hasWhitespaces = false;
while (c.IsWhitespace())
for (; Start <= End; Start++)
{
c = NextChar();
hasWhitespaces = true;
}
return hasWhitespaces;
}
public bool TrimStart(out int newLineCount)
{
bool hasWhitespaces = false;
newLineCount = 0;
var c = CurrentChar;
while (c.IsWhitespace())
{
if (c == '\n')
{
newLineCount++;
}
c = NextChar();
hasWhitespaces = true;
}
return hasWhitespaces;
}
public void TrimEnd(bool includeTabs = false)
{
for (int i = End; i >= Start; i--)
{
End = i;
var c = this[i];
if (!(includeTabs ? c.IsSpaceOrTab() : c.IsSpace()))
if (!Text[Start].IsWhitespace())
{
break;
}
}
return Start > End;
}
public bool TrimStart(out int spaceCount)
{
spaceCount = 0;
// Strip leading spaces
for (; Start <= End; Start++)
{
if (!Text[Start].IsWhitespace())
{
break;
}
spaceCount++;
}
return IsEndOfSlice;
}
public bool TrimEnd()
{
for (; Start <= End; End--)
{
if (!Text[End].IsWhitespace())
{
break;
}
}
return IsEndOfSlice;
}
public void Trim()
@@ -165,7 +180,7 @@ namespace Textamina.Markdig.Syntax
public override string ToString()
{
return Start <= End ? Text.Substring(Start, End - Start + 1) : string.Empty;
return Text != null && Start <= End ? Text.Substring(Start, End - Start + 1) : string.Empty;
}
}
}

View File

@@ -8,8 +8,6 @@ namespace Textamina.Markdig.Syntax
{
private static readonly StringSlice[] Empty = new StringSlice[0];
private StringSlice currentLine;
public StringSliceList()
{
Slices = Empty;
@@ -50,17 +48,17 @@ namespace Textamina.Markdig.Syntax
public override string ToString()
{
var stringBuilder = StringBuilderCache.Local();
var builder = StringBuilderCache.Local();
for(int i = 0; i < Count; i++)
{
if (i > 0)
{
stringBuilder.Append('\n');
builder.Append('\n');
}
stringBuilder.Append(Slices[i]);
builder.Append(Slices[i].Text, Slices[i].Start, Slices[i].Length);
}
var str = stringBuilder.ToString();
stringBuilder.Clear();
var str = builder.ToString();
builder.Clear();
return str;
}
}

View File

@@ -1,5 +1,4 @@
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Parsing;
using Textamina.Markdig.Parsers;
namespace Textamina.Markdig.Syntax
{
@@ -8,83 +7,11 @@ namespace Textamina.Markdig.Syntax
/// </summary>
public class ThematicBreakBlock : LeafBlock
{
public new static readonly BlockParser Parser = new ParserInternal();
public new static readonly BlockParser Parser = new ThematicBreakParser();
public ThematicBreakBlock(BlockParser parser) : base(parser)
{
NoInline = true;
}
private class ParserInternal : BlockParser
{
public override MatchLineResult Match(BlockParserState state)
{
if (state.IsCodeIndent)
{
return MatchLineResult.None;
}
var column = state.Column;
// 4.1 Thematic breaks
// A line consisting of 0-3 spaces of indentation, followed by a sequence of three or more matching -, _, or * characters, each followed optionally by any number of spaces
var c = state.CurrentChar;
int count = 0;
var matchChar = (char)0;
bool hasSpacesSinceLastMatch = false;
bool hasInnerSpaces = false;
int offset = 0;
while (c != '\0')
{
if (count == 0 && (c == '-' || c == '_' || c == '*'))
{
matchChar = c;
count++;
}
else if (c == matchChar)
{
if (hasSpacesSinceLastMatch)
{
hasInnerSpaces = true;
}
count++;
}
else if (!c.IsSpace() || count == 0)
{
return MatchLineResult.None;
}
else if (c.IsSpace())
{
hasSpacesSinceLastMatch = true;
}
offset++;
c = state.PeekChar(offset);
}
// If it as less than 3 chars or it is a setex heading and we are already in a paragraph, let the paragraph handle it
var previousParagraph = state.LastBlock as ParagraphBlock;
var isSetexHeading = previousParagraph != null && matchChar == '-' && !hasInnerSpaces;
if (isSetexHeading)
{
var parent = previousParagraph.Parent;
if (parent is QuoteBlock || (parent is ListItemBlock && previousParagraph.Column != column))
{
isSetexHeading = false;
}
}
if (count < 3 || isSetexHeading)
{
return MatchLineResult.None;
}
state.NewBlocks.Push(new ThematicBreakBlock(this) { Column = column });
return MatchLineResult.LastDiscard;
}
ProcessInlines = false;
}
}
}

View File

@@ -41,39 +41,52 @@
<Compile Include="Formatters\HtmlTextWriter.cs" />
<Compile Include="Helpers\LinkHelper.cs" />
<Compile Include="Helpers\StringBuilderCache.cs" />
<Compile Include="Parsing\InlineParser.cs" />
<Compile Include="Parsing\InlineParserState.cs" />
<Compile Include="Parsing\BlockParserState.cs" />
<Compile Include="Parsing\NewBlockParserState.cs" />
<Compile Include="Parsers\IBlockParser.cs" />
<Compile Include="Parsers\InlineParser.cs" />
<Compile Include="Parsers\InlineParserState.cs" />
<Compile Include="Parsers\BlockParserState.cs" />
<Compile Include="Parsers\ParserList.cs" />
<Compile Include="Syntax\BlankLineBlock.cs" />
<Compile Include="Parsers\CodeBlockParser.cs" />
<Compile Include="Parsers\FencedCodeBlockParser.cs" />
<Compile Include="Parsers\HeadingBlockParser.cs" />
<Compile Include="Parsers\HtmlBlockParser.cs" />
<Compile Include="Parsers\Inlines\AutolineInlineParser.cs" />
<Compile Include="Syntax\Inlines\AutolinkInline.cs" />
<Compile Include="Syntax\Inlines\CodeInline.cs" />
<Compile Include="Parsers\Inlines\CodeInlineParser.cs" />
<Compile Include="Syntax\Inlines\ContainerInline.cs" />
<Compile Include="Syntax\Inlines\DelimiterInline.cs" />
<Compile Include="Syntax\Inlines\DelimiterType.cs" />
<Compile Include="Syntax\Inlines\EmphasisDelimiterInline.cs" />
<Compile Include="Syntax\Inlines\EmphasisInline.cs" />
<Compile Include="Syntax\Inlines\EscapeInline.cs" />
<Compile Include="Parsers\Inlines\EmphasisInlineParser.cs" />
<Compile Include="Parsers\Inlines\EscapeInlineParser.cs" />
<Compile Include="Syntax\Inlines\HardlineBreakInline.cs" />
<Compile Include="Parsers\Inlines\HardlineBreakInlineParser.cs" />
<Compile Include="Syntax\Inlines\HtmlInline.cs" />
<Compile Include="Syntax\Inlines\InlineStack.cs" />
<Compile Include="Syntax\Inlines\LeafInline.cs" />
<Compile Include="Syntax\Inlines\LineBreakInline.cs" />
<Compile Include="Syntax\Inlines\LinkInline.cs" />
<Compile Include="Syntax\Inlines\LinkDelimiterInline.cs" />
<Compile Include="Parsers\Inlines\LinkInlineParser.cs" />
<Compile Include="Syntax\Inlines\LiteralInline.cs" />
<Compile Include="Parsers\Inlines\LiteralInlineParser.cs" />
<Compile Include="Syntax\Inlines\RawHtmlInline.cs" />
<Compile Include="Syntax\Block.cs" />
<Compile Include="Syntax\ContainerBlock.cs" />
<Compile Include="Syntax\HtmlBlock.cs" />
<Compile Include="Syntax\HtmlBlockType.cs" />
<Compile Include="Syntax\LeafBlock.cs" />
<Compile Include="Parsing\BlockParser.cs" />
<Compile Include="Parsing\MarkdownParser.cs" />
<Compile Include="Parsers\BlockParser.cs" />
<Compile Include="Parsers\MarkdownParser.cs" />
<Compile Include="Syntax\LinkReferenceDefinitionBlock.cs" />
<Compile Include="Syntax\ListBlock.cs" />
<Compile Include="Parsers\ListBlockParser.cs" />
<Compile Include="Syntax\ListItemBlock.cs" />
<Compile Include="Parsers\ParagraphBlockParser.cs" />
<Compile Include="Syntax\QuoteBlock.cs" />
<Compile Include="Parsers\QuoteBlockParser.cs" />
<Compile Include="Syntax\StringSliceListExtensions.cs" />
<Compile Include="Syntax\ThematicBreakBlock.cs" />
<Compile Include="Helpers\CharHelper.cs" />
@@ -81,17 +94,19 @@
<Compile Include="Syntax\Document.cs" />
<Compile Include="Syntax\FencedCodeBlock.cs" />
<Compile Include="Syntax\HeadingBlock.cs" />
<Compile Include="Parsing\MatchLineResult.cs" />
<Compile Include="Parsers\BlockState.cs" />
<Compile Include="MethodImplOptionPortable.cs" />
<Compile Include="Syntax\Inlines\Inline.cs" />
<Compile Include="Syntax\ParagraphBlock.cs" />
<Compile Include="Properties\AssemblyInfo.cs" />
<Compile Include="Syntax\StringSlice.cs" />
<Compile Include="Syntax\StringSliceList.cs" />
<Compile Include="Parsers\ThematicBreakParser.cs" />
</ItemGroup>
<ItemGroup>
<None Include="project.json" />
</ItemGroup>
<ItemGroup />
<Import Project="$(MSBuildExtensionsPath32)\Microsoft\Portable\$(TargetFrameworkVersion)\Microsoft.Portable.CSharp.targets" />
<!-- To modify your build process, add your task inside one of the targets below and uncomment it.
Other similar extension points exist, see Microsoft.Common.targets.

View File

@@ -1,2 +1,2 @@
<wpf:ResourceDictionary xml:space="preserve" xmlns:x="http://schemas.microsoft.com/winfx/2006/xaml" xmlns:s="clr-namespace:System;assembly=mscorlib" xmlns:ss="urn:shemas-jetbrains-com:settings-storage-xaml" xmlns:wpf="http://schemas.microsoft.com/winfx/2006/xaml/presentation">
<s:Boolean x:Key="/Default/CodeInspection/NamespaceProvider/NamespaceFoldersToSkip/=syntax_005Cinlines/@EntryIndexedValue">True</s:Boolean></wpf:ResourceDictionary>
<s:Boolean x:Key="/Default/CodeInspection/NamespaceProvider/NamespaceFoldersToSkip/=syntax_005Cinlines/@EntryIndexedValue">False</s:Boolean></wpf:ResourceDictionary>