Finish support for Emoji. Add TextMatcher

This commit is contained in:
Alexandre Mutel
2016-03-08 22:27:17 +09:00
parent 96e88251f3
commit 3e7bb6711e
11 changed files with 1322 additions and 1029 deletions

View File

@@ -8,6 +8,7 @@ using System.Linq;
using System.Text;
using System.Text.RegularExpressions;
using BenchmarkDotNet.Attributes;
using Textamina.Markdig.Helpers;
namespace Testamina.Markdig.Benchmarks
{
@@ -51,57 +52,6 @@ namespace Testamina.Markdig.Benchmarks
}
}
public class TextMatchHelper
{
private readonly string[] ordered;
public TextMatchHelper(HashSet<string> strings)
{
var orderedList = new List<string>(strings);
orderedList.Sort();
ordered = orderedList.ToArray();
}
public bool TryMatch(string text, int offset, int length, out string matchText)
{
matchText = null;
int start = 0;
int end = ordered.Length - 1;
while (start <= end)
{
int i = start + (end - start >> 1);
int num3 = Compare(text, offset, length, ordered[i]);
if (num3 == 0)
{
matchText = ordered[i];
return true;
}
if (num3 < 0)
start = i + 1;
else
end = i - 1;
}
return false;
}
private static int Compare(string text, int offset, int length, string value)
{
var maxLength = value.Length < length
? value.Length
: length;
for (int i = 0; i < maxLength; i++, offset++)
{
var result = value[i].CompareTo(text[offset]);
if (result != 0)
{
return result;
}
}
// Either we have a full match, or value string is longer than text
return maxLength == value.Length ? 0 : 1;
}
}
public class TestMatchPerf
{
@@ -119,7 +69,6 @@ namespace Testamina.Markdig.Benchmarks
matcher = new TextMatchHelper(new HashSet<string>(replacers.Keys));
}
[Benchmark]
public void TestMatch()
{
@@ -128,7 +77,7 @@ namespace Testamina.Markdig.Benchmarks
{
string matchText;
//var text = ":z150: this is a long string";
var text = ":z1";
var text = ":z1:";
matcher.TryMatch(text, 0, text.Length, out matchText);
}
}

View File

@@ -12,61 +12,68 @@ namespace Textamina.Markdig.Tests
[Test]
public void TestSimple()
{
var text = @"
Term 1
: This is a definition item
With a paragraph
> This is a block quote
- This is a list
- item2
```java
Test
```
And a lazy line
: This ia another definition item
Term2
Term3 *with some inline*
: This is another definition for term2
";
var text = @" This is a test with a :) and a :angry: smiley";
// var reader = new StringReader(@"> > toto tata
//> titi toto
//");
//var result = Markdown.ToHtml(text, new MarkdownPipeline().UseFootnotes().UseStrikethroughSuperAndSubScript());
var result = Markdown.ToHtml(text, new MarkdownPipeline().UseDefinitionList());
var result = Markdown.ToHtml(text, new MarkdownPipeline().UseEmojiAndSmiley());
//File.WriteAllText("test.html", result, Encoding.UTF8);
Console.WriteLine(result);
}
// Test for definition lists:
//
// var text = @"
//+-----------------------------------+--------------------------------------+
//| - this is a list | > We have a blockquote
//| - this is a second item |
//| |
//| ``` |
//| Yes |
//| ``` |
//+===================================+======================================+
//| This is a second line |
//+-----------------------------------+--------------------------------------+
//Term 1
//: This is a definition item
// With a paragraph
// > This is a block quote
//:::spoiler {#yessss}
//This is a spoiler
//:::
// - This is a list
// - item2
///| we have mult | paragraph |
///| we have a new colspan with a long line
///| and lots of text
// ```java
// Test
// ```
// And a lazy line
//: This ia another definition item
//Term2
//Term3 *with some inline*
//: This is another definition for term2
//";
// Test for grid table
// var text = @"
//+-----------------------------------+--------------------------------------+
//| - this is a list | > We have a blockquote
//| - this is a second item |
//| |
//| ``` |
//| Yes |
//| ``` |
//+===================================+======================================+
//| This is a second line |
//+-----------------------------------+--------------------------------------+
//:::spoiler {#yessss}
//This is a spoiler
//:::
///| we have mult | paragraph |
///| we have a new colspan with a long line
///| and lots of text
//";
}
}

View File

@@ -0,0 +1,22 @@
// Copyright (c) Alexandre Mutel. All rights reserved.
// This file is licensed under the BSD-Clause 2 license.
// See the license.txt file in the project root for more information.
namespace Textamina.Markdig.Extensions.Emoji
{
/// <summary>
/// Extension to allow emoji and smiley replacement.
/// </summary>
/// <seealso cref="Textamina.Markdig.IMarkdownExtension" />
public class EmojiExtension : IMarkdownExtension
{
public void Setup(MarkdownPipeline pipeline)
{
if (!pipeline.InlineParsers.Contains<EmojiParser>())
{
// Insert the parser before any other parsers
pipeline.InlineParsers.Insert(0, new EmojiParser());
}
}
}
}

View File

@@ -0,0 +1,37 @@
// Copyright (c) Alexandre Mutel. All rights reserved.
// This file is licensed under the BSD-Clause 2 license.
// See the license.txt file in the project root for more information.
using Textamina.Markdig.Helpers;
using Textamina.Markdig.Syntax.Inlines;
namespace Textamina.Markdig.Extensions.Emoji
{
/// <summary>
/// An emoji inline
/// </summary>
/// <seealso cref="Textamina.Markdig.Syntax.Inlines.Inline" />
public class EmojiInline : LiteralInline
{
/// <summary>
/// Initializes a new instance of the <see cref="EmojiInline"/> class.
/// </summary>
public EmojiInline()
{
}
/// <summary>
/// Initializes a new instance of the <see cref="EmojiInline"/> class.
/// </summary>
/// <param name="content">The content.</param>
public EmojiInline(string content)
{
Content = new StringSlice(content);
}
/// <summary>
/// Gets or sets the original match string (either an emoji or a text smiley)
/// </summary>
public string Match { get; set; }
}
}

File diff suppressed because it is too large Load Diff

View File

@@ -8,7 +8,7 @@ namespace Textamina.Markdig.Helpers
/// </summary>
/// <typeparam name="T">The type of item to cache</typeparam>
/// <seealso cref="Textamina.Markdig.Helpers.ObjectCache{T}" />
public class DefaultObjectCache<T> : ObjectCache<T> where T : class, new()
public abstract class DefaultObjectCache<T> : ObjectCache<T> where T : class, new()
{
protected override T NewInstance()
{

View File

@@ -22,6 +22,17 @@ namespace Textamina.Markdig.Helpers
builders = new Stack<T>();
}
/// <summary>
/// Clears this cache.
/// </summary>
public void Clear()
{
lock (builders)
{
builders.Clear();
}
}
/// <summary>
/// Gets a new instance.
/// </summary>
@@ -64,8 +75,6 @@ namespace Textamina.Markdig.Helpers
/// Resets the specified instance when <see cref="Release"/> is called before storing back to this cache.
/// </summary>
/// <param name="instance">The instance.</param>
protected virtual void Reset(T instance)
{
}
protected abstract void Reset(T instance);
}
}

View File

@@ -0,0 +1,175 @@
// Copyright (c) Alexandre Mutel. All rights reserved.
// This file is licensed under the BSD-Clause 2 license.
// See the license.txt file in the project root for more information.
using System;
using System.Collections.Generic;
namespace Textamina.Markdig.Helpers
{
/// <summary>
/// Match a text against a list of ASCII string using internally a tree to speedup the lookup
/// </summary>
public class TextMatchHelper
{
private readonly CharNode root;
private readonly ListCache listCache;
private readonly DictionaryCache dictCache;
/// <summary>
/// Initializes a new instance of the <see cref="TextMatchHelper"/> class.
/// </summary>
/// <param name="matches">The matches to match against.</param>
/// <exception cref="System.ArgumentNullException"></exception>
public TextMatchHelper(HashSet<string> matches)
{
if (matches == null) throw new ArgumentNullException(nameof(matches));
var list = new List<string>(matches);
root = new CharNode();
dictCache = new DictionaryCache();
listCache = new ListCache();
BuildMap(ref root, 0, list);
listCache.Clear();
dictCache.Clear();
}
/// <summary>
/// Tries to match in the text, at offset position, the list of string matches registered to this instance.
/// </summary>
/// <param name="text">The text.</param>
/// <param name="offset">The offset.</param>
/// <param name="length">The length.</param>
/// <param name="match">The match string if the match was successfull.</param>
/// <returns>
/// <c>true</c> if the match was successfull; <c>false</c> otherwise
/// </returns>
/// <exception cref="System.ArgumentNullException"></exception>
public bool TryMatch(string text, int offset, int length, out string match)
{
if (text == null) throw new ArgumentNullException(nameof(text));
// TODO(lazy): we should check offset and length for a better exception experience in case of wrong usage
var node = root;
match = null;
while (length > 0)
{
var c = text[offset];
var nextIndex = c - node.MinChar;
if (nextIndex < 0)
{
return false;
}
var nextNodes = node.NextNodes;
if (nextNodes == null || nextIndex >= nextNodes.Length)
{
return false;
}
node = nextNodes[nextIndex];
if (node == null)
{
return false;
}
if (node.Content != null)
{
match = node.Content;
return true;
}
offset++;
length--;
}
return false;
}
private void BuildMap(ref CharNode node, int index, List<string> list)
{
// TODO(lazy): This code for building the nodes is not very efficient in terms of memory usage and could be optimized (using structs and indices)
// At least, we are using a cache for the temporary objects build (List<string> and Dictionary<char, CharNode>)
var charSet = dictCache.Get();
int minChar = int.MaxValue;
int maxChar = 0;
for (int i = 0; i < list.Count; i++)
{
var str = list[i];
var c = str[index];
// Make sure that we don't get something to match that is too large
if (c > 127)
{
throw new InvalidOperationException($"The string [{str}] contains a non ASCII character `{c}`");
}
CharNode nextNode;
if (!charSet.TryGetValue(c, out nextNode))
{
nextNode = new CharNode();
charSet.Add(c, nextNode);
}
// We have found a string for this node
if (index + 1 == str.Length)
{
nextNode.Content = str;
}
else
{
if (nextNode.NextList == null)
{
nextNode.NextList = listCache.Get();
}
nextNode.NextList.Add(str);
}
if (c < minChar)
{
minChar = c;
}
if (c > maxChar)
{
maxChar = c;
}
}
node.MinChar = minChar;
var chars = new CharNode[maxChar - minChar + 1];
node.NextNodes = chars;
foreach (var charList in charSet)
{
var nodeIndex = charList.Key - minChar;
chars[nodeIndex] = charList.Value;
if (charList.Value.NextList != null)
{
BuildMap(ref chars[charList.Key - minChar], index + 1, charList.Value.NextList);
listCache.Release(charList.Value.NextList);
charList.Value.NextList = null;
}
}
dictCache.Release(charSet);
}
private class ListCache : DefaultObjectCache<List<string>>
{
protected override void Reset(List<string> instance)
{
instance.Clear();
}
}
private class DictionaryCache : DefaultObjectCache<Dictionary<char, CharNode>>
{
protected override void Reset(Dictionary<char, CharNode> instance)
{
instance.Clear();
}
}
private class CharNode
{
public CharNode[] NextNodes;
public int MinChar;
public List<string> NextList;
public string Content { get; set; }
}
}
}

View File

@@ -5,6 +5,7 @@ using Textamina.Markdig.Extensions;
using Textamina.Markdig.Extensions.Attributes;
using Textamina.Markdig.Extensions.CustomContainers;
using Textamina.Markdig.Extensions.DefinitionLists;
using Textamina.Markdig.Extensions.Emoji;
using Textamina.Markdig.Extensions.Footnotes;
using Textamina.Markdig.Extensions.Tables;
@@ -29,8 +30,9 @@ namespace Textamina.Markdig
.UsePipeTable()
.UseSoftlineBreakAsHardlineBreak()
.UseFootnotes()
.UseEmojiAndSmiley()
.UseStrikethroughSuperAndSubScript()
.UseAttributes();
.UseAttributes(); // Must be last as it is one parser that is modifying other parsers
}
/// <summary>
@@ -120,5 +122,16 @@ namespace Textamina.Markdig
pipeline.Extensions.AddIfNotAlready<AttributesExtension>();
return pipeline;
}
/// <summary>
/// Uses the emoji and smiley extension.
/// </summary>
/// <param name="pipeline">The pipeline.</param>
/// <returns>The modified pipeline</returns>
public static MarkdownPipeline UseEmojiAndSmiley(this MarkdownPipeline pipeline)
{
pipeline.Extensions.AddIfNotAlready<EmojiExtension>();
return pipeline;
}
}
}

View File

@@ -125,9 +125,19 @@ namespace Textamina.Markdig.Parsers
}
}
private class ContainerItemCache : DefaultObjectCache<ContainerItem>
{
protected override void Reset(ContainerItem instance)
{
instance.Container = null;
instance.Index = 0;
}
}
private void ProcessInlines()
{
var cache = new DefaultObjectCache<ContainerItem>();
var cache = new ContainerItemCache();
var blocks = new Stack<ContainerItem>();
// TODO: Use an ObjectCache for ContainerItem

View File

@@ -44,7 +44,9 @@
<Compile Include="Extensions\DefinitionLists\DefinitionList.cs" />
<Compile Include="Extensions\DefinitionLists\DefinitionListParser.cs" />
<Compile Include="Extensions\DefinitionLists\DefinitionTerm.cs" />
<Compile Include="Extensions\Emoji\EmojiExtension.cs" />
<Compile Include="Extensions\DefinitionLists\HtmlDefinitionListRenderer.cs" />
<Compile Include="Extensions\Emoji\EmojiInline.cs" />
<Compile Include="Extensions\Emoji\EmojiParser.cs" />
<Compile Include="Extensions\Footnotes\Footnote.cs" />
<Compile Include="Extensions\Footnotes\FootnoteParser.cs" />
@@ -73,6 +75,7 @@
<Compile Include="Helpers\DefaultObjectCache.cs" />
<Compile Include="Helpers\ObjectCache.cs" />
<Compile Include="Helpers\OrderedList.cs" />
<Compile Include="Helpers\TextMatcher.cs" />
<Compile Include="IMarkdownExtension.cs" />
<Compile Include="Markdown.cs" />
<Compile Include="MarkdownExtensions.cs" />