diff --git a/src/Textamina.Markdig/Formatters/HtmlFormatter.cs b/src/Textamina.Markdig/Formatters/HtmlFormatter.cs index 110075e5..56cdaa63 100644 --- a/src/Textamina.Markdig/Formatters/HtmlFormatter.cs +++ b/src/Textamina.Markdig/Formatters/HtmlFormatter.cs @@ -2,6 +2,7 @@ using System.Collections.Generic; using System.Globalization; using System.IO; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Syntax; namespace Textamina.Markdig.Formatters diff --git a/src/Textamina.Markdig/Formatters/HtmlTextWriter.cs b/src/Textamina.Markdig/Formatters/HtmlTextWriter.cs index 26c198c4..a2ade201 100644 --- a/src/Textamina.Markdig/Formatters/HtmlTextWriter.cs +++ b/src/Textamina.Markdig/Formatters/HtmlTextWriter.cs @@ -1,4 +1,5 @@ using System.IO; +using Textamina.Markdig.Helpers; namespace Textamina.Markdig.Formatters { diff --git a/src/Textamina.Markdig/Formatters/Scanner.cs b/src/Textamina.Markdig/Formatters/Scanner.cs deleted file mode 100644 index df1f3cfe..00000000 --- a/src/Textamina.Markdig/Formatters/Scanner.cs +++ /dev/null @@ -1,461 +0,0 @@ -using System; -using Textamina.Markdig.Parsing; - -namespace Textamina.Markdig.Formatters -{ - - /// - /// Contains the regular expressions that are used in the parsers. - /// - internal static partial class Scanner - { - /// - /// List of valid schemes of an URL. The array must be sorted. - /// - private static readonly string[] schemeArray = new[] - { - "AAA", "AAAS", "ABOUT", "ACAP", "ADIUMXTRA", "AFP", "AFS", "AIM", "APT", "ATTACHMENT", "AW", "BESHARE", - "BITCOIN", "BOLO", "CALLTO", "CAP", "CHROME", "CHROME-EXTENSION", "CID", "COAP", "COM-EVENTBRITE-ATTENDEE", - "CONTENT", "CRID", "CVS", "DATA", "DAV", "DICT", "DLNA-PLAYCONTAINER", "DLNA-PLAYSINGLE", "DNS", "DOI", - "DTN", "DVB", "ED2K", "FACETIME", "FEED", "FILE", "FINGER", "FISH", "FTP", "GEO", "GG", "GIT", - "GIZMOPROJECT", "GO", "GOPHER", "GTALK", "H323", "HCP", "HTTP", "HTTPS", "IAX", "ICAP", "ICON", "IM", "IMAP", - "INFO", "IPN", "IPP", "IRC", "IRC6", "IRCS", "IRIS", "IRIS.BEEP", "IRIS.LWZ", "IRIS.XPC", "IRIS.XPCS", - "ITMS", "JAR", "JAVASCRIPT", "JMS", "KEYPARC", "LASTFM", "LDAP", "LDAPS", "MAGNET", "MAILTO", "MAPS", - "MARKET", "MESSAGE", "MID", "MMS", "MS-HELP", "MSNIM", "MSRP", "MSRPS", "MTQP", "MUMBLE", "MUPDATE", "MVN", - "NEWS", "NFS", "NI", "NIH", "NNTP", "NOTES", "OID", "OPAQUELOCKTOKEN", "PALM", "PAPARAZZI", "PLATFORM", - "POP", "PRES", "PROXY", "PSYC", "QUERY", "RES", "RESOURCE", "RMI", "RSYNC", "RTMP", "RTSP", "SECONDLIFE", - "SERVICE", "SESSION", "SFTP", "SGN", "SHTTP", "SIEVE", "SIP", "SIPS", "SKYPE", "SMB", "SMS", "SNMP", - "SOAP.BEEP", "SOAP.BEEPS", "SOLDAT", "SPOTIFY", "SSH", "STEAM", "SVN", "TAG", "TEAMSPEAK", "TEL", "TELNET", - "TFTP", "THINGS", "THISMESSAGE", "TIP", "TN3270", "TV", "UDP", "UNREAL", "URN", "UT2004", "VEMMI", - "VENTRILO", "VIEW-SOURCE", "WEBCAL", "WS", "WSS", "WTAI", "WYCIWYG", "XCON", "XCON-USERID", "XFIRE", - "XMLRPC.BEEP", "XMLRPC.BEEPS", "XMPP", "XRI", "YMSGR", "Z39.50R", "Z39.50S" - }; - - /// - /// Try to match URI autolink after first <, returning number of chars matched. - /// - public static int scan_autolink_uri(string s, int pos, int sourceLength) - { - /*!re2c - scheme [:]([^\x00-\x20<>\\]|escaped_char)*[>] { return (p - start); } - .? { return 0; } - */ - // for now the tests do not include anything that would require the use of `escaped_char` part so it is ignored. - - // 24 is the maximum length of a valid scheme - var checkLen = sourceLength - pos; - if (checkLen > 24) - checkLen = 24; - - // PERF: potential small improvement - instead of using IndexOf, check char-by-char and return as soon as an invalid character is found ([^a-z0-9\.]) - // alternative approach (if we want to go crazy about performance - store the valid schemes as a prefix tree and lookup the valid scheme char by char and - // return as soon as the part does not match any prefix. - var colonpos = s.IndexOf(':', pos, checkLen); - if (colonpos == -1) - return 0; - - var potentialScheme = s.Substring(pos, colonpos - pos).ToUpperInvariant(); - if (Array.BinarySearch(schemeArray, potentialScheme, StringComparer.Ordinal) < 0) - return 0; - - for (var i = colonpos + 1; i < sourceLength; i++) - { - var c = s[i]; - if (c == '>') - return i - pos + 1; - - if (c == '<' || c <= 0x20) - return 0; - } - - return 0; - } - - /// - /// Try to match a link title (in single quotes, in double quotes, or - /// in parentheses), returning number of chars matched. Allow one - /// level of internal nesting (quotes within quotes). - /// - public static int scan_link_title(string s, int pos, int sourceLength) - { - /*!re2c - ["] (escaped_char|[^"\x00])* ["] { return (p - start); } - ['] (escaped_char|[^'\x00])* ['] { return (p - start); } - [(] (escaped_char|[^)\x00])* [)] { return (p - start); } - .? { return 0; } - */ - - if (pos + 2 >= sourceLength) - return 0; - - var c1 = s[pos]; - if (c1 != '"' && c1 != '\'' && c1 != '(') - return 0; - - if (c1 == '(') c1 = ')'; - - var nextEscaped = false; - for (var i = pos + 1; i < sourceLength; i++) - { - var c = s[i]; - if (c == c1 && !nextEscaped) - return i - pos + 1; - - nextEscaped = !nextEscaped && c == '\\'; - } - - return 0; - } - - /// - /// Match space characters, including newlines. - /// - public static int scan_spacechars(string s, int pos, int sourceLength) - { - /*!re2c - [ \t\n]* { return (p - start); } - . { return 0; } - */ - if (pos >= sourceLength) - return 0; - - for (var i = pos; i < sourceLength; i++) - { - if (!Utility.IsWhitespace(s[i])) - return i - pos; - } - - return sourceLength - pos; - } - - /// - /// Match ATX heading start. - /// - public static int scan_atx_heading_start(string s, int pos, int sourceLength, out int headingLevel) - { - /*!re2c - [#]{1,6} ([ ]+|[\n]) { return (p - start); } - .? { return 0; } - */ - - headingLevel = 1; - if (pos + 1 >= sourceLength) - return 0; - - if (s[pos] != '#') - return 0; - - var spaceExists = false; - for (var i = pos + 1; i < sourceLength; i++) - { - var c = s[i]; - - if (c == '#') - { - if (headingLevel == 6) - return 0; - - if (spaceExists) - return i - pos; - else - headingLevel++; - } - else if (c == ' ') - { - spaceExists = true; - } - else if (c == '\n') - { - return i - pos + 1; - } - else - { - return spaceExists ? i - pos : 0; - } - } - - if (spaceExists) - return sourceLength - pos; - - return 0; - } - - /// - /// Match sexext heading line. Return 1 for level-1 heading, - /// 2 for level-2, 0 for no match. - /// - public static int scan_setext_heading_line(string s, int pos, int sourceLength) - { - /*!re2c - [=]+ [ ]* [\n] { return 1; } - [-]+ [ ]* [\n] { return 2; } - .? { return 0; } - */ - - if (pos >= sourceLength) - return 0; - - var c1 = s[pos]; - - if (c1 != '=' && c1 != '-') - return 0; - - var fin = false; - for (var i = pos + 1; i < sourceLength; i++) - { - var c = s[i]; - if (c == c1 && !fin) - continue; - - fin = true; - if (c == ' ') - continue; - - if (c == '\n') - break; - - return 0; - } - - return c1 == '=' ? 1 : 2; - } - - /// - /// Scan a thematic break line: "...three or more hyphens, asterisks, - /// or underscores on a line by themselves. If you wish, you may use - /// spaces between the hyphens or asterisks." - /// - public static int scan_thematic_break(string s, int pos, int sourceLength) - { - // @"^([\*][ ]*){3,}[\s]*$", - // @"^([_][ ]*){3,}[\s]*$", - // @"^([-][ ]*){3,}[\s]*$", - - var count = 0; - var x = '\0'; - var ipos = pos; - while (ipos < sourceLength) - { - var c = s[ipos++]; - - if (c == ' ' || c == '\n') - continue; - if (count == 0) - { - if (c == '*' || c == '_' || c == '-') - x = c; - else - return 0; - - count = 1; - } - else if (c == x) - count++; - else - return 0; - } - - if (count < 3) - return 0; - - return sourceLength - pos; - } - - /// - /// Scan an opening code fence. Returns the number of characters forming the fence. - /// - public static int scan_open_code_fence(string s, int pos, int sourceLength) - { - /*!re2c - [`]{3,} / [^`\n\x00]*[\n] { return (p - start); } - [~]{3,} / [^~\n\x00]*[\n] { return (p - start); } - .? { return 0; } - */ - - if (pos + 3 >= sourceLength) - return 0; - - var fchar = s[pos]; - if (fchar != '`' && fchar != '~') - return 0; - - var cnt = 1; - var fenceDone = false; - for (var i = pos + 1; i < sourceLength; i++) - { - var c = s[i]; - - if (c == fchar) - { - if (fenceDone) - return 0; - - cnt++; - continue; - } - - fenceDone = true; - if (cnt < 3) - return 0; - - if (c == '\n') - return cnt; - } - - if (cnt < 3) - return 0; - - return cnt; - } - - /// - /// Scan a closing code fence with length at least len. - /// - public static int scan_close_code_fence(string s, int pos, int len, int sourceLength) - { - /*!re2c - ([`]{3,} | [~]{3,}) / spacechar* [\n] - { if (p - start > len) { - return (p - start); - } else { - return 0; - } } - .? { return 0; } - */ - if (pos + len >= sourceLength) - return 0; - - var c1 = s[pos]; - if (c1 != '`' && c1 != '~') - return 0; - - var cnt = 1; - var spaces = false; - for (var i = pos + 1; i < sourceLength; i++) - { - var c = s[i]; - if (c == c1 && !spaces) - cnt++; - else if (c == ' ') - spaces = true; - else if (c == '\n') - return cnt < len ? 0 : cnt; - else - return 0; - } - - return 0; - } - - /// - /// Scans an entity. - /// Returns number of chars matched. - /// - public static int scan_entity(string s, int pos, int length, out string namedEntity, out int numericEntity) - { - /*!re2c - [&] ([#] ([Xx][A-Fa-f0-9]{1,8}|[0-9]{1,8}) |[A-Za-z][A-Za-z0-9]{1,31} ) [;] - { return (p - start); } - .? { return 0; } - */ - - var lastPos = pos + length; - - namedEntity = null; - numericEntity = 0; - - if (pos + 3 >= lastPos) - return 0; - - if (s[pos] != '&') - return 0; - - char c; - int i; - int counter = 0; - if (s[pos + 1] == '#') - { - c = s[pos + 2]; - if (c == 'x' || c == 'X') - { - // expect 1-8 hex digits starting from pos+3 - for (i = pos + 3; i < lastPos; i++) - { - c = s[i]; - if (c >= '0' && c <= '9') - { - if (++counter == 9) return 0; - numericEntity = numericEntity*16 + (c - '0'); - continue; - } - else if (c >= 'A' && c <= 'F') - { - if (++counter == 9) return 0; - numericEntity = numericEntity*16 + (c - 'A' + 10); - continue; - } - else if (c >= 'a' && c <= 'f') - { - if (++counter == 9) return 0; - numericEntity = numericEntity*16 + (c - 'a' + 10); - continue; - } - - if (c == ';') - return counter == 0 ? 0 : i - pos + 1; - - return 0; - } - } - else - { - // expect 1-8 digits starting from pos+2 - for (i = pos + 2; i < lastPos; i++) - { - c = s[i]; - if (c >= '0' && c <= '9') - { - if (++counter == 9) return 0; - numericEntity = numericEntity*10 + (c - '0'); - continue; - } - - if (c == ';') - return counter == 0 ? 0 : i - pos + 1; - - return 0; - } - } - } - else - { - // expect a letter and 1-31 letters or digits - c = s[pos + 1]; - if ((c < 'A' || c > 'Z') && (c < 'a' && c > 'z')) - return 0; - - for (i = pos + 2; i < lastPos; i++) - { - c = s[i]; - if ((c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z')) - { - if (++counter == 32) - return 0; - - continue; - } - - if (c == ';') - { - namedEntity = s.Substring(pos + 1, counter + 1); - return counter == 0 ? 0 : i - pos + 1; - } - - return 0; - } - } - - return 0; - } - } -} \ No newline at end of file diff --git a/src/Textamina.Markdig/Utility.cs b/src/Textamina.Markdig/Helpers/CharHelper.cs similarity index 81% rename from src/Textamina.Markdig/Utility.cs rename to src/Textamina.Markdig/Helpers/CharHelper.cs index 592197be..647c2e23 100644 --- a/src/Textamina.Markdig/Utility.cs +++ b/src/Textamina.Markdig/Helpers/CharHelper.cs @@ -1,11 +1,11 @@ using System.Runtime.CompilerServices; -namespace Textamina.Markdig.Parsing +namespace Textamina.Markdig.Helpers { - internal class Utility + public static class CharHelper { [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsWhitespace(char c) + public static bool IsWhitespace(this char c) { // 2.1 Characters and lines // A whitespace character is a space(U + 0020), tab(U + 0009), newline(U + 000A), line tabulation (U + 000B), form feed (U + 000C), or carriage return (U + 000D). @@ -13,38 +13,38 @@ namespace Textamina.Markdig.Parsing } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsControl(char c) + public static bool IsControl(this char c) { return c < ' ' || char.IsControl(c); } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsEscapableSymbol(char c) + public static bool IsEscapableSymbol(this char c) { // char.IsSymbol also works with Unicode symbols that cannot be escaped based on the specification. return (c > ' ' && c < '0') || (c > '9' && c < 'A') || (c > 'Z' && c < 'a') || (c > 'z' && c < 127) || c == '•'; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsWhiteSpaceOrZero(char c) + public static bool IsWhiteSpaceOrZero(this char c) { return IsWhitespace(c) || IsZero(c); } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsNewLine(char c) + public static bool IsNewLine(this char c) { return c == '\n'; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsZero(char c) + public static bool IsZero(this char c) { return c == '\0'; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsSpace(char c) + public static bool IsSpace(this char c) { // 2.1 Characters and lines // A space is U+0020. @@ -52,7 +52,7 @@ namespace Textamina.Markdig.Parsing } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsTab(char c) + public static bool IsTab(this char c) { // 2.1 Characters and lines // A space is U+0009. @@ -60,13 +60,13 @@ namespace Textamina.Markdig.Parsing } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsSpaceOrTab(char c) + public static bool IsSpaceOrTab(this char c) { return IsSpace(c) || IsTab(c); } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static char EscapeInsecure(char c) + public static char EscapeInsecure(this char c) { // 2.3 Insecure characters // For security reasons, the Unicode character U+0000 must be replaced with the REPLACEMENT CHARACTER (U+FFFD). @@ -74,24 +74,24 @@ namespace Textamina.Markdig.Parsing } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsBulletListMarker(char c) + public static bool IsBulletListMarker(this char c) { return c == '-' || c == '+' || c == '*'; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsAlphaUpper(char c) + public static bool IsAlphaUpper(this char c) { return c >= 'A' && c <= 'Z'; } [MethodImpl(MethodImplOptionPortable.AggressiveInlining)] - public static bool IsAlphaNumeric(char c) + public static bool IsAlphaNumeric(this char c) { return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') || (c >= '0' && c <= '9'); } - public static bool IsASCIIPunctuation(char c) + public static bool IsAsciiPunctuation(this char c) { // 2.1 Characters and lines // An ASCII punctuation character is !, ", #, $, %, &, ', (, ), *, +, ,, -, ., /, :, ;, <, =, >, ?, @, [, \, ], ^, _, `, {, |, }, or ~. diff --git a/src/Textamina.Markdig/Formatters/EntityDecoder.cs b/src/Textamina.Markdig/Helpers/EntityHelper.cs similarity index 99% rename from src/Textamina.Markdig/Formatters/EntityDecoder.cs rename to src/Textamina.Markdig/Helpers/EntityHelper.cs index 4a00ebe7..af413367 100644 --- a/src/Textamina.Markdig/Formatters/EntityDecoder.cs +++ b/src/Textamina.Markdig/Helpers/EntityHelper.cs @@ -1,9 +1,9 @@ using System; using System.Collections.Generic; -namespace Textamina.Markdig.Formatters +namespace Textamina.Markdig.Helpers { - internal static class EntityDecoder + public static class EntityHelper { /// /// Decodes the given HTML entity to the matching Unicode characters. diff --git a/src/Textamina.Markdig/Formatters/HtmlHelper.cs b/src/Textamina.Markdig/Helpers/HtmlHelper.cs similarity index 61% rename from src/Textamina.Markdig/Formatters/HtmlHelper.cs rename to src/Textamina.Markdig/Helpers/HtmlHelper.cs index 926768fa..0bbb29fa 100644 --- a/src/Textamina.Markdig/Formatters/HtmlHelper.cs +++ b/src/Textamina.Markdig/Helpers/HtmlHelper.cs @@ -1,8 +1,8 @@ using System; using System.Text; -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Formatters; -namespace Textamina.Markdig.Formatters +namespace Textamina.Markdig.Helpers { internal static class HtmlHelper { @@ -31,8 +31,29 @@ namespace Textamina.Markdig.Formatters true, true, true, true, true, true, true, true, true, true, true, false, false, false, false, false }; - [ThreadStatic] - private static readonly StringBuilder TempBuilder = new StringBuilder(); + /// + /// List of valid schemes of an URL. The array must be sorted. + /// + private static readonly string[] SchemeArray = new[] + { + "AAA", "AAAS", "ABOUT", "ACAP", "ADIUMXTRA", "AFP", "AFS", "AIM", "APT", "ATTACHMENT", "AW", "BESHARE", + "BITCOIN", "BOLO", "CALLTO", "CAP", "CHROME", "CHROME-EXTENSION", "CID", "COAP", "COM-EVENTBRITE-ATTENDEE", + "CONTENT", "CRID", "CVS", "DATA", "DAV", "DICT", "DLNA-PLAYCONTAINER", "DLNA-PLAYSINGLE", "DNS", "DOI", + "DTN", "DVB", "ED2K", "FACETIME", "FEED", "FILE", "FINGER", "FISH", "FTP", "GEO", "GG", "GIT", + "GIZMOPROJECT", "GO", "GOPHER", "GTALK", "H323", "HCP", "HTTP", "HTTPS", "IAX", "ICAP", "ICON", "IM", "IMAP", + "INFO", "IPN", "IPP", "IRC", "IRC6", "IRCS", "IRIS", "IRIS.BEEP", "IRIS.LWZ", "IRIS.XPC", "IRIS.XPCS", + "ITMS", "JAR", "JAVASCRIPT", "JMS", "KEYPARC", "LASTFM", "LDAP", "LDAPS", "MAGNET", "MAILTO", "MAPS", + "MARKET", "MESSAGE", "MID", "MMS", "MS-HELP", "MSNIM", "MSRP", "MSRPS", "MTQP", "MUMBLE", "MUPDATE", "MVN", + "NEWS", "NFS", "NI", "NIH", "NNTP", "NOTES", "OID", "OPAQUELOCKTOKEN", "PALM", "PAPARAZZI", "PLATFORM", + "POP", "PRES", "PROXY", "PSYC", "QUERY", "RES", "RESOURCE", "RMI", "RSYNC", "RTMP", "RTSP", "SECONDLIFE", + "SERVICE", "SESSION", "SFTP", "SGN", "SHTTP", "SIEVE", "SIP", "SIPS", "SKYPE", "SMB", "SMS", "SNMP", + "SOAP.BEEP", "SOAP.BEEPS", "SOLDAT", "SPOTIFY", "SSH", "STEAM", "SVN", "TAG", "TEAMSPEAK", "TEL", "TELNET", + "TFTP", "THINGS", "THISMESSAGE", "TIP", "TN3270", "TV", "UDP", "UNREAL", "URN", "UT2004", "VEMMI", + "VENTRILO", "VIEW-SOURCE", "WEBCAL", "WS", "WSS", "WTAI", "WYCIWYG", "XCON", "XCON-USERID", "XFIRE", + "XMLRPC.BEEP", "XMLRPC.BEEPS", "XMPP", "XRI", "YMSGR", "Z39.50R", "Z39.50S" + }; + + [ThreadStatic] private static readonly StringBuilder TempBuilder = new StringBuilder(); /// /// Destructively unescape a string: remove backslashes before punctuation or symbol characters. @@ -45,7 +66,7 @@ namespace Textamina.Markdig.Formatters int lastPos = 0; int match; char c; - char[] search = new[] { '\\', '&' }; + char[] search = new[] {'\\', '&'}; var sb = TempBuilder; sb.Clear(); @@ -60,7 +81,7 @@ namespace Textamina.Markdig.Formatters break; c = url[searchPos]; - if (Utility.IsEscapableSymbol(c)) + if (CharHelper.IsEscapableSymbol(c)) { sb.Append(url, lastPos, searchPos - lastPos - 1); lastPos = searchPos; @@ -70,7 +91,8 @@ namespace Textamina.Markdig.Formatters { string namedEntity; int numericEntity; - match = Scanner.scan_entity(url, searchPos, url.Length - searchPos, out namedEntity, out numericEntity); + match = scan_entity(url, searchPos, url.Length - searchPos, out namedEntity, + out numericEntity); if (match == 0) { searchPos++; @@ -81,7 +103,7 @@ namespace Textamina.Markdig.Formatters if (namedEntity != null) { - var decoded = EntityDecoder.DecodeEntity(namedEntity); + var decoded = EntityHelper.DecodeEntity(namedEntity); if (decoded != null) { sb.Append(url, lastPos, searchPos - match - lastPos); @@ -91,7 +113,7 @@ namespace Textamina.Markdig.Formatters } else if (numericEntity > 0) { - var decoded = EntityDecoder.DecodeEntity(numericEntity); + var decoded = EntityHelper.DecodeEntity(numericEntity); if (decoded != null) { sb.Append(url, lastPos, searchPos - match - lastPos); @@ -122,7 +144,7 @@ namespace Textamina.Markdig.Formatters /// Escapes special URL characters. /// /// Orig: escape_html(inp, preserve_entities) - internal static void EscapeUrl(string input, HtmlTextWriter target) + public static void EscapeUrl(string input, HtmlTextWriter target) { if (input == null) return; @@ -188,7 +210,7 @@ namespace Textamina.Markdig.Formatters /// Escapes special HTML characters. /// /// Orig: escape_html(inp, preserve_entities) - internal static void EscapeHtml(string input, HtmlTextWriter target) + public static void EscapeHtml(string input, HtmlTextWriter target) { if (input.Length == 0) return; @@ -230,6 +252,7 @@ namespace Textamina.Markdig.Formatters target.Write(buffer, lastPos, input.Length - lastPos); } + /* /// @@ -282,5 +305,121 @@ namespace Textamina.Markdig.Formatters } } */ + + /// + /// Scans an entity. + /// Returns number of chars matched. + /// + public static int scan_entity(string s, int pos, int length, out string namedEntity, out int numericEntity) + { + /*!re2c + [&] ([#] ([Xx][A-Fa-f0-9]{1,8}|[0-9]{1,8}) |[A-Za-z][A-Za-z0-9]{1,31} ) [;] + { return (p - start); } + .? { return 0; } + */ + + var lastPos = pos + length; + + namedEntity = null; + numericEntity = 0; + + if (pos + 3 >= lastPos) + return 0; + + if (s[pos] != '&') + return 0; + + char c; + int i; + int counter = 0; + if (s[pos + 1] == '#') + { + c = s[pos + 2]; + if (c == 'x' || c == 'X') + { + // expect 1-8 hex digits starting from pos+3 + for (i = pos + 3; i < lastPos; i++) + { + c = s[i]; + if (c >= '0' && c <= '9') + { + if (++counter == 9) return 0; + numericEntity = numericEntity*16 + (c - '0'); + continue; + } + else if (c >= 'A' && c <= 'F') + { + if (++counter == 9) return 0; + numericEntity = numericEntity*16 + (c - 'A' + 10); + continue; + } + else if (c >= 'a' && c <= 'f') + { + if (++counter == 9) return 0; + numericEntity = numericEntity*16 + (c - 'a' + 10); + continue; + } + + if (c == ';') + return counter == 0 ? 0 : i - pos + 1; + + return 0; + } + } + else + { + // expect 1-8 digits starting from pos+2 + for (i = pos + 2; i < lastPos; i++) + { + c = s[i]; + if (c >= '0' && c <= '9') + { + if (++counter == 9) return 0; + numericEntity = numericEntity*10 + (c - '0'); + continue; + } + + if (c == ';') + return counter == 0 ? 0 : i - pos + 1; + + return 0; + } + } + } + else + { + // expect a letter and 1-31 letters or digits + c = s[pos + 1]; + if ((c < 'A' || c > 'Z') && (c < 'a' && c > 'z')) + return 0; + + for (i = pos + 2; i < lastPos; i++) + { + c = s[i]; + if ((c >= '0' && c <= '9') || (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z')) + { + if (++counter == 32) + return 0; + + continue; + } + + if (c == ';') + { + namedEntity = s.Substring(pos + 1, counter + 1); + return counter == 0 ? 0 : i - pos + 1; + } + + return 0; + } + } + + return 0; + } + + public static bool IsUrlScheme(string scheme) + { + return Array.BinarySearch(SchemeArray, scheme, StringComparer.Ordinal) >= 0; + } } } \ No newline at end of file diff --git a/src/Textamina.Markdig/Parsing/MarkdownParser.cs b/src/Textamina.Markdig/Parsing/MarkdownParser.cs index 9aee6f09..2f5fcb0b 100644 --- a/src/Textamina.Markdig/Parsing/MarkdownParser.cs +++ b/src/Textamina.Markdig/Parsing/MarkdownParser.cs @@ -3,6 +3,7 @@ using System.Collections.Generic; using System.IO; using System.Text; using System.Xml.Serialization; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Syntax; namespace Textamina.Markdig.Parsing @@ -381,7 +382,7 @@ namespace Textamina.Markdig.Parsing var c = (char) nextChar; // 2.3 Insecure characters - c = Utility.EscapeInsecure(c); + c = CharHelper.EscapeInsecure(c); // Go to next char, expecting most likely a \n, otherwise skip it // TODO: Should we treat it as an error in no \n is following? diff --git a/src/Textamina.Markdig/Syntax/BreakBlock.cs b/src/Textamina.Markdig/Syntax/BreakBlock.cs index 73b8dc45..d7b13dc2 100644 --- a/src/Textamina.Markdig/Syntax/BreakBlock.cs +++ b/src/Textamina.Markdig/Syntax/BreakBlock.cs @@ -1,4 +1,5 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax { @@ -38,7 +39,7 @@ namespace Textamina.Markdig.Syntax { count++; } - else if (!Utility.IsSpace(c) || count == 0) + else if (!CharHelper.IsSpace(c) || count == 0) { return MatchLineResult.None; } diff --git a/src/Textamina.Markdig/Syntax/CodeBlock.cs b/src/Textamina.Markdig/Syntax/CodeBlock.cs index c673cd03..3b782100 100644 --- a/src/Textamina.Markdig/Syntax/CodeBlock.cs +++ b/src/Textamina.Markdig/Syntax/CodeBlock.cs @@ -1,4 +1,5 @@ using System; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -33,8 +34,8 @@ namespace Textamina.Markdig.Syntax // 4.4 Indented code blocks var c = liner.Current; - var isTab = Utility.IsTab(c); - var isSpace = Utility.IsSpace(c); + var isTab = CharHelper.IsTab(c); + var isSpace = CharHelper.IsSpace(c); if ((isTab || (isSpace && (liner.Start - position) == 3)) && !liner.IsBlankLine()) { liner.NextChar(); diff --git a/src/Textamina.Markdig/Syntax/HeadingBlock.cs b/src/Textamina.Markdig/Syntax/HeadingBlock.cs index 0537e278..92e17b1d 100644 --- a/src/Textamina.Markdig/Syntax/HeadingBlock.cs +++ b/src/Textamina.Markdig/Syntax/HeadingBlock.cs @@ -1,4 +1,5 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax { @@ -44,7 +45,7 @@ namespace Textamina.Markdig.Syntax // closing # will be handled later, because anyway we have matched // A space is required after leading # - if (leadingCount > 0 && leadingCount <=6 && (Utility.IsSpace(c) || liner.IsEol)) + if (leadingCount > 0 && leadingCount <=6 && (CharHelper.IsSpace(c) || liner.IsEol)) { liner.NextChar(); state.Block = new HeadingBlock() {Level = leadingCount}; @@ -58,7 +59,7 @@ namespace Textamina.Markdig.Syntax c = liner[i]; if (endState == 0) { - if (Utility.IsSpace(c)) // TODO: Not clear if it is a space or space+tab in the specs + if (CharHelper.IsSpace(c)) // TODO: Not clear if it is a space or space+tab in the specs { continue; } @@ -74,7 +75,7 @@ namespace Textamina.Markdig.Syntax if (countClosingTags > 0) { - if (Utility.IsSpace(c)) + if (CharHelper.IsSpace(c)) { liner.End = i - 1; } diff --git a/src/Textamina.Markdig/Syntax/HtmlBlock.cs b/src/Textamina.Markdig/Syntax/HtmlBlock.cs index d5a03f0d..436a2dd9 100644 --- a/src/Textamina.Markdig/Syntax/HtmlBlock.cs +++ b/src/Textamina.Markdig/Syntax/HtmlBlock.cs @@ -1,4 +1,5 @@ using System; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -107,7 +108,7 @@ namespace Textamina.Markdig.Syntax for (int i = 0; i < 3; i++) { - if (!Utility.IsSpace(liner.PeekChar(index))) + if (!CharHelper.IsSpace(liner.PeekChar(index))) { break; } @@ -129,7 +130,7 @@ namespace Textamina.Markdig.Syntax { return CreateHtmlBlock(ref state, HtmlBlockType.Comment); // group 2 } - if (Utility.IsAlphaUpper(c)) + if (CharHelper.IsAlphaUpper(c)) { return CreateHtmlBlock(ref state, HtmlBlockType.DocumentType); // group 4 } @@ -157,14 +158,14 @@ namespace Textamina.Markdig.Syntax for (; count < tag.Length; index++, count++) { c = liner.PeekChar(index); - if (!Utility.IsAlphaNumeric(c)) + if (!CharHelper.IsAlphaNumeric(c)) { break; } tag[count] = char.ToLowerInvariant(c); } - if (!(c == '>' || (!hasLeadingClose && c == '/' && liner.PeekChar(index + 1) == '>') || Utility.IsWhitespace(c) || c == '\0')) + if (!(c == '>' || (!hasLeadingClose && c == '/' && liner.PeekChar(index + 1) == '>') || CharHelper.IsWhitespace(c) || c == '\0')) { return MatchLineResult.None; } diff --git a/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs b/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs index 64d8ec23..8633261f 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/CodeInline.cs @@ -1,3 +1,4 @@ +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -33,8 +34,8 @@ namespace Textamina.Markdig.Syntax while (c != '\0') { // Skip new lines - var isWhitespace = Utility.IsWhitespace(c); - if (!(Utility.IsNewLine(c) || (builder.Length == 0 && isWhitespace) || (lastWhiteSpace && isWhitespace))) + var isWhitespace = CharHelper.IsWhitespace(c); + if (!(CharHelper.IsNewLine(c) || (builder.Length == 0 && isWhitespace) || (lastWhiteSpace && isWhitespace))) { if (c == '`') { @@ -62,7 +63,7 @@ namespace Textamina.Markdig.Syntax int newLength = builder.Length; for (int i = builder.Length - 1; i >= 0; i--) { - if (Utility.IsWhitespace(builder[i])) + if (CharHelper.IsWhitespace(builder[i])) { newLength--; } diff --git a/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs b/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs index 67294159..818e2305 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/EmphasisInline.cs @@ -1,4 +1,5 @@ using System; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -41,11 +42,11 @@ namespace Textamina.Markdig.Syntax // (b) either not followed by a punctuation character, or preceded by Unicode whitespace // or a punctuation character. // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. - var afterIsPunctuation = Utility.IsASCIIPunctuation(c); - bool canOpen = !Utility.IsWhiteSpaceOrZero(c) && + var afterIsPunctuation = CharHelper.IsAsciiPunctuation(c); + bool canOpen = !CharHelper.IsWhiteSpaceOrZero(c) && (!afterIsPunctuation || - !Utility.IsWhiteSpaceOrZero(pc) || - !Utility.IsASCIIPunctuation(pc)); + !CharHelper.IsWhiteSpaceOrZero(pc) || + !CharHelper.IsAsciiPunctuation(pc)); // A right-flanking delimiter run is a delimiter run that is @@ -53,11 +54,11 @@ namespace Textamina.Markdig.Syntax // (b) either not preceded by a punctuation character, or followed by Unicode whitespace // or a punctuation character. // For purposes of this definition, the beginning and the end of the line count as Unicode whitespace. - var beforeIsPunctuation = Utility.IsASCIIPunctuation(pc); - bool canClose = !Utility.IsWhiteSpaceOrZero(pc) && + var beforeIsPunctuation = CharHelper.IsAsciiPunctuation(pc); + bool canClose = !CharHelper.IsWhiteSpaceOrZero(pc) && (!beforeIsPunctuation || - !Utility.IsWhiteSpaceOrZero(c) || - !Utility.IsASCIIPunctuation(c)); + !CharHelper.IsWhiteSpaceOrZero(c) || + !CharHelper.IsAsciiPunctuation(c)); if (delimiterChar == '_') { diff --git a/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs b/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs index 519bf688..18b70066 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/EscapeInline.cs @@ -1,3 +1,4 @@ +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -21,7 +22,7 @@ namespace Textamina.Markdig.Syntax // Go to escape character lines.NextChar(); - if (Utility.IsASCIIPunctuation(lines.Current)) + if (CharHelper.IsAsciiPunctuation(lines.Current)) { state.Inline = new EscapeInline() {EscapedChar = lines.Current}; lines.NextChar(); diff --git a/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs b/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs index 4fe9e521..931eccd1 100644 --- a/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs +++ b/src/Textamina.Markdig/Syntax/Inlines/LinkInline.cs @@ -1,6 +1,7 @@ using System; using System.Text; using Textamina.Markdig.Formatters; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -113,7 +114,7 @@ namespace Textamina.Markdig.Syntax break; } - if (hasEscape && !Utility.IsASCIIPunctuation(c)) + if (hasEscape && !CharHelper.IsAsciiPunctuation(c)) { buffer.Append('\\'); } @@ -173,7 +174,7 @@ namespace Textamina.Markdig.Syntax break; } - if (hasEscape && !Utility.IsASCIIPunctuation(c)) + if (hasEscape && !CharHelper.IsAsciiPunctuation(c)) { buffer.Append('\\'); } @@ -186,7 +187,7 @@ namespace Textamina.Markdig.Syntax hasEscape = false; - if (Utility.IsWhitespace(c)) // TODO: specs unclear. space is strict or relaxed? (includes tabs?) + if (CharHelper.IsWhitespace(c)) // TODO: specs unclear. space is strict or relaxed? (includes tabs?) { break; } @@ -231,7 +232,7 @@ namespace Textamina.Markdig.Syntax } } - if (hasEscape && !Utility.IsASCIIPunctuation(c)) + if (hasEscape && !CharHelper.IsAsciiPunctuation(c)) { buffer.Append('\\'); } @@ -246,7 +247,7 @@ namespace Textamina.Markdig.Syntax hasEscape = false; - if (Utility.IsSpaceOrTab(c) || Utility.IsControl(c)) // TODO: specs unclear. space is strict or relaxed? (includes tabs?) + if (CharHelper.IsSpaceOrTab(c) || CharHelper.IsControl(c)) // TODO: specs unclear. space is strict or relaxed? (includes tabs?) { isValid = true; break; diff --git a/src/Textamina.Markdig/Syntax/ListBlock.cs b/src/Textamina.Markdig/Syntax/ListBlock.cs index f4bcd432..f339f326 100644 --- a/src/Textamina.Markdig/Syntax/ListBlock.cs +++ b/src/Textamina.Markdig/Syntax/ListBlock.cs @@ -1,5 +1,7 @@ + +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -48,7 +50,7 @@ namespace Textamina.Markdig.Syntax var c = liner.Current; var startPosition = liner.Column; - while (Utility.IsSpaceOrTab(c)) + while (CharHelper.IsSpaceOrTab(c)) { c = liner.NextChar(); var endPosition = liner.Column; @@ -71,7 +73,7 @@ namespace Textamina.Markdig.Syntax liner.SkipLeadingSpaces3(); var preIndent = liner.Start - preStartPosition; var c = liner.Current; - if (Utility.IsBulletListMarker(c)) + if (CharHelper.IsBulletListMarker(c)) { var listType = c; @@ -82,7 +84,7 @@ namespace Textamina.Markdig.Syntax for (int i = 0; i < 4; i++) { c = liner.NextChar(); - if (!Utility.IsSpaceOrTab(c)) + if (!CharHelper.IsSpaceOrTab(c)) { break; } diff --git a/src/Textamina.Markdig/Syntax/ParagraphBlock.cs b/src/Textamina.Markdig/Syntax/ParagraphBlock.cs index e1bbd704..47255e4f 100644 --- a/src/Textamina.Markdig/Syntax/ParagraphBlock.cs +++ b/src/Textamina.Markdig/Syntax/ParagraphBlock.cs @@ -1,6 +1,7 @@ +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -55,7 +56,7 @@ namespace Textamina.Markdig.Syntax if (checkForSpaces) { - if (!Utility.IsSpaceOrTab(c)) + if (!CharHelper.IsSpaceOrTab(c)) { headingChar = (char)0; break; @@ -63,7 +64,7 @@ namespace Textamina.Markdig.Syntax } else if (c != headingChar) { - if (Utility.IsSpaceOrTab(c)) + if (CharHelper.IsSpaceOrTab(c)) { checkForSpaces = true; } @@ -93,7 +94,7 @@ namespace Textamina.Markdig.Syntax { // Remove leading spaces from paragraph var c = liner.Current; - while (Utility.IsSpaceOrTab(c)) + while (CharHelper.IsSpaceOrTab(c)) { c = liner.NextChar(); } diff --git a/src/Textamina.Markdig/Syntax/QuoteBlock.cs b/src/Textamina.Markdig/Syntax/QuoteBlock.cs index 21773ff5..81fb6c35 100644 --- a/src/Textamina.Markdig/Syntax/QuoteBlock.cs +++ b/src/Textamina.Markdig/Syntax/QuoteBlock.cs @@ -1,4 +1,5 @@ -using Textamina.Markdig.Parsing; +using Textamina.Markdig.Helpers; +using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax { @@ -23,7 +24,7 @@ namespace Textamina.Markdig.Syntax } c = liner.NextChar(); - if (Utility.IsSpace(c)) + if (CharHelper.IsSpace(c)) { liner.NextChar(); } diff --git a/src/Textamina.Markdig/Syntax/StringLine.cs b/src/Textamina.Markdig/Syntax/StringLine.cs index 2dfc2fbd..84c36633 100644 --- a/src/Textamina.Markdig/Syntax/StringLine.cs +++ b/src/Textamina.Markdig/Syntax/StringLine.cs @@ -1,5 +1,6 @@ using System; using System.Runtime.CompilerServices; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -63,7 +64,7 @@ namespace Textamina.Markdig.Syntax { // If previous character was a tab make the Column += 4 Column++; - if (Utility.IsTab(Current)) + if (CharHelper.IsTab(Current)) { // Align the tab on a column Column = ((Column + 3) / 4) * 4; @@ -156,7 +157,7 @@ namespace Textamina.Markdig.Syntax for (int i = Start; i <= End; i++) { - if (!Utility.IsSpace(Text[i])) + if (!CharHelper.IsSpace(Text[i])) { return false; } @@ -169,7 +170,7 @@ namespace Textamina.Markdig.Syntax { for (int i = 0; i < 3; i++) { - if (!Utility.IsSpace(Current)) + if (!CharHelper.IsSpace(Current)) { break; } @@ -186,7 +187,7 @@ namespace Textamina.Markdig.Syntax { // Strip leading spaces var c = Current; - while (Utility.IsSpace(c)) + while (CharHelper.IsSpace(c)) { c = NextChar(); } @@ -197,7 +198,7 @@ namespace Textamina.Markdig.Syntax for (int i = End; i >= Start; i--) { End = i; - if (!Utility.IsSpace(this[i])) + if (!CharHelper.IsSpace(this[i])) { break; } diff --git a/src/Textamina.Markdig/Syntax/StringLineGroup.cs b/src/Textamina.Markdig/Syntax/StringLineGroup.cs index 28a4642c..0209a5fd 100644 --- a/src/Textamina.Markdig/Syntax/StringLineGroup.cs +++ b/src/Textamina.Markdig/Syntax/StringLineGroup.cs @@ -1,6 +1,7 @@ using System; using System.Collections.ObjectModel; using System.Text; +using Textamina.Markdig.Helpers; using Textamina.Markdig.Parsing; namespace Textamina.Markdig.Syntax @@ -124,7 +125,7 @@ namespace Textamina.Markdig.Syntax public bool SkipWhiteSpaces() { bool hasWhitespaces = false; - while (Utility.IsWhitespace(Current)) + while (CharHelper.IsWhitespace(Current)) { NextChar(); hasWhitespaces = true; diff --git a/src/Textamina.Markdig/Textamina.Markdig.csproj b/src/Textamina.Markdig/Textamina.Markdig.csproj index 473cbd27..05479595 100644 --- a/src/Textamina.Markdig/Textamina.Markdig.csproj +++ b/src/Textamina.Markdig/Textamina.Markdig.csproj @@ -35,11 +35,10 @@ 4 - + - + - @@ -74,7 +73,7 @@ - +