Editor/HotCodeEditor/CodeView/SyntaxLanguage.cs
using System;
using System.Collections.Generic;
using System.IO;
public enum TokenKind
{
Text,
Keyword,
ControlKeyword,
Type,
Method,
String,
Number,
Comment,
Preprocessor,
Property
}
public readonly record struct Token( int Start, int Length, TokenKind Kind );
/// <summary>
/// What a line starts "inside of", for constructs that span lines.
/// </summary>
public enum LexState
{
Normal,
BlockComment,
VerbatimString,
RawString
}
/// <summary>
/// A small, line-based tokenizer configured per language. It isn't a real parser: types and
/// methods are guessed (PascalCase word = type, word followed by "(" = method), which is
/// right most of the time for C# and cheap enough to run on every keystroke.
/// </summary>
public class SyntaxLanguage
{
public string Name { get; init; }
public HashSet<string> Keywords { get; init; } = new();
public HashSet<string> ControlKeywords { get; init; } = new();
public string LineComment { get; init; }
public bool BlockComments { get; init; }
public bool CSharpStrings { get; init; }
public bool SingleQuoteStrings { get; init; }
public bool HashPreprocessor { get; init; }
public bool GuessTypesAndMethods { get; init; }
/// <summary>
/// CSS-style: identifiers may contain '-', and "name:" is a property.
/// </summary>
public bool CssIdentifiers { get; init; }
/// <summary>
/// No highlighting at all.
/// </summary>
public bool IsPlain { get; init; }
public static readonly SyntaxLanguage Plain = new() { Name = "Plain Text", IsPlain = true };
public static readonly SyntaxLanguage CSharp = new()
{
Name = "C#",
LineComment = "//",
BlockComments = true,
CSharpStrings = true,
HashPreprocessor = true,
GuessTypesAndMethods = true,
Keywords = new()
{
"abstract", "as", "base", "bool", "byte", "char", "checked", "class", "const", "decimal", "default", "delegate",
"double", "enum", "event", "explicit", "extern", "false", "fixed", "float", "implicit", "in", "int", "interface",
"internal", "is", "lock", "long", "namespace", "new", "null", "object", "operator", "out", "override", "params",
"private", "protected", "public", "readonly", "ref", "sbyte", "sealed", "short", "sizeof", "stackalloc", "static",
"string", "struct", "this", "true", "typeof", "uint", "ulong", "unchecked", "unsafe", "ushort", "using", "virtual",
"void", "volatile", "var", "dynamic", "async", "get", "set", "init", "value", "partial", "where", "record", "nameof",
"required", "global", "with", "not", "and", "or", "nint", "nuint", "file", "scoped", "field", "add", "remove", "let",
"from", "select", "orderby", "group", "into", "on", "equals", "by", "ascending", "descending", "join", "alias"
},
ControlKeywords = new()
{
"if", "else", "switch", "case", "for", "foreach", "while", "do", "break", "continue", "return", "goto", "try",
"catch", "finally", "throw", "yield", "await", "when"
}
};
public static readonly SyntaxLanguage Json = new()
{
Name = "JSON",
LineComment = "//",
BlockComments = true,
Keywords = new() { "true", "false", "null" }
};
public static readonly SyntaxLanguage Css = new()
{
Name = "SCSS",
LineComment = "//",
BlockComments = true,
SingleQuoteStrings = true,
CssIdentifiers = true,
Keywords = new() { "!important", "@import", "@use", "@media", "@keyframes", "@mixin", "@include", "@extend", "@if", "@else", "@each", "@for", "@function", "@return" }
};
public static SyntaxLanguage FromPath( string path ) => Path.GetExtension( path ).ToLowerInvariant() switch
{
".cs" => CSharp,
".json" or ".sbproj" => Json,
".scss" or ".css" => Css,
_ => Plain
};
/// <summary>
/// Appends the colored tokens of <paramref name="line"/> to <paramref name="tokens"/> (plain text gets no token)
/// and returns the state the next line starts in.
/// </summary>
public LexState Tokenize( string line, LexState state, List<Token> tokens )
{
if ( IsPlain ) return LexState.Normal;
int i = 0;
int n = line.Length;
// Continue whatever the previous line left open
switch ( state )
{
case LexState.BlockComment:
{
var end = line.IndexOf( "*/", StringComparison.Ordinal );
if ( end < 0 ) { Add( tokens, 0, n, TokenKind.Comment ); return state; }
Add( tokens, 0, end + 2, TokenKind.Comment );
i = end + 2;
break;
}
case LexState.VerbatimString:
{
var end = FindVerbatimEnd( line, 0 );
if ( end < 0 ) { Add( tokens, 0, n, TokenKind.String ); return state; }
Add( tokens, 0, end, TokenKind.String );
i = end;
break;
}
case LexState.RawString:
{
var end = line.IndexOf( "\"\"\"", StringComparison.Ordinal );
if ( end < 0 ) { Add( tokens, 0, n, TokenKind.String ); return state; }
Add( tokens, 0, end + 3, TokenKind.String );
i = end + 3;
break;
}
}
if ( HashPreprocessor && i == 0 )
{
var first = 0;
while ( first < n && char.IsWhiteSpace( line[first] ) ) first++;
if ( first < n && line[first] == '#' )
{
Add( tokens, first, n - first, TokenKind.Preprocessor );
return LexState.Normal;
}
}
while ( i < n )
{
var c = line[i];
if ( char.IsWhiteSpace( c ) ) { i++; continue; }
// Comments
if ( LineComment is not null && string.CompareOrdinal( line, i, LineComment, 0, LineComment.Length ) == 0 )
{
Add( tokens, i, n - i, TokenKind.Comment );
return LexState.Normal;
}
if ( BlockComments && c == '/' && i + 1 < n && line[i + 1] == '*' )
{
var end = line.IndexOf( "*/", i + 2, StringComparison.Ordinal );
if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.Comment ); return LexState.BlockComment; }
Add( tokens, i, end + 2 - i, TokenKind.Comment );
i = end + 2;
continue;
}
// Strings
if ( CSharpStrings && TryStartCSharpString( line, i, out var prefix, out var verbatim, out var raw ) )
{
var bodyStart = i + prefix;
if ( raw )
{
var end = line.IndexOf( "\"\"\"", bodyStart, StringComparison.Ordinal );
if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.String ); return LexState.RawString; }
Add( tokens, i, end + 3 - i, TokenKind.String );
i = end + 3;
}
else if ( verbatim )
{
var end = FindVerbatimEnd( line, bodyStart );
if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.String ); return LexState.VerbatimString; }
Add( tokens, i, end - i, TokenKind.String );
i = end;
}
else
{
var end = FindQuotedEnd( line, bodyStart, '"' );
Add( tokens, i, end - i, TokenKind.String );
i = end;
}
continue;
}
if ( c == '"' || (c == '\'' && (CSharpStrings || SingleQuoteStrings)) )
{
var end = FindQuotedEnd( line, i + 1, c );
Add( tokens, i, end - i, TokenKind.String );
i = end;
continue;
}
// Numbers
if ( char.IsDigit( c ) || (c == '.' && i + 1 < n && char.IsDigit( line[i + 1] )) )
{
var start = i;
while ( i < n && (char.IsLetterOrDigit( line[i] ) || line[i] == '.' || line[i] == '_') ) i++;
if ( CssIdentifiers && i < n && line[i] == '%' ) i++;
Add( tokens, start, i - start, TokenKind.Number );
continue;
}
// CSS hex colours: #fff, #a0b1c2
if ( CssIdentifiers && c == '#' && i + 1 < n && Uri.IsHexDigit( line[i + 1] ) )
{
var start = i++;
while ( i < n && char.IsLetterOrDigit( line[i] ) ) i++;
Add( tokens, start, i - start, TokenKind.Number );
continue;
}
// Words
if ( IsIdentStart( c ) || (c == '@' && i + 1 < n && IsIdentStart( line[i + 1] )) || (CssIdentifiers && (c == '$' || c == '!')) )
{
var start = i++;
while ( i < n && IsIdentPart( line[i] ) ) i++;
var word = line.Substring( start, i - start );
var kind = ClassifyWord( line, word, i );
if ( kind != TokenKind.Text ) Add( tokens, start, i - start, kind );
continue;
}
i++;
}
return LexState.Normal;
}
private TokenKind ClassifyWord( string line, string word, int after )
{
if ( ControlKeywords.Contains( word ) ) return TokenKind.ControlKeyword;
if ( Keywords.Contains( word ) ) return TokenKind.Keyword;
if ( CssIdentifiers )
{
if ( word[0] == '$' ) return TokenKind.Type;
// "color: red" is a property, but "a:hover {" is a selector
var isFirstOnLine = string.IsNullOrWhiteSpace( line[..(after - word.Length)] );
return isFirstOnLine && NextNonSpace( line, after ) == ':' && !line.Contains( '{' ) ? TokenKind.Property : TokenKind.Text;
}
if ( !GuessTypesAndMethods ) return TokenKind.Text;
var nextChar = NextNonSpace( line, after );
if ( nextChar == '(' ) return TokenKind.Method;
var bare = word[0] == '@' ? word[1..] : word;
if ( bare.Length > 0 && char.IsUpper( bare[0] ) ) return TokenKind.Type;
return TokenKind.Text;
}
private bool IsIdentStart( char c ) => char.IsLetter( c ) || c == '_';
private bool IsIdentPart( char c ) => char.IsLetterOrDigit( c ) || c == '_' || (CssIdentifiers && c == '-');
private static char SafeChar( string s, int i ) => i < s.Length ? s[i] : '\0';
private static char NextNonSpace( string line, int i )
{
while ( i < line.Length && line[i] == ' ' ) i++;
return SafeChar( line, i );
}
/// <summary>
/// Recognises the C# string openers: " $" @" $@" @$" """ $""" (and $$""" etc).
/// </summary>
private static bool TryStartCSharpString( string line, int i, out int prefixLength, out bool verbatim, out bool raw )
{
prefixLength = 0;
verbatim = false;
raw = false;
var j = i;
while ( j < line.Length && (line[j] == '$' || line[j] == '@') )
{
if ( line[j] == '@' ) verbatim = true;
j++;
}
if ( j >= line.Length || line[j] != '"' )
return false;
// Plain "..." is handled by the generic quote branch
if ( j == i && !(j + 2 < line.Length && line[j + 1] == '"' && line[j + 2] == '"') )
return false;
if ( !verbatim && j + 2 < line.Length && line[j + 1] == '"' && line[j + 2] == '"' )
{
raw = true;
var k = j;
while ( k < line.Length && line[k] == '"' ) k++;
prefixLength = k - i;
return true;
}
prefixLength = j + 1 - i;
return true;
}
/// <summary>
/// Index just past the closing quote of a verbatim string ("" is an escaped quote), or -1 if it runs off the line.
/// </summary>
private static int FindVerbatimEnd( string line, int i )
{
while ( i < line.Length )
{
if ( line[i] == '"' )
{
if ( i + 1 < line.Length && line[i + 1] == '"' ) { i += 2; continue; }
return i + 1;
}
i++;
}
return -1;
}
/// <summary>
/// Index just past the closing quote, honouring backslash escapes. Unterminated strings end at the line end.
/// </summary>
private static int FindQuotedEnd( string line, int i, char quote )
{
while ( i < line.Length )
{
if ( line[i] == '\\' ) { i += 2; continue; }
if ( line[i] == quote ) return i + 1;
i++;
}
return line.Length;
}
private static void Add( List<Token> tokens, int start, int length, TokenKind kind )
{
if ( length > 0 ) tokens.Add( new Token( start, length, kind ) );
}
}