Editor/HotCodeEditor/CodeView/SyntaxLanguage.cs
using System;
using System.Collections.Generic;
using System.IO;

public enum TokenKind
{
	Text,
	Keyword,
	ControlKeyword,
	Type,
	Method,
	String,
	Number,
	Comment,
	Preprocessor,
	Property
}

public readonly record struct Token( int Start, int Length, TokenKind Kind );

/// <summary>
/// What a line starts "inside of", for constructs that span lines.
/// </summary>
public enum LexState
{
	Normal,
	BlockComment,
	VerbatimString,
	RawString
}

/// <summary>
/// A small, line-based tokenizer configured per language. It isn't a real parser: types and
/// methods are guessed (PascalCase word = type, word followed by "(" = method), which is
/// right most of the time for C# and cheap enough to run on every keystroke.
/// </summary>
public class SyntaxLanguage
{
	public string Name { get; init; }
	public HashSet<string> Keywords { get; init; } = new();
	public HashSet<string> ControlKeywords { get; init; } = new();
	public string LineComment { get; init; }
	public bool BlockComments { get; init; }
	public bool CSharpStrings { get; init; }
	public bool SingleQuoteStrings { get; init; }
	public bool HashPreprocessor { get; init; }
	public bool GuessTypesAndMethods { get; init; }

	/// <summary>
	/// CSS-style: identifiers may contain '-', and "name:" is a property.
	/// </summary>
	public bool CssIdentifiers { get; init; }

	/// <summary>
	/// No highlighting at all.
	/// </summary>
	public bool IsPlain { get; init; }

	public static readonly SyntaxLanguage Plain = new() { Name = "Plain Text", IsPlain = true };

	public static readonly SyntaxLanguage CSharp = new()
	{
		Name = "C#",
		LineComment = "//",
		BlockComments = true,
		CSharpStrings = true,
		HashPreprocessor = true,
		GuessTypesAndMethods = true,
		Keywords = new()
		{
			"abstract", "as", "base", "bool", "byte", "char", "checked", "class", "const", "decimal", "default", "delegate",
			"double", "enum", "event", "explicit", "extern", "false", "fixed", "float", "implicit", "in", "int", "interface",
			"internal", "is", "lock", "long", "namespace", "new", "null", "object", "operator", "out", "override", "params",
			"private", "protected", "public", "readonly", "ref", "sbyte", "sealed", "short", "sizeof", "stackalloc", "static",
			"string", "struct", "this", "true", "typeof", "uint", "ulong", "unchecked", "unsafe", "ushort", "using", "virtual",
			"void", "volatile", "var", "dynamic", "async", "get", "set", "init", "value", "partial", "where", "record", "nameof",
			"required", "global", "with", "not", "and", "or", "nint", "nuint", "file", "scoped", "field", "add", "remove", "let",
			"from", "select", "orderby", "group", "into", "on", "equals", "by", "ascending", "descending", "join", "alias"
		},
		ControlKeywords = new()
		{
			"if", "else", "switch", "case", "for", "foreach", "while", "do", "break", "continue", "return", "goto", "try",
			"catch", "finally", "throw", "yield", "await", "when"
		}
	};

	public static readonly SyntaxLanguage Json = new()
	{
		Name = "JSON",
		LineComment = "//",
		BlockComments = true,
		Keywords = new() { "true", "false", "null" }
	};

	public static readonly SyntaxLanguage Css = new()
	{
		Name = "SCSS",
		LineComment = "//",
		BlockComments = true,
		SingleQuoteStrings = true,
		CssIdentifiers = true,
		Keywords = new() { "!important", "@import", "@use", "@media", "@keyframes", "@mixin", "@include", "@extend", "@if", "@else", "@each", "@for", "@function", "@return" }
	};

	public static SyntaxLanguage FromPath( string path ) => Path.GetExtension( path ).ToLowerInvariant() switch
	{
		".cs" => CSharp,
		".json" or ".sbproj" => Json,
		".scss" or ".css" => Css,
		_ => Plain
	};

	/// <summary>
	/// Appends the colored tokens of <paramref name="line"/> to <paramref name="tokens"/> (plain text gets no token)
	/// and returns the state the next line starts in.
	/// </summary>
	public LexState Tokenize( string line, LexState state, List<Token> tokens )
	{
		if ( IsPlain ) return LexState.Normal;

		int i = 0;
		int n = line.Length;

		// Continue whatever the previous line left open
		switch ( state )
		{
			case LexState.BlockComment:
			{
				var end = line.IndexOf( "*/", StringComparison.Ordinal );
				if ( end < 0 ) { Add( tokens, 0, n, TokenKind.Comment ); return state; }
				Add( tokens, 0, end + 2, TokenKind.Comment );
				i = end + 2;
				break;
			}
			case LexState.VerbatimString:
			{
				var end = FindVerbatimEnd( line, 0 );
				if ( end < 0 ) { Add( tokens, 0, n, TokenKind.String ); return state; }
				Add( tokens, 0, end, TokenKind.String );
				i = end;
				break;
			}
			case LexState.RawString:
			{
				var end = line.IndexOf( "\"\"\"", StringComparison.Ordinal );
				if ( end < 0 ) { Add( tokens, 0, n, TokenKind.String ); return state; }
				Add( tokens, 0, end + 3, TokenKind.String );
				i = end + 3;
				break;
			}
		}

		if ( HashPreprocessor && i == 0 )
		{
			var first = 0;
			while ( first < n && char.IsWhiteSpace( line[first] ) ) first++;
			if ( first < n && line[first] == '#' )
			{
				Add( tokens, first, n - first, TokenKind.Preprocessor );
				return LexState.Normal;
			}
		}

		while ( i < n )
		{
			var c = line[i];

			if ( char.IsWhiteSpace( c ) ) { i++; continue; }

			// Comments
			if ( LineComment is not null && string.CompareOrdinal( line, i, LineComment, 0, LineComment.Length ) == 0 )
			{
				Add( tokens, i, n - i, TokenKind.Comment );
				return LexState.Normal;
			}

			if ( BlockComments && c == '/' && i + 1 < n && line[i + 1] == '*' )
			{
				var end = line.IndexOf( "*/", i + 2, StringComparison.Ordinal );
				if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.Comment ); return LexState.BlockComment; }
				Add( tokens, i, end + 2 - i, TokenKind.Comment );
				i = end + 2;
				continue;
			}

			// Strings
			if ( CSharpStrings && TryStartCSharpString( line, i, out var prefix, out var verbatim, out var raw ) )
			{
				var bodyStart = i + prefix;
				if ( raw )
				{
					var end = line.IndexOf( "\"\"\"", bodyStart, StringComparison.Ordinal );
					if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.String ); return LexState.RawString; }
					Add( tokens, i, end + 3 - i, TokenKind.String );
					i = end + 3;
				}
				else if ( verbatim )
				{
					var end = FindVerbatimEnd( line, bodyStart );
					if ( end < 0 ) { Add( tokens, i, n - i, TokenKind.String ); return LexState.VerbatimString; }
					Add( tokens, i, end - i, TokenKind.String );
					i = end;
				}
				else
				{
					var end = FindQuotedEnd( line, bodyStart, '"' );
					Add( tokens, i, end - i, TokenKind.String );
					i = end;
				}
				continue;
			}

			if ( c == '"' || (c == '\'' && (CSharpStrings || SingleQuoteStrings)) )
			{
				var end = FindQuotedEnd( line, i + 1, c );
				Add( tokens, i, end - i, TokenKind.String );
				i = end;
				continue;
			}

			// Numbers
			if ( char.IsDigit( c ) || (c == '.' && i + 1 < n && char.IsDigit( line[i + 1] )) )
			{
				var start = i;
				while ( i < n && (char.IsLetterOrDigit( line[i] ) || line[i] == '.' || line[i] == '_') ) i++;
				if ( CssIdentifiers && i < n && line[i] == '%' ) i++;
				Add( tokens, start, i - start, TokenKind.Number );
				continue;
			}

			// CSS hex colours: #fff, #a0b1c2
			if ( CssIdentifiers && c == '#' && i + 1 < n && Uri.IsHexDigit( line[i + 1] ) )
			{
				var start = i++;
				while ( i < n && char.IsLetterOrDigit( line[i] ) ) i++;
				Add( tokens, start, i - start, TokenKind.Number );
				continue;
			}

			// Words
			if ( IsIdentStart( c ) || (c == '@' && i + 1 < n && IsIdentStart( line[i + 1] )) || (CssIdentifiers && (c == '$' || c == '!')) )
			{
				var start = i++;
				while ( i < n && IsIdentPart( line[i] ) ) i++;
				var word = line.Substring( start, i - start );
				var kind = ClassifyWord( line, word, i );
				if ( kind != TokenKind.Text ) Add( tokens, start, i - start, kind );
				continue;
			}

			i++;
		}

		return LexState.Normal;
	}

	private TokenKind ClassifyWord( string line, string word, int after )
	{
		if ( ControlKeywords.Contains( word ) ) return TokenKind.ControlKeyword;
		if ( Keywords.Contains( word ) ) return TokenKind.Keyword;

		if ( CssIdentifiers )
		{
			if ( word[0] == '$' ) return TokenKind.Type;

			// "color: red" is a property, but "a:hover {" is a selector
			var isFirstOnLine = string.IsNullOrWhiteSpace( line[..(after - word.Length)] );
			return isFirstOnLine && NextNonSpace( line, after ) == ':' && !line.Contains( '{' ) ? TokenKind.Property : TokenKind.Text;
		}

		if ( !GuessTypesAndMethods ) return TokenKind.Text;

		var nextChar = NextNonSpace( line, after );
		if ( nextChar == '(' ) return TokenKind.Method;

		var bare = word[0] == '@' ? word[1..] : word;
		if ( bare.Length > 0 && char.IsUpper( bare[0] ) ) return TokenKind.Type;

		return TokenKind.Text;
	}

	private bool IsIdentStart( char c ) => char.IsLetter( c ) || c == '_';
	private bool IsIdentPart( char c ) => char.IsLetterOrDigit( c ) || c == '_' || (CssIdentifiers && c == '-');

	private static char SafeChar( string s, int i ) => i < s.Length ? s[i] : '\0';

	private static char NextNonSpace( string line, int i )
	{
		while ( i < line.Length && line[i] == ' ' ) i++;
		return SafeChar( line, i );
	}

	/// <summary>
	/// Recognises the C# string openers: "  $"  @"  $@"  @$"  """  $"""  (and $$""" etc).
	/// </summary>
	private static bool TryStartCSharpString( string line, int i, out int prefixLength, out bool verbatim, out bool raw )
	{
		prefixLength = 0;
		verbatim = false;
		raw = false;

		var j = i;
		while ( j < line.Length && (line[j] == '$' || line[j] == '@') )
		{
			if ( line[j] == '@' ) verbatim = true;
			j++;
		}

		if ( j >= line.Length || line[j] != '"' )
			return false;

		// Plain "..." is handled by the generic quote branch
		if ( j == i && !(j + 2 < line.Length && line[j + 1] == '"' && line[j + 2] == '"') )
			return false;

		if ( !verbatim && j + 2 < line.Length && line[j + 1] == '"' && line[j + 2] == '"' )
		{
			raw = true;
			var k = j;
			while ( k < line.Length && line[k] == '"' ) k++;
			prefixLength = k - i;
			return true;
		}

		prefixLength = j + 1 - i;
		return true;
	}

	/// <summary>
	/// Index just past the closing quote of a verbatim string ("" is an escaped quote), or -1 if it runs off the line.
	/// </summary>
	private static int FindVerbatimEnd( string line, int i )
	{
		while ( i < line.Length )
		{
			if ( line[i] == '"' )
			{
				if ( i + 1 < line.Length && line[i + 1] == '"' ) { i += 2; continue; }
				return i + 1;
			}
			i++;
		}

		return -1;
	}

	/// <summary>
	/// Index just past the closing quote, honouring backslash escapes. Unterminated strings end at the line end.
	/// </summary>
	private static int FindQuotedEnd( string line, int i, char quote )
	{
		while ( i < line.Length )
		{
			if ( line[i] == '\\' ) { i += 2; continue; }
			if ( line[i] == quote ) return i + 1;
			i++;
		}

		return line.Length;
	}

	private static void Add( List<Token> tokens, int start, int length, TokenKind kind )
	{
		if ( length > 0 ) tokens.Add( new Token( start, length, kind ) );
	}
}