From 0338e35e758fb1fa4042fcae7a5ed4cae98a9436 Mon Sep 17 00:00:00 2001 From: hl-valdemar Date: Mon, 20 Jul 2026 08:57:12 +0200 Subject: [PATCH] checkpoint --- source/lexer.hon | 106 ---------- source/lexer/lexer.hon | 213 ++++++++++++++++++++ source/{ => lexer}/token.hon | 15 +- source/main.hon | 35 +++- source/{ => strpool}/strpool.hon | 0 std/hashmap/hashmap.hon | 4 +- std/static_string_map/static_string_map.hon | 2 +- std/std.hon | 4 +- std/testing/testing.hon | 32 ++- 9 files changed, 284 insertions(+), 127 deletions(-) delete mode 100644 source/lexer.hon create mode 100644 source/lexer/lexer.hon rename source/{ => lexer}/token.hon (60%) rename source/{ => strpool}/strpool.hon (100%) diff --git a/source/lexer.hon b/source/lexer.hon deleted file mode 100644 index 0cd3d87..0000000 --- a/source/lexer.hon +++ /dev/null @@ -1,106 +0,0 @@ -import "@std" -import "@std/mem" -import "@std/arraylist" - -scan func(tokens @mut std.ArrayList(Token), input []u8) void ! mem.AllocError { - cursor usize = 0 - while cursor < input.len { - char :: input[cursor] - - # whitespace - if char == '\n' { - try arraylist.append(tokens, Token{ kind = .newline, start = cursor }) - cursor += 1 - continue - } else if is_whitespace(char) { - cursor += 1 - continue - } - - # comments - if char == '#' { - while cursor < input.len and input[cursor] != '\n' : cursor += 1 {} - cursor += 1 - continue - } - - # identifiers and keywords - if is_alpha(char) or char == '_' { - start :: cursor - cursor += 1 - while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') { - cursor += 1 - } - try arraylist.append(tokens, Token{ kind = .ident, start = start }) - continue - } - - # integers literals - if is_digit(char) { - start :: cursor - cursor += 1 - while cursor < input.len and is_digit(input[cursor]) { - cursor += 1 - } - try arraylist.append(tokens, Token{ kind = .int, start = start }) - continue - } - - # string literals - if char == '"' { - start :: cursor - cursor += 1 - - while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 { - # ignore escaped characters - if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1 - } - - if cursor < input.len and input[cursor] == '"' { - cursor += 1 - try arraylist.append(tokens, Token{ kind = .string, start = start }) - } else { - try arraylist.append(tokens, Token{ kind = .invalid, start = start }) - } - - continue - } - - # mutable assignment - if char == '=' { - try arraylist.append(tokens, Token{ kind = .equal, start = cursor }) - cursor += 1 - continue - } - - # immutable assignment - cursor += 1 - if cursor < input.len and char == ':' and input[cursor] == ':' { - try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor }) - cursor += 1 - continue - } - - # invalid character - try arraylist.append(tokens, Token{ kind = .invalid, start = cursor }) - } - try arraylist.append(tokens, Token{ kind = .eof, start = cursor }) -} - -hide is_whitespace func(char u8) bool { - return char == ' ' or char == '\t' or char == '\n' or char == '\r' -} - -hide is_alpha func(char u8) bool { - return match char { - 'a'..'z', 'A'..'Z': true - else: false - } -} - -hide is_digit func(char u8) bool { - return match char { - '0'..'9': true - else: false - } -} diff --git a/source/lexer/lexer.hon b/source/lexer/lexer.hon new file mode 100644 index 0000000..b56f5f7 --- /dev/null +++ b/source/lexer/lexer.hon @@ -0,0 +1,213 @@ +import "@std" +import "@std/mem" +import "@std/arraylist" +#import "@std/static_string_map" + +#keywords std.StaticStringMap(TokenKind) :: static_string_map.init([ +# { "func", .func }, +# { "return", .return }, +# { "if", .if }, +# { "for", .for }, +# { "else", .else }, +#]) + +TokenIndex :: alias usize + +ScanDiagnostic :: struct { + token TokenIndex + message []u8 +} + +State :: struct { + tokens std.ArrayList(Token) + diagnostics std.ArrayList(ScanDiagnostic) +} + +init func(allocator mem.Allocator) State { + return State { + tokens = arraylist.init(allocator), + diagnostics = arraylist.init(allocator), + } +} + +deinit func(state @mut State) void { + arraylist.deinit(&state.tokens) + arraylist.deinit(&state.diagnostics) +} + +scan func(state @mut State, input []u8) void ! mem.AllocError { + tokens :: &state.tokens + diagnostics :: &state.diagnostics + + cursor usize = 0 + while cursor < input.len { + char :: input[cursor] + + # whitespace + if char == '\n' { + try arraylist.append(tokens, Token{ kind = .newline, start = cursor }) + cursor += 1 + continue + } else if is_whitespace(char) { + cursor += 1 + continue + } + + # comments + if char == '#' { + while cursor < input.len and input[cursor] != '\n' : cursor += 1 {} + cursor += 1 + continue + } + + # identifiers and keywords + if is_alpha(char) or char == '_' { + start :: cursor + cursor += 1 + while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') { + cursor += 1 + } + kind :: ident_keyword_map(input[start..cursor]) + try arraylist.append(tokens, Token{ kind = kind, start = start }) + continue + } + + # numeric literals + if is_digit(char) { + start :: cursor + has_decimal bool = false + + # scan integer part + while cursor < input.len and is_digit(input[cursor]) { + cursor += 1 + } + + # check for decimal point + if cursor < input.len and input[cursor] == '.' { + has_decimal = true + cursor += 1 + } + + # assert that decimals follow the decimal point + if has_decimal and cursor < input.len and !is_digit(input[cursor]) { + token :: tokens.items.len + try arraylist.append(tokens, Token{ kind = .invalid, start = start }) + try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "float must end with a digit" }) + continue + } + + # scan decimal part + while cursor < input.len and is_digit(input[cursor]) : cursor += 1 {} + + if has_decimal { + try arraylist.append(tokens, Token{ kind = .float, start = start }) + } else { + try arraylist.append(tokens, Token{ kind = .int, start = start }) + } + continue + } + + # string literals + if char == '"' { + start :: cursor + cursor += 1 + + while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 { + # ignore escaped characters + if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1 + } + + if cursor < input.len and input[cursor] == '"' { + cursor += 1 + try arraylist.append(tokens, Token{ kind = .string, start = start }) + } else { + try arraylist.append(tokens, Token{ kind = .invalid, start = start }) + } + + continue + } + + # mutable assignment + if char == '=' { + try arraylist.append(tokens, Token{ kind = .equal, start = cursor }) + cursor += 1 + continue + } + + # immutable assignment + if cursor + 1 < input.len and char == ':' and input[cursor + 1] == ':' { + try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor }) + cursor += 2 + continue + } + + # parentheses + if char == '(' { + try arraylist.append(tokens, Token{ kind = .open_paren, start = cursor }) + cursor += 1 + continue + } else if char == ')' { + try arraylist.append(tokens, Token{ kind = .close_paren, start = cursor }) + cursor += 1 + continue + } + + # curly braces + if char == '{' { + try arraylist.append(tokens, Token{ kind = .open_curly, start = cursor }) + cursor += 1 + continue + } else if char == '}' { + try arraylist.append(tokens, Token{ kind = .close_curly, start = cursor }) + cursor += 1 + continue + } + + # invalid character + token :: tokens.items.len + try arraylist.append(tokens, Token{ kind = .invalid, start = cursor }) + try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "invalid character" }) + } + + try arraylist.append(tokens, Token{ kind = .eof, start = cursor }) +} + +hide is_whitespace func(char u8) bool { + return char == ' ' or char == '\t' or char == '\n' or char == '\r' +} + +hide is_alpha func(char u8) bool { + return match char { + 'a'..='z', 'A'..='Z': true + else: false + } +} + +hide is_digit func(char u8) bool { + return match char { + '0'..='9': true + else: false + } +} + +# fixme: replace with a static string map once implemented +hide ident_keyword_map func(ident []u8) TokenKind { + if mem.eql(u8, ident, "if") return .if + if mem.eql(u8, ident, "else") return .else + if mem.eql(u8, ident, "for") return .for + if mem.eql(u8, ident, "while") return .while + if mem.eql(u8, ident, "func") return .func + if mem.eql(u8, ident, "return") return .return + return .ident +} + +# fixme(brolang): this function fails to infer the enum type from the return values due to the optional +#hide ident_keyword_map func(ident []u8) ?TokenKind { +# if mem.eql(u8, ident, "if") return .if +# if mem.eql(u8, ident, "else") return .else +# if mem.eql(u8, ident, "for") return .for +# if mem.eql(u8, ident, "while") return .while +# if mem.eql(u8, ident, "func") return .func +# if mem.eql(u8, ident, "return") return .return +# return null +#} diff --git a/source/token.hon b/source/lexer/token.hon similarity index 60% rename from source/token.hon rename to source/lexer/token.hon index 28833dc..85fd881 100644 --- a/source/token.hon +++ b/source/lexer/token.hon @@ -1,4 +1,11 @@ TokenKind :: enum { + if + else + for + while + func + return + ident int float @@ -7,10 +14,10 @@ TokenKind :: enum { equal double_colon - left_paren - right_paren - left_curly - right_curly + open_paren + close_paren + open_curly + close_curly newline diff --git a/source/main.hon b/source/main.hon index 5ce0b1c..02711af 100644 --- a/source/main.hon +++ b/source/main.hon @@ -1,13 +1,16 @@ -import "@std" import "@std/debug" import "@std/mem" -import "@std/arraylist" test import "@std/enums" test import "@std/arraylist" test import "@std/hashmap" test import "@std/static_string_map" +import "@source/strpool" +import "@source/lexer" + +test import "@source/strpool" + program :: `# these are immutable `x :: 32 @@ -15,21 +18,37 @@ program :: ` `# these are mutable `z u32 = 54 + ` + `# keywords + `if + `else + `for + `while + `func + `return + ` + `main func() void {} + +# cross-cutting concern, hence global singleton +strings StringPool = undefined main func() void { + strings = strpool.init(mem.c_allocator) + defer strpool.deinit(&strings) + debug.print("PROGRAM::[[\n{}\n]]\n\n", {program}) - tokens std.ArrayList(Token) = arraylist.init(mem.c_allocator) - defer arraylist.deinit(&tokens) + scan_state lexer.State = lexer.init(mem.c_allocator) + defer lexer.deinit(&scan_state) - scan(&tokens, program) catch |_| { - debug.print("failed to scan: out of memory\n", {}) + lexer.scan(&scan_state, program) catch |err| { + debug.print("failed to scan: {}\n", {err}) return } debug.print("TOKENS::[[\n", {}) - for tokens.items |token| { - debug.print("{}\n", { token.kind }) + for scan_state.tokens.items |token| { + debug.print("{}\n", {token.kind}) } debug.print("]]\n", {}) } diff --git a/source/strpool.hon b/source/strpool/strpool.hon similarity index 100% rename from source/strpool.hon rename to source/strpool/strpool.hon diff --git a/std/hashmap/hashmap.hon b/std/hashmap/hashmap.hon index 6141049..0080d05 100644 --- a/std/hashmap/hashmap.hon +++ b/std/hashmap/hashmap.hon @@ -40,8 +40,8 @@ init func( } } -# free the entries in the hash map. -# note: this operation invalidates the map. +#! free the entries in the hash map. +#! note: this operation invalidates the map. deinit func( $K, $V type, $hash_key func(key K) usize, diff --git a/std/static_string_map/static_string_map.hon b/std/static_string_map/static_string_map.hon index 7ae48a5..be282bd 100644 --- a/std/static_string_map/static_string_map.hon +++ b/std/static_string_map/static_string_map.hon @@ -64,7 +64,7 @@ init func($V type, $N usize, $entries [N]Pair(V)) StaticStringMap(V) { max_len u32 :: u32(keys[N - 1].len) len_indexes [usize(max_len) + 1]mut u32 = undefined entry_index usize = 0 - for 0..=(usize(max_len)) |length| { + for 0..=(usize(max_len)) |length| { # fixme: for casts and function calls, we should be able to omit the surrounding parentheses in the range while entry_index < N and keys[entry_index].len < length : entry_index += 1 {} len_indexes[length] = u32(entry_index) } diff --git a/std/std.hon b/std/std.hon index 5ea6cf9..16b47ba 100644 --- a/std/std.hon +++ b/std/std.hon @@ -1,9 +1,11 @@ import "io" import "enums" +import "hashmap" import "arraylist" import "static_string_map" Io :: alias io.Io -ArrayList :: alias arraylist.ArrayList EnumMap :: alias enums.EnumMap +ArrayList :: alias arraylist.ArrayList +StringHashMap :: alias hashmap.StringHashMap StaticStringMap :: alias static_string_map.StaticStringMap diff --git a/std/testing/testing.hon b/std/testing/testing.hon index 5e5e688..76188ca 100644 --- a/std/testing/testing.hon +++ b/std/testing/testing.hon @@ -13,7 +13,11 @@ SourceLocation :: struct { expect func(condition bool, location SourceLocation) void ! Error { if !condition { - debug.print("{s}:{d}:{d}: expectation failed\n", {location.file, location.line, location.column}) + debug.print("{s}:{d}:{d}: expectation failed\n", { + location.file, + location.line, + location.column, + }) return .expectation_failed } } @@ -26,20 +30,38 @@ expect_equal func($T type, expected, actual T, location SourceLocation) void ! E try expect_equal(expected_value, actual_value, location) return } - debug.print("{s}:{d}:{d}: expected an optional value, found null\n", {location.file, location.line, location.column}) + debug.print("{s}:{d}:{d}: expected an optional value, found null\n", { + location.file, + location.line, + location.column, + }) return .expectation_failed } if actual |_| { - debug.print("{s}:{d}:{d}: expected null, found an optional value\n", {location.file, location.line, location.column}) + debug.print("{s}:{d}:{d}: expected null, found an optional value\n", { + location.file, + location.line, + location.column, + }) return .expectation_failed } } .slice: if !mem.eql(expected, actual) { - debug.print("{s}:{d}:{d}: expected and actual slices differ\n", {location.file, location.line, location.column}) + debug.print("{s}:{d}:{d}: expected and actual slices differ\n", { + location.file, + location.line, + location.column, + }) return .expectation_failed } else: if expected != actual { - debug.print("{s}:{d}:{d}: expected {}, found {}\n", {location.file, location.line, location.column, expected, actual}) + debug.print("{s}:{d}:{d}: expected {}, found {}\n", { + location.file, + location.line, + location.column, + expected, + actual, + }) return .expectation_failed } }