checkpoint

This commit is contained in:
2026-07-20 08:57:12 +02:00
parent a4e6f96951
commit 0338e35e75
9 changed files with 284 additions and 127 deletions
-106
View File
@@ -1,106 +0,0 @@
import "@std"
import "@std/mem"
import "@std/arraylist"
scan func(tokens @mut std.ArrayList(Token), input []u8) void ! mem.AllocError {
cursor usize = 0
while cursor < input.len {
char :: input[cursor]
# whitespace
if char == '\n' {
try arraylist.append(tokens, Token{ kind = .newline, start = cursor })
cursor += 1
continue
} else if is_whitespace(char) {
cursor += 1
continue
}
# comments
if char == '#' {
while cursor < input.len and input[cursor] != '\n' : cursor += 1 {}
cursor += 1
continue
}
# identifiers and keywords
if is_alpha(char) or char == '_' {
start :: cursor
cursor += 1
while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') {
cursor += 1
}
try arraylist.append(tokens, Token{ kind = .ident, start = start })
continue
}
# integers literals
if is_digit(char) {
start :: cursor
cursor += 1
while cursor < input.len and is_digit(input[cursor]) {
cursor += 1
}
try arraylist.append(tokens, Token{ kind = .int, start = start })
continue
}
# string literals
if char == '"' {
start :: cursor
cursor += 1
while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 {
# ignore escaped characters
if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1
}
if cursor < input.len and input[cursor] == '"' {
cursor += 1
try arraylist.append(tokens, Token{ kind = .string, start = start })
} else {
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
}
continue
}
# mutable assignment
if char == '=' {
try arraylist.append(tokens, Token{ kind = .equal, start = cursor })
cursor += 1
continue
}
# immutable assignment
cursor += 1
if cursor < input.len and char == ':' and input[cursor] == ':' {
try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor })
cursor += 1
continue
}
# invalid character
try arraylist.append(tokens, Token{ kind = .invalid, start = cursor })
}
try arraylist.append(tokens, Token{ kind = .eof, start = cursor })
}
hide is_whitespace func(char u8) bool {
return char == ' ' or char == '\t' or char == '\n' or char == '\r'
}
hide is_alpha func(char u8) bool {
return match char {
'a'..'z', 'A'..'Z': true
else: false
}
}
hide is_digit func(char u8) bool {
return match char {
'0'..'9': true
else: false
}
}
+213
View File
@@ -0,0 +1,213 @@
import "@std"
import "@std/mem"
import "@std/arraylist"
#import "@std/static_string_map"
#keywords std.StaticStringMap(TokenKind) :: static_string_map.init([
# { "func", .func },
# { "return", .return },
# { "if", .if },
# { "for", .for },
# { "else", .else },
#])
TokenIndex :: alias usize
ScanDiagnostic :: struct {
token TokenIndex
message []u8
}
State :: struct {
tokens std.ArrayList(Token)
diagnostics std.ArrayList(ScanDiagnostic)
}
init func(allocator mem.Allocator) State {
return State {
tokens = arraylist.init(allocator),
diagnostics = arraylist.init(allocator),
}
}
deinit func(state @mut State) void {
arraylist.deinit(&state.tokens)
arraylist.deinit(&state.diagnostics)
}
scan func(state @mut State, input []u8) void ! mem.AllocError {
tokens :: &state.tokens
diagnostics :: &state.diagnostics
cursor usize = 0
while cursor < input.len {
char :: input[cursor]
# whitespace
if char == '\n' {
try arraylist.append(tokens, Token{ kind = .newline, start = cursor })
cursor += 1
continue
} else if is_whitespace(char) {
cursor += 1
continue
}
# comments
if char == '#' {
while cursor < input.len and input[cursor] != '\n' : cursor += 1 {}
cursor += 1
continue
}
# identifiers and keywords
if is_alpha(char) or char == '_' {
start :: cursor
cursor += 1
while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') {
cursor += 1
}
kind :: ident_keyword_map(input[start..cursor])
try arraylist.append(tokens, Token{ kind = kind, start = start })
continue
}
# numeric literals
if is_digit(char) {
start :: cursor
has_decimal bool = false
# scan integer part
while cursor < input.len and is_digit(input[cursor]) {
cursor += 1
}
# check for decimal point
if cursor < input.len and input[cursor] == '.' {
has_decimal = true
cursor += 1
}
# assert that decimals follow the decimal point
if has_decimal and cursor < input.len and !is_digit(input[cursor]) {
token :: tokens.items.len
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "float must end with a digit" })
continue
}
# scan decimal part
while cursor < input.len and is_digit(input[cursor]) : cursor += 1 {}
if has_decimal {
try arraylist.append(tokens, Token{ kind = .float, start = start })
} else {
try arraylist.append(tokens, Token{ kind = .int, start = start })
}
continue
}
# string literals
if char == '"' {
start :: cursor
cursor += 1
while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 {
# ignore escaped characters
if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1
}
if cursor < input.len and input[cursor] == '"' {
cursor += 1
try arraylist.append(tokens, Token{ kind = .string, start = start })
} else {
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
}
continue
}
# mutable assignment
if char == '=' {
try arraylist.append(tokens, Token{ kind = .equal, start = cursor })
cursor += 1
continue
}
# immutable assignment
if cursor + 1 < input.len and char == ':' and input[cursor + 1] == ':' {
try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor })
cursor += 2
continue
}
# parentheses
if char == '(' {
try arraylist.append(tokens, Token{ kind = .open_paren, start = cursor })
cursor += 1
continue
} else if char == ')' {
try arraylist.append(tokens, Token{ kind = .close_paren, start = cursor })
cursor += 1
continue
}
# curly braces
if char == '{' {
try arraylist.append(tokens, Token{ kind = .open_curly, start = cursor })
cursor += 1
continue
} else if char == '}' {
try arraylist.append(tokens, Token{ kind = .close_curly, start = cursor })
cursor += 1
continue
}
# invalid character
token :: tokens.items.len
try arraylist.append(tokens, Token{ kind = .invalid, start = cursor })
try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "invalid character" })
}
try arraylist.append(tokens, Token{ kind = .eof, start = cursor })
}
hide is_whitespace func(char u8) bool {
return char == ' ' or char == '\t' or char == '\n' or char == '\r'
}
hide is_alpha func(char u8) bool {
return match char {
'a'..='z', 'A'..='Z': true
else: false
}
}
hide is_digit func(char u8) bool {
return match char {
'0'..='9': true
else: false
}
}
# fixme: replace with a static string map once implemented
hide ident_keyword_map func(ident []u8) TokenKind {
if mem.eql(u8, ident, "if") return .if
if mem.eql(u8, ident, "else") return .else
if mem.eql(u8, ident, "for") return .for
if mem.eql(u8, ident, "while") return .while
if mem.eql(u8, ident, "func") return .func
if mem.eql(u8, ident, "return") return .return
return .ident
}
# fixme(brolang): this function fails to infer the enum type from the return values due to the optional
#hide ident_keyword_map func(ident []u8) ?TokenKind {
# if mem.eql(u8, ident, "if") return .if
# if mem.eql(u8, ident, "else") return .else
# if mem.eql(u8, ident, "for") return .for
# if mem.eql(u8, ident, "while") return .while
# if mem.eql(u8, ident, "func") return .func
# if mem.eql(u8, ident, "return") return .return
# return null
#}
+11 -4
View File
@@ -1,4 +1,11 @@
TokenKind :: enum { TokenKind :: enum {
if
else
for
while
func
return
ident ident
int int
float float
@@ -7,10 +14,10 @@ TokenKind :: enum {
equal equal
double_colon double_colon
left_paren open_paren
right_paren close_paren
left_curly open_curly
right_curly close_curly
newline newline
+27 -8
View File
@@ -1,13 +1,16 @@
import "@std"
import "@std/debug" import "@std/debug"
import "@std/mem" import "@std/mem"
import "@std/arraylist"
test import "@std/enums" test import "@std/enums"
test import "@std/arraylist" test import "@std/arraylist"
test import "@std/hashmap" test import "@std/hashmap"
test import "@std/static_string_map" test import "@std/static_string_map"
import "@source/strpool"
import "@source/lexer"
test import "@source/strpool"
program :: program ::
`# these are immutable `# these are immutable
`x :: 32 `x :: 32
@@ -15,21 +18,37 @@ program ::
` `
`# these are mutable `# these are mutable
`z u32 = 54 `z u32 = 54
`
`# keywords
`if
`else
`for
`while
`func
`return
`
`main func() void {}
# cross-cutting concern, hence global singleton
strings StringPool = undefined
main func() void { main func() void {
strings = strpool.init(mem.c_allocator)
defer strpool.deinit(&strings)
debug.print("PROGRAM::[[\n{}\n]]\n\n", {program}) debug.print("PROGRAM::[[\n{}\n]]\n\n", {program})
tokens std.ArrayList(Token) = arraylist.init(mem.c_allocator) scan_state lexer.State = lexer.init(mem.c_allocator)
defer arraylist.deinit(&tokens) defer lexer.deinit(&scan_state)
scan(&tokens, program) catch |_| { lexer.scan(&scan_state, program) catch |err| {
debug.print("failed to scan: out of memory\n", {}) debug.print("failed to scan: {}\n", {err})
return return
} }
debug.print("TOKENS::[[\n", {}) debug.print("TOKENS::[[\n", {})
for tokens.items |token| { for scan_state.tokens.items |token| {
debug.print("{}\n", { token.kind }) debug.print("{}\n", {token.kind})
} }
debug.print("]]\n", {}) debug.print("]]\n", {})
} }
+2 -2
View File
@@ -40,8 +40,8 @@ init func(
} }
} }
# free the entries in the hash map. #! free the entries in the hash map.
# note: this operation invalidates the map. #! note: this operation invalidates the map.
deinit func( deinit func(
$K, $V type, $K, $V type,
$hash_key func(key K) usize, $hash_key func(key K) usize,
+1 -1
View File
@@ -64,7 +64,7 @@ init func($V type, $N usize, $entries [N]Pair(V)) StaticStringMap(V) {
max_len u32 :: u32(keys[N - 1].len) max_len u32 :: u32(keys[N - 1].len)
len_indexes [usize(max_len) + 1]mut u32 = undefined len_indexes [usize(max_len) + 1]mut u32 = undefined
entry_index usize = 0 entry_index usize = 0
for 0..=(usize(max_len)) |length| { for 0..=(usize(max_len)) |length| { # fixme: for casts and function calls, we should be able to omit the surrounding parentheses in the range
while entry_index < N and keys[entry_index].len < length : entry_index += 1 {} while entry_index < N and keys[entry_index].len < length : entry_index += 1 {}
len_indexes[length] = u32(entry_index) len_indexes[length] = u32(entry_index)
} }
+3 -1
View File
@@ -1,9 +1,11 @@
import "io" import "io"
import "enums" import "enums"
import "hashmap"
import "arraylist" import "arraylist"
import "static_string_map" import "static_string_map"
Io :: alias io.Io Io :: alias io.Io
ArrayList :: alias arraylist.ArrayList
EnumMap :: alias enums.EnumMap EnumMap :: alias enums.EnumMap
ArrayList :: alias arraylist.ArrayList
StringHashMap :: alias hashmap.StringHashMap
StaticStringMap :: alias static_string_map.StaticStringMap StaticStringMap :: alias static_string_map.StaticStringMap
+27 -5
View File
@@ -13,7 +13,11 @@ SourceLocation :: struct {
expect func(condition bool, location SourceLocation) void ! Error { expect func(condition bool, location SourceLocation) void ! Error {
if !condition { if !condition {
debug.print("{s}:{d}:{d}: expectation failed\n", {location.file, location.line, location.column}) debug.print("{s}:{d}:{d}: expectation failed\n", {
location.file,
location.line,
location.column,
})
return .expectation_failed return .expectation_failed
} }
} }
@@ -26,20 +30,38 @@ expect_equal func($T type, expected, actual T, location SourceLocation) void ! E
try expect_equal(expected_value, actual_value, location) try expect_equal(expected_value, actual_value, location)
return return
} }
debug.print("{s}:{d}:{d}: expected an optional value, found null\n", {location.file, location.line, location.column}) debug.print("{s}:{d}:{d}: expected an optional value, found null\n", {
location.file,
location.line,
location.column,
})
return .expectation_failed return .expectation_failed
} }
if actual |_| { if actual |_| {
debug.print("{s}:{d}:{d}: expected null, found an optional value\n", {location.file, location.line, location.column}) debug.print("{s}:{d}:{d}: expected null, found an optional value\n", {
location.file,
location.line,
location.column,
})
return .expectation_failed return .expectation_failed
} }
} }
.slice: if !mem.eql(expected, actual) { .slice: if !mem.eql(expected, actual) {
debug.print("{s}:{d}:{d}: expected and actual slices differ\n", {location.file, location.line, location.column}) debug.print("{s}:{d}:{d}: expected and actual slices differ\n", {
location.file,
location.line,
location.column,
})
return .expectation_failed return .expectation_failed
} }
else: if expected != actual { else: if expected != actual {
debug.print("{s}:{d}:{d}: expected {}, found {}\n", {location.file, location.line, location.column, expected, actual}) debug.print("{s}:{d}:{d}: expected {}, found {}\n", {
location.file,
location.line,
location.column,
expected,
actual,
})
return .expectation_failed return .expectation_failed
} }
} }