checkpoint
This commit is contained in:
@@ -1,106 +0,0 @@
|
|||||||
import "@std"
|
|
||||||
import "@std/mem"
|
|
||||||
import "@std/arraylist"
|
|
||||||
|
|
||||||
scan func(tokens @mut std.ArrayList(Token), input []u8) void ! mem.AllocError {
|
|
||||||
cursor usize = 0
|
|
||||||
while cursor < input.len {
|
|
||||||
char :: input[cursor]
|
|
||||||
|
|
||||||
# whitespace
|
|
||||||
if char == '\n' {
|
|
||||||
try arraylist.append(tokens, Token{ kind = .newline, start = cursor })
|
|
||||||
cursor += 1
|
|
||||||
continue
|
|
||||||
} else if is_whitespace(char) {
|
|
||||||
cursor += 1
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# comments
|
|
||||||
if char == '#' {
|
|
||||||
while cursor < input.len and input[cursor] != '\n' : cursor += 1 {}
|
|
||||||
cursor += 1
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# identifiers and keywords
|
|
||||||
if is_alpha(char) or char == '_' {
|
|
||||||
start :: cursor
|
|
||||||
cursor += 1
|
|
||||||
while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') {
|
|
||||||
cursor += 1
|
|
||||||
}
|
|
||||||
try arraylist.append(tokens, Token{ kind = .ident, start = start })
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# integers literals
|
|
||||||
if is_digit(char) {
|
|
||||||
start :: cursor
|
|
||||||
cursor += 1
|
|
||||||
while cursor < input.len and is_digit(input[cursor]) {
|
|
||||||
cursor += 1
|
|
||||||
}
|
|
||||||
try arraylist.append(tokens, Token{ kind = .int, start = start })
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# string literals
|
|
||||||
if char == '"' {
|
|
||||||
start :: cursor
|
|
||||||
cursor += 1
|
|
||||||
|
|
||||||
while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 {
|
|
||||||
# ignore escaped characters
|
|
||||||
if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1
|
|
||||||
}
|
|
||||||
|
|
||||||
if cursor < input.len and input[cursor] == '"' {
|
|
||||||
cursor += 1
|
|
||||||
try arraylist.append(tokens, Token{ kind = .string, start = start })
|
|
||||||
} else {
|
|
||||||
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
|
|
||||||
}
|
|
||||||
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# mutable assignment
|
|
||||||
if char == '=' {
|
|
||||||
try arraylist.append(tokens, Token{ kind = .equal, start = cursor })
|
|
||||||
cursor += 1
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# immutable assignment
|
|
||||||
cursor += 1
|
|
||||||
if cursor < input.len and char == ':' and input[cursor] == ':' {
|
|
||||||
try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor })
|
|
||||||
cursor += 1
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
|
|
||||||
# invalid character
|
|
||||||
try arraylist.append(tokens, Token{ kind = .invalid, start = cursor })
|
|
||||||
}
|
|
||||||
try arraylist.append(tokens, Token{ kind = .eof, start = cursor })
|
|
||||||
}
|
|
||||||
|
|
||||||
hide is_whitespace func(char u8) bool {
|
|
||||||
return char == ' ' or char == '\t' or char == '\n' or char == '\r'
|
|
||||||
}
|
|
||||||
|
|
||||||
hide is_alpha func(char u8) bool {
|
|
||||||
return match char {
|
|
||||||
'a'..'z', 'A'..'Z': true
|
|
||||||
else: false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
hide is_digit func(char u8) bool {
|
|
||||||
return match char {
|
|
||||||
'0'..'9': true
|
|
||||||
else: false
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,213 @@
|
|||||||
|
import "@std"
|
||||||
|
import "@std/mem"
|
||||||
|
import "@std/arraylist"
|
||||||
|
#import "@std/static_string_map"
|
||||||
|
|
||||||
|
#keywords std.StaticStringMap(TokenKind) :: static_string_map.init([
|
||||||
|
# { "func", .func },
|
||||||
|
# { "return", .return },
|
||||||
|
# { "if", .if },
|
||||||
|
# { "for", .for },
|
||||||
|
# { "else", .else },
|
||||||
|
#])
|
||||||
|
|
||||||
|
TokenIndex :: alias usize
|
||||||
|
|
||||||
|
ScanDiagnostic :: struct {
|
||||||
|
token TokenIndex
|
||||||
|
message []u8
|
||||||
|
}
|
||||||
|
|
||||||
|
State :: struct {
|
||||||
|
tokens std.ArrayList(Token)
|
||||||
|
diagnostics std.ArrayList(ScanDiagnostic)
|
||||||
|
}
|
||||||
|
|
||||||
|
init func(allocator mem.Allocator) State {
|
||||||
|
return State {
|
||||||
|
tokens = arraylist.init(allocator),
|
||||||
|
diagnostics = arraylist.init(allocator),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
deinit func(state @mut State) void {
|
||||||
|
arraylist.deinit(&state.tokens)
|
||||||
|
arraylist.deinit(&state.diagnostics)
|
||||||
|
}
|
||||||
|
|
||||||
|
scan func(state @mut State, input []u8) void ! mem.AllocError {
|
||||||
|
tokens :: &state.tokens
|
||||||
|
diagnostics :: &state.diagnostics
|
||||||
|
|
||||||
|
cursor usize = 0
|
||||||
|
while cursor < input.len {
|
||||||
|
char :: input[cursor]
|
||||||
|
|
||||||
|
# whitespace
|
||||||
|
if char == '\n' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .newline, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
} else if is_whitespace(char) {
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# comments
|
||||||
|
if char == '#' {
|
||||||
|
while cursor < input.len and input[cursor] != '\n' : cursor += 1 {}
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# identifiers and keywords
|
||||||
|
if is_alpha(char) or char == '_' {
|
||||||
|
start :: cursor
|
||||||
|
cursor += 1
|
||||||
|
while cursor < input.len and (is_alpha(input[cursor]) or is_digit(input[cursor]) or input[cursor] == '_') {
|
||||||
|
cursor += 1
|
||||||
|
}
|
||||||
|
kind :: ident_keyword_map(input[start..cursor])
|
||||||
|
try arraylist.append(tokens, Token{ kind = kind, start = start })
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# numeric literals
|
||||||
|
if is_digit(char) {
|
||||||
|
start :: cursor
|
||||||
|
has_decimal bool = false
|
||||||
|
|
||||||
|
# scan integer part
|
||||||
|
while cursor < input.len and is_digit(input[cursor]) {
|
||||||
|
cursor += 1
|
||||||
|
}
|
||||||
|
|
||||||
|
# check for decimal point
|
||||||
|
if cursor < input.len and input[cursor] == '.' {
|
||||||
|
has_decimal = true
|
||||||
|
cursor += 1
|
||||||
|
}
|
||||||
|
|
||||||
|
# assert that decimals follow the decimal point
|
||||||
|
if has_decimal and cursor < input.len and !is_digit(input[cursor]) {
|
||||||
|
token :: tokens.items.len
|
||||||
|
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
|
||||||
|
try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "float must end with a digit" })
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# scan decimal part
|
||||||
|
while cursor < input.len and is_digit(input[cursor]) : cursor += 1 {}
|
||||||
|
|
||||||
|
if has_decimal {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .float, start = start })
|
||||||
|
} else {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .int, start = start })
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# string literals
|
||||||
|
if char == '"' {
|
||||||
|
start :: cursor
|
||||||
|
cursor += 1
|
||||||
|
|
||||||
|
while cursor < input.len and input[cursor] != '"' and input[cursor] != '\n' : cursor += 1 {
|
||||||
|
# ignore escaped characters
|
||||||
|
if (input[cursor] == '\\' and cursor + 1 < input.len) cursor += 1
|
||||||
|
}
|
||||||
|
|
||||||
|
if cursor < input.len and input[cursor] == '"' {
|
||||||
|
cursor += 1
|
||||||
|
try arraylist.append(tokens, Token{ kind = .string, start = start })
|
||||||
|
} else {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .invalid, start = start })
|
||||||
|
}
|
||||||
|
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# mutable assignment
|
||||||
|
if char == '=' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .equal, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# immutable assignment
|
||||||
|
if cursor + 1 < input.len and char == ':' and input[cursor + 1] == ':' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .double_colon, start = cursor })
|
||||||
|
cursor += 2
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# parentheses
|
||||||
|
if char == '(' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .open_paren, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
} else if char == ')' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .close_paren, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# curly braces
|
||||||
|
if char == '{' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .open_curly, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
} else if char == '}' {
|
||||||
|
try arraylist.append(tokens, Token{ kind = .close_curly, start = cursor })
|
||||||
|
cursor += 1
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
# invalid character
|
||||||
|
token :: tokens.items.len
|
||||||
|
try arraylist.append(tokens, Token{ kind = .invalid, start = cursor })
|
||||||
|
try arraylist.append(diagnostics, ScanDiagnostic{ token = token, message = "invalid character" })
|
||||||
|
}
|
||||||
|
|
||||||
|
try arraylist.append(tokens, Token{ kind = .eof, start = cursor })
|
||||||
|
}
|
||||||
|
|
||||||
|
hide is_whitespace func(char u8) bool {
|
||||||
|
return char == ' ' or char == '\t' or char == '\n' or char == '\r'
|
||||||
|
}
|
||||||
|
|
||||||
|
hide is_alpha func(char u8) bool {
|
||||||
|
return match char {
|
||||||
|
'a'..='z', 'A'..='Z': true
|
||||||
|
else: false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
hide is_digit func(char u8) bool {
|
||||||
|
return match char {
|
||||||
|
'0'..='9': true
|
||||||
|
else: false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
# fixme: replace with a static string map once implemented
|
||||||
|
hide ident_keyword_map func(ident []u8) TokenKind {
|
||||||
|
if mem.eql(u8, ident, "if") return .if
|
||||||
|
if mem.eql(u8, ident, "else") return .else
|
||||||
|
if mem.eql(u8, ident, "for") return .for
|
||||||
|
if mem.eql(u8, ident, "while") return .while
|
||||||
|
if mem.eql(u8, ident, "func") return .func
|
||||||
|
if mem.eql(u8, ident, "return") return .return
|
||||||
|
return .ident
|
||||||
|
}
|
||||||
|
|
||||||
|
# fixme(brolang): this function fails to infer the enum type from the return values due to the optional
|
||||||
|
#hide ident_keyword_map func(ident []u8) ?TokenKind {
|
||||||
|
# if mem.eql(u8, ident, "if") return .if
|
||||||
|
# if mem.eql(u8, ident, "else") return .else
|
||||||
|
# if mem.eql(u8, ident, "for") return .for
|
||||||
|
# if mem.eql(u8, ident, "while") return .while
|
||||||
|
# if mem.eql(u8, ident, "func") return .func
|
||||||
|
# if mem.eql(u8, ident, "return") return .return
|
||||||
|
# return null
|
||||||
|
#}
|
||||||
@@ -1,4 +1,11 @@
|
|||||||
TokenKind :: enum {
|
TokenKind :: enum {
|
||||||
|
if
|
||||||
|
else
|
||||||
|
for
|
||||||
|
while
|
||||||
|
func
|
||||||
|
return
|
||||||
|
|
||||||
ident
|
ident
|
||||||
int
|
int
|
||||||
float
|
float
|
||||||
@@ -7,10 +14,10 @@ TokenKind :: enum {
|
|||||||
equal
|
equal
|
||||||
double_colon
|
double_colon
|
||||||
|
|
||||||
left_paren
|
open_paren
|
||||||
right_paren
|
close_paren
|
||||||
left_curly
|
open_curly
|
||||||
right_curly
|
close_curly
|
||||||
|
|
||||||
newline
|
newline
|
||||||
|
|
||||||
+26
-7
@@ -1,13 +1,16 @@
|
|||||||
import "@std"
|
|
||||||
import "@std/debug"
|
import "@std/debug"
|
||||||
import "@std/mem"
|
import "@std/mem"
|
||||||
import "@std/arraylist"
|
|
||||||
|
|
||||||
test import "@std/enums"
|
test import "@std/enums"
|
||||||
test import "@std/arraylist"
|
test import "@std/arraylist"
|
||||||
test import "@std/hashmap"
|
test import "@std/hashmap"
|
||||||
test import "@std/static_string_map"
|
test import "@std/static_string_map"
|
||||||
|
|
||||||
|
import "@source/strpool"
|
||||||
|
import "@source/lexer"
|
||||||
|
|
||||||
|
test import "@source/strpool"
|
||||||
|
|
||||||
program ::
|
program ::
|
||||||
`# these are immutable
|
`# these are immutable
|
||||||
`x :: 32
|
`x :: 32
|
||||||
@@ -15,20 +18,36 @@ program ::
|
|||||||
`
|
`
|
||||||
`# these are mutable
|
`# these are mutable
|
||||||
`z u32 = 54
|
`z u32 = 54
|
||||||
|
`
|
||||||
|
`# keywords
|
||||||
|
`if
|
||||||
|
`else
|
||||||
|
`for
|
||||||
|
`while
|
||||||
|
`func
|
||||||
|
`return
|
||||||
|
`
|
||||||
|
`main func() void {}
|
||||||
|
|
||||||
|
# cross-cutting concern, hence global singleton
|
||||||
|
strings StringPool = undefined
|
||||||
|
|
||||||
main func() void {
|
main func() void {
|
||||||
|
strings = strpool.init(mem.c_allocator)
|
||||||
|
defer strpool.deinit(&strings)
|
||||||
|
|
||||||
debug.print("PROGRAM::[[\n{}\n]]\n\n", {program})
|
debug.print("PROGRAM::[[\n{}\n]]\n\n", {program})
|
||||||
|
|
||||||
tokens std.ArrayList(Token) = arraylist.init(mem.c_allocator)
|
scan_state lexer.State = lexer.init(mem.c_allocator)
|
||||||
defer arraylist.deinit(&tokens)
|
defer lexer.deinit(&scan_state)
|
||||||
|
|
||||||
scan(&tokens, program) catch |_| {
|
lexer.scan(&scan_state, program) catch |err| {
|
||||||
debug.print("failed to scan: out of memory\n", {})
|
debug.print("failed to scan: {}\n", {err})
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|
||||||
debug.print("TOKENS::[[\n", {})
|
debug.print("TOKENS::[[\n", {})
|
||||||
for tokens.items |token| {
|
for scan_state.tokens.items |token| {
|
||||||
debug.print("{}\n", {token.kind})
|
debug.print("{}\n", {token.kind})
|
||||||
}
|
}
|
||||||
debug.print("]]\n", {})
|
debug.print("]]\n", {})
|
||||||
|
|||||||
@@ -40,8 +40,8 @@ init func(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# free the entries in the hash map.
|
#! free the entries in the hash map.
|
||||||
# note: this operation invalidates the map.
|
#! note: this operation invalidates the map.
|
||||||
deinit func(
|
deinit func(
|
||||||
$K, $V type,
|
$K, $V type,
|
||||||
$hash_key func(key K) usize,
|
$hash_key func(key K) usize,
|
||||||
|
|||||||
@@ -64,7 +64,7 @@ init func($V type, $N usize, $entries [N]Pair(V)) StaticStringMap(V) {
|
|||||||
max_len u32 :: u32(keys[N - 1].len)
|
max_len u32 :: u32(keys[N - 1].len)
|
||||||
len_indexes [usize(max_len) + 1]mut u32 = undefined
|
len_indexes [usize(max_len) + 1]mut u32 = undefined
|
||||||
entry_index usize = 0
|
entry_index usize = 0
|
||||||
for 0..=(usize(max_len)) |length| {
|
for 0..=(usize(max_len)) |length| { # fixme: for casts and function calls, we should be able to omit the surrounding parentheses in the range
|
||||||
while entry_index < N and keys[entry_index].len < length : entry_index += 1 {}
|
while entry_index < N and keys[entry_index].len < length : entry_index += 1 {}
|
||||||
len_indexes[length] = u32(entry_index)
|
len_indexes[length] = u32(entry_index)
|
||||||
}
|
}
|
||||||
|
|||||||
+3
-1
@@ -1,9 +1,11 @@
|
|||||||
import "io"
|
import "io"
|
||||||
import "enums"
|
import "enums"
|
||||||
|
import "hashmap"
|
||||||
import "arraylist"
|
import "arraylist"
|
||||||
import "static_string_map"
|
import "static_string_map"
|
||||||
|
|
||||||
Io :: alias io.Io
|
Io :: alias io.Io
|
||||||
ArrayList :: alias arraylist.ArrayList
|
|
||||||
EnumMap :: alias enums.EnumMap
|
EnumMap :: alias enums.EnumMap
|
||||||
|
ArrayList :: alias arraylist.ArrayList
|
||||||
|
StringHashMap :: alias hashmap.StringHashMap
|
||||||
StaticStringMap :: alias static_string_map.StaticStringMap
|
StaticStringMap :: alias static_string_map.StaticStringMap
|
||||||
|
|||||||
+27
-5
@@ -13,7 +13,11 @@ SourceLocation :: struct {
|
|||||||
|
|
||||||
expect func(condition bool, location SourceLocation) void ! Error {
|
expect func(condition bool, location SourceLocation) void ! Error {
|
||||||
if !condition {
|
if !condition {
|
||||||
debug.print("{s}:{d}:{d}: expectation failed\n", {location.file, location.line, location.column})
|
debug.print("{s}:{d}:{d}: expectation failed\n", {
|
||||||
|
location.file,
|
||||||
|
location.line,
|
||||||
|
location.column,
|
||||||
|
})
|
||||||
return .expectation_failed
|
return .expectation_failed
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -26,20 +30,38 @@ expect_equal func($T type, expected, actual T, location SourceLocation) void ! E
|
|||||||
try expect_equal(expected_value, actual_value, location)
|
try expect_equal(expected_value, actual_value, location)
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
debug.print("{s}:{d}:{d}: expected an optional value, found null\n", {location.file, location.line, location.column})
|
debug.print("{s}:{d}:{d}: expected an optional value, found null\n", {
|
||||||
|
location.file,
|
||||||
|
location.line,
|
||||||
|
location.column,
|
||||||
|
})
|
||||||
return .expectation_failed
|
return .expectation_failed
|
||||||
}
|
}
|
||||||
if actual |_| {
|
if actual |_| {
|
||||||
debug.print("{s}:{d}:{d}: expected null, found an optional value\n", {location.file, location.line, location.column})
|
debug.print("{s}:{d}:{d}: expected null, found an optional value\n", {
|
||||||
|
location.file,
|
||||||
|
location.line,
|
||||||
|
location.column,
|
||||||
|
})
|
||||||
return .expectation_failed
|
return .expectation_failed
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
.slice: if !mem.eql(expected, actual) {
|
.slice: if !mem.eql(expected, actual) {
|
||||||
debug.print("{s}:{d}:{d}: expected and actual slices differ\n", {location.file, location.line, location.column})
|
debug.print("{s}:{d}:{d}: expected and actual slices differ\n", {
|
||||||
|
location.file,
|
||||||
|
location.line,
|
||||||
|
location.column,
|
||||||
|
})
|
||||||
return .expectation_failed
|
return .expectation_failed
|
||||||
}
|
}
|
||||||
else: if expected != actual {
|
else: if expected != actual {
|
||||||
debug.print("{s}:{d}:{d}: expected {}, found {}\n", {location.file, location.line, location.column, expected, actual})
|
debug.print("{s}:{d}:{d}: expected {}, found {}\n", {
|
||||||
|
location.file,
|
||||||
|
location.line,
|
||||||
|
location.column,
|
||||||
|
expected,
|
||||||
|
actual,
|
||||||
|
})
|
||||||
return .expectation_failed
|
return .expectation_failed
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user