multi-line string literals
This commit is contained in:
@@ -253,25 +253,29 @@
|
||||
- parser-only change (`parse_branch_body` in `compiler/parser/parser.odin`): a brace-less
|
||||
body is just a 1-element statement slice, so the checker and codegen are unchanged
|
||||
|
||||
17. add slice-by-range
|
||||
- allow the use of a range in slice expressions:
|
||||
```
|
||||
excl_range range :: 0..10
|
||||
some_arr[excl_range] # slice by named exclusive range
|
||||
|
||||
incl_range range :: 0..=10
|
||||
some_arr[incl_range] # slice by named inclusive range
|
||||
```
|
||||
17. multi-line strings (implemented; see below)
|
||||
- a multi-line string is an ordinary string literal under the hood: it lowers to the
|
||||
same `.String` expr / `@[N;0]u8` type, so the checker, lowering, and codegen are
|
||||
unchanged. only the lexer and parser change.
|
||||
- the lexer (`compiler/lexer/lexer.odin`, `` case '`' ``) collapses consecutive
|
||||
backtick-marked lines into one `Multiline_String` token; the trailing newline after
|
||||
the last line stays a `.Newline` so it terminates the statement normally
|
||||
- the parser (`decode_multiline_string` in `compiler/parser/parser.odin`) strips each
|
||||
line's leading indentation and `` ` ``, takes the rest of the line raw (no escapes),
|
||||
and joins lines with an implicit `\n` (no leading/trailing newline); an empty
|
||||
`` ` `` yields a blank line
|
||||
- the value may sit on the line after `=`/`::` (the existing post-operator
|
||||
`skip_newlines` already allows this)
|
||||
- `++` concatenation (the spec's "mixing" examples) is a separate, unimplemented
|
||||
operator and is out of scope here
|
||||
|
||||
18. add `defer` statement (inspired by zig)
|
||||
|
||||
19. multi-line strings (see below)
|
||||
19. unions and tagged unions
|
||||
|
||||
20. unions and tagged unions
|
||||
20. match statements with tagged unions payload unwrapping
|
||||
|
||||
21. match statements with tagged unions payload unwrapping
|
||||
|
||||
22. dynamic heap allocation
|
||||
21. dynamic heap allocation
|
||||
- see below for direction
|
||||
- notes below are too big in scope for a first pass and the language is not mature enough to support it yet
|
||||
- this first pass should focus on just basic heap allocation, so we have something to work with
|
||||
|
||||
@@ -277,6 +277,34 @@ lex :: proc(
|
||||
)
|
||||
append_token(&stream, source_file, .Invalid, start, cursor, diagnostic=id)
|
||||
}
|
||||
case '`':
|
||||
// Multi-line string: each line is marked with a leading '`'; raw
|
||||
// content (no escapes) runs to end of line. Consecutive backtick
|
||||
// lines collapse into one token; the trailing '\n' is left to
|
||||
// become the statement-terminating Newline.
|
||||
start := cursor
|
||||
content_end := cursor
|
||||
for {
|
||||
cursor += 1 // skip backtick
|
||||
for cursor < len(bytes) && bytes[cursor] != '\n' {
|
||||
cursor += 1
|
||||
}
|
||||
content_end = cursor
|
||||
look := cursor
|
||||
if look < len(bytes) && bytes[look] == '\n' {
|
||||
look += 1
|
||||
}
|
||||
for look < len(bytes) && (bytes[look] == ' ' || bytes[look] == '\t') {
|
||||
look += 1
|
||||
}
|
||||
if look < len(bytes) && bytes[look] == '`' {
|
||||
cursor = look
|
||||
continue
|
||||
}
|
||||
break
|
||||
}
|
||||
cursor = content_end
|
||||
append_token(&stream, source_file, .Multiline_String, start, content_end)
|
||||
case ';':
|
||||
append_token(&stream, source_file, .Semicolon, cursor, cursor+1)
|
||||
cursor += 1
|
||||
|
||||
@@ -560,9 +560,9 @@ parse_primary :: proc(parser: ^Parser, nesting: int) -> ast.Expr_Id {
|
||||
right=ast.INVALID_EXPR,
|
||||
diagnostic=source.INVALID_DIAGNOSTIC,
|
||||
})
|
||||
case .String:
|
||||
case .String, .Multiline_String:
|
||||
advance(parser)
|
||||
value := decode_import_path(parser, tok)
|
||||
value := decode_import_path(parser, tok) if tok.kind == .String else decode_multiline_string(parser, tok)
|
||||
id := u64(len(parser.module.strings))
|
||||
append(&parser.module.strings, value)
|
||||
return add_expr(parser, ast.Expr{
|
||||
@@ -1690,6 +1690,33 @@ decode_import_path :: proc(parser: ^Parser, tok: token.Token) -> string {
|
||||
return fmt.aprintf("%s", strings.to_string(builder), allocator=parser.module.allocator)
|
||||
}
|
||||
|
||||
decode_multiline_string :: proc(parser: ^Parser, tok: token.Token) -> string {
|
||||
text := token_text(parser, tok)
|
||||
builder := strings.builder_make(parser.module.allocator)
|
||||
defer strings.builder_destroy(&builder)
|
||||
first := true
|
||||
index := 0
|
||||
for index < len(text) {
|
||||
for index < len(text) && (text[index] == ' ' || text[index] == '\t') {
|
||||
index += 1
|
||||
}
|
||||
if index >= len(text) || text[index] != '`' {
|
||||
break
|
||||
}
|
||||
index += 1 // skip backtick
|
||||
if !first {
|
||||
strings.write_byte(&builder, '\n')
|
||||
}
|
||||
first = false
|
||||
for index < len(text) && text[index] != '\n' {
|
||||
strings.write_byte(&builder, text[index])
|
||||
index += 1
|
||||
}
|
||||
index += 1 // skip newline (terminates this line)
|
||||
}
|
||||
return fmt.aprintf("%s", strings.to_string(builder), allocator=parser.module.allocator)
|
||||
}
|
||||
|
||||
parse_import :: proc(parser: ^Parser, alias: token.Token, start: token.Token) {
|
||||
skip_newlines(parser)
|
||||
path_token := current(parser)
|
||||
|
||||
@@ -11,6 +11,7 @@ Kind :: enum u8 {
|
||||
Integer,
|
||||
Float,
|
||||
String,
|
||||
Multiline_String,
|
||||
Character,
|
||||
Underscore,
|
||||
Colon,
|
||||
|
||||
@@ -969,6 +969,47 @@ main :: func() void {
|
||||
testing.expect(t, strings.contains(llvm_text, "declare i32 @take_c_sentinel(ptr)"))
|
||||
}
|
||||
|
||||
@(test)
|
||||
multiline_strings_join_lines_and_strip_indentation :: proc(t: ^testing.T) {
|
||||
// Source written double-quoted (not a raw `...` literal) because the
|
||||
// backtick is the multi-line string marker. Covers basic join, the
|
||||
// trailing-newline form, a blank line in the middle, and value-on-next-line.
|
||||
text := "main :: func() void {\n" +
|
||||
"\tbasic ::\n\t\t`a\n\t\t`b\n" +
|
||||
"\ttrailing ::\n\t\t`hello\n\t\t`world\n\t\t`\n" +
|
||||
"\tgapped ::\n\t\t`x\n\t\t`\n\t\t`y\n" +
|
||||
"\t_ = basic\n\t_ = trailing\n\t_ = gapped\n}\n"
|
||||
source_file := source.Source{path="test.bro", text=text}
|
||||
diagnostics := source.init_diagnostics(&source_file)
|
||||
defer source.destroy_diagnostics(&diagnostics)
|
||||
symbols := symbol.init_table()
|
||||
defer symbol.destroy_table(&symbols)
|
||||
stream := lexer.lex(&source_file, &diagnostics, &symbols)
|
||||
defer delete(stream.items)
|
||||
ast_module := parser.parse(&stream, &source_file, &diagnostics)
|
||||
defer ast.destroy_module(&ast_module)
|
||||
hir_module := checker.check(&ast_module, &diagnostics, &symbols)
|
||||
defer hir.destroy_module(&hir_module)
|
||||
|
||||
testing.expect_value(t, len(diagnostics.items), 0)
|
||||
testing.expect_value(t, len(ast_module.strings), 3)
|
||||
testing.expect_value(t, ast_module.strings[0], "a\nb")
|
||||
testing.expect_value(t, ast_module.strings[1], "hello\nworld\n")
|
||||
testing.expect_value(t, ast_module.strings[2], "x\n\ny")
|
||||
|
||||
// A multi-line string is an ordinary string literal: @[N;0]u8.
|
||||
string_type := types.INVALID
|
||||
for expr in hir_module.exprs {
|
||||
if expr.kind == .String {
|
||||
string_type = expr.type
|
||||
break
|
||||
}
|
||||
}
|
||||
pointer, array, string_ok := types.array_pointer(string_type, &hir_module.types)
|
||||
testing.expect(t, string_ok && !pointer.mutable && array.child == types.U8 &&
|
||||
array.has_sentinel && array.sentinel == 0)
|
||||
}
|
||||
|
||||
@(test)
|
||||
array_pointer_and_c_string_coercion_restrictions_are_diagnosed :: proc(t: ^testing.T) {
|
||||
text := `take_c_string :: c_func(value *c_char) c_int
|
||||
|
||||
Reference in New Issue
Block a user