셀프호스팅에 손대기 전의 강제 함수다. 아픈 자리를 전부 건드린다: R4 아래의 토큰 구조체, 태그드 유니온, 진단 출력, 유닛 경계. 토큰은 자기가 나온 글자를 담지 않는다. R4 가 대여를 집합 저장소에서 막으므로, 어디서 시작해 얼마나 긴지를 적고 소스는 옆에서 같이 다닌다. 위치도 &mut usize 로 옆에서 다닌다 -- 슬라이스와 함께 구조체에 들어갈 수 없기 때문이다. 이것이 R11 이 말하는 모양이고, 쓸 수 있다. first keyword unit @1 / number 42 @3 / text "hi" @3 / arrow -> @5 keyword 6 name 7 number 1 text 1 punct 15 / total 30 길에서 고친 것: - binding.Type.Variant 가 안 풀렸다. 유닛 경계 이름 조회가 심볼만 보고 타입을 보지 않았다. - 문자열 const 전역이 빈 슬라이스로 나갔다. 포인터는 링커만 아는 수라서 바이트에 구멍을 두고 링커가 채우게 한다. - exec.py 가 OUTPUT 마커를 여러 개 적어도 마지막 하나만 검사했다. 고치자마자 readfile 의 낡은 기대가 드러났다. run.py 217/217, exec.py 27/27.
123 lines
4.1 KiB
Plaintext
123 lines
4.1 KiB
Plaintext
unit scan;
|
|
import tok;
|
|
|
|
// A scanner over a byte slice. The position travels in a `&mut usize` beside
|
|
// the source rather than inside a struct with it, because a struct cannot hold
|
|
// a slice: a slice is a borrowed view and R4 keeps borrows out of aggregates.
|
|
|
|
fn is_space(c: u8) -> bool { return c == 32 or c == 9 or c == 13 or c == 10; }
|
|
fn is_digit(c: u8) -> bool { return c >= 48 and c <= 57; }
|
|
|
|
fn is_name_start(c: u8) -> bool {
|
|
if c >= 97 and c <= 122 { return true; }
|
|
if c >= 65 and c <= 90 { return true; }
|
|
return c == 95;
|
|
}
|
|
|
|
fn is_name_part(c: u8) -> bool {
|
|
return is_name_start(c) or is_digit(c);
|
|
}
|
|
|
|
const KEYWORDS: usize = 12;
|
|
|
|
fn is_keyword(word: []u8) -> bool {
|
|
if same(word, "unit") { return true; }
|
|
if same(word, "import") { return true; }
|
|
if same(word, "pub") { return true; }
|
|
if same(word, "fn") { return true; }
|
|
if same(word, "struct") { return true; }
|
|
if same(word, "enum") { return true; }
|
|
if same(word, "let") { return true; }
|
|
if same(word, "var") { return true; }
|
|
if same(word, "if") { return true; }
|
|
if same(word, "else") { return true; }
|
|
if same(word, "while") { return true; }
|
|
if same(word, "return") { return true; }
|
|
return false;
|
|
}
|
|
|
|
fn same(a: []u8, b: []u8) -> bool {
|
|
if a.n != b.n { return false; }
|
|
var i: usize = 0;
|
|
while i < a.n {
|
|
if a[i] != b[i] { return false; }
|
|
i = i + 1;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
/// Step over anything that is not a token: spaces, newlines, and `//` to the
|
|
/// end of the line. `line` counts what was crossed so a token can say where it
|
|
/// came from.
|
|
fn skip_gaps(src: []u8, at: &mut usize, line: &mut usize) -> void {
|
|
while at.^ < src.n {
|
|
let c: u8 = src[at.^];
|
|
if c == 10 { line.^ = line.^ + 1; at.^ = at.^ + 1; }
|
|
else if is_space(c) { at.^ = at.^ + 1; }
|
|
else if c == 47 and at.^ + 1 < src.n and src[at.^ + 1] == 47 {
|
|
while at.^ < src.n {
|
|
if src[at.^] == 10 { break; }
|
|
at.^ = at.^ + 1;
|
|
}
|
|
}
|
|
else { break; }
|
|
}
|
|
}
|
|
|
|
pub fn next(src: []u8, at: &mut usize, line: &mut usize) -> tok.Token {
|
|
skip_gaps(src, at, line);
|
|
let start: usize = at.^;
|
|
let where: usize = line.^;
|
|
if start >= src.n {
|
|
return tok.Token{ kind: tok.Kind.End, from: start, len: 0, line: where };
|
|
}
|
|
let c: u8 = src[start];
|
|
if is_name_start(c) {
|
|
while at.^ < src.n {
|
|
if not is_name_part(src[at.^]) { break; }
|
|
at.^ = at.^ + 1;
|
|
}
|
|
let word: []u8 = src[start..at.^];
|
|
var kind: tok.Kind = tok.Kind.Name;
|
|
if is_keyword(word) { kind = tok.Kind.Keyword; }
|
|
return tok.Token{ kind: kind, from: start, len: at.^ - start,
|
|
line: where };
|
|
}
|
|
if is_digit(c) {
|
|
while at.^ < src.n {
|
|
if not is_digit(src[at.^]) { break; }
|
|
at.^ = at.^ + 1;
|
|
}
|
|
return tok.Token{ kind: tok.Kind.Number, from: start,
|
|
len: at.^ - start, line: where };
|
|
}
|
|
if c == 34 {
|
|
at.^ = at.^ + 1;
|
|
while at.^ < src.n {
|
|
if src[at.^] == 34 { break; }
|
|
if src[at.^] == 92 and at.^ + 1 < src.n { at.^ = at.^ + 1; }
|
|
at.^ = at.^ + 1;
|
|
}
|
|
if at.^ >= src.n {
|
|
return tok.Token{ kind: tok.Kind.Bad, from: start,
|
|
len: at.^ - start, line: where };
|
|
}
|
|
at.^ = at.^ + 1;
|
|
return tok.Token{ kind: tok.Kind.Text, from: start, len: at.^ - start,
|
|
line: where };
|
|
}
|
|
at.^ = at.^ + 1;
|
|
// Two-byte punctuation the language actually uses.
|
|
if at.^ < src.n {
|
|
let d: u8 = src[at.^];
|
|
if c == 45 and d == 62 { at.^ = at.^ + 1; }
|
|
else if c == 61 and d == 61 { at.^ = at.^ + 1; }
|
|
else if c == 33 and d == 61 { at.^ = at.^ + 1; }
|
|
else if c == 60 and d == 61 { at.^ = at.^ + 1; }
|
|
else if c == 62 and d == 61 { at.^ = at.^ + 1; }
|
|
else if c == 46 and d == 46 { at.^ = at.^ + 1; }
|
|
}
|
|
return tok.Token{ kind: tok.Kind.Punct, from: start, len: at.^ - start,
|
|
line: where };
|
|
}
|