Files
coolguy 76cb7e254c lexer: Ferro 의 렉서를 Ferro 로 쓴다
셀프호스팅에 손대기 전의 강제 함수다. 아픈 자리를 전부 건드린다: R4 아래의
토큰 구조체, 태그드 유니온, 진단 출력, 유닛 경계.

토큰은 자기가 나온 글자를 담지 않는다. R4 가 대여를 집합 저장소에서 막으므로,
어디서 시작해 얼마나 긴지를 적고 소스는 옆에서 같이 다닌다. 위치도 &mut usize
로 옆에서 다닌다 -- 슬라이스와 함께 구조체에 들어갈 수 없기 때문이다. 이것이
R11 이 말하는 모양이고, 쓸 수 있다.

  first keyword unit @1 / number 42 @3 / text "hi" @3 / arrow -> @5
  keyword 6 name 7 number 1 text 1 punct 15 / total 30

길에서 고친 것:

- binding.Type.Variant 가 안 풀렸다. 유닛 경계 이름 조회가 심볼만 보고 타입을
  보지 않았다.
- 문자열 const 전역이 빈 슬라이스로 나갔다. 포인터는 링커만 아는 수라서 바이트에
  구멍을 두고 링커가 채우게 한다.
- exec.py 가 OUTPUT 마커를 여러 개 적어도 마지막 하나만 검사했다. 고치자마자
  readfile 의 낡은 기대가 드러났다.

run.py 217/217, exec.py 27/27.
2026-08-17 07:35:21 +09:00

123 lines
4.1 KiB
Plaintext

unit scan;
import tok;
// A scanner over a byte slice. The position travels in a `&mut usize` beside
// the source rather than inside a struct with it, because a struct cannot hold
// a slice: a slice is a borrowed view and R4 keeps borrows out of aggregates.
fn is_space(c: u8) -> bool { return c == 32 or c == 9 or c == 13 or c == 10; }
fn is_digit(c: u8) -> bool { return c >= 48 and c <= 57; }
fn is_name_start(c: u8) -> bool {
if c >= 97 and c <= 122 { return true; }
if c >= 65 and c <= 90 { return true; }
return c == 95;
}
fn is_name_part(c: u8) -> bool {
return is_name_start(c) or is_digit(c);
}
const KEYWORDS: usize = 12;
fn is_keyword(word: []u8) -> bool {
if same(word, "unit") { return true; }
if same(word, "import") { return true; }
if same(word, "pub") { return true; }
if same(word, "fn") { return true; }
if same(word, "struct") { return true; }
if same(word, "enum") { return true; }
if same(word, "let") { return true; }
if same(word, "var") { return true; }
if same(word, "if") { return true; }
if same(word, "else") { return true; }
if same(word, "while") { return true; }
if same(word, "return") { return true; }
return false;
}
fn same(a: []u8, b: []u8) -> bool {
if a.n != b.n { return false; }
var i: usize = 0;
while i < a.n {
if a[i] != b[i] { return false; }
i = i + 1;
}
return true;
}
/// Step over anything that is not a token: spaces, newlines, and `//` to the
/// end of the line. `line` counts what was crossed so a token can say where it
/// came from.
fn skip_gaps(src: []u8, at: &mut usize, line: &mut usize) -> void {
while at.^ < src.n {
let c: u8 = src[at.^];
if c == 10 { line.^ = line.^ + 1; at.^ = at.^ + 1; }
else if is_space(c) { at.^ = at.^ + 1; }
else if c == 47 and at.^ + 1 < src.n and src[at.^ + 1] == 47 {
while at.^ < src.n {
if src[at.^] == 10 { break; }
at.^ = at.^ + 1;
}
}
else { break; }
}
}
pub fn next(src: []u8, at: &mut usize, line: &mut usize) -> tok.Token {
skip_gaps(src, at, line);
let start: usize = at.^;
let where: usize = line.^;
if start >= src.n {
return tok.Token{ kind: tok.Kind.End, from: start, len: 0, line: where };
}
let c: u8 = src[start];
if is_name_start(c) {
while at.^ < src.n {
if not is_name_part(src[at.^]) { break; }
at.^ = at.^ + 1;
}
let word: []u8 = src[start..at.^];
var kind: tok.Kind = tok.Kind.Name;
if is_keyword(word) { kind = tok.Kind.Keyword; }
return tok.Token{ kind: kind, from: start, len: at.^ - start,
line: where };
}
if is_digit(c) {
while at.^ < src.n {
if not is_digit(src[at.^]) { break; }
at.^ = at.^ + 1;
}
return tok.Token{ kind: tok.Kind.Number, from: start,
len: at.^ - start, line: where };
}
if c == 34 {
at.^ = at.^ + 1;
while at.^ < src.n {
if src[at.^] == 34 { break; }
if src[at.^] == 92 and at.^ + 1 < src.n { at.^ = at.^ + 1; }
at.^ = at.^ + 1;
}
if at.^ >= src.n {
return tok.Token{ kind: tok.Kind.Bad, from: start,
len: at.^ - start, line: where };
}
at.^ = at.^ + 1;
return tok.Token{ kind: tok.Kind.Text, from: start, len: at.^ - start,
line: where };
}
at.^ = at.^ + 1;
// Two-byte punctuation the language actually uses.
if at.^ < src.n {
let d: u8 = src[at.^];
if c == 45 and d == 62 { at.^ = at.^ + 1; }
else if c == 61 and d == 61 { at.^ = at.^ + 1; }
else if c == 33 and d == 61 { at.^ = at.^ + 1; }
else if c == 60 and d == 61 { at.^ = at.^ + 1; }
else if c == 62 and d == 61 { at.^ = at.^ + 1; }
else if c == 46 and d == 46 { at.^ = at.^ + 1; }
}
return tok.Token{ kind: tok.Kind.Punct, from: start, len: at.^ - start,
line: where };
}