lexer: Ferro 의 렉서를 Ferro 로 쓴다

셀프호스팅에 손대기 전의 강제 함수다. 아픈 자리를 전부 건드린다: R4 아래의
토큰 구조체, 태그드 유니온, 진단 출력, 유닛 경계.

토큰은 자기가 나온 글자를 담지 않는다. R4 가 대여를 집합 저장소에서 막으므로,
어디서 시작해 얼마나 긴지를 적고 소스는 옆에서 같이 다닌다. 위치도 &mut usize
로 옆에서 다닌다 -- 슬라이스와 함께 구조체에 들어갈 수 없기 때문이다. 이것이
R11 이 말하는 모양이고, 쓸 수 있다.

  first keyword unit @1 / number 42 @3 / text "hi" @3 / arrow -> @5
  keyword 6 name 7 number 1 text 1 punct 15 / total 30

길에서 고친 것:

- binding.Type.Variant 가 안 풀렸다. 유닛 경계 이름 조회가 심볼만 보고 타입을
  보지 않았다.
- 문자열 const 전역이 빈 슬라이스로 나갔다. 포인터는 링커만 아는 수라서 바이트에
  구멍을 두고 링커가 채우게 한다.
- exec.py 가 OUTPUT 마커를 여러 개 적어도 마지막 하나만 검사했다. 고치자마자
  readfile 의 낡은 기대가 드러났다.

run.py 217/217, exec.py 27/27.
This commit is contained in:
2026-08-17 07:35:21 +09:00
parent 4fe0073365
commit 76cb7e254c
10 changed files with 337 additions and 12 deletions
+61
View File
@@ -0,0 +1,61 @@
// EXIT:0
// OUTPUT:first keyword unit @1
// OUTPUT:number 42 @3
// OUTPUT:text "hi" @3
// OUTPUT:arrow -> @5
// OUTPUT:keyword 6 name 7 number 1 text 1 punct 15
// OUTPUT:total 30
unit main;
import std.io;
import tok;
import scan;
// The Ferro lexer, written in Ferro. This is the shape a self-hosted `fec`
// would take: read a source, hand back tokens, say where each came from.
const SOURCE: str = "unit demo;\n\nfn answer() { let n = 42; let s = \"hi\"; }\n// a comment\nfn arrow() -> i32 { return n; }\n";
fn main() -> i32 {
var at: usize = 0;
var line: usize = 1;
var keywords: usize = 0;
var names: usize = 0;
var numbers: usize = 0;
var texts: usize = 0;
var puncts: usize = 0;
var total: usize = 0;
var first: bool = true;
while true {
let t: tok.Token = scan.next(SOURCE, &mut at, &mut line);
if t.kind == tok.Kind.End { break; }
total = total + 1;
if first {
@print("first {} {} @{}\n", tok.name_of(t.kind),
tok.text(SOURCE, t), t.line);
first = false;
}
match t.kind {
Keyword => { keywords = keywords + 1; }
Name => { names = names + 1; }
Number => {
numbers = numbers + 1;
@print("number {} @{}\n", tok.text(SOURCE, t), t.line);
}
Text => {
texts = texts + 1;
@print("text {} @{}\n", tok.text(SOURCE, t), t.line);
}
Punct => {
puncts = puncts + 1;
if t.len == 2 {
@print("arrow {} @{}\n", tok.text(SOURCE, t), t.line);
}
}
_ => { @print("unexpected {}\n", tok.name_of(t.kind)); }
}
}
@print("keyword {} name {} number {} text {} punct {}\n",
keywords, names, numbers, texts, puncts);
@print("total {}\n", total);
return 0;
}
+122
View File
@@ -0,0 +1,122 @@
unit scan;
import tok;
// A scanner over a byte slice. The position travels in a `&mut usize` beside
// the source rather than inside a struct with it, because a struct cannot hold
// a slice: a slice is a borrowed view and R4 keeps borrows out of aggregates.
fn is_space(c: u8) -> bool { return c == 32 or c == 9 or c == 13 or c == 10; }
fn is_digit(c: u8) -> bool { return c >= 48 and c <= 57; }
fn is_name_start(c: u8) -> bool {
if c >= 97 and c <= 122 { return true; }
if c >= 65 and c <= 90 { return true; }
return c == 95;
}
fn is_name_part(c: u8) -> bool {
return is_name_start(c) or is_digit(c);
}
const KEYWORDS: usize = 12;
fn is_keyword(word: []u8) -> bool {
if same(word, "unit") { return true; }
if same(word, "import") { return true; }
if same(word, "pub") { return true; }
if same(word, "fn") { return true; }
if same(word, "struct") { return true; }
if same(word, "enum") { return true; }
if same(word, "let") { return true; }
if same(word, "var") { return true; }
if same(word, "if") { return true; }
if same(word, "else") { return true; }
if same(word, "while") { return true; }
if same(word, "return") { return true; }
return false;
}
fn same(a: []u8, b: []u8) -> bool {
if a.n != b.n { return false; }
var i: usize = 0;
while i < a.n {
if a[i] != b[i] { return false; }
i = i + 1;
}
return true;
}
/// Step over anything that is not a token: spaces, newlines, and `//` to the
/// end of the line. `line` counts what was crossed so a token can say where it
/// came from.
fn skip_gaps(src: []u8, at: &mut usize, line: &mut usize) -> void {
while at.^ < src.n {
let c: u8 = src[at.^];
if c == 10 { line.^ = line.^ + 1; at.^ = at.^ + 1; }
else if is_space(c) { at.^ = at.^ + 1; }
else if c == 47 and at.^ + 1 < src.n and src[at.^ + 1] == 47 {
while at.^ < src.n {
if src[at.^] == 10 { break; }
at.^ = at.^ + 1;
}
}
else { break; }
}
}
pub fn next(src: []u8, at: &mut usize, line: &mut usize) -> tok.Token {
skip_gaps(src, at, line);
let start: usize = at.^;
let where: usize = line.^;
if start >= src.n {
return tok.Token{ kind: tok.Kind.End, from: start, len: 0, line: where };
}
let c: u8 = src[start];
if is_name_start(c) {
while at.^ < src.n {
if not is_name_part(src[at.^]) { break; }
at.^ = at.^ + 1;
}
let word: []u8 = src[start..at.^];
var kind: tok.Kind = tok.Kind.Name;
if is_keyword(word) { kind = tok.Kind.Keyword; }
return tok.Token{ kind: kind, from: start, len: at.^ - start,
line: where };
}
if is_digit(c) {
while at.^ < src.n {
if not is_digit(src[at.^]) { break; }
at.^ = at.^ + 1;
}
return tok.Token{ kind: tok.Kind.Number, from: start,
len: at.^ - start, line: where };
}
if c == 34 {
at.^ = at.^ + 1;
while at.^ < src.n {
if src[at.^] == 34 { break; }
if src[at.^] == 92 and at.^ + 1 < src.n { at.^ = at.^ + 1; }
at.^ = at.^ + 1;
}
if at.^ >= src.n {
return tok.Token{ kind: tok.Kind.Bad, from: start,
len: at.^ - start, line: where };
}
at.^ = at.^ + 1;
return tok.Token{ kind: tok.Kind.Text, from: start, len: at.^ - start,
line: where };
}
at.^ = at.^ + 1;
// Two-byte punctuation the language actually uses.
if at.^ < src.n {
let d: u8 = src[at.^];
if c == 45 and d == 62 { at.^ = at.^ + 1; }
else if c == 61 and d == 61 { at.^ = at.^ + 1; }
else if c == 33 and d == 61 { at.^ = at.^ + 1; }
else if c == 60 and d == 61 { at.^ = at.^ + 1; }
else if c == 62 and d == 61 { at.^ = at.^ + 1; }
else if c == 46 and d == 46 { at.^ = at.^ + 1; }
}
return tok.Token{ kind: tok.Kind.Punct, from: start, len: at.^ - start,
line: where };
}
+39
View File
@@ -0,0 +1,39 @@
unit tok;
// What the lexer produces. A token does not hold the text it came from: R4
// keeps borrows out of aggregate storage, so it records where in the source it
// starts and how long it is, and the source travels beside it.
pub enum Kind {
End,
Name,
Number,
Text,
Punct,
Keyword,
Bad,
}
pub struct Token {
pub kind: Kind,
pub from: usize,
pub len: usize,
pub line: usize,
}
pub fn text(src: []u8, t: Token) -> []u8 {
return src[t.from..t.from + t.len];
}
pub fn name_of(k: Kind) -> []u8 {
match k {
End => { return "end"; }
Name => { return "name"; }
Number => { return "number"; }
Text => { return "text"; }
Punct => { return "punct"; }
Keyword => { return "keyword"; }
Bad => { return "bad"; }
}
return "?";
}