From 76cb7e254c7063e75420b6dec94fb03e5677dc36 Mon Sep 17 00:00:00 2001 From: Sebastian Jeong Date: Mon, 17 Aug 2026 07:35:21 +0900 Subject: [PATCH] =?UTF-8?q?lexer:=20Ferro=20=EC=9D=98=20=EB=A0=89=EC=84=9C?= =?UTF-8?q?=EB=A5=BC=20Ferro=20=EB=A1=9C=20=EC=93=B4=EB=8B=A4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 셀프호스팅에 손대기 전의 강제 함수다. 아픈 자리를 전부 건드린다: R4 아래의 토큰 구조체, 태그드 유니온, 진단 출력, 유닛 경계. 토큰은 자기가 나온 글자를 담지 않는다. R4 가 대여를 집합 저장소에서 막으므로, 어디서 시작해 얼마나 긴지를 적고 소스는 옆에서 같이 다닌다. 위치도 &mut usize 로 옆에서 다닌다 -- 슬라이스와 함께 구조체에 들어갈 수 없기 때문이다. 이것이 R11 이 말하는 모양이고, 쓸 수 있다. first keyword unit @1 / number 42 @3 / text "hi" @3 / arrow -> @5 keyword 6 name 7 number 1 text 1 punct 15 / total 30 길에서 고친 것: - binding.Type.Variant 가 안 풀렸다. 유닛 경계 이름 조회가 심볼만 보고 타입을 보지 않았다. - 문자열 const 전역이 빈 슬라이스로 나갔다. 포인터는 링커만 아는 수라서 바이트에 구멍을 두고 링커가 채우게 한다. - exec.py 가 OUTPUT 마커를 여러 개 적어도 마지막 하나만 검사했다. 고치자마자 readfile 의 낡은 기대가 드러났다. run.py 217/217, exec.py 27/27. --- fec/src/checkcal.c | 18 +++++- fec/src/ir.c | 14 ++++ fec/src/ir.h | 13 ++++ fec/src/lowerstm.c | 35 ++++++++++ fec/src/x86.c | 27 ++++++-- fec/tests/exec/lexer/main.fe | 61 ++++++++++++++++++ fec/tests/exec/lexer/scan.fe | 122 +++++++++++++++++++++++++++++++++++ fec/tests/exec/lexer/tok.fe | 39 +++++++++++ fec/tests/exec/readfile.fe | 2 +- tests/exec.py | 18 ++++-- 10 files changed, 337 insertions(+), 12 deletions(-) create mode 100644 fec/tests/exec/lexer/main.fe create mode 100644 fec/tests/exec/lexer/scan.fe create mode 100644 fec/tests/exec/lexer/tok.fe diff --git a/fec/src/checkcal.c b/fec/src/checkcal.c index 3dfb14b..f24f9e8 100644 --- a/fec/src/checkcal.c +++ b/fec/src/checkcal.c @@ -16,7 +16,23 @@ FeType *cross_unit_value(FeCheckerState *s, FeNode *n, int *handled) if (!home) return 0; *handled=1; sym=unit_member(s->c,home,n->b && n->b->text ? n->b->text : ""); - if (!sym) { err(s->c,n->loc,"unknown name"); return unknown(s->c); } + if (!sym) { + /* A name in another unit can be a type as well as a value -- + `binding.Enum.Variant` reaches one through the other. */ + FeType *there=unit_type(s->c,home,n->b && n->b->text ? n->b->text : ""); + FeNode *decl=unit_type_decl(s->c,home, + n->b && n->b->text ? n->b->text : ""); + if (there && decl) { + if (!decl_is_public(decl)) { + err(s->c,n->loc,"type is private to its unit"); + return unknown(s->c); + } + n->sem_type=there; + return there; + } + err(s->c,n->loc,"unknown name"); + return unknown(s->c); + } if (!decl_is_public(sym->decl)) { err(s->c,n->loc,"name is private to its unit"); return unknown(s->c); diff --git a/fec/src/ir.c b/fec/src/ir.c index 0bb795c..6b5f117 100644 --- a/fec/src/ir.c +++ b/fec/src/ir.c @@ -106,6 +106,20 @@ FeIrGlobal *fe_ir_global(FeIrModule *m, const char *name, FeIrType type, return g; } +void fe_ir_global_ref(FeIrModule *m, FeIrGlobal *g, unsigned long at, + const char *symbol) +{ + FeIrReloc *grown; + if (!g) return; + grown = (FeIrReloc *)ir_alloc(m, (g->reloc_count + 1) * sizeof(FeIrReloc)); + if (!grown) return; + if (g->relocs) memcpy(grown, g->relocs, g->reloc_count * sizeof(FeIrReloc)); + grown[g->reloc_count].at = at; + grown[g->reloc_count].symbol = symbol; + g->relocs = grown; + ++g->reloc_count; +} + const char *fe_ir_string(FeIrModule *m, const char *bytes, unsigned long length) { FeIrGlobal *g; diff --git a/fec/src/ir.h b/fec/src/ir.h index 4db1269..beff64f 100644 --- a/fec/src/ir.h +++ b/fec/src/ir.h @@ -121,12 +121,22 @@ struct FeIrFunc { struct FeIrFunc *next; }; +/* A place inside a global's bytes that holds the address of something else. + The value is not known until the linker places it, so the bytes carry a hole + and this says what fills it. */ +typedef struct FeIrReloc { + unsigned long at; + const char *symbol; +} FeIrReloc; + typedef struct FeIrGlobal { const char *name; FeIrType type; unsigned long size; unsigned align; const unsigned char *init; /* size bytes, or null for zero */ + FeIrReloc *relocs; + unsigned reloc_count; struct FeIrGlobal *next; } FeIrGlobal; @@ -155,6 +165,9 @@ FeIrBlock *fe_ir_block(FeIrModule *m, FeIrFunc *f); FeIrGlobal *fe_ir_global(FeIrModule *m, const char *name, FeIrType type, unsigned long size, unsigned align, const unsigned char *init); +/* Say that `at` bytes into `g` there is the address of `symbol`. */ +void fe_ir_global_ref(FeIrModule *m, FeIrGlobal *g, unsigned long at, + const char *symbol); /* A string literal's bytes, interned so the same text is stored once. */ const char *fe_ir_string(FeIrModule *m, const char *bytes, unsigned long length); diff --git a/fec/src/lowerstm.c b/fec/src/lowerstm.c index a9dff93..df09f18 100644 --- a/fec/src/lowerstm.c +++ b/fec/src/lowerstm.c @@ -504,6 +504,41 @@ void lower_global(Lower *L, FeNode *n) unsigned char *init = 0; unsigned long size = ir_size(t); if (!n->cname) return; + /* A text constant is a pointer and a length. The pointer is not a number + anyone knows yet, so the bytes carry a hole and the linker fills it. */ + if (n->b && n->b->kind == FE_N_LITERAL && n->b->text && + n->b->text[0] == '"' && t && + (t->kind == FE_TYPE_SLICE || t->kind == FE_TYPE_STR)) { + char text[1024]; + unsigned long raw = strlen(n->b->text); + unsigned long len = 0; + unsigned long i; + const char *label; + FeIrGlobal *g; + if (raw >= 2) raw -= 2; + for (i = 0; i < raw && len + 1 < sizeof text; ++i) { + char ch = n->b->text[1 + i]; + if (ch == 92 && i + 1 < raw) { + ++i; + switch (n->b->text[1 + i]) { + case 'n': ch = 10; break; + case 't': ch = 9; break; + case 'r': ch = 13; break; + case '0': ch = 0; break; + default: ch = n->b->text[1 + i]; break; + } + } + text[len++] = ch; + } + label = fe_ir_string(L->m, text, len); + init = (unsigned char *)fe_arena_alloc(&L->m->arena, 8); + if (!init || !label) return; + for (i = 0; i < 8; ++i) init[i] = 0; + for (i = 0; i < 4; ++i) init[4 + i] = (unsigned char)((len >> (i * 8)) & 0xFF); + g = fe_ir_global(L->m, n->cname, FE_IR_MEM, 8, 4, init); + fe_ir_global_ref(L->m, g, (unsigned long)SLICE_PTR_OFFSET, label); + return; + } if (n->b && n->b->kind == FE_N_LITERAL && size && size <= 8) { long v = literal_value(n->b); unsigned long i; diff --git a/fec/src/x86.c b/fec/src/x86.c index bec88fa..98e5824 100644 --- a/fec/src/x86.c +++ b/fec/src/x86.c @@ -634,10 +634,29 @@ void fe_x86_emit(const FeIrModule *m, FILE *out) fprintf(out, " db %lu dup(0)\n", g->size ? g->size : 1UL); continue; } - for (i = 0; i < g->size; ++i) { - if (i % 16 == 0) fputs(" db ", out); - fprintf(out, "%u%s", g->init[i], - (i + 1 == g->size || (i % 16) == 15) ? "\n" : ","); + for (i = 0; i < g->size; ) { + unsigned r; + unsigned long j; + for (r = 0; r < g->reloc_count; ++r) + if (g->relocs[r].at == i) break; + if (r < g->reloc_count) { + /* A hole the linker fills with an address. */ + fprintf(out, " dd offset %s\n", + g->relocs[r].symbol); + i += 4; + continue; + } + fputs(" db ", out); + j = 0; + while (i < g->size && j < 16) { + unsigned q; + for (q = 0; q < g->reloc_count; ++q) + if (g->relocs[q].at == i) break; + if (q < g->reloc_count) break; + fprintf(out, "%s%u", j ? "," : "", g->init[i]); + ++i; ++j; + } + fputc('\n', out); } if (!g->size) fputs(" db 0\n", out); } diff --git a/fec/tests/exec/lexer/main.fe b/fec/tests/exec/lexer/main.fe new file mode 100644 index 0000000..032dfa1 --- /dev/null +++ b/fec/tests/exec/lexer/main.fe @@ -0,0 +1,61 @@ +// EXIT:0 +// OUTPUT:first keyword unit @1 +// OUTPUT:number 42 @3 +// OUTPUT:text "hi" @3 +// OUTPUT:arrow -> @5 +// OUTPUT:keyword 6 name 7 number 1 text 1 punct 15 +// OUTPUT:total 30 +unit main; +import std.io; +import tok; +import scan; + +// The Ferro lexer, written in Ferro. This is the shape a self-hosted `fec` +// would take: read a source, hand back tokens, say where each came from. + +const SOURCE: str = "unit demo;\n\nfn answer() { let n = 42; let s = \"hi\"; }\n// a comment\nfn arrow() -> i32 { return n; }\n"; + +fn main() -> i32 { + var at: usize = 0; + var line: usize = 1; + var keywords: usize = 0; + var names: usize = 0; + var numbers: usize = 0; + var texts: usize = 0; + var puncts: usize = 0; + var total: usize = 0; + var first: bool = true; + while true { + let t: tok.Token = scan.next(SOURCE, &mut at, &mut line); + if t.kind == tok.Kind.End { break; } + total = total + 1; + if first { + @print("first {} {} @{}\n", tok.name_of(t.kind), + tok.text(SOURCE, t), t.line); + first = false; + } + match t.kind { + Keyword => { keywords = keywords + 1; } + Name => { names = names + 1; } + Number => { + numbers = numbers + 1; + @print("number {} @{}\n", tok.text(SOURCE, t), t.line); + } + Text => { + texts = texts + 1; + @print("text {} @{}\n", tok.text(SOURCE, t), t.line); + } + Punct => { + puncts = puncts + 1; + if t.len == 2 { + @print("arrow {} @{}\n", tok.text(SOURCE, t), t.line); + } + } + _ => { @print("unexpected {}\n", tok.name_of(t.kind)); } + } + } + @print("keyword {} name {} number {} text {} punct {}\n", + keywords, names, numbers, texts, puncts); + @print("total {}\n", total); + return 0; +} diff --git a/fec/tests/exec/lexer/scan.fe b/fec/tests/exec/lexer/scan.fe new file mode 100644 index 0000000..81e9ac5 --- /dev/null +++ b/fec/tests/exec/lexer/scan.fe @@ -0,0 +1,122 @@ +unit scan; +import tok; + +// A scanner over a byte slice. The position travels in a `&mut usize` beside +// the source rather than inside a struct with it, because a struct cannot hold +// a slice: a slice is a borrowed view and R4 keeps borrows out of aggregates. + +fn is_space(c: u8) -> bool { return c == 32 or c == 9 or c == 13 or c == 10; } +fn is_digit(c: u8) -> bool { return c >= 48 and c <= 57; } + +fn is_name_start(c: u8) -> bool { + if c >= 97 and c <= 122 { return true; } + if c >= 65 and c <= 90 { return true; } + return c == 95; +} + +fn is_name_part(c: u8) -> bool { + return is_name_start(c) or is_digit(c); +} + +const KEYWORDS: usize = 12; + +fn is_keyword(word: []u8) -> bool { + if same(word, "unit") { return true; } + if same(word, "import") { return true; } + if same(word, "pub") { return true; } + if same(word, "fn") { return true; } + if same(word, "struct") { return true; } + if same(word, "enum") { return true; } + if same(word, "let") { return true; } + if same(word, "var") { return true; } + if same(word, "if") { return true; } + if same(word, "else") { return true; } + if same(word, "while") { return true; } + if same(word, "return") { return true; } + return false; +} + +fn same(a: []u8, b: []u8) -> bool { + if a.n != b.n { return false; } + var i: usize = 0; + while i < a.n { + if a[i] != b[i] { return false; } + i = i + 1; + } + return true; +} + +/// Step over anything that is not a token: spaces, newlines, and `//` to the +/// end of the line. `line` counts what was crossed so a token can say where it +/// came from. +fn skip_gaps(src: []u8, at: &mut usize, line: &mut usize) -> void { + while at.^ < src.n { + let c: u8 = src[at.^]; + if c == 10 { line.^ = line.^ + 1; at.^ = at.^ + 1; } + else if is_space(c) { at.^ = at.^ + 1; } + else if c == 47 and at.^ + 1 < src.n and src[at.^ + 1] == 47 { + while at.^ < src.n { + if src[at.^] == 10 { break; } + at.^ = at.^ + 1; + } + } + else { break; } + } +} + +pub fn next(src: []u8, at: &mut usize, line: &mut usize) -> tok.Token { + skip_gaps(src, at, line); + let start: usize = at.^; + let where: usize = line.^; + if start >= src.n { + return tok.Token{ kind: tok.Kind.End, from: start, len: 0, line: where }; + } + let c: u8 = src[start]; + if is_name_start(c) { + while at.^ < src.n { + if not is_name_part(src[at.^]) { break; } + at.^ = at.^ + 1; + } + let word: []u8 = src[start..at.^]; + var kind: tok.Kind = tok.Kind.Name; + if is_keyword(word) { kind = tok.Kind.Keyword; } + return tok.Token{ kind: kind, from: start, len: at.^ - start, + line: where }; + } + if is_digit(c) { + while at.^ < src.n { + if not is_digit(src[at.^]) { break; } + at.^ = at.^ + 1; + } + return tok.Token{ kind: tok.Kind.Number, from: start, + len: at.^ - start, line: where }; + } + if c == 34 { + at.^ = at.^ + 1; + while at.^ < src.n { + if src[at.^] == 34 { break; } + if src[at.^] == 92 and at.^ + 1 < src.n { at.^ = at.^ + 1; } + at.^ = at.^ + 1; + } + if at.^ >= src.n { + return tok.Token{ kind: tok.Kind.Bad, from: start, + len: at.^ - start, line: where }; + } + at.^ = at.^ + 1; + return tok.Token{ kind: tok.Kind.Text, from: start, len: at.^ - start, + line: where }; + } + at.^ = at.^ + 1; + // Two-byte punctuation the language actually uses. + if at.^ < src.n { + let d: u8 = src[at.^]; + if c == 45 and d == 62 { at.^ = at.^ + 1; } + else if c == 61 and d == 61 { at.^ = at.^ + 1; } + else if c == 33 and d == 61 { at.^ = at.^ + 1; } + else if c == 60 and d == 61 { at.^ = at.^ + 1; } + else if c == 62 and d == 61 { at.^ = at.^ + 1; } + else if c == 46 and d == 46 { at.^ = at.^ + 1; } + } + return tok.Token{ kind: tok.Kind.Punct, from: start, len: at.^ - start, + line: where }; +} diff --git a/fec/tests/exec/lexer/tok.fe b/fec/tests/exec/lexer/tok.fe new file mode 100644 index 0000000..a5b9cb3 --- /dev/null +++ b/fec/tests/exec/lexer/tok.fe @@ -0,0 +1,39 @@ +unit tok; + +// What the lexer produces. A token does not hold the text it came from: R4 +// keeps borrows out of aggregate storage, so it records where in the source it +// starts and how long it is, and the source travels beside it. + +pub enum Kind { + End, + Name, + Number, + Text, + Punct, + Keyword, + Bad, +} + +pub struct Token { + pub kind: Kind, + pub from: usize, + pub len: usize, + pub line: usize, +} + +pub fn text(src: []u8, t: Token) -> []u8 { + return src[t.from..t.from + t.len]; +} + +pub fn name_of(k: Kind) -> []u8 { + match k { + End => { return "end"; } + Name => { return "name"; } + Number => { return "number"; } + Text => { return "text"; } + Punct => { return "punct"; } + Keyword => { return "keyword"; } + Bad => { return "bad"; } + } + return "?"; +} diff --git a/fec/tests/exec/readfile.fe b/fec/tests/exec/readfile.fe index a2888e0..f904eb8 100644 --- a/fec/tests/exec/readfile.fe +++ b/fec/tests/exec/readfile.fe @@ -1,5 +1,5 @@ // EXIT:0 -// OUTPUT:read 64 bytes +// OUTPUT:read 27 bytes // OUTPUT:first line: // EXIT:0 unit readfile; import std.io; diff --git a/tests/exec.py b/tests/exec.py index 92d0ea6..75ffd5f 100644 --- a/tests/exec.py +++ b/tests/exec.py @@ -35,14 +35,17 @@ def expectations(path: Path) -> dict: m = re.match(r"//\s*(EXIT|OUTPUT|NOCHECKS):(.*)", line) if m: key, value = m.group(1), m.group(2).strip() - want[key] = int(value) if key in ("EXIT", "NOCHECKS") else value + if key == "OUTPUT": + # Every OUTPUT line has to appear. Keeping only the last one + # would let the earlier ones rot unnoticed. + want.setdefault("OUTPUT", []).append(value) + else: + want[key] = int(value) return want def check_one(fec: Path, path: Path, out_dir: Path) -> tuple[bool, str]: want = expectations(path) - if "EXIT" not in want: - return False, "no // EXIT: marker" exe, log = builder.build(fec, path, out_dir) if not exe: @@ -52,8 +55,9 @@ def check_one(fec: Path, path: Path, out_dir: Path) -> tuple[bool, str]: code, text = builder.run(exe) if code != want["EXIT"]: return False, f"exited {code}, expected {want['EXIT']}\n {text.strip()}" - if "OUTPUT" in want and want["OUTPUT"] not in text: - return False, f"output has no {want['OUTPUT']!r}\n {text.strip()}" + for line in want.get("OUTPUT", []): + if line not in text: + return False, f"output has no {line!r}\n {text.strip()}" if "NOCHECKS" in want: exe2, log2 = builder.build(fec, path, out_dir / "nochecks", @@ -81,7 +85,9 @@ def main() -> int: if out_dir.exists(): shutil.rmtree(out_dir, ignore_errors=True) - cases = sorted(PROGRAMS.rglob("*.fe")) + # A file with no `// EXIT:` is a unit some program imports, not a program. + cases = [p for p in sorted(PROGRAMS.rglob("*.fe")) + if "EXIT" in expectations(p)] if args.select: cases = [p for p in cases if args.select in p.as_posix()] if not cases: