feat: Add language grammar and adjusted tokenizer

Signed-off-by: erick-alcachofa <erick@artichoke.dev>

This commit lays the foundational groundwork for the artichoke language
parser by introducing the formal language grammar specification.

The tokenizer was updated to include new operators and keywords, also
added the posibility to handle comments.

Key Additions:
- Implemented support for C-style block comments (`/* ... */`),
  including error handling for unclosed comments.
- Added all necessary tokens for missing keywords (e.g., `module`,
  `export`, `using`, `match`, `loop`) and operators (e.g., `+=`, `:=`,
  `.#`, `.*`, `.@`).
- The `Token` enum has been expanded to reflect the full language
  feature set.

Documentation:
- Added `docs/grammar.ebnf` which contains the official, well-structured
  EBNF grammar for the language.
- Added `docs/readme.md` providing a detailed technical overview of the
  language's features, syntax, and semantics.

BREAKING CHANGE: The `kwVariant` and `kwMut` tokens have been removed to
align with the updated language design defined in the new grammar.
This commit is contained in:
erick-alcachofa 2025-10-01 18:51:09 -06:00
parent f9051e1c21
commit d0599d374f
Signed by: me
GPG Key ID: 6FA5F8643444BAFA
7 changed files with 893 additions and 147 deletions

348
docs/grammar.ebnf Normal file
View File

@ -0,0 +1,348 @@
/*
================================================================================
| |
| The Artichoke Programming Language |
| Official EBNF Grammar |
| |
================================================================================
*/
/* --- Program Structure --- */
/* A program is a sequence of top-level declarations and statements. */
<program> =
( <import_statement>
| <module_statement>
| <alias_statement>
| <struct_declaration>
| <enum_declaration>
| <function_declaration> )*
<module_statement> =
"export"? "module" <namespaced_identifier> "{"
( <module_statement>
| <alias_statement>
| <struct_declaration>
| <enum_declaration>
| <function_declaration> )*
"}"
<import_statement> =
"import" <import_target> ";"
<import_target> =
<namespaced_identifier>
| <namespaced_identifier> "::" "*"
<alias_statement> =
"using" <identifier> "=" <namespaced_identifier> ";"
/* --- Declarations --- */
/* Rules for defining functions, structs, enums, and their components. */
<function_declaration> =
"export"? "fn" <identifier> <generic_params> "(" <fn_params> ")" ( "->" <type> )? <code_block>
<fn_params> =
<fn_params_list>?
<fn_params_list> =
"this" <type> ("," <fn_param> ( "," <fn_param> )* )?
| <fn_param> ( "," <fn_param> )*
<fn_param> =
<identifier> ":" <type>
<struct_declaration> =
"export"? "struct" <identifier> <generic_params> "{" <struct_members> "}"
<struct_members> =
<struct_member> ( "," <struct_member> )*
<struct_member> =
<identifier> ":" <type>
<enum_declaration> =
"export"? "enum" <identifier> <generic_params> "{" <enum_members> "}"
<enum_members> =
<enum_member> ( "," <enum_member> )*
<enum_member> =
<identifier> ( "(" <type> ")" )?
<generic_params> =
( "<" <generic_params_list> ">" )?
<generic_params_list> =
<generic_param> ( "," <generic_param> )*
<generic_param> =
"typename" <identifier>
/* --- Statements & Control Flow --- */
/* Rules for code blocks, variable declarations, and control structures. */
<code_block> =
"{" <statements>? "}"
<statements> =
<statement> ( <statement> )*
<statement> =
<variable_declaration> ";"
| <if_statement>
| <loop_statement>
| <defer_statement> ";"
| <errdefer_statement> ";"
| <return_statement> ";"
| <break_statement> ";"
| <continue_statement> ";"
| <expression_statement>
| <alias_statement>
| <match_statement>
| <switch_statement>
<variable_declaration> =
<variable_declarator> <identifier> ( ":" <type> )? "=" <expression>
| <variable_declarator> <identifier> ":" <type> ( "=" <expression> )?
<variable_declarator> =
"let"
| "def"
<if_statement> =
"if" "(" <expression> ")" <variable_unwrapper>? <code_block>
<else_statement>?
<else_statement> =
"else" <variable_unwrapper>? <code_block>
| "else" <if_statement>
<variable_unwrapper> =
"|" <identifier> "|"
<loop_statement> =
(<identifier> ":=")? (
<c_for_statement>
| <range_for_statement>
| <while_statement>
| <do_while_statement>
| <inf_loop_statement>
)
<c_for_statement> =
"for" "(" <variable_declaration>? ";" <expression> ";" <expression> ")"
<code_block>
<range_for_statement> =
"for" "(" <variable_declarator> <identifier> ":=" <expression> ")"
<code_block>
<while_statement> =
"while" "(" <expression> ")" <variable_unwrapper>? <code_block>
<else_statement>?
<do_while_statement> =
"do" <code_block> "while" "(" <expression> ")"
<inf_loop_statement> =
"loop" <code_block>
<match_statement> =
"match" "(" <expression> ")" "{" <match_case>* <default_case>? "}"
<switch_statement> =
"switch" "(" <expression> ")" "{" <switch_case>* <default_case>? "}"
<match_case> =
( <type_name> | <scoped_access_expression> ) ( "(" <identifier> ")" )? "->" <code_block>
<switch_case> =
<expression> "->" <code_block>
<default_case> =
"_" "->" <code_block>
<break_statement> =
"break" <identifier>?
<continue_statement> =
"continue" <identifier>?
<defer_statement> =
"defer" ( <expression> | <code_block> )
<errdefer_statement> =
"errdefer" ( <expression> | <code_block> )
<return_statement> =
"return" <expression>?
<expression_statement> =
<expression> ";"
/* --- Expressions & Operator Precedence --- */
/* The full expression hierarchy, from lowest to highest precedence. */
<expression> =
<assign_expression>
<assign_expression> =
<bool_or_expression> ( ( <assign_op> | <compound_assign_op> ) <expression> )?
<bool_or_expression> =
<bool_and_expression> ( ( "||" | "or" ) <bool_and_expression> )*
<bool_and_expression> =
<compare_expression> ( ( "&&" | "and" ) <compare_expression> )*
<compare_expression> =
<bitwise_expression> ( <compare_op> <bitwise_expression> )?
<bitwise_expression> =
<bitwise_shift_expression> ( <bitwise_op> <bitwise_shift_expression> )*
<bitwise_shift_expression> =
<addition_expression> ( <bitshift_op> <addition_expression> )*
<addition_expression> =
<multiply_expression> ( <addition_op> <multiply_expression> )*
<multiply_expression> =
<prefix_expression> ( <multiply_op> <prefix_expression> )*
<prefix_expression> =
<prefix_op>* <primary_expression>
<primary_expression> =
<primary_type_expression> ( <suffix_op> | <fn_call_arguments> )*
/* --- Primary Expressions & Literals --- */
/* The highest-precedence expressions, including literals and grouped expressions. */
<primary_type_expression> =
<char_literal>
| <null_literal>
| <string_literal>
| <number_literal>
| <boolean_literal>
| <grouped_expression>
| <identifier>
| <struct_literal>
| <scoped_access_expression>
| <reflection_expression>
<grouped_expression> =
"(" <expression> ")"
<scoped_access_expression> =
<type_name> "::" <identifier>
<reflection_expression> =
( <primary_expression> | <type_name> | <scoped_access_expression> ) ".@" <identifier>?
<fn_call_arguments> =
"(" <expression_list> ")"
<expression_list> =
(<expression> ",")* <expression>?
<struct_literal> =
<type> "{" ( <named_field_list> | <positional_field_list> )? ","? "}"
<named_field_list> =
<named_field_init> ( "," <named_field_init> )*
<named_field_init> =
<identifier> ":" <expression>
<positional_field_list> =
<expression> ( "," <expression> )*
<null_literal> =
"null"
<boolean_literal> =
"true"
| "false"
<number_literal> = /* Assumed to be defined by the tokenizer */
<string_literal> = /* Assumed to be defined by the tokenizer */
<char_literal> = /* Assumed to be defined by the tokenizer */
/* --- Operators --- */
/* Definitions for all operator token sets. */
<assign_op> = "="
<compound_assign_op> = "+=" | "-=" | "*=" | "/=" | "%=" | "&=" | "|=" | "<<=" | ">>=" | "||=" | "&&="
<compare_op> = "==" | "!=" | ">" | "<" | ">=" | "<="
<bitwise_op> = "&" | "^" | "|"
<bitshift_op> = "<<" | ">>"
<addition_op> = "+" | "-"
<multiply_op> = "*" | "/" | "%"
<prefix_op> = "!" | "-" | "~" | "&" | "*"
<suffix_op> =
"[" <expression>? ":" <expression>? "]"
| "[" <expression> "]"
| ".[" <expression> "]"
| "." <identifier>
| "->" <identifier>
| ".#"
| ".*"
/* --- Type System --- */
/* Rules for defining types, type names, and type qualifiers. */
<type> =
<type_qualifier_chain> <type_name>
<type_qualifier_chain> =
( "*" | "[]" ) <type_qualifier_chain>?
| "$" <type_qualifier_chain_after_mutable>?
| "?" <type_qualifier_chain_after_optional>?
<type_qualifier_chain_after_optional> =
( "*" | "[]" ) <type_qualifier_chain>?
| "$" <type_qualifier_chain_after_mutable>?
<type_qualifier_chain_after_mutable> =
( "*" | "[]" ) <type_qualifier_chain>?
| "?" <type_qualifier_chain_after_optional>?
<type_name> =
<namespaced_identifier> ( "<" <types_list> ">" )?
<namespaced_identifier> =
<identifier>
| <identifier> "::" <namespaced_identifier>
<types_list> =
<type> ( "," <types_list> )*
/* --- Lexical Tokens & Base Definitions --- */
/* The lowest-level building blocks of the language. */
<identifier> =
<nondigit> <identifier_tail>
<identifier_tail> =
<empty>
| <nondigit> <identifier_tail>
| <digit> <identifier_tail>
<nondigit> = "_" | [a-z] | [A-Z]
<digit> = <zero> | <nonzero_digit>
<zero> = "0"
<nonzero_digit> = [1-9]
<empty> = E /* Represents an empty terminal string */

286
docs/readme.md Normal file
View File

@ -0,0 +1,286 @@
# **The `artichoke` Programming Language: A Technical Overview**
## **1. Introduction**
`artichoke` is a statically-typed, general-purpose programming language designed
with an emphasis on performance, safety, and expressive syntax. It combines
low-level control over memory with modern, high-level features like generics,
algebraic data types, and integrated error handling. This document provides an
overview of the language's features as defined by its core grammar.
Is highly inspired by C, C++, Rust, and mostly Zig.
## **2. Basic Syntax & Structure**
### **Modules, Imports, and Aliases**
`artichoke` code is organized into modules. The `import` statement is used to bring
symbols from other modules into the current scope.
* **Importing a specific element:** `import my_module::some_function;`
* **Importing all direct elements of a module:** `import std::*;`
* **Importing an entire submodule:** `import std::memory;`
The `using` keyword creates a local, more convenient alias for a type, function,
or module name.
```
using mem = std::memory;
using FileHandle = std::fs::File;
```
### **Comments**
The language uses C-style block comments.
```
/* This is a multi-line
comment. */
```
## **3. The Type System**
`artichoke`'s type system is strong and static, with a rich set of features for
defining complex data structures.
### **Type Qualifiers**
Qualifiers modify the type to their immediate right, allowing for precise and
complex type definitions.
* **`*` (Pointer):** Creates a pointer to a type. Pointers cannot be `null`.
* **`$` (Mutable):** Marks a type as mutable. This is used for function
parameters, local variables, and struct fields to allow modification.
* **`?` (Optional):** Marks a type as nullable. An optional type can hold either a
value of its underlying type or `null`.
* **`[]` (Slice):** A "fat pointer" representing a view into a contiguous
sequence of elements. It contains both a pointer to the data and a length.
These qualifiers can be combined. For example, `*$?int` defines a **pointer to a
mutable optional integer**.
### **Generics**
Generics allow for writing flexible, reusable code that can operate on multiple
types. They are defined using `<typename T>`.
```
/* A generic struct */
struct Point<typename T> {
x: T,
y: T
}
/* A generic function */
fn scale<typename T>(lhs: *Point<T>, rhs: T) -> Point {
/* ... */
}
```
## **4. Declarations**
### **Variables**
Variables are declared using the `let` (mutable) and `def` (immutable/constant)
keywords.
* **Type inference** is supported when the type can be determined from the initializer.
* Variables must be initialized with either a type, a value, or both.
```
/* Mutable variable with explicit type */
let x: i32 = 10;
/* Immutable variable with type inference */
def do_you_get_it = meaning_of_life();
```
### **Structs**
Structs are composite data types that group together variables under one name.
They support generics.
```
struct Rectangle {
top: Point<i32>,
bot: Point<i32>
}
```
**Initialization:** Structs can be initialized using positional or named fields,
but not a mix of both.
```
/* Positional initialization */
def top_left = Point<i32>{ 0, 10 };
/* Named-field initialization */
def top_right = Point<i32>{ x: 10, y: 10 };
```
### **Enums (Tagged Unions)**
Enums define a type that can be one of several different variants. Variants can
optionally hold data.
```
enum AssetType {
Texture,
Model,
Sound,
}
enum Result<typename T, typename E> {
Ok(T),
Err(E)
}
```
**Initialization:** Enum variants are accessed using scope resolution (`::`).
```
def my_asset = AssetType::Texture;
def success = Result<i32, string>::Ok(100);
```
### **Functions**
Functions are defined with the fn keyword. The return type is specified after
the parameter list with `->`.
```
fn meaning_of_life() -> i32 {
return 42;
}
```
#### **Member Functions (`this` parameter)**
If the first parameter of a function is declared with the `this` keyword, it can
be called using "member function" syntax.
```
/* Definition */
fn add<typename T>(this *$Point<T>, other: *Point<T>) {
this->x += other->x;
this->y += other->y;
}
/* Can be called in two ways: */
/* Member function syntax */
my_point.add(&other_point);
/* Normal function syntax */
add(&my_point, &other_point);
```
## **5. Control Flow**
### **`if`/`else` Statements**
`artichoke` supports C-style `if`/`else` and `else if` chains. It also integrates a
powerful unwrapping feature for handling `Result` and optional (`?`) types.
```
/* Standard if/else */
if (argc < 2) {
return Result::Err(-1);
}
/* Unwrapping a Result */
if (foo()) |ok| {
/* `ok` holds the success value */
}
else |err| {
/* `err` holds the error value */
}
```
### **Loops**
The language provides a comprehensive set of looping constructs.
* **C-Style `for`:** `for (let i \= 0; i \< 10; i \+= 1\) { ... }`
* **Range-based `for`:** `for (let e := arrSlice) { ... }`
* **`while` Loop:** Can optionally have an `else` block that executes when the loop
condition is no longer met.
* **Iterator `while`:** Supports unwrapping `Result`/optional types, executing as
long as the value is valid.
* **`do-while` Loop:** Guarantees the body executes at least once.
* **Infinite `loop`:** `loop { ... }`
#### **Loop Labels and Control**
Loops can be labeled. The `break` and `continue` statements can optionally specify a
label to control nested loops.
```
outer_loop := while (condition) {
inner_loop := for (...) {
break outer_loop;
}
}
```
## **6. Expressions and Operators**
### **Pointer and Member Access**
* **`&` (Address-of):** Gets a pointer to a variable.
* **`*` (Dereference):** Accesses the value a pointer points to.
* **`.` (Member Access):** Accesses a member of a struct value.
* **`->` (Pointer Member Access):** Dereferences a pointer and accesses a member
(`p->x` is shorthand for `(*p).x`).
### **Slice Operators**
Slices have a dedicated set of operators for manipulation.
* **`[start:end]` (Slicing):** Creates a new slice from an existing one.
* **`.*` (Pointer Access):** Gets the underlying raw pointer of the slice.
* **`.#` (Length Access):** Gets the number of elements in the slice.
* **`.[length]` (Slice from Pointer):** Creates a slice from a raw pointer and a length.
### **Assignment**
The language supports simple (`=`) and compound assignment (`+=`, `*=`, etc.)
operators.
## **7. Advanced Features**
### **Resource Management (`defer` and `errdefer`)**
`artichoke` uses `defer` for deterministic resource management.
* **`defer`:** Schedules an expression or code block to be executed when the
current scope is exited. Deferred calls are executed in Last-In, First-Out
(LIFO) order.
* **`errdefer`:** Similar to `defer`, but the code is only executed if the scope is
exited due to a function returning an error (an `Err` variant of a `Result`).
```
defer call_cleanup();
errdefer {
log("An error occurred!");
}
```
### **Reflection (`.@`)**
The language provides a compile-time reflection mechanism via the `.@` operator.
It can be applied to values, types, and static members to query metadata.
* **On values:** `my_variable.@type`
* **On types:** `Point<u32>.@size, Point<u32>.@alignment`
* **On static members:** `Point<u32>::x.@offset`
```
/* Gets size in bytes */
def size_bytes = Point<u32>.@size;
/* Gets string representation of the type */
def point_name = Point<u32>.@typename;
```

View File

@ -23,84 +23,89 @@ namespace arti::lang {
tkCharacter, tkCharacter,
tkIdentifier, tkIdentifier,
opDot, opDot, /* . */
opMod, opMod, /* % */
opPlus, opPlus, /* + */
opHyphen, opHyphen, /* - */
opSlash, opSlash, /* / */
opBang, opBang, /* ! */
opStar, opStar, /* * */
opColon, opColon, /* : */
opComma, opComma, /* , */
opAssign, opAssign, /* = */
opAccess, opAccess, /* :: */
opSemicolon, opSemicolon, /* ; */
opCaret, /* ^ */
opTilde, /* ~ */
opEq, /* == */
opNeq, /* != */
opLt, /* < */
opGt, /* > */
opLtEq, /* <= */
opGtEq, /* >= */
opLShift, /* << */
opRShift, /* >> */
opBoolAnd, /* && */
opBoolOr, /* || */
opAnd, /* & */
opOr, /* | */
opLParen, /* ( */
opRParen, /* ) */
opLBracket, /* [ */
opRBracket, /* ] */
opLSquirly, /* { */
opRSquirly, /* } */
opArrow, /* -> */
opPlusAssign, /* += */
opHyphenAssign, /* -= */
opStarAssign, /* *= */
opSlashAssign, /* /= */
opModAssign, /* %= */
opAndAssign, /* &= */
opOrAssign, /* |= */
opLShiftAssign, /* <<= */
opRShiftAssign, /* >>= */
opBoolAndAssign, /* &&= */
opBoolORAssign, /* ||= */
opMut, /* $ */
opOpt, /* ? */
opSliceSize, /* .# */
opPtrSlice, /* .[ */
opSlicePtr, /* .* */
opReflect, /* .@ */
opLabel, /* := */
opCaret, /* Keywords */
opTilde, kwUnderscore, /* _ */
kwOr, /* or */
opEq, kwNot, /* not */
opNeq, kwAnd, /* and */
kwIf, /* if */
opLt, kwElse, /* else */
opGt, kwFn, /* fn */
kwEnum, /* enum */
opLtEq, kwStruct, /* struct */
opGtEq, kwDef, /* def */
kwLet, /* let */
opLShift, kwFor, /* for */
opRShift, kwLoop, /* loop */
kwBreak, /* break */
opBoolAnd, kwContinue, /* continue */
opBoolOr, kwWhile, /* while */
kwMatch, /* match */
opAnd, kwSwitch, /* switch */
opOr, kwReturn, /* return */
kwUnreachable, /* unreachable */
opLParen, kwDefer, /* defer */
opRParen, kwErrDefer, /* errdefer */
kwTrue, /* true */
opLBracket, kwFalse, /* false */
opRBracket, kwNull, /* null */
kwThis, /* this */
opLSquirly, kwImport, /* import */
opRSquirly, kwExport, /* export */
kwModule, /* module */
opArrow, kwUsing, /* using */
kwOr,
kwNot,
kwAnd,
kwIf,
kwElse,
kwFn,
kwEnum,
kwStruct,
kwVariant,
kwDef,
kwLet,
kwMut,
kwFor,
kwWhile,
kwReturn,
kwUnreachable,
kwDefer,
kwErrDefer,
kwTrue,
kwFalse,
kwNull,
kwImport,
kwExport,
kwModule,
}; };
std::string toString(const Token &value); std::string toString(const Token &value);

View File

@ -34,6 +34,7 @@ namespace arti::lang {
Generator<Expected<Token>> tokenize(); Generator<Expected<Token>> tokenize();
void skip_whitespace(); void skip_whitespace();
Expected<void> skip_comment();
Expected<Token> readNumber(); Expected<Token> readNumber();
Expected<Token> readString(); Expected<Token> readString();

View File

@ -16,6 +16,7 @@ namespace arti::lang {
ecInvalidLiteral, ecInvalidLiteral,
ecInvalidCharacter, ecInvalidCharacter,
ecInvalidIndex, ecInvalidIndex,
ecInvalidComment,
}; };
struct Exception { struct Exception {
@ -67,6 +68,9 @@ namespace arti::lang {
else if constexpr (code == ecInvalidIndex) { else if constexpr (code == ecInvalidIndex) {
return "Invalid index"; return "Invalid index";
} }
else if constexpr (code == ecInvalidComment) {
return "Invalid comment found, missing '*/' end of comment";
}
else { else {
return "Unknown error"; return "Unknown error";
} }

View File

@ -11,70 +11,94 @@ namespace arti::lang {
std::string_view tokenStr; std::string_view tokenStr;
switch (value.value) { switch (value.value) {
case tkEOF: return "Token{{ tkEOF }}"; case tkEOF: return "Token{ tkEOF }";
case tkString: tokenStr = "tkString"; break; case tkString: tokenStr = "tkString"; break;
case tkDecimal: tokenStr = "tkDecimal"; break; case tkDecimal: tokenStr = "tkDecimal"; break;
case tkInteger: tokenStr = "tkInteger"; break; case tkInteger: tokenStr = "tkInteger"; break;
case tkCharacter: tokenStr = "tkCharacter"; break; case tkCharacter: tokenStr = "tkCharacter"; break;
case tkIdentifier: tokenStr = "tkIdentifier"; break; case tkIdentifier: tokenStr = "tkIdentifier"; break;
case opDot: tokenStr = "opDot"; break; case opDot: tokenStr = "opDot"; break;
case opMod: tokenStr = "opMod"; break; case opMod: tokenStr = "opMod"; break;
case opPlus: tokenStr = "opPlus"; break; case opPlus: tokenStr = "opPlus"; break;
case opHyphen: tokenStr = "opHyphen"; break; case opHyphen: tokenStr = "opHyphen"; break;
case opSlash: tokenStr = "opSlash"; break; case opSlash: tokenStr = "opSlash"; break;
case opBang: tokenStr = "opBang"; break; case opBang: tokenStr = "opBang"; break;
case opStar: tokenStr = "opStar"; break; case opStar: tokenStr = "opStar"; break;
case opColon: tokenStr = "opColon"; break; case opColon: tokenStr = "opColon"; break;
case opComma: tokenStr = "opComma"; break; case opComma: tokenStr = "opComma"; break;
case opAssign: tokenStr = "opAssign"; break; case opAssign: tokenStr = "opAssign"; break;
case opAccess: tokenStr = "opAccess"; break; case opAccess: tokenStr = "opAccess"; break;
case opSemicolon: tokenStr = "opSemicolon"; break; case opSemicolon: tokenStr = "opSemicolon"; break;
case opCaret: tokenStr = "opCaret"; break; case opCaret: tokenStr = "opCaret"; break;
case opTilde: tokenStr = "opTilde"; break; case opTilde: tokenStr = "opTilde"; break;
case opEq: tokenStr = "opEq"; break; case opEq: tokenStr = "opEq"; break;
case opNeq: tokenStr = "opNeq"; break; case opNeq: tokenStr = "opNeq"; break;
case opLt: tokenStr = "opLt"; break; case opLt: tokenStr = "opLt"; break;
case opGt: tokenStr = "opGt"; break; case opGt: tokenStr = "opGt"; break;
case opLtEq: tokenStr = "opLtEq"; break; case opLtEq: tokenStr = "opLtEq"; break;
case opGtEq: tokenStr = "opGtEq"; break; case opGtEq: tokenStr = "opGtEq"; break;
case opLShift: tokenStr = "opLShift"; break; case opLShift: tokenStr = "opLShift"; break;
case opRShift: tokenStr = "opRShift"; break; case opRShift: tokenStr = "opRShift"; break;
case opBoolAnd: tokenStr = "opBoolAnd"; break; case opBoolAnd: tokenStr = "opBoolAnd"; break;
case opBoolOr: tokenStr = "opBoolOr"; break; case opBoolOr: tokenStr = "opBoolOr"; break;
case opAnd: tokenStr = "opAnd"; break; case opAnd: tokenStr = "opAnd"; break;
case opOr: tokenStr = "opOr"; break; case opOr: tokenStr = "opOr"; break;
case opLParen: tokenStr = "opLParen"; break; case opLParen: tokenStr = "opLParen"; break;
case opRParen: tokenStr = "opRParen"; break; case opRParen: tokenStr = "opRParen"; break;
case opLBracket: tokenStr = "opLBracket"; break; case opLBracket: tokenStr = "opLBracket"; break;
case opRBracket: tokenStr = "opRBracket"; break; case opRBracket: tokenStr = "opRBracket"; break;
case opLSquirly: tokenStr = "opLSquirly"; break; case opLSquirly: tokenStr = "opLSquirly"; break;
case opRSquirly: tokenStr = "opRSquirly"; break; case opRSquirly: tokenStr = "opRSquirly"; break;
case opArrow: tokenStr = "opArrow"; break; case opArrow: tokenStr = "opArrow"; break;
case kwOr: tokenStr = "kwOr"; break; case opPlusAssign: tokenStr = "opPlusAssign"; break;
case kwNot: tokenStr = "kwNot"; break; case opHyphenAssign: tokenStr = "opHyphenAssign"; break;
case kwAnd: tokenStr = "kwAnd"; break; case opStarAssign: tokenStr = "opStarAssign"; break;
case kwIf: tokenStr = "kwIf"; break; case opSlashAssign: tokenStr = "opSlashAssign"; break;
case kwElse: tokenStr = "kwElse"; break; case opModAssign: tokenStr = "opModAssign"; break;
case kwFn: tokenStr = "kwFn"; break; case opAndAssign: tokenStr = "opAndAssign"; break;
case kwEnum: tokenStr = "kwEnum"; break; case opOrAssign: tokenStr = "opOrAssign"; break;
case kwStruct: tokenStr = "kwStruct"; break; case opLShiftAssign: tokenStr = "opLShiftAssign"; break;
case kwVariant: tokenStr = "kwVariant"; break; case opRShiftAssign: tokenStr = "opRShiftAssign"; break;
case kwDef: tokenStr = "kwDef"; break; case opBoolAndAssign: tokenStr = "opBoolAndAssign"; break;
case kwLet: tokenStr = "kwLet"; break; case opBoolORAssign: tokenStr = "opBoolORAssign"; break;
case kwMut: tokenStr = "kwMut"; break; case opMut: tokenStr = "opMut"; break;
case kwFor: tokenStr = "kwFor"; break; case opOpt: tokenStr = "opOpt"; break;
case kwWhile: tokenStr = "kwWhile"; break; case opSliceSize: tokenStr = "opSliceSize"; break;
case kwReturn: tokenStr = "kwReturn"; break; case opPtrSlice: tokenStr = "opPtrSlice"; break;
case kwUnreachable: tokenStr = "kwUnreachable"; break; case opSlicePtr: tokenStr = "opSlicePtr"; break;
case kwDefer: tokenStr = "kwDefer"; break; case opReflect: tokenStr = "opReflect"; break;
case kwErrDefer: tokenStr = "kwErrDefer"; break; case opLabel: tokenStr = "opLabel"; break;
case kwTrue: tokenStr = "kwTrue"; break; case kwUnderscore: tokenStr = "kwUnderscore"; break;
case kwFalse: tokenStr = "kwFalse"; break; case kwOr: tokenStr = "kwOr"; break;
case kwNull: tokenStr = "kwNull"; break; case kwNot: tokenStr = "kwNot"; break;
case kwImport: tokenStr = "kwImport"; break; case kwAnd: tokenStr = "kwAnd"; break;
case kwExport: tokenStr = "kwExport"; break; case kwIf: tokenStr = "kwIf"; break;
case kwModule: tokenStr = "kwModule"; break; case kwElse: tokenStr = "kwElse"; break;
default: tokenStr = "<Undefined>"; break; case kwFn: tokenStr = "kwFn"; break;
case kwEnum: tokenStr = "kwEnum"; break;
case kwStruct: tokenStr = "kwStruct"; break;
case kwDef: tokenStr = "kwDef"; break;
case kwLet: tokenStr = "kwLet"; break;
case kwFor: tokenStr = "kwFor"; break;
case kwLoop: tokenStr = "kwLoop"; break;
case kwBreak: tokenStr = "kwBreak"; break;
case kwContinue: tokenStr = "kwContinue"; break;
case kwWhile: tokenStr = "kwWhile"; break;
case kwMatch: tokenStr = "kwMatch"; break;
case kwSwitch: tokenStr = "kwSwitch"; break;
case kwReturn: tokenStr = "kwReturn"; break;
case kwUnreachable: tokenStr = "kwUnreachable"; break;
case kwDefer: tokenStr = "kwDefer"; break;
case kwErrDefer: tokenStr = "kwErrDefer"; break;
case kwTrue: tokenStr = "kwTrue"; break;
case kwFalse: tokenStr = "kwFalse"; break;
case kwNull: tokenStr = "kwNull"; break;
case kwThis: tokenStr = "kwThis"; break;
case kwImport: tokenStr = "kwImport"; break;
case kwExport: tokenStr = "kwExport"; break;
case kwModule: tokenStr = "kwModule"; break;
case kwUsing: tokenStr = "kwUsing"; break;
default: tokenStr = "<Undefined>"; break;
} }
return std::format("Token{{ {}, {} }}", tokenStr, value.strValue); return std::format("Token{{ {}, {} }}", tokenStr, value.strValue);

View File

@ -129,6 +129,17 @@ namespace arti::lang {
else if (isFirstIdentChar(*iter)) { else if (isFirstIdentChar(*iter)) {
yield readIdentifier(); yield readIdentifier();
} }
else if (*iter == '/') {
if ((iter + 1) != source.end() && *(iter + 1) == '*') {
if (auto ok = skip_comment(); !ok) {
auto err = ok.error();
yield Unexpected<>{err};
}
}
else {
yield readOperator();
}
}
else { else {
yield readOperator(); yield readOperator();
} }
@ -156,6 +167,37 @@ namespace arti::lang {
} }
} }
Expected<void> Tokenizer::skip_comment() {
iter += 2;
column += 2;
bool isEnd = false;
while (iter != source.end()) {
if (*iter == '\n') {
column = 0;
line += 1;
}
else {
column += 1;
if (*iter == '*') {
if ((iter + 1) == source.end()) {
return langException<ExceptCode::ecInvalidComment>(line, column);
}
else if (*(iter + 1) == '/') {
iter += 2;
column += 2;
return {};
}
}
}
++iter;
}
return langException<ExceptCode::ecInvalidComment>(line, column);
}
Expected<Token> Tokenizer::readNumber() { Expected<Token> Tokenizer::readNumber() {
auto stIter = iter; auto stIter = iter;
@ -478,6 +520,9 @@ namespace arti::lang {
{ stIter, iter } { stIter, iter }
}; };
if (tok.strValue.compare("_") == 0) {
tok.value = TokenV::kwUnderscore;
}
if (tok.strValue.compare("or") == 0) { if (tok.strValue.compare("or") == 0) {
tok.value = TokenV::kwOr; tok.value = TokenV::kwOr;
} }
@ -502,24 +547,33 @@ namespace arti::lang {
else if (tok.strValue.compare("struct") == 0) { else if (tok.strValue.compare("struct") == 0) {
tok.value = TokenV::kwStruct; tok.value = TokenV::kwStruct;
} }
else if (tok.strValue.compare("variant") == 0) {
tok.value = TokenV::kwVariant;
}
else if (tok.strValue.compare("def") == 0) { else if (tok.strValue.compare("def") == 0) {
tok.value = TokenV::kwDef; tok.value = TokenV::kwDef;
} }
else if (tok.strValue.compare("let") == 0) { else if (tok.strValue.compare("let") == 0) {
tok.value = TokenV::kwLet; tok.value = TokenV::kwLet;
} }
else if (tok.strValue.compare("mut") == 0) {
tok.value = TokenV::kwMut;
}
else if (tok.strValue.compare("for") == 0) { else if (tok.strValue.compare("for") == 0) {
tok.value = TokenV::kwFor; tok.value = TokenV::kwFor;
} }
else if (tok.strValue.compare("loop") == 0) {
tok.value = TokenV::kwLoop;
}
else if (tok.strValue.compare("break") == 0) {
tok.value = TokenV::kwBreak;
}
else if (tok.strValue.compare("continue") == 0) {
tok.value = TokenV::kwContinue;
}
else if (tok.strValue.compare("while") == 0) { else if (tok.strValue.compare("while") == 0) {
tok.value = TokenV::kwWhile; tok.value = TokenV::kwWhile;
} }
else if (tok.strValue.compare("match") == 0) {
tok.value = TokenV::kwMatch;
}
else if (tok.strValue.compare("switch") == 0) {
tok.value = TokenV::kwSwitch;
}
else if (tok.strValue.compare("return") == 0) { else if (tok.strValue.compare("return") == 0) {
tok.value = TokenV::kwReturn; tok.value = TokenV::kwReturn;
} }
@ -541,6 +595,9 @@ namespace arti::lang {
else if (tok.strValue.compare("null") == 0) { else if (tok.strValue.compare("null") == 0) {
tok.value = TokenV::kwNull; tok.value = TokenV::kwNull;
} }
else if (tok.strValue.compare("this") == 0) {
tok.value = TokenV::kwThis;
}
else if (tok.strValue.compare("import") == 0) { else if (tok.strValue.compare("import") == 0) {
tok.value = TokenV::kwImport; tok.value = TokenV::kwImport;
} }
@ -550,6 +607,9 @@ namespace arti::lang {
else if (tok.strValue.compare("module") == 0) { else if (tok.strValue.compare("module") == 0) {
tok.value = TokenV::kwModule; tok.value = TokenV::kwModule;
} }
else if (tok.strValue.compare("using") == 0) {
tok.value = TokenV::kwUsing;
}
return tok; return tok;
} }
@ -654,6 +714,24 @@ namespace arti::lang {
tm.insert("{", TokenV::opLSquirly); tm.insert("{", TokenV::opLSquirly);
tm.insert("}", TokenV::opRSquirly); tm.insert("}", TokenV::opRSquirly);
tm.insert("->", TokenV::opArrow); tm.insert("->", TokenV::opArrow);
tm.insert("+=", TokenV::opPlusAssign);
tm.insert("-=", TokenV::opHyphenAssign);
tm.insert("*=", TokenV::opStarAssign);
tm.insert("/=", TokenV::opSlashAssign);
tm.insert("%=", TokenV::opModAssign);
tm.insert("&=", TokenV::opAndAssign);
tm.insert("|=", TokenV::opOrAssign);
tm.insert("<<=", TokenV::opLShiftAssign);
tm.insert(">>=", TokenV::opRShiftAssign);
tm.insert("&&=", TokenV::opBoolAndAssign);
tm.insert("||=", TokenV::opBoolORAssign);
tm.insert("$", TokenV::opMut);
tm.insert("?", TokenV::opOpt);
tm.insert(".#", TokenV::opSliceSize);
tm.insert(".[", TokenV::opPtrSlice);
tm.insert(".*", TokenV::opSlicePtr);
tm.insert(".@", TokenV::opReflect);
tm.insert(":=", TokenV::opLabel);
return tm; return tm;
} }