summaryrefslogtreecommitdiff
path: root/src/main/java/no/eliashaugsbakk
diff options
context:
space:
mode:
authorElias Haugsbakk <[email protected]>2026-09-19 13:08:30 +0200
committerElias Haugsbakk <[email protected]>2026-09-19 13:09:51 +0200
commit52a63a487f9d0102db339419aa392f9cd425c0c4 (patch)
treee6969a89960152364a932b4d0cbcfa8aed5115b3 /src/main/java/no/eliashaugsbakk
parent02cd0e25df3cf03b1b69557107cd16e89f3427d6 (diff)
add type declaration to lexer
Diffstat (limited to 'src/main/java/no/eliashaugsbakk')
-rw-r--r--src/main/java/no/eliashaugsbakk/kompilator/tokenization/Lexer.java108
-rw-r--r--src/main/java/no/eliashaugsbakk/kompilator/tokenization/LexerState.java3
-rw-r--r--src/main/java/no/eliashaugsbakk/kompilator/tokenization/TokenType.java17
3 files changed, 88 insertions, 40 deletions
diff --git a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/Lexer.java b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/Lexer.java
index 62d2359..c185650 100644
--- a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/Lexer.java
+++ b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/Lexer.java
@@ -2,8 +2,10 @@ package no.eliashaugsbakk.kompilator.tokenization;
import static java.lang.Character.isLetterOrDigit;
import static no.eliashaugsbakk.kompilator.tokenization.LexerState.IN_STRING;
+import static no.eliashaugsbakk.kompilator.tokenization.LexerState.IN_TYPE;
import static no.eliashaugsbakk.kompilator.tokenization.LexerState.IN_WORD;
import static no.eliashaugsbakk.kompilator.tokenization.LexerState.NORMAL;
+import static no.eliashaugsbakk.kompilator.tokenization.TokenType.ASSIGN;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.EOF;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.IDENTIFIER;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.KEYWORD;
@@ -11,11 +13,14 @@ import static no.eliashaugsbakk.kompilator.tokenization.TokenType.LPAREN;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.RPAREN;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.SEMICOLON;
import static no.eliashaugsbakk.kompilator.tokenization.TokenType.STRING;
+import static no.eliashaugsbakk.kompilator.tokenization.TokenType.TYPE;
+import static no.eliashaugsbakk.kompilator.tokenization.TokenType.TYPE_DECLARATION;
import java.util.ArrayList;
import java.util.List;
public class Lexer {
+ private char current;
private int line = 1;
private int column = 0;
@@ -28,13 +33,12 @@ public class Lexer {
public Lexer(String input) {
this.input = input;
+ this.current = input.charAt(position);
}
public List<Token> tokenize() {
while (position < input.length()) {
- char current = input.charAt(position);
-
if (current == '\n') {
line++;
column = 0;
@@ -42,52 +46,92 @@ public class Lexer {
column++;
}
- if (state == IN_WORD) {
- if (!isLetterOrDigit(current) || Character.isWhitespace(current)) {
- state = NORMAL;
- characterizeWord();
- } else {
- wordBuffer.append(current);
- }
- }
- if (state == NORMAL) {
- if (!Character.isWhitespace(current)) {
- if (current == '"') {
- state = IN_STRING;
- } else if (isLetterOrDigit(current)) {
- state = IN_WORD;
- wordBuffer.append(current);
-
- } else if (current == '(') {
- tokens.add(new Token(LPAREN, Character.toString(current), line, column));
- } else if (current == ')') {
- tokens.add(new Token(RPAREN, Character.toString(current), line, column));
- } else if (current == ';') {
- tokens.add(new Token(SEMICOLON, Character.toString(current), line, column));
- }
+ current = input.charAt(position);
+ if (!Character.isWhitespace(current)) {
+
+ if (state == IN_WORD) {
+ inWord();
+ } else if (state == IN_STRING) {
+ inString();
+ } else if (state == IN_TYPE) {
+ inType();
}
- } else if (state == IN_STRING) {
- if (current == '"') {
- state = NORMAL;
- tokens.add(new Token(STRING, wordBuffer.toString(), line, column));
- wordBuffer.delete(0, wordBuffer.length());
- } else {
- wordBuffer.append(current);
+ if (state == NORMAL) {
+ normal();
}
}
position++;
}
tokens.add(new Token(EOF, "End of File", line, column));
+ // tokens.forEach(token -> IO.println(token.type().toString() + ": " + token.value()));
return tokens;
}
+ private void normal() {
+ if (current == '"') {
+ state = IN_STRING;
+ } else if (isLetterOrDigit(current)) {
+ state = IN_WORD;
+ wordBuffer.append(current);
+
+ } else if (current == ':') {
+ tokens.add(new Token(TYPE_DECLARATION, ":", line, column));
+ state = IN_TYPE;
+
+ } else if (current == '=') {
+ tokens.add(new Token(ASSIGN, "=", line, column));
+ } else if (current == '(') {
+ tokens.add(new Token(LPAREN, Character.toString(current), line, column));
+ } else if (current == ')') {
+ tokens.add(new Token(RPAREN, Character.toString(current), line, column));
+ } else if (current == ';') {
+ tokens.add(new Token(SEMICOLON, Character.toString(current), line, column));
+ }
+ }
+
+ private void inType() {
+ if (current == '=' || current == ';') {
+ state = NORMAL;
+ tokens.add(new Token(TYPE, wordBuffer.toString(), line, column - wordBuffer.length()));
+ clearWordBuffer();
+ } else {
+ wordBuffer.append(current);
+ }
+ }
+
+ private void inString() {
+ if (current == '"') {
+ position++; // skip closing "
+ current = input.charAt(position);
+ state = NORMAL;
+ tokens.add(new Token(STRING, wordBuffer.toString(), line, column));
+ wordBuffer.delete(0, wordBuffer.length());
+ } else {
+ wordBuffer.append(current);
+ }
+ }
+
+ private void inWord() {
+ if (!isLetterOrDigit(current) && current != '_') {
+ state = NORMAL;
+ characterizeWord();
+ } else {
+ wordBuffer.append(current);
+ }
+
+ }
+
private void characterizeWord() {
if (Keywords.KEYWORDS.contains(wordBuffer.toString())) {
tokens.add(new Token(KEYWORD, wordBuffer.toString(), line, column - wordBuffer.length()));
} else {
tokens.add(new Token(IDENTIFIER, wordBuffer.toString(), line, column - wordBuffer.length()));
}
+ clearWordBuffer();
+ }
+
+ private void clearWordBuffer() {
wordBuffer.delete(0, wordBuffer.length());
}
}
diff --git a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/LexerState.java b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/LexerState.java
index 8f0a060..3d77df2 100644
--- a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/LexerState.java
+++ b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/LexerState.java
@@ -3,5 +3,6 @@ package no.eliashaugsbakk.kompilator.tokenization;
public enum LexerState {
NORMAL, // reading regular tokens
IN_STRING, // inside a string (after ")
- IN_WORD // reading a keyword or identifier
+ IN_WORD, // reading a keyword or identifier
+ IN_TYPE, // reading a type (i32, string, ...)
}
diff --git a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/TokenType.java b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/TokenType.java
index 696b46c..1d6382e 100644
--- a/src/main/java/no/eliashaugsbakk/kompilator/tokenization/TokenType.java
+++ b/src/main/java/no/eliashaugsbakk/kompilator/tokenization/TokenType.java
@@ -1,11 +1,14 @@
package no.eliashaugsbakk.kompilator.tokenization;
public enum TokenType {
- KEYWORD, // print, var, if, while, function, etc.
- IDENTIFIER, // variable_1
- STRING, // "Hello, World!"
- LPAREN, // (
- RPAREN, // )
- SEMICOLON, // ;
- EOF // End of File
+ KEYWORD, // print, var, if, while, function, etc.
+ IDENTIFIER, // variable_1
+ STRING, // "Hello, World!"
+ TYPE_DECLARATION, // : (x[:] int = ...)
+ TYPE, // String, i32, i16?, my_type, ... (? makes nullable)
+ ASSIGN, // =
+ LPAREN, // (
+ RPAREN, // )
+ SEMICOLON, // ;
+ EOF // End of File
}