Merge branch 'feat/tokenizer' into 'main'
Add server-side tokenizer See merge request cs108-fs26/Gruppe-13!29
This commit was merged in pull request #185.
This commit is contained in:
@@ -0,0 +1,4 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
/** Represents a raw (unclassified) token in the tokenizer. */
|
||||||
|
public record RawToken(RawTokenType type, String value, int line, int column) {}
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
public enum RawTokenType {
|
||||||
|
WORD,
|
||||||
|
STRING,
|
||||||
|
SEPARATOR,
|
||||||
|
NEWLINE,
|
||||||
|
EOF
|
||||||
|
}
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/** Internal state for the tokenizer. */
|
||||||
|
class State {
|
||||||
|
final String input;
|
||||||
|
int pos;
|
||||||
|
int line;
|
||||||
|
int column;
|
||||||
|
final List<RawToken> tokens;
|
||||||
|
|
||||||
|
State(String input) {
|
||||||
|
this.input = input;
|
||||||
|
this.pos = 0;
|
||||||
|
this.line = 1;
|
||||||
|
this.column = 1;
|
||||||
|
this.tokens = new ArrayList<>();
|
||||||
|
}
|
||||||
|
|
||||||
|
char current() {
|
||||||
|
return input.charAt(pos);
|
||||||
|
}
|
||||||
|
|
||||||
|
char peek() {
|
||||||
|
if (pos + 1 >= input.length()) {
|
||||||
|
return '\0';
|
||||||
|
}
|
||||||
|
return input.charAt(pos + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
void advance() {
|
||||||
|
pos++;
|
||||||
|
column++;
|
||||||
|
}
|
||||||
|
|
||||||
|
boolean isEof() {
|
||||||
|
return pos >= input.length();
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
/** Represents a token in the tokenizer. */
|
||||||
|
public record Token(TokenType type, String value, int line, int column) {}
|
||||||
@@ -0,0 +1,103 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
import java.util.ArrayList;
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
public class TokenClassifier {
|
||||||
|
public static List<Token> classify(List<RawToken> rawTokens) {
|
||||||
|
validateFirstToken(rawTokens);
|
||||||
|
|
||||||
|
List<Token> tokens = new ArrayList<>();
|
||||||
|
|
||||||
|
for (int i = 0; i < rawTokens.size(); i++) {
|
||||||
|
RawToken raw = rawTokens.get(i);
|
||||||
|
|
||||||
|
switch (raw.type()) {
|
||||||
|
case SEPARATOR -> {
|
||||||
|
validateSeparator(rawTokens, i);
|
||||||
|
tokens.add(
|
||||||
|
new Token(TokenType.SEPARATOR, raw.value(), raw.line(), raw.column()));
|
||||||
|
}
|
||||||
|
case WORD -> {
|
||||||
|
TokenType type = resolveWordType(rawTokens, i);
|
||||||
|
tokens.add(new Token(type, raw.value(), raw.line(), raw.column()));
|
||||||
|
}
|
||||||
|
case STRING -> {
|
||||||
|
validateString(rawTokens, i);
|
||||||
|
tokens.add(new Token(TokenType.VALUE, raw.value(), raw.line(), raw.column()));
|
||||||
|
}
|
||||||
|
|
||||||
|
case NEWLINE -> {
|
||||||
|
throw new TokenizerException(
|
||||||
|
"Unexpected newline in request", raw.line(), raw.column());
|
||||||
|
}
|
||||||
|
|
||||||
|
case EOF -> {
|
||||||
|
tokens.add(new Token(TokenType.EOF, raw.value(), raw.line(), raw.column()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return tokens;
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void validateFirstToken(List<RawToken> rawTokens) {
|
||||||
|
if (rawTokens.isEmpty() || rawTokens.get(0).type() != RawTokenType.WORD) {
|
||||||
|
throw new TokenizerException("Expected COMMAND as first token", 1, 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void validateSeparator(List<RawToken> rawTokens, int index) {
|
||||||
|
boolean missingKey = index == 0 || rawTokens.get(index - 1).type() != RawTokenType.WORD;
|
||||||
|
|
||||||
|
boolean missingValue =
|
||||||
|
index + 1 >= rawTokens.size()
|
||||||
|
|| (rawTokens.get(index + 1).type() != RawTokenType.WORD
|
||||||
|
&& rawTokens.get(index + 1).type() != RawTokenType.STRING);
|
||||||
|
|
||||||
|
boolean nextWordIsKey =
|
||||||
|
!missingValue
|
||||||
|
&& rawTokens.get(index + 1).type() == RawTokenType.WORD
|
||||||
|
&& index + 2 < rawTokens.size()
|
||||||
|
&& rawTokens.get(index + 2).type() == RawTokenType.SEPARATOR;
|
||||||
|
|
||||||
|
RawToken separator = rawTokens.get(index);
|
||||||
|
|
||||||
|
if (missingKey) {
|
||||||
|
throw new TokenizerException(
|
||||||
|
"Expected KEY before '='", separator.line(), separator.column());
|
||||||
|
}
|
||||||
|
|
||||||
|
if (missingValue || nextWordIsKey) {
|
||||||
|
throw new TokenizerException(
|
||||||
|
"Expected VALUE after '='", separator.line(), separator.column());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static void validateString(List<RawToken> rawTokens, int index) {
|
||||||
|
boolean afterSeparator =
|
||||||
|
index > 0 && rawTokens.get(index - 1).type() == RawTokenType.SEPARATOR;
|
||||||
|
|
||||||
|
if (!afterSeparator) {
|
||||||
|
RawToken token = rawTokens.get(index);
|
||||||
|
throw new TokenizerException("Unexpected string literal", token.line(), token.column());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
private static TokenType resolveWordType(List<RawToken> rawTokens, int index) {
|
||||||
|
boolean isFirst = index == 0;
|
||||||
|
boolean afterNewline = index > 0 && rawTokens.get(index - 1).type() == RawTokenType.NEWLINE;
|
||||||
|
boolean afterSeparator =
|
||||||
|
index > 0 && rawTokens.get(index - 1).type() == RawTokenType.SEPARATOR;
|
||||||
|
|
||||||
|
if (isFirst || afterNewline) {
|
||||||
|
return TokenType.COMMAND;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (afterSeparator) {
|
||||||
|
return TokenType.VALUE;
|
||||||
|
}
|
||||||
|
|
||||||
|
return TokenType.KEY;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
/** Enumeration of token types used in the tokenizer. */
|
||||||
|
public enum TokenType {
|
||||||
|
COMMAND,
|
||||||
|
KEY,
|
||||||
|
VALUE,
|
||||||
|
SEPARATOR,
|
||||||
|
EOF
|
||||||
|
}
|
||||||
@@ -0,0 +1,139 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
import java.util.List;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* A static utility class for tokenizing input strings into a sequence of tokens.
|
||||||
|
*
|
||||||
|
* <p>Strings can be enclosed in single quotes and may contain escaped single quotes using a
|
||||||
|
* backslash.
|
||||||
|
*/
|
||||||
|
public class Tokenizer {
|
||||||
|
/**
|
||||||
|
* Tokenizes the given input string into a list of tokens.
|
||||||
|
*
|
||||||
|
* @param input the string to tokenize
|
||||||
|
* @return a list of tokens representing the input
|
||||||
|
* @throws TokenizerException if the input contains unexpected characters, unterminated strings,
|
||||||
|
* or other syntax errors
|
||||||
|
*/
|
||||||
|
public static List<RawToken> tokenize(String input) {
|
||||||
|
State state = new State(input);
|
||||||
|
|
||||||
|
while (!state.isEof()) {
|
||||||
|
char c = state.current();
|
||||||
|
|
||||||
|
if (c == ' ' || c == '\t') {
|
||||||
|
state.advance();
|
||||||
|
|
||||||
|
} else if (c == '\n') {
|
||||||
|
readNewline(state);
|
||||||
|
|
||||||
|
} else if (c == '=') {
|
||||||
|
readSeparator(state);
|
||||||
|
|
||||||
|
} else if (c == '\'') {
|
||||||
|
readString(state);
|
||||||
|
|
||||||
|
} else if (Character.isLetterOrDigit(c) || c == '_') {
|
||||||
|
readWord(state);
|
||||||
|
|
||||||
|
} else {
|
||||||
|
throw new TokenizerException(
|
||||||
|
"Unexpected character '" + c + "'", state.line, state.column);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
state.tokens.add(new RawToken(RawTokenType.EOF, null, state.line, state.column));
|
||||||
|
return state.tokens;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads a newline character and creates a NEWLINE token.
|
||||||
|
*
|
||||||
|
* <p>Updates the line counter and resets the column to 1.
|
||||||
|
*
|
||||||
|
* @param state the current parsing state
|
||||||
|
*/
|
||||||
|
private static void readNewline(State state) {
|
||||||
|
state.tokens.add(new RawToken(RawTokenType.NEWLINE, "", state.line, state.column));
|
||||||
|
state.advance();
|
||||||
|
state.line++;
|
||||||
|
state.column = 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads a separator character ('=') and creates a SEPARATOR token.
|
||||||
|
*
|
||||||
|
* @param state the current parsing state
|
||||||
|
*/
|
||||||
|
private static void readSeparator(State state) {
|
||||||
|
state.tokens.add(new RawToken(RawTokenType.SEPARATOR, "=", state.line, state.column));
|
||||||
|
state.advance();
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads a word (sequence of alphanumeric characters and underscores) and creates an appropriate
|
||||||
|
* token (COMMAND, KEY, or VALUE) based on the parsing context.
|
||||||
|
*
|
||||||
|
* <p>The token type is determined by the {@link #resolveWordType(State)} method.
|
||||||
|
*
|
||||||
|
* @param state the current parsing state
|
||||||
|
*/
|
||||||
|
private static void readWord(State state) {
|
||||||
|
int startColumn = state.column;
|
||||||
|
StringBuilder sb = new StringBuilder();
|
||||||
|
|
||||||
|
while (!state.isEof()
|
||||||
|
&& (Character.isLetterOrDigit(state.current()) || state.current() == '_')) {
|
||||||
|
sb.append(state.current());
|
||||||
|
state.advance();
|
||||||
|
}
|
||||||
|
|
||||||
|
state.tokens.add(new RawToken(RawTokenType.WORD, sb.toString(), state.line, startColumn));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Reads a string literal enclosed in single quotes and creates a VALUE token.
|
||||||
|
*
|
||||||
|
* <p>Supports escaped single quotes using backslash notation (\'). Newlines within strings are
|
||||||
|
* properly tracked for line counting.
|
||||||
|
*
|
||||||
|
* @param state the current parsing state
|
||||||
|
* @throws TokenizerException if the string literal is not terminated before end of input
|
||||||
|
*/
|
||||||
|
private static void readString(State state) {
|
||||||
|
int startColumn = state.column;
|
||||||
|
state.advance();
|
||||||
|
StringBuilder sb = new StringBuilder();
|
||||||
|
|
||||||
|
while (true) {
|
||||||
|
if (state.isEof()) {
|
||||||
|
throw new TokenizerException(
|
||||||
|
"Unterminated string literal", state.line, startColumn);
|
||||||
|
}
|
||||||
|
|
||||||
|
char c = state.current();
|
||||||
|
|
||||||
|
if (c == '\\' && state.peek() == '\'') {
|
||||||
|
sb.append('\'');
|
||||||
|
state.advance();
|
||||||
|
state.advance();
|
||||||
|
|
||||||
|
} else if (c == '\'') {
|
||||||
|
state.advance();
|
||||||
|
break;
|
||||||
|
|
||||||
|
} else {
|
||||||
|
if (c == '\n') {
|
||||||
|
state.line++;
|
||||||
|
state.column = 1;
|
||||||
|
}
|
||||||
|
sb.append(c);
|
||||||
|
state.advance();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
state.tokens.add(new RawToken(RawTokenType.STRING, sb.toString(), state.line, startColumn));
|
||||||
|
}
|
||||||
|
}
|
||||||
+21
@@ -0,0 +1,21 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
/** Exception thrown during tokenization. */
|
||||||
|
public class TokenizerException extends RuntimeException {
|
||||||
|
private final int line;
|
||||||
|
private final int column;
|
||||||
|
|
||||||
|
public TokenizerException(String message, int line, int column) {
|
||||||
|
super(message);
|
||||||
|
this.line = line;
|
||||||
|
this.column = column;
|
||||||
|
}
|
||||||
|
|
||||||
|
public int getLine() {
|
||||||
|
return line;
|
||||||
|
}
|
||||||
|
|
||||||
|
public int getColumn() {
|
||||||
|
return column;
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,59 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||||
|
|
||||||
|
class StateTest {
|
||||||
|
@Test
|
||||||
|
void testConstructor() {
|
||||||
|
String input = "hello";
|
||||||
|
State state = new State(input);
|
||||||
|
assertEquals(input, state.input);
|
||||||
|
assertEquals(0, state.pos);
|
||||||
|
assertEquals(1, state.line);
|
||||||
|
assertEquals(1, state.column);
|
||||||
|
assertTrue(state.tokens.isEmpty());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void testCurrent() {
|
||||||
|
String input = "abc";
|
||||||
|
State state = new State(input);
|
||||||
|
assertEquals('a', state.current());
|
||||||
|
state.advance();
|
||||||
|
assertEquals('b', state.current());
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void testPeek() {
|
||||||
|
String input = "abc";
|
||||||
|
State state = new State(input);
|
||||||
|
assertEquals('b', state.peek());
|
||||||
|
state.advance();
|
||||||
|
assertEquals('c', state.peek());
|
||||||
|
state.advance();
|
||||||
|
assertEquals('\0', state.peek()); // End of input
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void testAdvance() {
|
||||||
|
String input = "abc";
|
||||||
|
State state = new State(input);
|
||||||
|
assertEquals(0, state.pos);
|
||||||
|
assertEquals(1, state.column);
|
||||||
|
state.advance();
|
||||||
|
assertEquals(1, state.pos);
|
||||||
|
assertEquals(2, state.column);
|
||||||
|
}
|
||||||
|
|
||||||
|
@Test
|
||||||
|
void testIsEof() {
|
||||||
|
String input = "a";
|
||||||
|
State state = new State(input);
|
||||||
|
assertFalse(state.isEof());
|
||||||
|
state.advance();
|
||||||
|
assertTrue(state.isEof());
|
||||||
|
}
|
||||||
|
}
|
||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
package ch.unibas.dmi.dbis.cs108.casono.server.tokenizer;
|
||||||
|
|
||||||
|
import org.junit.jupiter.api.Test;
|
||||||
|
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||||
|
|
||||||
|
class TokenizerExceptionTest {
|
||||||
|
@Test
|
||||||
|
void testConstructorAndGetters() {
|
||||||
|
String message = "Test error";
|
||||||
|
int line = 5;
|
||||||
|
int column = 10;
|
||||||
|
TokenizerException e = new TokenizerException(message, line, column);
|
||||||
|
assertEquals(message, e.getMessage());
|
||||||
|
assertEquals(line, e.getLine());
|
||||||
|
assertEquals(column, e.getColumn());
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user