diff --git a/src/main/java/ch/unibas/dmi/dbis/cs108/casono/common/tokenizer/Tokenizer.java b/src/main/java/ch/unibas/dmi/dbis/cs108/casono/common/tokenizer/Tokenizer.java new file mode 100644 index 0000000..667c101 --- /dev/null +++ b/src/main/java/ch/unibas/dmi/dbis/cs108/casono/common/tokenizer/Tokenizer.java @@ -0,0 +1,175 @@ +package ch.unibas.dmi.dbis.cs108.casono.common.tokenizer; + +import java.util.List; + +/** + * A static utility class for tokenizing input strings into a sequence of tokens. + * + *
Strings can be enclosed in single quotes and may contain escaped single quotes
+ * using a backslash.
+ */
+public class Tokenizer {
+ /**
+ * Tokenizes the given input string into a list of tokens.
+ *
+ * @param input the string to tokenize
+ * @return a list of tokens representing the input
+ * @throws TokenizerException if the input contains unexpected characters,
+ * unterminated strings, or other syntax errors
+ */
+ public static List Updates the line counter and resets the column to 1.
+ *
+ * @param state the current parsing state
+ */
+ private static void readNewline(State state) {
+ state.tokens.add(new Token(TokenType.NEWLINE, "", state.line, state.column));
+ state.advance();
+ state.line++;
+ state.column = 1;
+ }
+
+ /**
+ * Reads a separator character ('=') and creates a SEPARATOR token.
+ *
+ * @param state the current parsing state
+ */
+ private static void readSeparator(State state) {
+ state.tokens.add(new Token(TokenType.SEPARATOR, "=", state.line, state.column));
+ state.advance();
+ }
+
+ /**
+ * Reads a word (sequence of alphanumeric characters and underscores) and creates
+ * an appropriate token (COMMAND, KEY, or VALUE) based on the parsing context.
+ *
+ * The token type is determined by the {@link #resolveWordType(State)} method.
+ *
+ * @param state the current parsing state
+ */
+ private static void readWord(State state) {
+ int startColumn = state.column;
+ StringBuilder sb = new StringBuilder();
+
+ while (
+ !state.isEof() &&
+ (Character.isLetterOrDigit(state.current()) || state.current() == '_')
+ ) {
+ sb.append(state.current());
+ state.advance();
+ }
+
+ String word = sb.toString();
+ TokenType type = resolveWordType(state);
+ state.tokens.add(new Token(type, word, state.line, startColumn));
+ }
+
+ /**
+ * Reads a string literal enclosed in single quotes and creates a VALUE token.
+ *
+ * Supports escaped single quotes using backslash notation (\'). Newlines within
+ * strings are properly tracked for line counting.
+ *
+ * @param state the current parsing state
+ * @throws TokenizerException if the string literal is not terminated before end of input
+ */
+ private static void readString(State state) {
+ int startColumn = state.column;
+ state.advance();
+ StringBuilder sb = new StringBuilder();
+
+ while (true) {
+ if (state.isEof()) {
+ throw new TokenizerException(
+ "Unterminated string literal", state.line, startColumn);
+ }
+
+ char c = state.current();
+
+ if (c == '\\' && state.peek() == '\'') {
+ sb.append('\'');
+ state.advance();
+ state.advance();
+
+ } else if (c == '\'') {
+ state.advance();
+ break;
+
+ } else {
+ if (c == '\n') {
+ state.line++;
+ state.column = 1;
+ }
+ sb.append(c);
+ state.advance();
+ }
+ }
+
+ state.tokens.add(new Token(TokenType.VALUE, sb.toString(), state.line, startColumn));
+ }
+
+ /**
+ * Determines the token type for a word based on the parsing context.
+ *
+ * The type is resolved according to these rules:
+ *
+ *
+ *
+ * @param state the current parsing state
+ * @return the determined token type
+ */
+ private static TokenType resolveWordType(State state) {
+ if (state.tokens.isEmpty()) {
+ return TokenType.COMMAND;
+ }
+
+ Token last = state.tokens.get(state.tokens.size() - 1);
+
+ if (last.type() == TokenType.SEPARATOR) {
+ return TokenType.VALUE;
+ }
+
+ if (last.type() == TokenType.NEWLINE) {
+ return TokenType.COMMAND;
+ }
+
+ return TokenType.KEY;
+ }
+}