mm better token...
This commit is contained in:
parent
834783a279
commit
c6e8882be0
10
documentation/lang/Grammar.md
Normal file
10
documentation/lang/Grammar.md
Normal file
@ -0,0 +1,10 @@
|
|||||||
|
# Grammar
|
||||||
|
Basically just an outline of how to type things out and how things work syntactically.
|
||||||
|
$$
|
||||||
|
\begin{align}
|
||||||
|
[\text{exit}] &\to exit([\text{expr}]);
|
||||||
|
\\
|
||||||
|
[\text{expr}] &\to \text{int\_lit}
|
||||||
|
\end{align}
|
||||||
|
$$
|
||||||
|
### Note: i hate laytec or however you spell it. Its hell.
|
||||||
2
documentation/lang/Limitations.md
Normal file
2
documentation/lang/Limitations.md
Normal file
@ -0,0 +1,2 @@
|
|||||||
|
# Some limits of the lang
|
||||||
|
- Cannot have varible names or funtion names that start with a number, they MUST start with a letter. they however can contain numbers.
|
||||||
@ -1 +0,0 @@
|
|||||||
return 0;
|
|
||||||
1
examples/exitcompact.x
Normal file
1
examples/exitcompact.x
Normal file
@ -0,0 +1 @@
|
|||||||
|
exit 11;
|
||||||
62
src/main.cxx
62
src/main.cxx
@ -4,69 +4,14 @@
|
|||||||
#include <optional>
|
#include <optional>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
|
|
||||||
enum class TokenType {
|
#include "tokenization.hxx"
|
||||||
_return,
|
|
||||||
int_lit,
|
|
||||||
semi,
|
|
||||||
};
|
|
||||||
|
|
||||||
struct Token {
|
|
||||||
TokenType type;
|
|
||||||
std::optional<std::string> value {};
|
|
||||||
};
|
|
||||||
|
|
||||||
std::vector<Token> tokenize(const std::string& str) {
|
|
||||||
std::vector<Token> tokens;
|
|
||||||
|
|
||||||
std::string buffer;
|
|
||||||
for (int i = 0; i < str.length(); i++) {
|
|
||||||
// Words
|
|
||||||
if (std::isalpha(str.at(i))) {
|
|
||||||
buffer.push_back(str.at(i));
|
|
||||||
i++;
|
|
||||||
while (std::isalnum(str.at(i))) {
|
|
||||||
buffer.push_back(str.at(i));
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
i--;
|
|
||||||
if (buffer == "return") {
|
|
||||||
tokens.push_back({.type = TokenType::_return});
|
|
||||||
buffer.clear();
|
|
||||||
continue;
|
|
||||||
} else {
|
|
||||||
std::cerr << "xlang: \033[1;31merror:\033[0m \033[1;37munknown type name \'" << buffer << "\'\033[0m" << std::endl;
|
|
||||||
exit(EXIT_FAILURE);
|
|
||||||
}
|
|
||||||
} else if (std::isdigit(str.at(i))) { // Handle Numbers
|
|
||||||
buffer.push_back(str.at(i));
|
|
||||||
i++;
|
|
||||||
while (std::isdigit(str.at(i))) { // im going insane
|
|
||||||
buffer.push_back(str.at(i));
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
i--;
|
|
||||||
tokens.push_back({.type = TokenType::int_lit, .value = buffer});
|
|
||||||
buffer.clear();
|
|
||||||
} else if (str.at(i) == ';') {
|
|
||||||
tokens.push_back({.type = TokenType::semi});
|
|
||||||
} else if (std::isspace(str.at(i))) { // Handle White Space
|
|
||||||
continue;
|
|
||||||
} else {
|
|
||||||
std::cerr << "xlang: \033[1;31merror:\033[0m \033[1;37munknown type name \'" << buffer << "\'\033[0m" << std::endl;
|
|
||||||
exit(EXIT_FAILURE);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Return mah shit
|
|
||||||
return tokens;
|
|
||||||
}
|
|
||||||
|
|
||||||
std::string tokens_to_asm(std::vector<Token>& tokens) {
|
std::string tokens_to_asm(std::vector<Token>& tokens) {
|
||||||
std::stringstream output;
|
std::stringstream output;
|
||||||
output << "global _start\n_start:\n";
|
output << "global _start\n_start:\n";
|
||||||
for (int i = 0; i < tokens.size(); i++) {
|
for (int i = 0; i < tokens.size(); i++) {
|
||||||
const Token& token = tokens.at(i);
|
const Token& token = tokens.at(i);
|
||||||
if (token.type == TokenType::_return) {
|
if (token.type == TokenType::exit) {
|
||||||
if (i + 1 < tokens.size() && tokens.at(i + 1).type == TokenType::int_lit) {
|
if (i + 1 < tokens.size() && tokens.at(i + 1).type == TokenType::int_lit) {
|
||||||
if (i + 2 < tokens.size() && tokens.at(i + 2).type == TokenType::semi) {
|
if (i + 2 < tokens.size() && tokens.at(i + 2).type == TokenType::semi) {
|
||||||
output << " mov rax, 60\n";
|
output << " mov rax, 60\n";
|
||||||
@ -94,7 +39,8 @@ int main(int argc, char* argv[]) {
|
|||||||
contents = contents_stream.str();
|
contents = contents_stream.str();
|
||||||
}
|
}
|
||||||
|
|
||||||
std::vector<Token> tokens = tokenize(contents);
|
Tokenizer tokenizer(std::move(contents));
|
||||||
|
std::vector<Token> tokens = tokenizer.tokenize();
|
||||||
|
|
||||||
{
|
{
|
||||||
std::fstream output("out.asm", std::ios::out);
|
std::fstream output("out.asm", std::ios::out);
|
||||||
|
|||||||
84
src/tokenization.hxx
Normal file
84
src/tokenization.hxx
Normal file
@ -0,0 +1,84 @@
|
|||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include <iostream>
|
||||||
|
#include <fstream>
|
||||||
|
#include <sstream>
|
||||||
|
#include <optional>
|
||||||
|
#include <vector>
|
||||||
|
|
||||||
|
enum class TokenType {
|
||||||
|
exit,
|
||||||
|
int_lit,
|
||||||
|
semi,
|
||||||
|
};
|
||||||
|
|
||||||
|
struct Token {
|
||||||
|
TokenType type;
|
||||||
|
std::optional<std::string> value {};
|
||||||
|
};
|
||||||
|
|
||||||
|
class Tokenizer {
|
||||||
|
public:
|
||||||
|
inline explicit Tokenizer(const std::string& src) : m_src(std::move(src)) {
|
||||||
|
// Erm erm erm erm erm... i like erm
|
||||||
|
}
|
||||||
|
|
||||||
|
inline std::vector<Token> tokenize() {
|
||||||
|
std::vector<Token> tokens;
|
||||||
|
std::string buffer;
|
||||||
|
|
||||||
|
while (peak().has_value()) {
|
||||||
|
if (std::isalpha(peak().value())) {
|
||||||
|
buffer.push_back(consume());
|
||||||
|
while (peak().has_value() && std::isalnum(peak().value())) {
|
||||||
|
buffer.push_back(consume());
|
||||||
|
}
|
||||||
|
if (buffer == "exit") {
|
||||||
|
tokens.push_back({.type = TokenType::exit});
|
||||||
|
buffer.clear();
|
||||||
|
continue;
|
||||||
|
} else {
|
||||||
|
std::cerr << "xlang: \033[1;31merror:\033[0m \033[1;37munknown type name \'" << buffer << "\'\033[0m" << std::endl;
|
||||||
|
exit(EXIT_FAILURE);
|
||||||
|
}
|
||||||
|
} else if (std::isdigit(peak().value())) {
|
||||||
|
buffer.push_back(consume());
|
||||||
|
while (peak().has_value() && std::isdigit(peak().value())) {
|
||||||
|
buffer.push_back(consume());
|
||||||
|
}
|
||||||
|
tokens.push_back({.type = TokenType::int_lit, .value = buffer});
|
||||||
|
buffer.clear();
|
||||||
|
continue;
|
||||||
|
} else if (peak().value() == ';') {
|
||||||
|
consume();
|
||||||
|
tokens.push_back({.type = TokenType::semi});
|
||||||
|
continue;
|
||||||
|
} else if (std::isspace(peak().value())) {
|
||||||
|
consume();
|
||||||
|
continue;
|
||||||
|
} else {
|
||||||
|
std::cerr << "xlang: \033[1;31merror:\033[0m \033[1;37munknown type name \'" << buffer << "\'\033[0m" << std::endl;
|
||||||
|
exit(EXIT_FAILURE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Return mah shit
|
||||||
|
m_index = 0;
|
||||||
|
return tokens;
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
[[nodiscard]] std::optional<char> peak(int ahead = 1) const {
|
||||||
|
if (m_index + ahead > m_src.length()) {
|
||||||
|
return {};
|
||||||
|
} else {
|
||||||
|
return m_src.at(m_index);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
char consume() {
|
||||||
|
return m_src.at(m_index++);
|
||||||
|
}
|
||||||
|
|
||||||
|
const std::string m_src;
|
||||||
|
int m_index = 0;
|
||||||
|
};
|
||||||
Loading…
x
Reference in New Issue
Block a user