Language Initial

This commit is contained in:
IlyaShurupov 2024-02-03 12:18:10 +03:00 committed by Ilya Shurupov
parent 73897f11df
commit b8f2380c60
47 changed files with 831 additions and 14 deletions

View file

@ -0,0 +1,525 @@
#pragma once
#include "List.hpp"
#include "Buffer.hpp"
#include "Buffer2D.hpp"
#include "Utils.hpp"
#include "Map.hpp"
#ifdef ENV_OS_WINDOWS
#undef min
#undef max
#endif
namespace tp {
extern ModuleManifest gModuleTokenizer;
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class TransitionMatrix;
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class DFA;
// Non-Deterministic Finite-State Automata
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class NFA {
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
public:
struct Vertex;
private:
struct Edge {
Vertex* mVertex = nullptr;
bool mConsumesSymbol = false;
Range<tAlphabetType> mAcceptingRange;
bool mAcceptsAll = false;
bool mExclude = false;
bool isTransition(const tAlphabetType& symbol) {
if (symbol == 0) return false;
if (!mConsumesSymbol || mAcceptsAll) return true;
bool const in_range = (symbol >= mAcceptingRange.mBegin && symbol <= mAcceptingRange.mEnd);
return in_range != mExclude;
}
};
public:
struct Vertex {
List<Edge> edges;
tStateType termination_state = tNoStateVal;
#ifdef ENV_BUILD_DEBUG
ualni debug_idx = 0;
#endif
ualni flag = 0;
};
private:
friend DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
List<Vertex> mVertices;
Vertex* mStart = nullptr;
public:
NFA() = default;
Vertex* addVertex() {
auto node = mVertices.newNode();
#ifdef ENV_BUILD_DEBUG
node->data.debug_idx = mVertices.length() + 1;
#endif
mVertices.pushBack(node);
return &node->data;
}
void addTransition(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool consumes, bool accepts_all, bool exclude) {
Edge edge;
edge.mVertex = to;
edge.mConsumesSymbol = consumes;
edge.mAcceptingRange = range;
edge.mExclude = exclude;
edge.mAcceptsAll = accepts_all;
from->edges.pushBack(edge);
}
void setStartVertex(Vertex* start) {
mStart = start;
}
[[nodiscard]] Vertex* getStartVertex() const {
return mStart;
}
void setVertexState(Vertex* vertex, tStateType state) {
vertex->termination_state = state;
}
[[nodiscard]] bool isValid() const {
if (!mStart) {
return false;
}
return true;
}
Range<tAlphabetType> getAlphabetRange() const {
tAlphabetType start = 0, end = 0;
Range<tAlphabetType> all_range(std::numeric_limits<tAlphabetType>::min(), std::numeric_limits<tAlphabetType>::max());
bool first = true;
for (auto vertex : mVertices) {
for (auto edge : vertex.data().edges) {
if (!edge.data().mConsumesSymbol) continue;
auto const& tran_range = edge.data().mAcceptingRange;
if (edge.data().mAcceptsAll || edge.data().mExclude) return all_range;
if (tran_range.mBegin < start || first) start = tran_range.mBegin;
if (tran_range.mEnd > end || first) end = tran_range.mEnd;
first = false;
}
}
return Range<tAlphabetType>( start, end + 1 );
}
// vertices that are reachable from initial set with no input consumption (E-transitions)
// does not include initial set
void closure(const List<Vertex*>& set, List<Vertex*>& closure, ualni unique_call_id) {
List<Vertex*> marked;
marked = set;
while (marked.length()) {
auto first = marked.first()->data;
if (first->flag != unique_call_id) {
first->flag = unique_call_id;
closure.pushBack(first);
}
for (auto edge : first->edges) {
if (!edge.data().mConsumesSymbol && edge.data().mVertex->flag != unique_call_id) {
marked.pushBack(edge.data().mVertex);
}
}
marked.popFront();
}
}
// vertices that are reachable from initial set with symbol transition
void move(const List<Vertex*>& set, List<Vertex*>& reachable, tAlphabetType symbol, ualni unique_call_id) {
for (auto vertex : set) {
for (auto edge : vertex->edges) {
if (!edge.data().mConsumesSymbol) continue;
bool transition = edge.data().isTransition(symbol);
if (transition && edge.data().mVertex->flag != unique_call_id) {
edge.data().mVertex->flag = unique_call_id;
reachable.pushBack(edge.data().mVertex);
}
}
}
}
};
// Deterministic Finite-State Automata
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class DFA {
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
friend TransitionMatrix<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
struct Vertex {
struct Edge {
Vertex* vertex = nullptr;
tAlphabetType transition_code = nullptr;
};
List<Edge> edges;
tStateType termination_state = tNoStateVal;
bool marked = false;
};
List<Vertex> mVertices;
Vertex* mStart = nullptr;
const Vertex* mIter = nullptr;
Range<tAlphabetType> mAlphabetRange;
bool mTrapState = false;
typedef typename NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>::Vertex NState;
public:
explicit DFA(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& nfa) {
if (!nfa.isValid()) {
return;
}
mAlphabetRange = nfa.getAlphabetRange();
struct DStateKey {
const List<NState*>* nStates;
bool operator==(const DStateKey& in) const {
if (nStates->length() != in.nStates->length()) {
return false;
}
// FIXME : make linear time
for (auto state : *nStates) {
bool found = false;
for (auto in_state : *in.nStates) {
if (state.data() == in_state.data()) {
found = true;
break;
}
}
if (!found) {
return false;
}
}
return true;
}
static ualni dStateHashFunc(DStateKey key) {
alni out = 0;
for (auto state : *key.nStates) {
out += alni(state.data());
}
return out;
};
};
// all NFA states that are reachable from initial DFA State for specific symbol in alphabet
struct DState {
struct DTransition {
DState* state = nullptr;
tAlphabetType accepting_code;
};
List<NState*> nStates;
List<DTransition> transitions;
Vertex* dVertex= nullptr; // relevant DFA vertex
#ifdef ENV_BUILD_DEBUG
ualni debug_idx = 0;
#endif
};
// includes closure of NFA start state by definition
auto start_state = new DState();
nfa.closure({ nfa.getStartVertex() }, start_state->nStates, ualni(start_state));
Map<DStateKey, DState*, DefaultAllocator, DStateKey::dStateHashFunc, 256> dStates;
dStates.put({ &start_state->nStates }, start_state);
List<DState*> working_set = { start_state };
// while there is items to work with
auto currentDState = working_set.first();
while (currentDState) {
// check all possible transitions for any symbol
for (auto symbol : mAlphabetRange) {
List<NState*> reachableNStates;
nfa.move(currentDState->data->nStates, reachableNStates, symbol, ualni(currentDState + symbol));
nfa.closure(reachableNStates, reachableNStates, ualni(currentDState + symbol));
if (!reachableNStates.length()) {
continue;
}
DState* targetDState = nullptr;
// check if set of all reachable NFA states already forms existing DFA state
auto idx = dStates.presents({ &reachableNStates});
if (idx) {
targetDState = dStates.getSlotVal(idx);
}
if (!targetDState) {
// register new DFA state
targetDState = new DState();
#ifdef ENV_BUILD_DEBUG
targetDState->debug_idx = dStates.size();
#endif
targetDState->nStates = reachableNStates;
// append to working stack
working_set.pushBack(targetDState);
dStates.put({ &targetDState->nStates }, targetDState);
}
// add transition to DFA state
currentDState->data->transitions.pushBack({targetDState, symbol });
}
working_set.popFront();
currentDState = working_set.first();
}
// create own vertices
for (auto node : dStates) {
tStateType state = tNoStateVal;
for (auto iter : node->val->nStates) {
if (iter->termination_state != tNoStateVal) {
state = iter->termination_state;
break;
}
}
node->val->dVertex = addVertex(state);
}
// connect all vertices
for (auto node : dStates) {
for (auto edge : node->val->transitions) {
addTransition(node->val->dVertex, edge.data().state->dVertex, edge.data().accepting_code);
}
}
// set the starting vertex
mStart = start_state->dVertex;
// cleanup
for (auto node : dStates) {
delete node->val;
}
collapseEquivalentVertices();
mAlphabetRange = getAlphabetRange();
}
tStateType move(tAlphabetType symbol) {
if (mTrapState || !mIter) {
return tNoStateVal;
}
for (auto edge : mIter->edges) {
if (edge.data().transition_code == symbol) {
mIter = edge.data().vertex;
return mIter->termination_state;
}
}
mTrapState = true;
return tNoStateVal;
}
void start() {
mIter = mStart;
mTrapState = false;
}
[[nodiscard]] uhalni nVertices() const {
return (uhalni) mVertices.length();
}
[[nodiscard]] Range<tAlphabetType> getRange() const {
return mAlphabetRange;
}
private:
[[nodiscard]] Range<tAlphabetType> getAlphabetRange() const {
Range<tAlphabetType> out;
for (auto vertex : mVertices) {
vertex.data().marked = false;
}
bool first = true;
getAlphabetRangeUtil(mStart, &out, first);
out.mEnd++;
return out;
}
void getAlphabetRangeUtil(Vertex* vert, Range<tAlphabetType>* out, bool& first) const {
vert->marked = true;
for (auto edge : vert->edges) {
auto const code = edge.data().transition_code;
if (first) {
*out = { code, code };
first = false;
}
if (code < out->mBegin) {
out->mBegin = code;
}
if (code > out->mEnd) {
out->mEnd = code;
}
if (!edge.data().vertex->marked) {
getAlphabetRangeUtil(edge.data().vertex, out, first);
}
}
}
void collapseEquivalentVertices() {
// TODO
}
Vertex* addVertex(tStateType state) {
auto node = mVertices.addNodeBack();
node->data.termination_state = state;
return &node->data;
}
void addTransition(Vertex* from, Vertex* to, tAlphabetType transition_symbol) {
from->edges.pushBack({ to, transition_symbol });
}
};
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class TransitionMatrix {
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
Buffer2D<ualni> mTransitions;
Buffer<tStateType> mStates;
Range<tAlphabetType> mSymbolRange = { 0, 0 };
ualni mIter = 0;
ualni mIterPrev = 0;
ualni mStart = 0;
public:
TransitionMatrix() = default;
auto getStates() const { return &mStates; }
auto getTransitions() const { return &mTransitions; }
auto getStart() const { return mStart; }
void construct(const DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& dfa) {
mSymbolRange = dfa.getRange();
auto range_len = ualni(mSymbolRange.mEnd - mSymbolRange.mBegin);
auto sizeX = range_len ? range_len : 1;
auto sizeY = (ualni) (dfa.nVertices() + 1);
mTransitions.reserve({ sizeX, sizeY });
mTransitions.assign(dfa.nVertices());
mStates.reserve(sizeY);
ualni idx = 0;
for (auto vertex : dfa.mVertices) {
auto state = vertex.data().termination_state;
mStates[idx] = state;
idx++;
}
mStates[dfa.nVertices()] = tFailedStateVal;
idx = 0;
for (auto vertex : dfa.mVertices) {
if (&vertex.data() == dfa.mStart) {
mStart = mIter = mIterPrev = idx;
}
idx++;
}
ualni vertexIdx = 0;
for (auto vertex : dfa.mVertices) {
for (auto edge : vertex.data().edges) {
ualni vertex2Idx = 0;
for (auto vertex2 : dfa.mVertices) {
if (edge.data().vertex == &vertex2.data()) break;
vertex2Idx++;
}
auto const code = edge.data().transition_code;
mTransitions.set( { (ualni) (code - mSymbolRange.mBegin), (ualni) vertexIdx }, vertex2Idx);
}
vertexIdx++;
}
}
bool isTrapped() {
return mStates[mIter] == tFailedStateVal;
}
tStateType move(tAlphabetType symbol) {
if (symbol >= mSymbolRange.mBegin && symbol < mSymbolRange.mEnd) {
mIter = mTransitions.get({ (ualni) (symbol - mSymbolRange.mBegin), (ualni) mIter });
}
else {
mIter = mStates.size() - 1;
}
if (mIterPrev == mStart) {
if (mStates[mIter] == tFailedStateVal) {
reset();
return tFailedStateVal;
}
else {
mIterPrev = mIter;
return tNoStateVal;
}
}
else {
if (mStates[mIter] == tFailedStateVal) {
if (mStates[mIterPrev] != tNoStateVal) {
auto out = mStates[mIterPrev];
reset();
return out;
}
else {
reset();
return tFailedStateVal;
}
}
else {
mIterPrev = mIter;
return tNoStateVal;
}
}
mIterPrev = mIter;
return mStates[mIter];
}
void reset() {
mIter = mStart;
mIterPrev = mStart;
}
};
}

View file

@ -0,0 +1,519 @@
#pragma once
#include "AutomataGraph.h"
namespace tp {
extern ModuleManifest gModuleTokenizer;
}
namespace tp::RegEx {
struct AstNode {
enum Type {
NONE,
ANY,
OR,
IF,
CLASS,
COMPOUND,
REPEAT,
VAL,
} mType = NONE;
AstNode() = default;
virtual ~AstNode() = default;
};
template <typename tAlphabetType>
struct AstVal : public AstNode {
explicit AstVal(tAlphabetType val) : mVal(val) { mType = VAL; }
~AstVal() override = default;
tAlphabetType mVal;
};
struct AstCompound : public AstNode {
AstCompound() { mType = COMPOUND; }
~AstCompound() override {
for (auto iter : mChilds) {
delete iter.data();
}
mChilds.removeAll();
}
List<AstNode*> mChilds;
};
struct AstAlternation : public AstNode {
AstAlternation() { mType = OR; }
~AstAlternation() override {
delete mFirst;
delete mSecond;
}
AstNode* mFirst = nullptr;
AstNode* mSecond = nullptr;
};
struct AstIf : public AstNode {
AstIf() { mType = IF; }
~AstIf() override { delete mNode; }
AstNode* mNode = nullptr;
};
struct AstAny : public AstNode {
AstAny() { mType = ANY; }
~AstAny() override = default;
};
struct AstRepetition : public AstNode {
AstRepetition() { mType = REPEAT; }
~AstRepetition() override { delete mNode; }
AstNode* mNode = nullptr;
bool mPlus = false;
};
template <typename tAlphabetType>
struct AstClass : public AstNode {
AstClass() { mType = CLASS; }
~AstClass() override { mRanges.removeAll(); }
List<Range<tAlphabetType>> mRanges;
bool mExclude = false;
};
struct ParseError {
const char* description = nullptr;
uhalni offset = 0;
[[nodiscard]] bool isError() const { return description != nullptr; }
};
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal>
class Parser {
enum TokType : uint1 {
TOK_COMPOUND_START = 0,
TOK_COMPOUND_END,
TOK_CLASS_START,
TOK_CLASS_END,
TOK_CLASS_START_EXCLUDE,
TOK_CLASS_END_EXCLUDE,
TOK_OR,
TOK_IF,
TOK_ANY,
TOK_REPEAT,
TOK_REPEAT_PLUS,
TOK_HYPHEN,
TOK_SPECIALS_END_,
TOK_VAL,
TOK_NONE,
};
tAlphabetType SpecialSymbols[TOK_SPECIALS_END_] = {
'(', ')', '[', ']', '{', '}', '|', '?', '.', '*', '+', '-',
};
tAlphabetType mEscapeSymbol = '\\';
struct Token {
TokType type;
tAlphabetType val;
};
const tAlphabetType* mSource = nullptr;
uhalni mOffset = 0;
Token mCurToken;
uhalni mTokLength = 0;
public:
ParseError mError;
// regular expression must be a zero termination string
AstCompound* parse(const tAlphabetType* regex) {
mSource = regex;
return parseRegEx();
}
private:
AstCompound* parseRegEx() {
auto out = new AstCompound();
for (AstNode* node = parseElement(); node; node = parseElement()) {
out->mChilds.pushBack(node);
}
if (!out->mChilds.length()) {
genError("Expected A Expression");
}
if (mError.description) {
delete out;
return nullptr;
}
return out;
}
AstNode* parseElement() {
AstNode* out = nullptr;
switch (readTok().type) {
case TOK_COMPOUND_START: out = parseCompound(); break;
case TOK_CLASS_START: out = parseClass(); break;
case TOK_CLASS_START_EXCLUDE: out = parseClass(true); break;
case TOK_ANY: out = parseAny(); break;
case TOK_VAL: out = parseVal(); break;
case TOK_NONE: { discardTok(); return nullptr; };
default: break;
}
if (!out) {
discardTok();
return nullptr;
}
switch (readTok().type) {
case TOK_OR: out = parseAlternation(out); break;
case TOK_REPEAT: out = parseRepetition(out); break;
case TOK_REPEAT_PLUS: out = parseRepetition(out, true); break;
case TOK_IF: out = parseIf(out); break;
case TOK_NONE: break;
default: { discardTok(); }
}
return out;
}
AstCompound* parseCompound() {
auto out = new AstCompound();
for (AstNode* node = parseElement(); node; node = parseElement()) {
out->mChilds.pushBack(node);
}
if (readTok().type != TOK_COMPOUND_END) {
genError("Expected Compound End");
}
if (mError.description) {
delete out;
return nullptr;
}
return out;
}
AstClass<tAlphabetType>* parseClass(bool exclude = false) {
auto out = new AstClass<tAlphabetType>();
out->mExclude = exclude;
auto& ranges = out->mRanges;
readTok();
READ_VAL:
if (mCurToken.type != TOK_VAL) {
delete out;
genError("Expected A Value");
return nullptr;
}
char range_start = mCurToken.val;
readTok();
if (mCurToken.type != TOK_HYPHEN) {
delete out;
genError("Expected A Range");
return nullptr;
}
readTok();
if (mCurToken.type != TOK_VAL) {
delete out;
genError("Expected A Value");
return nullptr;
}
char range_end = mCurToken.val;
ranges.pushBack({ range_start, range_end });
readTok();
if ((mCurToken.type == TOK_CLASS_END && !exclude) || (mCurToken.type == TOK_CLASS_END_EXCLUDE && exclude)) {
return out;
}
else {
goto READ_VAL;
}
}
AstAny* parseAny() {
return new AstAny();
}
AstVal<tAlphabetType>* parseVal() {
auto out = new AstVal<tAlphabetType>(mCurToken.val);
return out;
}
AstAlternation* parseAlternation(AstNode* left) {
auto right = parseElement();
if (!right) {
genError("Expected Alternation right Side");
delete left;
return nullptr;
}
auto out = new AstAlternation();
out->mFirst = left;
out->mSecond = right;
return out;
}
AstRepetition* parseRepetition(AstNode* left, bool plus = false) {
auto out = new AstRepetition();
out->mNode = left;
out->mPlus = plus;
return out;
}
AstIf* parseIf(AstNode* left) {
auto out = new AstIf();
out->mNode = left;
return out;
}
void genError(const char* desc) {
mError = { desc, mOffset };
}
Token& readTok() {
const tAlphabetType* crs = mSource + mOffset;
// zero termination string
if (*crs == 0) {
mCurToken.type = TOK_NONE;
return mCurToken;
}
mTokLength = 1;
mCurToken.type = TOK_VAL;
mCurToken.val = crs[0];
if (crs[0] == mEscapeSymbol) {
mCurToken.val = crs[1];
mTokLength = 2;
}
else {
for (uhalni tok = 0; tok < TOK_SPECIALS_END_; tok++) {
if (SpecialSymbols[tok] == mCurToken.val) {
mCurToken.type = TokType(tok);
break;
}
}
}
mOffset += mTokLength;
return mCurToken;
}
void discardTok() {
mOffset -= mTokLength;
mTokLength = 0;
}
};
template <typename tStateType>
struct CompileError {
ParseError mParseError;
uhalni mRuleIndex = 0;
tStateType mRuleState;
const char* description = nullptr;
[[nodiscard]] bool isError() const { return description; }
};
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
class Compiler {
typedef NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> Graph;
typedef typename Graph::Vertex Vertex;
typedef Parser<tAlphabetType, tStateType, tNoStateVal> Parser;
struct Node {
Vertex* left = nullptr;
Vertex* right = nullptr;
};
private:
Graph* mGraph = nullptr;
public:
CompileError<tStateType> mError;
Node compile(Graph& graph, const tAlphabetType* regex, tStateType state) {
mGraph = &graph;
return compileUtil(regex, state);
}
Node compile(Graph& aGraph, InitialierList<Pair<const tAlphabetType*, tStateType>> aRules) {
mGraph = &aGraph;
auto left = mGraph->addVertex();
auto right = mGraph->addVertex();
halni idx = 0;
for (auto rule: aRules) {
auto node = compileUtil(rule.head, rule.tail);
if (!(node.left && node.right)) {
mError.mRuleIndex = idx;
return {};
}
transitionAny(left, node.left);
transitionAny(node.right, right);
idx++;
}
mGraph->setStartVertex(left);
return { left, right };
}
private:
Node compileUtil(const tAlphabetType* regex, tStateType state) {
Parser parser;
auto astNode = parser.parse(regex);
if (parser.mError.isError()) {
mGraph->setStartVertex(nullptr);
mError.description = "Parsing Of Regular Expression Failed";
mError.mRuleState = state;
mError.mParseError = parser.mError;
return {};
}
auto node = compileNode(astNode, nullptr, nullptr);
delete astNode;
mGraph->setVertexState(node.right, state);
mGraph->setStartVertex(node.left);
return node;
}
Node compileVal(AstVal<tAlphabetType>* val, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
auto left = aLeft ? aLeft : mGraph->addVertex();
auto right = aRight ? aRight : mGraph->addVertex();
transitionVal(left, right, val->mVal);
return { left, right };
}
Node compileAlternation(AstAlternation* alt, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
auto first_node = compileNode(alt->mFirst, aLeft, aRight);
auto second_node = compileNode(alt->mSecond);
transitionAny(first_node.left, second_node.left);
transitionAny(second_node.right, first_node.right);
return first_node;
}
Node compileAny(AstAny*, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
auto left = aLeft ? aLeft : mGraph->addVertex();
auto right = aRight ? aRight : mGraph->addVertex();
transitionAny(left, right, true);
return { left, right };
}
Node compileRepeat(AstRepetition* repeat, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
if (repeat->mPlus) {
auto middle = mGraph->addVertex();
auto left_node = compileNode(repeat->mNode, aLeft, middle);
auto right_node = compileNode(repeat->mNode, middle, aRight);
transitionAny(right_node.right, right_node.left);
transitionAny(right_node.left, right_node.right);
return { left_node.left, right_node.right };
}
else {
auto node = compileNode(repeat->mNode, aLeft, aRight);
transitionAny(node.right, node.left);
transitionAny(node.left, node.right);
return node;
}
}
Node compileIf(AstIf* ifNode, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
auto node = compileNode(ifNode->mNode, aLeft, aRight);
transitionAny(node.left, node.right);
return node;
}
Node compileClass(AstClass<tAlphabetType>* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
auto left = aLeft ? aLeft : mGraph->addVertex();
auto right = aRight ? aRight : mGraph->addVertex();
if (node->mRanges.length() == 1) {
auto const& range = node->mRanges.first()->data;
transitionRange(left, right, { range.mBegin, range.mEnd }, node->mExclude);
return { left, right };
}
for (auto range : node->mRanges) {
auto middle = mGraph->addVertex();
transitionRange(left, middle, { range.data().mBegin, range.data().mEnd }, node->mExclude);
transitionAny(middle, right);
}
return { left, right };
}
Node compileCompound(AstCompound* compound, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
Vertex* left = nullptr;
Vertex* rigth = nullptr;
ualni idx = 0;
for (auto child : compound->mChilds) {
auto pass_left = idx == 0 ? aLeft : rigth;
auto pass_right = idx == compound->mChilds.length() - 1 ? aRight : nullptr;
auto node = compileNode(child.data(), pass_left, pass_right);
if (!left) left = node.left;
rigth = node.right;
idx++;
}
return { left, rigth };
}
Node compileNode(AstNode* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
switch (node->mType) {
case AstNode::CLASS: return compileClass((AstClass<tAlphabetType>*)node, aLeft, aRight);
case AstNode::COMPOUND: return compileCompound((AstCompound*)node, aLeft, aRight);
case AstNode::IF: return compileIf((AstIf*)node, aLeft, aRight);
case AstNode::REPEAT: return compileRepeat((AstRepetition*)node, aLeft, aRight);
case AstNode::ANY: return compileAny((AstAny*)node, aLeft, aRight);
case AstNode::OR: return compileAlternation((AstAlternation*)node, aLeft, aRight);
case AstNode::VAL: return compileVal((AstVal<tAlphabetType>*)node, aLeft, aRight);
case AstNode::NONE:
break;
}
ASSERT(0)
return {};
}
void transitionAny(Vertex* from, Vertex* to, bool consumes = false) {
mGraph->addTransition(from, to, {}, consumes, true, false);
}
void transitionVal(Vertex* from, Vertex* to, tAlphabetType val) {
mGraph->addTransition(from, to, { val, val }, true, false, false);
}
void transitionRange(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool exclude) {
mGraph->addTransition(from, to, range, true, false, exclude);
}
};
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
CompileError<tStateType> compile(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out, const tAlphabetType* regex, tStateType state) {
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
compiler.compile(out, regex, state);
return compiler.mError;
}
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
CompileError<tStateType> compile(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out, const InitialierList<Pair<const tAlphabetType*, tStateType>>& rules) {
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
compiler.compile(out, rules);
return compiler.mError;
}
}

View file

@ -0,0 +1,184 @@
#pragma once
#include "RegularExpression.h"
#include "Strings.hpp"
namespace tp {
extern ModuleManifest gModuleTokenizer;
template <typename tAlphabetType, typename tTokType, tTokType tNoTokVal, tTokType tFailedTokVal>
class Tokenizer {
TransitionMatrix<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> mTransitionMatrix;
RegEx::CompileError<tTokType> mError;
bool scanFailed() { return mTransitionMatrix.isTrapped(); }
public:
// Some useful RE to be reused
static constexpr const tAlphabetType* etherRE = "\n|\t| |\r";
static constexpr const tAlphabetType* intRE = "((\\-)|(\\+))?[0-9]+i?";
static constexpr const tAlphabetType* floatRE = R"(((\-)|(\+))?([0-9]+)(\.)([0-9]*)?f?)";
static constexpr const tAlphabetType* commentBlockRE = R"((/\*){\*-\*}*(\*/))";
static constexpr const tAlphabetType* stringRE = R"("{"-"}*")";
static constexpr const tAlphabetType* idRE = "([a-z]|[A-Z]|_)+([a-z]|[A-Z]|[0-9]|_)*";
public:
Tokenizer() { MODULE_SANITY_CHECK(gModuleTokenizer) }
void build(const InitialierList<Pair<const tAlphabetType*, tTokType>>& rules) {
NFA<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> nfa;
mError = RegEx::compile(nfa, rules);
if (mError.isError()) {
return;
}
DFA<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> dfa(nfa);
mTransitionMatrix.construct(dfa);
}
[[nodiscard]] bool isBuild() const { return !mError.isError(); }
auto getMatrix() const { return &mTransitionMatrix; }
const RegEx::CompileError<tTokType>& getBuildError() { return mError; }
void resetMatrix() { mTransitionMatrix.reset(); }
tTokType advanceSymbol(tAlphabetType symbol) { return mTransitionMatrix.move(symbol); }
tTokType advanceToken(const tAlphabetType* source, ualni source_len, ualni* token_len) {
tTokType out = tNoTokVal;
*token_len = 0;
for (ualni idx = 0; idx < source_len; idx++) {
out = advanceSymbol(source[idx]);
if (out != tNoTokVal) {
*token_len = idx;
break;
}
}
return out;
}
~Tokenizer() = default;
};
template <typename tAlphabetType, typename tTokType, tTokType tNoTokVal, tTokType tFailedTokVal, tTokType tSourceEndTokVal>
class SimpleTokenizer {
public:
typedef Tokenizer<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> tTokenizer;
private:
tTokenizer mTokenizer;
const tAlphabetType* mSource = nullptr;
ualni mLastTokLen = 0;
ualni mSourceLen = 0;
ualni mAdvancedOffset = 0;
public:
struct Cursor {
const tAlphabetType* mSource = nullptr;
ualni mAdvancedOffset = 0;
const tAlphabetType* str() { return mSource + mAdvancedOffset; }
struct Location2D {
ualni line = 0;
ualni character = 0;
};
Location2D get2DLocation() {
Location2D out;
typedef typename StringLogic<tAlphabetType>::Index Idx;
Buffer<Idx> offsets;
StringLogic<tAlphabetType>::calcLineOffsets(mSource, StringLogic<tAlphabetType>::calcLength(mSource), offsets);
for (ualni idx = 1; idx < offsets.size(); idx++) {
if (offsets[idx] > mAdvancedOffset) {
out.line = idx - 1;
break;
}
}
out.character = mAdvancedOffset - offsets[out.line];
return out;
}
};
SimpleTokenizer() = default;
auto getTokenizer() const { return &mTokenizer; }
void build(const InitialierList<Pair<const tAlphabetType*, tTokType>>& rules) { mTokenizer.build(rules); }
[[nodiscard]] bool isBuild() const { return mTokenizer.isBuild(); }
const RegEx::CompileError<tTokType>& getBuildError() { return mTokenizer.getBuildError(); }
void bindSource(const tAlphabetType* source) {
mSource = source;
while (mSource[mSourceLen])
mSourceLen++;
mSourceLen++;
}
[[nodiscard]] bool isInputLeft() const {
if (!mSource) {
return false;
}
if (mSource[mAdvancedOffset] == 0) {
return false;
}
return true;
}
Cursor getCursor() const { return { mSource, mAdvancedOffset }; }
Cursor getCursorPrev() const { return { mSource, mAdvancedOffset - mLastTokLen }; }
void setCursor(const Cursor& crs) {
mAdvancedOffset = crs.mAdvancedOffset;
mLastTokLen = 0;
}
tTokType readTok() {
if (mSourceLen == mAdvancedOffset + 1) {
return tSourceEndTokVal;
}
tTokType out = mTokenizer.advanceToken(mSource + mAdvancedOffset, mSourceLen - mAdvancedOffset, &mLastTokLen);
mAdvancedOffset += mLastTokLen;
return out;
}
tTokType lookupTok() {
auto out = readTok();
discardTok();
return out;
}
void discardTok() {
mTokenizer.resetMatrix();
mAdvancedOffset -= mLastTokLen;
mLastTokLen = 0;
}
void skipTok() { readTok(); }
void reset() {
mAdvancedOffset = mLastTokLen = 0;
mTokenizer.resetMatrix();
}
[[nodiscard]] ualni lastTokLEn() const { return mLastTokLen; }
String extractVal() {
auto crs = getCursorPrev();
String out;
out.resize(mLastTokLen);
memCopy(out.write(), crs.str(), mLastTokLen);
return out;
}
};
}