Reuse Regular Automata functionality
This commit is contained in:
parent
b3e5bf0941
commit
56d63f2f18
17 changed files with 1033 additions and 1058 deletions
|
|
@ -1,15 +0,0 @@
|
||||||
project(Automatas)
|
|
||||||
|
|
||||||
### ---------------------- Static Library --------------------- ###
|
|
||||||
file(GLOB SOURCES "./private/*.cpp" "./private/*/*.cpp")
|
|
||||||
file(GLOB HEADERS "./public/*.hpp" "./public/*/*.hpp")
|
|
||||||
add_library(${PROJECT_NAME} STATIC ${SOURCES} ${HEADERS})
|
|
||||||
target_include_directories(${PROJECT_NAME} PUBLIC public/)
|
|
||||||
target_link_libraries(${PROJECT_NAME} PUBLIC Utils)
|
|
||||||
|
|
||||||
### -------------------------- Tests -------------------------- ###
|
|
||||||
enable_testing()
|
|
||||||
file(GLOB TEST_SOURCES "./tests/*.cpp")
|
|
||||||
add_executable(${PROJECT_NAME}Tests ${TEST_SOURCES})
|
|
||||||
target_link_libraries(${PROJECT_NAME}Tests ${PROJECT_NAME} Utils)
|
|
||||||
add_test(NAME ${PROJECT_NAME}Tests COMMAND ${PROJECT_NAME}Tests)
|
|
||||||
|
|
@ -1,8 +0,0 @@
|
||||||
|
|
||||||
#include "AutomatasCommon.hpp"
|
|
||||||
#include "Utils.hpp"
|
|
||||||
|
|
||||||
using namespace tp;
|
|
||||||
|
|
||||||
static ModuleManifest* sModuleDependencies[] = { &gModuleUtils, nullptr };
|
|
||||||
ModuleManifest tp::gModuleAutomatas = ModuleManifest("Automatas", nullptr, nullptr, sModuleDependencies);
|
|
||||||
|
|
@ -1,7 +0,0 @@
|
||||||
#pragma once
|
|
||||||
|
|
||||||
#include "Module.hpp"
|
|
||||||
|
|
||||||
namespace tp {
|
|
||||||
extern ModuleManifest gModuleAutomatas;
|
|
||||||
}
|
|
||||||
|
|
@ -1,14 +0,0 @@
|
||||||
#include "AutomatasCommon.hpp"
|
|
||||||
#include "Utils.hpp"
|
|
||||||
|
|
||||||
int main() {
|
|
||||||
|
|
||||||
tp::ModuleManifest* deps[] = { &tp::gModuleAutomatas, &tp::gModuleUtils, nullptr };
|
|
||||||
tp::ModuleManifest testModule("Test", nullptr, nullptr, deps);
|
|
||||||
|
|
||||||
if (!testModule.initialize()) {
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
testModule.deinitialize();
|
|
||||||
}
|
|
||||||
|
|
@ -16,7 +16,6 @@ add_subdirectory(Utils)
|
||||||
add_subdirectory(Containers)
|
add_subdirectory(Containers)
|
||||||
add_subdirectory(Math)
|
add_subdirectory(Math)
|
||||||
add_subdirectory(Allocators)
|
add_subdirectory(Allocators)
|
||||||
add_subdirectory(Automatas)
|
|
||||||
add_subdirectory(Strings)
|
add_subdirectory(Strings)
|
||||||
add_subdirectory(Language)
|
add_subdirectory(Language)
|
||||||
add_subdirectory(CommandLine)
|
add_subdirectory(CommandLine)
|
||||||
|
|
|
||||||
|
|
@ -165,7 +165,12 @@ namespace tp {
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
template <typename tType, class tAllocator = DefaultAllocator, ualni(tResizePolicy)(ualni) = BufferResizeScaling<2>, ualni(tResizePolicyDown)(ualni) = BufferResizeScalingDown<2>, ualni tMinSize = 4>
|
template <
|
||||||
|
typename tType,
|
||||||
|
class tAllocator = DefaultAllocator,
|
||||||
|
ualni(tResizePolicy)(ualni) = BufferResizeScaling<2>,
|
||||||
|
ualni(tResizePolicyDown)(ualni) = BufferResizeScalingDown<2>,
|
||||||
|
ualni tMinSize = 4>
|
||||||
class Buffer {
|
class Buffer {
|
||||||
typedef SelectValueOrReference<tType> Arg;
|
typedef SelectValueOrReference<tType> Arg;
|
||||||
|
|
||||||
|
|
@ -305,7 +310,7 @@ namespace tp {
|
||||||
mBuff = (tType*) mAllocator.allocate(sizeof(tType) * mSize);
|
mBuff = (tType*) mAllocator.allocate(sizeof(tType) * mSize);
|
||||||
mLoad = 0;
|
mLoad = 0;
|
||||||
for (const auto& val : input) {
|
for (const auto& val : input) {
|
||||||
mBuff[mLoad] = val;
|
new (&mBuff[mLoad]) tType(val);
|
||||||
mLoad++;
|
mLoad++;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -10,7 +10,7 @@ file(GLOB SOURCES "./private/*.cpp" "./private/*/*.cpp")
|
||||||
file(GLOB HEADERS "./public/*.hpp" "./public/*/*.hpp")
|
file(GLOB HEADERS "./public/*.hpp" "./public/*/*.hpp")
|
||||||
add_library(${PROJECT_NAME} STATIC ${SOURCES} ${HEADERS})
|
add_library(${PROJECT_NAME} STATIC ${SOURCES} ${HEADERS})
|
||||||
target_include_directories(${PROJECT_NAME} PUBLIC public/)
|
target_include_directories(${PROJECT_NAME} PUBLIC public/)
|
||||||
target_link_libraries(${PROJECT_NAME} PUBLIC Strings Automatas)
|
target_link_libraries(${PROJECT_NAME} PUBLIC Strings)
|
||||||
|
|
||||||
### -------------------------- Tests -------------------------- ###
|
### -------------------------- Tests -------------------------- ###
|
||||||
enable_testing()
|
enable_testing()
|
||||||
|
|
|
||||||
|
|
@ -16,510 +16,4 @@ namespace tp {
|
||||||
|
|
||||||
extern ModuleManifest gModuleTokenizer;
|
extern ModuleManifest gModuleTokenizer;
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
class TransitionMatrix;
|
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
class DFA;
|
|
||||||
|
|
||||||
// Non-Deterministic Finite-State Automata
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
class NFA {
|
|
||||||
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
|
||||||
|
|
||||||
public:
|
|
||||||
struct Vertex;
|
|
||||||
|
|
||||||
private:
|
|
||||||
struct Edge {
|
|
||||||
Vertex* mVertex = nullptr;
|
|
||||||
bool mConsumesSymbol = false;
|
|
||||||
Range<tAlphabetType> mAcceptingRange;
|
|
||||||
bool mAcceptsAll = false;
|
|
||||||
bool mExclude = false;
|
|
||||||
|
|
||||||
bool isTransition(const tAlphabetType& symbol) {
|
|
||||||
if (symbol == 0) return false;
|
|
||||||
if (!mConsumesSymbol || mAcceptsAll) return true;
|
|
||||||
bool const in_range = (symbol >= mAcceptingRange.mBegin && symbol <= mAcceptingRange.mEnd);
|
|
||||||
return in_range != mExclude;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
public:
|
|
||||||
struct Vertex {
|
|
||||||
List<Edge> edges;
|
|
||||||
tStateType termination_state = tNoStateVal;
|
|
||||||
#ifdef ENV_BUILD_DEBUG
|
|
||||||
ualni debug_idx = 0;
|
|
||||||
#endif
|
|
||||||
ualni flag = 0;
|
|
||||||
};
|
|
||||||
|
|
||||||
private:
|
|
||||||
friend DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
|
|
||||||
|
|
||||||
List<Vertex> mVertices;
|
|
||||||
Vertex* mStart = nullptr;
|
|
||||||
|
|
||||||
public:
|
|
||||||
NFA() = default;
|
|
||||||
|
|
||||||
Vertex* addVertex() {
|
|
||||||
auto node = mVertices.newNode();
|
|
||||||
#ifdef ENV_BUILD_DEBUG
|
|
||||||
node->data.debug_idx = mVertices.length() + 1;
|
|
||||||
#endif
|
|
||||||
mVertices.pushBack(node);
|
|
||||||
return &node->data;
|
|
||||||
}
|
|
||||||
|
|
||||||
void addTransition(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool consumes, bool accepts_all, bool exclude) {
|
|
||||||
Edge edge;
|
|
||||||
edge.mVertex = to;
|
|
||||||
edge.mConsumesSymbol = consumes;
|
|
||||||
edge.mAcceptingRange = range;
|
|
||||||
edge.mExclude = exclude;
|
|
||||||
edge.mAcceptsAll = accepts_all;
|
|
||||||
from->edges.pushBack(edge);
|
|
||||||
}
|
|
||||||
|
|
||||||
void setStartVertex(Vertex* start) {
|
|
||||||
mStart = start;
|
|
||||||
}
|
|
||||||
|
|
||||||
[[nodiscard]] Vertex* getStartVertex() const {
|
|
||||||
return mStart;
|
|
||||||
}
|
|
||||||
|
|
||||||
void setVertexState(Vertex* vertex, tStateType state) {
|
|
||||||
vertex->termination_state = state;
|
|
||||||
}
|
|
||||||
|
|
||||||
[[nodiscard]] bool isValid() const {
|
|
||||||
if (!mStart) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
Range<tAlphabetType> getAlphabetRange() const {
|
|
||||||
tAlphabetType start = 0, end = 0;
|
|
||||||
Range<tAlphabetType> all_range(std::numeric_limits<tAlphabetType>::min(), std::numeric_limits<tAlphabetType>::max());
|
|
||||||
bool first = true;
|
|
||||||
|
|
||||||
for (auto vertex : mVertices) {
|
|
||||||
for (auto edge : vertex.data().edges) {
|
|
||||||
if (!edge.data().mConsumesSymbol) continue;
|
|
||||||
auto const& tran_range = edge.data().mAcceptingRange;
|
|
||||||
if (edge.data().mAcceptsAll || edge.data().mExclude) return all_range;
|
|
||||||
if (tran_range.mBegin < start || first) start = tran_range.mBegin;
|
|
||||||
if (tran_range.mEnd > end || first) end = tran_range.mEnd;
|
|
||||||
first = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return Range<tAlphabetType>( start, end + 1 );
|
|
||||||
}
|
|
||||||
|
|
||||||
// vertices that are reachable from initial set with no input consumption (E-transitions)
|
|
||||||
// does not include initial set
|
|
||||||
void closure(const List<Vertex*>& set, List<Vertex*>& closure, ualni unique_call_id) {
|
|
||||||
List<Vertex*> marked;
|
|
||||||
marked = set;
|
|
||||||
|
|
||||||
while (marked.length()) {
|
|
||||||
auto first = marked.first()->data;
|
|
||||||
if (first->flag != unique_call_id) {
|
|
||||||
first->flag = unique_call_id;
|
|
||||||
closure.pushBack(first);
|
|
||||||
}
|
|
||||||
for (auto edge : first->edges) {
|
|
||||||
if (!edge.data().mConsumesSymbol && edge.data().mVertex->flag != unique_call_id) {
|
|
||||||
marked.pushBack(edge.data().mVertex);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
marked.popFront();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// vertices that are reachable from initial set with symbol transition
|
|
||||||
void move(const List<Vertex*>& set, List<Vertex*>& reachable, tAlphabetType symbol, ualni unique_call_id) {
|
|
||||||
for (auto vertex : set) {
|
|
||||||
for (auto edge : vertex->edges) {
|
|
||||||
if (!edge.data().mConsumesSymbol) continue;
|
|
||||||
bool transition = edge.data().isTransition(symbol);
|
|
||||||
if (transition && edge.data().mVertex->flag != unique_call_id) {
|
|
||||||
edge.data().mVertex->flag = unique_call_id;
|
|
||||||
reachable.pushBack(edge.data().mVertex);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
// Deterministic Finite-State Automata
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
class DFA {
|
|
||||||
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
|
||||||
friend TransitionMatrix<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
|
|
||||||
|
|
||||||
struct Vertex {
|
|
||||||
struct Edge {
|
|
||||||
Vertex* vertex = nullptr;
|
|
||||||
tAlphabetType transition_code = nullptr;
|
|
||||||
};
|
|
||||||
|
|
||||||
List<Edge> edges;
|
|
||||||
tStateType termination_state = tNoStateVal;
|
|
||||||
bool marked = false;
|
|
||||||
};
|
|
||||||
|
|
||||||
List<Vertex> mVertices;
|
|
||||||
Vertex* mStart = nullptr;
|
|
||||||
const Vertex* mIter = nullptr;
|
|
||||||
Range<tAlphabetType> mAlphabetRange;
|
|
||||||
bool mTrapState = false;
|
|
||||||
|
|
||||||
typedef typename NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>::Vertex NState;
|
|
||||||
|
|
||||||
public:
|
|
||||||
|
|
||||||
explicit DFA(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& nfa) {
|
|
||||||
if (!nfa.isValid()) {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
mAlphabetRange = nfa.getAlphabetRange();
|
|
||||||
|
|
||||||
struct DStateKey {
|
|
||||||
const List<NState*>* nStates;
|
|
||||||
bool operator==(const DStateKey& in) const {
|
|
||||||
if (nStates->length() != in.nStates->length()) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
// FIXME : make linear time
|
|
||||||
for (auto state : *nStates) {
|
|
||||||
bool found = false;
|
|
||||||
for (auto in_state : *in.nStates) {
|
|
||||||
if (state.data() == in_state.data()) {
|
|
||||||
found = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!found) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return true;
|
|
||||||
}
|
|
||||||
|
|
||||||
static ualni dStateHashFunc(DStateKey key) {
|
|
||||||
alni out = 0;
|
|
||||||
for (auto state : *key.nStates) {
|
|
||||||
out += alni(state.data());
|
|
||||||
}
|
|
||||||
return out;
|
|
||||||
};
|
|
||||||
};
|
|
||||||
|
|
||||||
// all NFA states that are reachable from initial DFA State for specific symbol in alphabet
|
|
||||||
struct DState {
|
|
||||||
struct DTransition {
|
|
||||||
DState* state = nullptr;
|
|
||||||
tAlphabetType accepting_code;
|
|
||||||
};
|
|
||||||
|
|
||||||
List<NState*> nStates;
|
|
||||||
List<DTransition> transitions;
|
|
||||||
|
|
||||||
Vertex* dVertex= nullptr; // relevant DFA vertex
|
|
||||||
|
|
||||||
#ifdef ENV_BUILD_DEBUG
|
|
||||||
ualni debug_idx = 0;
|
|
||||||
#endif
|
|
||||||
};
|
|
||||||
|
|
||||||
// includes closure of NFA start state by definition
|
|
||||||
auto start_state = new DState();
|
|
||||||
nfa.closure({ nfa.getStartVertex() }, start_state->nStates, ualni(start_state));
|
|
||||||
|
|
||||||
Map<DStateKey, DState*, DefaultAllocator, DStateKey::dStateHashFunc, 256> dStates;
|
|
||||||
|
|
||||||
dStates.put({ &start_state->nStates }, start_state);
|
|
||||||
|
|
||||||
List<DState*> working_set = { start_state };
|
|
||||||
|
|
||||||
// while there is items to work with
|
|
||||||
auto currentDState = working_set.first();
|
|
||||||
while (currentDState) {
|
|
||||||
|
|
||||||
// check all possible transitions for any symbol
|
|
||||||
for (auto symbol : mAlphabetRange) {
|
|
||||||
|
|
||||||
List<NState*> reachableNStates;
|
|
||||||
|
|
||||||
nfa.move(currentDState->data->nStates, reachableNStates, symbol, ualni(currentDState + symbol));
|
|
||||||
nfa.closure(reachableNStates, reachableNStates, ualni(currentDState + symbol));
|
|
||||||
|
|
||||||
if (!reachableNStates.length()) {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
DState* targetDState = nullptr;
|
|
||||||
|
|
||||||
// check if set of all reachable NFA states already forms existing DFA state
|
|
||||||
auto idx = dStates.presents({ &reachableNStates});
|
|
||||||
if (idx) {
|
|
||||||
targetDState = dStates.getSlotVal(idx);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!targetDState) {
|
|
||||||
// register new DFA state
|
|
||||||
targetDState = new DState();
|
|
||||||
|
|
||||||
#ifdef ENV_BUILD_DEBUG
|
|
||||||
targetDState->debug_idx = dStates.size();
|
|
||||||
#endif
|
|
||||||
|
|
||||||
targetDState->nStates = reachableNStates;
|
|
||||||
|
|
||||||
// append to working stack
|
|
||||||
working_set.pushBack(targetDState);
|
|
||||||
dStates.put({ &targetDState->nStates }, targetDState);
|
|
||||||
}
|
|
||||||
|
|
||||||
// add transition to DFA state
|
|
||||||
currentDState->data->transitions.pushBack({targetDState, symbol });
|
|
||||||
}
|
|
||||||
|
|
||||||
working_set.popFront();
|
|
||||||
currentDState = working_set.first();
|
|
||||||
}
|
|
||||||
|
|
||||||
// create own vertices
|
|
||||||
for (auto node : dStates) {
|
|
||||||
tStateType state = tNoStateVal;
|
|
||||||
for (auto iter : node->val->nStates) {
|
|
||||||
if (iter->termination_state != tNoStateVal) {
|
|
||||||
state = iter->termination_state;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
node->val->dVertex = addVertex(state);
|
|
||||||
}
|
|
||||||
|
|
||||||
// connect all vertices
|
|
||||||
for (auto node : dStates) {
|
|
||||||
for (auto edge : node->val->transitions) {
|
|
||||||
addTransition(node->val->dVertex, edge.data().state->dVertex, edge.data().accepting_code);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// set the starting vertex
|
|
||||||
mStart = start_state->dVertex;
|
|
||||||
|
|
||||||
// cleanup
|
|
||||||
for (auto node : dStates) {
|
|
||||||
delete node->val;
|
|
||||||
}
|
|
||||||
|
|
||||||
collapseEquivalentVertices();
|
|
||||||
mAlphabetRange = getAlphabetRange();
|
|
||||||
}
|
|
||||||
|
|
||||||
tStateType move(tAlphabetType symbol) {
|
|
||||||
if (mTrapState || !mIter) {
|
|
||||||
return tNoStateVal;
|
|
||||||
}
|
|
||||||
|
|
||||||
for (auto edge : mIter->edges) {
|
|
||||||
if (edge.data().transition_code == symbol) {
|
|
||||||
mIter = edge.data().vertex;
|
|
||||||
return mIter->termination_state;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
mTrapState = true;
|
|
||||||
return tNoStateVal;
|
|
||||||
}
|
|
||||||
|
|
||||||
void start() {
|
|
||||||
mIter = mStart;
|
|
||||||
mTrapState = false;
|
|
||||||
}
|
|
||||||
|
|
||||||
[[nodiscard]] uhalni nVertices() const {
|
|
||||||
return (uhalni) mVertices.length();
|
|
||||||
}
|
|
||||||
|
|
||||||
[[nodiscard]] Range<tAlphabetType> getRange() const {
|
|
||||||
return mAlphabetRange;
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
[[nodiscard]] Range<tAlphabetType> getAlphabetRange() const {
|
|
||||||
Range<tAlphabetType> out;
|
|
||||||
for (auto vertex : mVertices) {
|
|
||||||
vertex.data().marked = false;
|
|
||||||
}
|
|
||||||
|
|
||||||
bool first = true;
|
|
||||||
getAlphabetRangeUtil(mStart, &out, first);
|
|
||||||
out.mEnd++;
|
|
||||||
return out;
|
|
||||||
}
|
|
||||||
|
|
||||||
void getAlphabetRangeUtil(Vertex* vert, Range<tAlphabetType>* out, bool& first) const {
|
|
||||||
vert->marked = true;
|
|
||||||
for (auto edge : vert->edges) {
|
|
||||||
auto const code = edge.data().transition_code;
|
|
||||||
|
|
||||||
if (first) {
|
|
||||||
*out = { code, code };
|
|
||||||
first = false;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (code < out->mBegin) {
|
|
||||||
out->mBegin = code;
|
|
||||||
}
|
|
||||||
if (code > out->mEnd) {
|
|
||||||
out->mEnd = code;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!edge.data().vertex->marked) {
|
|
||||||
getAlphabetRangeUtil(edge.data().vertex, out, first);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
void collapseEquivalentVertices() {
|
|
||||||
// TODO
|
|
||||||
}
|
|
||||||
|
|
||||||
Vertex* addVertex(tStateType state) {
|
|
||||||
auto node = mVertices.addNodeBack();
|
|
||||||
node->data.termination_state = state;
|
|
||||||
return &node->data;
|
|
||||||
}
|
|
||||||
|
|
||||||
void addTransition(Vertex* from, Vertex* to, tAlphabetType transition_symbol) {
|
|
||||||
from->edges.pushBack({ to, transition_symbol });
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
class TransitionMatrix {
|
|
||||||
|
|
||||||
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
|
||||||
|
|
||||||
Buffer2D<ualni> mTransitions;
|
|
||||||
Buffer<tStateType> mStates;
|
|
||||||
Range<tAlphabetType> mSymbolRange = { 0, 0 };
|
|
||||||
|
|
||||||
ualni mIter = 0;
|
|
||||||
ualni mIterPrev = 0;
|
|
||||||
ualni mStart = 0;
|
|
||||||
|
|
||||||
public:
|
|
||||||
|
|
||||||
TransitionMatrix() = default;
|
|
||||||
|
|
||||||
auto getStates() const { return &mStates; }
|
|
||||||
auto getTransitions() const { return &mTransitions; }
|
|
||||||
auto getStart() const { return mStart; }
|
|
||||||
|
|
||||||
void construct(const DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& dfa) {
|
|
||||||
mSymbolRange = dfa.getRange();
|
|
||||||
auto range_len = ualni(mSymbolRange.mEnd - mSymbolRange.mBegin);
|
|
||||||
auto sizeX = range_len ? range_len : 1;
|
|
||||||
auto sizeY = (ualni) (dfa.nVertices() + 1);
|
|
||||||
|
|
||||||
mTransitions.reserve({ sizeX, sizeY });
|
|
||||||
mTransitions.assign(dfa.nVertices());
|
|
||||||
mStates.reserve(sizeY);
|
|
||||||
|
|
||||||
ualni idx = 0;
|
|
||||||
for (auto vertex : dfa.mVertices) {
|
|
||||||
auto state = vertex.data().termination_state;
|
|
||||||
mStates[idx] = state;
|
|
||||||
idx++;
|
|
||||||
}
|
|
||||||
|
|
||||||
mStates[dfa.nVertices()] = tFailedStateVal;
|
|
||||||
|
|
||||||
idx = 0;
|
|
||||||
for (auto vertex : dfa.mVertices) {
|
|
||||||
if (&vertex.data() == dfa.mStart) {
|
|
||||||
mStart = mIter = mIterPrev = idx;
|
|
||||||
}
|
|
||||||
idx++;
|
|
||||||
}
|
|
||||||
|
|
||||||
ualni vertexIdx = 0;
|
|
||||||
for (auto vertex : dfa.mVertices) {
|
|
||||||
for (auto edge : vertex.data().edges) {
|
|
||||||
ualni vertex2Idx = 0;
|
|
||||||
for (auto vertex2 : dfa.mVertices) {
|
|
||||||
if (edge.data().vertex == &vertex2.data()) break;
|
|
||||||
vertex2Idx++;
|
|
||||||
}
|
|
||||||
auto const code = edge.data().transition_code;
|
|
||||||
mTransitions.set( { (ualni) (code - mSymbolRange.mBegin), (ualni) vertexIdx }, vertex2Idx);
|
|
||||||
}
|
|
||||||
vertexIdx++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
bool isTrapped() {
|
|
||||||
return mStates[mIter] == tFailedStateVal;
|
|
||||||
}
|
|
||||||
|
|
||||||
tStateType move(tAlphabetType symbol) {
|
|
||||||
if (symbol >= mSymbolRange.mBegin && symbol < mSymbolRange.mEnd) {
|
|
||||||
mIter = mTransitions.get({ (ualni) (symbol - mSymbolRange.mBegin), (ualni) mIter });
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
mIter = mStates.size() - 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (mIterPrev == mStart) {
|
|
||||||
if (mStates[mIter] == tFailedStateVal) {
|
|
||||||
reset();
|
|
||||||
return tFailedStateVal;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
mIterPrev = mIter;
|
|
||||||
return tNoStateVal;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
if (mStates[mIter] == tFailedStateVal) {
|
|
||||||
if (mStates[mIterPrev] != tNoStateVal) {
|
|
||||||
auto out = mStates[mIterPrev];
|
|
||||||
reset();
|
|
||||||
return out;
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
reset();
|
|
||||||
return tFailedStateVal;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
mIterPrev = mIter;
|
|
||||||
return tNoStateVal;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
mIterPrev = mIter;
|
|
||||||
return mStates[mIter];
|
|
||||||
}
|
|
||||||
|
|
||||||
void reset() {
|
|
||||||
mIter = mStart;
|
|
||||||
mIterPrev = mStart;
|
|
||||||
}
|
|
||||||
};
|
|
||||||
}
|
}
|
||||||
|
|
@ -27,7 +27,10 @@ namespace tp::RegEx {
|
||||||
|
|
||||||
template <typename tAlphabetType>
|
template <typename tAlphabetType>
|
||||||
struct AstVal : public AstNode {
|
struct AstVal : public AstNode {
|
||||||
explicit AstVal(tAlphabetType val) : mVal(val) { mType = VAL; }
|
explicit AstVal(tAlphabetType val) :
|
||||||
|
mVal(val) {
|
||||||
|
mType = VAL;
|
||||||
|
}
|
||||||
~AstVal() override = default;
|
~AstVal() override = default;
|
||||||
tAlphabetType mVal;
|
tAlphabetType mVal;
|
||||||
};
|
};
|
||||||
|
|
@ -123,7 +126,6 @@ namespace tp::RegEx {
|
||||||
uhalni mTokLength = 0;
|
uhalni mTokLength = 0;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
|
|
||||||
ParseError mError;
|
ParseError mError;
|
||||||
|
|
||||||
// regular expression must be a zero termination string
|
// regular expression must be a zero termination string
|
||||||
|
|
@ -133,7 +135,6 @@ namespace tp::RegEx {
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
|
|
||||||
AstCompound* parseRegEx() {
|
AstCompound* parseRegEx() {
|
||||||
auto out = new AstCompound();
|
auto out = new AstCompound();
|
||||||
for (AstNode* node = parseElement(); node; node = parseElement()) {
|
for (AstNode* node = parseElement(); node; node = parseElement()) {
|
||||||
|
|
@ -157,7 +158,11 @@ namespace tp::RegEx {
|
||||||
case TOK_CLASS_START_EXCLUDE: out = parseClass(true); break;
|
case TOK_CLASS_START_EXCLUDE: out = parseClass(true); break;
|
||||||
case TOK_ANY: out = parseAny(); break;
|
case TOK_ANY: out = parseAny(); break;
|
||||||
case TOK_VAL: out = parseVal(); break;
|
case TOK_VAL: out = parseVal(); break;
|
||||||
case TOK_NONE: { discardTok(); return nullptr; };
|
case TOK_NONE:
|
||||||
|
{
|
||||||
|
discardTok();
|
||||||
|
return nullptr;
|
||||||
|
};
|
||||||
default: break;
|
default: break;
|
||||||
}
|
}
|
||||||
if (!out) {
|
if (!out) {
|
||||||
|
|
@ -170,7 +175,10 @@ namespace tp::RegEx {
|
||||||
case TOK_REPEAT_PLUS: out = parseRepetition(out, true); break;
|
case TOK_REPEAT_PLUS: out = parseRepetition(out, true); break;
|
||||||
case TOK_IF: out = parseIf(out); break;
|
case TOK_IF: out = parseIf(out); break;
|
||||||
case TOK_NONE: break;
|
case TOK_NONE: break;
|
||||||
default: { discardTok(); }
|
default:
|
||||||
|
{
|
||||||
|
discardTok();
|
||||||
|
}
|
||||||
}
|
}
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
@ -225,15 +233,12 @@ namespace tp::RegEx {
|
||||||
readTok();
|
readTok();
|
||||||
if ((mCurToken.type == TOK_CLASS_END && !exclude) || (mCurToken.type == TOK_CLASS_END_EXCLUDE && exclude)) {
|
if ((mCurToken.type == TOK_CLASS_END && !exclude) || (mCurToken.type == TOK_CLASS_END_EXCLUDE && exclude)) {
|
||||||
return out;
|
return out;
|
||||||
}
|
} else {
|
||||||
else {
|
|
||||||
goto READ_VAL;
|
goto READ_VAL;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
AstAny* parseAny() {
|
AstAny* parseAny() { return new AstAny(); }
|
||||||
return new AstAny();
|
|
||||||
}
|
|
||||||
|
|
||||||
AstVal<tAlphabetType>* parseVal() {
|
AstVal<tAlphabetType>* parseVal() {
|
||||||
auto out = new AstVal<tAlphabetType>(mCurToken.val);
|
auto out = new AstVal<tAlphabetType>(mCurToken.val);
|
||||||
|
|
@ -269,9 +274,7 @@ namespace tp::RegEx {
|
||||||
return out;
|
return out;
|
||||||
}
|
}
|
||||||
|
|
||||||
void genError(const char* desc) {
|
void genError(const char* desc) { mError = { desc, mOffset }; }
|
||||||
mError = { desc, mOffset };
|
|
||||||
}
|
|
||||||
|
|
||||||
Token& readTok() {
|
Token& readTok() {
|
||||||
|
|
||||||
|
|
@ -290,8 +293,7 @@ namespace tp::RegEx {
|
||||||
if (crs[0] == mEscapeSymbol) {
|
if (crs[0] == mEscapeSymbol) {
|
||||||
mCurToken.val = crs[1];
|
mCurToken.val = crs[1];
|
||||||
mTokLength = 2;
|
mTokLength = 2;
|
||||||
}
|
} else {
|
||||||
else {
|
|
||||||
for (uhalni tok = 0; tok < TOK_SPECIALS_END_; tok++) {
|
for (uhalni tok = 0; tok < TOK_SPECIALS_END_; tok++) {
|
||||||
if (SpecialSymbols[tok] == mCurToken.val) {
|
if (SpecialSymbols[tok] == mCurToken.val) {
|
||||||
mCurToken.type = TokType(tok);
|
mCurToken.type = TokType(tok);
|
||||||
|
|
@ -310,208 +312,20 @@ namespace tp::RegEx {
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
template <typename tStateType>
|
|
||||||
struct CompileError {
|
|
||||||
ParseError mParseError;
|
|
||||||
uhalni mRuleIndex = 0;
|
|
||||||
tStateType mRuleState;
|
|
||||||
const char* description = nullptr;
|
|
||||||
[[nodiscard]] bool isError() const { return description; }
|
|
||||||
};
|
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
class Compiler {
|
CompileError<tStateType> compile(
|
||||||
|
NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out, const tAlphabetType* regex, tStateType state
|
||||||
typedef NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> Graph;
|
) {
|
||||||
typedef typename Graph::Vertex Vertex;
|
|
||||||
typedef Parser<tAlphabetType, tStateType, tNoStateVal> Parser;
|
|
||||||
|
|
||||||
struct Node {
|
|
||||||
Vertex* left = nullptr;
|
|
||||||
Vertex* right = nullptr;
|
|
||||||
};
|
|
||||||
|
|
||||||
private:
|
|
||||||
Graph* mGraph = nullptr;
|
|
||||||
|
|
||||||
public:
|
|
||||||
CompileError<tStateType> mError;
|
|
||||||
|
|
||||||
Node compile(Graph& graph, const tAlphabetType* regex, tStateType state) {
|
|
||||||
mGraph = &graph;
|
|
||||||
return compileUtil(regex, state);
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compile(Graph& aGraph, InitialierList<Pair<const tAlphabetType*, tStateType>> aRules) {
|
|
||||||
mGraph = &aGraph;
|
|
||||||
|
|
||||||
auto left = mGraph->addVertex();
|
|
||||||
auto right = mGraph->addVertex();
|
|
||||||
|
|
||||||
halni idx = 0;
|
|
||||||
for (auto rule: aRules) {
|
|
||||||
|
|
||||||
auto node = compileUtil(rule.head, rule.tail);
|
|
||||||
|
|
||||||
if (!(node.left && node.right)) {
|
|
||||||
mError.mRuleIndex = idx;
|
|
||||||
return {};
|
|
||||||
}
|
|
||||||
|
|
||||||
transitionAny(left, node.left);
|
|
||||||
transitionAny(node.right, right);
|
|
||||||
|
|
||||||
idx++;
|
|
||||||
}
|
|
||||||
|
|
||||||
mGraph->setStartVertex(left);
|
|
||||||
|
|
||||||
return { left, right };
|
|
||||||
}
|
|
||||||
|
|
||||||
private:
|
|
||||||
|
|
||||||
Node compileUtil(const tAlphabetType* regex, tStateType state) {
|
|
||||||
Parser parser;
|
|
||||||
auto astNode = parser.parse(regex);
|
|
||||||
if (parser.mError.isError()) {
|
|
||||||
mGraph->setStartVertex(nullptr);
|
|
||||||
mError.description = "Parsing Of Regular Expression Failed";
|
|
||||||
mError.mRuleState = state;
|
|
||||||
mError.mParseError = parser.mError;
|
|
||||||
return {};
|
|
||||||
}
|
|
||||||
|
|
||||||
auto node = compileNode(astNode, nullptr, nullptr);
|
|
||||||
delete astNode;
|
|
||||||
|
|
||||||
mGraph->setVertexState(node.right, state);
|
|
||||||
mGraph->setStartVertex(node.left);
|
|
||||||
|
|
||||||
return node;
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileVal(AstVal<tAlphabetType>* val, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
auto left = aLeft ? aLeft : mGraph->addVertex();
|
|
||||||
auto right = aRight ? aRight : mGraph->addVertex();
|
|
||||||
transitionVal(left, right, val->mVal);
|
|
||||||
return { left, right };
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileAlternation(AstAlternation* alt, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
auto first_node = compileNode(alt->mFirst, aLeft, aRight);
|
|
||||||
auto second_node = compileNode(alt->mSecond);
|
|
||||||
transitionAny(first_node.left, second_node.left);
|
|
||||||
transitionAny(second_node.right, first_node.right);
|
|
||||||
return first_node;
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileAny(AstAny*, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
auto left = aLeft ? aLeft : mGraph->addVertex();
|
|
||||||
auto right = aRight ? aRight : mGraph->addVertex();
|
|
||||||
transitionAny(left, right, true);
|
|
||||||
return { left, right };
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileRepeat(AstRepetition* repeat, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
if (repeat->mPlus) {
|
|
||||||
auto middle = mGraph->addVertex();
|
|
||||||
|
|
||||||
auto left_node = compileNode(repeat->mNode, aLeft, middle);
|
|
||||||
|
|
||||||
auto right_node = compileNode(repeat->mNode, middle, aRight);
|
|
||||||
transitionAny(right_node.right, right_node.left);
|
|
||||||
transitionAny(right_node.left, right_node.right);
|
|
||||||
|
|
||||||
return { left_node.left, right_node.right };
|
|
||||||
}
|
|
||||||
else {
|
|
||||||
auto node = compileNode(repeat->mNode, aLeft, aRight);
|
|
||||||
transitionAny(node.right, node.left);
|
|
||||||
transitionAny(node.left, node.right);
|
|
||||||
return node;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileIf(AstIf* ifNode, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
auto node = compileNode(ifNode->mNode, aLeft, aRight);
|
|
||||||
transitionAny(node.left, node.right);
|
|
||||||
return node;
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileClass(AstClass<tAlphabetType>* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
auto left = aLeft ? aLeft : mGraph->addVertex();
|
|
||||||
auto right = aRight ? aRight : mGraph->addVertex();
|
|
||||||
|
|
||||||
if (node->mRanges.length() == 1) {
|
|
||||||
auto const& range = node->mRanges.first()->data;
|
|
||||||
transitionRange(left, right, { range.mBegin, range.mEnd }, node->mExclude);
|
|
||||||
return { left, right };
|
|
||||||
}
|
|
||||||
|
|
||||||
for (auto range : node->mRanges) {
|
|
||||||
auto middle = mGraph->addVertex();
|
|
||||||
transitionRange(left, middle, { range.data().mBegin, range.data().mEnd }, node->mExclude);
|
|
||||||
transitionAny(middle, right);
|
|
||||||
}
|
|
||||||
return { left, right };
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileCompound(AstCompound* compound, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
Vertex* left = nullptr;
|
|
||||||
Vertex* rigth = nullptr;
|
|
||||||
|
|
||||||
ualni idx = 0;
|
|
||||||
for (auto child : compound->mChilds) {
|
|
||||||
auto pass_left = idx == 0 ? aLeft : rigth;
|
|
||||||
auto pass_right = idx == compound->mChilds.length() - 1 ? aRight : nullptr;
|
|
||||||
auto node = compileNode(child.data(), pass_left, pass_right);
|
|
||||||
if (!left) left = node.left;
|
|
||||||
rigth = node.right;
|
|
||||||
idx++;
|
|
||||||
}
|
|
||||||
|
|
||||||
return { left, rigth };
|
|
||||||
}
|
|
||||||
|
|
||||||
Node compileNode(AstNode* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
|
||||||
switch (node->mType) {
|
|
||||||
case AstNode::CLASS: return compileClass((AstClass<tAlphabetType>*)node, aLeft, aRight);
|
|
||||||
case AstNode::COMPOUND: return compileCompound((AstCompound*)node, aLeft, aRight);
|
|
||||||
case AstNode::IF: return compileIf((AstIf*)node, aLeft, aRight);
|
|
||||||
case AstNode::REPEAT: return compileRepeat((AstRepetition*)node, aLeft, aRight);
|
|
||||||
case AstNode::ANY: return compileAny((AstAny*)node, aLeft, aRight);
|
|
||||||
case AstNode::OR: return compileAlternation((AstAlternation*)node, aLeft, aRight);
|
|
||||||
case AstNode::VAL: return compileVal((AstVal<tAlphabetType>*)node, aLeft, aRight);
|
|
||||||
case AstNode::NONE:
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
ASSERT(0)
|
|
||||||
return {};
|
|
||||||
}
|
|
||||||
|
|
||||||
void transitionAny(Vertex* from, Vertex* to, bool consumes = false) {
|
|
||||||
mGraph->addTransition(from, to, {}, consumes, true, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
void transitionVal(Vertex* from, Vertex* to, tAlphabetType val) {
|
|
||||||
mGraph->addTransition(from, to, { val, val }, true, false, false);
|
|
||||||
}
|
|
||||||
|
|
||||||
void transitionRange(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool exclude) {
|
|
||||||
mGraph->addTransition(from, to, range, true, false, exclude);
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
|
||||||
CompileError<tStateType> compile(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out, const tAlphabetType* regex, tStateType state) {
|
|
||||||
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
|
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
|
||||||
compiler.compile(out, regex, state);
|
compiler.compile(out, regex, state);
|
||||||
return compiler.mError;
|
return compiler.mError;
|
||||||
}
|
}
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
CompileError<tStateType> compile(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out, const InitialierList<Pair<const tAlphabetType*, tStateType>>& rules) {
|
CompileError<tStateType> compile(
|
||||||
|
NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& out,
|
||||||
|
const InitialierList<Pair<const tAlphabetType*, tStateType>>& rules
|
||||||
|
) {
|
||||||
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
|
Compiler<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> compiler;
|
||||||
compiler.compile(out, rules);
|
compiler.compile(out, rules);
|
||||||
return compiler.mError;
|
return compiler.mError;
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,5 @@
|
||||||
|
|
||||||
#include "Grammar.hpp"
|
#include "Grammar.hpp"
|
||||||
#include "NewPlacement.hpp"
|
|
||||||
|
|
||||||
using namespace tp;
|
using namespace tp;
|
||||||
|
|
||||||
|
|
@ -12,7 +11,9 @@ ContextFreeGrammar::Arg::Arg(const String& id, bool terminal, bool epsilon) {
|
||||||
|
|
||||||
const String& ContextFreeGrammar::Arg::getId() const { return mId; }
|
const String& ContextFreeGrammar::Arg::getId() const { return mId; }
|
||||||
|
|
||||||
bool ContextFreeGrammar::Arg::operator==(const Arg& in) const { return (mId == in.mId) && (mIsEpsilon == in.mIsEpsilon) && (mIsTerminal == in.mIsTerminal); }
|
bool ContextFreeGrammar::Arg::operator==(const Arg& in) const {
|
||||||
|
return (mId == in.mId) && (mIsEpsilon == in.mIsEpsilon) && (mIsTerminal == in.mIsTerminal);
|
||||||
|
}
|
||||||
|
|
||||||
ContextFreeGrammar::Rule::Rule(const String& id, const InitialierList<Arg>& args) {
|
ContextFreeGrammar::Rule::Rule(const String& id, const InitialierList<Arg>& args) {
|
||||||
mId = id;
|
mId = id;
|
||||||
|
|
|
||||||
|
|
@ -70,7 +70,7 @@ namespace tp {
|
||||||
|
|
||||||
template <typename tAlphabetType, typename tTokType>
|
template <typename tAlphabetType, typename tTokType>
|
||||||
class RegularGrammar {
|
class RegularGrammar {
|
||||||
|
public:
|
||||||
struct Node {
|
struct Node {
|
||||||
enum Type {
|
enum Type {
|
||||||
NONE,
|
NONE,
|
||||||
|
|
@ -97,7 +97,7 @@ namespace tp {
|
||||||
|
|
||||||
~ValueNode() override = default;
|
~ValueNode() override = default;
|
||||||
|
|
||||||
private:
|
public:
|
||||||
tAlphabetType mVal;
|
tAlphabetType mVal;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
@ -118,7 +118,7 @@ namespace tp {
|
||||||
mSequence.clear();
|
mSequence.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
public:
|
||||||
Buffer<const Node*> mSequence;
|
Buffer<const Node*> mSequence;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
@ -138,7 +138,7 @@ namespace tp {
|
||||||
delete mSecond;
|
delete mSecond;
|
||||||
}
|
}
|
||||||
|
|
||||||
private:
|
public:
|
||||||
const Node* mFirst = nullptr;
|
const Node* mFirst = nullptr;
|
||||||
const Node* mSecond = nullptr;
|
const Node* mSecond = nullptr;
|
||||||
};
|
};
|
||||||
|
|
@ -155,7 +155,7 @@ namespace tp {
|
||||||
|
|
||||||
~IfNode() override { delete mNode; }
|
~IfNode() override { delete mNode; }
|
||||||
|
|
||||||
private:
|
public:
|
||||||
const Node* mNode = nullptr;
|
const Node* mNode = nullptr;
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|
@ -180,7 +180,7 @@ namespace tp {
|
||||||
|
|
||||||
~RepetitionNode() override { delete mNode; }
|
~RepetitionNode() override { delete mNode; }
|
||||||
|
|
||||||
private:
|
public:
|
||||||
Node* mNode = nullptr;
|
Node* mNode = nullptr;
|
||||||
bool mPlus = false;
|
bool mPlus = false;
|
||||||
};
|
};
|
||||||
|
|
@ -197,7 +197,7 @@ namespace tp {
|
||||||
|
|
||||||
~ClassNode() override { mRanges.removeAll(); }
|
~ClassNode() override { mRanges.removeAll(); }
|
||||||
|
|
||||||
private:
|
public:
|
||||||
Buffer<Range<tAlphabetType>> mRanges;
|
Buffer<Range<tAlphabetType>> mRanges;
|
||||||
bool mExclude = false;
|
bool mExclude = false;
|
||||||
};
|
};
|
||||||
|
|
@ -220,9 +220,11 @@ namespace tp {
|
||||||
const Node* may(const Node* a) { return new IfNode(a); }
|
const Node* may(const Node* a) { return new IfNode(a); }
|
||||||
const Node* any() { return new AnyNode(); }
|
const Node* any() { return new AnyNode(); }
|
||||||
const Node* rep(const Node* rep, bool plus = false) { return new RepetitionNode(rep, plus); }
|
const Node* rep(const Node* rep, bool plus = false) { return new RepetitionNode(rep, plus); }
|
||||||
const Node* ranges(const Buffer<Range<tAlphabetType>>& ranges, bool exclude = false) { return new ClassNode(ranges, exclude); }
|
const Node* ranges(const Buffer<Range<tAlphabetType>>& ranges, bool exclude = false) {
|
||||||
|
return new ClassNode(ranges, exclude);
|
||||||
|
}
|
||||||
|
|
||||||
private:
|
public:
|
||||||
Buffer<Pair<const Node*, tTokType>> mRules;
|
Buffer<Pair<const Node*, tTokType>> mRules;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -2,5 +2,185 @@
|
||||||
#pragma once
|
#pragma once
|
||||||
|
|
||||||
#include "Grammar.hpp"
|
#include "Grammar.hpp"
|
||||||
|
#include "RegularAutomata.hpp"
|
||||||
|
|
||||||
namespace tp {}
|
namespace tp {
|
||||||
|
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class RegularCompiler {
|
||||||
|
|
||||||
|
typedef NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal> Graph;
|
||||||
|
typedef typename Graph::Vertex Vertex;
|
||||||
|
typedef RegularGrammar<tAlphabetType, tStateType> Grammar;
|
||||||
|
|
||||||
|
struct Node {
|
||||||
|
Vertex* left = nullptr;
|
||||||
|
Vertex* right = nullptr;
|
||||||
|
};
|
||||||
|
|
||||||
|
private:
|
||||||
|
Graph* mGraph = nullptr;
|
||||||
|
|
||||||
|
public:
|
||||||
|
struct CompileError {
|
||||||
|
uhalni mRuleIndex = 0;
|
||||||
|
tStateType mRuleState;
|
||||||
|
const char* description = nullptr;
|
||||||
|
[[nodiscard]] bool isError() const { return description; }
|
||||||
|
};
|
||||||
|
|
||||||
|
CompileError mError;
|
||||||
|
|
||||||
|
Node compile(Graph& graph, const tAlphabetType* regex, tStateType state) {
|
||||||
|
mGraph = &graph;
|
||||||
|
return compileUtil(regex, state);
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compile(Graph& aGraph, const Grammar& grammar) {
|
||||||
|
mGraph = &aGraph;
|
||||||
|
|
||||||
|
auto left = mGraph->addVertex();
|
||||||
|
auto right = mGraph->addVertex();
|
||||||
|
|
||||||
|
halni idx = 0;
|
||||||
|
for (auto rule : grammar.mRules) {
|
||||||
|
|
||||||
|
auto node = compileUtil(rule.data().first, rule.data().second);
|
||||||
|
|
||||||
|
if (!(node.left && node.right)) {
|
||||||
|
mError.mRuleIndex = idx;
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
transitionAny(left, node.left);
|
||||||
|
transitionAny(node.right, right);
|
||||||
|
|
||||||
|
idx++;
|
||||||
|
}
|
||||||
|
|
||||||
|
mGraph->setStartVertex(left);
|
||||||
|
|
||||||
|
return { left, right };
|
||||||
|
}
|
||||||
|
|
||||||
|
private:
|
||||||
|
Node compileUtil(const Grammar::Node* astNode, tStateType state) {
|
||||||
|
|
||||||
|
auto node = compileNode(astNode, nullptr, nullptr);
|
||||||
|
|
||||||
|
mGraph->setVertexState(node.right, state);
|
||||||
|
mGraph->setStartVertex(node.left);
|
||||||
|
|
||||||
|
return node;
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileVal(Grammar::ValueNode* val, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
auto left = aLeft ? aLeft : mGraph->addVertex();
|
||||||
|
auto right = aRight ? aRight : mGraph->addVertex();
|
||||||
|
transitionVal(left, right, val->mVal);
|
||||||
|
return { left, right };
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileAlternation(const Grammar::AlternationNode* alt, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
auto first_node = compileNode(alt->mFirst, aLeft, aRight);
|
||||||
|
auto second_node = compileNode(alt->mSecond);
|
||||||
|
transitionAny(first_node.left, second_node.left);
|
||||||
|
transitionAny(second_node.right, first_node.right);
|
||||||
|
return first_node;
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileAny(const Grammar::AnyNode*, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
auto left = aLeft ? aLeft : mGraph->addVertex();
|
||||||
|
auto right = aRight ? aRight : mGraph->addVertex();
|
||||||
|
transitionAny(left, right, true);
|
||||||
|
return { left, right };
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileRepeat(const Grammar::RepetitionNode* repeat, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
if (repeat->mPlus) {
|
||||||
|
auto middle = mGraph->addVertex();
|
||||||
|
|
||||||
|
auto left_node = compileNode(repeat->mNode, aLeft, middle);
|
||||||
|
|
||||||
|
auto right_node = compileNode(repeat->mNode, middle, aRight);
|
||||||
|
transitionAny(right_node.right, right_node.left);
|
||||||
|
transitionAny(right_node.left, right_node.right);
|
||||||
|
|
||||||
|
return { left_node.left, right_node.right };
|
||||||
|
} else {
|
||||||
|
auto node = compileNode(repeat->mNode, aLeft, aRight);
|
||||||
|
transitionAny(node.right, node.left);
|
||||||
|
transitionAny(node.left, node.right);
|
||||||
|
return node;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileIf(const Grammar::IfNode* ifNode, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
auto node = compileNode(ifNode->mNode, aLeft, aRight);
|
||||||
|
transitionAny(node.left, node.right);
|
||||||
|
return node;
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileClass(const Grammar::ClassNode* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
auto left = aLeft ? aLeft : mGraph->addVertex();
|
||||||
|
auto right = aRight ? aRight : mGraph->addVertex();
|
||||||
|
|
||||||
|
if (node->mRanges.size() == 1) {
|
||||||
|
auto const& range = node->mRanges.first();
|
||||||
|
transitionRange(left, right, { range.mBegin, range.mEnd }, node->mExclude);
|
||||||
|
return { left, right };
|
||||||
|
}
|
||||||
|
|
||||||
|
for (auto range : node->mRanges) {
|
||||||
|
auto middle = mGraph->addVertex();
|
||||||
|
transitionRange(left, middle, { range.data().mBegin, range.data().mEnd }, node->mExclude);
|
||||||
|
transitionAny(middle, right);
|
||||||
|
}
|
||||||
|
return { left, right };
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileCompound(const Grammar::CompoundNode* compound, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
Vertex* left = nullptr;
|
||||||
|
Vertex* rigth = nullptr;
|
||||||
|
|
||||||
|
ualni idx = 0;
|
||||||
|
for (auto child : compound->mSequence) {
|
||||||
|
auto pass_left = idx == 0 ? aLeft : rigth;
|
||||||
|
auto pass_right = idx == compound->mSequence.size() - 1 ? aRight : nullptr;
|
||||||
|
auto node = compileNode(child.data(), pass_left, pass_right);
|
||||||
|
if (!left) left = node.left;
|
||||||
|
rigth = node.right;
|
||||||
|
idx++;
|
||||||
|
}
|
||||||
|
|
||||||
|
return { left, rigth };
|
||||||
|
}
|
||||||
|
|
||||||
|
Node compileNode(const Grammar::Node* node, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
|
||||||
|
switch (node->mType) {
|
||||||
|
case Grammar::Node::CLASS: return compileClass((typename Grammar::ClassNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::COMPOUND: return compileCompound((typename Grammar::CompoundNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::IF: return compileIf((typename Grammar::IfNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::REPEAT: return compileRepeat((typename Grammar::RepetitionNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::ANY: return compileAny((typename Grammar::AnyNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::OR: return compileAlternation((typename Grammar::AlternationNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::VAL: return compileVal((typename Grammar::ValueNode*) node, aLeft, aRight);
|
||||||
|
case Grammar::Node::NONE: break;
|
||||||
|
}
|
||||||
|
ASSERT(0)
|
||||||
|
return {};
|
||||||
|
}
|
||||||
|
|
||||||
|
void transitionAny(Vertex* from, Vertex* to, bool consumes = false) {
|
||||||
|
mGraph->addTransition(from, to, {}, consumes, true, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void transitionVal(Vertex* from, Vertex* to, tAlphabetType val) {
|
||||||
|
mGraph->addTransition(from, to, { val, val }, true, false, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
void transitionRange(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool exclude) {
|
||||||
|
mGraph->addTransition(from, to, range, true, false, exclude);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
|
||||||
|
|
@ -6,16 +6,34 @@
|
||||||
|
|
||||||
namespace tp {
|
namespace tp {
|
||||||
|
|
||||||
template <typename tAlphabetType>
|
template <typename tAlphabetType, typename TokenType>
|
||||||
class Parser {
|
class Parser {
|
||||||
|
|
||||||
|
typedef TransitionMatrix<tAlphabetType, TokenType, TokenType::InTransition, TokenType::Failed> RegularTable;
|
||||||
|
typedef DFA<tAlphabetType, TokenType, TokenType::InTransition, TokenType::Failed> RegularGraph;
|
||||||
|
typedef RegularCompiler<tAlphabetType, TokenType, TokenType::InTransition, TokenType::Failed> RegularCompiler;
|
||||||
|
typedef NFA<tAlphabetType, TokenType, TokenType::InTransition, TokenType::Failed> RegularNonDetGraph;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
Parser() = default;
|
Parser() = default;
|
||||||
|
|
||||||
public:
|
public:
|
||||||
void compileTables(const ContextFreeGrammar& cfGrammar, const RegularGrammar<tAlphabetType, ualni>& reGrammar) {}
|
void compileTables(const ContextFreeGrammar& cfGrammar, const RegularGrammar<tAlphabetType, TokenType>& reGrammar) {
|
||||||
|
// Compile Regular Grammar
|
||||||
|
{
|
||||||
|
RegularNonDetGraph nfa;
|
||||||
|
RegularCompiler compiler;
|
||||||
|
compiler.compile(nfa, reGrammar);
|
||||||
|
|
||||||
|
RegularGraph dfa(nfa);
|
||||||
|
mRegularTable.construct(dfa);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
void parse(const tAlphabetType* sentence, ualni sentenceLength, AST& out) {}
|
void parse(const tAlphabetType* sentence, ualni sentenceLength, AST& out) {}
|
||||||
|
|
||||||
public:
|
public:
|
||||||
// save load compiled tables
|
// save load compiled tables
|
||||||
|
RegularTable mRegularTable;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
|
||||||
501
Language/public/RegularAutomata.hpp
Normal file
501
Language/public/RegularAutomata.hpp
Normal file
|
|
@ -0,0 +1,501 @@
|
||||||
|
|
||||||
|
#pragma once
|
||||||
|
|
||||||
|
#include "Strings.hpp"
|
||||||
|
#include "List.hpp"
|
||||||
|
#include "Map.hpp"
|
||||||
|
#include "Buffer2D.hpp"
|
||||||
|
|
||||||
|
namespace tp {
|
||||||
|
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class TransitionMatrix;
|
||||||
|
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class DFA;
|
||||||
|
|
||||||
|
// Non-Deterministic Finite-State Automata
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class NFA {
|
||||||
|
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
||||||
|
|
||||||
|
public:
|
||||||
|
struct Vertex;
|
||||||
|
|
||||||
|
private:
|
||||||
|
struct Edge {
|
||||||
|
Vertex* mVertex = nullptr;
|
||||||
|
bool mConsumesSymbol = false;
|
||||||
|
Range<tAlphabetType> mAcceptingRange;
|
||||||
|
bool mAcceptsAll = false;
|
||||||
|
bool mExclude = false;
|
||||||
|
|
||||||
|
bool isTransition(const tAlphabetType& symbol) {
|
||||||
|
if (symbol == 0) return false;
|
||||||
|
if (!mConsumesSymbol || mAcceptsAll) return true;
|
||||||
|
bool const in_range = (symbol >= mAcceptingRange.mBegin && symbol <= mAcceptingRange.mEnd);
|
||||||
|
return in_range != mExclude;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
public:
|
||||||
|
struct Vertex {
|
||||||
|
List<Edge> edges;
|
||||||
|
tStateType termination_state = tNoStateVal;
|
||||||
|
#ifdef ENV_BUILD_DEBUG
|
||||||
|
ualni debug_idx = 0;
|
||||||
|
#endif
|
||||||
|
ualni flag = 0;
|
||||||
|
};
|
||||||
|
|
||||||
|
private:
|
||||||
|
friend DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
|
||||||
|
|
||||||
|
List<Vertex> mVertices;
|
||||||
|
Vertex* mStart = nullptr;
|
||||||
|
|
||||||
|
public:
|
||||||
|
NFA() = default;
|
||||||
|
|
||||||
|
Vertex* addVertex() {
|
||||||
|
auto node = mVertices.newNode();
|
||||||
|
#ifdef ENV_BUILD_DEBUG
|
||||||
|
node->data.debug_idx = mVertices.length() + 1;
|
||||||
|
#endif
|
||||||
|
mVertices.pushBack(node);
|
||||||
|
return &node->data;
|
||||||
|
}
|
||||||
|
|
||||||
|
void
|
||||||
|
addTransition(Vertex* from, Vertex* to, Range<tAlphabetType> range, bool consumes, bool accepts_all, bool exclude) {
|
||||||
|
Edge edge;
|
||||||
|
edge.mVertex = to;
|
||||||
|
edge.mConsumesSymbol = consumes;
|
||||||
|
edge.mAcceptingRange = range;
|
||||||
|
edge.mExclude = exclude;
|
||||||
|
edge.mAcceptsAll = accepts_all;
|
||||||
|
from->edges.pushBack(edge);
|
||||||
|
}
|
||||||
|
|
||||||
|
void setStartVertex(Vertex* start) { mStart = start; }
|
||||||
|
|
||||||
|
[[nodiscard]] Vertex* getStartVertex() const { return mStart; }
|
||||||
|
|
||||||
|
void setVertexState(Vertex* vertex, tStateType state) { vertex->termination_state = state; }
|
||||||
|
|
||||||
|
[[nodiscard]] bool isValid() const {
|
||||||
|
if (!mStart) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
Range<tAlphabetType> getAlphabetRange() const {
|
||||||
|
tAlphabetType start = 0, end = 0;
|
||||||
|
Range<tAlphabetType> all_range(
|
||||||
|
std::numeric_limits<tAlphabetType>::min(), std::numeric_limits<tAlphabetType>::max()
|
||||||
|
);
|
||||||
|
bool first = true;
|
||||||
|
|
||||||
|
for (auto vertex : mVertices) {
|
||||||
|
for (auto edge : vertex.data().edges) {
|
||||||
|
if (!edge.data().mConsumesSymbol) continue;
|
||||||
|
auto const& tran_range = edge.data().mAcceptingRange;
|
||||||
|
if (edge.data().mAcceptsAll || edge.data().mExclude) return all_range;
|
||||||
|
if (tran_range.mBegin < start || first) start = tran_range.mBegin;
|
||||||
|
if (tran_range.mEnd > end || first) end = tran_range.mEnd;
|
||||||
|
first = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return Range<tAlphabetType>(start, end + 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
// vertices that are reachable from initial set with no input consumption (E-transitions)
|
||||||
|
// does not include initial set
|
||||||
|
void closure(const List<Vertex*>& set, List<Vertex*>& closure, ualni unique_call_id) {
|
||||||
|
List<Vertex*> marked;
|
||||||
|
marked = set;
|
||||||
|
|
||||||
|
while (marked.length()) {
|
||||||
|
auto first = marked.first()->data;
|
||||||
|
if (first->flag != unique_call_id) {
|
||||||
|
first->flag = unique_call_id;
|
||||||
|
closure.pushBack(first);
|
||||||
|
}
|
||||||
|
for (auto edge : first->edges) {
|
||||||
|
if (!edge.data().mConsumesSymbol && edge.data().mVertex->flag != unique_call_id) {
|
||||||
|
marked.pushBack(edge.data().mVertex);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
marked.popFront();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// vertices that are reachable from initial set with symbol transition
|
||||||
|
void move(const List<Vertex*>& set, List<Vertex*>& reachable, tAlphabetType symbol, ualni unique_call_id) {
|
||||||
|
for (auto vertex : set) {
|
||||||
|
for (auto edge : vertex->edges) {
|
||||||
|
if (!edge.data().mConsumesSymbol) continue;
|
||||||
|
bool transition = edge.data().isTransition(symbol);
|
||||||
|
if (transition && edge.data().mVertex->flag != unique_call_id) {
|
||||||
|
edge.data().mVertex->flag = unique_call_id;
|
||||||
|
reachable.pushBack(edge.data().mVertex);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Deterministic Finite-State Automata
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class DFA {
|
||||||
|
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
||||||
|
friend TransitionMatrix<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>;
|
||||||
|
|
||||||
|
struct Vertex {
|
||||||
|
struct Edge {
|
||||||
|
Vertex* vertex = nullptr;
|
||||||
|
tAlphabetType transition_code = nullptr;
|
||||||
|
};
|
||||||
|
|
||||||
|
List<Edge> edges;
|
||||||
|
tStateType termination_state = tNoStateVal;
|
||||||
|
bool marked = false;
|
||||||
|
};
|
||||||
|
|
||||||
|
List<Vertex> mVertices;
|
||||||
|
Vertex* mStart = nullptr;
|
||||||
|
const Vertex* mIter = nullptr;
|
||||||
|
Range<tAlphabetType> mAlphabetRange;
|
||||||
|
bool mTrapState = false;
|
||||||
|
|
||||||
|
typedef typename NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>::Vertex NState;
|
||||||
|
|
||||||
|
public:
|
||||||
|
explicit DFA(NFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& nfa) {
|
||||||
|
if (!nfa.isValid()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
mAlphabetRange = nfa.getAlphabetRange();
|
||||||
|
|
||||||
|
struct DStateKey {
|
||||||
|
const List<NState*>* nStates;
|
||||||
|
bool operator==(const DStateKey& in) const {
|
||||||
|
if (nStates->length() != in.nStates->length()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// FIXME : make linear time
|
||||||
|
for (auto state : *nStates) {
|
||||||
|
bool found = false;
|
||||||
|
for (auto in_state : *in.nStates) {
|
||||||
|
if (state.data() == in_state.data()) {
|
||||||
|
found = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!found) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
|
||||||
|
static ualni dStateHashFunc(DStateKey key) {
|
||||||
|
alni out = 0;
|
||||||
|
for (auto state : *key.nStates) {
|
||||||
|
out += alni(state.data());
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
};
|
||||||
|
};
|
||||||
|
|
||||||
|
// all NFA states that are reachable from initial DFA State for specific symbol in alphabet
|
||||||
|
struct DState {
|
||||||
|
struct DTransition {
|
||||||
|
DState* state = nullptr;
|
||||||
|
tAlphabetType accepting_code;
|
||||||
|
};
|
||||||
|
|
||||||
|
List<NState*> nStates;
|
||||||
|
List<DTransition> transitions;
|
||||||
|
|
||||||
|
Vertex* dVertex = nullptr; // relevant DFA vertex
|
||||||
|
|
||||||
|
#ifdef ENV_BUILD_DEBUG
|
||||||
|
ualni debug_idx = 0;
|
||||||
|
#endif
|
||||||
|
};
|
||||||
|
|
||||||
|
// includes closure of NFA start state by definition
|
||||||
|
auto start_state = new DState();
|
||||||
|
nfa.closure({ nfa.getStartVertex() }, start_state->nStates, ualni(start_state));
|
||||||
|
|
||||||
|
Map<DStateKey, DState*, DefaultAllocator, DStateKey::dStateHashFunc, 256> dStates;
|
||||||
|
|
||||||
|
dStates.put({ &start_state->nStates }, start_state);
|
||||||
|
|
||||||
|
List<DState*> working_set = { start_state };
|
||||||
|
|
||||||
|
// while there is items to work with
|
||||||
|
auto currentDState = working_set.first();
|
||||||
|
while (currentDState) {
|
||||||
|
|
||||||
|
// check all possible transitions for any symbol
|
||||||
|
for (auto symbol : mAlphabetRange) {
|
||||||
|
|
||||||
|
List<NState*> reachableNStates;
|
||||||
|
|
||||||
|
nfa.move(currentDState->data->nStates, reachableNStates, symbol, ualni(currentDState + symbol));
|
||||||
|
nfa.closure(reachableNStates, reachableNStates, ualni(currentDState + symbol));
|
||||||
|
|
||||||
|
if (!reachableNStates.length()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
DState* targetDState = nullptr;
|
||||||
|
|
||||||
|
// check if set of all reachable NFA states already forms existing DFA state
|
||||||
|
auto idx = dStates.presents({ &reachableNStates });
|
||||||
|
if (idx) {
|
||||||
|
targetDState = dStates.getSlotVal(idx);
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!targetDState) {
|
||||||
|
// register new DFA state
|
||||||
|
targetDState = new DState();
|
||||||
|
|
||||||
|
#ifdef ENV_BUILD_DEBUG
|
||||||
|
targetDState->debug_idx = dStates.size();
|
||||||
|
#endif
|
||||||
|
|
||||||
|
targetDState->nStates = reachableNStates;
|
||||||
|
|
||||||
|
// append to working stack
|
||||||
|
working_set.pushBack(targetDState);
|
||||||
|
dStates.put({ &targetDState->nStates }, targetDState);
|
||||||
|
}
|
||||||
|
|
||||||
|
// add transition to DFA state
|
||||||
|
currentDState->data->transitions.pushBack({ targetDState, symbol });
|
||||||
|
}
|
||||||
|
|
||||||
|
working_set.popFront();
|
||||||
|
currentDState = working_set.first();
|
||||||
|
}
|
||||||
|
|
||||||
|
// create own vertices
|
||||||
|
for (auto node : dStates) {
|
||||||
|
tStateType state = tNoStateVal;
|
||||||
|
for (auto iter : node->val->nStates) {
|
||||||
|
if (iter->termination_state != tNoStateVal) {
|
||||||
|
state = iter->termination_state;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
node->val->dVertex = addVertex(state);
|
||||||
|
}
|
||||||
|
|
||||||
|
// connect all vertices
|
||||||
|
for (auto node : dStates) {
|
||||||
|
for (auto edge : node->val->transitions) {
|
||||||
|
addTransition(node->val->dVertex, edge.data().state->dVertex, edge.data().accepting_code);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// set the starting vertex
|
||||||
|
mStart = start_state->dVertex;
|
||||||
|
|
||||||
|
// cleanup
|
||||||
|
for (auto node : dStates) {
|
||||||
|
delete node->val;
|
||||||
|
}
|
||||||
|
|
||||||
|
collapseEquivalentVertices();
|
||||||
|
mAlphabetRange = getAlphabetRange();
|
||||||
|
}
|
||||||
|
|
||||||
|
tStateType move(tAlphabetType symbol) {
|
||||||
|
if (mTrapState || !mIter) {
|
||||||
|
return tNoStateVal;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (auto edge : mIter->edges) {
|
||||||
|
if (edge.data().transition_code == symbol) {
|
||||||
|
mIter = edge.data().vertex;
|
||||||
|
return mIter->termination_state;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
mTrapState = true;
|
||||||
|
return tNoStateVal;
|
||||||
|
}
|
||||||
|
|
||||||
|
void start() {
|
||||||
|
mIter = mStart;
|
||||||
|
mTrapState = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
[[nodiscard]] uhalni nVertices() const { return (uhalni) mVertices.length(); }
|
||||||
|
|
||||||
|
[[nodiscard]] Range<tAlphabetType> getRange() const { return mAlphabetRange; }
|
||||||
|
|
||||||
|
private:
|
||||||
|
[[nodiscard]] Range<tAlphabetType> getAlphabetRange() const {
|
||||||
|
Range<tAlphabetType> out;
|
||||||
|
for (auto vertex : mVertices) {
|
||||||
|
vertex.data().marked = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool first = true;
|
||||||
|
getAlphabetRangeUtil(mStart, &out, first);
|
||||||
|
out.mEnd++;
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
void getAlphabetRangeUtil(Vertex* vert, Range<tAlphabetType>* out, bool& first) const {
|
||||||
|
vert->marked = true;
|
||||||
|
for (auto edge : vert->edges) {
|
||||||
|
auto const code = edge.data().transition_code;
|
||||||
|
|
||||||
|
if (first) {
|
||||||
|
*out = { code, code };
|
||||||
|
first = false;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (code < out->mBegin) {
|
||||||
|
out->mBegin = code;
|
||||||
|
}
|
||||||
|
if (code > out->mEnd) {
|
||||||
|
out->mEnd = code;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!edge.data().vertex->marked) {
|
||||||
|
getAlphabetRangeUtil(edge.data().vertex, out, first);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void collapseEquivalentVertices() {
|
||||||
|
// TODO
|
||||||
|
}
|
||||||
|
|
||||||
|
Vertex* addVertex(tStateType state) {
|
||||||
|
auto node = mVertices.addNodeBack();
|
||||||
|
node->data.termination_state = state;
|
||||||
|
return &node->data;
|
||||||
|
}
|
||||||
|
|
||||||
|
void addTransition(Vertex* from, Vertex* to, tAlphabetType transition_symbol) {
|
||||||
|
from->edges.pushBack({ to, transition_symbol });
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
template <typename tAlphabetType, typename tStateType, tStateType tNoStateVal, tStateType tFailedStateVal>
|
||||||
|
class TransitionMatrix {
|
||||||
|
|
||||||
|
static_assert(TypeTraits<tAlphabetType>::isIntegral, "tAlphabetType must be enumerable.");
|
||||||
|
|
||||||
|
Buffer2D<ualni> mTransitions;
|
||||||
|
Buffer<tStateType> mStates;
|
||||||
|
Range<tAlphabetType> mSymbolRange = { 0, 0 };
|
||||||
|
|
||||||
|
ualni mIter = 0;
|
||||||
|
ualni mIterPrev = 0;
|
||||||
|
ualni mStart = 0;
|
||||||
|
|
||||||
|
public:
|
||||||
|
TransitionMatrix() = default;
|
||||||
|
|
||||||
|
auto getStates() const { return &mStates; }
|
||||||
|
auto getTransitions() const { return &mTransitions; }
|
||||||
|
auto getStart() const { return mStart; }
|
||||||
|
|
||||||
|
void construct(const DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& dfa) {
|
||||||
|
mSymbolRange = dfa.getRange();
|
||||||
|
auto range_len = ualni(mSymbolRange.mEnd - mSymbolRange.mBegin);
|
||||||
|
auto sizeX = range_len ? range_len : 1;
|
||||||
|
auto sizeY = (ualni) (dfa.nVertices() + 1);
|
||||||
|
|
||||||
|
mTransitions.reserve({ sizeX, sizeY });
|
||||||
|
mTransitions.assign(dfa.nVertices());
|
||||||
|
mStates.reserve(sizeY);
|
||||||
|
|
||||||
|
ualni idx = 0;
|
||||||
|
for (auto vertex : dfa.mVertices) {
|
||||||
|
auto state = vertex.data().termination_state;
|
||||||
|
mStates[idx] = state;
|
||||||
|
idx++;
|
||||||
|
}
|
||||||
|
|
||||||
|
mStates[dfa.nVertices()] = tFailedStateVal;
|
||||||
|
|
||||||
|
idx = 0;
|
||||||
|
for (auto vertex : dfa.mVertices) {
|
||||||
|
if (&vertex.data() == dfa.mStart) {
|
||||||
|
mStart = mIter = mIterPrev = idx;
|
||||||
|
}
|
||||||
|
idx++;
|
||||||
|
}
|
||||||
|
|
||||||
|
ualni vertexIdx = 0;
|
||||||
|
for (auto vertex : dfa.mVertices) {
|
||||||
|
for (auto edge : vertex.data().edges) {
|
||||||
|
ualni vertex2Idx = 0;
|
||||||
|
for (auto vertex2 : dfa.mVertices) {
|
||||||
|
if (edge.data().vertex == &vertex2.data()) break;
|
||||||
|
vertex2Idx++;
|
||||||
|
}
|
||||||
|
auto const code = edge.data().transition_code;
|
||||||
|
mTransitions.set({ (ualni) (code - mSymbolRange.mBegin), (ualni) vertexIdx }, vertex2Idx);
|
||||||
|
}
|
||||||
|
vertexIdx++;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
bool isTrapped() { return mStates[mIter] == tFailedStateVal; }
|
||||||
|
|
||||||
|
tStateType move(tAlphabetType symbol) {
|
||||||
|
if (symbol >= mSymbolRange.mBegin && symbol < mSymbolRange.mEnd) {
|
||||||
|
mIter = mTransitions.get({ (ualni) (symbol - mSymbolRange.mBegin), (ualni) mIter });
|
||||||
|
} else {
|
||||||
|
mIter = mStates.size() - 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (mIterPrev == mStart) {
|
||||||
|
if (mStates[mIter] == tFailedStateVal) {
|
||||||
|
reset();
|
||||||
|
return tFailedStateVal;
|
||||||
|
} else {
|
||||||
|
mIterPrev = mIter;
|
||||||
|
return tNoStateVal;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
if (mStates[mIter] == tFailedStateVal) {
|
||||||
|
if (mStates[mIterPrev] != tNoStateVal) {
|
||||||
|
auto out = mStates[mIterPrev];
|
||||||
|
reset();
|
||||||
|
return out;
|
||||||
|
} else {
|
||||||
|
reset();
|
||||||
|
return tFailedStateVal;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
mIterPrev = mIter;
|
||||||
|
return tNoStateVal;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
mIterPrev = mIter;
|
||||||
|
return mStates[mIter];
|
||||||
|
}
|
||||||
|
|
||||||
|
void reset() {
|
||||||
|
mIter = mStart;
|
||||||
|
mIterPrev = mStart;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
@ -7,6 +7,8 @@ namespace tp {
|
||||||
// Gives ability to express grammar in the Unified Format as sentence
|
// Gives ability to express grammar in the Unified Format as sentence
|
||||||
template <typename tAlphabetType>
|
template <typename tAlphabetType>
|
||||||
class SimpleParser {
|
class SimpleParser {
|
||||||
|
enum UGTokens : ualni { InTransition = 0, Failed, TestSeq };
|
||||||
|
|
||||||
public:
|
public:
|
||||||
SimpleParser() {
|
SimpleParser() {
|
||||||
// Grammar for unified grammar format sentence that tables compiled from
|
// Grammar for unified grammar format sentence that tables compiled from
|
||||||
|
|
@ -20,10 +22,10 @@ namespace tp {
|
||||||
}
|
}
|
||||||
|
|
||||||
// Define Regular grammar
|
// Define Regular grammar
|
||||||
RegularGrammar<tAlphabetType, ualni> rg;
|
RegularGrammar<tAlphabetType, UGTokens> rg;
|
||||||
{
|
{
|
||||||
// this is basically ast from existing tokenizer
|
// this is basically ast from existing tokenizer
|
||||||
rg.addRule(rg.seq({ rg.val('a'), rg.val('b') }), 0);
|
rg.addRule(rg.seq({ rg.val('a'), rg.val('b') }), TestSeq);
|
||||||
}
|
}
|
||||||
|
|
||||||
mUnifiedGrammarParser.compileTables(contextFreeGrammar, rg);
|
mUnifiedGrammarParser.compileTables(contextFreeGrammar, rg);
|
||||||
|
|
@ -36,7 +38,7 @@ namespace tp {
|
||||||
|
|
||||||
// compile each ast into RegularGrammar and ContextFree Grammar api instructions
|
// compile each ast into RegularGrammar and ContextFree Grammar api instructions
|
||||||
ContextFreeGrammar userContextFreeGrammar;
|
ContextFreeGrammar userContextFreeGrammar;
|
||||||
RegularGrammar<tAlphabetType, ualni> userRegularGrammar;
|
RegularGrammar<tAlphabetType, UGTokens> userRegularGrammar;
|
||||||
|
|
||||||
// ...
|
// ...
|
||||||
// split ast into RE and CF part
|
// split ast into RE and CF part
|
||||||
|
|
@ -48,10 +50,12 @@ namespace tp {
|
||||||
mUserParser.compileTables(userContextFreeGrammar, userRegularGrammar);
|
mUserParser.compileTables(userContextFreeGrammar, userRegularGrammar);
|
||||||
}
|
}
|
||||||
|
|
||||||
void parse(const tAlphabetType* grammar, ualni grammarLength, AST& out) { mUserParser.parse(grammar, grammarLength, out); }
|
void parse(const tAlphabetType* grammar, ualni grammarLength, AST& out) {
|
||||||
|
mUserParser.parse(grammar, grammarLength, out);
|
||||||
|
}
|
||||||
|
|
||||||
private:
|
private:
|
||||||
Parser<tAlphabetType> mUnifiedGrammarParser;
|
Parser<tAlphabetType, UGTokens> mUnifiedGrammarParser;
|
||||||
Parser<tAlphabetType> mUserParser;
|
Parser<tAlphabetType, UGTokens> mUserParser;
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -1,5 +1,4 @@
|
||||||
|
|
||||||
#include "NewPlacement.hpp"
|
|
||||||
#include "Test.hpp"
|
#include "Test.hpp"
|
||||||
|
|
||||||
using namespace tp;
|
using namespace tp;
|
||||||
|
|
|
||||||
|
|
@ -36,12 +36,14 @@ namespace tp {
|
||||||
T1 t1;
|
T1 t1;
|
||||||
T1 head;
|
T1 head;
|
||||||
T1 x;
|
T1 x;
|
||||||
|
T1 first;
|
||||||
};
|
};
|
||||||
|
|
||||||
union {
|
union {
|
||||||
T2 t2;
|
T2 t2;
|
||||||
T2 tail;
|
T2 tail;
|
||||||
T2 y;
|
T2 y;
|
||||||
|
T2 second;
|
||||||
};
|
};
|
||||||
|
|
||||||
bool operator==(const Pair& in) const { return in.t1 == t1 && in.t2 == t2; }
|
bool operator==(const Pair& in) const { return in.t1 == t1 && in.t2 == t2; }
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue