Fixing tokenizer bug

This commit is contained in:
IlushaShurupov 2023-08-10 18:33:17 +03:00 committed by Ilya Shurupov
parent a938bf9bb1
commit 92e772cde1
7 changed files with 27 additions and 25 deletions

View file

@ -422,6 +422,10 @@ namespace tp {
TransitionMatrix() = default;
auto getStates() const { return &mStates; }
auto getTransitions() const { return &mTransitions; }
auto getStart() const { return mStart; }
void construct(const DFA<tAlphabetType, tStateType, tNoStateVal, tFailedStateVal>& dfa) {
mSymbolRange = dfa.getRange();
auto range_len = ualni(mSymbolRange.mEnd - mSymbolRange.mBegin);

View file

@ -351,17 +351,15 @@ namespace tp::RegEx {
halni idx = 0;
for (auto rule: aRules) {
auto node = idx ? compileUtil(rule.head, rule.tail) : compileUtil(rule.head, rule.tail, left, right);
auto node = compileUtil(rule.head, rule.tail);
if (!(node.left && node.right)) {
mError.mRuleIndex = idx;
return {};
}
if (idx) {
transitionAny(left, node.left);
transitionAny(node.right, right);
}
transitionAny(left, node.left);
transitionAny(node.right, right);
idx++;
}
@ -373,7 +371,7 @@ namespace tp::RegEx {
private:
Node compileUtil(const tAlphabetType* regex, tStateType state, Vertex* aLeft = nullptr, Vertex* aRight = nullptr) {
Node compileUtil(const tAlphabetType* regex, tStateType state) {
Parser parser;
auto astNode = parser.parse(regex);
if (parser.mError.isError()) {
@ -384,7 +382,7 @@ namespace tp::RegEx {
return {};
}
auto node = compileNode(astNode, aLeft, aRight);
auto node = compileNode(astNode, nullptr, nullptr);
delete astNode;
mGraph->setVertexState(node.right, state);

View file

@ -50,6 +50,8 @@ namespace tp {
return !mError.isError();
}
auto getMatrix() const { return &mTransitionMatrix; }
const RegEx::CompileError<tTokType>& getBuildError() {
return mError;
}
@ -80,8 +82,11 @@ namespace tp {
template <typename tAlphabetType, typename tTokType, tTokType tNoTokVal, tTokType tFailedTokVal, tTokType tSourceEndTokVal>
class SimpleTokenizer {
Tokenizer<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> mTokenizer;
public:
typedef Tokenizer<tAlphabetType, tTokType, tNoTokVal, tFailedTokVal> tTokenizer;
private:
tTokenizer mTokenizer;
const tAlphabetType* mSource = nullptr;
ualni mLastTokLen = 0;
@ -117,7 +122,7 @@ namespace tp {
};
SimpleTokenizer() = default;
auto getTokenizer() const { return mTokenizer; }
auto getTokenizer() const { return &mTokenizer; }
void build(const InitialierList<Pair<const tAlphabetType*, tTokType>>& rules) {
mTokenizer.build(rules);

View file

@ -30,7 +30,7 @@ return
)";
TEST_DEF_STATIC(Simple) {
TEST_DEF_STATIC(General) {
enum class TokType {
NONE,
@ -193,7 +193,7 @@ TEST_DEF_STATIC(Simple) {
TEST(outputHash == outputHashPassed);
}
TEST_DEF(Covarage) {
TEST_DEF(Simple) {
enum class TokType {
START = 0,
NONE = 1,
@ -207,9 +207,9 @@ TEST_DEF(Covarage) {
SimpleTokenizer<char, TokType, TokType::NONE, TokType::FAILED, TokType::TOK_SOURCE_END> lexer;
lexer.build({
{ " ", TokType::SPACE },
{ "([a-c]|A)+", TokType::ID },
{ "A", TokType::START },
{ " ", TokType::SPACE },
});
if (!lexer.isBuild()) {
@ -218,7 +218,6 @@ TEST_DEF(Covarage) {
}
lexer.bindSource(" A bc cb cc ");
/* 3 5 3 0 3 4 3 4 3 6 */
TokType tok;
ualni outputHash = 0;
@ -229,10 +228,10 @@ TEST_DEF(Covarage) {
printf(" %i ", int(tok));
} while (tok != TokType::TOK_SOURCE_END && tok != TokType::FAILED);
printf(" : %llu", outputHash);
TEST(outputHash == 90);
TEST(outputHash == 33);
}
TEST_DEF(Tokenizer) {
testGeneral();
testSimple();
// testCovarage(); TODO : test environment for an error
}
}