#include #include #include #include #include #include using namespace std; /*Detects and merges binary operators*/ smatch bin_ops_merger (vector&, size_t); /*Combines two vectors into a pair*/ template vector> merge_vectors (const vector&, const vector&); /*Assigns Matching Tokens to Lexemes*/ void tokenizer (vector&, vector>&); int main(int argc, char** argv) { //store full source file in a string string file, line; ifstream ifs("analyzethis.file"); while (getline(ifs, line, '\0')){ file+= line; } ifs.close(); /*Strip single and multi line comments*/ regex rexComments("(//.*)|(/\\*(?:.|[\\n\\r])*?\\*/)"); string result; regex_replace(std::back_inserter(result), file.begin(), file.end(), rexComments, " "); file= result; /* REGEX Patterns: * Not alphanumeric [\\W] * number: (\\d+).(\\d+) * string: (\".*\") */ regex rexPtrn("[\\W]|(\\d+).(\\d+)|(\".*\")"); //only alpha numeric, reverses negation regex_token_iterator rtiNS(file.begin(), file.end(), rexPtrn, -1); //only symbols regex_token_iterator rtiS(file.begin(), file.end(), rexPtrn); //end of line comparison regex_token_iterator rtiEnd; vector vData; //holds token while ((rtiNS!=rtiEnd)&&(rtiS!=rtiEnd)){ if((*rtiNS).length()>0) //if not symbol vData.push_back(*rtiNS); if((*rtiS).length()>0&&*rtiS!=" "&&*rtiS!="\t"&&*rtiS!="\n") vData.push_back(*rtiS); //advance iterators ++rtiNS; ++rtiS; } //FIND BINARY OPERATORS AND COMBINE THEM for (size_t ctr= 0; ctr < vData.size() - 1; ++ctr) { bin_ops_merger(vData, ctr); } /*Holds final matches*/ vector> vTokenLexeme; tokenizer(vData, vTokenLexeme); //Save tokenized lexemes ofstream ofs("tokenized.txt"); ofs< vector> merge_vectors (const vector& v1, const vector& v2) { vector> vOut; for(size_t i= 0; i< v1.size(); ++i){ vOut.emplace_back(v1.at(i), v2.at(i)); } return vOut; } /*Detects and merges binary operators*/ smatch bin_ops_merger (vector& vData, size_t ctr) { regex binaryOperatorsPattern("\\+=|-=|\\*=|/=|%=|&=|\\!=|==|\\|=|\\^=" "|<=|>=|--|\\+\\+|<<|>>|&&|\\|\\||->"); vector ::iterator curr, next; curr= vData.begin()+ctr; next= vData.begin()+ctr+1; string str= *curr+*next; //run regex pattern on this string smatch binOpsMatch; //stores the matched partion if (regex_match(str, binOpsMatch, binaryOperatorsPattern)) { *curr= *curr + *next; //merge operators vData.erase(next); //delete extra element } return binOpsMatch; } /*Assigns Matching Tokens to Lexemes*/ void tokenizer (vector& vLex, vector>& vTknLex) { //Reserved Key Words vector vKword; vKword={"string","include","auto","const","struct","unsigned","break", "continue","else","for","signed","switch","void","case","default", "enum","goto","register","sizeof","typedef","volatile","char","do", "extern","if","return","static","union","while","asm","dynamic_cast", "namespace","reinterpret_cast","try","bool","explicit","new","template", "static_cast","typeid","catch","false","operator","typename","public", "class","friend","private","this","using","const_cast","inline","throw", "virtual","delete","mutable","protected","true","elseif"}; vector vDataTypes; vDataTypes={"double","float","int","short","size_t","long","string"}; //Binary Operators vector vbotkn, vbolex; vbotkn={"+=", "-=", "*=", "/=", "%=", "&=", "!=", "==", "|=", "^=", "<=", ">=","--", "++", "<<", ">>", "&&", "||", "->",":"}; vbolex={"ADD_ASSIGN","SUB_ASSIGN","MUL_ASSIGN","DIV_ASSIGN","MOD_ASSIGN", "AND_ASSIGN","LOGIC_INEQ","LOGIC_EQ","OR_ASSIGN","XOR_ASSIGN", "LESS_OR_EQ","MORE_OR_EQ","DECREMENT","INCREMENT","INSERTION", "EXTRACTION","LOGIC_AND","LOGIC_OR","MEMBER_PTR","SCOPE_RES"}; vector> vBOpsTokens= merge_vectors(vbotkn, vbolex); //Unary Symbols vector vsymtkn, vsymlex; vsymtkn={".","#",",","=","-","+","/","*","%","(",")","{","}","[","]","~", "^","|","&","?",":",";","!",">","<"}; vsymlex={"MEMBER_OBJ","PREPROC","SEPARATOR","ASSIGN","SUB","ADD","DIV", "MUL_OR_DEREF","MOD","L_PAREN","R_PAREN","L_BRACE","R_BRACE","L_BRACKET", "R_BRACKET","COMPLEMENT","XOR","OR","AND","CONDITIONAL","COND_SEP", "SEMI_COLON","NOT","GREATER_THAN","LESS_THAN"}; vector> vUnaryTokens= merge_vectors(vsymtkn, vsymlex); //Library Objects vector vLibObj; vLibObj={"cout","cin","printf","size","sizeof","system","getline","endl", "to_string"}; /* * Search lexemes for token matches * */ for(size_t lexItr= 0; lexItr