Small tokenizer-related classes

git-svn-id: https://swig.svn.sourceforge.net/svnroot/swig/branches/gsoc2008-cherylfoil@10659 626c5289-ae23-0410-ae9c-e8d60b6d4f22
This commit is contained in:
Cheryl Foil 2008-07-13 04:51:51 +00:00
commit da4e3cd374
4 changed files with 180 additions and 0 deletions

View file

@ -0,0 +1,32 @@
#include "Token.h"
#include <cstdlib>
#include <iostream>
#include <string>
#include <list>
using namespace std;
Token::Token(int tType, string tString)
{
tokenType = tType;
tokenString = tString;
}
string Token::toString()
{
if (tokenType == END_LINE){
return "{END OF LINE}";
}
if (tokenType == PARAGRAPH_END){
return "{END OF PARAGRAPH}";
}
if (tokenType == PLAINSTRING){
return tokenString;
}
if (tokenType == COMMAND){
return "{COMMAND : " + tokenString+ "}";
}
return "";
}
Token:: ~Token(){}

View file

@ -0,0 +1,21 @@
#ifndef TOKEN_H_
#define TOKEN_H_
#include <string>
#define END_LINE 101
#define PARAGRAPH_END 102
#define PLAINSTRING 103
#define COMMAND 104
using namespace std;
class Token
{
public:
Token(int tType, string tString);
~Token();
int tokenType;
string tokenString;
string toString();
};
#endif /*TOKEN_H_*/

View file

@ -0,0 +1,103 @@
#include "TokenList.h"
#include <cstdlib>
#include <iostream>
#include <string>
#include <list>
#include "Token.h"
#define TOKENSPERLINE 8; //change this to change the printing behaviour of the token list
#define END_LINE 101
#define PARAGRAPH_END 102 //not used at the moment
#define PLAINSTRING 103
#define COMMAND 104
using namespace std;
list <Token> tokenList;
list<Token>::iterator tokenListIterator;
int noisy2 = 0;
/* The tokenizer*/
TokenList::TokenList(string doxygenString){
int currentIndex = 0;
//Regex whitespace("[ \t]+");
//Regex newLine("[\n]");
//Regex command("[@|\\]{1}[^ \t \n]+"); //the cheap solution
//Regex doxygenFluff("[/*!]+");
int nextIndex = 0;
int isFluff = 0;
string currentWord;
while (currentIndex < doxygenString.length()){
if(doxygenString[currentIndex] == '\n'){
tokenList.push_back(Token(END_LINE, currentWord));
currentIndex++;
}
while(currentIndex < doxygenString.length() && (doxygenString[currentIndex] == ' '
|| doxygenString[currentIndex]== '\t')) currentIndex ++;
if (currentIndex == doxygenString.length()) {} //do nothing since end of string was reached
else {nextIndex = currentIndex;
while (nextIndex < doxygenString.length() && (doxygenString[nextIndex] != ' '
&& doxygenString[nextIndex]!= '\t' && doxygenString[nextIndex]!= '\n')) nextIndex++;
currentWord = doxygenString.substr(currentIndex, nextIndex-currentIndex);
if(noisy2) cout << "Current Word: " << currentWord << endl;
if (currentWord[0] == '@' || currentWord[0] == '\\'){
currentWord = currentWord.substr(1, currentWord.length()-1);
tokenList.push_back(Token(COMMAND, currentWord));
}
else if (currentWord[0] == '\n'){
//if ((tokenList.back()).tokenType == END_LINE){}
tokenList.push_back(Token(END_LINE, currentWord));
}
else if (currentWord[0] == '*' || currentWord[0] == '/' ||currentWord[0] == '!'){
if (currentWord.length() == 1) {isFluff = 1;}
else { isFluff = 1;
for(int i = 1; i < currentWord.length(); i++){
if (currentWord[0] != '*' && currentWord[0] != '/' && currentWord[0] != '!') isFluff = 0;
}
}
if(!isFluff) tokenList.push_back(Token(PLAINSTRING, currentWord));
}
else tokenList.push_back(Token(PLAINSTRING, currentWord));
currentIndex = nextIndex;
}
}
tokenListIterator = tokenList.begin();
}
Token TokenList::peek(){
list<Token>::iterator p = tokenList.begin();
if(p != tokenList.end()){
p++;
Token returnedToken = (*p);
p--;
return returnedToken;
}
else
return Token(0, "");
}
Token TokenList::next(){
list<Token>::iterator p = tokenList.begin();
if(p != tokenList.end()){
p++;
return (*p);
}
else
return Token(0, "");
}
void TokenList::printList(){
list<Token>::iterator p = tokenList.begin();
int i = 1;
int b = 0;
while (p != tokenList.end()){
cout << (*p).toString() << " ";
b = i%TOKENSPERLINE;
if (b == 0) cout << endl;
p++; i++;
}
}
TokenList:: ~TokenList(){}

View file

@ -0,0 +1,24 @@
#ifndef TOKENLIST_H_
#define TOKENLIST_H_
#include <cstdlib>
#include <iostream>
#include <string>
#include <list>
#include "Token.h"
using namespace std;
/* a small class used to represent the sequence of tokens
* that can be derived from a formatted doxygen string
*/
class TokenList{
public:
/* constructor takes a blob of Doxygen comment */
TokenList(string doxygenString);
~TokenList();
Token peek(); /* returns next token without advancing */
Token next(); /* returns next token and advances */
void printList(); /* prints out the sequence of tokens */
};
#endif /*TOKENLIST_H_*/