mirror of
https://github.com/ValKmjolnir/Nasal-Interpreter.git
synced 2026-07-26 21:08:45 +08:00
update
This commit is contained in:
+121
-167
@@ -1,38 +1,17 @@
|
||||
#ifndef __NASAL_LEXER_H__
|
||||
#define __NASAL_LEXER_H__
|
||||
|
||||
/*
|
||||
__token_reserve_word:
|
||||
for,foreach,forindex,while : loop
|
||||
var,func : definition
|
||||
break,continue : in loop
|
||||
return : in function
|
||||
if,else,elsif : conditional expr
|
||||
and,or : calculation
|
||||
nil : special type
|
||||
__token_identifier:
|
||||
must begin with '_' or 'a'~'z' or 'A'~'Z'
|
||||
can include '_' or 'a'~'z' or 'A'~'Z' or '0'~'9'
|
||||
__token_string:
|
||||
example:
|
||||
"string"
|
||||
'string'
|
||||
if a string does not end with " or ' then lexer will throw an error
|
||||
__token_number:
|
||||
example:
|
||||
2147483647 (integer)
|
||||
2.71828 (float)
|
||||
0xdeadbeef (hex) or 0xDEADBEEF (hex)
|
||||
0o170001 (oct)
|
||||
1e-1234 (dec) or 10E2 (dec)
|
||||
__token_operator:
|
||||
! + - * / ~
|
||||
= += -= *= /= ~=
|
||||
== != > >= < <=
|
||||
('and' 'or' are operators too but they are recognized as operator in generate_detail_token())
|
||||
() [] {} ; , . : ?
|
||||
others: __unknown_operator
|
||||
*/
|
||||
#define IS_IDENTIFIER_HEAD(c) (c=='_')||('a'<=c && c<='z')||('A'<=c&&c<='Z')
|
||||
#define IS_IDENTIFIER_BODY(c) (c=='_')||('a'<=c && c<='z')||('A'<=c&&c<='Z')||('0'<=c&&c<='9')
|
||||
#define IS_NUMBER_HEAD(c) ('0'<=c&&c<='9')
|
||||
#define IS_NUMBER_BODY(c) ('0'<=c&&c<='9')||('a'<=c&&c<='f')||('A'<=c&&c<='F')||(c=='e'||c=='E'||c=='.'||c=='x'||c=='o')
|
||||
#define IS_STRING_HEAD(c) (c=='\''||c=='\"')
|
||||
// single operators have only one character
|
||||
#define IS_SINGLE_OPRATOR(c) (c=='('||c==')'||c=='['||c==']'||c=='{'||c=='}'||c==','||c==';'||c=='|'||c==':'||\
|
||||
c=='?'||c=='.'||c=='`'||c=='&'||c=='@'||c=='%'||c=='$'||c=='^'||c=='\\')
|
||||
// calculation operators may have two chars, for example: += -= *= /= ~= != == >= <=
|
||||
#define IS_CALC_OPERATOR(c) (c=='='||c=='+'||c=='-'||c=='*'||c=='!'||c=='/'||c=='<'||c=='>'||c=='~')
|
||||
#define IS_NOTE_HEAD(c) (c=='#')
|
||||
|
||||
/* filenames of lib files */
|
||||
#ifndef LIB_FILE_NUM
|
||||
@@ -52,6 +31,7 @@ const std::string lib_filename[LIB_FILE_NUM]=
|
||||
"lib/utf8.nas"
|
||||
};
|
||||
#endif
|
||||
|
||||
/* reserve words */
|
||||
#ifndef RESERVE_WORD_NUM
|
||||
#define RESERVE_WORD_NUM 15
|
||||
@@ -73,21 +53,21 @@ int is_reserve_word(std::string str)
|
||||
|
||||
class resource_file
|
||||
{
|
||||
private:
|
||||
std::vector<char> source_code;
|
||||
public:
|
||||
/*
|
||||
delete_all_source: clear all the source codes in std::list<char> resource
|
||||
input_file : input source codes by filenames
|
||||
load_lib_file : input lib source codes
|
||||
get_source : get the std::vector<char> source_code
|
||||
print_resource : print source codes
|
||||
*/
|
||||
void delete_all_source();
|
||||
void input_file(std::string);
|
||||
void load_lib_file();
|
||||
std::vector<char>& get_source();
|
||||
void print_resource();
|
||||
private:
|
||||
std::vector<char> source_code;
|
||||
public:
|
||||
/*
|
||||
delete_all_source: clear all the source codes in std::list<char> resource
|
||||
input_file : input source codes by filenames
|
||||
load_lib_file : input lib source codes
|
||||
get_source : get the std::vector<char> source_code
|
||||
print_resource : print source codes
|
||||
*/
|
||||
void delete_all_source();
|
||||
void input_file(std::string);
|
||||
void load_lib_file();
|
||||
std::vector<char>& get_source();
|
||||
void print_resource();
|
||||
};
|
||||
|
||||
/* struct token: mainly used in nasal_lexer and nasal_parse*/
|
||||
@@ -100,39 +80,39 @@ struct token
|
||||
{
|
||||
line=tmp.line;
|
||||
type=tmp.type;
|
||||
str=tmp.str;
|
||||
str =tmp.str;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
class nasal_lexer
|
||||
{
|
||||
private:
|
||||
std::list<token> token_list;
|
||||
std::list<token> detail_token_list;
|
||||
int error;
|
||||
std::string identifier_gen(std::vector<char>&,int&,int&);
|
||||
std::string number_gen (std::vector<char>&,int&,int&);
|
||||
std::string string_gen (std::vector<char>&,int&,int&);
|
||||
public:
|
||||
/*
|
||||
identifier_gen : scan the source codes and generate identifiers
|
||||
number_gen : scan the source codes and generate numbers
|
||||
string_gen : scan the source codes and generate strings
|
||||
print_token_list : print generated token list
|
||||
scanner : scan the source codes and generate tokens
|
||||
generate_detail_token: recognize and change token types to detailed types that can be processed by nasal_parse
|
||||
get_error : get the number of errors that occurred when generating tokens
|
||||
get_detail_token : output the detailed tokens,must be used after generate_detail_token()
|
||||
*/
|
||||
nasal_lexer();
|
||||
~nasal_lexer();
|
||||
void delete_all_tokens();
|
||||
void print_token_list();
|
||||
void scanner(std::vector<char>&);
|
||||
void generate_detail_token();
|
||||
int get_error();
|
||||
std::list<token>& get_detail_token_list();
|
||||
private:
|
||||
std::list<token> token_list;
|
||||
std::list<token> detail_token_list;
|
||||
int error;
|
||||
std::string identifier_gen(std::vector<char>&,int&,int&);
|
||||
std::string number_gen (std::vector<char>&,int&,int&);
|
||||
std::string string_gen (std::vector<char>&,int&,int&);
|
||||
public:
|
||||
/*
|
||||
identifier_gen : scan the source codes and generate identifiers
|
||||
number_gen : scan the source codes and generate numbers
|
||||
string_gen : scan the source codes and generate strings
|
||||
print_token_list : print generated token list
|
||||
scanner : scan the source codes and generate tokens
|
||||
generate_detail_token: recognize and change token types to detailed types that can be processed by nasal_parse
|
||||
get_error : get the number of errors that occurred when generating tokens
|
||||
get_detail_token : output the detailed tokens,must be used after generate_detail_token()
|
||||
*/
|
||||
nasal_lexer();
|
||||
~nasal_lexer();
|
||||
void delete_all_tokens();
|
||||
void print_token_list();
|
||||
void scanner(std::vector<char>&);
|
||||
void generate_detail_token();
|
||||
int get_error();
|
||||
std::list<token>& get_detail_token_list();
|
||||
};
|
||||
|
||||
|
||||
@@ -141,6 +121,7 @@ void resource_file::delete_all_source()
|
||||
std::vector<char> tmp;
|
||||
source_code.clear();
|
||||
source_code.swap(tmp);
|
||||
// use tmp's destructor to delete the memory space that source_code used before
|
||||
return;
|
||||
}
|
||||
void resource_file::input_file(std::string filename)
|
||||
@@ -156,9 +137,7 @@ void resource_file::input_file(std::string filename)
|
||||
while(!fin.eof())
|
||||
{
|
||||
c=fin.get();
|
||||
if(fin.eof())
|
||||
break;
|
||||
//source_code.push_back(c<0? '?':c);
|
||||
if(fin.eof()) break;
|
||||
source_code.push_back(c);
|
||||
}
|
||||
fin.close();
|
||||
@@ -178,9 +157,8 @@ void resource_file::load_lib_file()
|
||||
while(!fin.eof())
|
||||
{
|
||||
c=fin.get();
|
||||
if(fin.eof())
|
||||
break;
|
||||
source_code.push_back(c<0? '?':c);
|
||||
if(fin.eof()) break;
|
||||
source_code.push_back(c);
|
||||
}
|
||||
}
|
||||
fin.close();
|
||||
@@ -193,26 +171,25 @@ std::vector<char>& resource_file::get_source()
|
||||
}
|
||||
void resource_file::print_resource()
|
||||
{
|
||||
int size=source_code.size();
|
||||
int line=1;
|
||||
std::cout<<line<<"\t";
|
||||
for(int i=0;i<source_code.size();++i)
|
||||
for(int i=0;i<size;++i)
|
||||
{
|
||||
if(32<=source_code[i])
|
||||
std::cout<<source_code[i];
|
||||
if(32<=source_code[i]) std::cout<<source_code[i];
|
||||
else if(source_code[i]<0)
|
||||
{
|
||||
// print unicode
|
||||
std::string tmp="";
|
||||
for(;i<source_code.size();++i)
|
||||
for(;i<size;++i)
|
||||
{
|
||||
if(source_code[i]>=0)
|
||||
break;
|
||||
if(source_code[i]>=0) break;
|
||||
tmp.push_back(source_code[i]);
|
||||
}
|
||||
std::cout<<tmp;--i;
|
||||
}
|
||||
else
|
||||
std::cout<<" ";
|
||||
if(source_code[i]=='\n')
|
||||
else std::cout<<" ";
|
||||
if(i<size && source_code[i]=='\n')
|
||||
{
|
||||
++line;
|
||||
std::cout<<std::endl<<line<<"\t";
|
||||
@@ -225,42 +202,27 @@ void resource_file::print_resource()
|
||||
std::string nasal_lexer::identifier_gen(std::vector<char>& res,int& ptr,int& line)
|
||||
{
|
||||
std::string token_str="";
|
||||
while(res[ptr]=='_' || ('a'<=res[ptr] && res[ptr]<='z') || ('A'<=res[ptr] && res[ptr]<='Z') || ('0'<=res[ptr] && res[ptr]<='9'))
|
||||
while(IS_IDENTIFIER_BODY(res[ptr]))
|
||||
{
|
||||
token_str+=res[ptr];
|
||||
++ptr;
|
||||
if(ptr>=res.size())
|
||||
break;
|
||||
if(ptr>=res.size()) break;
|
||||
}
|
||||
// check dynamic identifier "..."
|
||||
if(res[ptr]=='.')
|
||||
if(ptr+2<res.size() && res[ptr]=='.' && res[ptr+1]=='.' && res[ptr+2]=='.')
|
||||
{
|
||||
++ptr;
|
||||
if(ptr<res.size() && res[ptr]=='.')
|
||||
{
|
||||
++ptr;
|
||||
if(ptr<res.size() && res[ptr]=='.')
|
||||
{
|
||||
token_str+="...";
|
||||
++ptr;
|
||||
}
|
||||
else
|
||||
ptr-=2;
|
||||
}
|
||||
else
|
||||
--ptr;
|
||||
token_str+="...";
|
||||
ptr+=3;
|
||||
}
|
||||
return token_str;
|
||||
// after running this process, ptr will point to the next token's beginning character
|
||||
}
|
||||
|
||||
std::string nasal_lexer::number_gen(std::vector<char>& res,int& ptr,int& line)
|
||||
{
|
||||
bool scientific_notation=false;
|
||||
bool scientific_notation=false;// numbers like 1e8 are scientific_notation
|
||||
std::string token_str="";
|
||||
while(('0'<=res[ptr] && res[ptr]<='9') ||
|
||||
('a'<=res[ptr] && res[ptr]<='f') ||
|
||||
('A'<=res[ptr] && res[ptr]<='F') ||
|
||||
res[ptr]=='.' || res[ptr]=='x' || res[ptr]=='o' ||
|
||||
res[ptr]=='e' || res[ptr]=='E')
|
||||
while(IS_NUMBER_BODY(res[ptr]))
|
||||
{
|
||||
token_str+=res[ptr];
|
||||
if(res[ptr]=='e' || res[ptr]=='E')
|
||||
@@ -299,14 +261,12 @@ std::string nasal_lexer::string_gen(std::vector<char>& res,int& ptr,int& line)
|
||||
std::string token_str="";
|
||||
char str_begin=res[ptr];
|
||||
++ptr;
|
||||
if(ptr>=res.size())
|
||||
return token_str;
|
||||
if(ptr>=res.size()) return token_str;
|
||||
while(ptr<res.size() && res[ptr]!=str_begin)
|
||||
{
|
||||
token_str+=res[ptr];
|
||||
if(res[ptr]=='\n')
|
||||
++line;
|
||||
if(res[ptr]=='\\')
|
||||
if(res[ptr]=='\n') ++line;
|
||||
if(res[ptr]=='\\' && ptr+1<res.size())
|
||||
{
|
||||
++ptr;
|
||||
switch(res[ptr])
|
||||
@@ -321,8 +281,7 @@ std::string nasal_lexer::string_gen(std::vector<char>& res,int& ptr,int& line)
|
||||
}
|
||||
}
|
||||
++ptr;
|
||||
if(ptr>=res.size())
|
||||
break;
|
||||
if(ptr>=res.size()) break;
|
||||
}
|
||||
// check if this string ends with a " or '
|
||||
if(ptr>=res.size())
|
||||
@@ -377,13 +336,12 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
{
|
||||
while(ptr<res.size() && (res[ptr]==' ' || res[ptr]=='\n' || res[ptr]=='\t' || res[ptr]=='\r' || res[ptr]<0))
|
||||
{
|
||||
if(res[ptr]=='\n')
|
||||
++line;
|
||||
// these characters will be ignored, and '\n' will cause ++line
|
||||
if(res[ptr]=='\n') ++line;
|
||||
++ptr;
|
||||
}
|
||||
if(ptr>=res.size())
|
||||
break;
|
||||
if(res[ptr]=='_' || ('a'<=res[ptr] && res[ptr]<='z') || ('A'<=res[ptr] && res[ptr]<='Z'))
|
||||
if(ptr>=res.size()) break;
|
||||
if(IS_IDENTIFIER_HEAD(res[ptr]))
|
||||
{
|
||||
token_str=identifier_gen(res,ptr,line);
|
||||
token new_token;
|
||||
@@ -392,7 +350,7 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
new_token.str=token_str;
|
||||
token_list.push_back(new_token);
|
||||
}
|
||||
else if('0'<=res[ptr] && res[ptr]<='9')
|
||||
else if(IS_NUMBER_HEAD(res[ptr]))
|
||||
{
|
||||
token_str=number_gen(res,ptr,line);
|
||||
token new_token;
|
||||
@@ -401,7 +359,7 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
new_token.str=token_str;
|
||||
token_list.push_back(new_token);
|
||||
}
|
||||
else if(res[ptr]=='\'' || res[ptr]=='\"')
|
||||
else if(IS_STRING_HEAD(res[ptr]))
|
||||
{
|
||||
token_str=string_gen(res,ptr,line);
|
||||
token new_token;
|
||||
@@ -410,10 +368,7 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
new_token.str=token_str;
|
||||
token_list.push_back(new_token);
|
||||
}
|
||||
else if(res[ptr]=='(' || res[ptr]==')' || res[ptr]=='[' || res[ptr]==']' || res[ptr]=='{' ||
|
||||
res[ptr]=='}' || res[ptr]==',' || res[ptr]==';' || res[ptr]=='|' || res[ptr]==':' ||
|
||||
res[ptr]=='?' || res[ptr]=='.' || res[ptr]=='`' || res[ptr]=='&' || res[ptr]=='@' ||
|
||||
res[ptr]=='%' || res[ptr]=='$' || res[ptr]=='^' || res[ptr]=='\\')
|
||||
else if(IS_SINGLE_OPRATOR(res[ptr]))
|
||||
{
|
||||
token_str="";
|
||||
token_str+=res[ptr];
|
||||
@@ -424,8 +379,7 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
token_list.push_back(new_token);
|
||||
++ptr;
|
||||
}
|
||||
else if(res[ptr]=='=' || res[ptr]=='+' || res[ptr]=='-' || res[ptr]=='*' || res[ptr]=='!' ||
|
||||
res[ptr]=='/' || res[ptr]=='<' || res[ptr]=='>' || res[ptr]=='~')
|
||||
else if(IS_CALC_OPERATOR(res[ptr]))
|
||||
{
|
||||
// get calculation operator
|
||||
token_str="";
|
||||
@@ -442,7 +396,7 @@ void nasal_lexer::scanner(std::vector<char>& res)
|
||||
new_token.str=token_str;
|
||||
token_list.push_back(new_token);
|
||||
}
|
||||
else if(res[ptr]=='#')
|
||||
else if(IS_NOTE_HEAD(res[ptr]))
|
||||
{
|
||||
// avoid note
|
||||
while(ptr<res.size() && res[ptr]!='\n')
|
||||
@@ -484,21 +438,21 @@ void nasal_lexer::generate_detail_token()
|
||||
{
|
||||
detail_token.line=i->line;
|
||||
detail_token.str ="";
|
||||
if (i->str=="for") detail_token.type=__for;
|
||||
else if(i->str=="foreach") detail_token.type=__foreach;
|
||||
if (i->str=="for" ) detail_token.type=__for;
|
||||
else if(i->str=="foreach" ) detail_token.type=__foreach;
|
||||
else if(i->str=="forindex") detail_token.type=__forindex;
|
||||
else if(i->str=="while") detail_token.type=__while;
|
||||
else if(i->str=="var") detail_token.type=__var;
|
||||
else if(i->str=="func") detail_token.type=__func;
|
||||
else if(i->str=="break") detail_token.type=__break;
|
||||
else if(i->str=="while" ) detail_token.type=__while;
|
||||
else if(i->str=="var" ) detail_token.type=__var;
|
||||
else if(i->str=="func" ) detail_token.type=__func;
|
||||
else if(i->str=="break" ) detail_token.type=__break;
|
||||
else if(i->str=="continue") detail_token.type=__continue;
|
||||
else if(i->str=="return") detail_token.type=__return;
|
||||
else if(i->str=="if") detail_token.type=__if;
|
||||
else if(i->str=="else") detail_token.type=__else;
|
||||
else if(i->str=="elsif") detail_token.type=__elsif;
|
||||
else if(i->str=="nil") detail_token.type=__nil;
|
||||
else if(i->str=="and") detail_token.type=__and_operator;
|
||||
else if(i->str=="or") detail_token.type=__or_operator;
|
||||
else if(i->str=="return" ) detail_token.type=__return;
|
||||
else if(i->str=="if" ) detail_token.type=__if;
|
||||
else if(i->str=="else" ) detail_token.type=__else;
|
||||
else if(i->str=="elsif" ) detail_token.type=__elsif;
|
||||
else if(i->str=="nil" ) detail_token.type=__nil;
|
||||
else if(i->str=="and" ) detail_token.type=__and_operator;
|
||||
else if(i->str=="or" ) detail_token.type=__or_operator;
|
||||
detail_token_list.push_back(detail_token);
|
||||
}
|
||||
else if(i->type==__token_identifier)
|
||||
@@ -527,35 +481,35 @@ void nasal_lexer::generate_detail_token()
|
||||
{
|
||||
detail_token.line=i->line;
|
||||
detail_token.str ="";
|
||||
if (i->str=="+") detail_token.type=__add_operator;
|
||||
else if(i->str=="-") detail_token.type=__sub_operator;
|
||||
else if(i->str=="*") detail_token.type=__mul_operator;
|
||||
else if(i->str=="/") detail_token.type=__div_operator;
|
||||
else if(i->str=="~") detail_token.type=__link_operator;
|
||||
if (i->str=="+" ) detail_token.type=__add_operator;
|
||||
else if(i->str=="-" ) detail_token.type=__sub_operator;
|
||||
else if(i->str=="*" ) detail_token.type=__mul_operator;
|
||||
else if(i->str=="/" ) detail_token.type=__div_operator;
|
||||
else if(i->str=="~" ) detail_token.type=__link_operator;
|
||||
else if(i->str=="+=") detail_token.type=__add_equal;
|
||||
else if(i->str=="-=") detail_token.type=__sub_equal;
|
||||
else if(i->str=="*=") detail_token.type=__mul_equal;
|
||||
else if(i->str=="/=") detail_token.type=__div_equal;
|
||||
else if(i->str=="~=") detail_token.type=__link_equal;
|
||||
else if(i->str=="=") detail_token.type=__equal;
|
||||
else if(i->str=="=" ) detail_token.type=__equal;
|
||||
else if(i->str=="==") detail_token.type=__cmp_equal;
|
||||
else if(i->str=="!=") detail_token.type=__cmp_not_equal;
|
||||
else if(i->str=="<") detail_token.type=__cmp_less;
|
||||
else if(i->str=="<" ) detail_token.type=__cmp_less;
|
||||
else if(i->str=="<=") detail_token.type=__cmp_less_or_equal;
|
||||
else if(i->str==">") detail_token.type=__cmp_more;
|
||||
else if(i->str==">" ) detail_token.type=__cmp_more;
|
||||
else if(i->str==">=") detail_token.type=__cmp_more_or_equal;
|
||||
else if(i->str==";") detail_token.type=__semi;
|
||||
else if(i->str==".") detail_token.type=__dot;
|
||||
else if(i->str==":") detail_token.type=__colon;
|
||||
else if(i->str==",") detail_token.type=__comma;
|
||||
else if(i->str=="?") detail_token.type=__ques_mark;
|
||||
else if(i->str=="!") detail_token.type=__nor_operator;
|
||||
else if(i->str=="[") detail_token.type=__left_bracket;
|
||||
else if(i->str=="]") detail_token.type=__right_bracket;
|
||||
else if(i->str=="(") detail_token.type=__left_curve;
|
||||
else if(i->str==")") detail_token.type=__right_curve;
|
||||
else if(i->str=="{") detail_token.type=__left_brace;
|
||||
else if(i->str=="}") detail_token.type=__right_brace;
|
||||
else if(i->str==";" ) detail_token.type=__semi;
|
||||
else if(i->str=="." ) detail_token.type=__dot;
|
||||
else if(i->str==":" ) detail_token.type=__colon;
|
||||
else if(i->str=="," ) detail_token.type=__comma;
|
||||
else if(i->str=="?" ) detail_token.type=__ques_mark;
|
||||
else if(i->str=="!" ) detail_token.type=__nor_operator;
|
||||
else if(i->str=="[" ) detail_token.type=__left_bracket;
|
||||
else if(i->str=="]" ) detail_token.type=__right_bracket;
|
||||
else if(i->str=="(" ) detail_token.type=__left_curve;
|
||||
else if(i->str==")" ) detail_token.type=__right_curve;
|
||||
else if(i->str=="{" ) detail_token.type=__left_brace;
|
||||
else if(i->str=="}" ) detail_token.type=__right_brace;
|
||||
else
|
||||
{
|
||||
++error;
|
||||
|
||||
Reference in New Issue
Block a user