X-Git-Url: https://www.kengrimes.com/gitweb/?p=henge%2Fwebcc.git;a=blobdiff_plain;f=src%2Fapc%2Flexer.c;h=f11c6a21d1736743b9059aa0e9fa110c1e353654;hp=6ef1b4c5ac85600365e3fa56fc70f049c9f5de26;hb=0f505368fa8abbc2e9ab0296b9a5e6bd4869345f;hpb=03efe0c43855ff6af7eb2194f6d4a5f022bc415d diff --git a/src/apc/lexer.c b/src/apc/lexer.c index 6ef1b4c..f11c6a2 100644 --- a/src/apc/lexer.c +++ b/src/apc/lexer.c @@ -14,11 +14,22 @@ /* Standard */ #include #include +#include #include /* Posix */ #include +#include +#include +#include +#include +#include #include +#include //realpath, NAME_MAX, PATH_MAX #include +/* Redefinitions of NAME_MAX and PATH_MAX */ +//#define NAME_MAX NAME_MAX/4 +//#define PATH_MAX PATH_MAX/4 + /* Local */ #include "parser.tab.h" #ifndef DE_STACKSIZE @@ -27,45 +38,49 @@ #ifndef TK_STACKSIZE #define TK_STACKSIZE 1024 #endif +#ifndef MAX_SETNAME_LEN //max setname length +#define MAX_SETNAME_LEN 32 +#endif + /* Public */ -int lexer_init(void); -int lexer(void); -void lexer_pushtok(int, YYSTYPE); -extern //lexer_lex.rl -int lexer_lex(const char*); -struct dirent* lexer_direntpa[DE_STACKSIZE], **lexer_direntpp; +int lexer_init(void); +int lexer(void); +int lexer_lexfile(const uint8_t*); +void lexer_pushtok(int, YYSTYPE); +uint8_t const* lexer_get_current_filepath(void); +int lexer_lexfilename(uint8_t*); +struct dirent* lexer_direntpa[DE_STACKSIZE],** lexer_direntpp,** lexer_direntpb; /* Private */ +extern //lexer_fsm.rl +int lexer_lexstring(uint8_t*, int); extern //scanner.c -int scanner_init(void); +int scanner_init(void); extern //scanner.c -int scanner(void); +int scanner(void); static inline -int dredge_current_depth(void); +int dredge_current_depth(void); extern //bison -YYSTYPE yylval; +YYSTYPE yylval; +static +uint8_t const* current_filename; static struct tok { YYSTYPE lval; //token val int tok_t; //token type -} token_stack[TK_STACKSIZE]; -static -union tokp -{ int* tpt; //token pointer type - struct tok* tok; - YYSTYPE* tvp; //token value pointer -} tks, tkx; +} token_stack[TK_STACKSIZE], *tsp, *tsx; /* Directory Entity Array/Stack Simple array for keeping track of dirents yet to be processed by the scanner. If this list is empty and there are no tokens, the lexer is done. This array is populated by the scanner as an array, and popped locally by the - lexer as a stack. + lexer as a stack, and is popped as a FIFO stack. */ #define DE_STACK (lexer_direntpa) #define DE_STACKP (lexer_direntpp) -#define DE_LEN() (DE_STACKP - DE_STACK) -#define DE_INIT() (DE_STACKP = DE_STACK) -#define DE_POP() (*--DE_STACKP) +#define DE_STACKB (lexer_direntpb) +#define DE_LEN() (DE_STACKP - DE_STACKB) +#define DE_INIT() (DE_STACKP = DE_STACKB = DE_STACK) +#define DE_POP() (*DE_STACKB++) /* Token Stack This is a FIFO stack whose pointers are a union of either a pointer to an @@ -76,20 +91,16 @@ union tokp times in a sequence! */ #define TK_STACK (token_stack) -#define TK_STACKP (tks.tok) -#define TK_STACKPI (tks.tpt) -#define TK_STACKPL (tks.tvp) -#define TK_STACKX (tkx.tok) -#define TK_STACKXI (tkx.tpt) +#define TK_STACKP (tsp) +#define TK_STACKX (tsx) #define TK_LEN() (TK_STACKX - TK_STACKP) #define TK_INIT() (TK_STACKP = TK_STACKX = TK_STACK) #define TK_POP() (*TK_STACKP++) -#define TK_POPI() (*TK_STACKPI++); -#define TK_POPL() (*TK_STACKPL++); #define TK_PUSH(T,L) (*TK_STACKX++ = (struct tok){L,T}) /* Initializer - The initializer returns boolean true if an error occurs, which may be handled with standard errno. + The initializer returns boolean true if an error occurs, which may be handled + with standard errno. */ int lexer_init () @@ -115,10 +126,13 @@ int lexer goto done; \ } while (0) () -{start: - while (DE_LEN() > 0) //lex any directory entries in our stack - if (lexer_lex(DE_POP()->d_name) == 0) //fail if it generates no tokens - FAIL("Lexer failed to tokenize [%s]\n",(*DE_STACKP)->d_name); +{ struct tok token; + start: + while (DE_LEN() > 0)//lex any directory entries in our stack + { + if (lexer_lexfile(DE_POP()->d_name) == 0) + FAIL("Lexer failed to tokenize [%s]\n",(*DE_STACKB)->d_name); + } if (TK_EMPTY) //if there are no tokens, { TK_INIT(); //initialize the token stack back to 0 switch (scanner()) @@ -130,8 +144,9 @@ int lexer goto start; //start over and lex them } } - yylval = TK_POPL(); - return TK_POPI(); + token = TK_POP(); + yylval = token.lval; + return token.tok_t; done: yylval.val = 0; return 0; @@ -150,5 +165,140 @@ void lexer_pushtok exit(EXIT_FAILURE); } TK_PUSH(tok, lval); - printf("Pushed Token %i | %i\n", TK_STACK[TK_LEN() - 1].tok_t, TK_STACK[TK_LEN() - 1].lval.val); } + +/* Lexical analysis of a file + Strips a filename to its base name, then sends it to lexer_lex +*/ +int lexer_lexfile +#define HIDDEN_WARNING "%s is hidden and will not be parsed!\n", filename +( const uint8_t *filename +) +{ static uint8_t fname[NAME_MAX]; + uint8_t *last_period = NULL, *iter; + + if (*filename == '.') + { fprintf (stderr, HIDDEN_WARNING); + return 0; + } + /* Copy the filename and remove its suffix */ + u8_strncpy(fname,filename,NAME_MAX); + last_period = NULL; + for (iter = fname; *iter; iter++) //find the last '.' char + if (*iter == '.') + last_period = iter; + if (last_period) //if we found one, + *last_period = 0; //truncate the string there + /* Register the current_filename */ + current_filename = filename; + + return lexer_lexfilename(fname); +} + +uint8_t const* lexer_get_current_filepath +() +{ static uint8_t current_path[PATH_MAX]; + static uint8_t const* last_filename; + if ((!last_filename || last_filename != current_filename) && + (realpath(current_filename, current_path) != (char*) current_path)) + { perror("realpath: "); + return NULL; + } + return (const char*)current_path; +} + +/* Scan filename and push the its tokens + onto the stack */ +int lexer_lexfilename +(uint8_t* str) +{ + int ntok, i, cmp, len, set_len, height, width; + char map_key[] = "_m_"; + static uint8_t set_name[MAX_SETNAME_LEN] = {0}; + uint8_t *first, *map_begin; + + printf("Starting lexerfilename on %s\n", str); + + + if(*str == 0) + printf("Lexfilename:: str is NULL so fail\n"); + printf("setname is %s\n", set_name); + + /* If last file was a mapfile, then its 5th to last token should + be a MOPEN. If this is the case, then we only pass MOPEN, height, + weight and name of the current file. */ + if( (TK_STACKX - 5)->tok_t == MOPEN ) + { printf("The last file was a mapfile\n"); + if( (map_begin = strstr(map_key, str)) ) //if the current file is a mapfile + { printf("The current file is a variant of the last mapfile\n"); + printf("Start lexing mapfile %s\n", str); + ntok += lexer_lexstring(map_begin, strlen(map_begin)); + } + printf("Current file is not a variant of the last mapfile\n"); + } + else //last file was not a mapfile + { printf("Last file was not a mapfile\n"); + + first = (uint8_t*) u8_strchr(str, '_'); //find the first '_' to find where str set_name ends + + if(set_name[0] != 0) //if there is a set_name from last str + { printf("There is a set_name (%s) present\n", set_name); + set_len = first - str; + + if(u8_strncmp(str, set_name, set_len) == 0) //check if it matches the current set_name + { str = str + set_len + 1; //if so, remove it from str + printf("str set_name matched last set_name, set str to %s\n", str); + } + else //update set_name to be str set_name + { u8_cpy(set_name, str, set_len); + set_name[set_len] = 0; + + } + } + else //set set_name + { u8_cpy(set_name, str, first-str); + } + /* Call lexer_lexstring to tokenize the string */ + printf("calling lexstring to tokenize str (%s) of len %d\n", str, u8_strlen(str)); + ntok += lexer_lexstring(str, u8_strlen(str)); + } + + /*TODO: if regfile, store full path for later */ + + printf("Ending lexer_lex on %s, %d tokens were lexed\n", str, ntok); + return ntok; +} + +/* int lexer_lexmapfile */ +/* #define INC_X() */ +/* (int height, int width) */ +/* { */ +/* int x, y; */ + +/* /\* Give scanner_scanpixels a buffer and a len. Iterate through */ +/* buf with buf[n]. If n == 0, do nothing. if n has a value, push x, */ +/* push y, push (z = n << 24), push (ref_id = n >> 8) *\/ */ +/* //scanner_scanpixels() */ + +/* for(i = 0; i < len; i++) */ +/* if(buf[i] == 0) */ +/* if(x == width) */ +/* x = 0; */ +/* else */ + + + + +/* } */ +/* fname_bytes = (uint8_t*)(DE_POP()->d_name); */ + /* printf("d_name is %s\n", fname_bytes); */ + /* for (fnp = filename, i = 0; i < NAME_MAX; i += unit_size, fnp++) */ + /* { unit_size = u8_mblen(fname_bytes + i, min(4, NAME_MAX - i)); */ + /* if (u8_mbtouc(fnp, fname_bytes + i, unit_size) == -1) //add ucs4 char to the filename */ + /* FAIL("Lexer failed to convert ^%s to unicode\n", (fname_bytes + i)); */ + /* if (*fnp == 0) //added a terminating char */ + /* break; */ + /* } */ + /* if(u8_mbtouc(filename, DE_POP()->d_name, NAME_MAX) == -1) */ + /* FAIL("Lexer failed to convert d_name into uint8_t\n"); */ + /* ulc_fprintf(stdout, "filename is %11U\n c", filename); */