@ -84,6 +84,9 @@ do_scrub_begin ()
# endif
}
/* Note: if any other character can be LEX_IS_STRINGQUOTE, the loop
in state 5 of do_scrub_chars must be changed . */
/* Note that these override the previous defaults, e.g. if ';' is a
comment char , then it isn ' t a line separator . */
for ( p = symbol_chars ; * p ; + + p )
@ -129,43 +132,14 @@ do_scrub_begin ()
}
} /* do_scrub_begin() */
FILE * scrub_file ;
int
scrub_from_file ( )
{
return getc ( scrub_file ) ;
}
void
scrub_to_file ( ch )
int ch ;
{
ungetc ( ch , scrub_file ) ;
} /* scrub_to_file() */
char * scrub_string ;
char * scrub_last_string ;
int
scrub_from_string ( )
{
return scrub_string = = scrub_last_string ? EOF : * scrub_string + + ;
} /* scrub_from_string() */
void
scrub_to_string ( ch )
int ch ;
{
* - - scrub_string = ch ;
} /* scrub_to_string() */
/* Saved state of the scrubber */
static int state ;
static int old_state ;
static char * out_string ;
static char out_buf [ 20 ] ;
static int add_newlines = 0 ;
static int add_newlines ;
static char * saved_input ;
static int saved_input_len ;
/* Data structure for saving the state of app across #include's. Note that
app is called asynchronously to the parsing of the . include ' s , so our
@ -179,9 +153,8 @@ struct app_save
char * out_string ;
char out_buf [ sizeof ( out_buf ) ] ;
int add_newlines ;
char * scrub_string ;
char * scrub_last_string ;
FILE * scrub_file ;
char * saved_input ;
int saved_input_len ;
} ;
char *
@ -195,11 +168,14 @@ app_push ()
saved - > out_string = out_string ;
memcpy ( saved - > out_buf , out_buf , sizeof ( out_buf ) ) ;
saved - > add_newlines = add_newlines ;
saved - > scrub_string = scrub_string ;
saved - > scrub_last_string = scrub_last_string ;
saved - > scrub_file = scrub_file ;
saved - > saved_input = saved_input ;
saved - > saved_input_len = saved_input_len ;
/* do_scrub_begin() is not useful, just wastes time. */
state = 0 ;
saved_input = NULL ;
return ( char * ) saved ;
}
@ -215,9 +191,8 @@ app_pop (arg)
out_string = saved - > out_string ;
memcpy ( out_buf , saved - > out_buf , sizeof ( out_buf ) ) ;
add_newlines = saved - > add_newlines ;
scrub_string = saved - > scrub_string ;
scrub_last_string = saved - > scrub_last_string ;
scrub_file = saved - > scrub_file ;
saved_input = saved - > saved_input ;
saved_input_len = saved - > saved_input_len ;
free ( arg ) ;
} /* app_pop() */
@ -248,11 +223,32 @@ process_escape (ch)
return ch ;
}
}
/* This function is called to process input characters. The GET
parameter is used to retrieve more input characters . GET should
set its parameter to point to a buffer , and return the length of
the buffer ; it should return 0 at end of file . The scrubbed output
characters are put into the buffer starting at TOSTART ; the TOSTART
buffer is TOLEN bytes in length . The function returns the number
of scrubbed characters put into TOSTART . This will be TOLEN unless
end of file was seen . This function is arranged as a state
machine , and saves its state so that it may return at any point .
This is the way the old code used to work . */
int
do_scrub_next_char ( get , unget )
int ( * get ) ( ) ;
void ( * unget ) ( ) ;
{
do_scrub_chars ( get , tostart , tolen )
int ( * get ) PARAMS ( ( char * * ) ) ;
char * tostart ;
int tolen ;
{
char * to = tostart ;
char * toend = tostart + tolen ;
char * from ;
char * fromend ;
int fromlen ;
register int ch , ch2 = 0 ;
int not_cpp_line = 0 ;
/*State 0: beginning of normal line
1 : After first whitespace on line ( flush more white )
2 : After first non - white ( opcode ) on line ( keep 1 white )
@ -279,27 +275,73 @@ do_scrub_next_char (get, unget)
correctly on the PA ( and any other target where colons are optional ) .
Jeff Law , law @ cs . utah . edu . */
/* This is purely an optimization hack, and relies on gcc's inlining
capability . */
# if defined (__GNUC__) && defined (__OPTIMIZE__)
# define GET() (get == scrub_from_file ? scrub_from_file () : (*get) ())
# else
# define GET() ((*get) ())
# endif
register int ch , ch2 = 0 ;
int not_cpp_line = 0 ;
/* This macro gets the next input character. */
# define GET() \
( from < fromend \
? * from + + \
: ( ( saved_input ! = NULL \
? ( free ( saved_input ) , \
saved_input = NULL , \
0 ) \
: 0 ) , \
fromlen = ( * get ) ( & from ) , \
fromend = from + fromlen , \
( fromlen = = 0 \
? EOF \
: * from + + ) ) )
/* This macro pushes a character back on the input stream. */
# define UNGET(uch) (*--from = (uch))
/* This macro puts a character into the output buffer. If this
character fills the output buffer , this macro jumps to the label
TOFULL . We use this rather ugly approach because we need to
handle two different termination conditions : EOF on the input
stream , and a full output buffer . It would be simpler if we
always read in the entire input stream before processing it , but
I don ' t want to make such a significant change to the assembler ' s
memory usage . */
# define PUT(pch) \
do \
{ \
* to + + = ( pch ) ; \
if ( to > = toend ) \
goto tofull ; \
} \
while ( 0 )
if ( saved_input ! = NULL )
{
from = saved_input ;
fromend = from + saved_input_len ;
}
else
{
fromlen = ( * get ) ( & from ) ;
if ( fromlen = = 0 )
return 0 ;
fromend = from + fromlen ;
}
while ( 1 )
{
/* The cases in this switch end with continue, in order to
branch back to the top of this while loop and generate the
next output character in the appropriate state . */
switch ( state )
{
case - 1 :
ch = * out_string + + ;
if ( * out_string = = 0 )
if ( * out_string = = ' \0 ' )
{
state = old_state ;
old_state = 3 ;
}
return ch ;
PUT ( ch ) ;
continue ;
case - 2 :
for ( ; ; )
@ -307,73 +349,130 @@ do_scrub_next_char (get, unget)
do
{
ch = GET ( ) ;
if ( ch = = EOF )
{
as_warn ( " end of file in comment " ) ;
goto fromeof ;
}
while ( ch ! = EOF & & ch ! = ' \n ' & & ch ! = ' * ' ) ;
if ( ch = = ' \n ' | | ch = = EOF )
return ch ;
/* At this point, ch must be a '*' */
if ( ch = = ' \n ' )
PUT ( ' \n ' ) ;
}
while ( ch ! = ' * ' ) ;
while ( ( ch = GET ( ) ) = = ' * ' )
{
;
if ( ch = = EOF )
{
as_warn ( " end of file in comment " ) ;
goto fromeof ;
}
if ( ch = = EOF | | ch = = ' / ' )
if ( ch = = ' / ' )
break ;
( * unget ) ( ch ) ;
UNGET ( ch ) ;
}
state = old_state ;
return ' ' ;
PUT ( ' ' ) ;
continue ;
case 4 :
ch = GET ( ) ;
if ( ch = = EOF | | ( ch > = ' 0 ' & & ch < = ' 9 ' ) )
return ch ;
if ( ch = = EOF )
goto fromeof ;
else if ( ch > = ' 0 ' & & ch < = ' 9 ' )
PUT ( ch ) ;
else
{
while ( ch ! = EOF & & IS_WHITESPACE ( ch ) )
ch = GET ( ) ;
if ( ch = = ' " ' )
{
( * unget ) ( ch ) ;
UNGET ( ch ) ;
out_string = " \n \t .appfile " ;
old_state = 7 ;
state = - 1 ;
return * out_string + + ;
PUT ( * out_string + + ) ;
}
else
{
while ( ch ! = EOF & & ch ! = ' \n ' )
ch = GET ( ) ;
state = 0 ;
return ch ;
PUT ( ch ) ;
}
}
continue ;
case 5 :
/* We are going to copy everything up to a quote character,
with special handling for a backslash . We try to
optimize the copying in the simple case without using the
GET and PUT macros . */
{
char * s ;
int len ;
for ( s = from ; s < fromend ; s + + )
{
ch = * s ;
/* This condition must be changed if the type of any
other character can be LEX_IS_STRINGQUOTE . */
if ( ch = = ' \\ '
| | ch = = ' " '
| | ch = = ' \' '
| | ch = = ' \n ' )
break ;
}
len = s - from ;
if ( len > toend - to )
len = toend - to ;
if ( len > 0 )
{
memcpy ( to , from , len ) ;
to + = len ;
from + = len ;
}
}
ch = GET ( ) ;
if ( lex [ ch ] = = LEX_IS_STRINGQUOTE )
if ( ch = = EOF )
{
as_warn ( " end of file in string: inserted ' \" ' " ) ;
state = old_state ;
return ch ;
UNGET ( ' \n ' ) ;
PUT ( ' " ' ) ;
}
else if ( lex [ ch ] = = LEX_IS_STRINGQUOTE )
{
state = old_state ;
PUT ( ch ) ;
}
# ifndef NO_STRING_ESCAPES
else if ( ch = = ' \\ ' )
{
state = 6 ;
return ch ;
PUT ( ch ) ;
}
# endif
else if ( ch = = EOF )
else if ( flag_mri & & ch = = ' \n ' )
{
as_warn ( " End of file in string: inserted ' \" ' " ) ;
/* Just quietly terminate the string. This permits lines like
bne label loop if we haven ' t reach end yet
*/
state = old_state ;
( * unget ) ( ' \n ' ) ;
return ' " ' ;
UNGET ( ch ) ;
PUT ( ' \' ' ) ;
}
else
{
return ch ;
PUT ( ch ) ;
}
continue ;
case 6 :
state = 5 ;
@ -383,9 +482,10 @@ do_scrub_next_char (get, unget)
/* Handle strings broken across lines, by turning '\n' into
' \\ ' and ' n ' . */
case ' \n ' :
( * unget ) ( ' n ' ) ;
UNGET ( ' n ' ) ;
add_newlines + + ;
return ' \\ ' ;
PUT ( ' \\ ' ) ;
continue ;
case ' " ' :
case ' \\ ' :
@ -418,22 +518,30 @@ do_scrub_next_char (get, unget)
case EOF :
as_warn ( " End of file in string: ' \" ' inserted " ) ;
return ' " ' ;
PUT ( ' " ' ) ;
continue ;
}
return ch ;
PUT ( ch ) ;
continue ;
case 7 :
ch = GET ( ) ;
state = 5 ;
old_state = 8 ;
return ch ;
if ( ch = = EOF )
goto fromeof ;
PUT ( ch ) ;
continue ;
case 8 :
do
ch = GET ( ) ;
while ( ch ! = ' \n ' ) ;
while ( ch ! = ' \n ' & & ch ! = EOF ) ;
if ( ch = = EOF )
goto fromeof ;
state = 0 ;
return ch ;
PUT ( ch ) ;
continue ;
}
/* OK, we are somewhere in states 0 through 4 or 9 through 11 */
@ -445,52 +553,55 @@ recycle:
{
if ( state ! = 0 )
{
as_warn ( " End of file not at end of a line: Newline inserted. " ) ;
as_warn ( " end of file not at end of a line; newline inserted " ) ;
state = 0 ;
return ' \n ' ;
PUT ( ' \n ' ) ;
}
return ch ;
goto fromeof ;
}
switch ( lex [ ch ] )
{
case LEX_IS_WHITESPACE :
do
/* Preserve a single whitespace character at the beginning of
a line . */
if ( state = = 0 )
{
/* Preserve a single whitespace character at the
beginning of a line . */
state = 1 ;
return ch ;
PUT ( ch ) ;
break ;
}
else
do
{
ch = GET ( ) ;
}
while ( ch ! = EOF & & IS_WHITESPACE ( ch ) ) ;
if ( ch = = EOF )
return ch ;
goto fromeof ;
if ( IS_COMMENT ( ch )
| | ( state = = 0 & & IS_LINE_COMMENT ( ch ) )
| | ch = = ' / '
| | IS_LINE_SEPARATOR ( ch ) )
{
/* cpp never outputs a leading space before the #, so try t o
avoid being confused . */
/* cpp never outputs a leading space before the #, so
try to avoid being confused . */
not_cpp_line = 1 ;
goto recycle ;
}
/* If we're in state 2 or 11, we've seen a non-white character
followed by whitespace . If the next character is ' : ' , this
is whitespace after a label name which we normally must
ignore . In MRI mode , though , spaces are not permitted
between the label and the colon . */
/* If we're in state 2 or 11, we've seen a non-white
character followed by whitespace . If the next character
is ' : ' , this is whitespace after a label name which we
normally must ignore . In MRI mode , though , spaces are
not permitted between the label and the colon . */
if ( ( state = = 2 | | state = = 11 )
& & lex [ ch ] = = LEX_IS_COLON
& & ! flag_mri )
{
state = 1 ;
return ch ;
PUT ( ch ) ;
break ;
}
switch ( state )
@ -499,23 +610,46 @@ recycle:
state + + ;
goto recycle ; /* Punted leading sp */
case 1 :
/* We can arrive here if we leave a leading whitespace character
at the beginning of a line . */
/* We can arrive here if we leave a leading whitespace
character at the beginning of a line . */
goto recycle ;
case 2 :
state = 3 ;
( * unget ) ( ch ) ;
return ' ' ; /* Sp after opco */
if ( to + 1 < toend )
{
/* Optimize common case by skipping UNGET/GET. */
PUT ( ' ' ) ; /* Sp after opco */
goto recycle ;
}
UNGET ( ch ) ;
PUT ( ' ' ) ;
break ;
case 3 :
if ( flag_mri )
{
/* In MRI mode, we keep these spaces. */
UNGET ( ch ) ;
PUT ( ' ' ) ;
break ;
}
goto recycle ; /* Sp in operands */
case 9 :
case 10 :
if ( flag_mri )
{
/* In MRI mode, we keep these spaces. */
state = 3 ;
UNGET ( ch ) ;
PUT ( ' ' ) ;
break ;
}
state = 10 ; /* Sp after symbol char */
goto recycle ;
case 11 :
state = 1 ;
( * unget ) ( ch ) ;
return ' ' ; /* Sp after label definition. */
UNGET ( ch ) ;
PUT ( ' ' ) ; /* Sp after label definition. */
break ;
default :
BAD_CASE ( state ) ;
}
@ -545,10 +679,10 @@ recycle:
if ( ch2 = = EOF
| | lex [ ch2 ] = = LEX_IS_TWOCHAR_COMMENT_1ST )
break ;
( * unget ) ( ch ) ;
UNGET ( ch ) ;
}
if ( ch2 = = EOF )
as_warn ( " E nd of file in multiline comment" ) ;
as_warn ( " e nd of file in multiline comment" ) ;
ch = ' ' ;
goto recycle ;
@ -556,10 +690,10 @@ recycle:
else
{
if ( ch2 ! = EOF )
( * unget ) ( ch2 ) ;
UNGET ( ch2 ) ;
if ( state = = 9 | | state = = 10 )
state = 3 ;
return ch ;
PUT ( ch ) ;
}
break ;
@ -567,51 +701,66 @@ recycle:
if ( state = = 10 )
{
/* Preserve the whitespace in foo "bar" */
( * unget ) ( ch ) ;
UNGET ( ch ) ;
state = 3 ;
return ' ' ;
PUT ( ' ' ) ;
/* PUT didn't jump out. We could just break, but we
know what will happen , so optimize a bit . */
ch = GET ( ) ;
old_state = 3 ;
}
else if ( state = = 9 )
old_state = 3 ;
else
old_state = state ;
state = 5 ;
return ch ;
PUT ( ch ) ;
break ;
# ifndef IEEE_STYLE
case LEX_IS_ONECHAR_QUOTE :
if ( state = = 10 )
{
/* Preserve the whitespace in foo 'b' */
( * unget ) ( ch ) ;
UNGET ( ch ) ;
state = 3 ;
return ' ' ;
PUT ( ' ' ) ;
break ;
}
ch = GET ( ) ;
if ( ch = = EOF )
{
as_warn ( " End-of- file after a one-character quote; \\ 00 0 inserted" ) ;
as_warn ( " end of file after a one-character quote; \\ 0 inserted " ) ;
ch = 0 ;
}
if ( ch = = ' \\ ' )
{
ch = GET ( ) ;
if ( ch = = EOF )
{
as_warn ( " end of file in escape character " ) ;
ch = ' \\ ' ;
}
else
ch = process_escape ( ch ) ;
}
sprintf ( out_buf , " %d " , ( int ) ( unsigned char ) ch ) ;
/* None of these 'x constants for us. We want 'x'. */
if ( ( ch = GET ( ) ) ! = ' \' ' )
{
# ifdef REQUIRE_CHAR_CLOSE_QUOTE
as_warn ( " Missing close quote: (assumed) " ) ;
# else
( * unget ) ( ch ) ;
if ( ch ! = EOF )
UNGET ( ch ) ;
# endif
}
if ( strlen ( out_buf ) = = 1 )
{
return out_buf [ 0 ] ;
PUT ( out_buf [ 0 ] ) ;
break ;
}
if ( state = = 9 )
old_state = 3 ;
@ -619,46 +768,51 @@ recycle:
old_state = state ;
state = - 1 ;
out_string = out_buf ;
return * out_string + + ;
PUT ( * out_string + + ) ;
break ;
# endif
case LEX_IS_COLON :
if ( state = = 9 | | state = = 10 )
state = 3 ;
else if ( state ! = 3 )
state = 1 ;
return ch ;
PUT ( ch ) ;
break ;
case LEX_IS_NEWLINE :
/* Roll out a bunch of newlines from inside comments, etc. */
if ( add_newlines )
{
- - add_newlines ;
( * unget ) ( ch ) ;
UNGET ( ch ) ;
}
/* fall thru into... */
case LEX_IS_LINE_SEPARATOR :
state = 0 ;
return ch ;
PUT ( ch ) ;
break ;
case LEX_IS_LINE_COMMENT_START :
if ( state = = 0 ) /* Only comment at start of line. */
{
/* FIXME-someday: The two character comment stuff was badly
thought out . On i386 , we want ' / ' as line comment start
AND we want C style comments . hence this hack . The
whole lexical process should be reworked . xoxorich . */
/* FIXME-someday: The two character comment stuff was
badly thought out . On i386 , we want ' / ' as line
comment start AND we want C style comments . hence
this hack . The whole lexical process should be
reworked . xoxorich . */
if ( ch = = ' / ' )
{
ch2 = GET ( ) ;
if ( ch2 = = ' * ' )
{
state = - 2 ;
return ( do_scrub_next_char ( get , unget ) ) ;
break ;
}
else
{
( * unget ) ( ch2 ) ;
UNGET ( ch2 ) ;
}
} /* bad hack */
@ -666,12 +820,15 @@ recycle:
not_cpp_line = 1 ;
do
{
ch = GET ( ) ;
}
while ( ch ! = EOF & & IS_WHITESPACE ( ch ) ) ;
if ( ch = = EOF )
{
as_warn ( " EOF in comment: Newline inserted " ) ;
return ' \n ' ;
as_warn ( " end of file in comment; newline inserted " ) ;
PUT ( ' \n ' ) ;
break ;
}
if ( ch < ' 0 ' | | ch > ' 9 ' | | not_cpp_line )
{
@ -681,44 +838,112 @@ recycle:
if ( ch = = EOF )
as_warn ( " EOF in Comment: Newline inserted " ) ;
state = 0 ;
return ' \n ' ;
PUT ( ' \n ' ) ;
break ;
}
/* Numerics begin comment. Perhaps CPP `# 123 "filename"' */
( * unget ) ( ch ) ;
UNGET ( ch ) ;
old_state = 4 ;
state = - 1 ;
out_string = " \t .appline " ;
return * out_string + + ;
PUT ( * out_string + + ) ;
break ;
}
/* We have a line comment character which is not at the start of
a line . If this is also a normal comment character , fall
through . Otherwise treat it as a default character . */
if ( ( flag_mri & & ( ch = = ' ! ' | | ch = = ' * ' ) )
| | strchr ( comment_chars , ch ) = = NULL )
/* We have a line comment character which is not at the
start of a line . If this is also a normal comment
character , fall through . Otherwise treat it as a default
character . */
if ( strchr ( comment_chars , ch ) = = NULL
& & ( ! flag_mri
| | ( ch ! = ' ! ' & & ch ! = ' * ' ) ) )
goto de_fault ;
if ( flag_mri
& & ( ch = = ' ! ' | | ch = = ' * ' )
& & state ! = 1
& & state ! = 10 )
goto de_fault ;
/* Fall through. */
case LEX_IS_COMMENT_START :
do
{
ch = GET ( ) ;
}
while ( ch ! = EOF & & ! IS_NEWLINE ( ch ) ) ;
if ( ch = = EOF )
as_warn ( " EOF in comment: N ewline inserted" ) ;
as_warn ( " end of file in comment; n ewline inserted" ) ;
state = 0 ;
return ' \n ' ;
PUT ( ' \n ' ) ;
break ;
case LEX_IS_SYMBOL_COMPONENT :
if ( state = = 10 )
{
/* This is a symbol character following another symbol
character , with whitespace in between . We skipped the
whitespace earlier , so output it now . */
( * unget ) ( ch ) ;
character , with whitespace in between . We skipped
the whitespace earlier , so output it now . */
UNGET ( ch ) ;
state = 3 ;
return ' ' ;
PUT ( ' ' ) ;
break ;
}
if ( state = = 3 )
state = 9 ;
/* This is a common case. Quickly copy CH and all the
following symbol component or normal characters . */
if ( to + 1 < toend )
{
char * s ;
int len ;
for ( s = from ; s < fromend ; s + + )
{
int type ;
ch2 = * s ;
type = lex [ ch2 ] ;
if ( type ! = 0
& & type ! = LEX_IS_SYMBOL_COMPONENT )
break ;
}
if ( s > from )
{
/* Handle the last character normally, for
simplicity . */
- - s ;
}
len = s - from ;
if ( len > ( toend - to ) - 1 )
len = ( toend - to ) - 1 ;
if ( len > 0 )
{
PUT ( ch ) ;
if ( len > 8 )
{
memcpy ( to , from , len ) ;
to + = len ;
from + = len ;
}
else
{
switch ( len )
{
case 8 : * to + + = * from + + ;
case 7 : * to + + = * from + + ;
case 6 : * to + + = * from + + ;
case 5 : * to + + = * from + + ;
case 4 : * to + + = * from + + ;
case 3 : * to + + = * from + + ;
case 2 : * to + + = * from + + ;
case 1 : * to + + = * from + + ;
}
}
ch = GET ( ) ;
}
}
/* Fall through. */
default :
de_fault :
@ -726,55 +951,54 @@ recycle:
if ( state = = 0 )
{
state = 11 ; /* Now seeing label definition */
return ch ;
}
else if ( state = = 1 )
{
state = 2 ; /* Ditto */
return ch ;
}
else if ( state = = 9 )
{
if ( lex [ ch ] ! = LEX_IS_SYMBOL_COMPONENT )
state = 3 ;
return ch ;
}
else if ( state = = 10 )
{
state = 3 ;
return ch ;
}
else
{
return ch ; /* Opcode or operands already */
}
PUT ( ch ) ;
break ;
}
return - 1 ;
# undef GET
}
# ifdef TEST
/*NOTREACHED*/
const char comment_chars [ ] = " | " ;
const char line_comment_chars [ ] = " # " ;
fromeof :
/* We have reached the end of the input. */
return to - tostart ;
main ( )
tofull :
/* The output buffer is full. Save any input we have not yet
processed . */
if ( fromend > from )
{
int ch ;
char * save ;
app_begin ( ) ;
while ( ( ch = do_scrub_next_char ( stdin ) ) ! = EOF )
putc ( ch , stdout ) ;
save = ( char * ) xmalloc ( fromend - from ) ;
memcpy ( save , from , fromend - from ) ;
if ( saved_input ! = NULL )
free ( saved_input ) ;
saved_input = save ;
saved_input_len = fromend - from ;
}
as_warn ( str )
char * str ;
else
{
fputs ( str , stderr ) ;
putc ( ' \n ' , stderr ) ;
if ( saved_input ! = NULL )
{
free ( saved_input ) ;
saved_input = NULL ;
}
}
return to - tostart ;
}
# endif
/* end of app.c */