indention start
This commit is contained in:
parent
1c21aede4b
commit
5062a9b588
12 changed files with 268 additions and 48 deletions
2
makefile
2
makefile
|
|
@ -11,7 +11,7 @@ $(TARGET): build/parser.o build/lexer.o build/main.o
|
|||
build/parser.cpp build/parser.hpp: src/parser.y | build
|
||||
$(BISON) -d -o build/parser.cpp src/parser.y
|
||||
|
||||
build/lexer.cpp build/lexer.hpp: src/lexer.l build/parser.hpp src/scanner.hpp | build
|
||||
build/lexer.cpp build/lexer.hpp: src/lexer.l build/parser.hpp src/scanner.hpp src/indent_helper.hpp| build
|
||||
$(FLEX) --header-file=build/lexer.hpp -o build/lexer.cpp src/lexer.l
|
||||
|
||||
build/%.o: build/%.cpp
|
||||
|
|
|
|||
65
src/indent_helper.hpp
Normal file
65
src/indent_helper.hpp
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
#pragma once
|
||||
#include <stack>
|
||||
|
||||
class IndentHelper
|
||||
{
|
||||
std::stack<int> indent_stack;
|
||||
int pending_dedents = 0;
|
||||
|
||||
public:
|
||||
IndentHelper() { indent_stack.push(0); }
|
||||
|
||||
int processLine(int spaces)
|
||||
{
|
||||
int current = indent_stack.top();
|
||||
|
||||
if (spaces > current)
|
||||
{
|
||||
indent_stack.push(spaces);
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (spaces == current)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
pending_dedents = 0;
|
||||
while (indent_stack.top() > spaces)
|
||||
{
|
||||
indent_stack.pop();
|
||||
pending_dedents++;
|
||||
}
|
||||
|
||||
if (indent_stack.top() != spaces)
|
||||
{
|
||||
throw std::runtime_error("Indentation error");
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
bool hasPendingDedents() const
|
||||
{
|
||||
return pending_dedents > 0;
|
||||
}
|
||||
|
||||
void consumePendingDedent()
|
||||
{
|
||||
if (pending_dedents > 0)
|
||||
{
|
||||
pending_dedents--;
|
||||
}
|
||||
}
|
||||
|
||||
int flushAllDedents()
|
||||
{
|
||||
int total_dedents = 0;
|
||||
while (indent_stack.size() > 1)
|
||||
{
|
||||
indent_stack.pop();
|
||||
total_dedents++;
|
||||
}
|
||||
return total_dedents;
|
||||
}
|
||||
};
|
||||
52
src/lexer.l
52
src/lexer.l
|
|
@ -5,28 +5,64 @@
|
|||
|
||||
%{
|
||||
#include "scanner.hpp"
|
||||
#include "indent_helper.hpp"
|
||||
|
||||
IndentHelper indent_helper;
|
||||
bool at_line_start = true;
|
||||
%}
|
||||
|
||||
%%
|
||||
|
||||
^[ \t]* {
|
||||
if (!at_line_start) {
|
||||
REJECT;
|
||||
}
|
||||
at_line_start = false;
|
||||
|
||||
int spaces = 0;
|
||||
for (int i = 0; i < yyleng; i++)
|
||||
spaces += (yytext[i] == '\t') ? 8 : 1;
|
||||
|
||||
int result = indent_helper.processLine(spaces);
|
||||
if (result > 0) return yy::parser::token::TOK_INDENT;
|
||||
if (result < 0) return yy::parser::token::TOK_DEDENT;
|
||||
}
|
||||
|
||||
\n {
|
||||
at_line_start = true;
|
||||
return yy::parser::token::TOK_NEWLINE;
|
||||
}
|
||||
|
||||
def { return yy::parser::token::TOK_KW_DEF; }
|
||||
if { return yy::parser::token::TOK_KW_IF; }
|
||||
return { return yy::parser::token::TOK_KW_RETURN; }
|
||||
for { return yy::parser::token::TOK_KW_FOR; }
|
||||
in { return yy::parser::token::TOK_KW_IF;}
|
||||
\n { return yy::parser::token::TOK_NEWLINE; }
|
||||
in { return yy::parser::token::TOK_KW_IN; }
|
||||
|
||||
\( { return yy::parser::token::TOK_L_PAREN; }
|
||||
\) { return yy::parser::token::TOK_R_PAREN; }
|
||||
\: { return yy::parser::token::TOK_COLON; }
|
||||
\= { return yy::parser::token::TOK_ASSIGN; }
|
||||
\+ { return yy::parser::token::TOK_PLUS; }
|
||||
\- { return yy::parser::token::TOK_MINUS; }
|
||||
\" { yymore(); BEGIN(STRING); }
|
||||
[0-9]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_NUMBER; }
|
||||
[_a-zA-Z]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_IDENTIFIER; }
|
||||
<<EOF>> { return yy::parser::token::TOK_EOF; }
|
||||
. { /* the last rule, ignore anything else (there should be nothing) */}
|
||||
|
||||
\" { yymore(); BEGIN(STRING); }
|
||||
<STRING>\"\" { yymore(); }
|
||||
<STRING>\" { BEGIN(INITIAL); yylval->as<std::string>() = yytext; return yy::parser::token::TOK_STRING; }
|
||||
<STRING>. { yymore(); }
|
||||
%%
|
||||
|
||||
[0-9]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_NUMBER; }
|
||||
[A-Za-z_][A-Za-z0-9_]* { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_IDENTIFIER; }
|
||||
|
||||
|
||||
<<EOF>> {
|
||||
if (indent_helper.hasPendingDedents()) {
|
||||
indent_helper.consumeDedent();
|
||||
return yy::parser::token::TOK_DEDENT;
|
||||
}
|
||||
int d;
|
||||
if ((d = indent_helper.flushDedent()) != 0) return yy::parser::token::TOK_DEDENT;
|
||||
return yy::parser::token::TOK_EOF;
|
||||
}
|
||||
|
||||
. { /* ignore other characters */ }
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@
|
|||
%token <int> NUMBER
|
||||
%token ASSIGN PLUS MINUS L_PAREN R_PAREN COLON
|
||||
%token KW_DEF KW_IF KW_RETURN KW_FOR KW_IN
|
||||
%token EOF NEWLINE INDENT_UP INDENT_DOWN
|
||||
%token EOF NEWLINE INDENT DEDENT
|
||||
|
||||
|
||||
%%
|
||||
|
|
|
|||
|
|
@ -4,28 +4,28 @@ L_PAREN
|
|||
R_PAREN
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT_UP
|
||||
INDENT
|
||||
KW_IF
|
||||
IDENTIFIER condition
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT_UP
|
||||
INDENT
|
||||
KW_RETURN
|
||||
IDENTIFIER value
|
||||
NEWLINE
|
||||
INDENT_DOWN
|
||||
DEDENT
|
||||
KW_FOR
|
||||
IDENTIFIER item
|
||||
KW_IN
|
||||
IDENTIFIER list
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT_UP
|
||||
INDENT
|
||||
IDENTIFIER print
|
||||
L_PAREN
|
||||
IDENTIFIER item
|
||||
R_PAREN
|
||||
NEWLINE
|
||||
INDENT_DOWN
|
||||
INDENT_DOWN
|
||||
DEDENT
|
||||
DEDENT
|
||||
EOF
|
||||
|
|
@ -1,3 +1,12 @@
|
|||
x == y != z < > <= >=
|
||||
a + b - c * d / e % f
|
||||
( ) [ ] { } , : . ;
|
||||
def outer():
|
||||
def inner1():
|
||||
if cond1:
|
||||
action1()
|
||||
if cond2:
|
||||
action2()
|
||||
def inner2():
|
||||
for i in range(3):
|
||||
if cond3:
|
||||
action3()
|
||||
action4()
|
||||
return result
|
||||
|
|
|
|||
73
test/lexer/3.out
Normal file
73
test/lexer/3.out
Normal file
|
|
@ -0,0 +1,73 @@
|
|||
KW_DEF
|
||||
IDENTIFIER outer
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
KW_DEF
|
||||
IDENTIFIER inner1
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
KW_IF
|
||||
IDENTIFIER cond1
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
IDENTIFIER action1
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
NEWLINE
|
||||
DEDENT
|
||||
KW_IF
|
||||
IDENTIFIER cond2
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
IDENTIFIER action2
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
NEWLINE
|
||||
DEDENT
|
||||
DEDENT
|
||||
KW_DEF
|
||||
IDENTIFIER inner2
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
KW_FOR
|
||||
IDENTIFIER i
|
||||
KW_IN
|
||||
IDENTIFIER range
|
||||
L_PAREN
|
||||
NUMBER 3
|
||||
R_PAREN
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
KW_IF
|
||||
IDENTIFIER cond3
|
||||
COLON
|
||||
NEWLINE
|
||||
INDENT
|
||||
IDENTIFIER action3
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
NEWLINE
|
||||
DEDENT
|
||||
IDENTIFIER action4
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
NEWLINE
|
||||
DEDENT
|
||||
DEDENT
|
||||
KW_RETURN
|
||||
IDENTIFIER result
|
||||
NEWLINE
|
||||
DEDENT
|
||||
EOF
|
||||
|
|
@ -1,3 +1,3 @@
|
|||
42 3.14 0xFF 0b1010
|
||||
"hello" 'world' """multiline"""
|
||||
'test' "escaped\"quote"
|
||||
x == y != z < > <= >=
|
||||
a + b - c * d / e % f
|
||||
( ) [ ] { } , : . ;
|
||||
|
|
|
|||
34
test/lexer/4.out
Normal file
34
test/lexer/4.out
Normal file
|
|
@ -0,0 +1,34 @@
|
|||
IDENTIFIER x
|
||||
EQ
|
||||
IDENTIFIER y
|
||||
NE
|
||||
IDENTIFIER z
|
||||
LT
|
||||
GT
|
||||
LTE
|
||||
GTE
|
||||
NEWLINE
|
||||
IDENTIFIER a
|
||||
PLUS
|
||||
IDENTIFIER b
|
||||
MINUS
|
||||
IDENTIFIER c
|
||||
MULTIPLY
|
||||
IDENTIFIER d
|
||||
DIVIDE
|
||||
IDENTIFIER e
|
||||
MODULO
|
||||
IDENTIFIER f
|
||||
NEWLINE
|
||||
L_PAREN
|
||||
R_PAREN
|
||||
L_BRACKET
|
||||
R_BRACKET
|
||||
L_BRACE
|
||||
R_BRACE
|
||||
COMMA
|
||||
COLON
|
||||
DOT
|
||||
SEMICOLON
|
||||
NEWLINE
|
||||
EOF
|
||||
|
|
@ -1,8 +1,3 @@
|
|||
def calculate(a, b=10):
|
||||
result = (a + b) * 2
|
||||
if result > 100:
|
||||
print("Large result:", result)
|
||||
return result
|
||||
|
||||
# This is a comment
|
||||
x = calculate(5, 15)
|
||||
42 3.14 0xFF 0b1010
|
||||
"hello" 'world' """multiline"""
|
||||
'test' "escaped\"quote"
|
||||
|
|
|
|||
|
|
@ -1,5 +1,8 @@
|
|||
_private = "valid"
|
||||
var123 = 456
|
||||
CamelCase = True
|
||||
snake_case = False
|
||||
very_long_variable_name_123 = "test"
|
||||
def calculate(a, b=10):
|
||||
result = (a + b) * 2
|
||||
if result > 100:
|
||||
print("Large result:", result)
|
||||
return result
|
||||
|
||||
# This is a comment
|
||||
x = calculate(5, 15)
|
||||
|
|
|
|||
5
test/lexer/7.in
Normal file
5
test/lexer/7.in
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
_private = "valid"
|
||||
var123 = 456
|
||||
CamelCase = True
|
||||
snake_case = False
|
||||
very_long_variable_name_123 = "test"
|
||||
Loading…
Add table
Add a link
Reference in a new issue