indention start

This commit is contained in:
bronku 2025-11-25 13:53:50 +01:00
parent 1c21aede4b
commit 5062a9b588
12 changed files with 268 additions and 48 deletions

View file

@ -11,7 +11,7 @@ $(TARGET): build/parser.o build/lexer.o build/main.o
build/parser.cpp build/parser.hpp: src/parser.y | build
$(BISON) -d -o build/parser.cpp src/parser.y
build/lexer.cpp build/lexer.hpp: src/lexer.l build/parser.hpp src/scanner.hpp | build
build/lexer.cpp build/lexer.hpp: src/lexer.l build/parser.hpp src/scanner.hpp src/indent_helper.hpp| build
$(FLEX) --header-file=build/lexer.hpp -o build/lexer.cpp src/lexer.l
build/%.o: build/%.cpp

65
src/indent_helper.hpp Normal file
View file

@ -0,0 +1,65 @@
#pragma once
#include <stack>
class IndentHelper
{
std::stack<int> indent_stack;
int pending_dedents = 0;
public:
IndentHelper() { indent_stack.push(0); }
int processLine(int spaces)
{
int current = indent_stack.top();
if (spaces > current)
{
indent_stack.push(spaces);
return 1;
}
if (spaces == current)
{
return 0;
}
pending_dedents = 0;
while (indent_stack.top() > spaces)
{
indent_stack.pop();
pending_dedents++;
}
if (indent_stack.top() != spaces)
{
throw std::runtime_error("Indentation error");
}
return -1;
}
bool hasPendingDedents() const
{
return pending_dedents > 0;
}
void consumePendingDedent()
{
if (pending_dedents > 0)
{
pending_dedents--;
}
}
int flushAllDedents()
{
int total_dedents = 0;
while (indent_stack.size() > 1)
{
indent_stack.pop();
total_dedents++;
}
return total_dedents;
}
};

View file

@ -5,28 +5,64 @@
%{
#include "scanner.hpp"
#include "indent_helper.hpp"
IndentHelper indent_helper;
bool at_line_start = true;
%}
%%
def { return yy::parser::token::TOK_KW_DEF;}
if { return yy::parser::token::TOK_KW_IF;}
return { return yy::parser::token::TOK_KW_RETURN;}
for { return yy::parser::token::TOK_KW_FOR;}
in { return yy::parser::token::TOK_KW_IF;}
\n { return yy::parser::token::TOK_NEWLINE; }
^[ \t]* {
if (!at_line_start) {
REJECT;
}
at_line_start = false;
int spaces = 0;
for (int i = 0; i < yyleng; i++)
spaces += (yytext[i] == '\t') ? 8 : 1;
int result = indent_helper.processLine(spaces);
if (result > 0) return yy::parser::token::TOK_INDENT;
if (result < 0) return yy::parser::token::TOK_DEDENT;
}
\n {
at_line_start = true;
return yy::parser::token::TOK_NEWLINE;
}
def { return yy::parser::token::TOK_KW_DEF; }
if { return yy::parser::token::TOK_KW_IF; }
return { return yy::parser::token::TOK_KW_RETURN; }
for { return yy::parser::token::TOK_KW_FOR; }
in { return yy::parser::token::TOK_KW_IN; }
\( { return yy::parser::token::TOK_L_PAREN; }
\) { return yy::parser::token::TOK_R_PAREN; }
\: { return yy::parser::token::TOK_COLON; }
\= { return yy::parser::token::TOK_ASSIGN; }
\+ { return yy::parser::token::TOK_PLUS; }
\- { return yy::parser::token::TOK_MINUS; }
\" { yymore(); BEGIN(STRING); }
[0-9]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_NUMBER; }
[_a-zA-Z]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_IDENTIFIER; }
<<EOF>> { return yy::parser::token::TOK_EOF; }
. { /* the last rule, ignore anything else (there should be nothing) */}
\" { yymore(); BEGIN(STRING); }
<STRING>\"\" { yymore(); }
<STRING>\" { BEGIN(INITIAL); yylval->as<std::string>() = yytext; return yy::parser::token::TOK_STRING; }
<STRING>. { yymore(); }
%%
[0-9]+ { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_NUMBER; }
[A-Za-z_][A-Za-z0-9_]* { yylval->as<std::string>() = yytext; return yy::parser::token::TOK_IDENTIFIER; }
<<EOF>> {
if (indent_helper.hasPendingDedents()) {
indent_helper.consumeDedent();
return yy::parser::token::TOK_DEDENT;
}
int d;
if ((d = indent_helper.flushDedent()) != 0) return yy::parser::token::TOK_DEDENT;
return yy::parser::token::TOK_EOF;
}
. { /* ignore other characters */ }

View file

@ -29,7 +29,7 @@
%token <int> NUMBER
%token ASSIGN PLUS MINUS L_PAREN R_PAREN COLON
%token KW_DEF KW_IF KW_RETURN KW_FOR KW_IN
%token EOF NEWLINE INDENT_UP INDENT_DOWN
%token EOF NEWLINE INDENT DEDENT
%%

View file

@ -4,28 +4,28 @@ L_PAREN
R_PAREN
COLON
NEWLINE
INDENT_UP
INDENT
KW_IF
IDENTIFIER condition
COLON
NEWLINE
INDENT_UP
INDENT
KW_RETURN
IDENTIFIER value
NEWLINE
INDENT_DOWN
DEDENT
KW_FOR
IDENTIFIER item
KW_IN
IDENTIFIER list
COLON
NEWLINE
INDENT_UP
INDENT
IDENTIFIER print
L_PAREN
IDENTIFIER item
R_PAREN
NEWLINE
INDENT_DOWN
INDENT_DOWN
DEDENT
DEDENT
EOF

View file

@ -1,3 +1,12 @@
x == y != z < > <= >=
a + b - c * d / e % f
( ) [ ] { } , : . ;
def outer():
def inner1():
if cond1:
action1()
if cond2:
action2()
def inner2():
for i in range(3):
if cond3:
action3()
action4()
return result

73
test/lexer/3.out Normal file
View file

@ -0,0 +1,73 @@
KW_DEF
IDENTIFIER outer
L_PAREN
R_PAREN
COLON
NEWLINE
INDENT
KW_DEF
IDENTIFIER inner1
L_PAREN
R_PAREN
COLON
NEWLINE
INDENT
KW_IF
IDENTIFIER cond1
COLON
NEWLINE
INDENT
IDENTIFIER action1
L_PAREN
R_PAREN
NEWLINE
DEDENT
KW_IF
IDENTIFIER cond2
COLON
NEWLINE
INDENT
IDENTIFIER action2
L_PAREN
R_PAREN
NEWLINE
DEDENT
DEDENT
KW_DEF
IDENTIFIER inner2
L_PAREN
R_PAREN
COLON
NEWLINE
INDENT
KW_FOR
IDENTIFIER i
KW_IN
IDENTIFIER range
L_PAREN
NUMBER 3
R_PAREN
COLON
NEWLINE
INDENT
KW_IF
IDENTIFIER cond3
COLON
NEWLINE
INDENT
IDENTIFIER action3
L_PAREN
R_PAREN
NEWLINE
DEDENT
IDENTIFIER action4
L_PAREN
R_PAREN
NEWLINE
DEDENT
DEDENT
KW_RETURN
IDENTIFIER result
NEWLINE
DEDENT
EOF

View file

@ -1,3 +1,3 @@
42 3.14 0xFF 0b1010
"hello" 'world' """multiline"""
'test' "escaped\"quote"
x == y != z < > <= >=
a + b - c * d / e % f
( ) [ ] { } , : . ;

34
test/lexer/4.out Normal file
View file

@ -0,0 +1,34 @@
IDENTIFIER x
EQ
IDENTIFIER y
NE
IDENTIFIER z
LT
GT
LTE
GTE
NEWLINE
IDENTIFIER a
PLUS
IDENTIFIER b
MINUS
IDENTIFIER c
MULTIPLY
IDENTIFIER d
DIVIDE
IDENTIFIER e
MODULO
IDENTIFIER f
NEWLINE
L_PAREN
R_PAREN
L_BRACKET
R_BRACKET
L_BRACE
R_BRACE
COMMA
COLON
DOT
SEMICOLON
NEWLINE
EOF

View file

@ -1,8 +1,3 @@
def calculate(a, b=10):
result = (a + b) * 2
if result > 100:
print("Large result:", result)
return result
# This is a comment
x = calculate(5, 15)
42 3.14 0xFF 0b1010
"hello" 'world' """multiline"""
'test' "escaped\"quote"

View file

@ -1,5 +1,8 @@
_private = "valid"
var123 = 456
CamelCase = True
snake_case = False
very_long_variable_name_123 = "test"
def calculate(a, b=10):
result = (a + b) * 2
if result > 100:
print("Large result:", result)
return result
# This is a comment
x = calculate(5, 15)

5
test/lexer/7.in Normal file
View file

@ -0,0 +1,5 @@
_private = "valid"
var123 = 456
CamelCase = True
snake_case = False
very_long_variable_name_123 = "test"