Compare commits

..

4 Commits

218 changed files with 2105 additions and 11796 deletions

View File

@ -17,7 +17,7 @@ jobs:
steps: steps:
- name: Install dependencies (Linux) - name: Install dependencies (Linux)
if: runner.os == 'Linux' if: runner.os == 'Linux'
run: sudo apt-get update && sudo apt-get install -y gcc gdc ldc valgrind run: sudo apt-get update && sudo apt-get install -y gcc gdc ldc
- name: Install dependencies (macOS) - name: Install dependencies (macOS)
if: runner.os == 'macOS' if: runner.os == 'macOS'
@ -31,9 +31,6 @@ jobs:
with: with:
ruby-version: ${{ matrix.ruby-version }} ruby-version: ${{ matrix.ruby-version }}
- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
- name: Install dependencies - name: Install dependencies
run: bundle install run: bundle install

View File

@ -1,169 +1,3 @@
## v5.1.0
### New Features
- Add a `node_id()` accessor to the C++ and D tree node handle types, for node
identity comparison. This matches the existing `p_node_id()` macro (C) and
`node_id()` method (Rust).
## v5.0.0
### New Features
- Add Rust target language output.
- Add Rust language detection in propane.vim.
### API Changes
- The matched text argument passed to lexer user code blocks is now named
`match_text` instead of `match`, since `match` is a keyword in Rust. The
`match_length` argument (C and C++) is unchanged.
- Tree generation mode now stores all tree nodes in a compact arena owned by
the parser context (a flat node array plus a shared child-link array).
This replaces the previous design of one heap allocation per node with
layout-punned typed structs.
- Tree nodes are now referenced by lightweight handles rather than pointers.
`p_result()` and the field accessors now return handle values in tree
generation mode.
- The whole tree is freed together with the context by `p_context_delete()`.
The `p_tree_delete()` / `p_tree_delete_XXX()` functions have been removed;
tree node handles are only valid while the context is alive.
- Tree node field access changed per target language:
- C: per-field accessor functions (e.g. `p_Start_pItems(node)`) plus tree
walk macros (e.g. `p_tree_walk_Start(node, pItems, pItem, pToken1, token)`),
and generic accessors `p_node_valid()`, `p_node_position()`,
`p_node_end_position()`, `p_node_n_fields()`, `p_node_data()`, `p_node_id()`.
- C++: handle methods called with `()` (e.g. `node.pItems().pToken1().token()`),
plus the same C-style functions/macros for convenience.
- D: `@property` accessors preserving the previous field-access syntax
(e.g. `node.pItems.pToken1.token`); null checks use `.valid` instead of
`is null`.
- Tree-mode parser rule user code: `$$` and `$1` etc. now yield node handles.
Reference child fields through the target-language accessors described above
rather than through struct pointer members.
### Improvements
- Improve D language detection in propane.vim
- Speed up specs
## v4.8.1
### Fixes
- Fix tree node struct type forward-declarations for C/C++
## v4.8.0
### New Features
- Add `p_parse_inner_XXX()` APIs that accept a caller-provided set of follow
tokens. These behave the same as `p_parse_XXX()` by parsing starting at the
given start rule, but instead of expecting the rest of the input to match
the start rule they allow specifying a set of tokens that may follow the
start rule.
- Add `p_set_position()` API to set the current text position stored in the
context. Useful for setting the initial text position to something other
than `(1, 1)` for a nested parse operation.
- Add `p_input_index()` API to get the current input text byte offset.
- Add `p_set_input_index()` API to set the current input text byte offset.
Useful together with `p_set_position()` to rewind the input part-way through
a parse in order to re-read an earlier section of the input.
## v4.7.0
### New Features
- Support parser rule user code blocks in tree generation mode.
### Fixes
- propane.vim: do not highlight rule components as propane keywords
## v4.6.0
### New Features
- Add lexer user code API to access matched input text positions
- Track rule component text positions and add parser user code API to access
### Fixes
- Fixed a few user guide and source comments related to text input positions
## v4.5.0
### New Features
- Add `noline` grammar statement to skip emitting `#line` directives
- Attempt to autodetect target language (D/C++) in extra/vim/syntax/propane.vim
### Fixes
- Fix #line reset directives
- Update keyword list in extra/vim/syntax/propane.vim
- Fix propane.vim keyword detection
## v4.4.0
### New Features
- Add p_value_get() / p_value_get_XXX() accessors
## v4.3.0
### New Features
- Use #line for user code blocks to report input grammar position for errors.
## v4.2.0
### New Features
- Add support for a custom lex function.
## v4.1.0
### New Features
- Add `p_context_delete()` and `p_tree_delete()` for D targets.
## v4.0.0
### New Features
- Add `context_user_fields` statement to allow custom context user fields.
- Add `token_user_fields` statement to allow custom token user fields.
- Add `on_token_node` statement to allow custom code when constructing token nodes.
- Add `free_token_node` statement to allow custom code when freeing token nodes.
- Add `p_context_delete()`.
- Allow `drop` patterns to execute lexer user code blocks.
### Breaking Changes
- Replace `p_context_init()` with `p_context_new()` and `p_context_delete()`.
- Renamed `p_free_tree()` to `p_tree_delete()`.
- The `free_token_node` statement now takes a user code block instead of a
function name parameter.
## v3.0.0
### New Features
- Add support for multiple starting rules (#38)
- Add `p_free_tree()` functions to reclaim generated tree memory
- Add `free_token_node` grammar statement to reclaim user-allocated memory stored in a Token tree node `pvalue` field
- Add valgrind memory leak tests to unit tests
- Fix build issues for C++ to officially support C++ target output
### Improvements
- Document `p_lex()` and `p_token_info_t` in user guide (#37)
### Breaking Changes
- Rename AST generation mode to tree generation mode (see [UPGRADING.md](UPGRADING.md))
## v2.3.0 ## v2.3.0
### New Features ### New Features

View File

@ -5,12 +5,12 @@ GEM
date (3.5.1) date (3.5.1)
diff-lcs (1.6.2) diff-lcs (1.6.2)
docile (1.4.1) docile (1.4.1)
erb (6.0.4) erb (6.0.1)
psych (5.4.0) psych (5.3.1)
date date
stringio stringio
rake (13.4.2) rake (13.3.1)
rdoc (7.2.0) rdoc (7.1.0)
erb erb
psych (>= 4.0.0) psych (>= 4.0.0)
tsort tsort
@ -24,10 +24,10 @@ GEM
rspec-expectations (3.13.5) rspec-expectations (3.13.5)
diff-lcs (>= 1.2.0, < 2.0) diff-lcs (>= 1.2.0, < 2.0)
rspec-support (~> 3.13.0) rspec-support (~> 3.13.0)
rspec-mocks (3.13.8) rspec-mocks (3.13.7)
diff-lcs (>= 1.2.0, < 2.0) diff-lcs (>= 1.2.0, < 2.0)
rspec-support (~> 3.13.0) rspec-support (~> 3.13.0)
rspec-support (3.13.7) rspec-support (3.13.6)
simplecov (0.22.0) simplecov (0.22.0)
docile (~> 1.1) docile (~> 1.1)
simplecov-html (~> 0.11) simplecov-html (~> 0.11)
@ -51,4 +51,4 @@ DEPENDENCIES
syntax syntax
BUNDLED WITH BUNDLED WITH
4.0.14 2.3.7

View File

@ -1,6 +1,6 @@
The MIT License (MIT) The MIT License (MIT)
Copyright (c) 2010-2026 Josh Holtrop Copyright (c) 2010-2024 Josh Holtrop
Permission is hereby granted, free of charge, to any person obtaining a copy Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal of this software and associated documentation files (the "Software"), to deal

View File

@ -6,10 +6,8 @@ Propane is a LALR Parser Generator (LPG) which:
* generates a built-in lexer to tokenize input * generates a built-in lexer to tokenize input
* supports UTF-8 lexer inputs * supports UTF-8 lexer inputs
* generates a table-driven shift/reduce parser to parse input in linear time * generates a table-driven shift/reduce parser to parse input in linear time
* targets C, C++, D, or Rust language outputs * targets C, C++, or D language outputs
* optionally supports automatic full parse tree generation * optionally supports automatic full AST generation
* supports starting parsing from multiple start rules
* tracks input text start and end positions for all matched tokens/rules
* is MIT-licensed * is MIT-licensed
* is distributable as a standalone Ruby script * is distributable as a standalone Ruby script
@ -69,7 +67,7 @@ token times /\*/;
token power /\*\*/; token power /\*\*/;
token integer /\d+/ << token integer /\d+/ <<
ulong v; ulong v;
foreach (c; match_text) foreach (c; match)
{ {
v *= 10; v *= 10;
v += (c - '0'); v += (c - '0');

View File

@ -1,4 +1,3 @@
require "fileutils"
require "rake/clean" require "rake/clean"
require "rspec/core/rake_task" require "rspec/core/rake_task"
require "simplecov" require "simplecov"
@ -12,10 +11,7 @@ end
RSpec::Core::RakeTask.new(:spec, :example_pattern) do |task, args| RSpec::Core::RakeTask.new(:spec, :example_pattern) do |task, args|
if args.example_pattern if args.example_pattern
ENV["partial_specs"] = "1"
task.rspec_opts = %W[-e "#{args.example_pattern}" -f documentation] task.rspec_opts = %W[-e "#{args.example_pattern}" -f documentation]
else
FileUtils.rm_rf("coverage")
end end
end end
task :spec do |task, args| task :spec do |task, args|
@ -23,7 +19,7 @@ task :spec do |task, args|
original_stdout = $stdout original_stdout = $stdout
sio = StringIO.new sio = StringIO.new
$stdout = sio $stdout = sio
SimpleCov.collate Dir["coverage/parts/*/.resultset.json"] SimpleCov.collate Dir["coverage/.resultset.json"]
$stdout = original_stdout $stdout = original_stdout
sio.string.lines.each do |line| sio.string.lines.each do |line|
$stdout.write(line) unless line =~ /Coverage report generated for/ $stdout.write(line) unless line =~ /Coverage report generated for/
@ -31,15 +27,6 @@ task :spec do |task, args|
end end
end end
task :valgrind do
begin
ENV["spec-valgrind"] = "1"
Rake::Task[:spec].execute
ensure
ENV.delete("spec-valgrind")
end
end
# dspec task is useful to test the distributable release script, but is not # dspec task is useful to test the distributable release script, but is not
# useful for coverage information. # useful for coverage information.
desc "Dist Specs" desc "Dist Specs"
@ -56,4 +43,4 @@ task :user_guide do
system("ruby", "-Ilib", "rb/gen_user_guide.rb") system("ruby", "-Ilib", "rb/gen_user_guide.rb")
end end
task :all => [:valgrind, :dspec, :user_guide] task :all => [:spec, :dspec, :user_guide]

View File

@ -1,110 +0,0 @@
## v5.0.0
The generated API for tree generation mode (`tree;`) has been changed
significantly for this version.
Aside from the lexer user code block matched text rename described below, the
lexer/parser value APIs for non-tree grammars are unchanged.
### Lexer user code block matched text
The matched text argument passed to lexer user code blocks has been renamed
from `match` to `match_text` for all target languages.
- C, C++, and D: rename references to `match` in lexer user code blocks to
`match_text` (for example `$$ = match[0];` becomes `$$ = match_text[0];`).
The `match_length` argument (C, C++) is unchanged.
### Tree memory management
- Remove all calls to `p_tree_delete()` / `p_tree_delete_XXX()`. Tree nodes now
live in the parser context and are freed by `p_context_delete()`.
- Tree node handles (returned by `p_result()` and the field accessors) are only
valid while the context is alive. Do not use them after `p_context_delete()`.
### Tree node field access
Tree nodes are now referenced by handle values instead of pointers, and field
access differs per target language:
- C: replace `node->field` with the accessor function `p_TYPE_field(node)`, or
use the tree walk macro `p_tree_walk_TYPE(node, field1, field2, ...)`. Replace
`x != NULL` / `x == NULL` node checks with `p_node_valid(x)` /
`!p_node_valid(x)`. Read positions with `p_node_position(node)` /
`p_node_end_position(node)`, token payload with `p_TYPE_token(node)` /
`p_TYPE_pvalue(node)` or `p_node_data(node)->field`, and compare node identity
with `p_node_id(a) == p_node_id(b)`.
- C++: replace `node->field` with the handle method `node.field()`. Use
`node.valid()`, `node.position()`, `node.token()`, `node.pvalue()`, and
`node.data()->field` for user token fields. (The C-style functions and macros
above are also available.)
- D: replace pointer declarations (`Start * s`) with value handles (`Start s`)
and replace `x !is null` / `x is null` with `x.valid` / `!x.valid`. Field
access syntax (`node.field.field`) is otherwise unchanged.
### Tree-mode parser rule user code
In tree generation mode `$$` and `$1`, `$2`, ... now expand to node handles.
Reference child fields through the target-language accessors above (for example
`$$->pA->pToken1->pvalue` becomes `p_tree_walk_Start($$, pA, pToken1, pvalue)`
in C, `$$.pA().pToken1().pvalue()` in C++, and `$$.pA.pToken1.pvalue` in D).
### Pointers into tree node storage
Tree nodes previously each had their own allocation, so a pointer to a node
stayed valid for the life of the tree. They are now held in a single array
which is reallocated as it grows, so a pointer or reference into that array may
be invalidated whenever a new node is created.
New nodes are created while parsing, so this matters for a pointer taken in a
tree-mode parser rule user code block, which runs before the parse has
finished. Keep the node handle instead, which stores a node ID rather than an
address and stays valid, and obtain the pointer from it when it is needed.
For example, replace a saved pointer:
```
context_user_fields <<
p_node_data_t * saved;
>>
Items -> Items a << ${context.saved} = p_node_data($$); >>
```
with a saved handle:
```
context_user_fields <<
Items saved_node;
>>
Items -> Items a << ${context.saved_node} = $$; >>
```
```
p_node_data_t * data = p_node_data(context->saved_node);
```
Once parsing has finished, no further nodes are created, so a pointer obtained
after `p_parse()` returns stays valid until the context is deleted, as long as
no further parsing is performed with the same context.
## v4.0.0
### API Changes
- Replace any calls to `p_context_init()` with `p_context_new()`.
- Replace any references to the address of a statically allocated context
structure with the pointer returned from `p_context_init()` (e.g. `&context`
-> `context`).
- Add a call to `p_context_delete()` (for C or C++) after lexing/parsing to
reclaim context memory.
- Rename `p_free_tree()` calls to `p_tree_delete()`.
- Change `free_token_node` statement calls from taking a function name argument
to taking a user code block.
## v3.0.0
### Grammar Changes
- Rename `ast;` statement to `tree;`.
- Rename `ast_prefix;` statement to `tree_prefix;`.
- Rename `ast_suffix;` statement to `tree_suffix;`.

View File

@ -43,82 +43,30 @@ const char * <%= @grammar.prefix %>token_names[] = {
*************************************************************************/ *************************************************************************/
/** /**
* Allocate and initialize lexer/parser context structure. * Initialize lexer/parser context structure.
*
* Deinitialize and deallocate with <%= @grammar.prefix %>context_delete().
* *
* @param[out] context
* Lexer/parser context structure.
* @param input * @param input
* Text input. * Text input.
* @param input_length * @param input_length
* Text input length. * Text input length.
*
* @return Context structure for lexer/parser.
*/ */
<%= @grammar.prefix %>context_t * <%= @grammar.prefix %>context_new(uint8_t const * input, size_t input_length) void <%= @grammar.prefix %>context_init(<%= @grammar.prefix %>context_t * context, uint8_t const * input, size_t input_length)
{ {
<% if @cpp %> /* New default-initialized context structure. */
<%= @grammar.prefix %>context_t * context = new <%= @grammar.prefix %>context_t(); <%= @grammar.prefix %>context_t newcontext;
<% else %> memset(&newcontext, 0, sizeof(newcontext));
<%= @grammar.prefix %>context_t * context = (<%= @grammar.prefix %>context_t *)calloc(1, sizeof(<%= @grammar.prefix %>context_t));
<% end %>
/* Lexer initialization. */ /* Lexer initialization. */
context->input = input; newcontext.input = input;
context->input_length = input_length; newcontext.input_length = input_length;
context->text_position.row = 1u; newcontext.text_position.row = 1u;
context->text_position.col = 1u; newcontext.text_position.col = 1u;
context->mode = <%= @lexer.mode_id("default") %>; newcontext.mode = <%= @lexer.mode_id("default") %>;
<% if @grammar.tree %>
/* Reserve node ID 0 as the null tree node. */ /* Copy to the user's context structure. */
<% if @cpp %> *context = newcontext;
context-><%= @grammar.prefix %>tree_nodes.resize(1);
<% else %>
context-><%= @grammar.prefix %>tree_nodes_capacity = 16u;
context-><%= @grammar.prefix %>tree_nodes = (<%= @grammar.prefix %>node_data_t *)malloc(16u * sizeof(<%= @grammar.prefix %>node_data_t));
memset(&context-><%= @grammar.prefix %>tree_nodes[0], 0, sizeof(<%= @grammar.prefix %>node_data_t));
context-><%= @grammar.prefix %>tree_nodes_length = 1u;
<% end %>
<% end %>
return context;
}
/**
* Deinitialize and deallocate lexer/parser context structure.
*
* For C++, destructors will be called for any context user fields. However, if
* pointers are used to store allocated resources, the user should free them
* before calling this function.
*
* @param context
* Lexer/parser context structure allocated with <%= @grammar.prefix %>context_new().
*/
void <%= @grammar.prefix %>context_delete(<%= @grammar.prefix %>context_t * context)
{
<% if @grammar.tree && @grammar.free_token_node != "" %>
<% if @cpp %>
for (size_t i = 0u; i < context-><%= @grammar.prefix %>tree_nodes.size(); i++)
<% else %>
for (size_t i = 0u; i < context-><%= @grammar.prefix %>tree_nodes_length; i++)
<% end %>
{
if (context-><%= @grammar.prefix %>tree_nodes[i].is_token)
{
<%= @grammar.prefix %>node_data_t * token_tree_node = &context-><%= @grammar.prefix %>tree_nodes[i];
<%= expand_code(@grammar.free_token_node, false, nil, nil) %>
}
}
<% end %>
<% if @cpp %>
delete context;
<% else %>
<% if @grammar.tree %>
free(context-><%= @grammar.prefix %>tree_nodes);
free(context-><%= @grammar.prefix %>tree_children);
<% end %>
free(context);
<% end %>
} }
/************************************************************************** /**************************************************************************
@ -319,7 +267,7 @@ static lexer_mode_t lexer_mode_table[] = {
* Lexer/parser context structure. * Lexer/parser context structure.
* @param code_id * @param code_id
* The ID of the user code block to execute. * The ID of the user code block to execute.
* @param match_text * @param match
* Matched text for this pattern. * Matched text for this pattern.
* @param match_length * @param match_length
* Matched text length. * Matched text length.
@ -330,7 +278,7 @@ static lexer_mode_t lexer_mode_table[] = {
* not explicitly return a token. * not explicitly return a token.
*/ */
static <%= @grammar.prefix %>token_t lexer_user_code(<%= @grammar.prefix %>context_t * context, static <%= @grammar.prefix %>token_t lexer_user_code(<%= @grammar.prefix %>context_t * context,
lexer_user_code_id_t code_id, uint8_t const * match_text, lexer_user_code_id_t code_id, uint8_t const * match,
size_t match_length, <%= @grammar.prefix %>token_info_t * out_token_info) size_t match_length, <%= @grammar.prefix %>token_info_t * out_token_info)
{ {
switch (code_id) switch (code_id)
@ -516,27 +464,11 @@ static size_t attempt_lex_token(<%= @grammar.prefix %>context_t * context, <%= @
case P_SUCCESS: case P_SUCCESS:
{ {
<%= @grammar.prefix %>token_t token_to_accept = match_info.accepting_state->token; <%= @grammar.prefix %>token_t token_to_accept = match_info.accepting_state->token;
/* Calculate the token length and start/end positions before invoking
* the lexer user code so that the user code can access them. The
* context input text position tracking is not updated until after the
* user code has run so that it is left unchanged if the user code
* requests to terminate the lexer. */
token_info.length = match_info.length;
if (match_info.end_delta_position.row != 0u)
{
token_info.end_position.row = token_info.position.row + match_info.end_delta_position.row;
token_info.end_position.col = match_info.end_delta_position.col;
}
else
{
token_info.end_position.row = token_info.position.row;
token_info.end_position.col = token_info.position.col + match_info.end_delta_position.col;
}
if (match_info.accepting_state->code_id != INVALID_USER_CODE_ID) if (match_info.accepting_state->code_id != INVALID_USER_CODE_ID)
{ {
uint8_t const * match_text = &context->input[context->input_index]; uint8_t const * match = &context->input[context->input_index];
<%= @grammar.prefix %>token_t user_code_token = lexer_user_code(context, <%= @grammar.prefix %>token_t user_code_token = lexer_user_code(context,
match_info.accepting_state->code_id, match_text, match_info.length, &token_info); match_info.accepting_state->code_id, match, match_info.length, &token_info);
/* A TERMINATE_TOKEN_ID return code from lexer_user_code() means /* A TERMINATE_TOKEN_ID return code from lexer_user_code() means
* that the user code is requesting to terminate the lexer. */ * that the user code is requesting to terminate the lexer. */
if (user_code_token == TERMINATE_TOKEN_ID) if (user_code_token == TERMINATE_TOKEN_ID)
@ -570,6 +502,17 @@ static size_t attempt_lex_token(<%= @grammar.prefix %>context_t * context, <%= @
return P_DROP; return P_DROP;
} }
token_info.token = token_to_accept; token_info.token = token_to_accept;
token_info.length = match_info.length;
if (match_info.end_delta_position.row != 0u)
{
token_info.end_position.row = token_info.position.row + match_info.end_delta_position.row;
token_info.end_position.col = match_info.end_delta_position.col;
}
else
{
token_info.end_position.row = token_info.position.row;
token_info.end_position.col = token_info.position.col + match_info.end_delta_position.col;
}
*out_token_info = token_info; *out_token_info = token_info;
} }
return P_SUCCESS; return P_SUCCESS;
@ -695,7 +638,7 @@ typedef struct
* reduce action. * reduce action.
*/ */
parser_state_id_t n_states; parser_state_id_t n_states;
<% if @grammar.tree %> <% if @grammar.ast %>
/** /**
* Map of rule components to rule set child fields. * Map of rule components to rule set child fields.
@ -703,7 +646,7 @@ typedef struct
uint16_t const * rule_set_node_field_index_map; uint16_t const * rule_set_node_field_index_map;
/** /**
* Number of rule set tree node fields. * Number of rule set AST node fields.
*/ */
uint16_t rule_set_node_field_array_size; uint16_t rule_set_node_field_array_size;
@ -742,17 +685,25 @@ typedef struct
/** Parser state ID. */ /** Parser state ID. */
size_t state_id; size_t state_id;
<% if @grammar.tree %>
/** Tree node ID. */
<%= @grammar.prefix %>node_id_t node_id;
<% else %>
<%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position;
/** Parser value from this state. */ /** Parser value from this state. */
<%= @grammar.prefix %>value_t pvalue; <%= @grammar.prefix %>value_t pvalue;
<% if @grammar.ast %>
/** AST node. */
void * ast_node;
<% end %> <% end %>
} state_value_t; } state_value_t;
/** Common AST node structure. */
typedef struct ASTNode_s
{
<%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position;
uint16_t n_fields;
uint8_t is_token;
struct ASTNode_s * fields[];
} ASTNode;
/** Parser shift table. */ /** Parser shift table. */
static const shift_t parser_shift_table[] = { static const shift_t parser_shift_table[] = {
<% @parser.shift_table.each do |shift| %> <% @parser.shift_table.each do |shift| %>
@ -760,7 +711,7 @@ static const shift_t parser_shift_table[] = {
<% end %> <% end %>
}; };
<% if @grammar.tree %> <% if @grammar.ast %>
<% @grammar.rules.each do |rule| %> <% @grammar.rules.each do |rule| %>
<% unless rule.flat_rule_set_node_field_index_map? %> <% unless rule.flat_rule_set_node_field_index_map? %>
const uint16_t r_<%= rule.name.gsub("$", "_") %><%= rule.id %>_node_field_index_map[<%= rule.rule_set_node_field_index_map.size %>] = {<%= rule.rule_set_node_field_index_map.map {|v| v.to_s}.join(", ") %>}; const uint16_t r_<%= rule.name.gsub("$", "_") %><%= rule.id %>_node_field_index_map[<%= rule.rule_set_node_field_index_map.size %>] = {<%= rule.rule_set_node_field_index_map.map {|v| v.to_s}.join(", ") %>};
@ -775,14 +726,14 @@ static const reduce_t parser_reduce_table[] = {
<%= reduce[:token_id] %>u, /* Token: <%= reduce[:token] ? reduce[:token].name : "(any)" %> */ <%= reduce[:token_id] %>u, /* Token: <%= reduce[:token] ? reduce[:token].name : "(any)" %> */
<%= reduce[:rule_id] %>u, /* Rule ID */ <%= reduce[:rule_id] %>u, /* Rule ID */
<%= reduce[:rule_set_id] %>u, /* Rule set ID (<%= reduce[:rule].rule_set.name %>) */ <%= reduce[:rule_set_id] %>u, /* Rule set ID (<%= reduce[:rule].rule_set.name %>) */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= reduce[:n_states] %>u, /* Number of states */ <%= reduce[:n_states] %>u, /* Number of states */
<% if reduce[:rule].flat_rule_set_node_field_index_map? %> <% if reduce[:rule].flat_rule_set_node_field_index_map? %>
NULL, /* No rule set node field index map (flat map) */ NULL, /* No rule set node field index map (flat map) */
<% else %> <% else %>
&r_<%= reduce[:rule].name.gsub("$", "_") %><%= reduce[:rule].id %>_node_field_index_map[0], /* Rule set node field index map */ &r_<%= reduce[:rule].name.gsub("$", "_") %><%= reduce[:rule].id %>_node_field_index_map[0], /* Rule set node field index map */
<% end %> <% end %>
<%= reduce[:rule].rule_set.tree_fields.size %>, /* Number of tree fields */ <%= reduce[:rule].rule_set.ast_fields.size %>, /* Number of AST fields */
<%= reduce[:propagate_optional_target] %>}, /* Propagate optional target? */ <%= reduce[:propagate_optional_target] %>}, /* Propagate optional target? */
<% else %> <% else %>
<%= reduce[:n_states] %>u}, <%= reduce[:n_states] %>u},
@ -890,131 +841,7 @@ static void state_values_stack_free(state_values_stack_t * stack)
free(stack->entries); free(stack->entries);
} }
<% if @grammar.tree %> <% unless @grammar.ast %>
/* Tree arena helpers. */
/**
* Allocate a new (zeroed) tree node in the context arena.
*
* @return The new node ID.
*/
static <%= @grammar.prefix %>node_id_t tree_new_node(<%= @grammar.prefix %>context_t * context)
{
<% if @cpp %>
<%= @grammar.prefix %>node_id_t id = (<%= @grammar.prefix %>node_id_t)context-><%= @grammar.prefix %>tree_nodes.size();
context-><%= @grammar.prefix %>tree_nodes.emplace_back();
return id;
<% else %>
if (context-><%= @grammar.prefix %>tree_nodes_length >= context-><%= @grammar.prefix %>tree_nodes_capacity)
{
size_t new_capacity = context-><%= @grammar.prefix %>tree_nodes_capacity * 2u;
<%= @grammar.prefix %>node_data_t * new_nodes = (<%= @grammar.prefix %>node_data_t *)malloc(new_capacity * sizeof(<%= @grammar.prefix %>node_data_t));
memcpy(new_nodes, context-><%= @grammar.prefix %>tree_nodes, context-><%= @grammar.prefix %>tree_nodes_length * sizeof(<%= @grammar.prefix %>node_data_t));
free(context-><%= @grammar.prefix %>tree_nodes);
context-><%= @grammar.prefix %>tree_nodes = new_nodes;
context-><%= @grammar.prefix %>tree_nodes_capacity = new_capacity;
}
<%= @grammar.prefix %>node_id_t id = (<%= @grammar.prefix %>node_id_t)context-><%= @grammar.prefix %>tree_nodes_length;
memset(&context-><%= @grammar.prefix %>tree_nodes[id], 0, sizeof(<%= @grammar.prefix %>node_data_t));
context-><%= @grammar.prefix %>tree_nodes_length += 1u;
return id;
<% end %>
}
/**
* Reserve n contiguous (zeroed) child slots in the shared children array.
*
* @return The offset of the first reserved slot.
*/
static <%= @grammar.prefix %>node_id_t tree_reserve_children(<%= @grammar.prefix %>context_t * context, size_t n)
{
<% if @cpp %>
<%= @grammar.prefix %>node_id_t offset = (<%= @grammar.prefix %>node_id_t)context-><%= @grammar.prefix %>tree_children.size();
context-><%= @grammar.prefix %>tree_children.resize(context-><%= @grammar.prefix %>tree_children.size() + n);
return offset;
<% else %>
size_t offset = context-><%= @grammar.prefix %>tree_children_length;
size_t needed = offset + n;
if (needed > context-><%= @grammar.prefix %>tree_children_capacity)
{
size_t new_capacity = context-><%= @grammar.prefix %>tree_children_capacity ? context-><%= @grammar.prefix %>tree_children_capacity : 1u;
while (new_capacity < needed)
{
new_capacity *= 2u;
}
<%= @grammar.prefix %>node_id_t * new_children = (<%= @grammar.prefix %>node_id_t *)malloc(new_capacity * sizeof(<%= @grammar.prefix %>node_id_t));
if (context-><%= @grammar.prefix %>tree_children != NULL)
{
memcpy(new_children, context-><%= @grammar.prefix %>tree_children, context-><%= @grammar.prefix %>tree_children_length * sizeof(<%= @grammar.prefix %>node_id_t));
free(context-><%= @grammar.prefix %>tree_children);
}
context-><%= @grammar.prefix %>tree_children = new_children;
context-><%= @grammar.prefix %>tree_children_capacity = new_capacity;
}
memset(&context-><%= @grammar.prefix %>tree_children[offset], 0, n * sizeof(<%= @grammar.prefix %>node_id_t));
context-><%= @grammar.prefix %>tree_children_length = needed;
return (<%= @grammar.prefix %>node_id_t)offset;
<% end %>
}
/* Tree node field accessor functions. */
<%= c_tree_accessor_defs %>
<% end %>
<% unless @grammar.tree %>
/**
* Get the rule position (start or end) for the currently matched rule.
*/
static <%= @grammar.prefix %>position_t get_rule_position(state_values_stack_t * statevalues, size_t i, size_t n_states, bool get_end)
{
if (n_states > 0u)
{
if (i == 0u)
{
if (get_end)
{
int stack_index = -1;
for (size_t j = 0u; j < n_states; j++)
{
state_value_t * sv = state_values_stack_index(statevalues, stack_index - (int)j);
if (<%= @grammar.prefix %>position_valid(sv->end_position))
{
return sv->end_position;
}
}
}
else
{
int stack_index = -(int)n_states;
for (size_t j = 0u; j < n_states; j++)
{
state_value_t * sv = state_values_stack_index(statevalues, stack_index + (int)j);
if (<%= @grammar.prefix %>position_valid(sv->position))
{
return sv->position;
}
}
}
}
else
{
if (get_end)
{
return state_values_stack_index(statevalues, -1 - (int)n_states + (int)i)->end_position;
}
else
{
return state_values_stack_index(statevalues, -1 - (int)n_states + (int)i)->position;
}
}
}
<%= @grammar.prefix %>position_t empty_pos;
memset(&empty_pos, 0, sizeof(empty_pos));
return empty_pos;
}
<% end %>
<% if !@grammar.tree || @grammar.parser_user_code_used? %>
/** /**
* Execute user code associated with a parser rule. * Execute user code associated with a parser rule.
* *
@ -1025,7 +852,7 @@ static <%= @grammar.prefix %>position_t get_rule_position(state_values_stack_t *
* @retval P_USER_TERMINATED * @retval P_USER_TERMINATED
* User requested to terminate parsing. * User requested to terminate parsing.
*/ */
static size_t parser_user_code(<%= @grammar.tree ? "#{@grammar.prefix}node_id_t _node_id" : "#{@grammar.prefix}value_t * _pvalue" %>, uint32_t rule, state_values_stack_t * statevalues, uint32_t n_states, <%= @grammar.prefix %>context_t * context) static size_t parser_user_code(<%= @grammar.prefix %>value_t * _pvalue, uint32_t rule, state_values_stack_t * statevalues, uint32_t n_states, <%= @grammar.prefix %>context_t * context)
{ {
switch (rule) switch (rule)
{ {
@ -1075,7 +902,7 @@ static size_t check_shift(size_t state_id, size_t symbol_id)
* @param token * @param token
* Incoming token. * Incoming token.
* *
* @return Reduce table index to reduce with, or INVALID_ID if none. * @return State to reduce to, or INVALID_ID if none.
*/ */
static size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token) static size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token)
{ {
@ -1097,17 +924,8 @@ static size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token)
* *
* @param context * @param context
* Lexer/parser context structure. * Lexer/parser context structure.
* @param start_state_id * @start_state_id
* ID of the state in which to start. * ID of the state in which to start.
* @param start_rule_set_id
* Rule set ID for the requested start rule. Only used when
* @p follow_tokens is non-NULL, to gate follow-token shift success.
* @param follow_tokens
* Optional array of caller-provided follow tokens (tokens expected to
* appear immediately after the start rule in some outer context). Used to
* drive the "parse inner" retry logic. May be NULL for a standard parse.
* @param n_follow_tokens
* Number of entries in @p follow_tokens.
* *
* @retval P_SUCCESS * @retval P_SUCCESS
* The parser successfully matched the input text. The parse result value * The parser successfully matched the input text. The parse result value
@ -1120,20 +938,15 @@ static size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token)
* @reval P_UNEXPECTED_INPUT * @reval P_UNEXPECTED_INPUT
* Input text does not match any lexer pattern. * Input text does not match any lexer pattern.
*/ */
static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start_state_id, static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start_state_id)
size_t start_rule_set_id,
<%= @grammar.prefix %>token_t const * follow_tokens, size_t n_follow_tokens)
{ {
<%= @grammar.prefix %>token_info_t token_info; <%= @grammar.prefix %>token_info_t token_info;
<%= @grammar.prefix %>token_t token = INVALID_TOKEN_ID; <%= @grammar.prefix %>token_t token = INVALID_TOKEN_ID;
state_values_stack_t statevalues; state_values_stack_t statevalues;
size_t reduced_rule_set = INVALID_ID; size_t reduced_rule_set = INVALID_ID;
size_t last_shifted_rule_set_id = INVALID_ID; <% if @grammar.ast %>
<% if @grammar.tree %> void * reduced_parser_node;
<%= @grammar.prefix %>node_id_t reduced_parser_node;
<% else %> <% else %>
<%= @grammar.prefix %>position_t reduced_position;
<%= @grammar.prefix %>position_t reduced_end_position;
<%= @grammar.prefix %>value_t reduced_parser_value; <%= @grammar.prefix %>value_t reduced_parser_value;
<% end %> <% end %>
state_values_stack_init(&statevalues); state_values_stack_init(&statevalues);
@ -1144,7 +957,7 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
{ {
if (token == INVALID_TOKEN_ID) if (token == INVALID_TOKEN_ID)
{ {
size_t lexer_result = <%= lex_fn %>(context, &token_info); size_t lexer_result = <%= @grammar.prefix %>lex(context, &token_info);
if (lexer_result != P_SUCCESS) if (lexer_result != P_SUCCESS)
{ {
result = lexer_result; result = lexer_result;
@ -1152,18 +965,6 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
} }
token = token_info.token; token = token_info.token;
} }
/* For a "parse inner" operation, determine once per iteration whether
* the current token is a member of the caller-provided follow token
* set. Used by both the shift-side and reduce-side retries below. */
bool token_is_follow = false;
for (size_t i = 0u; i < n_follow_tokens; i++)
{
if (token == follow_tokens[i])
{
token_is_follow = true;
break;
}
}
size_t shift_state = INVALID_ID; size_t shift_state = INVALID_ID;
if (reduced_rule_set != INVALID_ID) if (reduced_rule_set != INVALID_ID)
{ {
@ -1175,83 +976,44 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
if ((shift_state != INVALID_ID) && (token == TOKEN___EOF)) if ((shift_state != INVALID_ID) && (token == TOKEN___EOF))
{ {
/* Successful parse. */ /* Successful parse. */
<% if @grammar.tree %> <% if @grammar.ast %>
context->parse_result = state_values_stack_index(&statevalues, -1)->node_id; context->parse_result = state_values_stack_index(&statevalues, -1)->ast_node;
<% else %> <% else %>
context->parse_result = state_values_stack_index(&statevalues, -1)->pvalue; context->parse_result = state_values_stack_index(&statevalues, -1)->pvalue;
<% end %> <% end %>
result = P_SUCCESS; result = P_SUCCESS;
break; break;
} }
if ((shift_state == INVALID_ID) && token_is_follow)
{
/* For a "parse inner" operation, if the incoming token is one
* of the caller's follow tokens, retry the shift as
* TOKEN___EOF. Only consider the parse complete if the reduced
* start rule is the only thing on the parse stack (i.e. the
* initial state plus a single shifted start rule set entry). */
size_t retry_shift_state = check_shift(state_values_stack_index(&statevalues, -1)->state_id, TOKEN___EOF);
if ((retry_shift_state != INVALID_ID) &&
(statevalues.length == 2u) &&
(last_shifted_rule_set_id == start_rule_set_id))
{
/* Successful parse via follow token. Rewind the input
* position so that the follow token is not consumed from
* the input stream and remains available for a subsequent
* call to <%= @grammar.prefix %>lex() or a
* <%= @grammar.prefix %>parse*() function. */
context->input_index -= token_info.length;
context->text_position = token_info.position;
<% if @grammar.tree %>
context->parse_result = state_values_stack_index(&statevalues, -1)->node_id;
<% else %>
context->parse_result = state_values_stack_index(&statevalues, -1)->pvalue;
<% end %>
result = P_SUCCESS;
break;
}
}
} }
if (shift_state != INVALID_ID) if (shift_state != INVALID_ID)
{ {
/* We have something to shift. Track the last shifted rule set ID /* We have something to shift. */
* (INVALID_ID if we just shifted a token) so the follow-token
* shift retry can gate success on the reduced start rule being the
* only thing on top of the initial state. */
last_shifted_rule_set_id = reduced_rule_set;
state_values_stack_push(&statevalues); state_values_stack_push(&statevalues);
state_value_t * new_state_info = state_values_stack_index(&statevalues, -1); state_values_stack_index(&statevalues, -1)->state_id = shift_state;
new_state_info->state_id = shift_state;
if (reduced_rule_set == INVALID_ID) if (reduced_rule_set == INVALID_ID)
{ {
/* We shifted a token, mark it consumed. */ /* We shifted a token, mark it consumed. */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= @grammar.prefix %>node_id_t token_node_id = tree_new_node(context); <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %> * token_ast_node = (<%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %> *)malloc(sizeof(<%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %>));
<%= @grammar.prefix %>node_data_t * token_tree_node = &context-><%= @grammar.prefix %>tree_nodes[token_node_id]; token_ast_node->position = token_info.position;
token_tree_node->position = token_info.position; token_ast_node->end_position = token_info.end_position;
token_tree_node->end_position = token_info.end_position; token_ast_node->n_fields = 0u;
token_tree_node->n_fields = 0u; token_ast_node->is_token = 1u;
token_tree_node->is_token = 1u; token_ast_node->token = token;
token_tree_node->token = token; token_ast_node->pvalue = token_info.pvalue;
token_tree_node->pvalue = token_info.pvalue; state_values_stack_index(&statevalues, -1)->ast_node = token_ast_node;
<%= expand_code(@grammar.on_token_node, false, nil, nil) %>
new_state_info->node_id = token_node_id;
<% else %> <% else %>
new_state_info->position = token_info.position; state_values_stack_index(&statevalues, -1)->pvalue = token_info.pvalue;
new_state_info->end_position = token_info.end_position;
new_state_info->pvalue = token_info.pvalue;
<% end %> <% end %>
token = INVALID_TOKEN_ID; token = INVALID_TOKEN_ID;
} }
else else
{ {
/* We shifted a RuleSet. */ /* We shifted a RuleSet. */
<% if @grammar.tree %> <% if @grammar.ast %>
new_state_info->node_id = reduced_parser_node; state_values_stack_index(&statevalues, -1)->ast_node = reduced_parser_node;
<% else %> <% else %>
new_state_info->pvalue = reduced_parser_value; state_values_stack_index(&statevalues, -1)->pvalue = reduced_parser_value;
new_state_info->position = reduced_position;
new_state_info->end_position = reduced_end_position;
<%= @grammar.prefix %>value_t new_parse_result; <%= @grammar.prefix %>value_t new_parse_result;
memset(&new_parse_result, 0, sizeof(new_parse_result)); memset(&new_parse_result, 0, sizeof(new_parse_result));
reduced_parser_value = new_parse_result; reduced_parser_value = new_parse_result;
@ -1262,77 +1024,57 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
} }
size_t reduce_index = check_reduce(state_values_stack_index(&statevalues, -1)->state_id, token); size_t reduce_index = check_reduce(state_values_stack_index(&statevalues, -1)->state_id, token);
if ((reduce_index == INVALID_ID) && token_is_follow)
{
/* For a "parse inner" operation, if the incoming token is one of
* the caller's follow tokens, retry the reduce lookup as
* TOKEN___EOF. Whatever reduce_index results (if any) is used
* regardless of which rule set it reduces to; this allows chains
* of reductions leading up to the start rule. */
reduce_index = check_reduce(state_values_stack_index(&statevalues, -1)->state_id, TOKEN___EOF);
}
if (reduce_index != INVALID_ID) if (reduce_index != INVALID_ID)
{ {
/* We have something to reduce. */ /* We have something to reduce. */
<% if @grammar.tree %> <% if @grammar.ast %>
if (parser_reduce_table[reduce_index].propagate_optional_target) if (parser_reduce_table[reduce_index].propagate_optional_target)
{ {
reduced_parser_node = state_values_stack_index(&statevalues, -1)->node_id; reduced_parser_node = state_values_stack_index(&statevalues, -1)->ast_node;
} }
else if (parser_reduce_table[reduce_index].n_states > 0) else if (parser_reduce_table[reduce_index].n_states > 0)
{ {
uint16_t n_fields = parser_reduce_table[reduce_index].rule_set_node_field_array_size; size_t n_fields = parser_reduce_table[reduce_index].rule_set_node_field_array_size;
/* Reserve child slots. New slots are zero-initialized size_t bytes = sizeof(ASTNode) + n_fields * sizeof(void *);
* (null node ID) so absent optional children remain null. */ ASTNode * node = (ASTNode *)malloc(bytes);
<%= @grammar.prefix %>node_id_t child_offset = tree_reserve_children(context, n_fields); memset(node, 0, bytes);
node->position = INVALID_POSITION;
node->end_position = INVALID_POSITION;
node->n_fields = n_fields;
if (parser_reduce_table[reduce_index].rule_set_node_field_index_map == NULL) if (parser_reduce_table[reduce_index].rule_set_node_field_index_map == NULL)
{ {
for (size_t i = 0; i < parser_reduce_table[reduce_index].n_states; i++) for (size_t i = 0; i < parser_reduce_table[reduce_index].n_states; i++)
{ {
context-><%= @grammar.prefix %>tree_children[child_offset + i] = state_values_stack_index(&statevalues, -(int)parser_reduce_table[reduce_index].n_states + (int)i)->node_id; node->fields[i] = (ASTNode *)state_values_stack_index(&statevalues, -(int)parser_reduce_table[reduce_index].n_states + (int)i)->ast_node;
} }
} }
else else
{ {
for (size_t i = 0; i < parser_reduce_table[reduce_index].n_states; i++) for (size_t i = 0; i < parser_reduce_table[reduce_index].n_states; i++)
{ {
context-><%= @grammar.prefix %>tree_children[child_offset + parser_reduce_table[reduce_index].rule_set_node_field_index_map[i]] = state_values_stack_index(&statevalues, -(int)parser_reduce_table[reduce_index].n_states + (int)i)->node_id; node->fields[parser_reduce_table[reduce_index].rule_set_node_field_index_map[i]] = (ASTNode *)state_values_stack_index(&statevalues, -(int)parser_reduce_table[reduce_index].n_states + (int)i)->ast_node;
} }
} }
<%= @grammar.prefix %>node_id_t node_id = tree_new_node(context);
<%= @grammar.prefix %>node_data_t * node = &context-><%= @grammar.prefix %>tree_nodes[node_id];
node->position = INVALID_POSITION;
node->end_position = INVALID_POSITION;
node->child_offset = child_offset;
node->n_fields = n_fields;
node->is_token = 0u;
bool position_found = false; bool position_found = false;
for (uint16_t i = 0; i < n_fields; i++) for (size_t i = 0; i < n_fields; i++)
{ {
<%= @grammar.prefix %>node_id_t child_id = context-><%= @grammar.prefix %>tree_children[child_offset + i]; ASTNode * child = node->fields[i];
if ((child_id != 0u) && <%= @grammar.prefix %>position_valid(context-><%= @grammar.prefix %>tree_nodes[child_id].position)) if ((child != NULL) && <%= @grammar.prefix %>position_valid(child->position))
{ {
if (!position_found) if (!position_found)
{ {
node->position = context-><%= @grammar.prefix %>tree_nodes[child_id].position; node->position = child->position;
position_found = true; position_found = true;
} }
node->end_position = context-><%= @grammar.prefix %>tree_nodes[child_id].end_position; node->end_position = child->end_position;
} }
} }
reduced_parser_node = node_id; reduced_parser_node = node;
} }
else else
{ {
reduced_parser_node = 0u; reduced_parser_node = NULL;
} }
<% if @grammar.parser_user_code_used? %>
if (parser_user_code(reduced_parser_node, parser_reduce_table[reduce_index].rule, &statevalues, parser_reduce_table[reduce_index].n_states, context) == P_USER_TERMINATED)
{
state_values_stack_free(&statevalues);
return P_USER_TERMINATED;
}
<% end %>
<% else %> <% else %>
<%= @grammar.prefix %>value_t reduced_parser_value2; <%= @grammar.prefix %>value_t reduced_parser_value2;
memset(&reduced_parser_value2, 0, sizeof(reduced_parser_value2)); memset(&reduced_parser_value2, 0, sizeof(reduced_parser_value2));
@ -1342,16 +1084,6 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
return P_USER_TERMINATED; return P_USER_TERMINATED;
} }
reduced_parser_value = reduced_parser_value2; reduced_parser_value = reduced_parser_value2;
if (parser_reduce_table[reduce_index].n_states > 0u)
{
reduced_position = get_rule_position(&statevalues, 0u, parser_reduce_table[reduce_index].n_states, false);
reduced_end_position = get_rule_position(&statevalues, 0u, parser_reduce_table[reduce_index].n_states, true);
}
else
{
memset(&reduced_position, 0, sizeof(reduced_position));
memset(&reduced_end_position, 0, sizeof(reduced_end_position));
}
<% end %> <% end %>
reduced_rule_set = parser_reduce_table[reduce_index].rule_set; reduced_rule_set = parser_reduce_table[reduce_index].rule_set;
state_values_stack_pop(&statevalues, parser_reduce_table[reduce_index].n_states); state_values_stack_pop(&statevalues, parser_reduce_table[reduce_index].n_states);
@ -1374,20 +1106,14 @@ static size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start
size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context) size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context)
{ {
return parse_from(context, 0u, <%= @parser.rule_sets[@grammar.start_rules[0]].id %>u, NULL, 0u); return parse_from(context, 0u);
} }
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context) size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context)
{ {
return parse_from(context, <%= i %>u, <%= @parser.rule_sets[start_rule].id %>u, NULL, 0u); return parse_from(context, <%= i %>u);
}
size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %>context_t * context,
<%= @grammar.prefix %>token_t const * follow_tokens, size_t n_follow_tokens)
{
return parse_from(context, <%= i %>u, <%= @parser.rule_sets[start_rule].id %>u, follow_tokens, n_follow_tokens);
} }
<% end %> <% end %>
@ -1399,15 +1125,15 @@ size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %
* *
* @return Parse result value. * @return Parse result value.
*/ */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= h_type(@grammar.start_rules[0]) %> <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context) <%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context)
{ {
return <%= tree_handle(h_type(@grammar.start_rules[0]), "context->parse_result") %>; return (<%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> *) context->parse_result;
} }
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
<%= h_type(start_rule) %> <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context) <%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context)
{ {
return <%= tree_handle(h_type(start_rule), "context->parse_result") %>; return (<%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> *) context->parse_result;
} }
<% end %> <% end %>
<% else %> <% else %>
@ -1436,58 +1162,6 @@ size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %
return context->text_position; return context->text_position;
} }
/**
* Set the current text input position.
*
* This can be used to set the initial text position to something other than
* (1, 1) for a nested parse operation so that error positions reported by
* subsequent lexer/parser calls are relative to a larger enclosing document.
*
* @param context
* Lexer/parser context structure.
* @param position
* Text position to set.
*/
void <%= @grammar.prefix %>set_position(<%= @grammar.prefix %>context_t * context, <%= @grammar.prefix %>position_t position)
{
context->text_position = position;
}
/**
* Get the current input text byte offset.
*
* @param context
* Lexer/parser context structure.
*
* @return Current input text byte offset (measured from the start of the
* input text passed to <%= @grammar.prefix %>context_new()).
*/
size_t <%= @grammar.prefix %>input_index(<%= @grammar.prefix %>context_t * context)
{
return context->input_index;
}
/**
* Set the current input text byte offset.
*
* This moves the lexer's read cursor to the given byte offset (measured from
* the start of the input text passed to <%= @grammar.prefix %>context_new()).
* It can be used together with <%= @grammar.prefix %>set_position() to rewind
* the input part-way through a parse in order to re-read an earlier section of
* the input. The byte offset is not validated; the caller is responsible for
* providing an offset within the bounds of the input text. A value previously
* returned by <%= @grammar.prefix %>input_index() is a suitable argument.
*
* @param context
* Lexer/parser context structure.
* @param input_index
* Input text byte offset to set.
*/
void <%= @grammar.prefix %>set_input_index(<%= @grammar.prefix %>context_t * context, size_t input_index)
{
context->input_index = input_index;
}
/** /**
* Get the user terminate code. * Get the user terminate code.
* *
@ -1510,3 +1184,45 @@ size_t <%= @grammar.prefix %>user_terminate_code(<%= @grammar.prefix %>context_t
{ {
return context->token; return context->token;
} }
<% if @grammar.ast %>
static void free_ast_node(ASTNode * node)
{
if (node->is_token)
{
<% if @grammar.free_token_node %>
<%= @grammar.free_token_node %>((<%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %> *) node);
<% end %>
/* TODO: free value_t */
}
else if (node->n_fields > 0u)
{
for (size_t i = 0u; i < node->n_fields; i++)
{
if (node->fields[i] != NULL)
{
free_ast_node(node->fields[i]);
}
}
}
free(node);
}
/**
* Free all AST node memory.
*/
void <%= @grammar.prefix %>free_ast(<%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> * ast)
{
free_ast_node((ASTNode *)ast);
}
<% @grammar.start_rules.each_with_index do |start_rule, i| %>
/**
* Free all AST node memory.
*/
void <%= @grammar.prefix %>free_ast_<%= start_rule %>(<%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> * ast)
{
free_ast_node((ASTNode *)ast);
}
<% end %>
<% end %>

View File

@ -9,7 +9,7 @@ module <%= @grammar.modulename %>;
<% end %> <% end %>
import core.memory; import core.memory;
import core.stdc.stdlib : malloc, free; import core.stdc.stdlib : malloc;
/************************************************************************** /**************************************************************************
* User code blocks * User code blocks
@ -59,10 +59,10 @@ public alias <%= @grammar.prefix %>code_point_t = uint;
*/ */
public struct <%= @grammar.prefix %>position_t public struct <%= @grammar.prefix %>position_t
{ {
/** Input text row (1-based). */ /** Input text row (0-based). */
uint row; uint row;
/** Input text column (1-based). */ /** Input text column (0-based). */
uint col; uint col;
/** Invalid position value. */ /** Invalid position value. */
@ -75,7 +75,7 @@ public struct <%= @grammar.prefix %>position_t
} }
} }
<% if @grammar.tree %> <% if @grammar.ast %>
/** Parser values type. */ /** Parser values type. */
public alias <%= @grammar.prefix %>value_t = <%= @grammar.ptype %>; public alias <%= @grammar.prefix %>value_t = <%= @grammar.ptype %>;
<% else %> <% else %>
@ -86,139 +86,41 @@ public union <%= @grammar.prefix %>value_t
<%= typestring %> v_<%= name %>; <%= typestring %> v_<%= name %>;
<% end %> <% end %>
} }
/** Parser value constructor(s). */
<% @grammar.ptypes.each do |name, typestring| %>
public <%= @grammar.prefix %>value_t <%= @grammar.prefix %>value<%= name == "default" ? "" : "_#{name}" %>(T)(T v)
{
return <%= @grammar.prefix %>value_t(v_<%= name %>: v);
}
<% end %> <% end %>
/** Parser value accessor(s). */ <% if @grammar.ast %>
<% @grammar.ptypes.each do |name, typestring| %> /** Common AST node structure. */
public <%= typestring %> <%= @grammar.prefix %>value_get<%= name == "default" ? "" : "_#{name}" %>(<%= @grammar.prefix %>value_t * pvalue) private struct ASTNode
{
return pvalue.v_<%= name %>;
}
<% end %>
<% end %>
<% if @grammar.tree %>
/** Tree node ID type (index into the context node arena). ID 0 is null. */
public alias <%= @grammar.prefix %>node_id_t = uint;
/**
* Tree node record.
*
* All tree nodes are stored contiguously in the context node arena. Child
* links are stored in a shared children array: a node's children
* occupy children[child_offset .. child_offset + n_fields]. Token payload
* fields (token, pvalue, and any user fields) are only meaningful when
* is_token is true.
*/
private struct <%= @grammar.prefix %>node_data_t
{ {
<%= @grammar.prefix %>position_t position; <%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position; <%= @grammar.prefix %>position_t end_position;
<%= @grammar.prefix %>node_id_t child_offset; void *[0] fields;
ushort n_fields; }
bool is_token;
/** AST node types. @{ */
public struct <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %>
{
/* ASTNode fields must be present in the same order here. */
<%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position;
<%= @grammar.prefix %>token_t token; <%= @grammar.prefix %>token_t token;
<%= @grammar.prefix %>value_t pvalue; <%= @grammar.prefix %>value_t pvalue;
<%= @grammar.token_user_fields %>
} }
/** Tree node handle types. @{ */ <% @parser.rule_sets.each do |name, rule_set| %>
<% next if name.start_with?("$") %>
/** Token tree node handle. */ <% next if rule_set.optional? %>
public struct <%= @grammar.tree_prefix %>Token<%= @grammar.tree_suffix %> public struct <%= @grammar.ast_prefix %><%= name %><%= @grammar.ast_suffix %>
{ {
private <%= @grammar.prefix %>context_t * __context; <%= @grammar.prefix %>position_t position;
private <%= @grammar.prefix %>node_id_t __id; <%= @grammar.prefix %>position_t end_position;
<% rule_set.ast_fields.each do |fields| %>
this(<%= @grammar.prefix %>context_t * context, <%= @grammar.prefix %>node_id_t id) union
{ {
this.__context = context;
this.__id = id;
}
/** Return whether this handle refers to a valid (non-null) node. */
@property bool valid()
{
return __id != 0u;
}
/** Return the node ID (for identity comparison). */
@property <%= @grammar.prefix %>node_id_t node_id()
{
return __id;
}
/** Access the underlying node record (token, pvalue, and user fields). */
@property ref <%= @grammar.prefix %>node_data_t __node()
{
return __context.<%= @grammar.prefix %>tree_nodes[__id];
}
alias __node this;
}
<% tree_node_rule_sets.each do |rule_set| %>
/** <%= rule_set.name %> tree node handle. */
public struct <%= @grammar.tree_prefix %><%= rule_set.name %><%= @grammar.tree_suffix %>
{
private <%= @grammar.prefix %>context_t * __context;
private <%= @grammar.prefix %>node_id_t __id;
this(<%= @grammar.prefix %>context_t * context, <%= @grammar.prefix %>node_id_t id)
{
this.__context = context;
this.__id = id;
}
/** Return whether this handle refers to a valid (non-null) node. */
@property bool valid()
{
return __id != 0u;
}
/** Return the node ID (for identity comparison). */
@property <%= @grammar.prefix %>node_id_t node_id()
{
return __id;
}
/** Text position of the first code point spanned by this node. */
@property <%= @grammar.prefix %>position_t position()
{
return __context.<%= @grammar.prefix %>tree_nodes[__id].position;
}
/** Text position of the last code point spanned by this node. */
@property <%= @grammar.prefix %>position_t end_position()
{
return __context.<%= @grammar.prefix %>tree_nodes[__id].end_position;
}
/** Number of child fields in this node. */
@property ushort n_fields()
{
return __id ? __context.<%= @grammar.prefix %>tree_nodes[__id].n_fields : cast(ushort)0u;
}
<% rule_set.tree_fields.each_with_index do |fields, i| %>
<% fields.each do |field_name, type| %> <% fields.each do |field_name, type| %>
<%= type %> * <%= field_name %>;
/** Access the <%= field_name %> child node. */
@property <%= type %> <%= field_name %>()
{
if (__id == 0u)
{
return <%= type %>(__context, 0u);
}
return <%= type %>(__context, __context.<%= @grammar.prefix %>tree_children[__context.<%= @grammar.prefix %>tree_nodes[__id].child_offset + <%= i %>u]);
}
<% end %> <% end %>
}
<% end %> <% end %>
} }
@ -270,14 +172,8 @@ public struct <%= @grammar.prefix %>context_t
/* Parser context data. */ /* Parser context data. */
/** Parse result value. */ /** Parse result value. */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= @grammar.prefix %>node_id_t parse_result; void * parse_result;
/** Tree node arena. Node ID 0 is reserved as the null node. */
<%= @grammar.prefix %>node_data_t[] <%= @grammar.prefix %>tree_nodes;
/** Shared tree child links. */
<%= @grammar.prefix %>node_id_t[] <%= @grammar.prefix %>tree_children;
<% else %> <% else %>
<%= @grammar.prefix %>value_t parse_result; <%= @grammar.prefix %>value_t parse_result;
<% end %> <% end %>
@ -287,8 +183,6 @@ public struct <%= @grammar.prefix %>context_t
/** User terminate code. */ /** User terminate code. */
size_t user_terminate_code; size_t user_terminate_code;
<%= @grammar.context_user_fields %>
} }
/************************************************************************** /**************************************************************************
@ -328,52 +222,26 @@ private enum size_t INVALID_ID = cast(size_t)-1;
*************************************************************************/ *************************************************************************/
/** /**
* Allocate and initialize lexer/parser context structure. * Initialize lexer/parser context structure.
*
* Deinitialize and deallocate with <%= @grammar.prefix %>context_delete().
* *
* @param[out] context
* Lexer/parser context structure.
* @param input * @param input
* Text input. * Text input.
*
* @return Context structure for lexer/parser.
*/ */
<%= @grammar.prefix %>context_t * <%= @grammar.prefix %>context_new(string input) public void <%= @grammar.prefix %>context_init(<%= @grammar.prefix %>context_t * context, string input)
{ {
/* New default-initialized context structure. */ /* New default-initialized context structure. */
<%= @grammar.prefix %>context_t * context = new <%= @grammar.prefix %>context_t; <%= @grammar.prefix %>context_t newcontext;
/* Lexer initialization. */ /* Lexer initialization. */
context.input = input; newcontext.input = input;
context.text_position.row = 1u; newcontext.text_position.row = 1u;
context.text_position.col = 1u; newcontext.text_position.col = 1u;
context.mode = <%= @lexer.mode_id("default") %>; newcontext.mode = <%= @lexer.mode_id("default") %>;
<% if @grammar.tree %>
/* Reserve node ID 0 as the null tree node. */ /* Copy to the user's context structure. */
context.<%= @grammar.prefix %>tree_nodes = new <%= @grammar.prefix %>node_data_t[](1); *context = newcontext;
<% end %>
return context;
}
/**
* Deinitialize and deallocate lexer/parser context structure.
*
* @param context
* Lexer/parser context structure allocated with <%= @grammar.prefix %>context_new().
*/
void <%= @grammar.prefix %>context_delete(<%= @grammar.prefix %>context_t * context)
{
<% if @grammar.tree && @grammar.free_token_node != "" %>
foreach (ref node; context.<%= @grammar.prefix %>tree_nodes)
{
if (node.is_token)
{
<%= @grammar.prefix %>node_data_t * token_tree_node = &node;
<%= expand_code(@grammar.free_token_node, false, nil, nil) %>
}
}
<% end %>
} }
/************************************************************************** /**************************************************************************
@ -572,7 +440,7 @@ private immutable lexer_mode_t[] lexer_mode_table = [
* Lexer/parser context structure. * Lexer/parser context structure.
* @param code_id * @param code_id
* The ID of the user code block to execute. * The ID of the user code block to execute.
* @param match_text * @param match
* Matched text for this pattern. * Matched text for this pattern.
* @param out_token_info * @param out_token_info
* Lexer token info in progress. * Lexer token info in progress.
@ -581,7 +449,7 @@ private immutable lexer_mode_t[] lexer_mode_table = [
* not explicitly return a token. * not explicitly return a token.
*/ */
private <%= @grammar.prefix %>token_t lexer_user_code(<%= @grammar.prefix %>context_t * context, private <%= @grammar.prefix %>token_t lexer_user_code(<%= @grammar.prefix %>context_t * context,
lexer_user_code_id_t code_id, string match_text, lexer_user_code_id_t code_id, string match,
<%= @grammar.prefix %>token_info_t * out_token_info) <%= @grammar.prefix %>token_info_t * out_token_info)
{ {
switch (code_id) switch (code_id)
@ -759,27 +627,11 @@ private size_t attempt_lex_token(<%= @grammar.prefix %>context_t * context, <%=
{ {
case P_SUCCESS: case P_SUCCESS:
<%= @grammar.prefix %>token_t token_to_accept = match_info.accepting_state.token; <%= @grammar.prefix %>token_t token_to_accept = match_info.accepting_state.token;
/* Calculate the token length and start/end positions before invoking
* the lexer user code so that the user code can access them. The
* context input text position tracking is not updated until after the
* user code has run so that it is left unchanged if the user code
* requests to terminate the lexer. */
token_info.length = match_info.length;
if (match_info.end_delta_position.row != 0u)
{
token_info.end_position.row = token_info.position.row + match_info.end_delta_position.row;
token_info.end_position.col = match_info.end_delta_position.col;
}
else
{
token_info.end_position.row = token_info.position.row;
token_info.end_position.col = token_info.position.col + match_info.end_delta_position.col;
}
if (match_info.accepting_state.code_id != INVALID_USER_CODE_ID) if (match_info.accepting_state.code_id != INVALID_USER_CODE_ID)
{ {
string match_text = context.input[context.input_index..(context.input_index + match_info.length)]; string match = context.input[context.input_index..(context.input_index + match_info.length)];
<%= @grammar.prefix %>token_t user_code_token = lexer_user_code(context, <%= @grammar.prefix %>token_t user_code_token = lexer_user_code(context,
match_info.accepting_state.code_id, match_text, &token_info); match_info.accepting_state.code_id, match, &token_info);
/* A TERMINATE_TOKEN_ID return code from lexer_user_code() means /* A TERMINATE_TOKEN_ID return code from lexer_user_code() means
* that the user code is requesting to terminate the lexer. */ * that the user code is requesting to terminate the lexer. */
if (user_code_token == TERMINATE_TOKEN_ID) if (user_code_token == TERMINATE_TOKEN_ID)
@ -813,6 +665,17 @@ private size_t attempt_lex_token(<%= @grammar.prefix %>context_t * context, <%=
return P_DROP; return P_DROP;
} }
token_info.token = token_to_accept; token_info.token = token_to_accept;
token_info.length = match_info.length;
if (match_info.end_delta_position.row != 0u)
{
token_info.end_position.row = token_info.position.row + match_info.end_delta_position.row;
token_info.end_position.col = match_info.end_delta_position.col;
}
else
{
token_info.end_position.row = token_info.position.row;
token_info.end_position.col = token_info.position.col + match_info.end_delta_position.col;
}
*out_token_info = token_info; *out_token_info = token_info;
return P_SUCCESS; return P_SUCCESS;
@ -934,7 +797,7 @@ private struct reduce_t
* reduce action. * reduce action.
*/ */
parser_state_id_t n_states; parser_state_id_t n_states;
<% if @grammar.tree %> <% if @grammar.ast %>
/** /**
* Map of rule components to rule set child fields. * Map of rule components to rule set child fields.
@ -942,7 +805,7 @@ private struct reduce_t
immutable(ushort) * rule_set_node_field_index_map; immutable(ushort) * rule_set_node_field_index_map;
/** /**
* Number of rule set tree node fields. * Number of rule set AST node fields.
*/ */
ushort rule_set_node_field_array_size; ushort rule_set_node_field_array_size;
@ -981,14 +844,12 @@ private struct state_value_t
/** Parser state ID. */ /** Parser state ID. */
size_t state_id; size_t state_id;
<% if @grammar.tree %>
/** Tree node ID. */
<%= @grammar.prefix %>node_id_t node_id;
<% else %>
<%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position;
/** Parser value from this state. */ /** Parser value from this state. */
<%= @grammar.prefix %>value_t pvalue; <%= @grammar.prefix %>value_t pvalue;
<% if @grammar.ast %>
/** AST node. */
void * ast_node;
<% end %> <% end %>
this(size_t state_id) this(size_t state_id)
@ -1004,7 +865,7 @@ private immutable shift_t[] parser_shift_table = [
<% end %> <% end %>
]; ];
<% if @grammar.tree %> <% if @grammar.ast %>
<% @grammar.rules.each do |rule| %> <% @grammar.rules.each do |rule| %>
<% unless rule.flat_rule_set_node_field_index_map? %> <% unless rule.flat_rule_set_node_field_index_map? %>
immutable ushort[<%= rule.rule_set_node_field_index_map.size %>] r_<%= rule.name.gsub("$", "_") %><%= rule.id %>_node_field_index_map = [<%= rule.rule_set_node_field_index_map.map {|v| v.to_s}.join(", ") %>]; immutable ushort[<%= rule.rule_set_node_field_index_map.size %>] r_<%= rule.name.gsub("$", "_") %><%= rule.id %>_node_field_index_map = [<%= rule.rule_set_node_field_index_map.map {|v| v.to_s}.join(", ") %>];
@ -1019,14 +880,14 @@ private immutable reduce_t[] parser_reduce_table = [
<%= reduce[:token_id] %>u, /* Token: <%= reduce[:token] ? reduce[:token].name : "(any)" %> */ <%= reduce[:token_id] %>u, /* Token: <%= reduce[:token] ? reduce[:token].name : "(any)" %> */
<%= reduce[:rule_id] %>u, /* Rule ID */ <%= reduce[:rule_id] %>u, /* Rule ID */
<%= reduce[:rule_set_id] %>u, /* Rule set ID (<%= reduce[:rule].rule_set.name %>) */ <%= reduce[:rule_set_id] %>u, /* Rule set ID (<%= reduce[:rule].rule_set.name %>) */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= reduce[:n_states] %>u, /* Number of states */ <%= reduce[:n_states] %>u, /* Number of states */
<% if reduce[:rule].flat_rule_set_node_field_index_map? %> <% if reduce[:rule].flat_rule_set_node_field_index_map? %>
null, /* No rule set node field index map (flat map) */ null, /* No rule set node field index map (flat map) */
<% else %> <% else %>
&r_<%= reduce[:rule].name.gsub("$", "_") %><%= reduce[:rule].id %>_node_field_index_map[0], /* Rule set node field index map */ &r_<%= reduce[:rule].name.gsub("$", "_") %><%= reduce[:rule].id %>_node_field_index_map[0], /* Rule set node field index map */
<% end %> <% end %>
<%= reduce[:rule].rule_set.tree_fields.size %>, /* Number of tree fields */ <%= reduce[:rule].rule_set.ast_fields.size %>, /* Number of AST fields */
<%= reduce[:propagate_optional_target] %>), /* Propagate optional target? */ <%= reduce[:propagate_optional_target] %>), /* Propagate optional target? */
<% else %> <% else %>
<%= reduce[:n_states] %>u), /* Number of states */ <%= reduce[:n_states] %>u), /* Number of states */
@ -1041,56 +902,7 @@ private immutable parser_state_t[] parser_state_table = [
<% end %> <% end %>
]; ];
<% unless @grammar.tree %> <% unless @grammar.ast %>
/**
* Get the rule position (start or end) for the currently matched rule.
*/
private <%= @grammar.prefix %>position_t get_rule_position(state_value_t[] statevalues, size_t i, size_t n_states, bool get_end)
{
if (n_states > 0u)
{
if (i == 0u)
{
if (get_end)
{
for (size_t j = 0u; j < n_states; j++)
{
state_value_t * sv = &statevalues[$-1-j];
if (sv.end_position.valid)
{
return sv.end_position;
}
}
}
else
{
for (size_t j = 0u; j < n_states; j++)
{
state_value_t * sv = &statevalues[$-n_states+j];
if (sv.position.valid)
{
return sv.position;
}
}
}
}
else
{
if (get_end)
{
return statevalues[$-1-n_states+i].end_position;
}
else
{
return statevalues[$-1-n_states+i].position;
}
}
}
return <%= @grammar.prefix %>position_t.INVALID;
}
<% end %>
<% if !@grammar.tree || @grammar.parser_user_code_used? %>
/** /**
* Execute user code associated with a parser rule. * Execute user code associated with a parser rule.
* *
@ -1101,7 +913,7 @@ private <%= @grammar.prefix %>position_t get_rule_position(state_value_t[] state
* @retval P_USER_TERMINATED * @retval P_USER_TERMINATED
* User requested to terminate parsing. * User requested to terminate parsing.
*/ */
private size_t parser_user_code(<%= @grammar.tree ? "#{@grammar.prefix}node_id_t _node_id" : "#{@grammar.prefix}value_t * _pvalue" %>, uint rule, state_value_t[] statevalues, uint n_states, <%= @grammar.prefix %>context_t * context) private size_t parser_user_code(<%= @grammar.prefix %>value_t * _pvalue, uint rule, state_value_t[] statevalues, uint n_states, <%= @grammar.prefix %>context_t * context)
{ {
switch (rule) switch (rule)
{ {
@ -1151,7 +963,7 @@ private size_t check_shift(size_t state_id, size_t symbol_id)
* @param token * @param token
* Incoming token. * Incoming token.
* *
* @return Reduce table index to reduce with, or INVALID_ID if none. * @return State to reduce to, or INVALID_ID if none.
*/ */
private size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token) private size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token)
{ {
@ -1173,16 +985,8 @@ private size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token
* *
* @param context * @param context
* Lexer/parser context structure. * Lexer/parser context structure.
* @param start_state_id * @start_state_id
* ID of the state in which to start. * ID of the state in which to start.
* @param start_rule_set_id
* Rule set ID for the requested start rule. Only used when
* @p follow_tokens is non-empty, to gate follow-token shift success.
* @param follow_tokens
* Optional slice of caller-provided follow tokens (tokens expected to
* appear immediately after the start rule in some outer context). Used to
* drive the "parse inner" retry logic. May be null/empty for a standard
* parse.
* *
* @retval P_SUCCESS * @retval P_SUCCESS
* The parser successfully matched the input text. The parse result value * The parser successfully matched the input text. The parse result value
@ -1195,46 +999,29 @@ private size_t check_reduce(size_t state_id, <%= @grammar.prefix %>token_t token
* @reval P_UNEXPECTED_INPUT * @reval P_UNEXPECTED_INPUT
* Input text does not match any lexer pattern. * Input text does not match any lexer pattern.
*/ */
private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start_state_id, private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t start_state_id)
size_t start_rule_set_id,
const(<%= @grammar.prefix %>token_t)[] follow_tokens)
{ {
<%= @grammar.prefix %>token_info_t token_info; <%= @grammar.prefix %>token_info_t token_info;
<%= @grammar.prefix %>token_t token = INVALID_TOKEN_ID; <%= @grammar.prefix %>token_t token = INVALID_TOKEN_ID;
state_value_t[] statevalues = new state_value_t[](1); state_value_t[] statevalues = new state_value_t[](1);
statevalues[0].state_id = start_state_id; statevalues[0].state_id = start_state_id;
size_t reduced_rule_set = INVALID_ID; size_t reduced_rule_set = INVALID_ID;
size_t last_shifted_rule_set_id = INVALID_ID; <% if @grammar.ast %>
<% if @grammar.tree %> void * reduced_parser_node;
<%= @grammar.prefix %>node_id_t reduced_parser_node;
<% else %> <% else %>
<%= @grammar.prefix %>position_t reduced_position;
<%= @grammar.prefix %>position_t reduced_end_position;
<%= @grammar.prefix %>value_t reduced_parser_value; <%= @grammar.prefix %>value_t reduced_parser_value;
<% end %> <% end %>
for (;;) for (;;)
{ {
if (token == INVALID_TOKEN_ID) if (token == INVALID_TOKEN_ID)
{ {
size_t lexer_result = <%= lex_fn %>(context, &token_info); size_t lexer_result = <%= @grammar.prefix %>lex(context, &token_info);
if (lexer_result != P_SUCCESS) if (lexer_result != P_SUCCESS)
{ {
return lexer_result; return lexer_result;
} }
token = token_info.token; token = token_info.token;
} }
/* For a "parse inner" operation, determine once per iteration whether
* the current token is a member of the caller-provided follow token
* set. Used by both the shift-side and reduce-side retries below. */
bool token_is_follow = false;
foreach (eof_token; follow_tokens)
{
if (token == eof_token)
{
token_is_follow = true;
break;
}
}
size_t shift_state = INVALID_ID; size_t shift_state = INVALID_ID;
if (reduced_rule_set != INVALID_ID) if (reduced_rule_set != INVALID_ID)
{ {
@ -1246,67 +1033,25 @@ private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t star
if ((shift_state != INVALID_ID) && (token == TOKEN___EOF)) if ((shift_state != INVALID_ID) && (token == TOKEN___EOF))
{ {
/* Successful parse. */ /* Successful parse. */
<% if @grammar.tree %> <% if @grammar.ast %>
context.parse_result = statevalues[$-1].node_id; context.parse_result = statevalues[$-1].ast_node;
<% else %> <% else %>
context.parse_result = statevalues[$-1].pvalue; context.parse_result = statevalues[$-1].pvalue;
<% end %> <% end %>
return P_SUCCESS; return P_SUCCESS;
} }
if ((shift_state == INVALID_ID) && token_is_follow)
{
/* For a "parse inner" operation, if the incoming token is one
* of the caller's follow tokens, retry the shift as
* TOKEN___EOF. Only consider the parse complete if the reduced
* start rule is the only thing on the parse stack (i.e. the
* initial state plus a single shifted start rule set entry). */
size_t retry_shift_state = check_shift(statevalues[$-1].state_id, TOKEN___EOF);
if ((retry_shift_state != INVALID_ID) &&
(statevalues.length == 2u) &&
(last_shifted_rule_set_id == start_rule_set_id))
{
/* Successful parse via follow token. Rewind the input
* position so that the follow token is not consumed from
* the input stream and remains available for a subsequent
* call to <%= @grammar.prefix %>lex() or a
* <%= @grammar.prefix %>parse*() function. */
context.input_index -= token_info.length;
context.text_position = token_info.position;
<% if @grammar.tree %>
context.parse_result = statevalues[$-1].node_id;
<% else %>
context.parse_result = statevalues[$-1].pvalue;
<% end %>
return P_SUCCESS;
}
}
} }
if (shift_state != INVALID_ID) if (shift_state != INVALID_ID)
{ {
/* We have something to shift. Track the last shifted rule set ID /* We have something to shift. */
* (INVALID_ID if we just shifted a token) so the follow-token
* shift retry can gate success on the reduced start rule being the
* only thing on top of the initial state. */
last_shifted_rule_set_id = reduced_rule_set;
statevalues ~= state_value_t(shift_state); statevalues ~= state_value_t(shift_state);
if (reduced_rule_set == INVALID_ID) if (reduced_rule_set == INVALID_ID)
{ {
/* We shifted a token, mark it consumed. */ /* We shifted a token, mark it consumed. */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= @grammar.prefix %>node_id_t token_node_id = cast(<%= @grammar.prefix %>node_id_t)context.<%= @grammar.prefix %>tree_nodes.length; <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %> * token_ast_node = new <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %>(token_info.position, token_info.end_position, token, token_info.pvalue);
context.<%= @grammar.prefix %>tree_nodes ~= <%= @grammar.prefix %>node_data_t.init; statevalues[$-1].ast_node = token_ast_node;
<%= @grammar.prefix %>node_data_t * token_tree_node = &context.<%= @grammar.prefix %>tree_nodes[token_node_id];
token_tree_node.position = token_info.position;
token_tree_node.end_position = token_info.end_position;
token_tree_node.n_fields = 0u;
token_tree_node.is_token = true;
token_tree_node.token = token;
token_tree_node.pvalue = token_info.pvalue;
<%= expand_code(@grammar.on_token_node, false, nil, nil) %>
statevalues[$-1].node_id = token_node_id;
<% else %> <% else %>
statevalues[$-1].position = token_info.position;
statevalues[$-1].end_position = token_info.end_position;
statevalues[$-1].pvalue = token_info.pvalue; statevalues[$-1].pvalue = token_info.pvalue;
<% end %> <% end %>
token = INVALID_TOKEN_ID; token = INVALID_TOKEN_ID;
@ -1314,12 +1059,10 @@ private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t star
else else
{ {
/* We shifted a RuleSet. */ /* We shifted a RuleSet. */
<% if @grammar.tree %> <% if @grammar.ast %>
statevalues[$-1].node_id = reduced_parser_node; statevalues[$-1].ast_node = reduced_parser_node;
<% else %> <% else %>
statevalues[$-1].pvalue = reduced_parser_value; statevalues[$-1].pvalue = reduced_parser_value;
statevalues[$-1].position = reduced_position;
statevalues[$-1].end_position = reduced_end_position;
<%= @grammar.prefix %>value_t new_parse_result; <%= @grammar.prefix %>value_t new_parse_result;
reduced_parser_value = new_parse_result; reduced_parser_value = new_parse_result;
<% end %> <% end %>
@ -1329,78 +1072,60 @@ private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t star
} }
size_t reduce_index = check_reduce(statevalues[$-1].state_id, token); size_t reduce_index = check_reduce(statevalues[$-1].state_id, token);
if ((reduce_index == INVALID_ID) && token_is_follow)
{
/* For a "parse inner" operation, if the incoming token is one of
* the caller's follow tokens, retry the reduce lookup as
* TOKEN___EOF. Whatever reduce_index results (if any) is used
* regardless of which rule set it reduces to; this allows chains
* of reductions leading up to the start rule. */
reduce_index = check_reduce(statevalues[$-1].state_id, TOKEN___EOF);
}
if (reduce_index != INVALID_ID) if (reduce_index != INVALID_ID)
{ {
/* We have something to reduce. */ /* We have something to reduce. */
<% if @grammar.tree %> <% if @grammar.ast %>
if (parser_reduce_table[reduce_index].propagate_optional_target) if (parser_reduce_table[reduce_index].propagate_optional_target)
{ {
reduced_parser_node = statevalues[$ - 1].node_id; reduced_parser_node = statevalues[$ - 1].ast_node;
} }
else if (parser_reduce_table[reduce_index].n_states > 0) else if (parser_reduce_table[reduce_index].n_states > 0)
{ {
ushort n_fields = parser_reduce_table[reduce_index].rule_set_node_field_array_size; size_t n_fields = parser_reduce_table[reduce_index].rule_set_node_field_array_size;
/* Reserve child slots. New slots are zero-initialized size_t node_size = ASTNode.sizeof + n_fields * (void *).sizeof;
* (null node ID) so absent optional children remain null. */ ASTNode * node = cast(ASTNode *)malloc(node_size);
<%= @grammar.prefix %>node_id_t child_offset = cast(<%= @grammar.prefix %>node_id_t)context.<%= @grammar.prefix %>tree_children.length; GC.addRange(node, node_size);
context.<%= @grammar.prefix %>tree_children.length += n_fields; node.position = <%= @grammar.prefix %>position_t.INVALID;
node.end_position = <%= @grammar.prefix %>position_t.INVALID;
foreach (i; 0..n_fields)
{
node.fields[i] = null;
}
if (parser_reduce_table[reduce_index].rule_set_node_field_index_map is null) if (parser_reduce_table[reduce_index].rule_set_node_field_index_map is null)
{ {
foreach (i; 0..parser_reduce_table[reduce_index].n_states) foreach (i; 0..parser_reduce_table[reduce_index].n_states)
{ {
context.<%= @grammar.prefix %>tree_children[child_offset + i] = statevalues[$ - parser_reduce_table[reduce_index].n_states + i].node_id; node.fields[i] = statevalues[$ - parser_reduce_table[reduce_index].n_states + i].ast_node;
} }
} }
else else
{ {
foreach (i; 0..parser_reduce_table[reduce_index].n_states) foreach (i; 0..parser_reduce_table[reduce_index].n_states)
{ {
context.<%= @grammar.prefix %>tree_children[child_offset + parser_reduce_table[reduce_index].rule_set_node_field_index_map[i]] = statevalues[$ - parser_reduce_table[reduce_index].n_states + i].node_id; node.fields[parser_reduce_table[reduce_index].rule_set_node_field_index_map[i]] = statevalues[$ - parser_reduce_table[reduce_index].n_states + i].ast_node;
} }
} }
<%= @grammar.prefix %>node_id_t node_id = cast(<%= @grammar.prefix %>node_id_t)context.<%= @grammar.prefix %>tree_nodes.length;
context.<%= @grammar.prefix %>tree_nodes ~= <%= @grammar.prefix %>node_data_t.init;
<%= @grammar.prefix %>node_data_t * node = &context.<%= @grammar.prefix %>tree_nodes[node_id];
node.position = <%= @grammar.prefix %>position_t.INVALID;
node.end_position = <%= @grammar.prefix %>position_t.INVALID;
node.child_offset = child_offset;
node.n_fields = n_fields;
node.is_token = false;
bool position_found = false; bool position_found = false;
foreach (i; 0..n_fields) foreach (i; 0..n_fields)
{ {
<%= @grammar.prefix %>node_id_t child_id = context.<%= @grammar.prefix %>tree_children[child_offset + i]; ASTNode * child = cast(ASTNode *)node.fields[i];
if (child_id != 0u && context.<%= @grammar.prefix %>tree_nodes[child_id].position.valid) if (child && child.position.valid)
{ {
if (!position_found) if (!position_found)
{ {
node.position = context.<%= @grammar.prefix %>tree_nodes[child_id].position; node.position = child.position;
position_found = true; position_found = true;
} }
node.end_position = context.<%= @grammar.prefix %>tree_nodes[child_id].end_position; node.end_position = child.end_position;
} }
} }
reduced_parser_node = node_id; reduced_parser_node = node;
} }
else else
{ {
reduced_parser_node = 0u; reduced_parser_node = null;
} }
<% if @grammar.parser_user_code_used? %>
if (parser_user_code(reduced_parser_node, parser_reduce_table[reduce_index].rule, statevalues, parser_reduce_table[reduce_index].n_states, context) == P_USER_TERMINATED)
{
return P_USER_TERMINATED;
}
<% end %>
<% else %> <% else %>
<%= @grammar.prefix %>value_t reduced_parser_value2; <%= @grammar.prefix %>value_t reduced_parser_value2;
if (parser_user_code(&reduced_parser_value2, parser_reduce_table[reduce_index].rule, statevalues, parser_reduce_table[reduce_index].n_states, context) == P_USER_TERMINATED) if (parser_user_code(&reduced_parser_value2, parser_reduce_table[reduce_index].rule, statevalues, parser_reduce_table[reduce_index].n_states, context) == P_USER_TERMINATED)
@ -1408,16 +1133,6 @@ private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t star
return P_USER_TERMINATED; return P_USER_TERMINATED;
} }
reduced_parser_value = reduced_parser_value2; reduced_parser_value = reduced_parser_value2;
if (parser_reduce_table[reduce_index].n_states > 0u)
{
reduced_position = get_rule_position(statevalues, 0u, parser_reduce_table[reduce_index].n_states, false);
reduced_end_position = get_rule_position(statevalues, 0u, parser_reduce_table[reduce_index].n_states, true);
}
else
{
reduced_position = <%= @grammar.prefix %>position_t.INVALID;
reduced_end_position = <%= @grammar.prefix %>position_t.INVALID;
}
<% end %> <% end %>
reduced_rule_set = parser_reduce_table[reduce_index].rule_set; reduced_rule_set = parser_reduce_table[reduce_index].rule_set;
statevalues.length -= parser_reduce_table[reduce_index].n_states; statevalues.length -= parser_reduce_table[reduce_index].n_states;
@ -1437,20 +1152,14 @@ private size_t parse_from(<%= @grammar.prefix %>context_t * context, size_t star
public size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context) public size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context)
{ {
return parse_from(context, 0u, <%= @parser.rule_sets[@grammar.start_rules[0]].id %>u, null); return parse_from(context, 0u);
} }
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
public size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context) public size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context)
{ {
return parse_from(context, <%= i %>u, <%= @parser.rule_sets[start_rule].id %>u, null); return parse_from(context, <%= i %>u);
}
public size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %>context_t * context,
const(<%= @grammar.prefix %>token_t)[] follow_tokens)
{
return parse_from(context, <%= i %>u, <%= @parser.rule_sets[start_rule].id %>u, follow_tokens);
} }
<% end %> <% end %>
@ -1462,15 +1171,15 @@ public size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.p
* *
* @return Parse result value. * @return Parse result value.
*/ */
<% if @grammar.tree %> <% if @grammar.ast %>
public <%= @grammar.tree_prefix %><%= @grammar.start_rules[0] %><%= @grammar.tree_suffix %> <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context) public <%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context)
{ {
return <%= @grammar.tree_prefix %><%= @grammar.start_rules[0] %><%= @grammar.tree_suffix %>(context, context.parse_result); return cast(<%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> *)context.parse_result;
} }
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
public <%= @grammar.tree_prefix %><%= start_rule %><%= @grammar.tree_suffix %> <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context) public <%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context)
{ {
return <%= @grammar.tree_prefix %><%= start_rule %><%= @grammar.tree_suffix %>(context, context.parse_result); return cast(<%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> *)context.parse_result;
} }
<% end %> <% end %>
<% else %> <% else %>
@ -1499,58 +1208,6 @@ public <%= @grammar.prefix %>position_t <%= @grammar.prefix %>position(<%= @gram
return context.text_position; return context.text_position;
} }
/**
* Set the current text input position.
*
* This can be used to set the initial text position to something other than
* (1, 1) for a nested parse operation so that error positions reported by
* subsequent lexer/parser calls are relative to a larger enclosing document.
*
* @param context
* Lexer/parser context structure.
* @param position
* Text position to set.
*/
public void <%= @grammar.prefix %>set_position(<%= @grammar.prefix %>context_t * context, <%= @grammar.prefix %>position_t position)
{
context.text_position = position;
}
/**
* Get the current input text byte offset.
*
* @param context
* Lexer/parser context structure.
*
* @return Current input text byte offset (measured from the start of the
* input text passed to <%= @grammar.prefix %>context_new()).
*/
public size_t <%= @grammar.prefix %>input_index(<%= @grammar.prefix %>context_t * context)
{
return context.input_index;
}
/**
* Set the current input text byte offset.
*
* This moves the lexer's read cursor to the given byte offset (measured from
* the start of the input text passed to <%= @grammar.prefix %>context_new()).
* It can be used together with <%= @grammar.prefix %>set_position() to rewind
* the input part-way through a parse in order to re-read an earlier section of
* the input. The byte offset is not validated; the caller is responsible for
* providing an offset within the bounds of the input text. A value previously
* returned by <%= @grammar.prefix %>input_index() is a suitable argument.
*
* @param context
* Lexer/parser context structure.
* @param input_index
* Input text byte offset to set.
*/
public void <%= @grammar.prefix %>set_input_index(<%= @grammar.prefix %>context_t * context, size_t input_index)
{
context.input_index = input_index;
}
/** /**
* Get the user terminate code. * Get the user terminate code.
* *

View File

@ -8,9 +8,6 @@
#include <stdint.h> #include <stdint.h>
#include <stddef.h> #include <stddef.h>
<% if @cpp %>
#include <vector>
<% end %>
/************************************************************************** /**************************************************************************
* Public types * Public types
@ -48,10 +45,10 @@ typedef uint32_t <%= @grammar.prefix %>code_point_t;
*/ */
typedef struct typedef struct
{ {
/** Input text row (1-based). */ /** Input text row (0-based). */
uint32_t row; uint32_t row;
/** Input text column (1-based). */ /** Input text column (0-based). */
uint32_t col; uint32_t col;
} <%= @grammar.prefix %>position_t; } <%= @grammar.prefix %>position_t;
@ -61,7 +58,7 @@ typedef struct
/** User header code blocks. */ /** User header code blocks. */
<%= @grammar.code_blocks.fetch("header", "") %> <%= @grammar.code_blocks.fetch("header", "") %>
<% if @grammar.tree %> <% if @grammar.ast %>
/** Parser values type. */ /** Parser values type. */
typedef <%= @grammar.ptype %> <%= @grammar.prefix %>value_t; typedef <%= @grammar.ptype %> <%= @grammar.prefix %>value_t;
<% else %> <% else %>
@ -72,48 +69,49 @@ typedef union
<%= typestring %> v_<%= name %>; <%= typestring %> v_<%= name %>;
<% end %> <% end %>
} <%= @grammar.prefix %>value_t; } <%= @grammar.prefix %>value_t;
/** Parser value constructor(s). */
<% @grammar.ptypes.each do |name, typestring| %>
static inline <%= @grammar.prefix %>value_t <%= @grammar.prefix %>value<%= name == "default" ? "" : "_#{name}" %>(<%= typestring %> v)
{
return (<%= @grammar.prefix %>value_t){.v_<%= name %> = v};
}
<% end %> <% end %>
/** Parser value accessor(s). */ <% if @grammar.ast %>
<% @grammar.ptypes.each do |name, typestring| %> /** AST node types. @{ */
static inline <%= typestring %> <%= @grammar.prefix %>value_get<%= name == "default" ? "" : "_#{name}" %>(<%= @grammar.prefix %>value_t const * pvalue) typedef struct <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %>
{
return pvalue->v_<%= name %>;
}
<% end %>
<% end %>
<% if @grammar.tree %>
/** Tree node ID type (index into the context node arena). ID 0 is null. */
typedef uint32_t <%= @grammar.prefix %>node_id_t;
/**
* Tree node record.
*
* All tree nodes are stored contiguously in the context node arena. Child
* links are stored in a shared children array: a node's children
* occupy children[child_offset .. child_offset + n_fields]. Token payload
* fields (token, pvalue, and any user fields) are only meaningful when
* is_token is nonzero.
*/
typedef struct
{ {
<% # ASTNode fields must be present in the same order here. # %>
<%= @grammar.prefix %>position_t position; <%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position; <%= @grammar.prefix %>position_t end_position;
<%= @grammar.prefix %>node_id_t child_offset;
uint16_t n_fields; uint16_t n_fields;
uint8_t is_token; uint8_t is_token;
<%= @grammar.prefix %>token_t token; <%= @grammar.prefix %>token_t token;
<%= @grammar.prefix %>value_t pvalue; <%= @grammar.prefix %>value_t pvalue;
<%= @grammar.token_user_fields %> } <%= @grammar.ast_prefix %>Token<%= @grammar.ast_suffix %>;
} <%= @grammar.prefix %>node_data_t;
<% @parser.rule_sets.each do |name, rule_set| %>
<% next if name.start_with?("$") %>
<% next if rule_set.optional? %>
struct <%= name %>;
<% end %>
<% @parser.rule_sets.each do |name, rule_set| %>
<% next if name.start_with?("$") %>
<% next if rule_set.optional? %>
typedef struct <%= @grammar.ast_prefix %><%= name %><%= @grammar.ast_suffix %>
{
<% # ASTNode fields must be present in the same order here. # %>
<%= @grammar.prefix %>position_t position;
<%= @grammar.prefix %>position_t end_position;
uint16_t n_fields;
uint8_t is_token;
<% rule_set.ast_fields.each do |fields| %>
union
{
<% fields.each do |field_name, type| %>
struct <%= type %> * <%= field_name %>;
<% end %>
};
<% end %>
} <%= @grammar.ast_prefix %><%= name %><%= @grammar.ast_suffix %>;
<% end %>
/** @} */
<% end %> <% end %>
/** Lexed token information. */ /** Lexed token information. */
@ -135,19 +133,13 @@ typedef struct
<%= @grammar.prefix %>value_t pvalue; <%= @grammar.prefix %>value_t pvalue;
} <%= @grammar.prefix %>token_info_t; } <%= @grammar.prefix %>token_info_t;
typedef struct <%= @grammar.prefix %>context_s <%= @grammar.prefix %>context_t;
<% if @grammar.tree %>
<%= c_tree_handle_types_header %>
<% end %>
/** /**
* Lexer and parser context. * Lexer and parser context.
* *
* The user must allocate an instance of this structure and pass it to any * The user must allocate an instance of this structure and pass it to any
* public API function. * public API function.
*/ */
struct <%= @grammar.prefix %>context_s typedef struct
{ {
/* Lexer context data. */ /* Lexer context data. */
@ -169,26 +161,8 @@ struct <%= @grammar.prefix %>context_s
/* Parser context data. */ /* Parser context data. */
/** Parse result value. */ /** Parse result value. */
<% if @grammar.tree %> <% if @grammar.ast %>
<%= @grammar.prefix %>node_id_t parse_result; void * parse_result;
<% if @cpp %>
/** Tree node arena. Node ID 0 is reserved as the null node. */
std::vector<<%= @grammar.prefix %>node_data_t> <%= @grammar.prefix %>tree_nodes;
/** Shared tree child links. */
std::vector<<%= @grammar.prefix %>node_id_t> <%= @grammar.prefix %>tree_children;
<% else %>
/** Tree node arena. Node ID 0 is reserved as the null node. */
<%= @grammar.prefix %>node_data_t * <%= @grammar.prefix %>tree_nodes;
size_t <%= @grammar.prefix %>tree_nodes_length;
size_t <%= @grammar.prefix %>tree_nodes_capacity;
/** Shared tree child links. */
<%= @grammar.prefix %>node_id_t * <%= @grammar.prefix %>tree_children;
size_t <%= @grammar.prefix %>tree_children_length;
size_t <%= @grammar.prefix %>tree_children_capacity;
<% end %>
<% else %> <% else %>
<%= @grammar.prefix %>value_t parse_result; <%= @grammar.prefix %>value_t parse_result;
<% end %> <% end %>
@ -198,13 +172,7 @@ struct <%= @grammar.prefix %>context_s
/** User terminate code. */ /** User terminate code. */
size_t user_terminate_code; size_t user_terminate_code;
} <%= @grammar.prefix %>context_t;
<%= @grammar.context_user_fields %>
};
<% if @grammar.tree %>
<%= c_tree_types_header %>
<% end %>
/************************************************************************** /**************************************************************************
* Public data * Public data
@ -213,9 +181,7 @@ struct <%= @grammar.prefix %>context_s
/** Token names. */ /** Token names. */
extern const char * <%= @grammar.prefix %>token_names[]; extern const char * <%= @grammar.prefix %>token_names[];
<%= @grammar.prefix %>context_t * <%= @grammar.prefix %>context_new(uint8_t const * input, size_t input_length); void <%= @grammar.prefix %>context_init(<%= @grammar.prefix %>context_t * context, uint8_t const * input, size_t input_length);
void <%= @grammar.prefix %>context_delete(<%= @grammar.prefix %>context_t * context);
size_t <%= @grammar.prefix %>decode_code_point(uint8_t const * input, size_t input_length, size_t <%= @grammar.prefix %>decode_code_point(uint8_t const * input, size_t input_length,
<%= @grammar.prefix %>code_point_t * out_code_point, uint8_t * out_code_point_length); <%= @grammar.prefix %>code_point_t * out_code_point, uint8_t * out_code_point_length);
@ -225,14 +191,12 @@ size_t <%= @grammar.prefix %>lex(<%= @grammar.prefix %>context_t * context, <%=
size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context); size_t <%= @grammar.prefix %>parse(<%= @grammar.prefix %>context_t * context);
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context); size_t <%= @grammar.prefix %>parse_<%= start_rule %>(<%= @grammar.prefix %>context_t * context);
size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %>context_t * context,
<%= @grammar.prefix %>token_t const * follow_tokens, size_t n_follow_tokens);
<% end %> <% end %>
<% if @grammar.tree %> <% if @grammar.ast %>
<%= h_type(@grammar.start_rules[0]) %> <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context); <%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context);
<% @grammar.start_rules.each_with_index do |start_rule, i| %> <% @grammar.start_rules.each_with_index do |start_rule, i| %>
<%= h_type(start_rule) %> <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context); <%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> * <%= @grammar.prefix %>result_<%= start_rule %>(<%= @grammar.prefix %>context_t * context);
<% end %> <% end %>
<% else %> <% else %>
<%= start_rule_type[1] %> <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context); <%= start_rule_type[1] %> <%= @grammar.prefix %>result(<%= @grammar.prefix %>context_t * context);
@ -241,14 +205,15 @@ size_t <%= @grammar.prefix %>parse_inner_<%= start_rule %>(<%= @grammar.prefix %
<% end %> <% end %>
<% end %> <% end %>
<% if @grammar.ast %>
void <%= @grammar.prefix %>free_ast(<%= @grammar.ast_prefix %><%= @grammar.start_rules[0] %><%= @grammar.ast_suffix %> * ast);
<% @grammar.start_rules.each_with_index do |start_rule, i| %>
void <%= @grammar.prefix %>free_ast_<%= start_rule %>(<%= @grammar.ast_prefix %><%= start_rule %><%= @grammar.ast_suffix %> * ast);
<% end %>
<% end %>
<%= @grammar.prefix %>position_t <%= @grammar.prefix %>position(<%= @grammar.prefix %>context_t * context); <%= @grammar.prefix %>position_t <%= @grammar.prefix %>position(<%= @grammar.prefix %>context_t * context);
void <%= @grammar.prefix %>set_position(<%= @grammar.prefix %>context_t * context, <%= @grammar.prefix %>position_t position);
size_t <%= @grammar.prefix %>input_index(<%= @grammar.prefix %>context_t * context);
void <%= @grammar.prefix %>set_input_index(<%= @grammar.prefix %>context_t * context, size_t input_index);
size_t <%= @grammar.prefix %>user_terminate_code(<%= @grammar.prefix %>context_t * context); size_t <%= @grammar.prefix %>user_terminate_code(<%= @grammar.prefix %>context_t * context);
<%= @grammar.prefix %>token_t <%= @grammar.prefix %>token(<%= @grammar.prefix %>context_t * context); <%= @grammar.prefix %>token_t <%= @grammar.prefix %>token(<%= @grammar.prefix %>context_t * context);

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@ -7,46 +7,8 @@ if exists("b:current_syntax")
finish finish
endif endif
" Guess the language of the user code blocks from their contents so that the
" matching syntax file can be included below. b:propane_subtype may also be set
" before this file is sourced to select the language explicitly.
if !exists("b:propane_subtype") if !exists("b:propane_subtype")
" Rust markers. Each keyword requires the syntax that follows it in Rust so
" that a plain identifier of the same name in another language does not match
" (`int fn = 3;' in C, for example). Type names are only accepted within a
" `ptype' statement for the same reason.
let s:rust = '\<let\s\+\%(mut\s\+\)\?\w'
let s:rust .= '\|\<fn\s\+\w\+\s*('
let s:rust .= '\|&mut\>\|\<pub\s\+\w\|\<impl\s\+\w'
let s:rust .= '\|#\[\|\<use\s\+\%(std\|core\)::'
let s:rust .= '\|\<ptype\>[^;]*\<\%(isize\|usize\|i8\|i16\|i32\|i64\|i128'
let s:rust .= '\|u8\|u16\|u32\|u64\|u128\|f32\|f64\|String\)\>'
" D markers. These are spellings that have no valid C, C++, or Rust
" equivalent, so `import' is deliberately not among them: it is a D keyword
" but is also a C++20 module declaration.
let s:d = '\<foreach\%(_reverse\)\?\s*([^)]*;'
let s:d .= '\|\~=\|\<static\s\+if\s*(\|\<version\s*(\s*\w\+\s*)'
let s:d .= '\|\<scope\s*(\s*\%(exit\|failure\|success\)\s*)'
let s:d .= '\|\<\%(unittest\|mixin\|immutable\|__gshared\|invariant\)\>'
let s:d .= '\|\<alias\s\+\w\+\s*=\|\<enum\s\+\w\+\s*='
let s:d .= '\|@\%(property\|safe\|trusted\|system\|nogc\|disable\)\>'
let s:d .= '\|\<is\s\+null\>\|\<cast\s*(\s*\w\+\s*)'
let s:d .= '\|\<write\%(ln\|fln\|f\)\s*('
let s:d .= '\|\<\%(dchar\|dstring\|wstring\|cent\|ucent\)\>'
" A module import on its own is ambiguous between D and C++20, so only take
" it as D when nothing else in the file looks like C++.
let s:import = '\<import\s\+[A-Za-z_][A-Za-z0-9_.]*\s*;'
let s:cpp = '::\|\<template\s*<\|\<namespace\>\|\<nullptr\>\|#include\s*[<"]'
if search(s:rust, 'nw') > 0
let b:propane_subtype = "rust"
elseif search(s:d, 'nw') > 0
let b:propane_subtype = "d" let b:propane_subtype = "d"
elseif search(s:import, 'nw') > 0 && search(s:cpp, 'nw') == 0
let b:propane_subtype = "d"
else
let b:propane_subtype = "cpp"
endif
unlet s:rust s:d s:import s:cpp
endif endif
exe "syn include @propaneTarget syntax/".b:propane_subtype.".vim" exe "syn include @propaneTarget syntax/".b:propane_subtype.".vim"
@ -54,32 +16,18 @@ exe "syn include @propaneTarget syntax/".b:propane_subtype.".vim"
syn region propaneTarget matchgroup=propaneDelimiter start="<<" end=">>$" contains=@propaneTarget keepend syn region propaneTarget matchgroup=propaneDelimiter start="<<" end=">>$" contains=@propaneTarget keepend
syn match propaneComment "#.*" syn match propaneComment "#.*"
syn match propaneOperator "->"
syn match propaneFieldAlias ":[a-zA-Z0-9_]\+" contains=propaneFieldOperator syn match propaneFieldAlias ":[a-zA-Z0-9_]\+" contains=propaneFieldOperator
syn match propaneFieldOperator ":" contained syn match propaneFieldOperator ":" contained
syn match propaneOperator "?" syn match propaneOperator "?"
" The right-hand side of a rule (after '->' up to '<<' or ';') lists symbol syn keyword propaneKeyword ast ast_prefix ast_suffix drop module prefix ptype start token tokenid
" names that may coincide with propane keywords (e.g. 'start', 'token',
" 'tree'). Wrap it in a region that excludes keyword matches so those names
" are not highlighted as keywords. The '<<' is left unconsumed so the
" propaneTarget region can still match it.
syn region propaneRuleRhs matchgroup=propaneOperator start="->" end="\ze<<" end=";" contains=propaneFieldAlias,propaneRuleOperator,propaneComment keepend
syn match propaneRuleOperator "?" contained
" Keywords that introduce a user-defined name. The name is consumed by
" propaneName via nextgroup so a name matching a keyword (e.g. 'token start')
" is not highlighted as a keyword. These must be a match (not syn keyword)
" because a syn keyword always wins over a contained nextgroup match.
syn match propaneNameDecl "\<\%(tokenid\|token\|lex_fn\|module\|start\|tree_prefix\|tree_suffix\)\>" nextgroup=propaneName skipwhite
syn match propaneName "\<\h\w*\>" contained
syn match propaneKeyword "\<\%(context_user_fields\|drop\|free_token_node\|noline\|on_token_node\|prefix\|ptype\|token_user_fields\|tree\)\>"
syn region propaneRegex start="/" end="/" skip="\v\\\\|\\/" syn region propaneRegex start="/" end="/" skip="\v\\\\|\\/"
hi def link propaneComment Comment hi def link propaneComment Comment
hi def link propaneKeyword Keyword hi def link propaneKeyword Keyword
hi def link propaneNameDecl Keyword
hi def link propaneRegex String hi def link propaneRegex String
hi def link propaneOperator Operator hi def link propaneOperator Operator
hi def link propaneRuleOperator Operator
hi def link propaneFieldOperator Operator hi def link propaneFieldOperator Operator
hi def link propaneDelimiter Delimiter hi def link propaneDelimiter Delimiter
hi def link propaneFieldAlias Identifier hi def link propaneFieldAlias Identifier

View File

@ -33,7 +33,7 @@ class Propane
def run(input_file, output_file, log_file, options) def run(input_file, output_file, log_file, options)
begin begin
grammar = Grammar.new(File.read(input_file), input_file) grammar = Grammar.new(File.read(input_file))
generator = Generator.new(grammar, output_file, log_file, options) generator = Generator.new(grammar, output_file, log_file, options)
generator.generate generator.generate
rescue Error => e rescue Error => e

View File

@ -13,15 +13,8 @@ class Propane
@language = @language =
if output_file.end_with?(".d") if output_file.end_with?(".d")
"d" "d"
elsif output_file.end_with?(".c")
"c"
elsif output_file =~ %r{\.(cc|cpp|cxx)$}
@cpp = true
"c"
elsif output_file.end_with?(".rs")
"rust"
else else
raise Error.new("Could not determine target language from output file name (#{output_file})") "c"
end end
@options = options @options = options
process_grammar! process_grammar!
@ -33,37 +26,14 @@ class Propane
extensions += %w[h] extensions += %w[h]
end end
extensions.each do |extension| extensions.each do |extension|
template_language = @language == "rust" ? "rs" : @language template = Assets.get("parser.#{extension || @language}.erb")
template = Assets.get("parser.#{extension || template_language}.erb")
if extension if extension
output_file = @output_file.sub(%r{\.[a-z]+$}, ".#{extension}") output_file = @output_file.sub(%r{\.[a-z]+$}, ".#{extension}")
else else
output_file = @output_file output_file = @output_file
end end
erb = ERB.new(template, trim_mode: "<>") erb = ERB.new(template, trim_mode: "<>")
# Rust has no #line directive support. For a Rust target the directives result = erb.result(binding.clone)
# that the grammar embeds around user code blocks are replaced with
# comments naming the grammar file and line number the code came from,
# so that the origin of a section of user code can still be found by
# reading up from a compiler diagnostic pointing into the generated
# module.
user_code_origin = nil
result = erb.result(binding.clone).lines.each_with_index.map do |line, i|
if @language == "rust"
if md = line.match(/^#line (\d+) "([^"]*)"/)
user_code_origin = "#{md[2]} line #{md[1]}"
line.sub(/^#line \d+ "[^"]*"/, %[/* Begin user code from #{user_code_origin}. */])
elsif line == "#linereset\n"
%[/* End user code from #{user_code_origin}. */\n]
else
line
end
elsif line == "#linereset\n"
%[#line #{i + 2} "#{output_file}"\n]
else
line
end
end.join
File.open(output_file, "wb") do |fh| File.open(output_file, "wb") do |fh|
fh.write(result) fh.write(result)
end end
@ -204,8 +174,7 @@ class Propane
end end
end end
@grammar.rules << Rule.new(component, [], nil, ptypename, rule.line_number) @grammar.rules << Rule.new(component, [], nil, ptypename, rule.line_number)
optcode = @grammar.tree ? nil : "$$ = $1;\n" @grammar.rules << Rule.new(component, [c], "$$ = $1;\n", ptypename, rule.line_number)
@grammar.rules << Rule.new(component, [c], optcode, ptypename, rule.line_number)
optional_rules_added << component optional_rules_added << component
end end
end end
@ -294,88 +263,47 @@ class Propane
"context->user_terminate_code = (#{user_terminate_code}); return #{retval};" "context->user_terminate_code = (#{user_terminate_code}); return #{retval};"
when "d" when "d"
"context.user_terminate_code = (#{user_terminate_code}); return #{retval};" "context.user_terminate_code = (#{user_terminate_code}); return #{retval};"
when "rust"
"context.user_terminate_code = (#{user_terminate_code}); return #{retval};"
end
end
code = code.gsub(/\$\{context\.(\w+)\}/) do |match|
fieldname = $1
case @language
when "c"
"context->#{fieldname}"
when "d"
"context.#{fieldname}"
when "rust"
"context.#{fieldname}"
end
end
code = code.gsub(/\$\{token\.(\w+)\}/) do |match|
fieldname = $1
case @language
when "c"
"token_tree_node->#{fieldname}"
when "d"
"token_tree_node.#{fieldname}"
when "rust"
"token_tree_node.#{fieldname}"
end end
end end
if parser if parser
code = code.gsub(/\$\$/) do |match| code = code.gsub(/\$\$/) do |match|
if @grammar.tree
typename = "#{@grammar.tree_prefix}#{rule.name}#{@grammar.tree_suffix}"
case @language
when "c"
tree_handle(typename, "_node_id")
when "d"
tree_handle(typename, "_node_id")
when "rust"
tree_handle(typename, "_node_id")
end
else
case @language case @language
when "c" when "c"
"_pvalue->v_#{rule.ptypename}" "_pvalue->v_#{rule.ptypename}"
when "d" when "d"
"_pvalue.v_#{rule.ptypename}" "_pvalue.v_#{rule.ptypename}"
when "rust"
"(*_pvalue.v_#{rule.ptypename}_mut())"
end
end end
end end
code = code.gsub(/\$(\d+)/) do |match| code = code.gsub(/\$(\d+)/) do |match|
parser_component_reference(rule, $1.to_i)
end
code = code.gsub(/\$\{(\$|\d+)\.position\}/) do |match|
index = $1.to_i index = $1.to_i
"get_rule_position(statevalues, #{index}, n_states, false)" case @language
when "c"
"state_values_stack_index(statevalues, -1 - (int)n_states + #{index})->pvalue.v_#{rule.components[index - 1].ptypename}"
when "d"
"statevalues[$-1-n_states+#{index}].pvalue.v_#{rule.components[index - 1].ptypename}"
end end
code = code.gsub(/\$\{(\$|\d+)\.end_position\}/) do |match|
index = $1.to_i
"get_rule_position(statevalues, #{index}, n_states, true)"
end end
code = code.gsub(/\$\{(\w+)\}/) do |match| code = code.gsub(/\$\{(\w+)\}/) do |match|
aliasname = $1 aliasname = $1
if index = rule.aliases[aliasname] if index = rule.aliases[aliasname]
# Field aliases are just a named reference to a positional rule case @language
# component, so reuse the same expansion as `$1', `$2', etc. Note when "c"
# that rule.aliases stores a 0-based component index, so add 1 to "state_values_stack_index(statevalues, -(int)n_states + #{index})->pvalue.v_#{rule.components[index].ptypename}"
# convert it to the 1-based index used for positional references. when "d"
parser_component_reference(rule, index + 1) "statevalues[$-n_states+#{index}].pvalue.v_#{rule.components[index].ptypename}"
end
else else
raise Error.new("Field alias '#{aliasname}' not found") raise Error.new("Field alias '#{aliasname}' not found")
end end
end end
else else
code = code.gsub(/\$\$/) do |match| code = code.gsub(/\$\$/) do |match|
if @grammar.tree if @grammar.ast
case @language case @language
when "c" when "c"
"out_token_info->pvalue" "out_token_info->pvalue"
when "d" when "d"
"out_token_info.pvalue" "out_token_info.pvalue"
when "rust"
"out_token_info.pvalue"
end end
else else
case @language case @language
@ -383,31 +311,9 @@ class Propane
"out_token_info->pvalue.v_#{pattern.ptypename}" "out_token_info->pvalue.v_#{pattern.ptypename}"
when "d" when "d"
"out_token_info.pvalue.v_#{pattern.ptypename}" "out_token_info.pvalue.v_#{pattern.ptypename}"
when "rust"
"(*out_token_info.pvalue.v_#{pattern.ptypename}_mut())"
end end
end end
end end
code = code.gsub(/\$\{position\}/) do |match|
case @language
when "c"
"out_token_info->position"
when "d"
"out_token_info.position"
when "rust"
"out_token_info.position"
end
end
code = code.gsub(/\$\{end_position\}/) do |match|
case @language
when "c"
"out_token_info->end_position"
when "d"
"out_token_info.end_position"
when "rust"
"out_token_info.end_position"
end
end
code = code.gsub(/\$mode\(([a-zA-Z_][a-zA-Z_0-9]*)\)/) do |match| code = code.gsub(/\$mode\(([a-zA-Z_][a-zA-Z_0-9]*)\)/) do |match|
mode_name = $1 mode_name = $1
mode_id = @lexer.mode_id(mode_name) mode_id = @lexer.mode_id(mode_name)
@ -419,411 +325,12 @@ class Propane
"context->mode = #{mode_id}u" "context->mode = #{mode_id}u"
when "d" when "d"
"context.mode = #{mode_id}u" "context.mode = #{mode_id}u"
when "rust"
"context.mode = #{mode_id}"
end end
end end
end end
code code
end end
# Expand a positional reference to a parser rule component.
#
# This is used to expand `$1', `$2', etc. as well as field aliases (which
# are just named references to a positional rule component).
#
# @param rule [Rule]
# The Rule containing the user code.
# @param index [Integer]
# 1-based index of the rule component to reference.
#
# @return [String]
# Expanded rule component reference.
def parser_component_reference(rule, index)
component = rule.components[index - 1]
if @grammar.tree
# In tree mode a component reference yields a handle to that
# component's tree node. An optional component propagates its target
# node (or null), so use the optional target's node type.
if component.is_a?(RuleSet) && component.optional?
component = component.option_target
end
node_name = component.is_a?(Token) ? "Token" : component.name
typename = "#{@grammar.tree_prefix}#{node_name}#{@grammar.tree_suffix}"
case @language
when "c"
tree_handle(typename, "state_values_stack_index(statevalues, -1 - (int)n_states + #{index})->node_id")
when "d"
tree_handle(typename, "statevalues[$-1-n_states+#{index}].node_id")
when "rust"
tree_handle(typename, "statevalues[statevalues.len() - 1 - n_states + #{index}].node_id")
end
else
case @language
when "c"
"state_values_stack_index(statevalues, -1 - (int)n_states + #{index})->pvalue.v_#{component.ptypename}"
when "d"
"statevalues[$-1-n_states+#{index}].pvalue.v_#{component.ptypename}"
when "rust"
"statevalues[statevalues.len() - 1 - n_states + #{index}].pvalue.get_v_#{component.ptypename}()"
end
end
end
# Construct a tree node handle expression for the target language.
#
# A handle is a small value pairing the parser context with a node ID
# (an index into the context's node arena). All handle types share this
# layout; the distinct types exist for documentation and, in C, to drive
# the tree walk macro's type threading.
#
# @param typename [String]
# Handle type name.
# @param id_expr [String]
# Expression yielding the node ID.
# @param parenthesize [Boolean]
# Whether to parenthesize the expression. Parentheses are required where
# the expression is substituted into a user code block, since the
# expression could be followed there by a field access or appear in a
# position where a bare Rust struct literal is not accepted. They are
# unnecessary where the expression stands alone, and Rust warns about
# them there, so this can be disabled for those uses.
#
# @return [String]
# Handle constructor expression.
def tree_handle(typename, id_expr, parenthesize = true)
if @cpp
"(#{typename}{context, #{id_expr}})"
elsif @language == "c"
"((#{typename}){context, #{id_expr}})"
elsif @language == "rust"
expr = "#{typename} { context, id: #{id_expr} }"
parenthesize ? "(#{expr})" : expr
else
"#{typename}(context, #{id_expr})"
end
end
# Get the list of non-optional, non-internal rule sets that get a tree node
# handle type generated for them.
#
# @return [Array<Propane::RuleSet>]
# Rule sets with generated tree node handle types.
def tree_node_rule_sets
@parser.rule_sets.reject do |name, rule_set|
name.start_with?("$") || rule_set.optional?
end.map {|name, rule_set| rule_set}
end
# Maximum number of chained fields supported by a single C tree walk macro
# invocation. Deeper navigation can be expressed by nesting walk calls.
C_TREE_WALK_MAX = 16
# Get the tree node handle type name for a node name.
#
# @param name [String]
# Rule set name, or "Token".
#
# @return [String]
# Handle type name.
def h_type(name)
"#{@grammar.tree_prefix}#{name}#{@grammar.tree_suffix}"
end
# Get the list of all tree node handle type names (Token plus rule sets).
#
# @return [Array<String>]
# Handle type names.
def tree_handle_types
[h_type("Token")] + tree_node_rule_sets.map {|rs| h_type(rs.name)}
end
# Enumerate the navigation fields of a rule set's tree node.
#
# @yield [rtype, field_name, child_type, slot]
# Handle type name, field accessor name, child handle type, and child
# slot index.
def each_tree_field(rule_set)
rtype = h_type(rule_set.name)
rule_set.tree_fields.each_with_index do |fields, slot|
fields.each do |field_name, child_type|
yield rtype, field_name, child_type, slot
end
end
end
# Generate the tree node handle type declarations for the header.
#
# These are emitted before the context structure definition so that a
# context_user_fields block can declare a field of a handle type.
#
# @return [String]
# Handle type declarations.
def c_tree_handle_types_header
@cpp ? cpp_tree_handle_types_header : c_only_tree_handle_types_header
end
# Generate the remainder of the tree node section for the header.
#
# This is emitted after the context structure definition since it
# dereferences the context and so requires the complete type.
#
# @return [String]
# Accessors, macros, and out-of-line handle method definitions.
def c_tree_types_header
@cpp ? cpp_tree_types_header : c_only_tree_types_header
end
# Generate the C (non-C++) tree node handle type section for the header.
def c_only_tree_handle_types_header
p = @grammar.prefix
out = []
out << "/** Tree node handle types. @{ */"
tree_handle_types.each do |t|
out << "typedef struct { #{p}context_t * __context; #{p}node_id_t __id; } #{t};"
end
out << ""
out.join("\n")
end
def c_only_tree_types_header
out = []
out << c_common_accessors_header
out << "/** @} */"
out.join("\n")
end
# Generate the C-style (function + macro) tree node accessors shared by the
# C and C++ headers. In C++ these are provided in addition to the handle
# methods so that C-style code (and the tree walk macros) also works.
def c_common_accessors_header
p = @grammar.prefix
out = []
out << "/** Generic tree node accessors (usable on any handle type). */"
out << "#define #{p}node_valid(h) ((h).__id != 0u)"
out << "#define #{p}node_id(h) ((h).__id)"
out << "#define #{p}node_data(h) (&(h).__context->#{p}tree_nodes[(h).__id])"
out << "#define #{p}node_position(h) ((h).__context->#{p}tree_nodes[(h).__id].position)"
out << "#define #{p}node_end_position(h) ((h).__context->#{p}tree_nodes[(h).__id].end_position)"
out << "#define #{p}node_n_fields(h) ((h).__id ? (h).__context->#{p}tree_nodes[(h).__id].n_fields : (uint16_t)0u)"
out << ""
out << "/** Tree node field accessor functions. */"
out << "#{p}token_t #{p}#{h_type("Token")}_token(#{h_type("Token")} node);"
out << "#{p}value_t #{p}#{h_type("Token")}_pvalue(#{h_type("Token")} node);"
tree_node_rule_sets.each do |rule_set|
each_tree_field(rule_set) do |rtype, field_name, child_type, slot|
out << "#{child_type} #{p}#{rtype}_#{field_name}(#{rtype} node);"
end
end
out << ""
out << c_tree_walk_macros
out.join("\n")
end
# Generate the C tree walk macro machinery.
def c_tree_walk_macros
p = @grammar.prefix
max = C_TREE_WALK_MAX
out = []
out << "/* Tree walk macros: p_tree_walk_<Type>(handle, field, ...). */"
out << "#define #{p}CAT_(a, b) a##b"
out << "#define #{p}CAT(a, b) #{p}CAT_(a, b)"
out << "#define #{p}TA(t, f) #{p}CAT(#{p}CAT(#{p}CAT(#{p}TYPEAFTER_, t), _), f)"
out << "#define #{p}ACC(t, f) #{p}CAT(#{p}CAT(#{p}CAT(#{p}, t), _), f)"
argn = (1..max).map {|i| "_#{i}"}.join(", ")
rseq = (0..max).to_a.reverse.join(", ")
out << "#define #{p}ARG_N(#{argn}, N, ...) N"
out << "#define #{p}NARG(...) #{p}ARG_N(__VA_ARGS__, #{rseq})"
(1..max).each do |n|
fparams = (1..n).map {|k| "f#{k}"}.join(", ")
call = "h"
(1..n).each do |k|
texpr = "R"
(1...k).each {|j| texpr = "#{p}TA(#{texpr}, f#{j})"}
call = "#{p}ACC(#{texpr}, f#{k})(#{call})"
end
out << "#define #{p}tree_walk_#{n}(R, h, #{fparams}) #{call}"
end
out << "#define #{p}tree_walk_dispatch(R, h, ...) #{p}CAT(#{p}tree_walk_, #{p}NARG(__VA_ARGS__))(R, h, __VA_ARGS__)"
# Type transition map (navigation fields only).
tree_node_rule_sets.each do |rule_set|
each_tree_field(rule_set) do |rtype, field_name, child_type, slot|
out << "#define #{p}TYPEAFTER_#{rtype}_#{field_name} #{child_type}"
end
end
# Per-handle-type walk entry points.
tree_handle_types.each do |t|
out << "#define #{p}tree_walk_#{t}(...) #{p}tree_walk_dispatch(#{t}, __VA_ARGS__)"
end
out.join("\n")
end
# Generate the C tree node accessor function definitions for the source.
#
# @return [String]
# Accessor function definitions.
def c_tree_accessor_defs
p = @grammar.prefix
tt = h_type("Token")
out = []
out << "#{p}token_t #{p}#{tt}_token(#{tt} node)"
out << "{"
out << " return node.__context->#{p}tree_nodes[node.__id].token;"
out << "}"
out << ""
out << "#{p}value_t #{p}#{tt}_pvalue(#{tt} node)"
out << "{"
out << " return node.__context->#{p}tree_nodes[node.__id].pvalue;"
out << "}"
tree_node_rule_sets.each do |rule_set|
each_tree_field(rule_set) do |rtype, field_name, child_type, slot|
out << ""
out << "#{child_type} #{p}#{rtype}_#{field_name}(#{rtype} node)"
out << "{"
out << " #{child_type} result;"
out << " result.__context = node.__context;"
out << " if (node.__id == 0u)"
out << " {"
out << " result.__id = 0u;"
out << " return result;"
out << " }"
out << " result.__id = node.__context->#{p}tree_children[node.__context->#{p}tree_nodes[node.__id].child_offset + #{slot}u];"
out << " return result;"
out << "}"
end
end
out.join("\n")
end
# Generate the C++ tree node handle class declarations for the header.
# Only valid() and node_id() are defined inline; every other method
# dereferences the context, which is still an incomplete type here, so
# those are declared and defined out of line once the context is
# complete.
def cpp_tree_handle_types_header
p = @grammar.prefix
out = []
out << "/** Tree node handle types. @{ */"
tree_handle_types.each {|t| out << "struct #{t};"}
out << ""
tt = h_type("Token")
out << "struct #{tt}"
out << "{"
out << " #{p}context_t * __context;"
out << " #{p}node_id_t __id;"
out << " bool valid() const { return __id != 0u; }"
out << " #{p}node_id_t node_id() const { return __id; }"
out << " #{p}node_data_t * data() const;"
out << " #{p}position_t position() const;"
out << " #{p}position_t end_position() const;"
out << " uint16_t n_fields() const;"
out << " #{p}token_t token() const;"
out << " #{p}value_t pvalue() const;"
out << "};"
out << ""
tree_node_rule_sets.each do |rule_set|
rtype = h_type(rule_set.name)
out << "struct #{rtype}"
out << "{"
out << " #{p}context_t * __context;"
out << " #{p}node_id_t __id;"
out << " bool valid() const { return __id != 0u; }"
out << " #{p}node_id_t node_id() const { return __id; }"
out << " #{p}node_data_t * data() const;"
out << " #{p}position_t position() const;"
out << " #{p}position_t end_position() const;"
out << " uint16_t n_fields() const;"
each_tree_field(rule_set) do |rt, field_name, child_type, slot|
out << " #{child_type} #{field_name}() const;"
end
out << "};"
out << ""
end
out.join("\n")
end
# Generate the out-of-line C++ handle method definitions plus the C-style
# accessors. Emitted after the context structure definition.
def cpp_tree_types_header
p = @grammar.prefix
out = []
# Common node methods, now that the context type is complete.
([h_type("Token")] + tree_node_rule_sets.map {|rs| h_type(rs.name)}).each do |ht|
out << "inline #{p}node_data_t * #{ht}::data() const { return &__context->#{p}tree_nodes[__id]; }"
out << "inline #{p}position_t #{ht}::position() const { return __context->#{p}tree_nodes[__id].position; }"
out << "inline #{p}position_t #{ht}::end_position() const { return __context->#{p}tree_nodes[__id].end_position; }"
out << "inline uint16_t #{ht}::n_fields() const { return __id ? __context->#{p}tree_nodes[__id].n_fields : (uint16_t)0u; }"
end
tt = h_type("Token")
out << "inline #{p}token_t #{tt}::token() const { return __context->#{p}tree_nodes[__id].token; }"
out << "inline #{p}value_t #{tt}::pvalue() const { return __context->#{p}tree_nodes[__id].pvalue; }"
out << ""
# Out-of-line navigation method bodies (all handle types now complete).
tree_node_rule_sets.each do |rule_set|
rtype = h_type(rule_set.name)
each_tree_field(rule_set) do |rt, field_name, child_type, slot|
out << "inline #{child_type} #{rtype}::#{field_name}() const"
out << "{"
out << " if (__id == 0u)"
out << " {"
out << " return #{child_type}{__context, 0u};"
out << " }"
out << " return #{child_type}{__context, __context->#{p}tree_children[__context->#{p}tree_nodes[__id].child_offset + #{slot}u]};"
out << "}"
end
end
out << ""
out << "/*"
out << " * C-style function and macro accessors, provided in addition to the handle"
out << " * methods above so that C-style code and the tree walk macros also work."
out << " */"
out << c_common_accessors_header
out << "/** @} */"
out.join("\n")
end
# Rust keywords that must be escaped as raw identifiers when used as a
# generated identifier (e.g. a field alias named `type`).
RUST_KEYWORDS = %w[
as break const continue dyn else enum extern false fn for if impl in let
loop match mod move mut pub ref return static struct trait true type
unsafe use where while async await abstract become box do final macro
override priv typeof unsized virtual yield try gen
]
# Escape a name as a Rust raw identifier if it is a reserved keyword.
#
# @param name [String]
# Identifier name.
#
# @return [String]
# Name, escaped as a raw identifier if necessary.
def rust_ident(name)
RUST_KEYWORDS.include?(name) ? "r##{name}" : name
end
# Map a ptype type string to a valid Rust type.
#
# The default ptype is a C "void *"; for Rust with no declared ptype we use
# the unit type instead.
#
# @param typestring [String]
# ptype type string.
#
# @return [String]
# Rust type string.
def rust_ptype(typestring)
typestring == "void *" ? "()" : typestring
end
# Get the lex function to use.
#
# @return [String]
# Lex function to use.
def lex_fn
@grammar.lex_fn || "#{@grammar.prefix}lex"
end
# Get the parser value type for the start rule. # Get the parser value type for the start rule.
# #
# @return [Array<String>] # @return [Array<String>]
@ -849,8 +356,6 @@ class Propane
"uint8_t" "uint8_t"
when "d" when "d"
"ubyte" "ubyte"
when "rust"
"u8"
end end
elsif max <= 0xFFFF elsif max <= 0xFFFF
case @language case @language
@ -858,15 +363,11 @@ class Propane
"uint16_t" "uint16_t"
when "d" when "d"
"ushort" "ushort"
when "rust"
"u16"
end end
else else
case @language case @language
when "c" when "c"
"uint32_t" "uint32_t"
when "rust"
"u32"
else else
"uint" "uint"
end end

View File

@ -5,11 +5,9 @@ class Propane
# Reserve identifiers beginning with a double-underscore for internal use. # Reserve identifiers beginning with a double-underscore for internal use.
IDENTIFIER_REGEX = /(?:[a-zA-Z]|_[a-zA-Z0-9])[a-zA-Z_0-9]*/ IDENTIFIER_REGEX = /(?:[a-zA-Z]|_[a-zA-Z0-9])[a-zA-Z_0-9]*/
attr_reader :context_user_fields attr_reader :ast
attr_reader :lex_fn attr_reader :ast_prefix
attr_reader :tree attr_reader :ast_suffix
attr_reader :tree_prefix
attr_reader :tree_suffix
attr_reader :free_token_node attr_reader :free_token_node
attr_reader :modulename attr_reader :modulename
attr_reader :patterns attr_reader :patterns
@ -19,11 +17,8 @@ class Propane
attr_reader :code_blocks attr_reader :code_blocks
attr_reader :ptypes attr_reader :ptypes
attr_reader :prefix attr_reader :prefix
attr_reader :on_token_node
attr_reader :token_user_fields
def initialize(input, filename) def initialize(input)
@filename = filename
@patterns = [] @patterns = []
@start_rules = [] @start_rules = []
@tokens = [] @tokens = []
@ -35,13 +30,10 @@ class Propane
@input = input.gsub("\r\n", "\n") @input = input.gsub("\r\n", "\n")
@ptypes = {"default" => "void *"} @ptypes = {"default" => "void *"}
@prefix = "p_" @prefix = "p_"
@tree = false @ast = false
@tree_prefix = "" @ast_prefix = ""
@tree_suffix = "" @ast_suffix = ""
@free_token_node = "" @free_token_node = nil
@context_user_fields = nil
@on_token_node = ""
@token_user_fields = nil
parse_grammar! parse_grammar!
@start_rules << "Start" if @start_rules.empty? @start_rules << "Start" if @start_rules.empty?
end end
@ -58,10 +50,6 @@ class Propane
@tokens.size + 1 @tokens.size + 1
end end
def parser_user_code_used?
@rules.any? {|r| r.code}
end
private private
def parse_grammar! def parse_grammar!
@ -74,15 +62,11 @@ class Propane
if parse_white_space! if parse_white_space!
elsif parse_comment_line! elsif parse_comment_line!
elsif @modeline.nil? && parse_mode_label! elsif @modeline.nil? && parse_mode_label!
elsif parse_context_user_fields_statement! elsif parse_ast_statement!
elsif parse_lex_fn! elsif parse_ast_prefix_statement!
elsif parse_tree_statement! elsif parse_ast_suffix_statement!
elsif parse_tree_prefix_statement!
elsif parse_tree_suffix_statement!
elsif parse_free_token_node_statement! elsif parse_free_token_node_statement!
elsif parse_module_statement! elsif parse_module_statement!
elsif parse_on_token_node_statement!
elsif parse_token_user_fields_statement!
elsif parse_ptype_statement! elsif parse_ptype_statement!
elsif parse_pattern_statement! elsif parse_pattern_statement!
elsif parse_start_statement! elsif parse_start_statement!
@ -92,7 +76,6 @@ class Propane
elsif parse_rule_statement! elsif parse_rule_statement!
elsif parse_code_block_statement! elsif parse_code_block_statement!
elsif parse_prefix_statement! elsif parse_prefix_statement!
elsif parse_noline_statement!
else else
if @input.size > 25 if @input.size > 25
@input = @input.slice(0..20) + "..." @input = @input.slice(0..20) + "..."
@ -115,37 +98,27 @@ class Propane
consume!(/#.*\n/) consume!(/#.*\n/)
end end
def parse_context_user_fields_statement! def parse_ast_statement!
if md = consume!(/context_user_fields\b\s*/) if consume!(/ast\s*;/)
unless code = parse_code_block! @ast = true
raise Error.new("Line #{@line_number}: expected code block")
end
@context_user_fields ||= ""
@context_user_fields += code
end end
end end
def parse_lex_fn! def parse_ast_prefix_statement!
if md = consume!(/lex_fn\b\s*(\w+)\s*;/) if md = consume!(/ast_prefix\s+(\w+)\s*;/)
@lex_fn = md[1] @ast_prefix = md[1]
end end
end end
def parse_tree_statement! def parse_ast_suffix_statement!
if consume!(/tree\s*;/) if md = consume!(/ast_suffix\s+(\w+)\s*;/)
@tree = true @ast_suffix = md[1]
end end
end end
def parse_tree_prefix_statement! def parse_free_token_node_statement!
if md = consume!(/tree_prefix\s+(\w+)\s*;/) if md = consume!(/free_token_node\s+(\w+)\s*;/)
@tree_prefix = md[1] @free_token_node = md[1]
end
end
def parse_tree_suffix_statement!
if md = consume!(/tree_suffix\s+(\w+)\s*;/)
@tree_suffix = md[1]
end end
end end
@ -159,40 +132,12 @@ class Propane
end end
end end
def parse_on_token_node_statement!
if md = consume!(/on_token_node\b\s*/)
unless code = parse_code_block!
raise Error.new("Line #{@line_number}: expected code block")
end
@on_token_node += code
end
end
def parse_token_user_fields_statement!
if md = consume!(/token_user_fields\b\s*/)
unless code = parse_code_block!
raise Error.new("Line #{@line_number}: expected code block")
end
@token_user_fields ||= ""
@token_user_fields += code
end
end
def parse_free_token_node_statement!
if md = consume!(/free_token_node\b\s*/)
unless code = parse_code_block!
raise Error.new("Line #{@line_number}: expected code block")
end
@free_token_node += code
end
end
def parse_ptype_statement! def parse_ptype_statement!
if consume!(/ptype\s+/) if consume!(/ptype\s+/)
name = "default" name = "default"
if md = consume!(/(#{IDENTIFIER_REGEX})\s*=\s*/) if md = consume!(/(#{IDENTIFIER_REGEX})\s*=\s*/)
if @tree if @ast
raise Error.new("Multiple ptypes are unsupported in tree mode") raise Error.new("Multiple ptypes are unsupported in AST mode")
end end
name = md[1] name = md[1]
end end
@ -206,8 +151,8 @@ class Propane
md = consume!(/(#{IDENTIFIER_REGEX})\s*/, "expected token name") md = consume!(/(#{IDENTIFIER_REGEX})\s*/, "expected token name")
name = md[1] name = md[1]
if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/) if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/)
if @tree if @ast
raise Error.new("Multiple ptypes are unsupported in tree mode") raise Error.new("Multiple ptypes are unsupported in AST mode")
end end
ptypename = md[1] ptypename = md[1]
end end
@ -230,8 +175,8 @@ class Propane
md = consume!(/(#{IDENTIFIER_REGEX})\s*/, "expected token name") md = consume!(/(#{IDENTIFIER_REGEX})\s*/, "expected token name")
name = md[1] name = md[1]
if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/) if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/)
if @tree if @ast
raise Error.new("Multiple ptypes are unsupported in tree mode") raise Error.new("Multiple ptypes are unsupported in AST mode")
end end
ptypename = md[1] ptypename = md[1]
end end
@ -250,10 +195,8 @@ class Propane
raise Error.new("Line #{@line_number}: expected pattern to follow `drop'") raise Error.new("Line #{@line_number}: expected pattern to follow `drop'")
end end
consume!(/\s+/) consume!(/\s+/)
unless code = parse_code_block! consume!(/;/, "expected `;'")
consume!(/;/, "expected `;' or code block") @patterns << Pattern.new(pattern: pattern, line_number: @line_number, modes: get_modes_from_modeline)
end
@patterns << Pattern.new(pattern: pattern, line_number: @line_number, code: code, modes: get_modes_from_modeline)
@modeline = nil @modeline = nil
true true
end end
@ -262,14 +205,18 @@ class Propane
def parse_rule_statement! def parse_rule_statement!
if md = consume!(/(#{IDENTIFIER_REGEX})\s*(?:\((#{IDENTIFIER_REGEX})\))?\s*->\s*/) if md = consume!(/(#{IDENTIFIER_REGEX})\s*(?:\((#{IDENTIFIER_REGEX})\))?\s*->\s*/)
rule_name, ptypename = *md[1, 2] rule_name, ptypename = *md[1, 2]
if @tree && ptypename if @ast && ptypename
raise Error.new("Multiple ptypes are unsupported in tree mode") raise Error.new("Multiple ptypes are unsupported in AST mode")
end end
md = consume!(/((?:#{IDENTIFIER_REGEX}\??(?::#{IDENTIFIER_REGEX})?\s*)*)\s*/, "expected rule component list") md = consume!(/((?:#{IDENTIFIER_REGEX}\??(?::#{IDENTIFIER_REGEX})?\s*)*)\s*/, "expected rule component list")
components = md[1].strip.split(/\s+/) components = md[1].strip.split(/\s+/)
if @ast
consume!(/;/, "expected `;'")
else
unless code = parse_code_block! unless code = parse_code_block!
consume!(/;/, "expected `;' or code block") consume!(/;/, "expected `;' or code block")
end end
end
@rules << Rule.new(rule_name, components, code, ptypename, @line_number) @rules << Rule.new(rule_name, components, code, ptypename, @line_number)
@modeline = nil @modeline = nil
true true
@ -280,8 +227,8 @@ class Propane
if pattern = parse_pattern! if pattern = parse_pattern!
consume!(/\s+/) consume!(/\s+/)
if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/) if md = consume!(/\((#{IDENTIFIER_REGEX})\)\s*/)
if @tree if @ast
raise Error.new("Multiple ptypes are unsupported in tree mode") raise Error.new("Multiple ptypes are unsupported in AST mode")
end end
ptypename = md[1] ptypename = md[1]
end end
@ -306,14 +253,8 @@ class Propane
def parse_code_block_statement! def parse_code_block_statement!
if md = consume!(/<<([a-z]*)(.*?)>>\n/m) if md = consume!(/<<([a-z]*)(.*?)>>\n/m)
name, code = md[1..2] name, code = md[1..2]
code = code.chomp code.sub!(/\A\n/, "")
unless @noline code += "\n" unless code.end_with?("\n")
if code.start_with?("\n")
code = %[#line #{@line_number + 1} "#{@filename}"#{code}\n#linereset\n]
else
code = %[#line #{@line_number} "#{@filename}"\n#{code}\n#linereset\n]
end
end
if @code_blocks[name] if @code_blocks[name]
@code_blocks[name] += code @code_blocks[name] += code
else else
@ -331,13 +272,6 @@ class Propane
end end
end end
def parse_noline_statement!
if md = consume!(/noline\s*;/)
@noline = true
true
end
end
def parse_pattern! def parse_pattern!
if md = consume!(%r{/}) if md = consume!(%r{/})
pattern = "" pattern = ""
@ -351,8 +285,6 @@ class Propane
end end
elsif md = consume!(%r{(.)}) elsif md = consume!(%r{(.)})
pattern += md[1] pattern += md[1]
elsif @input == "" || @input.start_with?("\n")
raise Error.new("Line #{@line_number}: Unterminated pattern; expected `/`")
end end
end end
pattern pattern
@ -361,14 +293,9 @@ class Propane
def parse_code_block! def parse_code_block!
if md = consume!(/<<(.*?)>>\n/m) if md = consume!(/<<(.*?)>>\n/m)
code = md[1].chomp code = md[1]
unless @noline code.sub!(/\A\n/, "")
if code.start_with?("\n") code += "\n" unless code.end_with?("\n")
code = %[#line #{@line_number + 1} "#{@filename}"#{code}\n#linereset\n]
else
code = %[#line #{@line_number} "#{@filename}"\n#{code}\n#linereset\n]
end
end
code code
end end
end end

View File

@ -36,7 +36,7 @@ class Propane
# @return [Array<Integer>] # @return [Array<Integer>]
# Map this rule's components to their positions in the parent RuleSet's # Map this rule's components to their positions in the parent RuleSet's
# node field pointer array. This is used for tree construction. # node field pointer array. This is used for AST construction.
attr_accessor :rule_set_node_field_index_map attr_accessor :rule_set_node_field_index_map
# Construct a Rule. # Construct a Rule.

View File

@ -4,8 +4,8 @@ class Propane
class RuleSet class RuleSet
# @return [Array<Hash>] # @return [Array<Hash>]
# tree fields. # AST fields.
attr_reader :tree_fields attr_reader :ast_fields
# @return [Integer] # @return [Integer]
# ID of the RuleSet. # ID of the RuleSet.
@ -100,28 +100,28 @@ class Propane
# Finalize a RuleSet after adding all Rules to it. # Finalize a RuleSet after adding all Rules to it.
def finalize(grammar) def finalize(grammar)
if grammar.tree if grammar.ast
build_tree_fields(grammar) build_ast_fields(grammar)
end end
end end
private private
# Build the set of tree fields for this RuleSet. # Build the set of AST fields for this RuleSet.
# #
# This is an Array of Hashes. Each entry in the Array corresponds to a # This is an Array of Hashes. Each entry in the Array corresponds to a
# field location in the tree node. The entry is a Hash. It could have one or # field location in the AST node. The entry is a Hash. It could have one or
# two keys. It will always have the field name with a positional suffix as # two keys. It will always have the field name with a positional suffix as
# a key. It may also have the field name without the positional suffix if # a key. It may also have the field name without the positional suffix if
# that field only exists in one position across all Rules in the RuleSet. # that field only exists in one position across all Rules in the RuleSet.
# #
# @return [void] # @return [void]
def build_tree_fields(grammar) def build_ast_fields(grammar)
field_tree_node_indexes = {} field_ast_node_indexes = {}
field_indexes_across_all_rules = {} field_indexes_across_all_rules = {}
# Stores the index into @tree_fields by field alias name. # Stores the index into @ast_fields by field alias name.
field_aliases = {} field_aliases = {}
@tree_fields = [] @ast_fields = []
@rules.each do |rule| @rules.each do |rule|
rule.components.each_with_index do |component, i| rule.components.each_with_index do |component, i|
if component.is_a?(RuleSet) && component.optional? if component.is_a?(RuleSet) && component.optional?
@ -132,25 +132,25 @@ class Propane
else else
node_name = component.name node_name = component.name
end end
struct_name = "#{grammar.tree_prefix}#{node_name}#{grammar.tree_suffix}" struct_name = "#{grammar.ast_prefix}#{node_name}#{grammar.ast_suffix}"
field_name = "p#{node_name}#{i + 1}" field_name = "p#{node_name}#{i + 1}"
unless field_tree_node_indexes[field_name] unless field_ast_node_indexes[field_name]
field_tree_node_indexes[field_name] = @tree_fields.size field_ast_node_indexes[field_name] = @ast_fields.size
@tree_fields << {field_name => struct_name} @ast_fields << {field_name => struct_name}
end end
rule.aliases.each do |alias_name, index| rule.aliases.each do |alias_name, index|
if index == i if index == i
alias_tree_fields_index = field_tree_node_indexes[field_name] alias_ast_fields_index = field_ast_node_indexes[field_name]
if field_aliases[alias_name] && field_aliases[alias_name] != alias_tree_fields_index if field_aliases[alias_name] && field_aliases[alias_name] != alias_ast_fields_index
raise Error.new("Error: conflicting tree node field positions for alias `#{alias_name}` in rule #{rule.name} defined on line #{rule.line_number}") raise Error.new("Error: conflicting AST node field positions for alias `#{alias_name}` in rule #{rule.name} defined on line #{rule.line_number}")
end end
field_aliases[alias_name] = alias_tree_fields_index field_aliases[alias_name] = alias_ast_fields_index
@tree_fields[alias_tree_fields_index][alias_name] = @tree_fields[alias_tree_fields_index].first[1] @ast_fields[alias_ast_fields_index][alias_name] = @ast_fields[alias_ast_fields_index].first[1]
end end
end end
field_indexes_across_all_rules[node_name] ||= Set.new field_indexes_across_all_rules[node_name] ||= Set.new
field_indexes_across_all_rules[node_name] << field_tree_node_indexes[field_name] field_indexes_across_all_rules[node_name] << field_ast_node_indexes[field_name]
rule.rule_set_node_field_index_map[i] = field_tree_node_indexes[field_name] rule.rule_set_node_field_index_map[i] = field_ast_node_indexes[field_name]
end end
end end
field_indexes_across_all_rules.each do |node_name, indexes_across_all_rules| field_indexes_across_all_rules.each do |node_name, indexes_across_all_rules|
@ -158,8 +158,8 @@ class Propane
# If this field was only seen in one position across all rules, # If this field was only seen in one position across all rules,
# then add an alias to the positional field name that does not # then add an alias to the positional field name that does not
# include the position. # include the position.
@tree_fields[indexes_across_all_rules.first]["p#{node_name}"] = @ast_fields[indexes_across_all_rules.first]["p#{node_name}"] =
"#{grammar.tree_prefix}#{node_name}#{grammar.tree_suffix}" "#{grammar.ast_prefix}#{node_name}#{grammar.ast_suffix}"
end end
end end
end end

View File

@ -1,3 +1,3 @@
class Propane class Propane
VERSION = "5.1.0" VERSION = "2.3.0"
end end

View File

@ -1,5 +1,5 @@
tree; ast;
tree_prefix P; ast_prefix P;
<<header <<header
#include <stdio.h> #include <stdio.h>
@ -46,7 +46,7 @@ token hex_int_l /0[xX][0-9a-fA-F][0-9a-fA-F_]*/ <<
# Identifier. # Identifier.
token ident /\$?[a-zA-Z_][a-zA-Z_0-9]*\??/ << token ident /\$?[a-zA-Z_][a-zA-Z_0-9]*\??/ <<
$$.s = match_text; $$.s = match;
$mode(default); $mode(default);
return $token(ident); return $token(ident);
>> >>

View File

@ -1,5 +1,5 @@
tree; ast;
tree_prefix P; ast_prefix P;
<< <<
import std.bigint; import std.bigint;
@ -42,8 +42,8 @@ token semicolon /;/;
# Integer literals. # Integer literals.
token hex_int_l /0[xX][0-9a-fA-F][0-9a-fA-F_]*/ << token hex_int_l /0[xX][0-9a-fA-F][0-9a-fA-F_]*/ <<
$$.bi = BigInt(match_text[0..3]); $$.bi = BigInt(match[0..3]);
foreach (c; match_text[3..$]) foreach (c; match[3..$])
{ {
if (('0' <= c) && (c <= '9')) if (('0' <= c) && (c <= '9'))
{ {
@ -65,13 +65,13 @@ token hex_int_l /0[xX][0-9a-fA-F][0-9a-fA-F_]*/ <<
# Identifier. # Identifier.
token ident /\$?[a-zA-Z_][a-zA-Z_0-9]*\??/ << token ident /\$?[a-zA-Z_][a-zA-Z_0-9]*\??/ <<
if (match_text[0] == '$') if (match[0] == '$')
{ {
$$.s = match_text[1..$]; $$.s = match[1..$];
} }
else else
{ {
$$.s = match_text; $$.s = match;
} }
$mode(default); $mode(default);
return $token(ident); return $token(ident);

View File

@ -21,46 +21,46 @@ token number /-?(0|[1-9][0-9]*)(\.[0-9]+)?([eE][-+]?[0-9]+)?/ <<
double n = 0.0; double n = 0.0;
bool negative = false; bool negative = false;
size_t i = 0u; size_t i = 0u;
if (match_text[i] == '-') if (match[i] == '-')
{ {
negative = true; negative = true;
i++; i++;
} }
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
n *= 10.0; n *= 10.0;
n += (match_text[i] - '0'); n += (match[i] - '0');
i++; i++;
} }
if (match_text[i] == '.') if (match[i] == '.')
{ {
i++; i++;
double mult = 0.1; double mult = 0.1;
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
n += mult * (match_text[i] - '0'); n += mult * (match[i] - '0');
mult /= 10.0; mult /= 10.0;
i++; i++;
} }
} }
if (match_text[i] == 'e' || match_text[i] == 'E') if (match[i] == 'e' || match[i] == 'E')
{ {
bool exp_negative = false; bool exp_negative = false;
i++; i++;
if (match_text[i] == '-') if (match[i] == '-')
{ {
exp_negative = true; exp_negative = true;
i++; i++;
} }
else if (match_text[i] == '+') else if (match[i] == '+')
{ {
i++; i++;
} }
long exp = 0.0; long exp = 0.0;
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
exp *= 10; exp *= 10;
exp += (match_text[i] - '0'); exp += (match[i] - '0');
i++; i++;
} }
if (exp_negative) if (exp_negative)
@ -120,11 +120,11 @@ string: /\\t/ <<
>> >>
string: /\\u[0-9a-fA-F]{4}/ << string: /\\u[0-9a-fA-F]{4}/ <<
/* Not actually going to encode the code point for this example... */ /* Not actually going to encode the code point for this example... */
char s[] = {'{', (char)match_text[2], (char)match_text[3], (char)match_text[4], (char)match_text[5], '}', 0}; char s[] = {'{', (char)match[2], (char)match[3], (char)match[4], (char)match[5], '}', 0};
str_append(&string_value, s); str_append(&string_value, s);
>> >>
string: /[^\\]/ << string: /[^\\]/ <<
char s[] = {(char)match_text[0], 0}; char s[] = {(char)match[0], 0};
str_append(&string_value, s); str_append(&string_value, s);
>> >>
Start -> Value << Start -> Value <<

View File

@ -20,46 +20,46 @@ token number /-?(0|[1-9][0-9]*)(\.[0-9]+)?([eE][-+]?[0-9]+)?/ <<
double n; double n;
bool negative; bool negative;
size_t i = 0u; size_t i = 0u;
if (match_text[i] == '-') if (match[i] == '-')
{ {
negative = true; negative = true;
i++; i++;
} }
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
n *= 10.0; n *= 10.0;
n += (match_text[i] - '0'); n += (match[i] - '0');
i++; i++;
} }
if (match_text[i] == '.') if (match[i] == '.')
{ {
i++; i++;
double mult = 0.1; double mult = 0.1;
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
n += mult * (match_text[i] - '0'); n += mult * (match[i] - '0');
mult /= 10.0; mult /= 10.0;
i++; i++;
} }
} }
if (match_text[i] == 'e' || match_text[i] == 'E') if (match[i] == 'e' || match[i] == 'E')
{ {
bool exp_negative; bool exp_negative;
i++; i++;
if (match_text[i] == '-') if (match[i] == '-')
{ {
exp_negative = true; exp_negative = true;
i++; i++;
} }
else if (match_text[i] == '+') else if (match[i] == '+')
{ {
i++; i++;
} }
long exp; long exp;
while ('0' <= match_text[i] && match_text[i] <= '9') while ('0' <= match[i] && match[i] <= '9')
{ {
exp *= 10; exp *= 10;
exp += (match_text[i] - '0'); exp += (match[i] - '0');
i++; i++;
} }
if (exp_negative) if (exp_negative)
@ -117,10 +117,10 @@ string: /\\t/ <<
>> >>
string: /\\u[0-9a-fA-F]{4}/ << string: /\\u[0-9a-fA-F]{4}/ <<
/* Not actually going to encode the code point for this example... */ /* Not actually going to encode the code point for this example... */
string_value ~= "{" ~ match_text[2..6] ~ "}"; string_value ~= "{" ~ match[2..6] ~ "}";
>> >>
string: /[^\\]/ << string: /[^\\]/ <<
string_value ~= match_text; string_value ~= match;
>> >>
Start -> Value << Start -> Value <<
$$ = $1; $$ = $1;

View File

@ -1,176 +0,0 @@
<<
pub const JSON_OBJECT: usize = 0;
pub const JSON_ARRAY: usize = 1;
pub const JSON_NUMBER: usize = 2;
pub const JSON_STRING: usize = 3;
pub const JSON_TRUE: usize = 4;
pub const JSON_FALSE: usize = 5;
pub const JSON_NULL: usize = 6;
#[derive(Clone, Default)]
pub enum JSONValue {
#[default]
Null,
Object(Vec<(String, JSONValue)>),
Array(Vec<JSONValue>),
Number(f64),
StringVal(String),
True,
False,
}
impl JSONValue {
pub fn id(&self) -> usize {
match self {
JSONValue::Object(_) => JSON_OBJECT,
JSONValue::Array(_) => JSON_ARRAY,
JSONValue::Number(_) => JSON_NUMBER,
JSONValue::StringVal(_) => JSON_STRING,
JSONValue::True => JSON_TRUE,
JSONValue::False => JSON_FALSE,
JSONValue::Null => JSON_NULL,
}
}
pub fn number(&self) -> f64 {
if let JSONValue::Number(n) = self { *n } else { 0.0 }
}
pub fn string(&self) -> &str {
if let JSONValue::StringVal(s) = self { s.as_str() } else { "" }
}
pub fn object_len(&self) -> usize {
if let JSONValue::Object(e) = self { e.len() } else { 0 }
}
pub fn array_len(&self) -> usize {
if let JSONValue::Array(e) = self { e.len() } else { 0 }
}
}
>>
context_user_fields <<
pub string_value: String,
>>
ptype JSONValue;
drop /\s+/;
token lbrace /\{/;
token rbrace /\}/;
token lbracket /\[/;
token rbracket /\]/;
token comma /,/;
token colon /:/;
token number /-?(0|[1-9][0-9]*)(\.[0-9]+)?([eE][-+]?[0-9]+)?/ <<
let n: f64 = std::str::from_utf8(match_text).unwrap().parse().unwrap();
$$ = JSONValue::Number(n);
>>
token true <<
$$ = JSONValue::True;
>>
token false <<
$$ = JSONValue::False;
>>
token null <<
$$ = JSONValue::Null;
>>
/"/ <<
$mode(string);
${context.string_value} = String::new();
>>
string: token string /"/ <<
$$ = JSONValue::StringVal(std::mem::take(&mut ${context.string_value}));
$mode(default);
>>
string: /\\"/ <<
${context.string_value}.push('"');
>>
string: /\\\\/ <<
${context.string_value}.push('\\');
>>
string: /\\\// <<
${context.string_value}.push('/');
>>
string: /\\b/ <<
${context.string_value}.push('\u{0008}');
>>
string: /\\f/ <<
${context.string_value}.push('\u{000C}');
>>
string: /\\n/ <<
${context.string_value}.push('\n');
>>
string: /\\r/ <<
${context.string_value}.push('\r');
>>
string: /\\t/ <<
${context.string_value}.push('\t');
>>
string: /\\u[0-9a-fA-F]{4}/ <<
/* Not actually going to encode the code point for this example... */
let s: String = ['{', match_text[2] as char, match_text[3] as char, match_text[4] as char, match_text[5] as char, '}'].iter().collect();
${context.string_value}.push_str(&s);
>>
string: /[^\\]/ <<
${context.string_value}.push(match_text[0] as char);
>>
Start -> Value <<
$$ = $1;
>>
Value -> string <<
$$ = $1;
>>
Value -> number <<
$$ = $1;
>>
Value -> Object <<
$$ = $1;
>>
Value -> Array <<
$$ = $1;
>>
Value -> true <<
$$ = $1;
>>
Value -> false <<
$$ = $1;
>>
Value -> null <<
$$ = $1;
>>
Object -> lbrace rbrace <<
$$ = JSONValue::Object(Vec::new());
>>
Object -> lbrace KeyValues rbrace <<
$$ = $2;
>>
KeyValues -> KeyValue <<
$$ = $1;
>>
KeyValues -> KeyValues comma KeyValue <<
let mut obj = $1;
if let JSONValue::Object(kve) = $3 {
if let JSONValue::Object(entries) = &mut obj {
entries.extend(kve);
}
}
$$ = obj;
>>
KeyValue -> string colon Value <<
let name = if let JSONValue::StringVal(s) = $1 { s } else { String::new() };
$$ = JSONValue::Object(vec![(name, $3)]);
>>
Array -> lbracket rbracket <<
$$ = JSONValue::Array(Vec::new());
>>
Array -> lbracket Values rbracket <<
$$ = $2;
>>
Values -> Value <<
$$ = $1;
>>
Values -> Values comma Value <<
let mut arr = $1;
if let JSONValue::Array(elems) = &mut arr {
elems.push($3);
}
$$ = arr;
>>

View File

@ -1,25 +0,0 @@
<<
#include <stdlib.h>
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info);
void record(int v);
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token macro;
token macroname /@[a-zA-Z_]\w*/;
token num /\d+/ << char b[100]; memcpy(b, match_text, match_length); b[match_length] = '\0'; $$ = atoi(b); >>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Statement -> MacroStart;
Add -> num plus num << $$ = $1 + $3; record($$); >>
MacroStart -> macro macroname lbrace;

View File

@ -1,31 +0,0 @@
<<
import test_macros;
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token macro;
token macroname /@[a-zA-Z_]\w*/;
token num /\d+/ <<
int n = 0;
foreach (c; match_text)
{
n *= 10;
n += (c - '0');
}
$$ = n;
>>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Statement -> MacroStart;
Add -> num plus num << $$ = $1 + $3; record($$); >>
MacroStart -> macro macroname lbrace;

View File

@ -1,80 +0,0 @@
<<
fn mylexfn(context: &mut p_context_t, out_token_info: &mut p_token_info_t) -> usize {
loop {
if context.expanding {
let ei = context.expand_i;
context.expand_i += 1;
if context.expand_i >= context.token_infos.len() {
context.expanding = false;
}
*out_token_info = context.token_infos[ei].clone();
return P_SUCCESS;
}
let lex_result = p_lex(context, out_token_info);
if lex_result != P_SUCCESS {
return lex_result;
}
if out_token_info.token == TOKEN_macro {
context.defining = true;
} else if out_token_info.token == TOKEN_macroname {
if !context.defining {
context.expanding = true;
context.expand_i = 0;
continue;
}
} else if out_token_info.token == TOKEN_lbrace {
if context.defining {
/* Capture the macro body tokens (up to the closing '}'). */
let mut infos: Vec<p_token_info_t> = Vec::new();
loop {
let mut ti = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(context, &mut ti));
if ti.token == TOKEN_rbrace {
break;
}
infos.push(ti);
}
context.token_infos = infos;
context.defining = false;
}
} else {
context.defining = false;
}
return lex_result;
}
}
>>
context_user_fields <<
pub defining: bool,
pub expanding: bool,
pub expand_i: usize,
pub token_infos: Vec<p_token_info_t>,
pub nums: Vec<i64>,
>>
ptype i64;
lex_fn mylexfn;
drop /\s+/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token macro;
token macroname /@[a-zA-Z_]\w*/;
token num /\d+/ <<
let mut v: i64 = 0;
for c in match_text { v = v * 10 + (*c - b'0') as i64; }
$$ = v;
>>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Statement -> MacroStart;
Add -> num plus num << $$ = $1 + $3; ${context.nums}.push($$); >>
MacroStart -> macro macroname lbrace;

View File

@ -1,19 +0,0 @@
<<
#include <stdlib.h>
#include <string.h>
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info);
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/ << char b[32]; memcpy(b, match_text, match_length); b[match_length] = '\0'; $$ = atoi(b); >>
Start -> Expr << $$ = $1; >>
Expr -> num << $$ = $1; >>
Expr -> Expr plus num << $$ = $1 + $3; >>

View File

@ -1,17 +0,0 @@
<<
import test_parse_inner_nested;
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/ << int n = 0; foreach (ch; match_text) { n *= 10; n += (ch - '0'); } $$ = n; >>
Start -> Expr << $$ = $1; >>
Expr -> num << $$ = $1; >>
Expr -> Expr plus num << $$ = $1 + $3; >>

View File

@ -1,41 +0,0 @@
<<
fn mylexfn(context: &mut p_context_t, out_token_info: &mut p_token_info_t) -> usize {
let result = p_lex(context, out_token_info);
if result != P_SUCCESS {
return result;
}
if out_token_info.token == TOKEN_lparen {
/* Reentrant nested parse of the parenthesized sub-expression. */
let inner_result = p_parse_inner_Start(context, &[TOKEN_rparen]);
if inner_result != P_SUCCESS {
return inner_result;
}
let value = p_result_Start(context);
/* p_parse_inner rewound the input so ')' was not consumed; consume it. */
let mut rparen_info = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(context, &mut rparen_info));
assert_eq!(TOKEN_rparen, rparen_info.token);
out_token_info.token = TOKEN_num;
out_token_info.pvalue = p_value(value);
}
P_SUCCESS
}
>>
ptype i64;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/ <<
let mut v: i64 = 0;
for c in match_text { v = v * 10 + (*c - b'0') as i64; }
$$ = v;
>>
Start -> Expr << $$ = $1; >>
Expr -> num << $$ = $1; >>
Expr -> Expr plus num << $$ = $1 + $3; >>

View File

@ -1,17 +0,0 @@
<<
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info);
>>
tree;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/;
Start -> Expr;
Expr -> num;
Expr -> Expr plus num;

View File

@ -1,17 +0,0 @@
<<
import test_parse_inner_nested_tree;
>>
tree;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/;
Start -> Expr;
Expr -> num;
Expr -> Expr plus num;

View File

@ -1,44 +0,0 @@
<<
fn mylexfn(context: &mut p_context_t, out_token_info: &mut p_token_info_t) -> usize {
let result = p_lex(context, out_token_info);
if result != P_SUCCESS {
return result;
}
if out_token_info.token == TOKEN_lparen {
let start_position = out_token_info.position;
let inner_result = p_parse_inner_Start(context, &[TOKEN_rparen]);
if inner_result != P_SUCCESS {
return inner_result;
}
/* Read the inner subtree's span before re-borrowing context to lex. */
let inner = p_result_Start(context);
assert!(inner.valid());
let inner_start_col = inner.position().col;
let inner_end_col = inner.end_position().col;
let mut rparen_info = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(context, &mut rparen_info));
assert_eq!(TOKEN_rparen, rparen_info.token);
assert_eq!(start_position.col + 1, inner_start_col);
assert_eq!(rparen_info.position.col - 1, inner_end_col);
/* Synthesize a num token spanning the whole "( ... )" group. */
out_token_info.token = TOKEN_num;
out_token_info.position = start_position;
out_token_info.end_position = rparen_info.end_position;
}
P_SUCCESS
}
>>
tree;
lex_fn mylexfn;
drop /\s+/;
token lparen /\(/;
token rparen /\)/;
token plus /\+/;
token num /\d+/;
Start -> Expr;
Expr -> num;
Expr -> Expr plus num;

View File

@ -28,7 +28,7 @@ B -> <<
b = 0; b = 0;
>> >>
EOF EOF
grammar = Grammar.new(input, "test.propane") grammar = Grammar.new(input)
expect(grammar.modulename).to eq "a.b" expect(grammar.modulename).to eq "a.b"
expect(grammar.ptype).to eq "XYZ *" expect(grammar.ptype).to eq "XYZ *"
expect(grammar.ptypes).to eq("default" => "XYZ *") expect(grammar.ptypes).to eq("default" => "XYZ *")
@ -62,7 +62,7 @@ EOF
expect(o).to_not be_nil expect(o).to_not be_nil
expect(o.pattern).to eq "token_with_code" expect(o.pattern).to eq "token_with_code"
expect(o.line_number).to eq 11 expect(o.line_number).to eq 11
expect(o.code).to eq %[#line 12 "test.propane"\nCode for the token\n#linereset\n] expect(o.code).to eq "Code for the token\n"
o = grammar.tokens.find {|token| token.name == "token_with_no_pattern"} o = grammar.tokens.find {|token| token.name == "token_with_no_pattern"}
expect(o).to_not be_nil expect(o).to_not be_nil
@ -83,7 +83,7 @@ EOF
expect(o.name).to eq "A" expect(o.name).to eq "A"
expect(o.components).to eq %w[B] expect(o.components).to eq %w[B]
expect(o.line_number).to eq 19 expect(o.line_number).to eq 19
expect(o.code).to eq %[#line 20 "test.propane"\n a = 42;\n#linereset\n] expect(o.code).to eq " a = 42;\n"
o = grammar.rules[1] o = grammar.rules[1]
expect(o.name).to eq "B" expect(o.name).to eq "B"
@ -95,7 +95,7 @@ EOF
expect(o.name).to eq "B" expect(o.name).to eq "B"
expect(o.components).to eq [] expect(o.components).to eq []
expect(o.line_number).to eq 23 expect(o.line_number).to eq 23
expect(o.code).to eq %[#line 24 "test.propane"\n b = 0;\n#linereset\n] expect(o.code).to eq " b = 0;\n"
end end
it "parses code segments with semicolons" do it "parses code segments with semicolons" do
@ -113,7 +113,7 @@ tokenid token_with_no_pattern;
prefix myparser_; prefix myparser_;
EOF EOF
grammar = Grammar.new(input, "test.propane") grammar = Grammar.new(input)
expect(grammar.prefix).to eq "myparser_" expect(grammar.prefix).to eq "myparser_"
o = grammar.tokens.find {|token| token.name == "code1"} o = grammar.tokens.find {|token| token.name == "code1"}
@ -122,7 +122,7 @@ EOF
o = grammar.patterns.find {|pattern| pattern.token == o} o = grammar.patterns.find {|pattern| pattern.token == o}
expect(o).to_not be_nil expect(o).to_not be_nil
expect(o.code).to eq %[#line 2 "test.propane"\n a = b;\n return c;\n#linereset\n] expect(o.code).to eq " a = b;\n return c;\n"
o = grammar.tokens.find {|token| token.name == "code2"} o = grammar.tokens.find {|token| token.name == "code2"}
expect(o).to_not be_nil expect(o).to_not be_nil
@ -130,42 +130,7 @@ EOF
o = grammar.patterns.find {|pattern| pattern.token == o} o = grammar.patterns.find {|pattern| pattern.token == o}
expect(o).to_not be_nil expect(o).to_not be_nil
expect(o.code).to eq %[#line 7 "test.propane"\n writeln("Hello there");\n#linereset\n] expect(o.code).to eq %[ writeln("Hello there");\n]
end
it "does not emit #line directives with noline statement" do
input = <<EOF
noline;
token code1 <<
a = b;
return c;
>>
token code2 <<
writeln("Hello there");
>>
tokenid token_with_no_pattern;
prefix myparser_;
EOF
grammar = Grammar.new(input, "test.propane")
expect(grammar.prefix).to eq "myparser_"
o = grammar.tokens.find {|token| token.name == "code1"}
expect(o).to_not be_nil
o = grammar.patterns.find {|pattern| pattern.token == o}
expect(o).to_not be_nil
expect(o.code).to eq %[\n a = b;\n return c;]
o = grammar.tokens.find {|token| token.name == "code2"}
expect(o).to_not be_nil
o = grammar.patterns.find {|pattern| pattern.token == o}
expect(o).to_not be_nil
expect(o.code).to eq %[\n writeln("Hello there");]
end end
it "supports mode labels" do it "supports mode labels" do
@ -179,7 +144,7 @@ m2: /bar/ <<
drop /q/; drop /q/;
m3: drop /r/; m3: drop /r/;
EOF EOF
grammar = Grammar.new(input, "test.propane") grammar = Grammar.new(input)
o = grammar.tokens.find {|token| token.name == "a"} o = grammar.tokens.find {|token| token.name == "a"}
expect(o).to_not be_nil expect(o).to_not be_nil
@ -232,7 +197,7 @@ tokenid int(integer);
Start (node) -> R; Start (node) -> R;
R -> abc int; R -> abc int;
EOF EOF
grammar = Grammar.new(input, "test.propane") grammar = Grammar.new(input)
o = grammar.tokens.find {|token| token.name == "abc"} o = grammar.tokens.find {|token| token.name == "abc"}
expect(o).to_not be_nil expect(o).to_not be_nil

View File

@ -51,7 +51,7 @@ class TestLexer
end end
def run(grammar, input) def run(grammar, input)
grammar = Propane::Grammar.new(grammar, "test.propane") grammar = Propane::Grammar.new(grammar)
token_dfa = Propane::Lexer::DFA.new(grammar.patterns) token_dfa = Propane::Lexer::DFA.new(grammar.patterns)
test_lexer = TestLexer.new(token_dfa) test_lexer = TestLexer.new(token_dfa)
test_lexer.lex(input) test_lexer.lex(input)

File diff suppressed because it is too large Load Diff

View File

@ -1,23 +0,0 @@
<<
#include <stdlib.h>
#include <string.h>
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info);
void record(int value);
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token repeat /repeat/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token num /\d+/ << char b[32]; memcpy(b, match_text, match_length); b[match_length] = '\0'; $$ = atoi(b); >>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Add -> num plus num << record($1 + $3); >>

View File

@ -1,20 +0,0 @@
<<
import test_rewind;
>>
ptype int;
lex_fn mylexfn;
drop /\s+/;
token repeat /repeat/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token num /\d+/ << int n = 0; foreach (ch; match_text) { n *= 10; n += (ch - '0'); } $$ = n; >>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Add -> num plus num << record($1 + $3); >>

View File

@ -1,67 +0,0 @@
<<
fn mylexfn(context: &mut p_context_t, out_token_info: &mut p_token_info_t) -> usize {
loop {
let result = p_lex(context, out_token_info);
if result != P_SUCCESS {
return result;
}
if out_token_info.token == TOKEN_repeat {
let mut count_info = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(context, &mut count_info));
assert_eq!(TOKEN_num, count_info.token);
let mut brace_info = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(context, &mut brace_info));
assert_eq!(TOKEN_lbrace, brace_info.token);
context.remaining = p_value_get(&count_info.pvalue);
context.body_index = p_input_index(context);
context.body_position = p_position(context);
continue;
}
if out_token_info.token == TOKEN_rbrace {
if context.remaining > 1 {
context.remaining -= 1;
let bi = context.body_index;
let bp = context.body_position;
p_set_input_index(context, bi);
p_set_position(context, bp);
continue;
}
context.remaining = 0;
continue;
}
if out_token_info.token == TOKEN_num {
context.num_cols.push(out_token_info.position.col);
}
return result;
}
}
>>
context_user_fields <<
pub nums: Vec<i64>,
pub num_cols: Vec<u32>,
pub remaining: i64,
pub body_index: usize,
pub body_position: p_position_t,
>>
ptype i64;
lex_fn mylexfn;
drop /\s+/;
token repeat /repeat/;
token lbrace /\{/;
token rbrace /\}/;
token plus /\+/;
token num /\d+/ <<
let mut v: i64 = 0;
for c in match_text { v = v * 10 + (*c - b'0') as i64; }
$$ = v;
>>
Start -> Statements;
Statements -> ;
Statements -> Statement Statements;
Statement -> Add;
Add -> num plus num << ${context.nums}.push($1 + $3); >>

View File

@ -15,10 +15,6 @@ unless ENV["dist_specs"]
command_name "RSpec" command_name "RSpec"
end end
project_name "Propane" project_name "Propane"
# Keep this process's results separate from the propane subprocess results
# so that nothing has to merge on the fly; the spec Rake task collates all
# of the parts once the suite is done.
coverage_dir "coverage/parts/rspec"
merge_timeout 3600 merge_timeout 3600
formatter(MyFormatter) formatter(MyFormatter)
end end

61
spec/test_ast.c Normal file
View File

@ -0,0 +1,61 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "a, ((b)), b";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
Start * start = p_result(&context);
assert(start->pItems1 != NULL);
assert(start->pItems != NULL);
Items * items = start->pItems;
assert(items->pItem != NULL);
assert(items->pItem->pToken1 != NULL);
assert_eq(TOKEN_a, items->pItem->pToken1->token);
assert_eq(11, items->pItem->pToken1->pvalue);
assert(items->pItemsMore != NULL);
ItemsMore * itemsmore = items->pItemsMore;
assert(itemsmore->pItem != NULL);
assert(itemsmore->pItem->pItem != NULL);
assert(itemsmore->pItem->pItem->pItem != NULL);
assert(itemsmore->pItem->pItem->pItem->pToken1 != NULL);
assert_eq(TOKEN_b, itemsmore->pItem->pItem->pItem->pToken1->token);
assert_eq(22, itemsmore->pItem->pItem->pItem->pToken1->pvalue);
assert(itemsmore->pItemsMore != NULL);
itemsmore = itemsmore->pItemsMore;
assert(itemsmore->pItem != NULL);
assert(itemsmore->pItem->pToken1 != NULL);
assert_eq(TOKEN_b, itemsmore->pItem->pToken1->token);
assert_eq(22, itemsmore->pItem->pToken1->pvalue);
assert(itemsmore->pItemsMore == NULL);
p_free_ast(start);
input = "";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start->pItems == NULL);
p_free_ast(start);
input = "2 1";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start->pItems != NULL);
assert(start->pItems->pItem != NULL);
assert(start->pItems->pItem->pDual != NULL);
assert(start->pItems->pItem->pDual->pTwo1 != NULL);
assert(start->pItems->pItem->pDual->pOne2 != NULL);
assert(start->pItems->pItem->pDual->pTwo2 == NULL);
assert(start->pItems->pItem->pDual->pOne1 == NULL);
p_free_ast(start);
return 0;
}

57
spec/test_ast.d Normal file
View File

@ -0,0 +1,57 @@
import testparser;
import std.stdio;
import testutils;
int main()
{
return 0;
}
unittest
{
string input = "a, ((b)), b";
p_context_t context;
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
Start * start = p_result(&context);
assert(start.pItems1 !is null);
assert(start.pItems !is null);
Items * items = start.pItems;
assert(items.pItem !is null);
assert(items.pItem.pToken1 !is null);
assert_eq(TOKEN_a, items.pItem.pToken1.token);
assert_eq(11, items.pItem.pToken1.pvalue);
assert(items.pItemsMore !is null);
ItemsMore * itemsmore = items.pItemsMore;
assert(itemsmore.pItem !is null);
assert(itemsmore.pItem.pItem !is null);
assert(itemsmore.pItem.pItem.pItem !is null);
assert(itemsmore.pItem.pItem.pItem.pToken1 !is null);
assert_eq(TOKEN_b, itemsmore.pItem.pItem.pItem.pToken1.token);
assert_eq(22, itemsmore.pItem.pItem.pItem.pToken1.pvalue);
assert(itemsmore.pItemsMore !is null);
itemsmore = itemsmore.pItemsMore;
assert(itemsmore.pItem !is null);
assert(itemsmore.pItem.pToken1 !is null);
assert_eq(TOKEN_b, itemsmore.pItem.pToken1.token);
assert_eq(22, itemsmore.pItem.pToken1.pvalue);
assert(itemsmore.pItemsMore is null);
input = "";
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start.pItems is null);
input = "2 1";
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start.pItems !is null);
assert(start.pItems.pItem !is null);
assert(start.pItems.pItem.pDual !is null);
assert(start.pItems.pItem.pDual.pTwo1 !is null);
assert(start.pItems.pItem.pDual.pOne2 !is null);
assert(start.pItems.pItem.pDual.pTwo2 is null);
assert(start.pItems.pItem.pDual.pOne1 is null);
}

View File

@ -0,0 +1,21 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "\na\nb\nc";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(TOKEN_a, start->first->pToken->token);
assert_eq(TOKEN_b, start->second->pToken->token);
assert_eq(TOKEN_c, start->third->pToken->token);
p_free_ast(start);
return 0;
}

View File

@ -10,13 +10,12 @@ int main()
unittest unittest
{ {
string input = "\na\nb\nc"; string input = "\na\nb\nc";
p_context_t * context = p_context_new(input); p_context_t context;
assert(p_parse(context) == P_SUCCESS); p_context_init(&context, input);
Start start = p_result(context); assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(TOKEN_a, start.first.pToken.token); assert_eq(TOKEN_a, start.first.pToken.token);
assert_eq(TOKEN_b, start.second.pToken.token); assert_eq(TOKEN_b, start.second.pToken.token);
assert_eq(TOKEN_c, start.third.pToken.token); assert_eq(TOKEN_c, start.third.pToken.token);
p_context_delete(context);
} }

View File

@ -0,0 +1,110 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "\na\n bb ccc";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(2, start->pT1->pToken->position.row);
assert_eq(1, start->pT1->pToken->position.col);
assert_eq(2, start->pT1->pToken->end_position.row);
assert_eq(1, start->pT1->pToken->end_position.col);
assert(p_position_valid(start->pT1->pA->position));
assert_eq(3, start->pT1->pA->position.row);
assert_eq(3, start->pT1->pA->position.col);
assert_eq(3, start->pT1->pA->end_position.row);
assert_eq(8, start->pT1->pA->end_position.col);
assert_eq(2, start->pT1->position.row);
assert_eq(1, start->pT1->position.col);
assert_eq(3, start->pT1->end_position.row);
assert_eq(8, start->pT1->end_position.col);
assert_eq(2, start->position.row);
assert_eq(1, start->position.col);
assert_eq(3, start->end_position.row);
assert_eq(8, start->end_position.col);
p_free_ast(start);
input = "a\nbb";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
start = p_result(&context);
assert_eq(1, start->pT1->pToken->position.row);
assert_eq(1, start->pT1->pToken->position.col);
assert_eq(1, start->pT1->pToken->end_position.row);
assert_eq(1, start->pT1->pToken->end_position.col);
assert(p_position_valid(start->pT1->pA->position));
assert_eq(2, start->pT1->pA->position.row);
assert_eq(1, start->pT1->pA->position.col);
assert_eq(2, start->pT1->pA->end_position.row);
assert_eq(2, start->pT1->pA->end_position.col);
assert_eq(1, start->pT1->position.row);
assert_eq(1, start->pT1->position.col);
assert_eq(2, start->pT1->end_position.row);
assert_eq(2, start->pT1->end_position.col);
assert_eq(1, start->position.row);
assert_eq(1, start->position.col);
assert_eq(2, start->end_position.row);
assert_eq(2, start->end_position.col);
p_free_ast(start);
input = "a\nc\nc";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
start = p_result(&context);
assert_eq(1, start->pT1->pToken->position.row);
assert_eq(1, start->pT1->pToken->position.col);
assert_eq(1, start->pT1->pToken->end_position.row);
assert_eq(1, start->pT1->pToken->end_position.col);
assert(p_position_valid(start->pT1->pA->position));
assert_eq(2, start->pT1->pA->position.row);
assert_eq(1, start->pT1->pA->position.col);
assert_eq(3, start->pT1->pA->end_position.row);
assert_eq(1, start->pT1->pA->end_position.col);
assert_eq(1, start->pT1->position.row);
assert_eq(1, start->pT1->position.col);
assert_eq(3, start->pT1->end_position.row);
assert_eq(1, start->pT1->end_position.col);
assert_eq(1, start->position.row);
assert_eq(1, start->position.col);
assert_eq(3, start->end_position.row);
assert_eq(1, start->end_position.col);
p_free_ast(start);
input = "a";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
start = p_result(&context);
assert_eq(1, start->pT1->pToken->position.row);
assert_eq(1, start->pT1->pToken->position.col);
assert_eq(1, start->pT1->pToken->end_position.row);
assert_eq(1, start->pT1->pToken->end_position.col);
assert(!p_position_valid(start->pT1->pA->position));
assert_eq(1, start->pT1->position.row);
assert_eq(1, start->pT1->position.col);
assert_eq(1, start->pT1->end_position.row);
assert_eq(1, start->pT1->end_position.col);
assert_eq(1, start->position.row);
assert_eq(1, start->position.col);
assert_eq(1, start->end_position.row);
assert_eq(1, start->end_position.col);
p_free_ast(start);
return 0;
}

View File

@ -10,9 +10,10 @@ int main()
unittest unittest
{ {
string input = "\na\n bb ccc"; string input = "\na\n bb ccc";
p_context_t * context = p_context_new(input); p_context_t context;
assert(p_parse(context) == P_SUCCESS); p_context_init(&context, input);
Start start = p_result(context); assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(2, start.pT1.pToken.position.row); assert_eq(2, start.pT1.pToken.position.row);
assert_eq(1, start.pT1.pToken.position.col); assert_eq(1, start.pT1.pToken.position.col);
@ -33,12 +34,10 @@ unittest
assert_eq(3, start.end_position.row); assert_eq(3, start.end_position.row);
assert_eq(8, start.end_position.col); assert_eq(8, start.end_position.col);
p_context_delete(context);
input = "a\nbb"; input = "a\nbb";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
start = p_result(context); start = p_result(&context);
assert_eq(1, start.pT1.pToken.position.row); assert_eq(1, start.pT1.pToken.position.row);
assert_eq(1, start.pT1.pToken.position.col); assert_eq(1, start.pT1.pToken.position.col);
@ -59,12 +58,10 @@ unittest
assert_eq(2, start.end_position.row); assert_eq(2, start.end_position.row);
assert_eq(2, start.end_position.col); assert_eq(2, start.end_position.col);
p_context_delete(context);
input = "a\nc\nc"; input = "a\nc\nc";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
start = p_result(context); start = p_result(&context);
assert_eq(1, start.pT1.pToken.position.row); assert_eq(1, start.pT1.pToken.position.row);
assert_eq(1, start.pT1.pToken.position.col); assert_eq(1, start.pT1.pToken.position.col);
@ -85,12 +82,10 @@ unittest
assert_eq(3, start.end_position.row); assert_eq(3, start.end_position.row);
assert_eq(1, start.end_position.col); assert_eq(1, start.end_position.col);
p_context_delete(context);
input = "a"; input = "a";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
start = p_result(context); start = p_result(&context);
assert_eq(1, start.pT1.pToken.position.row); assert_eq(1, start.pT1.pToken.position.row);
assert_eq(1, start.pT1.pToken.position.col); assert_eq(1, start.pT1.pToken.position.col);
@ -106,6 +101,4 @@ unittest
assert_eq(1, start.position.col); assert_eq(1, start.position.col);
assert_eq(1, start.end_position.row); assert_eq(1, start.end_position.row);
assert_eq(1, start.end_position.col); assert_eq(1, start.end_position.col);
p_context_delete(context);
} }

View File

@ -369,50 +369,50 @@ int main(int argc, char * argv[])
{"size_t_to_ulong", TOKEN_ulong}, {"size_t_to_ulong", TOKEN_ulong},
{"main", TOKEN_int}, {"main", TOKEN_int},
}; };
p_context_t * context; p_context_t context;
context = p_context_new((const uint8_t *)input, strlen(input)); p_context_init(&context, (const uint8_t *)input, strlen(input));
size_t result = p_parse(context); size_t result = p_parse(&context);
assert_eq(P_SUCCESS, result); assert_eq(P_SUCCESS, result);
PModule pmod = p_result(context); PModule * pmod = p_result(&context);
PModuleItems pmis = p_PModule_pModuleItems(pmod); PModuleItems * pmis = pmod->pModuleItems;
PFunctionDefinition * pfds; PFunctionDefinition ** pfds;
size_t n_pfds = 0u; size_t n_pfds = 0u;
while (p_node_valid(pmis)) while (pmis != NULL)
{ {
PModuleItem pmi = p_PModuleItems_pModuleItem(pmis); PModuleItem * pmi = pmis->pModuleItem;
if (p_node_valid(p_PModuleItem_pFunctionDefinition(pmi))) if (pmi->pFunctionDefinition != NULL)
{ {
n_pfds++; n_pfds++;
} }
pmis = p_PModuleItems_pModuleItems(pmis); pmis = pmis->pModuleItems;
} }
pfds = (PFunctionDefinition *)malloc(n_pfds * sizeof(PFunctionDefinition)); pfds = (PFunctionDefinition **)malloc(n_pfds * sizeof(PModuleItems *));
pmis = p_PModule_pModuleItems(pmod); pmis = pmod->pModuleItems;
size_t pfd_i = n_pfds; size_t pfd_i = n_pfds;
while (p_node_valid(pmis)) while (pmis != NULL)
{ {
PModuleItem pmi = p_PModuleItems_pModuleItem(pmis); PModuleItem * pmi = pmis->pModuleItem;
PFunctionDefinition pfd = p_PModuleItem_pFunctionDefinition(pmi); PFunctionDefinition * pfd = pmi->pFunctionDefinition;
if (p_node_valid(pfd)) if (pfd != NULL)
{ {
pfd_i--; pfd_i--;
assert(pfd_i < n_pfds); assert(pfd_i < n_pfds);
pfds[pfd_i] = pfd; pfds[pfd_i] = pfd;
} }
pmis = p_PModuleItems_pModuleItems(pmis); pmis = pmis->pModuleItems;
} }
assert_eq(51, n_pfds); assert_eq(51, n_pfds);
for (size_t i = 0; i < n_pfds; i++) for (size_t i = 0; i < n_pfds; i++)
{ {
if (strncmp(expected[i].name, (const char *)p_node_data(p_PFunctionDefinition_name(pfds[i]))->pvalue.s, strlen(expected[i].name)) != 0 || if (strncmp(expected[i].name, (const char *)pfds[i]->name->pvalue.s, strlen(expected[i].name)) != 0 ||
(expected[i].token != p_tree_walk_PFunctionDefinition(pfds[i], returntype, pType, pTypeBase, pToken1, token))) (expected[i].token != pfds[i]->returntype->pType->pTypeBase->pToken1->token))
{ {
fprintf(stderr, "Index %lu: expected %s/%u, got %u\n", i, expected[i].name, expected[i].token, p_tree_walk_PFunctionDefinition(pfds[i], returntype, pType, pTypeBase, pToken1, token)); fprintf(stderr, "Index %lu: expected %s/%u, got %u\n", i, expected[i].name, expected[i].token, pfds[i]->returntype->pType->pTypeBase->pToken1->token);
} }
} }
free(pfds); free(pfds);
p_context_delete(context); p_free_ast(pmod);
return 0; return 0;
} }

View File

@ -374,23 +374,23 @@ def main() -> int
Expected("size_t_to_ulong", TOKEN_ulong), Expected("size_t_to_ulong", TOKEN_ulong),
Expected("main", TOKEN_int), Expected("main", TOKEN_int),
]; ];
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
size_t result = p_parse(context); size_t result = p_parse(&context);
assert_eq(P_SUCCESS, result); assert_eq(P_SUCCESS, result);
PModule pmod = p_result(context); PModule * pmod = p_result(&context);
PModuleItems pmis = pmod.pModuleItems; PModuleItems * pmis = pmod.pModuleItems;
PFunctionDefinition[] pfds; PFunctionDefinition *[] pfds;
while (pmis.valid) while (pmis !is null)
{ {
PModuleItem pmi = pmis.pModuleItem; PModuleItem * pmi = pmis.pModuleItem;
if (!pmi.valid) if (pmi is null)
{ {
stderr.writeln("pmi is null!!!?"); stderr.writeln("pmi is null!!!?");
assert(0); assert(0);
} }
PFunctionDefinition pfd = pmi.pFunctionDefinition; PFunctionDefinition * pfd = pmi.pFunctionDefinition;
if (pfd.valid) if (pfd !is null)
{ {
pfds = [pfd] ~ pfds; pfds = [pfd] ~ pfds;
} }
@ -405,5 +405,4 @@ def main() -> int
stderr.writeln("Index ", i, ": expected ", expected[i].name, "/", expected[i].token, ", got ", pfds[i].name.pvalue.s, "/", pfds[i].returntype.pType.pTypeBase.pToken1.token); stderr.writeln("Index ", i, ": expected ", expected[i].name, "/", expected[i].token, ", got ", pfds[i].name.pvalue.s, "/", pfds[i].returntype.pType.pTypeBase.pToken1.token);
} }
} }
p_context_delete(context);
} }

61
spec/test_ast_ps.c Normal file
View File

@ -0,0 +1,61 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "a, ((b)), b";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
PStartS * start = p_result(&context);
assert(start->pItems1 != NULL);
assert(start->pItems != NULL);
PItemsS * items = start->pItems;
assert(items->pItem != NULL);
assert(items->pItem->pToken1 != NULL);
assert_eq(TOKEN_a, items->pItem->pToken1->token);
assert_eq(11, items->pItem->pToken1->pvalue);
assert(items->pItemsMore != NULL);
PItemsMoreS * itemsmore = items->pItemsMore;
assert(itemsmore->pItem != NULL);
assert(itemsmore->pItem->pItem != NULL);
assert(itemsmore->pItem->pItem->pItem != NULL);
assert(itemsmore->pItem->pItem->pItem->pToken1 != NULL);
assert_eq(TOKEN_b, itemsmore->pItem->pItem->pItem->pToken1->token);
assert_eq(22, itemsmore->pItem->pItem->pItem->pToken1->pvalue);
assert(itemsmore->pItemsMore != NULL);
itemsmore = itemsmore->pItemsMore;
assert(itemsmore->pItem != NULL);
assert(itemsmore->pItem->pToken1 != NULL);
assert_eq(TOKEN_b, itemsmore->pItem->pToken1->token);
assert_eq(22, itemsmore->pItem->pToken1->pvalue);
assert(itemsmore->pItemsMore == NULL);
p_free_ast(start);
input = "";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start->pItems == NULL);
p_free_ast(start);
input = "2 1";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start->pItems != NULL);
assert(start->pItems->pItem != NULL);
assert(start->pItems->pItem->pDual != NULL);
assert(start->pItems->pItem->pDual->pTwo1 != NULL);
assert(start->pItems->pItem->pDual->pOne2 != NULL);
assert(start->pItems->pItem->pDual->pTwo2 == NULL);
assert(start->pItems->pItem->pDual->pOne1 == NULL);
p_free_ast(start);
return 0;
}

57
spec/test_ast_ps.d Normal file
View File

@ -0,0 +1,57 @@
import testparser;
import std.stdio;
import testutils;
int main()
{
return 0;
}
unittest
{
string input = "a, ((b)), b";
p_context_t context;
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
PStartS * start = p_result(&context);
assert(start.pItems1 !is null);
assert(start.pItems !is null);
PItemsS * items = start.pItems;
assert(items.pItem !is null);
assert(items.pItem.pToken1 !is null);
assert_eq(TOKEN_a, items.pItem.pToken1.token);
assert_eq(11, items.pItem.pToken1.pvalue);
assert(items.pItemsMore !is null);
PItemsMoreS * itemsmore = items.pItemsMore;
assert(itemsmore.pItem !is null);
assert(itemsmore.pItem.pItem !is null);
assert(itemsmore.pItem.pItem.pItem !is null);
assert(itemsmore.pItem.pItem.pItem.pToken1 !is null);
assert_eq(TOKEN_b, itemsmore.pItem.pItem.pItem.pToken1.token);
assert_eq(22, itemsmore.pItem.pItem.pItem.pToken1.pvalue);
assert(itemsmore.pItemsMore !is null);
itemsmore = itemsmore.pItemsMore;
assert(itemsmore.pItem !is null);
assert(itemsmore.pItem.pToken1 !is null);
assert_eq(TOKEN_b, itemsmore.pItem.pToken1.token);
assert_eq(22, itemsmore.pItem.pToken1.pvalue);
assert(itemsmore.pItemsMore is null);
input = "";
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start.pItems is null);
input = "2 1";
p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(&context));
start = p_result(&context);
assert(start.pItems !is null);
assert(start.pItems.pItem !is null);
assert(start.pItems.pItem.pDual !is null);
assert(start.pItems.pItem.pDual.pTwo1 !is null);
assert(start.pItems.pItem.pDual.pOne2 !is null);
assert(start.pItems.pItem.pDual.pTwo2 is null);
assert(start.pItems.pItem.pDual.pOne1 is null);
}

View File

@ -0,0 +1,88 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "abbccc";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(1, start->pT1->pToken->position.row);
assert_eq(1, start->pT1->pToken->position.col);
assert_eq(1, start->pT1->pToken->end_position.row);
assert_eq(1, start->pT1->pToken->end_position.col);
assert_eq(1, start->pT1->position.row);
assert_eq(1, start->pT1->position.col);
assert_eq(1, start->pT1->end_position.row);
assert_eq(1, start->pT1->end_position.col);
assert_eq(1, start->pT2->pToken->position.row);
assert_eq(2, start->pT2->pToken->position.col);
assert_eq(1, start->pT2->pToken->end_position.row);
assert_eq(3, start->pT2->pToken->end_position.col);
assert_eq(1, start->pT2->position.row);
assert_eq(2, start->pT2->position.col);
assert_eq(1, start->pT2->end_position.row);
assert_eq(3, start->pT2->end_position.col);
assert_eq(1, start->pT3->pToken->position.row);
assert_eq(4, start->pT3->pToken->position.col);
assert_eq(1, start->pT3->pToken->end_position.row);
assert_eq(6, start->pT3->pToken->end_position.col);
assert_eq(1, start->pT3->position.row);
assert_eq(4, start->pT3->position.col);
assert_eq(1, start->pT3->end_position.row);
assert_eq(6, start->pT3->end_position.col);
assert_eq(1, start->position.row);
assert_eq(1, start->position.col);
assert_eq(1, start->end_position.row);
assert_eq(6, start->end_position.col);
p_free_ast(start);
input = "\n\n bb\nc\ncc\n\n a";
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(&context) == P_SUCCESS);
start = p_result(&context);
assert_eq(3, start->pT1->pToken->position.row);
assert_eq(3, start->pT1->pToken->position.col);
assert_eq(3, start->pT1->pToken->end_position.row);
assert_eq(4, start->pT1->pToken->end_position.col);
assert_eq(3, start->pT1->position.row);
assert_eq(3, start->pT1->position.col);
assert_eq(3, start->pT1->end_position.row);
assert_eq(4, start->pT1->end_position.col);
assert_eq(4, start->pT2->pToken->position.row);
assert_eq(1, start->pT2->pToken->position.col);
assert_eq(5, start->pT2->pToken->end_position.row);
assert_eq(2, start->pT2->pToken->end_position.col);
assert_eq(4, start->pT2->position.row);
assert_eq(1, start->pT2->position.col);
assert_eq(5, start->pT2->end_position.row);
assert_eq(2, start->pT2->end_position.col);
assert_eq(7, start->pT3->pToken->position.row);
assert_eq(6, start->pT3->pToken->position.col);
assert_eq(7, start->pT3->pToken->end_position.row);
assert_eq(6, start->pT3->pToken->end_position.col);
assert_eq(7, start->pT3->position.row);
assert_eq(6, start->pT3->position.col);
assert_eq(7, start->pT3->end_position.row);
assert_eq(6, start->pT3->end_position.col);
assert_eq(3, start->position.row);
assert_eq(3, start->position.col);
assert_eq(7, start->end_position.row);
assert_eq(6, start->end_position.col);
p_free_ast(start);
return 0;
}

View File

@ -10,9 +10,10 @@ int main()
unittest unittest
{ {
string input = "abbccc"; string input = "abbccc";
p_context_t * context = p_context_new(input); p_context_t context;
assert(p_parse(context) == P_SUCCESS); p_context_init(&context, input);
Start start = p_result(context); assert(p_parse(&context) == P_SUCCESS);
Start * start = p_result(&context);
assert_eq(1, start.pT1.pToken.position.row); assert_eq(1, start.pT1.pToken.position.row);
assert_eq(1, start.pT1.pToken.position.col); assert_eq(1, start.pT1.pToken.position.col);
@ -46,12 +47,10 @@ unittest
assert_eq(1, start.end_position.row); assert_eq(1, start.end_position.row);
assert_eq(6, start.end_position.col); assert_eq(6, start.end_position.col);
p_context_delete(context);
input = "\n\n bb\nc\ncc\n\n a"; input = "\n\n bb\nc\ncc\n\n a";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
start = p_result(context); start = p_result(&context);
assert_eq(3, start.pT1.pToken.position.row); assert_eq(3, start.pT1.pToken.position.row);
assert_eq(3, start.pT1.pToken.position.col); assert_eq(3, start.pT1.pToken.position.col);
@ -84,6 +83,4 @@ unittest
assert_eq(3, start.position.col); assert_eq(3, start.position.col);
assert_eq(7, start.end_position.row); assert_eq(7, start.end_position.row);
assert_eq(6, start.end_position.col); assert_eq(6, start.end_position.col);
p_context_delete(context);
} }

View File

@ -5,29 +5,25 @@
int main() int main()
{ {
char const * input = "1 + 2 * 3 + 4"; char const * input = "1 + 2 * 3 + 4";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(11, p_result(context)); assert_eq(11, p_result(&context));
p_context_delete(context);
input = "1 * 2 ** 4 * 3"; input = "1 * 2 ** 4 * 3";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(48, p_result(context)); assert_eq(48, p_result(&context));
p_context_delete(context);
input = "(1 + 2) * 3 + 4"; input = "(1 + 2) * 3 + 4";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(13, p_result(context)); assert_eq(13, p_result(&context));
p_context_delete(context);
input = "(2 * 2) ** 3 + 4 + 5"; input = "(2 * 2) ** 3 + 4 + 5";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(73, p_result(context)); assert_eq(73, p_result(&context));
p_context_delete(context);
return 0; return 0;
} }

View File

@ -10,23 +10,23 @@ int main()
unittest unittest
{ {
string input = "1 + 2 * 3 + 4"; string input = "1 + 2 * 3 + 4";
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(11, p_result(context)); assert_eq(11, p_result(&context));
input = "1 * 2 ** 4 * 3"; input = "1 * 2 ** 4 * 3";
context = p_context_new(input); p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(48, p_result(context)); assert_eq(48, p_result(&context));
input = "(1 + 2) * 3 + 4"; input = "(1 + 2) * 3 + 4";
context = p_context_new(input); p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(13, p_result(context)); assert_eq(13, p_result(&context));
input = "(2 * 2) ** 3 + 4 + 5"; input = "(2 * 2) ** 3 + 4 + 5";
context = p_context_new(input); p_context_init(&context, input);
assert_eq(P_SUCCESS, p_parse(context)); assert_eq(P_SUCCESS, p_parse(&context));
assert_eq(73, p_result(context)); assert_eq(73, p_result(&context));
} }

View File

@ -1,16 +0,0 @@
use testparser::*;
fn main() {
let cases: [(&[u8], u64); 4] = [
(b"1 + 2 * 3 + 4", 11),
(b"1 * 2 ** 4 * 3", 48),
(b"(1 + 2) * 3 + 4", 13),
(b"(2 * 2) ** 3 + 4 + 5", 73),
];
for (input, expected) in cases {
let mut context = p_context_new(input);
assert_eq!(P_SUCCESS, p_parse(&mut context));
assert_eq!(expected, p_result(&context));
p_context_delete(context);
}
}

View File

@ -1,39 +0,0 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char input[128];
size_t i;
p_context_t * context;
Token token;
/* Enough tokens that the tree node arena is reallocated during the parse. */
memset(input, 0, sizeof(input));
for (i = 0u; i < 40u; i++)
{
input[i] = 'a';
}
context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1u, context->have_first);
assert(p_node_valid(context->first_item));
token = p_Item_pToken1(context->first_item);
assert(p_node_valid(token));
assert_eq(TOKEN_a, p_Token_token(token));
assert_eq(7u, p_Token_pvalue(token));
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, p_node_position(context->first_item).row);
assert_eq(1u, p_node_position(context->first_item).col);
p_context_delete(context);
return 0;
}

View File

@ -1,36 +0,0 @@
#include "testparser.h"
#include <cassert>
#include <cstring>
#include "testutils.h"
int main()
{
char input[128];
/* Enough tokens that the tree node arena is reallocated during the parse. */
memset(input, 0, sizeof(input));
for (size_t i = 0u; i < 40u; i++)
{
input[i] = 'a';
}
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1u, context->have_first);
assert(context->first_item.valid());
Token token = context->first_item.pToken1();
assert(token.valid());
assert_eq(TOKEN_a, token.token());
assert_eq(7u, token.pvalue());
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, context->first_item.position().row);
assert_eq(1u, context->first_item.position().col);
p_context_delete(context);
return 0;
}

View File

@ -1,35 +0,0 @@
import testparser;
import testutils;
int main()
{
return 0;
}
unittest
{
/* Enough tokens that the tree node array is reallocated during the parse. */
string input;
foreach (i; 0 .. 40)
{
input ~= "a";
}
p_context_t * context = p_context_new(input);
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1, context.have_first);
assert(context.first_item.valid);
Token token = context.first_item.pToken1;
assert(token.valid);
assert_eq(TOKEN_a, token.token);
assert_eq(7, token.pvalue);
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, context.first_item.position.row);
assert_eq(1u, context.first_item.position.col);
p_context_delete(context);
}

View File

@ -1,15 +0,0 @@
#include "testparser.h"
#include "testutils.h"
#include <string.h>
int main()
{
char const * input = "cbacba";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context));
size_t result = p_result(context);
assert_eq(0x932187932187, result);
p_context_delete(context);
return 0;
}

View File

@ -1,18 +0,0 @@
import testparser;
import std.stdio;
import testutils;
int main()
{
return 0;
}
unittest
{
string input = "cbacba";
p_context_t * context = p_context_new(input);
assert_eq(P_SUCCESS, p_parse(context));
size_t result = p_result(context);
assert_eq(0x932187932187, result);
p_context_delete(context);
}

View File

@ -1,8 +0,0 @@
use testparser::*;
fn main() {
let mut c = p_context_new(b"cbacba");
assert_eq!(P_SUCCESS, p_parse(&mut c));
assert_eq!(0x932187932187, p_result(&c));
p_context_delete(c);
}

View File

@ -1,15 +0,0 @@
#include "testparser.h"
#include <assert.h>
#include <stdio.h>
#include <string.h>
int main()
{
char const * input = " # comment 1\n# comment 2\na\n";
p_context_t * context;
context = p_context_new((uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS);
p_context_delete(context);
return 0;
}

View File

@ -1,16 +0,0 @@
import testparser;
import std.stdio;
import testutils;
int main()
{
return 0;
}
unittest
{
string input = " # comment 1\n# comment 2\na\n";
p_context_t * context;
context = p_context_new(input);
assert(p_parse(context) == P_SUCCESS);
}

View File

@ -1,7 +0,0 @@
use testparser::*;
fn main() {
let mut c = p_context_new(b" # comment 1\n# comment 2\na\n");
assert_eq!(P_SUCCESS, p_parse(&mut c));
p_context_delete(c);
}

View File

@ -5,43 +5,38 @@
int main() int main()
{ {
char const * input = "a 42"; char const * input = "a 42";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
p_context_delete(context);
input = "a\n123\na a"; input = "a\n123\na a";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_UNEXPECTED_TOKEN); assert(p_parse(&context) == P_UNEXPECTED_TOKEN);
assert(p_position(context).row == 3); assert(p_position(&context).row == 3);
assert(p_position(context).col == 4); assert(p_position(&context).col == 4);
assert(p_token(context) == TOKEN_a); assert(p_token(&context) == TOKEN_a);
p_context_delete(context);
input = "12"; input = "12";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_UNEXPECTED_TOKEN); assert(p_parse(&context) == P_UNEXPECTED_TOKEN);
assert(p_position(context).row == 1); assert(p_position(&context).row == 1);
assert(p_position(context).col == 1); assert(p_position(&context).col == 1);
assert(p_token(context) == TOKEN_num); assert(p_token(&context) == TOKEN_num);
p_context_delete(context);
input = "a 12\n\nab"; input = "a 12\n\nab";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_UNEXPECTED_INPUT); assert(p_parse(&context) == P_UNEXPECTED_INPUT);
assert(p_position(context).row == 3); assert(p_position(&context).row == 3);
assert(p_position(context).col == 2); assert(p_position(&context).col == 2);
p_context_delete(context);
input = "a 12\n\na\n\n77\na \xAA"; input = "a 12\n\na\n\n77\na \xAA";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_DECODE_ERROR); assert(p_parse(&context) == P_DECODE_ERROR);
assert(p_position(context).row == 6); assert(p_position(&context).row == 6);
assert(p_position(context).col == 5); assert(p_position(&context).col == 5);
assert(strcmp(p_token_names[TOKEN_a], "a") == 0); assert(strcmp(p_token_names[TOKEN_a], "a") == 0);
assert(strcmp(p_token_names[TOKEN_num], "num") == 0); assert(strcmp(p_token_names[TOKEN_num], "num") == 0);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,31 +9,31 @@ int main()
unittest unittest
{ {
string input = "a 42"; string input = "a 42";
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
input = "a\n123\na a"; input = "a\n123\na a";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_UNEXPECTED_TOKEN); assert(p_parse(&context) == P_UNEXPECTED_TOKEN);
assert(p_position(context) == p_position_t(3, 4)); assert(p_position(&context) == p_position_t(3, 4));
assert(p_token(context) == TOKEN_a); assert(p_token(&context) == TOKEN_a);
input = "12"; input = "12";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_UNEXPECTED_TOKEN); assert(p_parse(&context) == P_UNEXPECTED_TOKEN);
assert(p_position(context) == p_position_t(1, 1)); assert(p_position(&context) == p_position_t(1, 1));
assert(p_token(context) == TOKEN_num); assert(p_token(&context) == TOKEN_num);
input = "a 12\n\nab"; input = "a 12\n\nab";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_UNEXPECTED_INPUT); assert(p_parse(&context) == P_UNEXPECTED_INPUT);
assert(p_position(context) == p_position_t(3, 2)); assert(p_position(&context) == p_position_t(3, 2));
input = "a 12\n\na\n\n77\na \xAA"; input = "a 12\n\na\n\n77\na \xAA";
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_DECODE_ERROR); assert(p_parse(&context) == P_DECODE_ERROR);
assert(p_position(context) == p_position_t(6, 5)); assert(p_position(&context) == p_position_t(6, 5));
assert(p_token_names[TOKEN_a] == "a"); assert(p_token_names[TOKEN_a] == "a");
assert(p_token_names[TOKEN_num] == "num"); assert(p_token_names[TOKEN_num] == "num");

View File

@ -1,35 +0,0 @@
use testparser::*;
fn main() {
let mut c = p_context_new(b"a 42");
assert_eq!(P_SUCCESS, p_parse(&mut c));
p_context_delete(c);
let mut c = p_context_new(b"a\n123\na a");
assert_eq!(P_UNEXPECTED_TOKEN, p_parse(&mut c));
assert_eq!(3, p_position(&c).row);
assert_eq!(4, p_position(&c).col);
assert_eq!(TOKEN_a, p_token(&c));
p_context_delete(c);
let mut c = p_context_new(b"12");
assert_eq!(P_UNEXPECTED_TOKEN, p_parse(&mut c));
assert_eq!(1, p_position(&c).row);
assert_eq!(1, p_position(&c).col);
assert_eq!(TOKEN_num, p_token(&c));
p_context_delete(c);
let mut c = p_context_new(b"a 12\n\nab");
assert_eq!(P_UNEXPECTED_INPUT, p_parse(&mut c));
assert_eq!(3, p_position(&c).row);
assert_eq!(2, p_position(&c).col);
p_context_delete(c);
let mut c = p_context_new(b"a 12\n\na\n\n77\na \xAA");
assert_eq!(P_DECODE_ERROR, p_parse(&mut c));
assert_eq!(6, p_position(&c).row);
assert_eq!(5, p_position(&c).col);
assert_eq!("a", p_token_names[TOKEN_a as usize]);
assert_eq!("num", p_token_names[TOKEN_num as usize]);
p_context_delete(c);
}

View File

@ -6,9 +6,8 @@
int main() int main()
{ {
char const * input = "foo1\nbar2"; char const * input = "foo1\nbar2";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,7 +9,7 @@ int main()
unittest unittest
{ {
string input = "foo1\nbar2"; string input = "foo1\nbar2";
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
} }

View File

@ -1,7 +0,0 @@
use testparser::*;
fn main() {
let mut c = p_context_new(b"foo1\nbar2");
assert_eq!(P_SUCCESS, p_parse(&mut c));
p_context_delete(c);
}

View File

@ -0,0 +1,19 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char const * input = "ab";
p_context_t context;
p_context_init(&context, (uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(&context));
Start * start = p_result(&context);
assert(start->a != NULL);
assert(*start->a->pvalue == 1);
assert(start->b != NULL);
assert(*start->b->pvalue == 2);
p_free_ast(start);
}

View File

@ -1,60 +0,0 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
/* Grammar (simple):
* drop /\\s+/;
* token a; token b;
* Start -> a b;
*
* Verifies that p_input_index() reports the parser/lexer's current byte
* offset into the input text. */
/* Fresh context: input_index starts at 0. */
{
char const * input = "ab";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(0u, p_input_index(context));
p_context_delete(context);
}
/* After each successful lex the byte offset advances past the token. */
{
char const * input = "a b";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
p_token_info_t token_info;
assert(p_lex(context, &token_info) == P_SUCCESS);
assert_eq((size_t)TOKEN_a, (size_t)token_info.token);
assert_eq(1u, p_input_index(context));
assert(p_lex(context, &token_info) == P_SUCCESS);
assert_eq((size_t)TOKEN_b, (size_t)token_info.token);
/* The dropped space between `a` and `b` advances input_index too. */
assert_eq(3u, p_input_index(context));
p_context_delete(context);
}
/* After a full successful parse, input_index has reached the end. */
{
char const * input = "ab";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
assert(p_parse_Start(context) == P_SUCCESS);
assert_eq(2u, p_input_index(context));
p_context_delete(context);
}
/* When parse_inner completes via a follow token, the follow token is not
* consumed, so input_index points at the start of the follow token. */
{
char const * input = "abb";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
p_token_t follow_tokens[] = { TOKEN_b };
assert(p_parse_inner_Start(context, follow_tokens, 1u) == P_SUCCESS);
assert_eq(2u, p_input_index(context));
p_context_delete(context);
}
return 0;
}

View File

@ -1,51 +0,0 @@
import testparser;
import std.stdio;
import testutils;
int main()
{
return 0;
}
unittest
{
/* See test_input_index.c for details on the grammar and cases. */
/* Fresh context: input_index starts at 0. */
{
string input = "ab";
p_context_t * context = p_context_new(input);
assert(p_input_index(context) == 0);
}
/* After each successful lex the byte offset advances past the token. */
{
string input = "a b";
p_context_t * context = p_context_new(input);
p_token_info_t token_info;
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_a);
assert(p_input_index(context) == 1);
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_b);
assert(p_input_index(context) == 3);
}
/* After a full successful parse, input_index has reached the end. */
{
string input = "ab";
p_context_t * context = p_context_new(input);
assert(p_parse_Start(context) == P_SUCCESS);
assert(p_input_index(context) == 2);
}
/* When parse_inner completes via a follow token, the follow token is not
* consumed, so input_index points at the start of the follow token. */
{
string input = "abb";
p_context_t * context = p_context_new(input);
p_token_t[] follow_tokens = [TOKEN_b];
assert(p_parse_inner_Start(context, follow_tokens) == P_SUCCESS);
assert(p_input_index(context) == 2);
}
}

View File

@ -1,28 +0,0 @@
use testparser::*;
fn main() {
let c = p_context_new(b"ab");
assert_eq!(0, p_input_index(&c));
p_context_delete(c);
let mut c = p_context_new(b"a b");
let mut ti = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_a, ti.token);
assert_eq!(1, p_input_index(&c));
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_b, ti.token);
assert_eq!(3, p_input_index(&c));
p_context_delete(c);
let mut c = p_context_new(b"ab");
assert_eq!(P_SUCCESS, p_parse_Start(&mut c));
assert_eq!(2, p_input_index(&c));
p_context_delete(c);
let mut c = p_context_new(b"abb");
let follow = [TOKEN_b];
assert_eq!(P_SUCCESS, p_parse_inner_Start(&mut c, &follow));
assert_eq!(2, p_input_index(&c));
p_context_delete(c);
}

View File

@ -38,75 +38,73 @@ int main()
p_token_info_t token_info; p_token_info_t token_info;
char const * input = "5 + 4 * \n677 + 567"; char const * input = "5 + 4 * \n677 + 567";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 1u); assert(token_info.position.row == 1u);
assert(token_info.position.col == 1u); assert(token_info.position.col == 1u);
assert(token_info.end_position.row == 1u); assert(token_info.end_position.row == 1u);
assert(token_info.end_position.col == 1u); assert(token_info.end_position.col == 1u);
assert(token_info.length == 1u); assert(token_info.length == 1u);
assert(token_info.token == TOKEN_int); assert(token_info.token == TOKEN_int);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 1u); assert(token_info.position.row == 1u);
assert(token_info.position.col == 3u); assert(token_info.position.col == 3u);
assert(token_info.end_position.row == 1u); assert(token_info.end_position.row == 1u);
assert(token_info.end_position.col == 3u); assert(token_info.end_position.col == 3u);
assert(token_info.length == 1u); assert(token_info.length == 1u);
assert(token_info.token == TOKEN_plus); assert(token_info.token == TOKEN_plus);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 1u); assert(token_info.position.row == 1u);
assert(token_info.position.col == 5u); assert(token_info.position.col == 5u);
assert(token_info.end_position.row == 1u); assert(token_info.end_position.row == 1u);
assert(token_info.end_position.col == 5u); assert(token_info.end_position.col == 5u);
assert(token_info.length == 1u); assert(token_info.length == 1u);
assert(token_info.token == TOKEN_int); assert(token_info.token == TOKEN_int);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 1u); assert(token_info.position.row == 1u);
assert(token_info.position.col == 7u); assert(token_info.position.col == 7u);
assert(token_info.end_position.row == 1u); assert(token_info.end_position.row == 1u);
assert(token_info.end_position.col == 7u); assert(token_info.end_position.col == 7u);
assert(token_info.length == 1u); assert(token_info.length == 1u);
assert(token_info.token == TOKEN_times); assert(token_info.token == TOKEN_times);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 2u); assert(token_info.position.row == 2u);
assert(token_info.position.col == 1u); assert(token_info.position.col == 1u);
assert(token_info.end_position.row == 2u); assert(token_info.end_position.row == 2u);
assert(token_info.end_position.col == 3u); assert(token_info.end_position.col == 3u);
assert(token_info.length == 3u); assert(token_info.length == 3u);
assert(token_info.token == TOKEN_int); assert(token_info.token == TOKEN_int);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 2u); assert(token_info.position.row == 2u);
assert(token_info.position.col == 5u); assert(token_info.position.col == 5u);
assert(token_info.end_position.row == 2u); assert(token_info.end_position.row == 2u);
assert(token_info.end_position.col == 5u); assert(token_info.end_position.col == 5u);
assert(token_info.length == 1u); assert(token_info.length == 1u);
assert(token_info.token == TOKEN_plus); assert(token_info.token == TOKEN_plus);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 2u); assert(token_info.position.row == 2u);
assert(token_info.position.col == 7u); assert(token_info.position.col == 7u);
assert(token_info.end_position.row == 2u); assert(token_info.end_position.row == 2u);
assert(token_info.end_position.col == 9u); assert(token_info.end_position.col == 9u);
assert(token_info.length == 3u); assert(token_info.length == 3u);
assert(token_info.token == TOKEN_int); assert(token_info.token == TOKEN_int);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 2u); assert(token_info.position.row == 2u);
assert(token_info.position.col == 10u); assert(token_info.position.col == 10u);
assert(token_info.end_position.row == 2u); assert(token_info.end_position.row == 2u);
assert(token_info.end_position.col == 10u); assert(token_info.end_position.col == 10u);
assert(token_info.length == 0u); assert(token_info.length == 0u);
assert(token_info.token == TOKEN___EOF); assert(token_info.token == TOKEN___EOF);
p_context_delete(context);
context = p_context_new((uint8_t const *)"", 0u); p_context_init(&context, (uint8_t const *)"", 0u);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info.position.row == 1u); assert(token_info.position.row == 1u);
assert(token_info.position.col == 1u); assert(token_info.position.col == 1u);
assert(token_info.end_position.row == 1u); assert(token_info.end_position.row == 1u);
assert(token_info.end_position.col == 1u); assert(token_info.end_position.col == 1u);
assert(token_info.length == 0u); assert(token_info.length == 0u);
assert(token_info.token == TOKEN___EOF); assert(token_info.token == TOKEN___EOF);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -44,26 +44,26 @@ unittest
{ {
p_token_info_t token_info; p_token_info_t token_info;
string input = "5 + 4 * \n677 + 567"; string input = "5 + 4 * \n677 + 567";
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(1, 1), p_position_t(1, 1), 1, TOKEN_int)); assert(token_info == p_token_info_t(p_position_t(1, 1), p_position_t(1, 1), 1, TOKEN_int));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(1, 3), p_position_t(1, 3), 1, TOKEN_plus)); assert(token_info == p_token_info_t(p_position_t(1, 3), p_position_t(1, 3), 1, TOKEN_plus));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(1, 5), p_position_t(1, 5), 1, TOKEN_int)); assert(token_info == p_token_info_t(p_position_t(1, 5), p_position_t(1, 5), 1, TOKEN_int));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(1, 7), p_position_t(1, 7), 1, TOKEN_times)); assert(token_info == p_token_info_t(p_position_t(1, 7), p_position_t(1, 7), 1, TOKEN_times));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(2, 1), p_position_t(2, 3), 3, TOKEN_int)); assert(token_info == p_token_info_t(p_position_t(2, 1), p_position_t(2, 3), 3, TOKEN_int));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(2, 5), p_position_t(2, 5), 1, TOKEN_plus)); assert(token_info == p_token_info_t(p_position_t(2, 5), p_position_t(2, 5), 1, TOKEN_plus));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(2, 7), p_position_t(2, 9), 3, TOKEN_int)); assert(token_info == p_token_info_t(p_position_t(2, 7), p_position_t(2, 9), 3, TOKEN_int));
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(2, 10), p_position_t(2, 10), 0, TOKEN___EOF)); assert(token_info == p_token_info_t(p_position_t(2, 10), p_position_t(2, 10), 0, TOKEN___EOF));
context = p_context_new(""); p_context_init(&context, "");
assert(p_lex(context, &token_info) == P_SUCCESS); assert(p_lex(&context, &token_info) == P_SUCCESS);
assert(token_info == p_token_info_t(p_position_t(1, 1), p_position_t(1, 1), 0, TOKEN___EOF)); assert(token_info == p_token_info_t(p_position_t(1, 1), p_position_t(1, 1), 0, TOKEN___EOF));
} }

View File

@ -1,49 +0,0 @@
use testparser::*;
fn chk(ti: &p_token_info_t, row: u32, col: u32, erow: u32, ecol: u32, len: usize, token: p_token_t) {
assert_eq!(row, ti.position.row);
assert_eq!(col, ti.position.col);
assert_eq!(erow, ti.end_position.row);
assert_eq!(ecol, ti.end_position.col);
assert_eq!(len, ti.length);
assert_eq!(token, ti.token);
}
fn main() {
let mut cp: p_code_point_t = 0;
let mut cpl: u8 = 0;
assert_eq!(P_SUCCESS, p_decode_code_point(b"5", &mut cp, &mut cpl));
assert_eq!('5' as u32, cp);
assert_eq!(1, cpl);
assert_eq!(P_EOF, p_decode_code_point(b"", &mut cp, &mut cpl));
assert_eq!(P_SUCCESS, p_decode_code_point(b"\xC2\xA9", &mut cp, &mut cpl));
assert_eq!(0xA9, cp);
assert_eq!(2, cpl);
assert_eq!(P_SUCCESS, p_decode_code_point(b"\xf0\x9f\xa7\xa1", &mut cp, &mut cpl));
assert_eq!(0x1F9E1, cp);
assert_eq!(4, cpl);
assert_eq!(P_DECODE_ERROR, p_decode_code_point(b"\xf0\x9f\x27", &mut cp, &mut cpl));
assert_eq!(P_DECODE_ERROR, p_decode_code_point(b"\xf0\x9f\xa7\xFF", &mut cp, &mut cpl));
assert_eq!(P_DECODE_ERROR, p_decode_code_point(b"\xfe", &mut cp, &mut cpl));
let mut context = p_context_new(b"5 + 4 * \n677 + 567");
let mut ti = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 1, 1, 1, 1, 1, TOKEN_int);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 1, 3, 1, 3, 1, TOKEN_plus);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 1, 5, 1, 5, 1, TOKEN_int);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 1, 7, 1, 7, 1, TOKEN_times);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 2, 1, 2, 3, 3, TOKEN_int);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 2, 5, 2, 5, 1, TOKEN_plus);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 2, 7, 2, 9, 3, TOKEN_int);
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 2, 10, 2, 10, 0, TOKEN___EOF);
p_context_delete(context);
let mut context = p_context_new(b"");
assert_eq!(P_SUCCESS, p_lex(&mut context, &mut ti)); chk(&ti, 1, 1, 1, 1, 0, TOKEN___EOF);
p_context_delete(context);
}

View File

@ -6,11 +6,10 @@
int main() int main()
{ {
char const * input = "identifier_123"; char const * input = "identifier_123";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
printf("pass1\n"); printf("pass1\n");
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,8 +9,8 @@ int main()
unittest unittest
{ {
string input = `identifier_123`; string input = `identifier_123`;
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
writeln("pass1"); writeln("pass1");
} }

View File

@ -1,8 +0,0 @@
use testparser::*;
fn main() {
let mut context = p_context_new(b"identifier_123");
assert_eq!(P_SUCCESS, p_parse(&mut context));
println!("pass1");
p_context_delete(context);
}

View File

@ -6,17 +6,15 @@
int main() int main()
{ {
char const * input = "abc \"a string\" def"; char const * input = "abc \"a string\" def";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
printf("pass1\n"); printf("pass1\n");
p_context_delete(context);
input = "abc \"abc def\" def"; input = "abc \"abc def\" def";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
printf("pass2\n"); printf("pass2\n");
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,13 +9,13 @@ int main()
unittest unittest
{ {
string input = `abc "a string" def`; string input = `abc "a string" def`;
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
writeln("pass1"); writeln("pass1");
input = `abc "abc def" def`; input = `abc "abc def" def`;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
writeln("pass2"); writeln("pass2");
} }

View File

@ -1,13 +0,0 @@
use testparser::*;
fn main() {
let mut context = p_context_new(b"abc \"a string\" def");
assert_eq!(P_SUCCESS, p_parse(&mut context));
println!("pass1");
p_context_delete(context);
let mut context = p_context_new(b"abc \"abc def\" def");
assert_eq!(P_SUCCESS, p_parse(&mut context));
println!("pass2");
p_context_delete(context);
}

View File

@ -6,17 +6,15 @@
int main() int main()
{ {
char const * input = "abc.def"; char const * input = "abc.def";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
printf("pass1\n"); printf("pass1\n");
p_context_delete(context);
input = "abc . abc"; input = "abc . abc";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
printf("pass2\n"); printf("pass2\n");
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,13 +9,13 @@ int main()
unittest unittest
{ {
string input = `abc.def`; string input = `abc.def`;
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
writeln("pass1"); writeln("pass1");
input = `abc . abc`; input = `abc . abc`;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
writeln("pass2"); writeln("pass2");
} }

View File

@ -1,13 +0,0 @@
use testparser::*;
fn main() {
let mut context = p_context_new(b"abc.def");
assert_eq!(P_SUCCESS, p_parse(&mut context));
println!("pass1");
p_context_delete(context);
let mut context = p_context_new(b"abc . abc");
assert_eq!(P_SUCCESS, p_parse(&mut context));
println!("pass2");
p_context_delete(context);
}

View File

@ -1,50 +0,0 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
int main()
{
char const * input = "abc\n defg hi\n!";
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
p_token_info_t token_info;
/* First token "abc" on row 1, cols 1-3. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context->last_start.row == 1u);
assert(context->last_start.col == 1u);
assert(context->last_end.row == 1u);
assert(context->last_end.col == 3u);
/* The lexer code block observed the same positions reported to the caller. */
assert(context->last_start.row == token_info.position.row);
assert(context->last_start.col == token_info.position.col);
assert(context->last_end.row == token_info.end_position.row);
assert(context->last_end.col == token_info.end_position.col);
/* Second token "defg" on row 2, cols 3-6. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context->last_start.row == 2u);
assert(context->last_start.col == 3u);
assert(context->last_end.row == 2u);
assert(context->last_end.col == 6u);
/* Third token "hi" on row 2, cols 8-9. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context->last_start.row == 2u);
assert(context->last_start.col == 8u);
assert(context->last_end.row == 2u);
assert(context->last_end.col == 9u);
/* The "!" stop token terminates the lexer. The context input text position
* must not be updated when the lexer user code requests termination, so it
* still points at the "!" token on row 3, col 1. */
assert(p_lex(context, &token_info) == P_USER_TERMINATED);
assert(p_user_terminate_code(context) == 42u);
assert(context->text_position.row == 3u);
assert(context->text_position.col == 1u);
p_context_delete(context);
return 0;
}

View File

@ -1,42 +0,0 @@
import testparser;
import std.stdio;
int main()
{
return 0;
}
unittest
{
string input = "abc\n defg hi\n!";
p_context_t * context = p_context_new(input);
p_token_info_t token_info;
/* First token "abc" on row 1, cols 1-3. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context.last_start == p_position_t(1, 1));
assert(context.last_end == p_position_t(1, 3));
/* The lexer code block observed the same positions reported to the caller. */
assert(context.last_start == token_info.position);
assert(context.last_end == token_info.end_position);
/* Second token "defg" on row 2, cols 3-6. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context.last_start == p_position_t(2, 3));
assert(context.last_end == p_position_t(2, 6));
/* Third token "hi" on row 2, cols 8-9. */
assert(p_lex(context, &token_info) == P_SUCCESS);
assert(token_info.token == TOKEN_word);
assert(context.last_start == p_position_t(2, 8));
assert(context.last_end == p_position_t(2, 9));
/* The "!" stop token terminates the lexer. The context input text position
* must not be updated when the lexer user code requests termination, so it
* still points at the "!" token on row 3, col 1. */
assert(p_lex(context, &token_info) == P_USER_TERMINATED);
assert(p_user_terminate_code(context) == 42u);
assert(context.text_position == p_position_t(3, 1));
}

View File

@ -1,38 +0,0 @@
use testparser::*;
fn main() {
let mut c = p_context_new(b"abc\n defg hi\n!");
let mut ti = p_token_info_t::default();
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_word, ti.token);
assert_eq!(1, c.last_start.row);
assert_eq!(1, c.last_start.col);
assert_eq!(1, c.last_end.row);
assert_eq!(3, c.last_end.col);
assert_eq!(c.last_start.row, ti.position.row);
assert_eq!(c.last_start.col, ti.position.col);
assert_eq!(c.last_end.row, ti.end_position.row);
assert_eq!(c.last_end.col, ti.end_position.col);
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_word, ti.token);
assert_eq!(2, c.last_start.row);
assert_eq!(3, c.last_start.col);
assert_eq!(2, c.last_end.row);
assert_eq!(6, c.last_end.col);
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_word, ti.token);
assert_eq!(2, c.last_start.row);
assert_eq!(8, c.last_start.col);
assert_eq!(2, c.last_end.row);
assert_eq!(9, c.last_end.col);
assert_eq!(P_USER_TERMINATED, p_lex(&mut c, &mut ti));
assert_eq!(42, p_user_terminate_code(&c));
assert_eq!(3, p_position(&c).row);
assert_eq!(1, p_position(&c).col);
p_context_delete(c);
}

View File

@ -5,17 +5,15 @@
int main() int main()
{ {
char const * input = "x"; char const * input = "x";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 1u); assert(p_result(&context) == 1u);
p_context_delete(context);
input = "fabulous"; input = "fabulous";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 8u); assert(p_result(&context) == 8u);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,13 +9,13 @@ int main()
unittest unittest
{ {
string input = `x`; string input = `x`;
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 1u); assert(p_result(&context) == 1u);
input = `fabulous`; input = `fabulous`;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 8u); assert(p_result(&context) == 8u);
} }

View File

@ -1,13 +0,0 @@
use testparser::*;
fn main() {
let mut context = p_context_new(b"x");
assert_eq!(P_SUCCESS, p_parse(&mut context));
assert_eq!(1, p_result(&context));
p_context_delete(context);
let mut context = p_context_new(b"fabulous");
assert_eq!(P_SUCCESS, p_parse(&mut context));
assert_eq!(8, p_result(&context));
p_context_delete(context);
}

View File

@ -5,16 +5,14 @@
int main() int main()
{ {
char const * input = "x"; char const * input = "x";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_UNEXPECTED_INPUT); assert(p_parse(&context) == P_UNEXPECTED_INPUT);
p_context_delete(context);
input = "123"; input = "123";
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 123u); assert(p_result(&context) == 123u);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,12 +9,12 @@ int main()
unittest unittest
{ {
string input = `x`; string input = `x`;
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_UNEXPECTED_INPUT); assert(p_parse(&context) == P_UNEXPECTED_INPUT);
input = `123`; input = `123`;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
assert(p_result(context) == 123u); assert(p_result(&context) == 123u);
} }

View File

@ -1,12 +0,0 @@
use testparser::*;
fn main() {
let mut context = p_context_new(b"x");
assert_eq!(P_UNEXPECTED_INPUT, p_parse(&mut context));
p_context_delete(context);
let mut context = p_context_new(b"123");
assert_eq!(P_SUCCESS, p_parse(&mut context));
assert_eq!(123, p_result(&context));
p_context_delete(context);
}

View File

@ -1,118 +0,0 @@
#include "testparser.h"
#include "testutils.h"
#include <string.h>
#include <assert.h>
#include <stddef.h>
#include <stdbool.h>
static p_context_t * context;
size_t n_tokens;
p_token_info_t token_infos[10];
/* Capture the macro body tokens (everything up to the closing '}') into
* token_infos[]. Called from mylexfn() right after the definition's '{' has
* been lexed, so the input cursor is positioned at the first body token. */
static void capture_macro_body(void)
{
n_tokens = 0u;
for (;;)
{
size_t result = p_lex(context, &token_infos[n_tokens]);
assert_eq(result, P_SUCCESS);
if (token_infos[n_tokens].token == TOKEN_rbrace)
{
break;
}
n_tokens++;
assert(n_tokens < sizeof(token_infos) / sizeof(token_infos[0]));
}
}
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info)
{
static bool defining;
static bool expanding;
static size_t expand_i;
for (;;)
{
if (expanding)
{
size_t ei = expand_i++;
if (expand_i >= n_tokens)
{
expanding = false;
}
*out_token_info = token_infos[ei];
return P_SUCCESS;
}
size_t lex_result = p_lex(context, out_token_info);
if (lex_result != P_SUCCESS)
{
return lex_result;
}
switch (out_token_info->token)
{
case TOKEN_macro:
/* Start of a macro definition: "macro macroname { ... }". */
defining = true;
break;
case TOKEN_macroname:
if (!defining)
{
/* Use of a macro: replay its captured body tokens instead of
* returning the macroname to the parser. */
expanding = true;
expand_i = 0u;
continue;
}
/* Definition name: pass through and keep waiting for '{'. */
break;
case TOKEN_lbrace:
if (defining)
{
/* Consume and store the macro body now, before the parser gets
* a chance to read its lookahead token (which would otherwise
* swallow the first body token). */
capture_macro_body();
defining = false;
}
break;
default:
defining = false;
break;
}
return lex_result;
}
}
size_t n_nums;
int nums[10];
void record(int v)
{
nums[n_nums++] = v;
}
int main()
{
char const * input =
"macro @m { 23 + 200 }\n"
"66 + 100\n"
"@m\n"
"33 + 55\n"
"@m\n";
context = p_context_new((uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS);
p_context_delete(context);
assert_eq(n_nums, 4);
assert_eq(nums[0], 166);
assert_eq(nums[1], 223);
assert_eq(nums[2], 88);
assert_eq(nums[3], 223);
return 0;
}

View File

@ -1,116 +0,0 @@
import testparser;
import testutils;
size_t n_tokens;
p_token_info_t[10] token_infos;
// Capture the macro body tokens (everything up to the closing '}') into
// token_infos[]. Called from mylexfn() right after the definition's '{' has
// been lexed, so the input cursor is positioned at the first body token.
void capture_macro_body(p_context_t * context)
{
n_tokens = 0u;
for (;;)
{
size_t result = p_lex(context, &token_infos[n_tokens]);
assert(result == P_SUCCESS);
if (token_infos[n_tokens].token == TOKEN_rbrace)
{
break;
}
n_tokens++;
assert(n_tokens < token_infos.length);
}
}
size_t mylexfn(p_context_t * context, p_token_info_t * out_token_info)
{
static bool defining;
static bool expanding;
static size_t expand_i;
for (;;)
{
if (expanding)
{
size_t ei = expand_i++;
if (expand_i >= n_tokens)
{
expanding = false;
}
*out_token_info = token_infos[ei];
return P_SUCCESS;
}
size_t lex_result = p_lex(context, out_token_info);
if (lex_result != P_SUCCESS)
{
return lex_result;
}
switch (out_token_info.token)
{
case TOKEN_macro:
// Start of a macro definition: "macro macroname { ... }".
defining = true;
break;
case TOKEN_macroname:
if (!defining)
{
// Use of a macro: replay its captured body tokens instead of
// returning the macroname to the parser.
expanding = true;
expand_i = 0u;
continue;
}
// Definition name: pass through and keep waiting for '{'.
break;
case TOKEN_lbrace:
if (defining)
{
// Consume and store the macro body now, before the parser gets
// a chance to read its lookahead token (which would otherwise
// swallow the first body token).
capture_macro_body(context);
defining = false;
}
break;
default:
defining = false;
break;
}
return lex_result;
}
}
size_t n_nums;
int[10] nums;
void record(int v)
{
nums[n_nums++] = v;
}
int main()
{
return 0;
}
unittest
{
string input =
"macro @m { 23 + 200 }\n" ~
"66 + 100\n" ~
"@m\n" ~
"33 + 55\n" ~
"@m\n";
p_context_t * context = p_context_new(input);
assert(p_parse(context) == P_SUCCESS);
p_context_delete(context);
assert(n_nums == 4);
assert(nums[0] == 166);
assert(nums[1] == 223);
assert(nums[2] == 88);
assert(nums[3] == 223);
}

View File

@ -1,9 +0,0 @@
use testparser::*;
fn main() {
let input = b"macro @m { 23 + 200 }\n66 + 100\n@m\n33 + 55\n@m\n";
let mut c = p_context_new(input);
assert_eq!(P_SUCCESS, p_parse(&mut c));
assert_eq!(vec![166, 223, 88, 223], c.nums);
p_context_delete(c);
}

View File

@ -5,10 +5,9 @@
int main() int main()
{ {
char const * input = "\a\b\t\n\v\f\rt"; char const * input = "\a\b\t\n\v\f\rt";
p_context_t * context; p_context_t context;
context = p_context_new((uint8_t const *)input, strlen(input)); p_context_init(&context, (uint8_t const *)input, strlen(input));
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
p_context_delete(context);
return 0; return 0;
} }

View File

@ -9,7 +9,7 @@ int main()
unittest unittest
{ {
string input = "\a\b\t\n\v\f\rt"; string input = "\a\b\t\n\v\f\rt";
p_context_t * context; p_context_t context;
context = p_context_new(input); p_context_init(&context, input);
assert(p_parse(context) == P_SUCCESS); assert(p_parse(&context) == P_SUCCESS);
} }

Some files were not shown because too many files have changed in this diff Show More