Compare commits

..

No commits in common. "master" and "rust" have entirely different histories.
master ... rust

18 changed files with 74 additions and 515 deletions

View File

@ -31,9 +31,6 @@ jobs:
with:
ruby-version: ${{ matrix.ruby-version }}
- name: Set up Rust
uses: dtolnay/rust-toolchain@stable
- name: Install dependencies
run: bundle install

View File

@ -1,30 +1,22 @@
## v5.1.0
### New Features
- Add a `node_id()` accessor to the C++ and D tree node handle types, for node
identity comparison. This matches the existing `p_node_id()` macro (C) and
`node_id()` method (Rust).
## v5.0.0
### New Features
- Add Rust target language output.
- Add Rust language detection in propane.vim.
### API Changes
- The matched text argument passed to lexer user code blocks is now named
`match_text` instead of `match`, since `match` is a keyword in Rust. The
`match_text` for every target language. It was previously named `match` for
C, C++, and D, and `match_` for Rust. Any lexer user code block which
references the matched text must be updated to use the new name. The
`match_length` argument (C and C++) is unchanged.
- Tree generation mode now stores all tree nodes in a compact arena owned by
the parser context (a flat node array plus a shared child-link array).
This replaces the previous design of one heap allocation per node with
layout-punned typed structs.
- Tree nodes are now referenced by lightweight handles rather than pointers.
`p_result()` and the field accessors now return handle values in tree
generation mode.
`p_result()` and the field accessors return handle values.
- The whole tree is freed together with the context by `p_context_delete()`.
The `p_tree_delete()` / `p_tree_delete_XXX()` functions have been removed;
tree node handles are only valid while the context is alive.
@ -42,11 +34,6 @@
Reference child fields through the target-language accessors described above
rather than through struct pointer members.
### Improvements
- Improve D language detection in propane.vim
- Speed up specs
## v4.8.1
### Fixes

View File

@ -14,6 +14,8 @@ from `match` to `match_text` for all target languages.
`match_text` (for example `$$ = match[0];` becomes `$$ = match_text[0];`).
The `match_length` argument (C, C++) is unchanged.
This rename only affects lexer user code blocks; parser rule user code blocks
never had a matched text argument.
### Tree memory management
@ -49,44 +51,6 @@ Reference child fields through the target-language accessors above (for example
`$$->pA->pToken1->pvalue` becomes `p_tree_walk_Start($$, pA, pToken1, pvalue)`
in C, `$$.pA().pToken1().pvalue()` in C++, and `$$.pA.pToken1.pvalue` in D).
### Pointers into tree node storage
Tree nodes previously each had their own allocation, so a pointer to a node
stayed valid for the life of the tree. They are now held in a single array
which is reallocated as it grows, so a pointer or reference into that array may
be invalidated whenever a new node is created.
New nodes are created while parsing, so this matters for a pointer taken in a
tree-mode parser rule user code block, which runs before the parse has
finished. Keep the node handle instead, which stores a node ID rather than an
address and stays valid, and obtain the pointer from it when it is needed.
For example, replace a saved pointer:
```
context_user_fields <<
p_node_data_t * saved;
>>
Items -> Items a << ${context.saved} = p_node_data($$); >>
```
with a saved handle:
```
context_user_fields <<
Items saved_node;
>>
Items -> Items a << ${context.saved_node} = $$; >>
```
```
p_node_data_t * data = p_node_data(context->saved_node);
```
Once parsing has finished, no further nodes are created, so a pointer obtained
after `p_parse()` returns stays valid until the context is deleted, as long as
no further parsing is performed with the same context.
## v4.0.0
### API Changes

View File

@ -149,12 +149,6 @@ public struct <%= @grammar.tree_prefix %>Token<%= @grammar.tree_suffix %>
return __id != 0u;
}
/** Return the node ID (for identity comparison). */
@property <%= @grammar.prefix %>node_id_t node_id()
{
return __id;
}
/** Access the underlying node record (token, pvalue, and user fields). */
@property ref <%= @grammar.prefix %>node_data_t __node()
{
@ -183,12 +177,6 @@ public struct <%= @grammar.tree_prefix %><%= rule_set.name %><%= @grammar.tree_s
return __id != 0u;
}
/** Return the node ID (for identity comparison). */
@property <%= @grammar.prefix %>node_id_t node_id()
{
return __id;
}
/** Text position of the first code point spanned by this node. */
@property <%= @grammar.prefix %>position_t position()
{

View File

@ -135,19 +135,13 @@ typedef struct
<%= @grammar.prefix %>value_t pvalue;
} <%= @grammar.prefix %>token_info_t;
typedef struct <%= @grammar.prefix %>context_s <%= @grammar.prefix %>context_t;
<% if @grammar.tree %>
<%= c_tree_handle_types_header %>
<% end %>
/**
* Lexer and parser context.
*
* The user must allocate an instance of this structure and pass it to any
* public API function.
*/
struct <%= @grammar.prefix %>context_s
typedef struct
{
/* Lexer context data. */
@ -200,7 +194,7 @@ struct <%= @grammar.prefix %>context_s
size_t user_terminate_code;
<%= @grammar.context_user_fields %>
};
} <%= @grammar.prefix %>context_t;
<% if @grammar.tree %>
<%= c_tree_types_header %>

View File

@ -5,6 +5,9 @@
#![allow(non_camel_case_types)]
#![allow(non_snake_case)]
#![allow(non_upper_case_globals)]
#![allow(dead_code)]
#![allow(unused_variables)]
#![allow(unused_parens)]
/**************************************************************************
* User code blocks
@ -46,7 +49,7 @@ pub type <%= @grammar.prefix %>code_point_t = u32;
*
* This is useful for reporting errors, etc...
*/
#[derive(Clone, Copy, Debug, Default, PartialEq)]
#[derive(Clone, Copy, Default, PartialEq)]
pub struct <%= @grammar.prefix %>position_t {
/** Input text row (1-based). */
pub row: u32,
@ -80,8 +83,6 @@ pub enum <%= @grammar.prefix %>value_t {
impl <%= @grammar.prefix %>value_t {
<% @grammar.ptypes.each do |name, typestring| %>
/* A grammar need not assign to $$ for every declared ptype. */
#[allow(dead_code)]
fn v_<%= name %>_mut(&mut self) -> &mut <%= rust_ptype(typestring) %> {
match self { <%= @grammar.prefix %>value_t::v_<%= name %>(v) => v, _ => unreachable!() }
}
@ -297,36 +298,28 @@ pub fn <%= @grammar.prefix %>context_new(input: &[u8]) -> <%= @grammar.prefix %>
context
}
<% free_token_node_used = @grammar.tree && @grammar.free_token_node != "" %>
<% if free_token_node_used %>
impl Drop for <%= @grammar.prefix %>context_t {
/* Run the free_token_node user code block for every token node in the tree. */
fn drop(&mut self) {
/* Named so that ${context.<field>} expansions resolve here. */
let context = self;
for i in 0..context.<%= @grammar.prefix %>tree_nodes.len() {
if context.<%= @grammar.prefix %>tree_nodes[i].is_token {
let token_node_id = i;
<%= expand_code(@grammar.free_token_node, false, nil, nil).gsub(/\btoken_tree_node\b/, "context.#{@grammar.prefix}tree_nodes[token_node_id]") %>
}
}
}
}
<% end %>
/**
* Deinitialize and deallocate lexer/parser context structure.
*
* The memory owned by the context is released when the context is dropped, so
* this function only has to consume it. It is provided for symmetry with
* <%= @grammar.prefix %>context_new() and with the other target languages;
* letting the context go out of scope has the same effect.
*
* @param context
* Lexer/parser context structure.
*/
#[allow(unused_variables)]
pub fn <%= @grammar.prefix %>context_delete(context: <%= @grammar.prefix %>context_t) {
<% free_token_node_used = @grammar.tree && @grammar.free_token_node != "" %>
<% if free_token_node_used %>
/* The context is taken as mutable for the benefit of the free_token_node user
* code block below, which is permitted but not required to modify the node it
* is freeing. */
#[allow(unused_mut)]
<% end %>
pub fn <%= @grammar.prefix %>context_delete(<%= free_token_node_used ? "mut " : "" %>context: <%= @grammar.prefix %>context_t) {
<% if free_token_node_used %>
for i in 0..context.<%= @grammar.prefix %>tree_nodes.len() {
if context.<%= @grammar.prefix %>tree_nodes[i].is_token {
let token_node_id = i;
<%= expand_code(@grammar.free_token_node, false, nil, nil).gsub(/\btoken_tree_node\b/, "context.#{@grammar.prefix}tree_nodes[token_node_id]") %>
}
}
<% end %>
}
/**************************************************************************
@ -501,12 +494,6 @@ static lexer_mode_table: [lexer_mode_t; <%= @lexer.mode_table.size %>] = [
* @return Token to accept, or invalid token if the user code does
* not explicitly return a token.
*/
/* The parameters of this function are provided for the user code blocks
* inlined into it, which are not obliged to reference any of them. A $$ or $N
* reference in a user code block also expands to a parenthesized dereference,
* since it may be followed there by a field or method access; those
* parentheses are redundant where the reference is a complete argument. */
#[allow(unused_parens, unused_variables)]
fn lexer_user_code(context: &mut <%= @grammar.prefix %>context_t,
code_id: <%= get_type_for(user_code_id_count) %>, match_text: &[u8],
out_token_info: &mut <%= @grammar.prefix %>token_info_t) -> <%= @grammar.prefix %>token_t {
@ -782,11 +769,8 @@ type symbol_id_t = <%= get_type_for(@parser.rule_sets.map(&:last).map(&:id).max)
/** Parser state ID type. */
type parser_state_id_t = <%= get_type_for(@parser.state_table.size) %>;
<% parser_user_code_called = !@grammar.tree || @grammar.parser_user_code_used? %>
<% if parser_user_code_called %>
/** Parser rule ID type. */
type rule_id_t = <%= get_type_for(@grammar.rules.size) %>;
<% end %>
/** Parser shift ID type. */
type shift_id_t = <%= get_type_for(@parser.shift_table.size) %>;
@ -805,7 +789,6 @@ struct shift_t {
struct reduce_t {
/** Lookahead token. */
token: <%= @grammar.prefix %>token_t,
<% if parser_user_code_called %>
/**
* Rule ID.
@ -814,7 +797,6 @@ struct reduce_t {
* grammar rule.
*/
rule: rule_id_t,
<% end %>
/**
* Rule set ID.
@ -909,9 +891,7 @@ static parser_reduce_table: [reduce_t; <%= @parser.reduce_table.size %>] = [
<% @parser.reduce_table.each do |reduce| %>
reduce_t {
token: <%= reduce[:token_id] %>, /* Token: <%= reduce[:token] ? reduce[:token].name : "(any)" %> */
<% if parser_user_code_called %>
rule: <%= reduce[:rule_id] %>, /* Rule ID */
<% end %>
rule_set: <%= reduce[:rule_set_id] %>, /* Rule set ID (<%= reduce[:rule].rule_set.name %>) */
n_states: <%= reduce[:n_states] %>, /* Number of states */
<% if @grammar.tree %>
@ -999,12 +979,6 @@ fn get_rule_position(statevalues: &[state_value_t], i: usize, n_states: usize, g
* @retval P_USER_TERMINATED
* User requested to terminate parsing.
*/
/* The parameters of this function are provided for the user code blocks
* inlined into it, which are not obliged to reference any of them. A $$ or $N
* reference in a user code block also expands to a parenthesized dereference,
* since it may be followed there by a field or method access; those
* parentheses are redundant where the reference is a complete argument. */
#[allow(unused_parens, unused_variables)]
fn parser_user_code(context: &mut <%= @grammar.prefix %>context_t, <%= @grammar.tree ? "_node_id: #{@grammar.prefix}node_id_t" : "_pvalue: &mut #{@grammar.prefix}value_t" %>, rule: u32, statevalues: &[state_value_t], n_states: usize) -> usize {
match rule {
<% @grammar.rules.each do |rule| %>
@ -1327,11 +1301,11 @@ pub fn <%= @grammar.prefix %>parse_inner_<%= start_rule %>(context: &mut <%= @gr
*/
<% if @grammar.tree %>
pub fn <%= @grammar.prefix %>result(context: &<%= @grammar.prefix %>context_t) -> <%= h_type(@grammar.start_rules[0]) %><'_> {
<%= tree_handle(h_type(@grammar.start_rules[0]), "context.parse_result", false) %>
<%= tree_handle(h_type(@grammar.start_rules[0]), "context.parse_result") %>
}
<% @grammar.start_rules.each_with_index do |start_rule, i| %>
pub fn <%= @grammar.prefix %>result_<%= start_rule %>(context: &<%= @grammar.prefix %>context_t) -> <%= h_type(start_rule) %><'_> {
<%= tree_handle(h_type(start_rule), "context.parse_result", false) %>
<%= tree_handle(h_type(start_rule), "context.parse_result") %>
}
<% end %>
<% else %>

View File

@ -349,11 +349,10 @@ accessors on a node handle:
for token payload and user fields), and `p_node_id(node)` (for identity
comparison).
* C++: handle methods called with `()`, e.g. `node.field()`, `node.valid()`,
`node.position()`, `node.token()`, `node.pvalue()`, `node.data()`, and
`node.node_id()` (for identity comparison). The C-style functions and
macros above are also available.
`node.position()`, `node.token()`, `node.pvalue()`, and `node.data()`. The
C-style functions and macros above are also available.
* D: `@property` accessors, e.g. `node.field`, `node.valid`, `node.position`,
`node.token`, `node.pvalue`, and `node.node_id` (for identity comparison).
`node.token`, `node.pvalue`.
* Rust: handle methods called with `()`, e.g. `node.field()`, `node.valid()`,
`node.position()`, `node.end_position()`, `node.n_fields()`,
`node.token()`, `node.pvalue()`, `node.data()` (a reference to the node
@ -531,16 +530,15 @@ Start -> a:a b:b;
The `free_token_node` statement user code block is not emitted for D language
since D has a garbage collector.
The code block is emitted for the Rust target, where it is run from a `Drop`
implementation generated for `p_context_t`.
It therefore runs exactly once however the context is disposed of, whether that
is by calling `p_context_delete()` or by simply letting the context go out of
scope.
The code block is emitted for the Rust target, where it runs from
`p_context_delete()`.
A `ptype` or token user field which owns its memory (a `String`, a `Vec`, a
`Box`, and so on) is released when the context is dropped and does not need a
`free_token_node` code block.
The statement is only needed for memory which Rust does not track, such as a
raw pointer obtained from `Box::into_raw()`.
Note that the generated `p_context_t` does not implement `Drop`, so a
`free_token_node` code block only runs if `p_context_delete()` is called.
##> `lex_fn` statement - specifying a custom lexer function
@ -697,21 +695,10 @@ generated output without any surrounding `#line` directives.
This can be useful when debugging the generated parser itself, or when the
`#line` directives interfere with other tooling.
Rust has no `#line` directive equivalent.
For a Rust target, Propane instead emits a comment before and after each
section of user code naming the grammar file and the line number the code was
taken from:
```
/* Begin user code from myparser.propane line 42. */
let mut v: i64 = 0;
/* End user code from myparser.propane line 42. */
```
A compiler diagnostic that points into the generated Rust module can be traced
back to the grammar by reading up to the nearest such comment.
The `noline` statement suppresses these comments for a Rust target, in the same
way that it suppresses `#line` directives for the other targets.
The `noline` statement only affects the C, C++, and D targets.
Rust has no `#line` directive equivalent, so `#line` directives are never
emitted into Rust output and the `noline` statement has no effect for the Rust
target.
##> `on_tree_node` statement - custom initialization of a token tree node
@ -1693,9 +1680,9 @@ provide a code block which frees that memory; if specified, the
For Rust targets, `p_context_delete()` takes the context by value and consumes
it.
Everything the context owns is released when it is dropped, including running
any `free_token_node` code block, so calling this function is optional for a
Rust target; letting the context go out of scope has the same effect.
The memory owned by the context is released when the context is dropped, so the
call is only strictly required when the grammar supplies a `free_token_node`
code block, which runs from `p_context_delete()`.
Rust example:

View File

@ -7,46 +7,12 @@ if exists("b:current_syntax")
finish
endif
" Guess the language of the user code blocks from their contents so that the
" matching syntax file can be included below. b:propane_subtype may also be set
" before this file is sourced to select the language explicitly.
if !exists("b:propane_subtype")
" Rust markers. Each keyword requires the syntax that follows it in Rust so
" that a plain identifier of the same name in another language does not match
" (`int fn = 3;' in C, for example). Type names are only accepted within a
" `ptype' statement for the same reason.
let s:rust = '\<let\s\+\%(mut\s\+\)\?\w'
let s:rust .= '\|\<fn\s\+\w\+\s*('
let s:rust .= '\|&mut\>\|\<pub\s\+\w\|\<impl\s\+\w'
let s:rust .= '\|#\[\|\<use\s\+\%(std\|core\)::'
let s:rust .= '\|\<ptype\>[^;]*\<\%(isize\|usize\|i8\|i16\|i32\|i64\|i128'
let s:rust .= '\|u8\|u16\|u32\|u64\|u128\|f32\|f64\|String\)\>'
" D markers. These are spellings that have no valid C, C++, or Rust
" equivalent, so `import' is deliberately not among them: it is a D keyword
" but is also a C++20 module declaration.
let s:d = '\<foreach\%(_reverse\)\?\s*([^)]*;'
let s:d .= '\|\~=\|\<static\s\+if\s*(\|\<version\s*(\s*\w\+\s*)'
let s:d .= '\|\<scope\s*(\s*\%(exit\|failure\|success\)\s*)'
let s:d .= '\|\<\%(unittest\|mixin\|immutable\|__gshared\|invariant\)\>'
let s:d .= '\|\<alias\s\+\w\+\s*=\|\<enum\s\+\w\+\s*='
let s:d .= '\|@\%(property\|safe\|trusted\|system\|nogc\|disable\)\>'
let s:d .= '\|\<is\s\+null\>\|\<cast\s*(\s*\w\+\s*)'
let s:d .= '\|\<write\%(ln\|fln\|f\)\s*('
let s:d .= '\|\<\%(dchar\|dstring\|wstring\|cent\|ucent\)\>'
" A module import on its own is ambiguous between D and C++20, so only take
" it as D when nothing else in the file looks like C++.
let s:import = '\<import\s\+[A-Za-z_][A-Za-z0-9_.]*\s*;'
let s:cpp = '::\|\<template\s*<\|\<namespace\>\|\<nullptr\>\|#include\s*[<"]'
if search(s:rust, 'nw') > 0
let b:propane_subtype = "rust"
elseif search(s:d, 'nw') > 0
let b:propane_subtype = "d"
elseif search(s:import, 'nw') > 0 && search(s:cpp, 'nw') == 0
if search('\<import\s\+\%(std\|core\)\.', 'nw') > 0
let b:propane_subtype = "d"
else
let b:propane_subtype = "cpp"
endif
unlet s:rust s:d s:import s:cpp
endif
exe "syn include @propaneTarget syntax/".b:propane_subtype.".vim"

View File

@ -41,23 +41,12 @@ class Propane
output_file = @output_file
end
erb = ERB.new(template, trim_mode: "<>")
# Rust has no #line directive support. For a Rust target the directives
# that the grammar embeds around user code blocks are replaced with
# comments naming the grammar file and line number the code came from,
# so that the origin of a section of user code can still be found by
# reading up from a compiler diagnostic pointing into the generated
# module.
user_code_origin = nil
result = erb.result(binding.clone).lines.each_with_index.map do |line, i|
if @language == "rust"
if md = line.match(/^#line (\d+) "([^"]*)"/)
user_code_origin = "#{md[2]} line #{md[1]}"
line.sub(/^#line \d+ "[^"]*"/, %[/* Begin user code from #{user_code_origin}. */])
elsif line == "#linereset\n"
%[/* End user code from #{user_code_origin}. */\n]
else
line
end
# Rust has no #line directive support, so strip the directives that
# the grammar embeds in user code blocks.
line = line.sub(/^#line \d+ "[^"]*"/, "")
line == "#linereset\n" ? "" : line
elsif line == "#linereset\n"
%[#line #{i + 2} "#{output_file}"\n]
else
@ -481,24 +470,16 @@ class Propane
# Handle type name.
# @param id_expr [String]
# Expression yielding the node ID.
# @param parenthesize [Boolean]
# Whether to parenthesize the expression. Parentheses are required where
# the expression is substituted into a user code block, since the
# expression could be followed there by a field access or appear in a
# position where a bare Rust struct literal is not accepted. They are
# unnecessary where the expression stands alone, and Rust warns about
# them there, so this can be disabled for those uses.
#
# @return [String]
# Handle constructor expression.
def tree_handle(typename, id_expr, parenthesize = true)
def tree_handle(typename, id_expr)
if @cpp
"(#{typename}{context, #{id_expr}})"
elsif @language == "c"
"((#{typename}){context, #{id_expr}})"
elsif @language == "rust"
expr = "#{typename} { context, id: #{id_expr} }"
parenthesize ? "(#{expr})" : expr
"(#{typename} { context, id: #{id_expr} })"
else
"#{typename}(context, #{id_expr})"
end
@ -552,30 +533,16 @@ class Propane
end
end
# Generate the tree node handle type declarations for the header.
#
# These are emitted before the context structure definition so that a
# context_user_fields block can declare a field of a handle type.
# Generate the C/C++ tree node handle type section for the header.
#
# @return [String]
# Handle type declarations.
def c_tree_handle_types_header
@cpp ? cpp_tree_handle_types_header : c_only_tree_handle_types_header
end
# Generate the remainder of the tree node section for the header.
#
# This is emitted after the context structure definition since it
# dereferences the context and so requires the complete type.
#
# @return [String]
# Accessors, macros, and out-of-line handle method definitions.
# Header handle section.
def c_tree_types_header
@cpp ? cpp_tree_types_header : c_only_tree_types_header
end
# Generate the C (non-C++) tree node handle type section for the header.
def c_only_tree_handle_types_header
def c_only_tree_types_header
p = @grammar.prefix
out = []
out << "/** Tree node handle types. @{ */"
@ -583,11 +550,6 @@ class Propane
out << "typedef struct { #{p}context_t * __context; #{p}node_id_t __id; } #{t};"
end
out << ""
out.join("\n")
end
def c_only_tree_types_header
out = []
out << c_common_accessors_header
out << "/** @} */"
out.join("\n")
@ -695,32 +657,29 @@ class Propane
out.join("\n")
end
# Generate the C++ tree node handle class declarations for the header.
# Only valid() and node_id() are defined inline; every other method
# dereferences the context, which is still an incomplete type here, so
# those are declared and defined out of line once the context is
# complete.
def cpp_tree_handle_types_header
# Generate the C++ tree node handle type section for the header.
def cpp_tree_types_header
p = @grammar.prefix
out = []
out << "/** Tree node handle types. @{ */"
tree_handle_types.each {|t| out << "struct #{t};"}
out << ""
# Token handle (all methods inline; no handle-typed returns).
tt = h_type("Token")
out << "struct #{tt}"
out << "{"
out << " #{p}context_t * __context;"
out << " #{p}node_id_t __id;"
out << " bool valid() const { return __id != 0u; }"
out << " #{p}node_id_t node_id() const { return __id; }"
out << " #{p}node_data_t * data() const;"
out << " #{p}position_t position() const;"
out << " #{p}position_t end_position() const;"
out << " uint16_t n_fields() const;"
out << " #{p}token_t token() const;"
out << " #{p}value_t pvalue() const;"
out << " #{p}node_data_t * data() const { return &__context->#{p}tree_nodes[__id]; }"
out << " #{p}position_t position() const { return __context->#{p}tree_nodes[__id].position; }"
out << " #{p}position_t end_position() const { return __context->#{p}tree_nodes[__id].end_position; }"
out << " uint16_t n_fields() const { return __id ? __context->#{p}tree_nodes[__id].n_fields : (uint16_t)0u; }"
out << " #{p}token_t token() const { return __context->#{p}tree_nodes[__id].token; }"
out << " #{p}value_t pvalue() const { return __context->#{p}tree_nodes[__id].pvalue; }"
out << "};"
out << ""
# Rule set handles: navigation methods declared, defined out-of-line below.
tree_node_rule_sets.each do |rule_set|
rtype = h_type(rule_set.name)
out << "struct #{rtype}"
@ -728,36 +687,16 @@ class Propane
out << " #{p}context_t * __context;"
out << " #{p}node_id_t __id;"
out << " bool valid() const { return __id != 0u; }"
out << " #{p}node_id_t node_id() const { return __id; }"
out << " #{p}node_data_t * data() const;"
out << " #{p}position_t position() const;"
out << " #{p}position_t end_position() const;"
out << " uint16_t n_fields() const;"
out << " #{p}node_data_t * data() const { return &__context->#{p}tree_nodes[__id]; }"
out << " #{p}position_t position() const { return __context->#{p}tree_nodes[__id].position; }"
out << " #{p}position_t end_position() const { return __context->#{p}tree_nodes[__id].end_position; }"
out << " uint16_t n_fields() const { return __id ? __context->#{p}tree_nodes[__id].n_fields : (uint16_t)0u; }"
each_tree_field(rule_set) do |rt, field_name, child_type, slot|
out << " #{child_type} #{field_name}() const;"
end
out << "};"
out << ""
end
out.join("\n")
end
# Generate the out-of-line C++ handle method definitions plus the C-style
# accessors. Emitted after the context structure definition.
def cpp_tree_types_header
p = @grammar.prefix
out = []
# Common node methods, now that the context type is complete.
([h_type("Token")] + tree_node_rule_sets.map {|rs| h_type(rs.name)}).each do |ht|
out << "inline #{p}node_data_t * #{ht}::data() const { return &__context->#{p}tree_nodes[__id]; }"
out << "inline #{p}position_t #{ht}::position() const { return __context->#{p}tree_nodes[__id].position; }"
out << "inline #{p}position_t #{ht}::end_position() const { return __context->#{p}tree_nodes[__id].end_position; }"
out << "inline uint16_t #{ht}::n_fields() const { return __id ? __context->#{p}tree_nodes[__id].n_fields : (uint16_t)0u; }"
end
tt = h_type("Token")
out << "inline #{p}token_t #{tt}::token() const { return __context->#{p}tree_nodes[__id].token; }"
out << "inline #{p}value_t #{tt}::pvalue() const { return __context->#{p}tree_nodes[__id].pvalue; }"
out << ""
# Out-of-line navigation method bodies (all handle types now complete).
tree_node_rule_sets.each do |rule_set|
rtype = h_type(rule_set.name)

View File

@ -1,3 +1,3 @@
class Propane
VERSION = "5.1.0"
VERSION = "4.8.1"
end

View File

@ -2059,36 +2059,9 @@ EOF
expect(results.status).to eq 0
end
# D is excluded since it is garbage collected, so Propane does not emit a
# free_token_node code block for it.
if %w[c cpp rust].include?(language)
if %w[c cpp].include?(language)
it "allows a user function to free token node memory in tree mode" do
if language == "rust"
write_grammar <<EOF
<<
use std::sync::atomic::{AtomicU32, Ordering};
/** Number of token nodes freed by the free_token_node code block. */
pub static FREED: AtomicU32 = AtomicU32::new(0);
>>
tree;
free_token_node <<
if !${token.pvalue}.is_null() {
unsafe { drop(Box::from_raw(${token.pvalue})); }
FREED.fetch_add(1, Ordering::SeqCst);
}
>>
ptype *mut i32;
token a <<
$$ = Box::into_raw(Box::new(1));
>>
token b <<
$$ = Box::into_raw(Box::new(2));
>>
Start -> a:a b:b;
EOF
else
write_grammar <<EOF
write_grammar <<EOF
tree;
free_token_node <<
free(${token.pvalue});
@ -2104,7 +2077,6 @@ token b <<
>>
Start -> a:a b:b;
EOF
end
run_propane(language: language)
compile("spec/test_tree_delete_token_node_memory.#{language}", language: language)
results = run_test(language: language)
@ -2113,76 +2085,6 @@ EOF
end
end
# Rust is excluded since a Rust tree node handle borrows the context, so
# storing one in a context field would make the context self-referential.
if %w[c cpp d].include?(language)
it "allows a tree node handle type in a context user field" do
write_grammar <<EOF
tree;
ptype int;
context_user_fields <<
Item first_item;
int have_first;
>>
drop /\\s+/;
token a /a/ << $$ = 7; >>
Item -> a;
Items -> ;
Items -> Items Item <<
if (${context.have_first} == 0)
{
${context.first_item} = $2;
${context.have_first} = 1;
}
>>
Start -> Items;
EOF
run_propane(language: language)
compile("spec/test_context_field_handle.#{language}", language: language)
results = run_test(language: language)
expect(results.stderr).to eq ""
expect(results.status).to eq 0
end
end
if language == "rust"
it "marks user code sections with their grammar file and line number" do
write_grammar <<EOF
ptype i64;
drop /\\s+/;
token num /\\d+/ <<
$$ = 42;
>>
Start -> num << $$ = $1; >>
EOF
run_propane(language: language)
parser = File.binread("spec/run/testparser.rs")
# The lexer code block body begins on grammar line 4 and the parser
# rule code block is on grammar line 6.
expect(parser).to include %[/* Begin user code from spec/run/testparser.propane line 4. */]
expect(parser).to include %[/* End user code from spec/run/testparser.propane line 4. */]
expect(parser).to include %[/* Begin user code from spec/run/testparser.propane line 6. */]
expect(parser).to include %[/* End user code from spec/run/testparser.propane line 6. */]
expect(parser).to_not include "#line"
end
it "omits user code section markers when noline is specified" do
write_grammar <<EOF
noline;
ptype i64;
drop /\\s+/;
token num /\\d+/ <<
$$ = 42;
>>
Start -> num << $$ = $1; >>
EOF
run_propane(language: language)
parser = File.binread("spec/run/testparser.rs")
expect(parser).to_not include "user code from"
expect(parser).to_not include "#line"
end
end
it "executes code blocks associated with drop statements" do
if language == "rust"
write_grammar <<EOF

View File

@ -1,39 +0,0 @@
#include "testparser.h"
#include <assert.h>
#include <string.h>
#include "testutils.h"
int main()
{
char input[128];
size_t i;
p_context_t * context;
Token token;
/* Enough tokens that the tree node arena is reallocated during the parse. */
memset(input, 0, sizeof(input));
for (i = 0u; i < 40u; i++)
{
input[i] = 'a';
}
context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1u, context->have_first);
assert(p_node_valid(context->first_item));
token = p_Item_pToken1(context->first_item);
assert(p_node_valid(token));
assert_eq(TOKEN_a, p_Token_token(token));
assert_eq(7u, p_Token_pvalue(token));
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, p_node_position(context->first_item).row);
assert_eq(1u, p_node_position(context->first_item).col);
p_context_delete(context);
return 0;
}

View File

@ -1,36 +0,0 @@
#include "testparser.h"
#include <cassert>
#include <cstring>
#include "testutils.h"
int main()
{
char input[128];
/* Enough tokens that the tree node arena is reallocated during the parse. */
memset(input, 0, sizeof(input));
for (size_t i = 0u; i < 40u; i++)
{
input[i] = 'a';
}
p_context_t * context = p_context_new((uint8_t const *)input, strlen(input));
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1u, context->have_first);
assert(context->first_item.valid());
Token token = context->first_item.pToken1();
assert(token.valid());
assert_eq(TOKEN_a, token.token());
assert_eq(7u, token.pvalue());
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, context->first_item.position().row);
assert_eq(1u, context->first_item.position().col);
p_context_delete(context);
return 0;
}

View File

@ -1,35 +0,0 @@
import testparser;
import testutils;
int main()
{
return 0;
}
unittest
{
/* Enough tokens that the tree node array is reallocated during the parse. */
string input;
foreach (i; 0 .. 40)
{
input ~= "a";
}
p_context_t * context = p_context_new(input);
assert_eq(P_SUCCESS, p_parse(context));
/* The handle was stored in a context user field during the parse, before
* the remaining nodes were created. It still refers to the same node. */
assert_eq(1, context.have_first);
assert(context.first_item.valid);
Token token = context.first_item.pToken1;
assert(token.valid);
assert_eq(TOKEN_a, token.token);
assert_eq(7, token.pvalue);
/* The stored handle refers to the first Item, which starts at column 1. */
assert_eq(1u, context.first_item.position.row);
assert_eq(1u, context.first_item.position.col);
p_context_delete(context);
}

View File

@ -31,7 +31,6 @@ unittest
assert(start.pR3.valid);
assert(start.pR.valid);
assert(start.pR == start.pR3);
assert_eq(start.pR.node_id, start.pR3.node_id);
assert_eq(TOKEN_c, start.pR.pToken1.token);
p_context_delete(context);

View File

@ -17,7 +17,9 @@ fn main() {
/* p_set_position overrides the initial position. */
let mut c = p_context_new(b"ab");
p_set_position(&mut c, p_position_t { row: 5, col: 20 });
assert_eq!(p_position_t { row: 5, col: 20 }, p_position(&c));
let pos = p_position(&c);
assert_eq!(5, pos.row);
assert_eq!(20, pos.col);
assert_eq!(P_SUCCESS, p_lex(&mut c, &mut ti));
assert_eq!(TOKEN_a, ti.token);
assert_eq!(5, ti.position.row);

View File

@ -13,8 +13,6 @@ int main()
assert(start.pItems1().valid());
assert(start.pItems().valid());
Items items = start.pItems();
assert_ne(0u, items.node_id());
assert_eq(start.pItems().node_id(), items.node_id());
assert(items.pItem().valid());
assert(items.pItem().pToken1().valid());
assert_eq(TOKEN_a, items.pItem().pToken1().token());
@ -42,7 +40,6 @@ int main()
assert_eq(P_SUCCESS, p_parse(context));
start = p_result(context);
assert(!start.pItems().valid());
assert_eq(0u, start.pItems().node_id());
p_context_delete(context);

View File

@ -1,27 +0,0 @@
use testparser::*;
use std::sync::atomic::Ordering;
fn main() {
let mut context = p_context_new(b"ab");
assert_eq!(P_SUCCESS, p_parse(&mut context));
let start = p_result(&context);
assert!(start.a().valid());
assert_eq!(1, unsafe { *start.a().pvalue() });
assert!(start.b().valid());
assert_eq!(2, unsafe { *start.b().pvalue() });
/* The free_token_node code block runs when the context is disposed of, not
* before, and frees each of the two token nodes exactly once. */
assert_eq!(0, FREED.load(Ordering::SeqCst));
p_context_delete(context);
assert_eq!(2, FREED.load(Ordering::SeqCst));
/* Letting the context go out of scope runs the code block too, so a caller
* which never calls p_context_delete() does not leak. */
{
let mut context = p_context_new(b"ab");
assert_eq!(P_SUCCESS, p_parse(&mut context));
assert_eq!(2, FREED.load(Ordering::SeqCst));
}
assert_eq!(4, FREED.load(Ordering::SeqCst));
}