From 59f354a07a2efd44f576c005e0bd322814b1b8f1 Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 10 Sep 2026 22:41:30 -0700 Subject: [PATCH 01/38] Add API for user defined grmtools section entries in GrammarAST --- cfgrammar/src/lib/header.rs | 2 +- cfgrammar/src/lib/markmap.rs | 2 +- cfgrammar/src/lib/yacc/ast.rs | 295 ++++++++++++++++++++++++++++++- cfgrammar/src/lib/yacc/parser.rs | 3 +- 4 files changed, 298 insertions(+), 4 deletions(-) diff --git a/cfgrammar/src/lib/header.rs b/cfgrammar/src/lib/header.rs index 645277f43..05a9ab19f 100644 --- a/cfgrammar/src/lib/header.rs +++ b/cfgrammar/src/lib/header.rs @@ -48,7 +48,7 @@ impl Spanned for HeaderError { // This is essentially a tuple that needs a newtype so we can implement `From` for it. // Thus we aren't worried about it being `pub`. -#[derive(Debug, PartialEq)] +#[derive(Debug, PartialEq, Clone)] #[doc(hidden)] pub struct HeaderValue(pub T, pub Value); diff --git a/cfgrammar/src/lib/markmap.rs b/cfgrammar/src/lib/markmap.rs index b2f910290..bf367751f 100644 --- a/cfgrammar/src/lib/markmap.rs +++ b/cfgrammar/src/lib/markmap.rs @@ -21,7 +21,7 @@ use std::fmt; /// /// Merge behaviors configure how the merge operator handles cases where both `MarkMaps` being merged /// contain a particular key. -#[derive(Debug, PartialEq, Eq)] +#[derive(Debug, PartialEq, Eq, Clone)] #[doc(hidden)] pub struct MarkMap { default_merge_behavior: MergeBehavior, diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index fb1ef5cb2..f42735b05 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,7 +14,7 @@ use super::{ use crate::{ Span, - header::{GrmtoolsSectionParser, HeaderError, HeaderErrorKind, HeaderValue}, + header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, yacc::YaccOriginalActionKind, }; @@ -178,6 +178,7 @@ pub struct GrammarAST { // The set of symbol names that, if unused in a // grammar, will not cause a warning or error. pub expect_unused: Vec, + pub grmtools_section: Option>, } #[derive(Debug, Clone)] @@ -255,6 +256,7 @@ impl GrammarAST { parse_generics: None, programs: None, expect_unused: Vec::new(), + grmtools_section: None, } } @@ -543,6 +545,39 @@ impl GrammarAST { }), ) } + + /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. + /// If the entry is found it marks the key as `used`, for the purposes of `unused_grmtools_section_keys_for_crate`. + pub fn grmtools_section_value_for_crate( + &mut self, + crate_name: &str, + key_name: &str, + ) -> Option<(Span, &Value)> { + let key = format!("{crate_name}.{key_name}"); + if let Some(HeaderValue(span, value)) = self.grmtools_section.as_mut().and_then(|map| { + map.mark_used(&key); + map.get(&key) + }) { + Some((*span, value)) + } else { + None + } + } + + pub fn unused_grmtools_section_keys_for_crate(&self, crate_name: &str) -> Vec { + if let Some(map) = &self.grmtools_section { + map.unused() + .iter() + .filter(|key_name| { + let crate_prefix = format!("{crate_name}."); + key_name.starts_with(&crate_prefix) + }) + .cloned() + .collect::>() + } else { + vec![] + } + } } #[cfg(test)] @@ -984,4 +1019,262 @@ start -> () : "a" {$;;;; }; }] ); } + + #[test] + fn test_grmtools_section_values() { + use super::*; + use crate::header::Value; + let src = r#" +%grmtools { + yacckind: Grmtools, + lrpar.recoverer: CPCTPlus, + test.Flag, + !test.Negative, + test.string: "Foo", + test.vec: ["Aaaa", "Bbbb"], + test.num: 1234, + test.unused: 5678 +} +%token a +%% +start -> () : "a" { () }; +"#; + let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let test_flag_span = src.find_span("test.Flag"); + let test_neg_span = src.find_span("test.Negative"); + let test_neg_val_span = src.find_span("!test.Negative"); + let test_string_span = src.find_span("test.string"); + let test_string_val_span = src.find_span("Foo"); + let test_vec_span = src.find_span("test.vec"); + let test_vec_a_span = src.find_span("Aaaa"); + let test_vec_b_span = src.find_span("Bbbb"); + let test_vec_val_span = src.find_span("[\"Aaaa\", \"Bbbb\"]"); + let test_num_span = src.find_span("test.num"); + let test_num_val_span = src.find_span("1234"); + let mut test_crate_expected = HashMap::new(); + test_crate_expected.insert( + "Flag".to_string(), + (test_flag_span, Value::Bool(true, test_flag_span)), + ); + test_crate_expected.insert( + "Negative".to_string(), + (test_neg_span, Value::Bool(false, test_neg_val_span)), + ); + test_crate_expected.insert( + "string".to_string(), + ( + test_string_span, + Value::String("Foo".to_string(), test_string_val_span), + ), + ); + test_crate_expected.insert( + "vec".to_string(), + ( + test_vec_span, + Value::Array( + vec![ + Value::String("Aaaa".to_string(), test_vec_a_span), + Value::String("Bbbb".to_string(), test_vec_b_span), + ], + test_vec_val_span, + ), + ), + ); + test_crate_expected.insert( + "num".to_string(), + (test_num_span, Value::Num(1234, test_num_val_span)), + ); + for (key, (expected_span, expected_value)) in test_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("test", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert_eq!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("test"), + vec!["test.unused"] + ); + + let mut cfgrammar_crate_expected = HashMap::new(); + let yacckind_span = src.find_span("yacckind"); + let yacckind_val_span = src.find_span("Grmtools"); + cfgrammar_crate_expected.insert( + "yacckind".to_string(), + ( + yacckind_span, + // The actual value we receive has been lower cased + Value::Namespaced("Grmtools".to_string(), yacckind_val_span), + ), + ); + for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("cfgrammar") + .is_empty() + ); + + let mut lrpar_crate_expected = HashMap::new(); + let recoverer_span = src.find_span("lrpar.recoverer"); + let recoverer_val_span = src.find_span("CPCTPlus"); + lrpar_crate_expected.insert( + "recoverer".to_string(), + ( + recoverer_span, + Value::Namespaced("CPCTPlus".to_string(), recoverer_val_span), + ), + ); + for (key, (expected_span, expected_value)) in lrpar_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("lrpar", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("lrpar") + .is_empty() + ); + } + + #[test] + fn test_grmtools_section_values2() { + use super::*; + let src = r#" +%grmtools { + yacckind: Original(YaccOriginalActionKind::UserAction), +} +%token a +%actiontype () +%% +start: "a" { () }; +"#; + let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let mut cfgrammar_crate_expected = HashMap::new(); + let yacckind_span = src.find_span("yacckind"); + let yacckind_val_span = src.find_span("Original(YaccOriginalActionKind::UserAction)"); + cfgrammar_crate_expected.insert( + "yacckind".to_string(), + ( + yacckind_span, + Value::Namespaced( + "Original(YaccOriginalActionKind::UserAction)".to_string(), + yacckind_val_span, + ), + ), + ); + for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("cfgrammar") + .is_empty() + ); + } + + #[test] + fn test_grmtools_section_values3() { + use super::*; + let src = r#" +%grmtools { + yacckind: YaccKind::Original(YaccOriginalActionKind::UserAction), +} +%token a +%actiontype () +%% +start: "a" { () }; +"#; + let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let mut cfgrammar_crate_expected = HashMap::new(); + let yacckind_span = src.find_span("yacckind"); + let yacckind_val_span = + src.find_span("YaccKind::Original(YaccOriginalActionKind::UserAction)"); + cfgrammar_crate_expected.insert( + "yacckind".to_string(), + ( + yacckind_span, + Value::Namespaced( + "YaccKind::Original(YaccOriginalActionKind::UserAction)".to_string(), + yacckind_val_span, + ), + ), + ); + for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("cfgrammar") + .is_empty() + ); + } + + #[test] + fn test_grmtools_section_values4() { + use super::*; + let src = r#" +%grmtools { + yacckind: YaccKind::Original(UserAction), +} +%token a +%actiontype () +%% +start: "a" { () }; +"#; + let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let mut cfgrammar_crate_expected = HashMap::new(); + let yacckind_span = src.find_span("yacckind"); + let yacckind_val_span = src.find_span("YaccKind::Original(UserAction)"); + cfgrammar_crate_expected.insert( + "yacckind".to_string(), + ( + yacckind_span, + Value::Namespaced( + "YaccKind::Original(UserAction)".to_string(), + yacckind_val_span, + ), + ), + ); + for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { + let value = ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", &key); + assert_eq!(value, Some((expected_span, &expected_value))); + } + assert!( + ast_validity + .ast + .unused_grmtools_section_keys_for_crate("cfgrammar") + .is_empty() + ); + } + + trait FindSpan { + fn find_span(&self, s: &str) -> Span; + } + + impl FindSpan for &'_ str { + #[track_caller] + fn find_span(&self, s: &str) -> Span { + let start_pos = self.find(s).unwrap(); + Span::new(start_pos, start_pos + s.len()) + } + } } diff --git a/cfgrammar/src/lib/yacc/parser.rs b/cfgrammar/src/lib/yacc/parser.rs index 4ee8eec99..13197929b 100644 --- a/cfgrammar/src/lib/yacc/parser.rs +++ b/cfgrammar/src/lib/yacc/parser.rs @@ -337,9 +337,10 @@ impl YaccParser<'_> { pub(crate) fn parse(&mut self) -> YaccGrammarResult { let mut errs = Vec::new(); - let (_, pos) = GrmtoolsSectionParser::new(self.src, false) + let (header, pos) = GrmtoolsSectionParser::new(self.src, false) .parse() .map_err(|mut errs| errs.drain(..).map(|e| e.into()).collect::>())?; + self.ast.grmtools_section = Some(header); // We pass around an index into the *bytes* of self.src. We guarantee that at all times // this points to the beginning of a UTF-8 character (since multibyte characters exist, not // every byte within the string is also a valid character). From eff44808165a9d16215938ae0840d3e267a48501 Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 10 Sep 2026 23:49:12 -0700 Subject: [PATCH 02/38] Clean up new test cases --- cfgrammar/src/lib/yacc/ast.rs | 204 ++++++++++++++-------------------- 1 file changed, 83 insertions(+), 121 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index f42735b05..bf30e6ff6 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -1040,51 +1040,49 @@ start -> () : "a" {$;;;; }; start -> () : "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); - let test_flag_span = src.find_span("test.Flag"); - let test_neg_span = src.find_span("test.Negative"); - let test_neg_val_span = src.find_span("!test.Negative"); - let test_string_span = src.find_span("test.string"); - let test_string_val_span = src.find_span("Foo"); - let test_vec_span = src.find_span("test.vec"); - let test_vec_a_span = src.find_span("Aaaa"); - let test_vec_b_span = src.find_span("Bbbb"); - let test_vec_val_span = src.find_span("[\"Aaaa\", \"Bbbb\"]"); - let test_num_span = src.find_span("test.num"); - let test_num_val_span = src.find_span("1234"); - let mut test_crate_expected = HashMap::new(); - test_crate_expected.insert( - "Flag".to_string(), - (test_flag_span, Value::Bool(true, test_flag_span)), - ); - test_crate_expected.insert( - "Negative".to_string(), - (test_neg_span, Value::Bool(false, test_neg_val_span)), - ); - test_crate_expected.insert( - "string".to_string(), + for (key, (expected_span, expected_value)) in [ ( - test_string_span, - Value::String("Foo".to_string(), test_string_val_span), + "Flag".to_string(), + ( + src.find_span("test.Flag"), + Value::Bool(true, src.find_span("test.Flag")), + ), ), - ); - test_crate_expected.insert( - "vec".to_string(), ( - test_vec_span, - Value::Array( - vec![ - Value::String("Aaaa".to_string(), test_vec_a_span), - Value::String("Bbbb".to_string(), test_vec_b_span), - ], - test_vec_val_span, + "Negative".to_string(), + ( + src.find_span("test.Negative"), + Value::Bool(false, src.find_span("!test.Negative")), ), ), - ); - test_crate_expected.insert( - "num".to_string(), - (test_num_span, Value::Num(1234, test_num_val_span)), - ); - for (key, (expected_span, expected_value)) in test_crate_expected { + ( + "string".to_string(), + ( + src.find_span("test.string"), + Value::String("Foo".to_string(), src.find_span("Foo")), + ), + ), + ( + "vec".to_string(), + ( + src.find_span("test.vec"), + Value::Array( + vec![ + Value::String("Aaaa".to_string(), src.find_span("Aaaa")), + Value::String("Bbbb".to_string(), src.find_span("Bbbb")), + ], + src.find_span("[\"Aaaa\", \"Bbbb\"]"), + ), + ), + ), + ( + "num".to_string(), + ( + src.find_span("test.num"), + Value::Num(1234, src.find_span("1234")), + ), + ), + ] { let value = ast_validity .ast .grmtools_section_value_for_crate("test", &key); @@ -1096,24 +1094,16 @@ start -> () : "a" { () }; .unused_grmtools_section_keys_for_crate("test"), vec!["test.unused"] ); - - let mut cfgrammar_crate_expected = HashMap::new(); - let yacckind_span = src.find_span("yacckind"); - let yacckind_val_span = src.find_span("Grmtools"); - cfgrammar_crate_expected.insert( - "yacckind".to_string(), - ( - yacckind_span, - // The actual value we receive has been lower cased - Value::Namespaced("Grmtools".to_string(), yacckind_val_span), - ), - ); - for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { - let value = ast_validity + assert_eq!( + ast_validity .ast - .grmtools_section_value_for_crate("cfgrammar", &key); - assert_eq!(value, Some((expected_span, &expected_value))); - } + .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + Some(( + src.find_span("yacckind"), + &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) + )) + ); + assert!( ast_validity .ast @@ -1121,22 +1111,16 @@ start -> () : "a" { () }; .is_empty() ); - let mut lrpar_crate_expected = HashMap::new(); - let recoverer_span = src.find_span("lrpar.recoverer"); - let recoverer_val_span = src.find_span("CPCTPlus"); - lrpar_crate_expected.insert( - "recoverer".to_string(), - ( - recoverer_span, - Value::Namespaced("CPCTPlus".to_string(), recoverer_val_span), - ), - ); - for (key, (expected_span, expected_value)) in lrpar_crate_expected { - let value = ast_validity + assert_eq!( + ast_validity .ast - .grmtools_section_value_for_crate("lrpar", &key); - assert_eq!(value, Some((expected_span, &expected_value))); - } + .grmtools_section_value_for_crate("lrpar", "recoverer"), + Some(( + src.find_span("lrpar.recoverer"), + &Value::Namespaced("CPCTPlus".to_string(), src.find_span("CPCTPlus")) + )) + ); + assert!( ast_validity .ast @@ -1158,25 +1142,18 @@ start -> () : "a" { () }; start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); - let mut cfgrammar_crate_expected = HashMap::new(); - let yacckind_span = src.find_span("yacckind"); - let yacckind_val_span = src.find_span("Original(YaccOriginalActionKind::UserAction)"); - cfgrammar_crate_expected.insert( - "yacckind".to_string(), - ( - yacckind_span, - Value::Namespaced( + assert_eq!( + ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + Some(( + src.find_span("yacckind"), + &Value::Namespaced( "Original(YaccOriginalActionKind::UserAction)".to_string(), - yacckind_val_span, + src.find_span("Original(YaccOriginalActionKind::UserAction)"), ), - ), + )) ); - for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { - let value = ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", &key); - assert_eq!(value, Some((expected_span, &expected_value))); - } assert!( ast_validity .ast @@ -1198,26 +1175,18 @@ start: "a" { () }; start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); - let mut cfgrammar_crate_expected = HashMap::new(); - let yacckind_span = src.find_span("yacckind"); - let yacckind_val_span = - src.find_span("YaccKind::Original(YaccOriginalActionKind::UserAction)"); - cfgrammar_crate_expected.insert( - "yacckind".to_string(), - ( - yacckind_span, - Value::Namespaced( + assert_eq!( + ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + Some(( + src.find_span("yacckind"), + &Value::Namespaced( "YaccKind::Original(YaccOriginalActionKind::UserAction)".to_string(), - yacckind_val_span, + src.find_span("YaccKind::Original(YaccOriginalActionKind::UserAction)"), ), - ), + )) ); - for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { - let value = ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", &key); - assert_eq!(value, Some((expected_span, &expected_value))); - } assert!( ast_validity .ast @@ -1239,25 +1208,18 @@ start: "a" { () }; start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); - let mut cfgrammar_crate_expected = HashMap::new(); - let yacckind_span = src.find_span("yacckind"); - let yacckind_val_span = src.find_span("YaccKind::Original(UserAction)"); - cfgrammar_crate_expected.insert( - "yacckind".to_string(), - ( - yacckind_span, - Value::Namespaced( + assert_eq!( + ast_validity + .ast + .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + Some(( + src.find_span("yacckind"), + &Value::Namespaced( "YaccKind::Original(UserAction)".to_string(), - yacckind_val_span, + src.find_span("YaccKind::Original(UserAction)"), ), - ), + )) ); - for (key, (expected_span, expected_value)) in cfgrammar_crate_expected { - let value = ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", &key); - assert_eq!(value, Some((expected_span, &expected_value))); - } assert!( ast_validity .ast From 14537c3c3df7ab12215179f8a086c2081d408745 Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 11 Sep 2026 00:03:43 -0700 Subject: [PATCH 03/38] Remove unneeded `to_string()` in test --- cfgrammar/src/lib/yacc/ast.rs | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index bf30e6ff6..7742a8527 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -1042,28 +1042,28 @@ start -> () : "a" { () }; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); for (key, (expected_span, expected_value)) in [ ( - "Flag".to_string(), + "Flag", ( src.find_span("test.Flag"), Value::Bool(true, src.find_span("test.Flag")), ), ), ( - "Negative".to_string(), + "Negative", ( src.find_span("test.Negative"), Value::Bool(false, src.find_span("!test.Negative")), ), ), ( - "string".to_string(), + "string", ( src.find_span("test.string"), Value::String("Foo".to_string(), src.find_span("Foo")), ), ), ( - "vec".to_string(), + "vec", ( src.find_span("test.vec"), Value::Array( @@ -1076,7 +1076,7 @@ start -> () : "a" { () }; ), ), ( - "num".to_string(), + "num", ( src.find_span("test.num"), Value::Num(1234, src.find_span("1234")), @@ -1085,7 +1085,7 @@ start -> () : "a" { () }; ] { let value = ast_validity .ast - .grmtools_section_value_for_crate("test", &key); + .grmtools_section_value_for_crate("test", key); assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( From a96e799d62cb5b83f8befbd47fa0f33cd24db537 Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 11 Sep 2026 02:52:21 -0700 Subject: [PATCH 04/38] Move ownership of field to ASTWithValidityInfo --- cfgrammar/src/lib/yacc/ast.rs | 107 ++++++++++++------------------- cfgrammar/src/lib/yacc/parser.rs | 13 ++-- 2 files changed, 50 insertions(+), 70 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 7742a8527..0bab7a686 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -39,6 +39,7 @@ impl fmt::Display for ASTModificationError { pub struct ASTWithValidityInfo { yacc_kind: YaccKind, ast: GrammarAST, + grmtools_section: Header, errs: Vec, } @@ -52,18 +53,19 @@ impl ASTWithValidityInfo { /// already extracted the `YaccKind` if any. pub fn new(yacc_kind: YaccKind, s: &str) -> Self { let mut errs = Vec::new(); - let ast = { + let (ast, grmtools_section) = { let mut yp = YaccParser::new(yacc_kind, s); yp.parse().map_err(|e| errs.extend(e)).ok(); - let mut ast = yp.build(); + let (mut ast, grmtools_section) = yp.build(); ast.complete_and_validate(Some(yacc_kind)) .map_err(|e| errs.push(e)) .ok(); - ast + (ast, grmtools_section) }; ASTWithValidityInfo { ast, errs, + grmtools_section, yacc_kind, } } @@ -108,6 +110,34 @@ impl ASTWithValidityInfo { }) } } + + /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. + /// If the entry is found it marks the key as `used`, for the purposes of `unused_grmtools_section_keys_for_crate`. + pub fn grmtools_section_value_for_crate( + &mut self, + crate_name: &str, + key_name: &str, + ) -> Option<(Span, &Value)> { + let key = format!("{crate_name}.{key_name}"); + self.grmtools_section.mark_used(&key); + if let Some(HeaderValue(span, value)) = self.grmtools_section.get(&key) { + Some((*span, value)) + } else { + None + } + } + + pub fn unused_grmtools_section_keys_for_crate(&self, crate_name: &str) -> Vec { + self.grmtools_section + .unused() + .iter() + .filter(|key_name| { + let crate_prefix = format!("{crate_name}."); + key_name.starts_with(&crate_prefix) + }) + .cloned() + .collect::>() + } } impl FromStr for ASTWithValidityInfo { @@ -124,7 +154,7 @@ impl FromStr for ASTWithValidityInfo { // We don't want to strip off the header so that span's will be correct. let mut yp = YaccParser::new(yacc_kind, src); yp.parse().map_err(|e| errs.extend(e)).ok(); - let mut ast = yp.build(); + let (mut ast, _) = yp.build(); ast.complete_and_validate(Some(yacc_kind)) .map_err(|e| errs.push(e)) .ok(); @@ -133,6 +163,7 @@ impl FromStr for ASTWithValidityInfo { Ok(ASTWithValidityInfo { ast, errs, + grmtools_section: header, yacc_kind, }) } else { @@ -178,7 +209,6 @@ pub struct GrammarAST { // The set of symbol names that, if unused in a // grammar, will not cause a warning or error. pub expect_unused: Vec, - pub grmtools_section: Option>, } #[derive(Debug, Clone)] @@ -256,7 +286,6 @@ impl GrammarAST { parse_generics: None, programs: None, expect_unused: Vec::new(), - grmtools_section: None, } } @@ -545,39 +574,6 @@ impl GrammarAST { }), ) } - - /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. - /// If the entry is found it marks the key as `used`, for the purposes of `unused_grmtools_section_keys_for_crate`. - pub fn grmtools_section_value_for_crate( - &mut self, - crate_name: &str, - key_name: &str, - ) -> Option<(Span, &Value)> { - let key = format!("{crate_name}.{key_name}"); - if let Some(HeaderValue(span, value)) = self.grmtools_section.as_mut().and_then(|map| { - map.mark_used(&key); - map.get(&key) - }) { - Some((*span, value)) - } else { - None - } - } - - pub fn unused_grmtools_section_keys_for_crate(&self, crate_name: &str) -> Vec { - if let Some(map) = &self.grmtools_section { - map.unused() - .iter() - .filter(|key_name| { - let crate_prefix = format!("{crate_name}."); - key_name.starts_with(&crate_prefix) - }) - .cloned() - .collect::>() - } else { - vec![] - } - } } #[cfg(test)] @@ -1083,21 +1079,15 @@ start -> () : "a" { () }; ), ), ] { - let value = ast_validity - .ast - .grmtools_section_value_for_crate("test", key); + let value = ast_validity.grmtools_section_value_for_crate("test", key); assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( - ast_validity - .ast - .unused_grmtools_section_keys_for_crate("test"), + ast_validity.unused_grmtools_section_keys_for_crate("test"), vec!["test.unused"] ); assert_eq!( - ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) @@ -1106,15 +1096,12 @@ start -> () : "a" { () }; assert!( ast_validity - .ast .unused_grmtools_section_keys_for_crate("cfgrammar") .is_empty() ); assert_eq!( - ast_validity - .ast - .grmtools_section_value_for_crate("lrpar", "recoverer"), + ast_validity.grmtools_section_value_for_crate("lrpar", "recoverer"), Some(( src.find_span("lrpar.recoverer"), &Value::Namespaced("CPCTPlus".to_string(), src.find_span("CPCTPlus")) @@ -1123,7 +1110,6 @@ start -> () : "a" { () }; assert!( ast_validity - .ast .unused_grmtools_section_keys_for_crate("lrpar") .is_empty() ); @@ -1143,9 +1129,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1156,7 +1140,6 @@ start: "a" { () }; ); assert!( ast_validity - .ast .unused_grmtools_section_keys_for_crate("cfgrammar") .is_empty() ); @@ -1176,9 +1159,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1189,7 +1170,6 @@ start: "a" { () }; ); assert!( ast_validity - .ast .unused_grmtools_section_keys_for_crate("cfgrammar") .is_empty() ); @@ -1209,9 +1189,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity - .ast - .grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1222,7 +1200,6 @@ start: "a" { () }; ); assert!( ast_validity - .ast .unused_grmtools_section_keys_for_crate("cfgrammar") .is_empty() ); diff --git a/cfgrammar/src/lib/yacc/parser.rs b/cfgrammar/src/lib/yacc/parser.rs index 13197929b..6604a03ae 100644 --- a/cfgrammar/src/lib/yacc/parser.rs +++ b/cfgrammar/src/lib/yacc/parser.rs @@ -16,7 +16,7 @@ use wincode::{SchemaRead, SchemaWrite}; use crate::{ Span, Spanned, - header::{GrmtoolsSectionParser, HeaderErrorKind}, + header::{GrmtoolsSectionParser, Header, HeaderErrorKind}, }; pub type YaccGrammarResult = Result>; @@ -294,6 +294,7 @@ pub(crate) struct YaccParser<'a> { src: &'a str, num_newlines: usize, ast: GrammarAST, + header: Option>, global_actiontype: Option<(String, Span)>, } @@ -331,6 +332,7 @@ impl YaccParser<'_> { src, num_newlines: 0, ast: GrammarAST::new(), + header: None, global_actiontype: None, } } @@ -340,7 +342,7 @@ impl YaccParser<'_> { let (header, pos) = GrmtoolsSectionParser::new(self.src, false) .parse() .map_err(|mut errs| errs.drain(..).map(|e| e.into()).collect::>())?; - self.ast.grmtools_section = Some(header); + self.header = Some(header); // We pass around an index into the *bytes* of self.src. We guarantee that at all times // this points to the beginning of a UTF-8 character (since multibyte characters exist, not // every byte within the string is also a valid character). @@ -372,8 +374,8 @@ impl YaccParser<'_> { } } - pub(crate) fn build(self) -> GrammarAST { - self.ast + pub(crate) fn build(self) -> (GrammarAST, Header) { + (self.ast, self.header.expect("set by parse()")) } fn parse_declarations( @@ -1084,7 +1086,8 @@ mod test { fn parse(yacc_kind: YaccKind, s: &str) -> Result> { let mut yp = YaccParser::new(yacc_kind, s); yp.parse()?; - Ok(yp.build()) + let (ast, _) = yp.build(); + Ok(ast) } fn rule(n: &str) -> Symbol { From 86a4457a22bda501212c28ee08f7800a23c8be2a Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 11 Sep 2026 10:12:26 -0700 Subject: [PATCH 05/38] Use the grmtools section from `YaccParser::build` --- cfgrammar/src/lib/yacc/ast.rs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 0bab7a686..8bc806c20 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -150,20 +150,20 @@ impl FromStr for ASTWithValidityInfo { .map_err(|mut errs| errs.drain(..).map(|e| e.into()).collect::>())?; if let Some(HeaderValue(_, yk_val)) = header.get("cfgrammar.yacckind") { let yacc_kind = YaccKind::try_from(yk_val).map_err(|e| vec![e.into()])?; - let ast = { + let (ast, grmtools_section) = { // We don't want to strip off the header so that span's will be correct. let mut yp = YaccParser::new(yacc_kind, src); yp.parse().map_err(|e| errs.extend(e)).ok(); - let (mut ast, _) = yp.build(); + let (mut ast, grmtools_section) = yp.build(); ast.complete_and_validate(Some(yacc_kind)) .map_err(|e| errs.push(e)) .ok(); - ast + (ast, grmtools_section) }; Ok(ASTWithValidityInfo { ast, errs, - grmtools_section: header, + grmtools_section, yacc_kind, }) } else { From adb16f4c6bd2c9ee6125dcad8f8b21317f845f35 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 00:36:49 -0700 Subject: [PATCH 06/38] First attempt at allowing downstream crate keys in CTParserBuilder --- lrpar/src/lib/codegen.rs | 95 ++++++++++++++++++++++++++++++++++++-- lrpar/src/lib/ctbuilder.rs | 15 +++++- 2 files changed, 105 insertions(+), 5 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index ecb342641..7651bb039 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -13,7 +13,7 @@ use crate::{ use cfgrammar::{ Location, RIdx, Span, Symbol, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue}, + header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue, RE_CRATE_DOT}, markmap::MergeError, yacc::{ YaccGrammar, YaccGrammarError, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo, @@ -32,6 +32,7 @@ const ACTIONS_KIND: &str = "__GtActionsKind"; const ACTIONS_KIND_PREFIX: &str = "Ak"; const ACTIONS_KIND_HIDDEN: &str = "__GtActionsKindHidden"; +#[derive(Debug)] #[non_exhaustive] pub(crate) enum ParserSrcEnvError { GrmtoolsSectionParseError(Vec>), @@ -41,6 +42,7 @@ pub(crate) enum ParserSrcEnvError { MissingModName, } +#[derive(Debug)] #[non_exhaustive] pub(crate) enum ParserBuildEnvError where @@ -53,6 +55,7 @@ where GrmtoolsSectionMissingRequiredKeys(Vec), } +#[derive(Debug)] #[non_exhaustive] pub(crate) enum CodegenError { ProcMacro2Error(proc_macro2::LexError), @@ -456,8 +459,23 @@ where self.ast_with_validity_info.yacc_kind() } - pub(crate) fn check_unused_header_keys(&self) -> Result<(), ParserBuildEnvError> { - let unused_keys = self.header.unused(); + pub(crate) fn check_unused_header_keys_for_crate( + &self, + crate_name: Option<&str>, + ) -> Result<(), ParserBuildEnvError> { + let unused_keys = self + .header + .unused() + .iter() + .filter(|s| { + if let Some(crate_name) = crate_name { + s.starts_with(&format!("{crate_name}.")) + } else { + !RE_CRATE_DOT.is_match(s) + } + }) + .map(|s| s.to_string()) + .collect::>(); if !unused_keys.is_empty() { return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_keys)); } @@ -1294,3 +1312,74 @@ pub(crate) fn make_generics(parse_generics: Option<&str>) -> Result)) } } + +#[cfg(test)] +mod test { + use crate::test_utils::TestLexerTypes; + use cfgrammar::{header::Header, span::Location}; + + use super::*; + #[test] + fn test_unused_crate_header_entry() { + let src = r#" + %grmtools{ + yacckind: Grmtools, + test.foo: "test crate value", + } + %% + start -> () : "A" { () }; + "#; + let empty_header = Header::::new(); + let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let build_env = src_env + .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("cfgrammar")) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("lrpar")) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("lrlex")) + .unwrap(); + build_env.check_unused_header_keys_for_crate(None).unwrap(); + let codegen = build_env.code_generator("timestamp").unwrap(); + let out = codegen.generate(&build_env).unwrap(); + assert!(!out.is_empty()); + } + + #[test] + fn test_unused_header_entry() { + let src = r#" + %grmtools{ + yacckind: Grmtools, + testfoo: "values which do not specify a crate origin should show up as unused", + } + %% + start -> () : "A" { () }; + "#; + let empty_header = Header::::new(); + let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let build_env = src_env + .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("cfgrammar")) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("lrpar")) + .unwrap(); + build_env + .check_unused_header_keys_for_crate(Some("lrlex")) + .unwrap(); + match build_env.check_unused_header_keys_for_crate(None) { + Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) + if keys == vec!["testfoo".to_string()] => {} + _ => panic!("Unexpected return value for unused header keys check"), + } + let codegen = build_env.code_generator("timestamp").unwrap(); + let out = codegen.generate(&build_env).unwrap(); + assert!(!out.is_empty()); + } +} diff --git a/lrpar/src/lib/ctbuilder.rs b/lrpar/src/lib/ctbuilder.rs index 0df742103..1d7a1feaf 100644 --- a/lrpar/src/lib/ctbuilder.rs +++ b/lrpar/src/lib/ctbuilder.rs @@ -747,10 +747,21 @@ where inspector_rt(build_env.header_mut(), rt, &rule_ids, grmp)? } + // Catch any typos in key names for cfgrammar or lrpar build_env - .check_unused_header_keys() + .check_unused_header_keys_for_crate(Some("cfgrammar")) + .map_err(|e| ErrorString(e.to_string()))?; + build_env + .check_unused_header_keys_for_crate(Some("lrpar")) + .map_err(|e| ErrorString(e.to_string()))?; + // Catch any stray lrlex keys that accidentally make their way into the parser src. + build_env + .check_unused_header_keys_for_crate(Some("lrlex")) + .map_err(|e| ErrorString(e.to_string()))?; + // Catch any stray keys without a crate prefix. + build_env + .check_unused_header_keys_for_crate(None) .map_err(|e| ErrorString(e.to_string()))?; - self.output_file( &code_gen, outp, From fd3fa865b61736efe9b8e7a4eda488a381e6755c Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 02:08:56 -0700 Subject: [PATCH 07/38] Use same naming convention as the codegen module --- cfgrammar/src/lib/yacc/ast.rs | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 8bc806c20..ae9a1fa22 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -112,7 +112,7 @@ impl ASTWithValidityInfo { } /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. - /// If the entry is found it marks the key as `used`, for the purposes of `unused_grmtools_section_keys_for_crate`. + /// If the entry is found it marks the key as `used`, for the purposes of `unused_header_keys_for_crate`. pub fn grmtools_section_value_for_crate( &mut self, crate_name: &str, @@ -127,7 +127,7 @@ impl ASTWithValidityInfo { } } - pub fn unused_grmtools_section_keys_for_crate(&self, crate_name: &str) -> Vec { + pub fn unused_header_keys_for_crate(&self, crate_name: &str) -> Vec { self.grmtools_section .unused() .iter() @@ -1083,7 +1083,7 @@ start -> () : "a" { () }; assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( - ast_validity.unused_grmtools_section_keys_for_crate("test"), + ast_validity.unused_header_keys_for_crate("test"), vec!["test.unused"] ); assert_eq!( @@ -1096,7 +1096,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_grmtools_section_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); @@ -1110,7 +1110,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_grmtools_section_keys_for_crate("lrpar") + .unused_header_keys_for_crate("lrpar") .is_empty() ); } @@ -1140,7 +1140,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_grmtools_section_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } @@ -1170,7 +1170,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_grmtools_section_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } @@ -1200,7 +1200,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_grmtools_section_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } From fcba11b571ca53d7c7e49f91f25b6aae027f2dff Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 02:56:24 -0700 Subject: [PATCH 08/38] Preemptively call `mark_used` on grmtools crate entries in header --- cfgrammar/src/lib/yacc/parser.rs | 15 +++++- lrpar/src/lib/codegen.rs | 89 ++++++++++++++++++++++++++++++++ 2 files changed, 102 insertions(+), 2 deletions(-) diff --git a/cfgrammar/src/lib/yacc/parser.rs b/cfgrammar/src/lib/yacc/parser.rs index 6604a03ae..39e8f60c9 100644 --- a/cfgrammar/src/lib/yacc/parser.rs +++ b/cfgrammar/src/lib/yacc/parser.rs @@ -16,7 +16,7 @@ use wincode::{SchemaRead, SchemaWrite}; use crate::{ Span, Spanned, - header::{GrmtoolsSectionParser, Header, HeaderErrorKind}, + header::{CRATE_KEY_MAP, GrmtoolsSectionParser, Header, HeaderErrorKind}, }; pub type YaccGrammarResult = Result>; @@ -375,7 +375,18 @@ impl YaccParser<'_> { } pub(crate) fn build(self) -> (GrammarAST, Header) { - (self.ast, self.header.expect("set by parse()")) + let mut header = self.header.expect("set by parse()"); + // Preemptively mark the keys for lrpar and cfgrammar as used in the header. + // If a downstream crate checks the keys in the ast. The lrpar crate works on a + // local instance which merges the keys from ast with keys from the `CTBuilder`. + // + // It is difficult to do later due to shared references. + for (key_name, crate_name) in CRATE_KEY_MAP.iter() { + if ["cfgrammar", "lrpar"].contains(crate_name) { + header.mark_used(&format!("{crate_name}.{key_name}")); + } + } + (self.ast, header) } fn parse_declarations( diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 7651bb039..8cf2cde65 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1334,15 +1334,33 @@ mod test { let build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("cfgrammar") + .is_empty() + ); build_env .check_unused_header_keys_for_crate(Some("cfgrammar")) .unwrap(); build_env .check_unused_header_keys_for_crate(Some("lrpar")) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrpar") + .is_empty() + ); build_env .check_unused_header_keys_for_crate(Some("lrlex")) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrpar") + .is_empty() + ); build_env.check_unused_header_keys_for_crate(None).unwrap(); let codegen = build_env.code_generator("timestamp").unwrap(); let out = codegen.generate(&build_env).unwrap(); @@ -1367,12 +1385,30 @@ mod test { build_env .check_unused_header_keys_for_crate(Some("cfgrammar")) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("cfgrammar") + .is_empty() + ); build_env .check_unused_header_keys_for_crate(Some("lrpar")) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrpar") + .is_empty() + ); build_env .check_unused_header_keys_for_crate(Some("lrlex")) .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrlex") + .is_empty() + ); match build_env.check_unused_header_keys_for_crate(None) { Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) if keys == vec!["testfoo".to_string()] => {} @@ -1382,4 +1418,57 @@ mod test { let out = codegen.generate(&build_env).unwrap(); assert!(!out.is_empty()); } + + #[test] + fn test_unused_grmtools_header_entry() { + let src = r#" + %grmtools{ + yacckind: Grmtools, + cfgrammar.unknown: "should be unused", + lrpar.unknown: "should be unused", + + } + %% + start -> () : "A" { () }; + "#; + let empty_header = Header::::new(); + let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let build_env = src_env + .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) + .unwrap(); + assert!( + build_env + .check_unused_header_keys_for_crate(Some("cfgrammar")) + .is_err() + ); + assert_eq!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("cfgrammar"), + vec!["cfgrammar.unknown"] + ); + assert!( + build_env + .check_unused_header_keys_for_crate(Some("lrpar")) + .is_err() + ); + assert_eq!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrpar"), + vec!["lrpar.unknown"] + ); + build_env + .check_unused_header_keys_for_crate(Some("lrlex")) + .unwrap(); + assert!( + build_env + .ast_with_validity_info() + .unused_header_keys_for_crate("lrlex") + .is_empty() + ); + let codegen = build_env.code_generator("timestamp").unwrap(); + let out = codegen.generate(&build_env).unwrap(); + assert!(!out.is_empty()); + } } From 751a4d4dcacaaadc704117ef24fde7eedc17eb75 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 02:58:32 -0700 Subject: [PATCH 09/38] stray whitespace --- lrpar/src/lib/codegen.rs | 1 - 1 file changed, 1 deletion(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 8cf2cde65..cbf180335 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1426,7 +1426,6 @@ mod test { yacckind: Grmtools, cfgrammar.unknown: "should be unused", lrpar.unknown: "should be unused", - } %% start -> () : "A" { () }; From e8012c93efd8c71ef2657628599a3ef2f7db97a2 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 03:26:08 -0700 Subject: [PATCH 10/38] Make unused header entries check return spans --- cfgrammar/src/lib/markmap.rs | 11 +++++++---- cfgrammar/src/lib/yacc/ast.rs | 13 ++++++++----- lrlex/src/lib/ctbuilder.rs | 7 ++++++- lrlex/src/main.rs | 6 +++++- lrpar/src/lib/codegen.rs | 23 +++++++++++++++++++---- 5 files changed, 45 insertions(+), 15 deletions(-) diff --git a/cfgrammar/src/lib/markmap.rs b/cfgrammar/src/lib/markmap.rs index bf367751f..9e2a3d245 100644 --- a/cfgrammar/src/lib/markmap.rs +++ b/cfgrammar/src/lib/markmap.rs @@ -484,12 +484,15 @@ impl MarkMap { } /// Returns a `Vec` containing all the keys that are not marked as used. - pub fn unused(&self) -> Vec { + pub fn unused(&self) -> Vec<(K, V)> + where + V: Clone, + { let mut ret = Vec::new(); for (k, mark, v) in &self.contents { let used_mark = Mark::Used.repr(); if v.is_some() && mark & used_mark == 0 { - ret.push(k.to_owned()) + ret.push((k.to_owned(), v.as_ref().unwrap().clone())) } } ret @@ -711,7 +714,7 @@ mod test { assert!(mm.insert("a", "test").is_none()); mm.mark_used(&"a"); assert_eq!(mm.get_mark(&"a"), Some(Mark::Used.repr())); - let empty: &[&String] = &[]; + let empty: &[(&str, &str)] = &[]; assert_eq!(mm.unused().as_slice(), empty); } @@ -722,7 +725,7 @@ mod test { assert!(mm.insert("b", "unused").is_none()); assert_eq!(mm.get_mark(&"a"), Some(Mark::Used.repr())); assert_eq!(mm.get_mark(&"b"), Some(0)); - assert_eq!(mm.unused().as_slice(), &["b"]); + assert_eq!(mm.unused().as_slice(), &[("b", "unused")]); } } diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index ae9a1fa22..f9a8bc92e 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -127,15 +127,18 @@ impl ASTWithValidityInfo { } } - pub fn unused_header_keys_for_crate(&self, crate_name: &str) -> Vec { + pub fn unused_header_keys_for_crate(&self, crate_name: &str) -> Vec<(String, Span)> { self.grmtools_section .unused() .iter() - .filter(|key_name| { + .filter_map(|(key_name, HeaderValue(key_span, _))| { let crate_prefix = format!("{crate_name}."); - key_name.starts_with(&crate_prefix) + if key_name.starts_with(&crate_prefix) { + Some((key_name.clone(), *key_span)) + } else { + None + } }) - .cloned() .collect::>() } } @@ -1084,7 +1087,7 @@ start -> () : "a" { () }; } assert_eq!( ast_validity.unused_header_keys_for_crate("test"), - vec!["test.unused"] + vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), diff --git a/lrlex/src/lib/ctbuilder.rs b/lrlex/src/lib/ctbuilder.rs index 7b6983e46..6a1eea4bd 100644 --- a/lrlex/src/lib/ctbuilder.rs +++ b/lrlex/src/lib/ctbuilder.rs @@ -520,7 +520,12 @@ where None }; - let unused_header_values = build_env.header().unused(); + let unused_header_values = build_env + .header() + .unused() + .iter() + .map(|(s, _)| s.to_string()) + .collect::>(); if !unused_header_values.is_empty() { return Err( format!("Unused header values: {}", unused_header_values.join(", ")).into(), diff --git a/lrlex/src/main.rs b/lrlex/src/main.rs index 30ebbb326..bb87599f2 100644 --- a/lrlex/src/main.rs +++ b/lrlex/src/main.rs @@ -123,7 +123,11 @@ fn main() -> Result<(), Box> { } }; { - let unused_header_values = header.unused(); + let unused_header_values = header + .unused() + .iter() + .map(|(s, _)| s.to_string()) + .collect::>(); if !unused_header_values.is_empty() { Err(ErrorString(format!( "Unused header values: {}", diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index cbf180335..c589a6643 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -467,14 +467,14 @@ where .header .unused() .iter() - .filter(|s| { + .filter(|(s, _)| { if let Some(crate_name) = crate_name { s.starts_with(&format!("{crate_name}.")) } else { !RE_CRATE_DOT.is_match(s) } }) - .map(|s| s.to_string()) + .map(|(s, _)| s.to_string()) .collect::>(); if !unused_keys.is_empty() { return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_keys)); @@ -1444,7 +1444,10 @@ mod test { build_env .ast_with_validity_info() .unused_header_keys_for_crate("cfgrammar"), - vec!["cfgrammar.unknown"] + vec![( + "cfgrammar.unknown".to_string(), + src.find_span("cfgrammar.unknown") + )] ); assert!( build_env @@ -1455,7 +1458,7 @@ mod test { build_env .ast_with_validity_info() .unused_header_keys_for_crate("lrpar"), - vec!["lrpar.unknown"] + vec![("lrpar.unknown".to_string(), src.find_span("lrpar.unknown"))] ); build_env .check_unused_header_keys_for_crate(Some("lrlex")) @@ -1470,4 +1473,16 @@ mod test { let out = codegen.generate(&build_env).unwrap(); assert!(!out.is_empty()); } + + trait FindSpan { + fn find_span(&self, s: &str) -> Span; + } + + impl FindSpan for &'_ str { + #[track_caller] + fn find_span(&self, s: &str) -> Span { + let start_pos = self.find(s).unwrap(); + Span::new(start_pos, start_pos + s.len()) + } + } } From 9bb74bfb86844f7a3b5a40720c10cc49760d306d Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 06:18:00 -0700 Subject: [PATCH 11/38] Rename lookup function to use similar naming scheme. --- cfgrammar/src/lib/yacc/ast.rs | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index f9a8bc92e..8c7c37595 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -113,7 +113,7 @@ impl ASTWithValidityInfo { /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. /// If the entry is found it marks the key as `used`, for the purposes of `unused_header_keys_for_crate`. - pub fn grmtools_section_value_for_crate( + pub fn header_value_for_crate( &mut self, crate_name: &str, key_name: &str, @@ -1082,7 +1082,7 @@ start -> () : "a" { () }; ), ), ] { - let value = ast_validity.grmtools_section_value_for_crate("test", key); + let value = ast_validity.header_value_for_crate("test", key); assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( @@ -1090,7 +1090,7 @@ start -> () : "a" { () }; vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( - ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) @@ -1104,7 +1104,7 @@ start -> () : "a" { () }; ); assert_eq!( - ast_validity.grmtools_section_value_for_crate("lrpar", "recoverer"), + ast_validity.header_value_for_crate("lrpar", "recoverer"), Some(( src.find_span("lrpar.recoverer"), &Value::Namespaced("CPCTPlus".to_string(), src.find_span("CPCTPlus")) @@ -1132,7 +1132,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1162,7 +1162,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1192,7 +1192,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.grmtools_section_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar", "yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( From 2c47d2952c39bcede1c60ea6a53d13a617ff2bb1 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 13:03:45 -0700 Subject: [PATCH 12/38] Make crate name optional in `unused_header_keys_for_crate` --- cfgrammar/src/lib/yacc/ast.rs | 35 +++++++++++++++++++++++------------ lrpar/src/lib/codegen.rs | 18 +++++++++--------- 2 files changed, 32 insertions(+), 21 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 8c7c37595..ae470223f 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,7 +14,10 @@ use super::{ use crate::{ Span, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, + header::{ + GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, RE_CRATE_DOT, + Value, + }, yacc::YaccOriginalActionKind, }; @@ -127,16 +130,24 @@ impl ASTWithValidityInfo { } } - pub fn unused_header_keys_for_crate(&self, crate_name: &str) -> Vec<(String, Span)> { + pub fn unused_header_keys_for_crate(&self, crate_name: Option<&str>) -> Vec<(String, Span)> { self.grmtools_section .unused() .iter() .filter_map(|(key_name, HeaderValue(key_span, _))| { - let crate_prefix = format!("{crate_name}."); - if key_name.starts_with(&crate_prefix) { - Some((key_name.clone(), *key_span)) + if let Some(crate_name) = crate_name { + let crate_prefix = format!("{crate_name}."); + if key_name.starts_with(&crate_prefix) { + Some((key_name.clone(), *key_span)) + } else { + None + } } else { - None + if !RE_CRATE_DOT.is_match(key_name) { + Some((key_name.clone(), *key_span)) + } else { + None + } } }) .collect::>() @@ -1086,7 +1097,7 @@ start -> () : "a" { () }; assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( - ast_validity.unused_header_keys_for_crate("test"), + ast_validity.unused_header_keys_for_crate(Some("test")), vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( @@ -1099,7 +1110,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); @@ -1113,7 +1124,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate("lrpar") + .unused_header_keys_for_crate(Some("lrpar")) .is_empty() ); } @@ -1143,7 +1154,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); } @@ -1173,7 +1184,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); } @@ -1203,7 +1214,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); } diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index c589a6643..345002597 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1337,7 +1337,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); build_env @@ -1349,7 +1349,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar") + .unused_header_keys_for_crate(Some("lrpar")) .is_empty() ); build_env @@ -1358,7 +1358,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar") + .unused_header_keys_for_crate(Some("lrpar")) .is_empty() ); build_env.check_unused_header_keys_for_crate(None).unwrap(); @@ -1388,7 +1388,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("cfgrammar") + .unused_header_keys_for_crate(Some("cfgrammar")) .is_empty() ); build_env @@ -1397,7 +1397,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar") + .unused_header_keys_for_crate(Some("lrpar")) .is_empty() ); build_env @@ -1406,7 +1406,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrlex") + .unused_header_keys_for_crate(Some("lrlex")) .is_empty() ); match build_env.check_unused_header_keys_for_crate(None) { @@ -1443,7 +1443,7 @@ mod test { assert_eq!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("cfgrammar"), + .unused_header_keys_for_crate(Some("cfgrammar")), vec![( "cfgrammar.unknown".to_string(), src.find_span("cfgrammar.unknown") @@ -1457,7 +1457,7 @@ mod test { assert_eq!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar"), + .unused_header_keys_for_crate(Some("lrpar")), vec![("lrpar.unknown".to_string(), src.find_span("lrpar.unknown"))] ); build_env @@ -1466,7 +1466,7 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate("lrlex") + .unused_header_keys_for_crate(Some("lrlex")) .is_empty() ); let codegen = build_env.code_generator("timestamp").unwrap(); From 1cde52891a1ddbcef5b00272ccd6fa710d242f3f Mon Sep 17 00:00:00 2001 From: matt rice Date: Sat, 12 Sep 2026 13:13:52 -0700 Subject: [PATCH 13/38] Document the unused value checks --- cfgrammar/src/lib/yacc/ast.rs | 3 +++ lrpar/src/lib/codegen.rs | 3 +++ 2 files changed, 6 insertions(+) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index ae470223f..1526958b0 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -130,6 +130,9 @@ impl ASTWithValidityInfo { } } + /// Returns all key names given in the header specified by a `%grmtools` directive with the + /// `crate_name.` prefix for the given crate. If the `crate_name` is None returns any unused + /// keys with no crate prefix specified. pub fn unused_header_keys_for_crate(&self, crate_name: Option<&str>) -> Vec<(String, Span)> { self.grmtools_section .unused() diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 345002597..1ad7ed751 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -459,6 +459,9 @@ where self.ast_with_validity_info.yacc_kind() } + /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with a + /// `crate_name.` prefix for `crate_name` value are found. If the `crate_name` is None returns + /// an error if any unused keys with no crate prefix specified are found. pub(crate) fn check_unused_header_keys_for_crate( &self, crate_name: Option<&str>, From 9e0c16401bf0c733b4474b2059f44bd2756d5928 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sun, 13 Sep 2026 09:20:54 -0700 Subject: [PATCH 14/38] Move the find_span helper to test_utils --- cfgrammar/src/lib/mod.rs | 3 +++ cfgrammar/src/lib/yacc/ast.rs | 13 +------------ lrpar/src/lib/codegen.rs | 14 +------------- lrpar/src/lib/mod.rs | 5 +++-- lrpar/src/lib/test_utils.rs | 13 ++++++++++++- lrtable/src/lib/mod.rs | 3 +++ 6 files changed, 23 insertions(+), 28 deletions(-) diff --git a/cfgrammar/src/lib/mod.rs b/cfgrammar/src/lib/mod.rs index fb68ceed2..bb4d47144 100644 --- a/cfgrammar/src/lib/mod.rs +++ b/cfgrammar/src/lib/mod.rs @@ -68,6 +68,9 @@ pub mod yacc; pub use newlinecache::NewlineCache; pub use span::{Location, Span, Spanned}; +#[cfg(test)] +pub mod test_utils; + /// A type specifically for rule indices. pub use crate::idxnewtype::{PIdx, RIdx, SIdx, TIdx}; diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 1526958b0..df6ed652e 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -599,6 +599,7 @@ mod test { super::{AssocKind, Precedence}, GrammarAST, Span, Symbol, YaccGrammarError, YaccGrammarErrorKind, }; + use crate::test_utils::FindSpan as _; fn rule(n: &str) -> Symbol { Symbol::Rule(n.to_string(), Span::new(0, 0)) @@ -1221,16 +1222,4 @@ start: "a" { () }; .is_empty() ); } - - trait FindSpan { - fn find_span(&self, s: &str) -> Span; - } - - impl FindSpan for &'_ str { - #[track_caller] - fn find_span(&self, s: &str) -> Span { - let start_pos = self.find(s).unwrap(); - Span::new(start_pos, start_pos + s.len()) - } - } } diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 1ad7ed751..824b9a8be 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1318,7 +1318,7 @@ pub(crate) fn make_generics(parse_generics: Option<&str>) -> Result Span; - } - - impl FindSpan for &'_ str { - #[track_caller] - fn find_span(&self, s: &str) -> Span { - let start_pos = self.find(s).unwrap(); - Span::new(start_pos, start_pos + s.len()) - } - } } diff --git a/lrpar/src/lib/mod.rs b/lrpar/src/lib/mod.rs index a9ed046f3..da84e843a 100644 --- a/lrpar/src/lib/mod.rs +++ b/lrpar/src/lib/mod.rs @@ -202,11 +202,12 @@ mod dijkstra; pub mod lex_api; #[doc(hidden)] pub mod parser; -#[cfg(test)] -pub mod test_utils; mod codegen; +#[cfg(test)] +pub mod test_utils; + pub use crate::{ ctbuilder::{ CTParser, CTParserBuilder, ParserData, RustEdition, SerialisationFormat, Visibility, diff --git a/lrpar/src/lib/test_utils.rs b/lrpar/src/lib/test_utils.rs index 0fcf94dda..7b66e2cdb 100644 --- a/lrpar/src/lib/test_utils.rs +++ b/lrpar/src/lib/test_utils.rs @@ -1,4 +1,3 @@ -#![allow(clippy::len_without_is_empty)] #![allow(unused)] use std::{error::Error, fmt, hash::Hash}; @@ -87,3 +86,15 @@ impl fmt::Display for TestLexError { unreachable!(); } } + +pub trait FindSpan { + fn find_span(&self, s: &str) -> Span; +} + +impl FindSpan for &'_ str { + #[track_caller] + fn find_span(&self, s: &str) -> Span { + let start_pos = self.find(s).unwrap(); + Span::new(start_pos, start_pos + s.len()) + } +} diff --git a/lrtable/src/lib/mod.rs b/lrtable/src/lib/mod.rs index a9cde69f2..936ae2966 100644 --- a/lrtable/src/lib/mod.rs +++ b/lrtable/src/lib/mod.rs @@ -17,6 +17,9 @@ mod pager; mod stategraph; pub mod statetable; +#[cfg(test)] +mod test_utils; + pub use crate::{ stategraph::StateGraph, statetable::{Action, StateTable, StateTableError, StateTableErrorKind}, From f3079aaae2bbabfdb585a13ef2fbd001a9246b29 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sun, 13 Sep 2026 16:41:54 -0700 Subject: [PATCH 15/38] Make unknown keys without a crate prefix an error --- cfgrammar/src/lib/header.rs | 10 +++++--- lrpar/cttests/src/lib.rs | 40 ++++++++++++++--------------- lrpar/src/lib/codegen.rs | 51 ++++++++++--------------------------- 3 files changed, 40 insertions(+), 61 deletions(-) diff --git a/cfgrammar/src/lib/header.rs b/cfgrammar/src/lib/header.rs index 05a9ab19f..677676e02 100644 --- a/cfgrammar/src/lib/header.rs +++ b/cfgrammar/src/lib/header.rs @@ -14,7 +14,7 @@ use std::{collections::HashMap, error::Error, fmt, sync::LazyLock}; /// /// * An error during parsing the section. /// * An error resulting from a value in the section having an invalid value. -#[derive(Debug, Clone)] +#[derive(Debug, Clone, PartialEq, Eq)] #[doc(hidden)] pub struct HeaderError { pub kind: HeaderErrorKind, @@ -358,7 +358,11 @@ impl<'input> GrmtoolsSectionParser<'input> { if let Some(crate_name) = CRATE_KEY_MAP.get(key.as_str()) { format!("{crate_name}.{key}") } else { - key + errs.push(HeaderError { + kind: HeaderErrorKind::IllegalName, + locations: vec![key_loc], + }); + return Err(errs); } } else { key @@ -588,7 +592,7 @@ mod test { #[test] fn test_header_duplicates() { - let src = "%grmtools {dupe, !dupe, dupe: test}"; + let src = "%grmtools {test.dupe, !test.dupe, test.dupe: test}"; for flag in [true, false] { let parser = GrmtoolsSectionParser::new(src, flag); let res = parser.parse(); diff --git a/lrpar/cttests/src/lib.rs b/lrpar/cttests/src/lib.rs index 6b0eeb61f..f9d898988 100644 --- a/lrpar/cttests/src/lib.rs +++ b/lrpar/cttests/src/lib.rs @@ -399,26 +399,26 @@ fn test_grmtools_section_files() { fn test_grmtools_section_strings() { let srcs = [ "%grmtools{}", - "%grmtools{x}", - "%grmtools{x,}", - "%grmtools{!x}", - "%grmtools{!x,}", - "%grmtools{x: y}", - "%grmtools{x: y,}", - "%grmtools{x, y}", - "%grmtools{x, y,}", - "%grmtools{x, !y}", - "%grmtools{x, !y,}", - "%grmtools{x: y(z)}", - "%grmtools{x: y(z),}", - "%grmtools{a, x: y(z),}", - "%grmtools{a, x: y(z)}", - "%grmtools{a, !b, x: y(z), e: f}", - "%grmtools{a, !b, x: y(z), e: f,}", - "%grmtools{a, !b, x: w::y(z), e: f}", - "%grmtools{a, !b, x: w::y(z), e: f,}", - "%grmtools{a, !b, x: w::y(z), e: g::f}", - "%grmtools{a, !b, x: w::y(z), e: g::f,}", + "%grmtools{test.x}", + "%grmtools{test.x,}", + "%grmtools{!test.x}", + "%grmtools{!test.x,}", + "%grmtools{test.x: test.y}", + "%grmtools{test.x: test.y,}", + "%grmtools{test.x, test.y}", + "%grmtools{test.x, test.y,}", + "%grmtools{test.x, !test.y}", + "%grmtools{test.x, !test.y,}", + "%grmtools{test.x: y(z)}", + "%grmtools{test.x: y(z),}", + "%grmtools{test.a, test.x: y(z),}", + "%grmtools{test.a, test.x: y(z)}", + "%grmtools{test.a, !test.b, test.x: y(z), test.e: f}", + "%grmtools{test.a, !test.b, test.x: y(z), test.e: f,}", + "%grmtools{test.a, !test.b, test.x: w::y(z), test.e: f}", + "%grmtools{test.a, !test.b, test.x: w::y(z), test.e: f,}", + "%grmtools{test.a, !test.b, test.x: w::y(z), test.e: g::f}", + "%grmtools{test.a, !test.b, test.x: w::y(z), test.e: g::f,}", ]; let lexerdef = grmtools_section_l::lexerdef(); diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 824b9a8be..c0207a4ba 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1319,7 +1319,10 @@ pub(crate) fn make_generics(parse_generics: Option<&str>) -> Result::new(); let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); - let build_env = src_env - .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) - .unwrap(); - build_env - .check_unused_header_keys_for_crate(Some("cfgrammar")) - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate(Some("cfgrammar")) - .is_empty() - ); - build_env - .check_unused_header_keys_for_crate(Some("lrpar")) - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrpar")) - .is_empty() - ); - build_env - .check_unused_header_keys_for_crate(Some("lrlex")) - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrlex")) - .is_empty() - ); - match build_env.check_unused_header_keys_for_crate(None) { - Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) - if keys == vec!["testfoo".to_string()] => {} - _ => panic!("Unexpected return value for unused header keys check"), + let expected_errs = vec![HeaderError { + kind: HeaderErrorKind::IllegalName, + locations: vec![src.find_span("testfoo")], + }]; + match src_env.build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) { + Err(ParserSrcEnvError::GrmtoolsSectionParseError(errs)) => { + assert_eq!(errs, expected_errs) + } + _ => panic!("Unexpected err"), } - let codegen = build_env.code_generator("timestamp").unwrap(); - let out = codegen.generate(&build_env).unwrap(); - assert!(!out.is_empty()); } #[test] From ab83ba4deb829b9ec08aef3dd9bd6e8817898dec Mon Sep 17 00:00:00 2001 From: matt rice Date: Sun, 13 Sep 2026 20:40:29 -0700 Subject: [PATCH 16/38] Simplify unused header check arguments --- cfgrammar/src/lib/yacc/ast.rs | 38 ++++++++++++------------------ lrpar/src/lib/codegen.rs | 44 ++++++++++++++++++----------------- lrpar/src/lib/ctbuilder.rs | 8 +++---- 3 files changed, 42 insertions(+), 48 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index df6ed652e..8e1191b99 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,10 +14,7 @@ use super::{ use crate::{ Span, - header::{ - GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, RE_CRATE_DOT, - Value, - }, + header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, yacc::YaccOriginalActionKind, }; @@ -133,24 +130,19 @@ impl ASTWithValidityInfo { /// Returns all key names given in the header specified by a `%grmtools` directive with the /// `crate_name.` prefix for the given crate. If the `crate_name` is None returns any unused /// keys with no crate prefix specified. - pub fn unused_header_keys_for_crate(&self, crate_name: Option<&str>) -> Vec<(String, Span)> { + pub fn unused_header_keys_for_crate(&self, crate_prefix: &str) -> Vec<(String, Span)> { self.grmtools_section .unused() .iter() .filter_map(|(key_name, HeaderValue(key_span, _))| { - if let Some(crate_name) = crate_name { - let crate_prefix = format!("{crate_name}."); - if key_name.starts_with(&crate_prefix) { - Some((key_name.clone(), *key_span)) - } else { - None - } + if crate_prefix.is_empty() + || key_name + .strip_prefix(crate_prefix) + .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) + { + Some((key_name.clone(), *key_span)) } else { - if !RE_CRATE_DOT.is_match(key_name) { - Some((key_name.clone(), *key_span)) - } else { - None - } + None } }) .collect::>() @@ -1101,7 +1093,7 @@ start -> () : "a" { () }; assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( - ast_validity.unused_header_keys_for_crate(Some("test")), + ast_validity.unused_header_keys_for_crate("test"), vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( @@ -1114,7 +1106,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate(Some("cfgrammar")) + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); @@ -1128,7 +1120,7 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate(Some("lrpar")) + .unused_header_keys_for_crate("lrpar") .is_empty() ); } @@ -1158,7 +1150,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate(Some("cfgrammar")) + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } @@ -1188,7 +1180,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate(Some("cfgrammar")) + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } @@ -1218,7 +1210,7 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate(Some("cfgrammar")) + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); } diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index c0207a4ba..bd1b3c060 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -13,7 +13,7 @@ use crate::{ use cfgrammar::{ Location, RIdx, Span, Symbol, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue, RE_CRATE_DOT}, + header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue}, markmap::MergeError, yacc::{ YaccGrammar, YaccGrammarError, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo, @@ -464,18 +464,17 @@ where /// an error if any unused keys with no crate prefix specified are found. pub(crate) fn check_unused_header_keys_for_crate( &self, - crate_name: Option<&str>, + crate_prefix: &str, ) -> Result<(), ParserBuildEnvError> { let unused_keys = self .header .unused() .iter() - .filter(|(s, _)| { - if let Some(crate_name) = crate_name { - s.starts_with(&format!("{crate_name}.")) - } else { - !RE_CRATE_DOT.is_match(s) - } + .filter(|(key_name, _)| { + crate_prefix.is_empty() + || key_name + .strip_prefix(crate_prefix) + .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) }) .map(|(s, _)| s.to_string()) .collect::>(); @@ -1343,31 +1342,34 @@ mod test { assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("cfgrammar")) + .unused_header_keys_for_crate("cfgrammar") .is_empty() ); build_env - .check_unused_header_keys_for_crate(Some("cfgrammar")) + .check_unused_header_keys_for_crate("cfgrammar") .unwrap(); build_env - .check_unused_header_keys_for_crate(Some("lrpar")) + .check_unused_header_keys_for_crate("lrpar") .unwrap(); assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrpar")) + .unused_header_keys_for_crate("lrpar") .is_empty() ); build_env - .check_unused_header_keys_for_crate(Some("lrlex")) + .check_unused_header_keys_for_crate("lrlex") .unwrap(); assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrpar")) + .unused_header_keys_for_crate("lrpar") .is_empty() ); - build_env.check_unused_header_keys_for_crate(None).unwrap(); + match build_env.check_unused_header_keys_for_crate("") { + Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => assert_eq!(&keys, &["test.foo".to_string()]), + _ => panic!("Unexpected error result"), + } let codegen = build_env.code_generator("timestamp").unwrap(); let out = codegen.generate(&build_env).unwrap(); assert!(!out.is_empty()); @@ -1415,13 +1417,13 @@ mod test { .unwrap(); assert!( build_env - .check_unused_header_keys_for_crate(Some("cfgrammar")) + .check_unused_header_keys_for_crate("cfgrammar") .is_err() ); assert_eq!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("cfgrammar")), + .unused_header_keys_for_crate("cfgrammar"), vec![( "cfgrammar.unknown".to_string(), src.find_span("cfgrammar.unknown") @@ -1429,22 +1431,22 @@ mod test { ); assert!( build_env - .check_unused_header_keys_for_crate(Some("lrpar")) + .check_unused_header_keys_for_crate("lrpar") .is_err() ); assert_eq!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrpar")), + .unused_header_keys_for_crate("lrpar"), vec![("lrpar.unknown".to_string(), src.find_span("lrpar.unknown"))] ); build_env - .check_unused_header_keys_for_crate(Some("lrlex")) + .check_unused_header_keys_for_crate("lrlex") .unwrap(); assert!( build_env .ast_with_validity_info() - .unused_header_keys_for_crate(Some("lrlex")) + .unused_header_keys_for_crate("lrlex") .is_empty() ); let codegen = build_env.code_generator("timestamp").unwrap(); diff --git a/lrpar/src/lib/ctbuilder.rs b/lrpar/src/lib/ctbuilder.rs index 1d7a1feaf..588de7237 100644 --- a/lrpar/src/lib/ctbuilder.rs +++ b/lrpar/src/lib/ctbuilder.rs @@ -749,18 +749,18 @@ where // Catch any typos in key names for cfgrammar or lrpar build_env - .check_unused_header_keys_for_crate(Some("cfgrammar")) + .check_unused_header_keys_for_crate("cfgrammar") .map_err(|e| ErrorString(e.to_string()))?; build_env - .check_unused_header_keys_for_crate(Some("lrpar")) + .check_unused_header_keys_for_crate("lrpar") .map_err(|e| ErrorString(e.to_string()))?; // Catch any stray lrlex keys that accidentally make their way into the parser src. build_env - .check_unused_header_keys_for_crate(Some("lrlex")) + .check_unused_header_keys_for_crate("lrlex") .map_err(|e| ErrorString(e.to_string()))?; // Catch any stray keys without a crate prefix. build_env - .check_unused_header_keys_for_crate(None) + .check_unused_header_keys_for_crate("") .map_err(|e| ErrorString(e.to_string()))?; self.output_file( &code_gen, From 75127c6d7513f07af0110632f5b339781db2e7cb Mon Sep 17 00:00:00 2001 From: matt rice Date: Sun, 13 Sep 2026 20:51:55 -0700 Subject: [PATCH 17/38] make header_value_for_crate take one arg --- cfgrammar/src/lib/yacc/ast.rs | 33 ++++++++++++++------------------- lrpar/src/lib/codegen.rs | 6 ++++-- 2 files changed, 18 insertions(+), 21 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 8e1191b99..376f7f40c 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -113,14 +113,9 @@ impl ASTWithValidityInfo { /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. /// If the entry is found it marks the key as `used`, for the purposes of `unused_header_keys_for_crate`. - pub fn header_value_for_crate( - &mut self, - crate_name: &str, - key_name: &str, - ) -> Option<(Span, &Value)> { - let key = format!("{crate_name}.{key_name}"); - self.grmtools_section.mark_used(&key); - if let Some(HeaderValue(span, value)) = self.grmtools_section.get(&key) { + pub fn header_value_for_crate(&mut self, key: &str) -> Option<(Span, &Value)> { + self.grmtools_section.mark_used(&key.to_string()); + if let Some(HeaderValue(span, value)) = self.grmtools_section.get(key) { Some((*span, value)) } else { None @@ -1048,28 +1043,28 @@ start -> () : "a" { () }; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); for (key, (expected_span, expected_value)) in [ ( - "Flag", + "test.Flag", ( src.find_span("test.Flag"), Value::Bool(true, src.find_span("test.Flag")), ), ), ( - "Negative", + "test.Negative", ( src.find_span("test.Negative"), Value::Bool(false, src.find_span("!test.Negative")), ), ), ( - "string", + "test.string", ( src.find_span("test.string"), Value::String("Foo".to_string(), src.find_span("Foo")), ), ), ( - "vec", + "test.vec", ( src.find_span("test.vec"), Value::Array( @@ -1082,14 +1077,14 @@ start -> () : "a" { () }; ), ), ( - "num", + "test.num", ( src.find_span("test.num"), Value::Num(1234, src.find_span("1234")), ), ), ] { - let value = ast_validity.header_value_for_crate("test", key); + let value = ast_validity.header_value_for_crate(key); assert_eq!(value, Some((expected_span, &expected_value))); } assert_eq!( @@ -1097,7 +1092,7 @@ start -> () : "a" { () }; vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) @@ -1111,7 +1106,7 @@ start -> () : "a" { () }; ); assert_eq!( - ast_validity.header_value_for_crate("lrpar", "recoverer"), + ast_validity.header_value_for_crate("lrpar.recoverer"), Some(( src.find_span("lrpar.recoverer"), &Value::Namespaced("CPCTPlus".to_string(), src.find_span("CPCTPlus")) @@ -1139,7 +1134,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1169,7 +1164,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1199,7 +1194,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar", "yacckind"), + ast_validity.header_value_for_crate("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index bd1b3c060..9e02f0e1a 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1367,8 +1367,10 @@ mod test { .is_empty() ); match build_env.check_unused_header_keys_for_crate("") { - Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => assert_eq!(&keys, &["test.foo".to_string()]), - _ => panic!("Unexpected error result"), + Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { + assert_eq!(&keys, &["test.foo".to_string()]) + } + _ => panic!("Unexpected error result"), } let codegen = build_env.code_generator("timestamp").unwrap(); let out = codegen.generate(&build_env).unwrap(); From a1d8a8f3deed1a8c9b47e978da6eeb877db10bd4 Mon Sep 17 00:00:00 2001 From: matt rice Date: Sun, 13 Sep 2026 20:59:10 -0700 Subject: [PATCH 18/38] Update docs --- cfgrammar/src/lib/yacc/ast.rs | 4 ++-- lrpar/src/lib/codegen.rs | 6 +++--- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 376f7f40c..2d3f4f2c4 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -123,8 +123,8 @@ impl ASTWithValidityInfo { } /// Returns all key names given in the header specified by a `%grmtools` directive with the - /// `crate_name.` prefix for the given crate. If the `crate_name` is None returns any unused - /// keys with no crate prefix specified. + /// `crate_prefix.` prefix for the given crate. If the `crate_prefix` is empty returns all + /// unused keys regardless of crate. pub fn unused_header_keys_for_crate(&self, crate_prefix: &str) -> Vec<(String, Span)> { self.grmtools_section .unused() diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 9e02f0e1a..14913d459 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -459,9 +459,9 @@ where self.ast_with_validity_info.yacc_kind() } - /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with a - /// `crate_name.` prefix for `crate_name` value are found. If the `crate_name` is None returns - /// an error if any unused keys with no crate prefix specified are found. + /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with + /// `crate_prefix.` are found. If the `crate_prefix` is empty returns an error if any unused + /// keys are found. pub(crate) fn check_unused_header_keys_for_crate( &self, crate_prefix: &str, From 0ca6955033305e46c33f278ac138e1ce969c1418 Mon Sep 17 00:00:00 2001 From: matt rice Date: Tue, 15 Sep 2026 23:42:08 -0700 Subject: [PATCH 19/38] First try at avoiding multiple header structures --- cfgrammar/src/lib/yacc/ast.rs | 24 ++++++ lrlex/src/lib/ctbuilder.rs | 1 - lrpar/src/lib/codegen.rs | 115 +++++++++++--------------- lrpar/src/lib/ctbuilder.rs | 21 +++-- lrpar/src/lib/parser.rs | 10 +-- nimbleparse/src/main.rs | 150 +++++++++++++--------------------- 6 files changed, 151 insertions(+), 170 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 2d3f4f2c4..e3d0ba28c 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -142,6 +142,30 @@ impl ASTWithValidityInfo { }) .collect::>() } + + pub fn check_missing_required_keys_for_crate(&self, crate_prefix: &str) -> Vec { + self.grmtools_section + .missing() + .iter() + .cloned() + .filter_map(|key_name| { + if crate_prefix.is_empty() + || key_name + .strip_prefix(crate_prefix) + .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) + { + Some(key_name.clone()) + } else { + None + } + }) + .collect::>() + } + + #[doc(hidden)] + pub fn header(&self) -> &Header { + &self.grmtools_section + } } impl FromStr for ASTWithValidityInfo { diff --git a/lrlex/src/lib/ctbuilder.rs b/lrlex/src/lib/ctbuilder.rs index 6a1eea4bd..43c290bfd 100644 --- a/lrlex/src/lib/ctbuilder.rs +++ b/lrlex/src/lib/ctbuilder.rs @@ -448,7 +448,6 @@ where .map(|(x, y)| (&**x, *y)) .collect::>(); closure_lexerdef.set_rule_ids(&owned_map); - yacc_header.mark_used(&"lrpar.test_files".to_string()); let grammar = rtpb.grammar(); let test_glob = yacc_header.get("lrpar.test_files"); let mut err_str = None; diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 14913d459..465410c8d 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -37,7 +37,7 @@ const ACTIONS_KIND_HIDDEN: &str = "__GtActionsKindHidden"; pub(crate) enum ParserSrcEnvError { GrmtoolsSectionParseError(Vec>), GrmtoolsSectionMergeError(MergeError>>), - GrmtoolsSectionLookupError(HeaderError), + GrmtoolsSectionLookupError(HeaderError), MissingYaccKind, MissingModName, } @@ -51,7 +51,7 @@ where { StateTableError(StateTableError), YaccGrammarErrors(Vec), - GrmtoolsSectionUnusedKeys(Vec), + GrmtoolsSectionUnusedKeys(Vec<(String, Span)>), GrmtoolsSectionMissingRequiredKeys(Vec), } @@ -76,8 +76,8 @@ impl From>>> for ParserSrcEnvError } } -impl From> for ParserSrcEnvError { - fn from(it: HeaderError) -> Self { +impl From> for ParserSrcEnvError { + fn from(it: HeaderError) -> Self { ParserSrcEnvError::GrmtoolsSectionLookupError(it) } } @@ -144,6 +144,7 @@ where .collect::>() .join("\n"), Self::GrmtoolsSectionUnusedKeys(keys) => { + let keys = keys.iter().cloned().map(|(s, _)| s).collect::>(); format!("Unused keys in %grmtools section: {}", keys.join(", ")) } Self::GrmtoolsSectionMissingRequiredKeys(keys) => format!( @@ -175,7 +176,8 @@ where src: &'a str, fallback_modname: Option, grammar_path_cache_entry: Option, - header: Header, + yacckind: Option, + recoverer: Option, phantom: PhantomData, } @@ -203,7 +205,6 @@ where phantom_storaget: PhantomData, mod_name: String, grammar_path: Option, - header: Header, } pub(crate) struct ParserCodegen @@ -269,11 +270,7 @@ where LexerTypesT: LexerTypes, usize: num_traits::AsPrimitive, { - pub(crate) fn new_with_header( - src: &'a str, - path: Option<&Path>, - header: Header, - ) -> ParserSrcEnv<'a, LexerTypesT> { + pub(crate) fn new(src: &'a str, path: Option<&Path>) -> ParserSrcEnv<'a, LexerTypesT> { let fallback_modname = if let Some(path) = path { // When the user hasn't specified a module name, so we create one automatically: what we // do is strip off all the filename extensions (note that it's likely that inp ends @@ -296,18 +293,20 @@ where src, fallback_modname, grammar_path_cache_entry, - header, phantom: PhantomData, + recoverer: None, + yacckind: None, } } - fn merge_headers(&mut self) -> Result<(), ParserSrcEnvError> { - let (parsed_header, _) = self.parse_header()?; - Ok(self.header.merge_from(parsed_header)?) + pub(crate) fn yacckind(mut self, yacckind: Option) -> Self { + self.yacckind = yacckind; + self } - fn parse_header(&self) -> Result<(Header, usize), Vec>> { - GrmtoolsSectionParser::new(self.src, false).parse() + pub(crate) fn recoverer(mut self, recoverykind: Option) -> Self { + self.recoverer = recoverykind; + self } /// Looks up the `yacckind` field from the header, marks the field @@ -315,12 +314,13 @@ where fn resolve_ast_with_validity_info( &mut self, from_ast: Option<&ASTWithValidityInfo>, + header: &Header, ) -> Result { - self.header.mark_used(&"cfgrammar.yacckind".to_string()); if let Some(ast) = from_ast { Ok(ast.clone()) - } else if let Some(yk) = self - .header + } else if let Some(yk) = self.yacckind { + Ok(ASTWithValidityInfo::new(yk, self.src)) + } else if let Some(yk) = header .get("cfgrammar.yacckind") .map(|HeaderValue(_, val)| val) .map(YaccKind::try_from) @@ -334,13 +334,16 @@ where /// Looks up the `recoverer` field in the header, marks the field /// as used, and defaulting to `CPCTPlus` if unfound. - fn resolve_recoverer(&mut self) -> Result { - self.header.mark_used(&"lrpar.recoverer".to_string()); - let rk_val = self - .header + fn resolve_recoverer( + &mut self, + header: &Header, + ) -> Result { + let rk_val = header .get("lrpar.recoverer") .map(|HeaderValue(_, rk_val)| rk_val); - if let Some(rk_val) = rk_val { + if let Some(rk) = self.recoverer { + Ok(rk) + } else if let Some(rk_val) = rk_val { Ok(RecoveryKind::try_from(rk_val)?) } else { // Fallback to the default recoverykind. @@ -350,11 +353,11 @@ where /// Looks up the `serialisation_format` field in the header, marks the field /// as used, and defaults to `VariableSizedInteger` if unfound. - fn resolve_serialisation_format(&mut self) -> Result { - self.header - .mark_used(&"lrpar.serialisation_format".to_string()); - if let Some(ec_val) = self - .header + fn resolve_serialisation_format( + &mut self, + header: &Header, + ) -> Result { + if let Some(ec_val) = header .get("lrpar.serialisation_format") .map(|HeaderValue(_, ec_val)| ec_val) { @@ -385,11 +388,11 @@ where LexerTypesT: LexerTypes, usize: num_traits::AsPrimitive, { - self.merge_headers()?; + let (header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; let ast_with_validity_info = - self.resolve_ast_with_validity_info(args.ast_with_validity_info)?; - let recoverer = self.resolve_recoverer()?; - let serialisation_format = self.resolve_serialisation_format()?; + self.resolve_ast_with_validity_info(args.ast_with_validity_info, &header)?; + let recoverer = self.resolve_recoverer(&header)?; + let serialisation_format = self.resolve_serialisation_format(&header)?; let mod_name = self.resolve_mod_name(&args)?; let grammar_path = self.grammar_path_cache_entry; @@ -400,7 +403,6 @@ where serialisation_format, mod_name, grammar_path, - header: self.header, phantom_storaget: PhantomData, }) } @@ -415,10 +417,6 @@ where &self.ast_with_validity_info } - pub(crate) fn header_mut(&mut self) -> &mut Header { - &mut self.header - } - pub(crate) fn serialisation_format(&self) -> &SerialisationFormat { &self.serialisation_format } @@ -467,26 +465,14 @@ where crate_prefix: &str, ) -> Result<(), ParserBuildEnvError> { let unused_keys = self - .header - .unused() - .iter() - .filter(|(key_name, _)| { - crate_prefix.is_empty() - || key_name - .strip_prefix(crate_prefix) - .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) - }) - .map(|(s, _)| s.to_string()) - .collect::>(); + .ast_with_validity_info() + .unused_header_keys_for_crate(crate_prefix); if !unused_keys.is_empty() { return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_keys)); } let missing_keys = self - .header - .missing() - .iter() - .map(|s| s.to_string()) - .collect::>(); + .ast_with_validity_info() + .check_missing_required_keys_for_crate(crate_prefix); if !missing_keys.is_empty() { Err(ParserBuildEnvError::GrmtoolsSectionMissingRequiredKeys( missing_keys, @@ -1318,10 +1304,7 @@ pub(crate) fn make_generics(parse_generics: Option<&str>) -> Result () : "A" { () }; "#; - let empty_header = Header::::new(); - let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let src_env = ParserSrcEnv::::new(src, None); let build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); @@ -1368,7 +1350,10 @@ mod test { ); match build_env.check_unused_header_keys_for_crate("") { Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { - assert_eq!(&keys, &["test.foo".to_string()]) + assert_eq!( + &keys, + &[("test.foo".to_string(), src.find_span("test.foo"))] + ) } _ => panic!("Unexpected error result"), } @@ -1387,8 +1372,7 @@ mod test { %% start -> () : "A" { () }; "#; - let empty_header = Header::::new(); - let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let src_env = ParserSrcEnv::::new(src, None); let expected_errs = vec![HeaderError { kind: HeaderErrorKind::IllegalName, locations: vec![src.find_span("testfoo")], @@ -1412,8 +1396,7 @@ mod test { %% start -> () : "A" { () }; "#; - let empty_header = Header::::new(); - let src_env = ParserSrcEnv::::new_with_header(src, None, empty_header); + let src_env = ParserSrcEnv::::new(src, None); let build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); diff --git a/lrpar/src/lib/ctbuilder.rs b/lrpar/src/lib/ctbuilder.rs index 588de7237..304a8fb84 100644 --- a/lrpar/src/lib/ctbuilder.rs +++ b/lrpar/src/lib/ctbuilder.rs @@ -26,7 +26,7 @@ use crate::{ use crate::unstable_api::UnstableApi; use cfgrammar::{ - Location, + Location, Span, header::{Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, markmap::{Entry, MergeBehavior}, yacc::{YaccGrammar, YaccKind, ast::ASTWithValidityInfo}, @@ -201,7 +201,7 @@ where inspect_rt: Option< Box< dyn for<'b> FnMut( - &'b mut Header, + &'b Header, RTParserBuilder, &'b HashMap, &PathBuf, @@ -442,7 +442,7 @@ where mut self, cb: Box< dyn for<'b, 'y> FnMut( - &'b mut Header, + &'b Header, RTParserBuilder<'y, StorageT, LexerTypesT>, &'b HashMap, &PathBuf, @@ -575,7 +575,9 @@ where read_to_string(grmp).map_err(|e| format!("When reading '{}': {e}", grmp.display()))? }; - let src_env = ParserSrcEnv::new_with_header(&inc, Some(grmp), header); + let src_env = ParserSrcEnv::new(&inc, Some(grmp)) + .yacckind(self.yacckind) + .recoverer(self.recoverer); let yacc_diag = SpannedDiagnosticFormatter::new(&inc, grmp); let build_args = ParserBuildEnvArgs::new() .ast_with_validity_info(self.from_ast.as_ref()) @@ -585,7 +587,7 @@ where .warnings_are_errors(self.warnings_are_errors) .visibility(self.visibility.clone()) .rust_edition(self.rust_edition); - let mut build_env = src_env.build_env(build_args).map_err(|e| match e { + let build_env = src_env.build_env(build_args).map_err(|e| match e { ParserSrcEnvError::GrmtoolsSectionParseError(es) => { let mut out = String::new(); out.push_str(&format!( @@ -737,14 +739,19 @@ where } } - if let Some(ref mut inspector_rt) = self.inspect_rt { + if let Some(inspector_rt) = &mut self.inspect_rt { let rt: RTParserBuilder<'_, StorageT, LexerTypesT> = RTParserBuilder::new(grm, stable); let rt = if let Some(rk) = self.recoverer { rt.recoverer(rk) } else { rt }; - inspector_rt(build_env.header_mut(), rt, &rule_ids, grmp)? + inspector_rt( + build_env.ast_with_validity_info().header(), + rt, + &rule_ids, + grmp, + )? } // Catch any typos in key names for cfgrammar or lrpar diff --git a/lrpar/src/lib/parser.rs b/lrpar/src/lib/parser.rs index 34ca6adbd..8315657f6 100644 --- a/lrpar/src/lib/parser.rs +++ b/lrpar/src/lib/parser.rs @@ -648,9 +648,9 @@ impl TryFrom for Value { } } -impl TryFrom<&Value> for RecoveryKind { - type Error = cfgrammar::header::HeaderError; - fn try_from(rk: &Value) -> Result { +impl TryFrom<&Value> for RecoveryKind { + type Error = cfgrammar::header::HeaderError; + fn try_from(rk: &Value) -> Result { match rk { Value::Namespaced(rs, loc) => match rs.as_str() { "RecoveryKind::CPCTPlus" | "CPCTPlus" => Ok(RecoveryKind::CPCTPlus), @@ -660,7 +660,7 @@ impl TryFrom<&Value> for RecoveryKind { "RecoveryKind", "Cannot convert to RecoveryKind", ), - locations: vec![loc.clone()], + locations: vec![*loc], }), }, value => Err(HeaderError { @@ -668,7 +668,7 @@ impl TryFrom<&Value> for RecoveryKind { "RecoveryKind", "Cannot convert to RecoveryKind", ), - locations: vec![value.primary_location().clone()], + locations: vec![*value.primary_location()], }), } } diff --git a/nimbleparse/src/main.rs b/nimbleparse/src/main.rs index 7acf86dde..05ce7e7cd 100644 --- a/nimbleparse/src/main.rs +++ b/nimbleparse/src/main.rs @@ -1,7 +1,6 @@ use cfgrammar::{ - Location, RIdx, Span, TIdx, + RIdx, Span, TIdx, header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue, Value}, - markmap::Entry, yacc::{YaccGrammar, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo}, }; use getopts::Options; @@ -173,47 +172,17 @@ fn main() { let dump_state_graph = matches.opt_present("d"); let quiet = matches.opt_present("q"); - let mut header = Header::new(); - match matches.opt_str("r") { - None => (), - Some(s) => { - header.set_merge_behavior( - &"lrpar.recoverer".to_string(), - cfgrammar::markmap::MergeBehavior::Ours, - ); - header.insert( - "lrpar.recoverer".to_string(), - HeaderValue( - Location::CommandLine, - Value::try_from(match &*s.to_lowercase() { - "cpctplus" => RecoveryKind::CPCTPlus, - "none" => RecoveryKind::None, - _ => usage(prog, &format!("Unknown recoverer '{}'.", s)), - }) - .expect("All these RecoveryKinds should convert without error"), - ), - ); - } - }; - let entry = match header.entry("cfgrammar.yacckind".to_string()) { - Entry::Occupied(_) => unreachable!("Header should be empty"), - Entry::Vacant(v) => v, - }; - match matches.opt_str("y") { - None => {} - Some(s) => { - entry.insert_entry(HeaderValue( - Location::CommandLine, - Value::try_from(match &*s.to_lowercase() { - "eco" => YaccKind::Eco, - "grmtools" => YaccKind::Grmtools, - "original" => YaccKind::Original(YaccOriginalActionKind::GenericParseTree), - _ => usage(prog, &format!("Unknown Yacc variant '{}'.", s)), - }) - .expect("All these yacckinds should convert without error"), - )); - } - }; + let rk_arg = matches.opt_str("r").map(|s| match &*s.to_lowercase() { + "cpctplus" => RecoveryKind::CPCTPlus, + "none" => RecoveryKind::None, + _ => usage(prog, &format!("Unknown recoverer '{}'.", s)), + }); + let yk_arg = matches.opt_str("y").map(|s| match &*s.to_lowercase() { + "eco" => YaccKind::Eco, + "grmtools" => YaccKind::Grmtools, + "original" => YaccKind::Original(YaccOriginalActionKind::GenericParseTree), + _ => usage(prog, &format!("Unknown Yacc variant '{}'.", s)), + }); let args_len = matches.free.len(); if args_len < 2 { usage(prog, "Too few arguments given."); @@ -236,39 +205,23 @@ fn main() { let yacc_y_path = PathBuf::from(&matches.free[1]); let yacc_src = read_file(&yacc_y_path); let yacc_diag = SpannedDiagnosticFormatter::new(&yacc_src, &yacc_y_path); - let yk_val = header.get("cfgrammar.yacckind"); - if yk_val.is_none() { - let parsed_header = GrmtoolsSectionParser::new(&yacc_src, true).parse(); - match parsed_header { - Ok((parsed_header, _)) => { - header - .merge_from(parsed_header) - .expect("Specified merge behavior cannot fail"); - } - Err(errs) => { - eprintln!( - "{ERROR}{}", - yacc_diag.file_location_msg(" parsing the `%grmtools` section:", None) - ); - for e in errs { - eprintln!("{}", indent(" ", &yacc_diag.format_error(e).to_string())); - } - std::process::exit(1); + let parsed_header = GrmtoolsSectionParser::new(&yacc_src, true).parse(); + let parsed_header = match parsed_header { + Ok((parsed_header, _)) => parsed_header, + Err(errs) => { + eprintln!( + "{ERROR}{}", + yacc_diag.file_location_msg(" parsing the `%grmtools` section:", None) + ); + for e in errs { + eprintln!("{}", indent(" ", &yacc_diag.format_error(e).to_string())); } + std::process::exit(1); } - } - let yk_val = header.get("cfgrammar.yacckind"); - if yk_val.is_none() { - eprintln!( - "yacckind not specified in the %grmtools section of the grammar or via the '-y' parameter" - ); - std::process::exit(1); - } - let HeaderValue(_, yk_val) = yk_val.unwrap(); - let yacc_kind = YaccKind::try_from(yk_val).unwrap_or(YaccKind::Grmtools); - let ast_validation = ASTWithValidityInfo::new(yacc_kind, &yacc_src); - let recoverykind = if let Some(HeaderValue(_, rk_val)) = header.get("lrpar.recoverer") { - match RecoveryKind::try_from(rk_val) { + }; + let yk_header_val = parsed_header + .get("cfgrammar.yacckind") + .map(|HeaderValue(_, value)| match YaccKind::try_from(value) { Err(e) => { eprintln!( "{ERROR}{}", @@ -276,27 +229,42 @@ fn main() { ); let spanned_e: HeaderError = HeaderError { kind: e.kind, - locations: e - .locations - .iter() - .map(|l| match l { - Location::Span(span) => *span, - _ => unreachable!("All reachable errors should contain spans"), - }) - .collect::>(), + locations: e.locations.to_vec(), }; eprintln!( "{}", indent(" ", &yacc_diag.format_error(spanned_e).to_string()) ); - process::exit(1) + std::process::exit(1); } - Ok(rk) => rk, - } - } else { - // Fallback to the default recoverykind - RecoveryKind::CPCTPlus - }; + + Ok(yacc_kind) => yacc_kind, + }); + let yacc_kind = yk_arg.unwrap_or(yk_header_val.unwrap_or(YaccKind::Grmtools)); + let ast_validation = ASTWithValidityInfo::new(yacc_kind, &yacc_src); + let rk_header_val = parsed_header + .get("lrpar.recoverer") + .map( + |HeaderValue(_, value)| match RecoveryKind::try_from(value) { + Err(e) => { + eprintln!( + "{ERROR}{}", + yacc_diag.file_location_msg(" parsing the `%grmtools` section:", None) + ); + let spanned_e: HeaderError = HeaderError { + kind: e.kind, + locations: e.locations.to_vec(), + }; + eprintln!( + "{}", + indent(" ", &yacc_diag.format_error(spanned_e).to_string()) + ); + process::exit(1) + } + Ok(rk) => rk, + }, + ); + let recoverykind = rk_arg.unwrap_or(rk_header_val.unwrap_or(RecoveryKind::CPCTPlus)); let warnings = ast_validation.ast().warnings(); let res = YaccGrammar::new_from_ast_with_validity_info(&ast_validation); let grm = match res { @@ -414,7 +382,7 @@ fn main() { } let parser_build_ctxt = ParserBuildCtxt { - header, + header: parsed_header, lexerdef, grm, stable, @@ -448,7 +416,7 @@ where LexerTypesT: LexerTypes, usize: AsPrimitive, { - header: Header, + header: Header, lexerdef: LRNonStreamingLexerDef, grm: YaccGrammar, yacc_y_path: PathBuf, From 6cca7a6e9205c65a9f4407db01e2ea1a16f77199 Mon Sep 17 00:00:00 2001 From: matt rice Date: Wed, 16 Sep 2026 11:32:01 -0700 Subject: [PATCH 20/38] No longer need to build a header in CTBuilder --- cfgrammar/src/lib/header.rs | 8 ---- lrpar/src/lib/codegen.rs | 16 +++----- lrpar/src/lib/ctbuilder.rs | 74 +++++-------------------------------- lrpar/src/lib/parser.rs | 9 ----- 4 files changed, 15 insertions(+), 92 deletions(-) diff --git a/cfgrammar/src/lib/header.rs b/cfgrammar/src/lib/header.rs index 677676e02..fad804718 100644 --- a/cfgrammar/src/lib/header.rs +++ b/cfgrammar/src/lib/header.rs @@ -481,14 +481,6 @@ impl<'input> GrmtoolsSectionParser<'input> { #[doc(hidden)] pub type Header = MarkMap>; -impl TryFrom for Value { - type Error = HeaderError; - fn try_from(kind: YaccKind) -> Result, HeaderError> { - let from_loc = Location::Other("From".to_string()); - Ok(Value::Namespaced(format!("YaccKind::{kind:?}"), from_loc)) - } -} - impl TryFrom<&Value> for YaccKind { type Error = HeaderError; fn try_from(value: &Value) -> Result> { diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 465410c8d..60be14b1e 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -12,9 +12,8 @@ use crate::{ }; use cfgrammar::{ - Location, RIdx, Span, Symbol, + RIdx, Span, Symbol, header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue}, - markmap::MergeError, yacc::{ YaccGrammar, YaccGrammarError, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo, }, @@ -36,7 +35,6 @@ const ACTIONS_KIND_HIDDEN: &str = "__GtActionsKindHidden"; #[non_exhaustive] pub(crate) enum ParserSrcEnvError { GrmtoolsSectionParseError(Vec>), - GrmtoolsSectionMergeError(MergeError>>), GrmtoolsSectionLookupError(HeaderError), MissingYaccKind, MissingModName, @@ -70,12 +68,6 @@ impl From>> for ParserSrcEnvError { } } -impl From>>> for ParserSrcEnvError { - fn from(it: MergeError>>) -> Self { - ParserSrcEnvError::GrmtoolsSectionMergeError(it) - } -} - impl From> for ParserSrcEnvError { fn from(it: HeaderError) -> Self { ParserSrcEnvError::GrmtoolsSectionLookupError(it) @@ -122,7 +114,6 @@ impl fmt::Display for ParserSrcEnvError { .map(|e| e.to_string()) .collect::>() .join("\n"), - Self::GrmtoolsSectionMergeError(e) => e.to_string(), Self::GrmtoolsSectionLookupError(e) => e.to_string(), Self::MissingYaccKind => "Code generator cannot resolve yacc kind".to_string(), Self::MissingModName => "Code generator requires a mod name".to_string(), @@ -388,7 +379,10 @@ where LexerTypesT: LexerTypes, usize: num_traits::AsPrimitive, { - let (header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; + let (mut header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; + if self.yacckind.is_none() { + header.mark_required(&"cfgrammar.yacckind".to_string()); + } let ast_with_validity_info = self.resolve_ast_with_validity_info(args.ast_with_validity_info, &header)?; let recoverer = self.resolve_recoverer(&header)?; diff --git a/lrpar/src/lib/ctbuilder.rs b/lrpar/src/lib/ctbuilder.rs index 304a8fb84..f1e0affba 100644 --- a/lrpar/src/lib/ctbuilder.rs +++ b/lrpar/src/lib/ctbuilder.rs @@ -26,9 +26,8 @@ use crate::{ use crate::unstable_api::UnstableApi; use cfgrammar::{ - Location, Span, - header::{Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, - markmap::{Entry, MergeBehavior}, + Span, + header::{Header, HeaderError, HeaderErrorKind, Value}, yacc::{YaccGrammar, YaccKind, ast::ASTWithValidityInfo}, }; use filetime::FileTime; @@ -135,22 +134,11 @@ pub enum SerialisationFormat { VariableSizedInteger, } -impl TryFrom for Value { - type Error = cfgrammar::header::HeaderError; - fn try_from(kind: SerialisationFormat) -> Result, HeaderError> { - let from_loc = Location::Other("From".to_string()); - Ok(Value::Namespaced( - format!("SerialisationFormat::{kind:?}"), - from_loc, - )) - } -} - -impl TryFrom<&Value> for SerialisationFormat { - type Error = HeaderError; - fn try_from(value: &Value) -> Result> { +impl TryFrom<&Value> for SerialisationFormat { + type Error = HeaderError; + fn try_from(value: &Value) -> Result> { match value { - Value::Namespaced(serialisation_fmt, loc) => match serialisation_fmt.as_str() { + Value::Namespaced(serialisation_fmt, span) => match serialisation_fmt.as_str() { "SerialisationFormat::FixedSizeInteger" | "FixedSizeInteger" => { Ok(SerialisationFormat::FixedSizeInteger) } @@ -159,12 +147,12 @@ impl TryFrom<&Value> for SerialisationFormat { } _ => Err(HeaderError { kind: HeaderErrorKind::InvalidEntry("serialisation_format"), - locations: vec![loc.clone()], + locations: vec![*span], }), }, val => Err(HeaderError { kind: HeaderErrorKind::InvalidEntry("serialisation_format"), - locations: vec![val.primary_location().clone()], + locations: vec![*val.primary_location()], }), } } @@ -514,51 +502,9 @@ where .output_path .as_ref() .expect("output_path must be specified before processing."); - let mut header = Header::new(); - - match header.entry("cfgrammar.yacckind".to_string()) { - Entry::Occupied(_) => unreachable!(), - Entry::Vacant(mut v) => match self.yacckind { - Some(YaccKind::Eco) => panic!("Eco compile-time grammar generation not supported."), - Some(yk) => { - let yk_value = Value::try_from(yk)?; - let mut o = v.insert_entry(HeaderValue( - Location::Other("CTParserBuilder".to_string()), - yk_value, - )); - o.set_merge_behavior(MergeBehavior::Ours); - } - None => { - v.mark_required(); - } - }, - } - if let Some(recoverer) = self.recoverer { - match header.entry("lrpar.recoverer".to_string()) { - Entry::Occupied(_) => unreachable!(), - Entry::Vacant(v) => { - let rk_value = Value::try_from(recoverer)?; - let mut o = v.insert_entry(HeaderValue( - Location::Other("CTParserBuilder".to_string()), - rk_value, - )); - o.set_merge_behavior(MergeBehavior::Ours); - } - } - } - if let Some(encoding) = self.serialisation_format { - match header.entry("lrpar.serialisation_format".to_string()) { - Entry::Occupied(_) => unreachable!(), - Entry::Vacant(v) => { - let rk_value = Value::try_from(encoding)?; - let mut o = v.insert_entry(HeaderValue( - Location::Other("CTParserBuilder".to_string()), - rk_value, - )); - o.set_merge_behavior(MergeBehavior::Ours); - } - } + if let Some(YaccKind::Eco) = self.yacckind { + panic!("Eco compile-time grammar generation not supported.") } { diff --git a/lrpar/src/lib/parser.rs b/lrpar/src/lib/parser.rs index 8315657f6..c8b987940 100644 --- a/lrpar/src/lib/parser.rs +++ b/lrpar/src/lib/parser.rs @@ -17,7 +17,6 @@ use cactus::Cactus; use cfgrammar::{ RIdx, Span, TIdx, header::{HeaderError, HeaderErrorKind, Value}, - span::Location, yacc::YaccGrammar, }; use lrtable::{Action, StIdx, StateTable}; @@ -640,14 +639,6 @@ pub enum RecoveryKind { None, } -impl TryFrom for Value { - type Error = cfgrammar::header::HeaderError; - fn try_from(rk: RecoveryKind) -> Result, Self::Error> { - let from_loc = Location::Other("From".to_string()); - Ok(Value::Namespaced(format!("RecoveryKind::{rk:?}"), from_loc)) - } -} - impl TryFrom<&Value> for RecoveryKind { type Error = cfgrammar::header::HeaderError; fn try_from(rk: &Value) -> Result { From a4e14d5db78b1256ad2e90a3edbdff5a4d32cd0d Mon Sep 17 00:00:00 2001 From: matt rice Date: Wed, 16 Sep 2026 16:49:24 -0700 Subject: [PATCH 21/38] remove mark_required call, MissingYaccKind error covers this case --- lrpar/src/lib/codegen.rs | 3 --- 1 file changed, 3 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 60be14b1e..f16f76346 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -380,9 +380,6 @@ where usize: num_traits::AsPrimitive, { let (mut header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; - if self.yacckind.is_none() { - header.mark_required(&"cfgrammar.yacckind".to_string()); - } let ast_with_validity_info = self.resolve_ast_with_validity_info(args.ast_with_validity_info, &header)?; let recoverer = self.resolve_recoverer(&header)?; From 45fca2b50a351233e24d03de33fc379ba20278e5 Mon Sep 17 00:00:00 2001 From: matt rice Date: Wed, 16 Sep 2026 22:57:34 -0700 Subject: [PATCH 22/38] Perform unused header checks automatically during the codegen process --- lrpar/src/lib/codegen.rs | 156 +++++++++++++++++---------------------- 1 file changed, 68 insertions(+), 88 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index f16f76346..4cccabd9f 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -196,6 +196,7 @@ where phantom_storaget: PhantomData, mod_name: String, grammar_path: Option, + crates_to_check: Vec, } pub(crate) struct ParserCodegen @@ -379,17 +380,22 @@ where LexerTypesT: LexerTypes, usize: num_traits::AsPrimitive, { - let (mut header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; + let (header, _) = GrmtoolsSectionParser::new(self.src, false).parse()?; let ast_with_validity_info = self.resolve_ast_with_validity_info(args.ast_with_validity_info, &header)?; let recoverer = self.resolve_recoverer(&header)?; let serialisation_format = self.resolve_serialisation_format(&header)?; let mod_name = self.resolve_mod_name(&args)?; let grammar_path = self.grammar_path_cache_entry; - + let crates_to_check = vec![ + "cfgrammar".to_string(), + "lrpar".to_string(), + "lrlex".to_string(), + ]; Ok(ParserBuildEnv { ast_with_validity_info, cache_args: args, + crates_to_check, recoverer, serialisation_format, mod_name, @@ -448,6 +454,16 @@ where self.ast_with_validity_info.yacc_kind() } + /// Causes the `code_generator()` function to check for unused entries in the grmtools section + /// starting for entries starting with `crate_prefix`. + #[allow(unused)] + pub(crate) fn add_value_checks_for_crate(&mut self, crate_prefix: &str) { + let crate_prefix = crate_prefix.to_string(); + if !self.crates_to_check.contains(&crate_prefix) { + self.crates_to_check.push(crate_prefix.to_string()); + } + } + /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with /// `crate_prefix.` are found. If the `crate_prefix` is empty returns an error if any unused /// keys are found. @@ -477,6 +493,18 @@ where &self, timestamp: &str, ) -> Result, ParserBuildEnvError> { + let mut unused_vals = Vec::new(); + for crate_prefix in &self.crates_to_check { + let result = self.check_unused_header_keys_for_crate(crate_prefix); + if let Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) = result { + unused_vals.extend(keys); + } else if result.is_err() { + result? + } + } + if !unused_vals.is_empty() { + return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_vals)); + } let grm = YaccGrammar::::new_from_ast_with_validity_info( &self.ast_with_validity_info, )?; @@ -1300,57 +1328,31 @@ mod test { use super::*; #[test] fn test_unused_crate_header_entry() { - let src = r#" - %grmtools{ - yacckind: Grmtools, - test.foo: "test crate value", - } - %% - start -> () : "A" { () }; - "#; - let src_env = ParserSrcEnv::::new(src, None); - let build_env = src_env - .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("cfgrammar") - .is_empty() - ); - build_env - .check_unused_header_keys_for_crate("cfgrammar") - .unwrap(); - build_env - .check_unused_header_keys_for_crate("lrpar") - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar") - .is_empty() - ); - build_env - .check_unused_header_keys_for_crate("lrlex") - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar") - .is_empty() - ); - match build_env.check_unused_header_keys_for_crate("") { - Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { - assert_eq!( - &keys, - &[("test.foo".to_string(), src.find_span("test.foo"))] - ) + for crate_prefix in ["test", ""] { + let src = r#" + %grmtools{ + yacckind: Grmtools, + test.foo: "test crate value", + } + %% + start -> () : "A" { () }; + "#; + let src_env = ParserSrcEnv::::new(src, None); + let mut build_env = src_env + .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) + .unwrap(); + build_env.add_value_checks_for_crate(crate_prefix); + match build_env.code_generator("timestamp") { + Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { + assert_eq!( + &keys, + &[("test.foo".to_string(), src.find_span("test.foo"))] + ) + } + Err(e) => panic!("Unexpected error result: {:?}", e), + _ => panic!("Unexpected Ok return value"), } - _ => panic!("Unexpected error result"), } - let codegen = build_env.code_generator("timestamp").unwrap(); - let out = codegen.generate(&build_env).unwrap(); - assert!(!out.is_empty()); } #[test] @@ -1391,42 +1393,20 @@ mod test { let build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); - assert!( - build_env - .check_unused_header_keys_for_crate("cfgrammar") - .is_err() - ); - assert_eq!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("cfgrammar"), - vec![( - "cfgrammar.unknown".to_string(), - src.find_span("cfgrammar.unknown") - )] - ); - assert!( - build_env - .check_unused_header_keys_for_crate("lrpar") - .is_err() - ); - assert_eq!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("lrpar"), - vec![("lrpar.unknown".to_string(), src.find_span("lrpar.unknown"))] - ); - build_env - .check_unused_header_keys_for_crate("lrlex") - .unwrap(); - assert!( - build_env - .ast_with_validity_info() - .unused_header_keys_for_crate("lrlex") - .is_empty() - ); - let codegen = build_env.code_generator("timestamp").unwrap(); - let out = codegen.generate(&build_env).unwrap(); - assert!(!out.is_empty()); + match build_env.code_generator("timestamp") { + Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { + assert_eq!( + &keys, + &[ + ( + "cfgrammar.unknown".to_string(), + src.find_span("cfgrammar.unknown") + ), + ("lrpar.unknown".to_string(), src.find_span("lrpar.unknown")) + ] + ); + } + _ => panic!("Unexpected error result"), + } } } From c587fab757626fd44699de847bf86e505fb3af13 Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 17 Sep 2026 00:09:02 -0700 Subject: [PATCH 23/38] Remove manual unused checks --- lrpar/src/lib/ctbuilder.rs | 15 --------------- 1 file changed, 15 deletions(-) diff --git a/lrpar/src/lib/ctbuilder.rs b/lrpar/src/lib/ctbuilder.rs index f1e0affba..e534789e3 100644 --- a/lrpar/src/lib/ctbuilder.rs +++ b/lrpar/src/lib/ctbuilder.rs @@ -700,21 +700,6 @@ where )? } - // Catch any typos in key names for cfgrammar or lrpar - build_env - .check_unused_header_keys_for_crate("cfgrammar") - .map_err(|e| ErrorString(e.to_string()))?; - build_env - .check_unused_header_keys_for_crate("lrpar") - .map_err(|e| ErrorString(e.to_string()))?; - // Catch any stray lrlex keys that accidentally make their way into the parser src. - build_env - .check_unused_header_keys_for_crate("lrlex") - .map_err(|e| ErrorString(e.to_string()))?; - // Catch any stray keys without a crate prefix. - build_env - .check_unused_header_keys_for_crate("") - .map_err(|e| ErrorString(e.to_string()))?; self.output_file( &code_gen, outp, From c3b3f6cd6c27b7e2ad118687f1c5b578a608eabd Mon Sep 17 00:00:00 2001 From: matt rice Date: Mon, 21 Sep 2026 15:20:07 -0700 Subject: [PATCH 24/38] Update after review --- cfgrammar/src/lib/markmap.rs | 24 ++++---- cfgrammar/src/lib/yacc/ast.rs | 110 ++++++++++++++++++---------------- lrlex/src/lib/ctbuilder.rs | 1 - lrlex/src/main.rs | 1 - lrpar/src/lib/codegen.rs | 52 +++++----------- nimbleparse/src/main.rs | 35 ++++++++++- 6 files changed, 118 insertions(+), 105 deletions(-) diff --git a/cfgrammar/src/lib/markmap.rs b/cfgrammar/src/lib/markmap.rs index 9e2a3d245..6841a0b00 100644 --- a/cfgrammar/src/lib/markmap.rs +++ b/cfgrammar/src/lib/markmap.rs @@ -484,18 +484,15 @@ impl MarkMap { } /// Returns a `Vec` containing all the keys that are not marked as used. - pub fn unused(&self) -> Vec<(K, V)> - where - V: Clone, - { - let mut ret = Vec::new(); - for (k, mark, v) in &self.contents { + pub fn unused(&self) -> impl Iterator { + self.contents.iter().filter_map(|(k, mark, v)| { let used_mark = Mark::Used.repr(); if v.is_some() && mark & used_mark == 0 { - ret.push((k.to_owned(), v.as_ref().unwrap().clone())) + Some((k, v.as_ref().unwrap())) + } else { + None } - } - ret + }) } /// Returns a `Vec` containing all the keys that are marked as required, @@ -714,8 +711,8 @@ mod test { assert!(mm.insert("a", "test").is_none()); mm.mark_used(&"a"); assert_eq!(mm.get_mark(&"a"), Some(Mark::Used.repr())); - let empty: &[(&str, &str)] = &[]; - assert_eq!(mm.unused().as_slice(), empty); + let empty: &[(&&str, &&str)] = &[]; + assert_eq!(mm.unused().collect::>(), empty); } { @@ -725,7 +722,10 @@ mod test { assert!(mm.insert("b", "unused").is_none()); assert_eq!(mm.get_mark(&"a"), Some(Mark::Used.repr())); assert_eq!(mm.get_mark(&"b"), Some(0)); - assert_eq!(mm.unused().as_slice(), &[("b", "unused")]); + assert_eq!( + mm.unused().collect::>(), + vec![(&"b", &"unused")] + ); } } diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index e3d0ba28c..32f9840d7 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,7 +14,10 @@ use super::{ use crate::{ Span, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, + header::{ + GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, RE_CRATE_DOT, + Value, + }, yacc::YaccOriginalActionKind, }; @@ -113,7 +116,7 @@ impl ASTWithValidityInfo { /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. /// If the entry is found it marks the key as `used`, for the purposes of `unused_header_keys_for_crate`. - pub fn header_value_for_crate(&mut self, key: &str) -> Option<(Span, &Value)> { + pub fn header_value_get(&mut self, key: &str) -> Option<(Span, &Value)> { self.grmtools_section.mark_used(&key.to_string()); if let Some(HeaderValue(span, value)) = self.grmtools_section.get(key) { Some((*span, value)) @@ -125,41 +128,36 @@ impl ASTWithValidityInfo { /// Returns all key names given in the header specified by a `%grmtools` directive with the /// `crate_prefix.` prefix for the given crate. If the `crate_prefix` is empty returns all /// unused keys regardless of crate. - pub fn unused_header_keys_for_crate(&self, crate_prefix: &str) -> Vec<(String, Span)> { + #[doc(hidden)] + pub fn iter_unused_header_values( + &self, + prefixes: HashSet, + ) -> impl Iterator { self.grmtools_section .unused() - .iter() - .filter_map(|(key_name, HeaderValue(key_span, _))| { - if crate_prefix.is_empty() - || key_name - .strip_prefix(crate_prefix) - .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) - { - Some((key_name.clone(), *key_span)) - } else { - None + .filter_map(move |(key_name, HeaderValue(key_span, _))| { + if prefixes.contains("") { + return Some((key_name.clone(), *key_span)); } - }) - .collect::>() - } - - pub fn check_missing_required_keys_for_crate(&self, crate_prefix: &str) -> Vec { - self.grmtools_section - .missing() - .iter() - .cloned() - .filter_map(|key_name| { - if crate_prefix.is_empty() - || key_name - .strip_prefix(crate_prefix) - .is_some_and(|rest| crate_prefix.ends_with('.') || rest.starts_with('.')) - { - Some(key_name.clone()) + if let Some(prefix_match) = RE_CRATE_DOT.find(key_name) { + let key_prefix = prefix_match.as_str(); + if let Some(key_prefix) = key_prefix.strip_suffix('.') { + if prefixes.contains(key_prefix) { + Some((key_name.clone(), *key_span)) + } else { + if prefixes.contains(key_prefix) { + Some((key_name.clone(), *key_span)) + } else { + None + } + } + } else { + None + } } else { None } }) - .collect::>() } #[doc(hidden)] @@ -1067,28 +1065,28 @@ start -> () : "a" { () }; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); for (key, (expected_span, expected_value)) in [ ( - "test.Flag", + "test.Flag".to_string(), ( src.find_span("test.Flag"), Value::Bool(true, src.find_span("test.Flag")), ), ), ( - "test.Negative", + "test.Negative".to_string(), ( src.find_span("test.Negative"), Value::Bool(false, src.find_span("!test.Negative")), ), ), ( - "test.string", + "test.string".to_string(), ( src.find_span("test.string"), Value::String("Foo".to_string(), src.find_span("Foo")), ), ), ( - "test.vec", + "test.vec".to_string(), ( src.find_span("test.vec"), Value::Array( @@ -1101,22 +1099,25 @@ start -> () : "a" { () }; ), ), ( - "test.num", + "test.num".to_string(), ( src.find_span("test.num"), Value::Num(1234, src.find_span("1234")), ), ), ] { - let value = ast_validity.header_value_for_crate(key); + let value = ast_validity.header_value_get(&key); assert_eq!(value, Some((expected_span, &expected_value))); } + let crate_prefixes = HashSet::from_iter(["test".to_string()]); assert_eq!( - ast_validity.unused_header_keys_for_crate("test"), + ast_validity + .iter_unused_header_values(crate_prefixes) + .collect::>(), vec![("test.unused".to_string(), src.find_span("test.unused"))] ); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar.yacckind"), + ast_validity.header_value_get("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) @@ -1125,12 +1126,13 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") - .is_empty() + .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .next() + .is_none() ); assert_eq!( - ast_validity.header_value_for_crate("lrpar.recoverer"), + ast_validity.header_value_get("lrpar.recoverer"), Some(( src.find_span("lrpar.recoverer"), &Value::Namespaced("CPCTPlus".to_string(), src.find_span("CPCTPlus")) @@ -1139,8 +1141,9 @@ start -> () : "a" { () }; assert!( ast_validity - .unused_header_keys_for_crate("lrpar") - .is_empty() + .iter_unused_header_values(HashSet::from_iter(["lrpar".to_string()])) + .next() + .is_none() ); } @@ -1158,7 +1161,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar.yacckind"), + ast_validity.header_value_get("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1169,8 +1172,9 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") - .is_empty() + .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .next() + .is_none() ); } @@ -1188,7 +1192,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar.yacckind"), + ast_validity.header_value_get("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1199,8 +1203,9 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") - .is_empty() + .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .next() + .is_none() ); } @@ -1218,7 +1223,7 @@ start: "a" { () }; "#; let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( - ast_validity.header_value_for_crate("cfgrammar.yacckind"), + ast_validity.header_value_get("cfgrammar.yacckind"), Some(( src.find_span("yacckind"), &Value::Namespaced( @@ -1229,8 +1234,9 @@ start: "a" { () }; ); assert!( ast_validity - .unused_header_keys_for_crate("cfgrammar") - .is_empty() + .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .next() + .is_none() ); } } diff --git a/lrlex/src/lib/ctbuilder.rs b/lrlex/src/lib/ctbuilder.rs index 43c290bfd..7fab89341 100644 --- a/lrlex/src/lib/ctbuilder.rs +++ b/lrlex/src/lib/ctbuilder.rs @@ -522,7 +522,6 @@ where let unused_header_values = build_env .header() .unused() - .iter() .map(|(s, _)| s.to_string()) .collect::>(); if !unused_header_values.is_empty() { diff --git a/lrlex/src/main.rs b/lrlex/src/main.rs index bb87599f2..0320ddae7 100644 --- a/lrlex/src/main.rs +++ b/lrlex/src/main.rs @@ -125,7 +125,6 @@ fn main() -> Result<(), Box> { { let unused_header_values = header .unused() - .iter() .map(|(s, _)| s.to_string()) .collect::>(); if !unused_header_values.is_empty() { diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 4cccabd9f..ebefe9e37 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1,5 +1,6 @@ use std::{ any::type_name, + collections::HashSet, fmt::{self, Write}, hash::Hash, marker::PhantomData, @@ -50,7 +51,6 @@ where StateTableError(StateTableError), YaccGrammarErrors(Vec), GrmtoolsSectionUnusedKeys(Vec<(String, Span)>), - GrmtoolsSectionMissingRequiredKeys(Vec), } #[derive(Debug)] @@ -138,10 +138,6 @@ where let keys = keys.iter().cloned().map(|(s, _)| s).collect::>(); format!("Unused keys in %grmtools section: {}", keys.join(", ")) } - Self::GrmtoolsSectionMissingRequiredKeys(keys) => format!( - "Required keys are missing from %grmtools section: {}", - keys.join(", ") - ), }) } } @@ -196,7 +192,7 @@ where phantom_storaget: PhantomData, mod_name: String, grammar_path: Option, - crates_to_check: Vec, + crates_to_check: HashSet, } pub(crate) struct ParserCodegen @@ -395,7 +391,7 @@ where Ok(ParserBuildEnv { ast_with_validity_info, cache_args: args, - crates_to_check, + crates_to_check: HashSet::from_iter(crates_to_check), recoverer, serialisation_format, mod_name, @@ -457,54 +453,34 @@ where /// Causes the `code_generator()` function to check for unused entries in the grmtools section /// starting for entries starting with `crate_prefix`. #[allow(unused)] - pub(crate) fn add_value_checks_for_crate(&mut self, crate_prefix: &str) { - let crate_prefix = crate_prefix.to_string(); - if !self.crates_to_check.contains(&crate_prefix) { - self.crates_to_check.push(crate_prefix.to_string()); - } + pub(crate) fn register_header_key_prefix(&mut self, crate_prefix: &str) { + self.crates_to_check.insert(crate_prefix.to_string()); } /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with /// `crate_prefix.` are found. If the `crate_prefix` is empty returns an error if any unused /// keys are found. - pub(crate) fn check_unused_header_keys_for_crate( + pub(crate) fn check_unused_header_keys( &self, - crate_prefix: &str, + crate_prefixes: HashSet, ) -> Result<(), ParserBuildEnvError> { let unused_keys = self .ast_with_validity_info() - .unused_header_keys_for_crate(crate_prefix); + .iter_unused_header_values(crate_prefixes) + .collect::>(); if !unused_keys.is_empty() { return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_keys)); } - let missing_keys = self - .ast_with_validity_info() - .check_missing_required_keys_for_crate(crate_prefix); - if !missing_keys.is_empty() { - Err(ParserBuildEnvError::GrmtoolsSectionMissingRequiredKeys( - missing_keys, - )) - } else { - Ok(()) - } + Ok(()) } pub(crate) fn code_generator( &self, timestamp: &str, ) -> Result, ParserBuildEnvError> { - let mut unused_vals = Vec::new(); - for crate_prefix in &self.crates_to_check { - let result = self.check_unused_header_keys_for_crate(crate_prefix); - if let Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) = result { - unused_vals.extend(keys); - } else if result.is_err() { - result? - } - } - if !unused_vals.is_empty() { - return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_vals)); - } + let mut crate_prefixes = HashSet::new(); + crate_prefixes.extend(self.crates_to_check.clone()); + self.check_unused_header_keys(crate_prefixes)?; let grm = YaccGrammar::::new_from_ast_with_validity_info( &self.ast_with_validity_info, )?; @@ -1341,7 +1317,7 @@ mod test { let mut build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); - build_env.add_value_checks_for_crate(crate_prefix); + build_env.register_header_key_prefix(crate_prefix); match build_env.code_generator("timestamp") { Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { assert_eq!( diff --git a/nimbleparse/src/main.rs b/nimbleparse/src/main.rs index 05ce7e7cd..bcf4dd4ad 100644 --- a/nimbleparse/src/main.rs +++ b/nimbleparse/src/main.rs @@ -14,6 +14,7 @@ use lrtable::{Minimiser, StateTable, from_yacc}; use num_traits::ToPrimitive as _; use num_traits::{AsPrimitive, PrimInt, Unsigned}; use std::{ + collections::HashSet, env, error::Error, fmt, @@ -205,7 +206,8 @@ fn main() { let yacc_y_path = PathBuf::from(&matches.free[1]); let yacc_src = read_file(&yacc_y_path); let yacc_diag = SpannedDiagnosticFormatter::new(&yacc_src, &yacc_y_path); - let parsed_header = GrmtoolsSectionParser::new(&yacc_src, true).parse(); + // If we were given no yk_arg on the command line, require a grmtools section. + let parsed_header = GrmtoolsSectionParser::new(&yacc_src, yk_arg.is_none()).parse(); let parsed_header = match parsed_header { Ok((parsed_header, _)) => parsed_header, Err(errs) => { @@ -242,6 +244,37 @@ fn main() { }); let yacc_kind = yk_arg.unwrap_or(yk_header_val.unwrap_or(YaccKind::Grmtools)); let ast_validation = ASTWithValidityInfo::new(yacc_kind, &yacc_src); + // Note we don't expect to find any used lrlex keys, we want to produce an error if unused ones are found. + let crate_prefixes = HashSet::from_iter([ + "cfgrammar".to_string(), + "lrpar".to_string(), + "lrlex".to_string(), + ]); + let unused_keys = ast_validation + .iter_unused_header_values(crate_prefixes) + .collect::>(); + if !unused_keys.is_empty() { + eprintln!( + "{ERROR}{}", + yacc_diag.file_location_msg(" parsing the `%grmtools` section:", None) + ); + for (_, span) in unused_keys { + eprintln!( + "{}", + indent( + " ", + &yacc_diag + .underline_span_with_text( + span, + "Unused key in grmtools section".to_string(), + '^' + ) + .to_string() + ) + ); + } + process::exit(1) + } let rk_header_val = parsed_header .get("lrpar.recoverer") .map( From e8e1e7fb248c801fd0abcbaaa5c3c7b811be31c4 Mon Sep 17 00:00:00 2001 From: matt rice Date: Mon, 21 Sep 2026 15:30:05 -0700 Subject: [PATCH 25/38] Update comment --- cfgrammar/src/lib/yacc/parser.rs | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/cfgrammar/src/lib/yacc/parser.rs b/cfgrammar/src/lib/yacc/parser.rs index 39e8f60c9..62371126c 100644 --- a/cfgrammar/src/lib/yacc/parser.rs +++ b/cfgrammar/src/lib/yacc/parser.rs @@ -376,11 +376,10 @@ impl YaccParser<'_> { pub(crate) fn build(self) -> (GrammarAST, Header) { let mut header = self.header.expect("set by parse()"); - // Preemptively mark the keys for lrpar and cfgrammar as used in the header. - // If a downstream crate checks the keys in the ast. The lrpar crate works on a - // local instance which merges the keys from ast with keys from the `CTBuilder`. - // - // It is difficult to do later due to shared references. + // At this point we still have mutable access to the header, so it is a convenient place + // to mark keys used. It would be less error prone if we did this at the point where keys + // are used. However at some points where we do lookups, there are shared references to the + // header making it difficult to get mutable access. for (key_name, crate_name) in CRATE_KEY_MAP.iter() { if ["cfgrammar", "lrpar"].contains(crate_name) { header.mark_used(&format!("{crate_name}.{key_name}")); From 22d5f114560c16bc0240ee3ff205e1e877a6099a Mon Sep 17 00:00:00 2001 From: matt rice Date: Tue, 22 Sep 2026 04:30:46 -0700 Subject: [PATCH 26/38] Take set of crate/key prefixes by ref --- cfgrammar/src/lib/yacc/ast.rs | 14 +++++++------- lrpar/src/lib/codegen.rs | 6 ++---- nimbleparse/src/main.rs | 2 +- 3 files changed, 10 insertions(+), 12 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 32f9840d7..eee3e9caa 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -131,7 +131,7 @@ impl ASTWithValidityInfo { #[doc(hidden)] pub fn iter_unused_header_values( &self, - prefixes: HashSet, + prefixes: &HashSet, ) -> impl Iterator { self.grmtools_section .unused() @@ -1112,7 +1112,7 @@ start -> () : "a" { () }; let crate_prefixes = HashSet::from_iter(["test".to_string()]); assert_eq!( ast_validity - .iter_unused_header_values(crate_prefixes) + .iter_unused_header_values(&crate_prefixes) .collect::>(), vec![("test.unused".to_string(), src.find_span("test.unused"))] ); @@ -1126,7 +1126,7 @@ start -> () : "a" { () }; assert!( ast_validity - .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) .next() .is_none() ); @@ -1141,7 +1141,7 @@ start -> () : "a" { () }; assert!( ast_validity - .iter_unused_header_values(HashSet::from_iter(["lrpar".to_string()])) + .iter_unused_header_values(&HashSet::from_iter(["lrpar".to_string()])) .next() .is_none() ); @@ -1172,7 +1172,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) .next() .is_none() ); @@ -1203,7 +1203,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) .next() .is_none() ); @@ -1234,7 +1234,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(HashSet::from_iter(["cfgrammar".to_string()])) + .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) .next() .is_none() ); diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index ebefe9e37..82ee04772 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -462,7 +462,7 @@ where /// keys are found. pub(crate) fn check_unused_header_keys( &self, - crate_prefixes: HashSet, + crate_prefixes: &HashSet, ) -> Result<(), ParserBuildEnvError> { let unused_keys = self .ast_with_validity_info() @@ -478,9 +478,7 @@ where &self, timestamp: &str, ) -> Result, ParserBuildEnvError> { - let mut crate_prefixes = HashSet::new(); - crate_prefixes.extend(self.crates_to_check.clone()); - self.check_unused_header_keys(crate_prefixes)?; + self.check_unused_header_keys(&self.crates_to_check)?; let grm = YaccGrammar::::new_from_ast_with_validity_info( &self.ast_with_validity_info, )?; diff --git a/nimbleparse/src/main.rs b/nimbleparse/src/main.rs index bcf4dd4ad..b98ab4383 100644 --- a/nimbleparse/src/main.rs +++ b/nimbleparse/src/main.rs @@ -251,7 +251,7 @@ fn main() { "lrlex".to_string(), ]); let unused_keys = ast_validation - .iter_unused_header_values(crate_prefixes) + .iter_unused_header_values(&crate_prefixes) .collect::>(); if !unused_keys.is_empty() { eprintln!( From 74c2ba019dc901cb18c427e339036378c0c33877 Mon Sep 17 00:00:00 2001 From: matt rice Date: Tue, 22 Sep 2026 04:38:13 -0700 Subject: [PATCH 27/38] Update doc strings. --- cfgrammar/src/lib/yacc/ast.rs | 2 +- lrpar/src/lib/codegen.rs | 5 ++--- 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index eee3e9caa..5c492ed9d 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -115,7 +115,7 @@ impl ASTWithValidityInfo { } /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. - /// If the entry is found it marks the key as `used`, for the purposes of `unused_header_keys_for_crate`. + /// If the entry is found it marks the key as `used`, for the purposes of `iter_unused_header_values`. pub fn header_value_get(&mut self, key: &str) -> Option<(Span, &Value)> { self.grmtools_section.mark_used(&key.to_string()); if let Some(HeaderValue(span, value)) = self.grmtools_section.get(key) { diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 82ee04772..a8c469b27 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -457,9 +457,8 @@ where self.crates_to_check.insert(crate_prefix.to_string()); } - /// Returns an error if any unused keys specified in a `%grmtools` directive that begin with - /// `crate_prefix.` are found. If the `crate_prefix` is empty returns an error if any unused - /// keys are found. + /// Checks all the keys staring with `crate_prefixes`. If any of them are `unused`, return an error. + /// If `crate_prefixes` contains the empty string, returns an arror if any key is unused. pub(crate) fn check_unused_header_keys( &self, crate_prefixes: &HashSet, From 70ce6f13cd4df83aa81a70b5ef95ac0cc54958ef Mon Sep 17 00:00:00 2001 From: matt rice Date: Tue, 22 Sep 2026 04:51:21 -0700 Subject: [PATCH 28/38] Rename a few functions ofr uniformity --- lrpar/src/lib/codegen.rs | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index a8c469b27..f2ae9e257 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -453,13 +453,13 @@ where /// Causes the `code_generator()` function to check for unused entries in the grmtools section /// starting for entries starting with `crate_prefix`. #[allow(unused)] - pub(crate) fn register_header_key_prefix(&mut self, crate_prefix: &str) { + pub(crate) fn register_header_value_prefix(&mut self, crate_prefix: &str) { self.crates_to_check.insert(crate_prefix.to_string()); } /// Checks all the keys staring with `crate_prefixes`. If any of them are `unused`, return an error. /// If `crate_prefixes` contains the empty string, returns an arror if any key is unused. - pub(crate) fn check_unused_header_keys( + fn check_unused_header_values( &self, crate_prefixes: &HashSet, ) -> Result<(), ParserBuildEnvError> { @@ -477,7 +477,7 @@ where &self, timestamp: &str, ) -> Result, ParserBuildEnvError> { - self.check_unused_header_keys(&self.crates_to_check)?; + self.check_unused_header_values(&self.crates_to_check)?; let grm = YaccGrammar::::new_from_ast_with_validity_info( &self.ast_with_validity_info, )?; @@ -1314,7 +1314,7 @@ mod test { let mut build_env = src_env .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); - build_env.register_header_key_prefix(crate_prefix); + build_env.register_header_value_prefix(crate_prefix); match build_env.code_generator("timestamp") { Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { assert_eq!( From d7512adb9899fbc3bc7931b5760c218aeab3afec Mon Sep 17 00:00:00 2001 From: matt rice Date: Tue, 22 Sep 2026 17:23:15 -0700 Subject: [PATCH 29/38] Remove reference to hidden function in docs --- cfgrammar/src/lib/yacc/ast.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 5c492ed9d..60a8af409 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -115,7 +115,7 @@ impl ASTWithValidityInfo { } /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. - /// If the entry is found it marks the key as `used`, for the purposes of `iter_unused_header_values`. + /// If the entry is found it marks the key as `used`. pub fn header_value_get(&mut self, key: &str) -> Option<(Span, &Value)> { self.grmtools_section.mark_used(&key.to_string()); if let Some(HeaderValue(span, value)) = self.grmtools_section.get(key) { From 804a34d661af067a1aec23f04a722e36d71b03fe Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 1 Oct 2026 17:18:13 -0700 Subject: [PATCH 30/38] initial work towards external checking of header values. This migrates the checking to be done externally, so removes some test cases. It doesn't yet support "exhaustive" checking. In that it only currently checks for the provided prefixes. Further work is required to ensure that there are no unrecognized prefixes. A bug was also fixed in one of the `IntoIterator` implementations of `MarkMap`. --- cfgrammar/src/lib/header.rs | 49 ++++++++++++- cfgrammar/src/lib/markmap.rs | 15 ++-- cfgrammar/src/lib/yacc/ast.rs | 93 ++++++++++++------------- cfgrammar/src/lib/yacc/parser.rs | 4 +- lrpar/cttests/src/grmtools_section.test | 6 +- lrpar/src/lib/codegen.rs | 66 ++++++------------ nimbleparse/src/main.rs | 18 ++--- 7 files changed, 133 insertions(+), 118 deletions(-) diff --git a/cfgrammar/src/lib/header.rs b/cfgrammar/src/lib/header.rs index fad804718..5a8bea6b8 100644 --- a/cfgrammar/src/lib/header.rs +++ b/cfgrammar/src/lib/header.rs @@ -6,7 +6,12 @@ use crate::{ }, }; use regex::{Regex, RegexBuilder}; -use std::{collections::HashMap, error::Error, fmt, sync::LazyLock}; +use std::{ + collections::{HashMap, HashSet}, + error::Error, + fmt, + sync::LazyLock, +}; /// An error regarding the `%grmtools` header section. /// @@ -166,7 +171,7 @@ pub static RE_CRATE_DOT: LazyLock = LazyLock::new(|| { static RE_DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"^[0-9]+").unwrap()); static RE_STRING: LazyLock = LazyLock::new(|| Regex::new(r#"^\"(\\.|[^"\\])*\""#).unwrap()); #[doc(hidden)] -pub static CRATE_KEY_MAP: LazyLock> = LazyLock::new(|| { +pub static KEY_CRATE_MAP: LazyLock> = LazyLock::new(|| { let mut map = HashMap::new(); let cfgrammar = ["yacckind"]; let lrpar = ["recoverer", "test_files", "serialisation_format"]; @@ -199,6 +204,44 @@ pub static CRATE_KEY_MAP: LazyLock> = LazyLo map }); +#[doc(hidden)] +pub static CFGRAMMAR_KEYS: LazyLock> = + LazyLock::new(|| HashSet::from_iter(["cfgrammar.yacckind"])); + +#[doc(hidden)] +pub static LRPAR_KEYS: LazyLock> = LazyLock::new(|| { + HashSet::from_iter([ + "lrpar.recoverer", + "lrpar.test_files", + "lrpar.serialisation_format", + ]) +}); + +#[doc(hidden)] +pub static LRLEX_KEYS: LazyLock> = LazyLock::new(|| { + HashSet::from_iter([ + "lrlex.lexerkind", + "lrlex.allow_wholeline_comments", + "lrlex.posix_escapes", + ]) +}); + +#[doc(hidden)] +pub static REGEX_KEYS: LazyLock> = LazyLock::new(|| { + HashSet::from_iter([ + "regex.case_insensitive", + "regex.dot_matches_new_line", + "regex.multi_line", + "regex.octal", + "regex.swap_greed", + "regex.ignore_whitespace", + "regex.unicode", + "regex.size_limit", + "regex.dfa_size_limit", + "regex.nest_limit", + ]) +}); + const MAGIC: &str = "%grmtools"; fn add_duplicate_occurrence( @@ -355,7 +398,7 @@ impl<'input> GrmtoolsSectionParser<'input> { let (key, key_loc, val, j) = match self.parse_key_value(i) { Ok((key, key_loc, val, pos)) => { let key = if !RE_CRATE_DOT.is_match(&key) { - if let Some(crate_name) = CRATE_KEY_MAP.get(key.as_str()) { + if let Some(crate_name) = KEY_CRATE_MAP.get(key.as_str()) { format!("{crate_name}.{key}") } else { errs.push(HeaderError { diff --git a/cfgrammar/src/lib/markmap.rs b/cfgrammar/src/lib/markmap.rs index 6841a0b00..142887181 100644 --- a/cfgrammar/src/lib/markmap.rs +++ b/cfgrammar/src/lib/markmap.rs @@ -546,11 +546,18 @@ impl<'a, K, V> Iterator for MarkMapIterRef<'a, K, V> { type Item = (&'a K, &'a V); fn next(&mut self) -> Option { - if let Some((k, _, v)) = self.map.contents.get(self.pos) { + loop { + if self.pos >= self.map.contents.len() { + return None; + } + let pos = self.pos; self.pos += 1; - v.as_ref().map(|v| (k, v)) - } else { - None + if self.map.contents[pos].2.is_some() { + return Some(( + &self.map.contents[pos].0, + self.map.contents[pos].2.as_ref().unwrap(), + )); + } } } } diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 60a8af409..3e62030d6 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,10 +14,7 @@ use super::{ use crate::{ Span, - header::{ - GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, RE_CRATE_DOT, - Value, - }, + header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, yacc::YaccOriginalActionKind, }; @@ -115,9 +112,7 @@ impl ASTWithValidityInfo { } /// Performs a lookup in the grmtools section for an entry with the key `crate_name.key_name` and returns it. - /// If the entry is found it marks the key as `used`. - pub fn header_value_get(&mut self, key: &str) -> Option<(Span, &Value)> { - self.grmtools_section.mark_used(&key.to_string()); + pub fn header_value_get(&self, key: &str) -> Option<(Span, &Value)> { if let Some(HeaderValue(span, value)) = self.grmtools_section.get(key) { Some((*span, value)) } else { @@ -125,41 +120,30 @@ impl ASTWithValidityInfo { } } - /// Returns all key names given in the header specified by a `%grmtools` directive with the - /// `crate_prefix.` prefix for the given crate. If the `crate_prefix` is empty returns all - /// unused keys regardless of crate. - #[doc(hidden)] - pub fn iter_unused_header_values( - &self, - prefixes: &HashSet, - ) -> impl Iterator { - self.grmtools_section - .unused() - .filter_map(move |(key_name, HeaderValue(key_span, _))| { - if prefixes.contains("") { - return Some((key_name.clone(), *key_span)); - } - if let Some(prefix_match) = RE_CRATE_DOT.find(key_name) { - let key_prefix = prefix_match.as_str(); - if let Some(key_prefix) = key_prefix.strip_suffix('.') { - if prefixes.contains(key_prefix) { - Some((key_name.clone(), *key_span)) - } else { - if prefixes.contains(key_prefix) { - Some((key_name.clone(), *key_span)) - } else { - None - } - } - } else { - None - } + pub fn iter_prefix_keys(&self, prefix: &str) -> impl Iterator { + let prefix = format!("{prefix}."); + (&self.grmtools_section) + .into_iter() + .filter_map(move |(key, val)| { + let HeaderValue(span, _) = val; + eprintln!("prefix: {prefix:?} key: {key:?}"); + if key.starts_with(&prefix) { + Some((key.as_str(), *span)) } else { None } }) } + pub fn unrecognized_keys_for_prefix( + &self, + prefix: &str, + valid_keys: &HashSet<&str>, + ) -> impl Iterator { + self.iter_prefix_keys(prefix) + .filter(|(key, _)| !valid_keys.contains(*key)) + } + #[doc(hidden)] pub fn header(&self) -> &Header { &self.grmtools_section @@ -608,7 +592,10 @@ mod test { super::{AssocKind, Precedence}, GrammarAST, Span, Symbol, YaccGrammarError, YaccGrammarErrorKind, }; - use crate::test_utils::FindSpan as _; + use crate::{ + header::{CFGRAMMAR_KEYS, LRPAR_KEYS}, + test_utils::FindSpan as _, + }; fn rule(n: &str) -> Symbol { Symbol::Rule(n.to_string(), Span::new(0, 0)) @@ -1047,6 +1034,13 @@ start -> () : "a" {$;;;; }; fn test_grmtools_section_values() { use super::*; use crate::header::Value; + let valid_test_keys = HashSet::from_iter([ + "test.string", + "test.vec", + "test.num", + "test.Negative", + "test.Flag", + ]); let src = r#" %grmtools { yacckind: Grmtools, @@ -1062,7 +1056,7 @@ start -> () : "a" {$;;;; }; %% start -> () : "a" { () }; "#; - let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); for (key, (expected_span, expected_value)) in [ ( "test.Flag".to_string(), @@ -1109,12 +1103,12 @@ start -> () : "a" { () }; let value = ast_validity.header_value_get(&key); assert_eq!(value, Some((expected_span, &expected_value))); } - let crate_prefixes = HashSet::from_iter(["test".to_string()]); + eprintln!("umm {src}"); assert_eq!( ast_validity - .iter_unused_header_values(&crate_prefixes) + .unrecognized_keys_for_prefix("test", &valid_test_keys) .collect::>(), - vec![("test.unused".to_string(), src.find_span("test.unused"))] + vec![("test.unused", src.find_span("test.unused"))] ); assert_eq!( ast_validity.header_value_get("cfgrammar.yacckind"), @@ -1123,10 +1117,9 @@ start -> () : "a" { () }; &Value::Namespaced("Grmtools".to_string(), src.find_span("Grmtools")) )) ); - assert!( ast_validity - .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) + .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) .next() .is_none() ); @@ -1141,7 +1134,7 @@ start -> () : "a" { () }; assert!( ast_validity - .iter_unused_header_values(&HashSet::from_iter(["lrpar".to_string()])) + .unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS) .next() .is_none() ); @@ -1159,7 +1152,7 @@ start -> () : "a" { () }; %% start: "a" { () }; "#; - let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( ast_validity.header_value_get("cfgrammar.yacckind"), Some(( @@ -1172,7 +1165,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) + .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) .next() .is_none() ); @@ -1190,7 +1183,7 @@ start: "a" { () }; %% start: "a" { () }; "#; - let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( ast_validity.header_value_get("cfgrammar.yacckind"), Some(( @@ -1203,7 +1196,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) + .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) .next() .is_none() ); @@ -1221,7 +1214,7 @@ start: "a" { () }; %% start: "a" { () }; "#; - let mut ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); + let ast_validity = ASTWithValidityInfo::from_str(src).unwrap(); assert_eq!( ast_validity.header_value_get("cfgrammar.yacckind"), Some(( @@ -1234,7 +1227,7 @@ start: "a" { () }; ); assert!( ast_validity - .iter_unused_header_values(&HashSet::from_iter(["cfgrammar".to_string()])) + .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) .next() .is_none() ); diff --git a/cfgrammar/src/lib/yacc/parser.rs b/cfgrammar/src/lib/yacc/parser.rs index 62371126c..7af9885ba 100644 --- a/cfgrammar/src/lib/yacc/parser.rs +++ b/cfgrammar/src/lib/yacc/parser.rs @@ -16,7 +16,7 @@ use wincode::{SchemaRead, SchemaWrite}; use crate::{ Span, Spanned, - header::{CRATE_KEY_MAP, GrmtoolsSectionParser, Header, HeaderErrorKind}, + header::{GrmtoolsSectionParser, Header, HeaderErrorKind, KEY_CRATE_MAP}, }; pub type YaccGrammarResult = Result>; @@ -380,7 +380,7 @@ impl YaccParser<'_> { // to mark keys used. It would be less error prone if we did this at the point where keys // are used. However at some points where we do lookups, there are shared references to the // header making it difficult to get mutable access. - for (key_name, crate_name) in CRATE_KEY_MAP.iter() { + for (key_name, crate_name) in KEY_CRATE_MAP.iter() { if ["cfgrammar", "lrpar"].contains(crate_name) { header.mark_used(&format!("{crate_name}.{key_name}")); } diff --git a/lrpar/cttests/src/grmtools_section.test b/lrpar/cttests/src/grmtools_section.test index e28ef37c1..69f3d0522 100644 --- a/lrpar/cttests/src/grmtools_section.test +++ b/lrpar/cttests/src/grmtools_section.test @@ -20,7 +20,7 @@ grammar: | : valbind { let ((key, key_loc), val) = $1; let key = if !RE_CRATE_DOT.is_match(&key) { - if let Some(crate_name) = CRATE_KEY_MAP.get(key.as_str()) { + if let Some(crate_name) = KEY_CRATE_MAP.get(key.as_str()) { format!("{crate_name}.{key}") } else { key @@ -48,7 +48,7 @@ grammar: | | val_seq ',' valbind { let ((key, key_loc), val) = $3; let key = if !RE_CRATE_DOT.is_match(&key) { - if let Some(crate_name) = CRATE_KEY_MAP.get(key.as_str()) { + if let Some(crate_name) = KEY_CRATE_MAP.get(key.as_str()) { format!("{crate_name}.{key}") } else { key @@ -164,7 +164,7 @@ grammar: | HeaderErrorKind, HeaderValue, Value, - CRATE_KEY_MAP, + KEY_CRATE_MAP, RE_CRATE_DOT, }, markmap::Entry, diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index f2ae9e257..270388ff5 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -14,7 +14,7 @@ use crate::{ use cfgrammar::{ RIdx, Span, Symbol, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue}, + header::{CFGRAMMAR_KEYS, GrmtoolsSectionParser, Header, HeaderError, HeaderValue, LRPAR_KEYS}, yacc::{ YaccGrammar, YaccGrammarError, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo, }, @@ -459,16 +459,22 @@ where /// Checks all the keys staring with `crate_prefixes`. If any of them are `unused`, return an error. /// If `crate_prefixes` contains the empty string, returns an arror if any key is unused. - fn check_unused_header_values( - &self, - crate_prefixes: &HashSet, - ) -> Result<(), ParserBuildEnvError> { - let unused_keys = self - .ast_with_validity_info() - .iter_unused_header_values(crate_prefixes) - .collect::>(); - if !unused_keys.is_empty() { - return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(unused_keys)); + fn check_unused_grmtools_header_values(&self) -> Result<(), ParserBuildEnvError> { + let mut unrecognized_keys = Vec::new(); + unrecognized_keys.extend( + self.ast_with_validity_info() + .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) + .map(|(s, span)| (s.to_string(), span)), + ); + unrecognized_keys.extend( + self.ast_with_validity_info() + .unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS) + .map(|(s, span)| (s.to_string(), span)), + ); + if !unrecognized_keys.is_empty() { + return Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys( + unrecognized_keys, + )); } Ok(()) } @@ -477,7 +483,7 @@ where &self, timestamp: &str, ) -> Result, ParserBuildEnvError> { - self.check_unused_header_values(&self.crates_to_check)?; + self.check_unused_grmtools_header_values()?; let grm = YaccGrammar::::new_from_ast_with_validity_info( &self.ast_with_validity_info, )?; @@ -1295,45 +1301,15 @@ pub(crate) fn make_generics(parse_generics: Option<&str>) -> Result () : "A" { () }; - "#; - let src_env = ParserSrcEnv::::new(src, None); - let mut build_env = src_env - .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) - .unwrap(); - build_env.register_header_value_prefix(crate_prefix); - match build_env.code_generator("timestamp") { - Err(ParserBuildEnvError::GrmtoolsSectionUnusedKeys(keys)) => { - assert_eq!( - &keys, - &[("test.foo".to_string(), src.find_span("test.foo"))] - ) - } - Err(e) => panic!("Unexpected error result: {:?}", e), - _ => panic!("Unexpected Ok return value"), - } - } - } - + use crate::test_utils::{FindSpan as _, TestLexerTypes}; + use cfgrammar::header::HeaderErrorKind; #[test] fn test_unused_header_entry() { let src = r#" %grmtools{ yacckind: Grmtools, - testfoo: "values which do not specify a crate origin should show up as unused", + testfoo: "non-grmtools values which do not specify a crate origin should produce errors", } %% start -> () : "A" { () }; diff --git a/nimbleparse/src/main.rs b/nimbleparse/src/main.rs index b98ab4383..4e6a5765f 100644 --- a/nimbleparse/src/main.rs +++ b/nimbleparse/src/main.rs @@ -1,6 +1,8 @@ use cfgrammar::{ RIdx, Span, TIdx, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderValue, Value}, + header::{ + CFGRAMMAR_KEYS, GrmtoolsSectionParser, Header, HeaderError, HeaderValue, LRPAR_KEYS, Value, + }, yacc::{YaccGrammar, YaccKind, YaccOriginalActionKind, ast::ASTWithValidityInfo}, }; use getopts::Options; @@ -14,7 +16,6 @@ use lrtable::{Minimiser, StateTable, from_yacc}; use num_traits::ToPrimitive as _; use num_traits::{AsPrimitive, PrimInt, Unsigned}; use std::{ - collections::HashSet, env, error::Error, fmt, @@ -244,15 +245,10 @@ fn main() { }); let yacc_kind = yk_arg.unwrap_or(yk_header_val.unwrap_or(YaccKind::Grmtools)); let ast_validation = ASTWithValidityInfo::new(yacc_kind, &yacc_src); - // Note we don't expect to find any used lrlex keys, we want to produce an error if unused ones are found. - let crate_prefixes = HashSet::from_iter([ - "cfgrammar".to_string(), - "lrpar".to_string(), - "lrlex".to_string(), - ]); - let unused_keys = ast_validation - .iter_unused_header_values(&crate_prefixes) - .collect::>(); + let mut unused_keys = Vec::new(); + unused_keys.extend(ast_validation.unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS)); + unused_keys.extend(ast_validation.unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS)); + if !unused_keys.is_empty() { eprintln!( "{ERROR}{}", From 534631b6e93b31bb6dbf2a5f2727f21b8d5dd43c Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 1 Oct 2026 19:48:45 -0700 Subject: [PATCH 31/38] Avoid using helper function for now --- cfgrammar/src/lib/yacc/ast.rs | 34 +++++++++++++--------------------- lrpar/src/lib/codegen.rs | 6 ++++-- nimbleparse/src/main.rs | 13 +++++++++++-- 3 files changed, 28 insertions(+), 25 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 3e62030d6..974f38b08 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -135,15 +135,6 @@ impl ASTWithValidityInfo { }) } - pub fn unrecognized_keys_for_prefix( - &self, - prefix: &str, - valid_keys: &HashSet<&str>, - ) -> impl Iterator { - self.iter_prefix_keys(prefix) - .filter(|(key, _)| !valid_keys.contains(*key)) - } - #[doc(hidden)] pub fn header(&self) -> &Header { &self.grmtools_section @@ -1034,7 +1025,7 @@ start -> () : "a" {$;;;; }; fn test_grmtools_section_values() { use super::*; use crate::header::Value; - let valid_test_keys = HashSet::from_iter([ + let valid_test_keys: HashSet<&str> = HashSet::from_iter([ "test.string", "test.vec", "test.num", @@ -1106,7 +1097,8 @@ start -> () : "a" { () }; eprintln!("umm {src}"); assert_eq!( ast_validity - .unrecognized_keys_for_prefix("test", &valid_test_keys) + .iter_prefix_keys("test") + .filter(|(key, _)| !valid_test_keys.contains(*key)) .collect::>(), vec![("test.unused", src.find_span("test.unused"))] ); @@ -1119,8 +1111,8 @@ start -> () : "a" { () }; ); assert!( ast_validity - .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) - .next() + .iter_prefix_keys("cfgrammar") + .find(|(key, _)| !CFGRAMMAR_KEYS.contains(key)) .is_none() ); @@ -1134,8 +1126,8 @@ start -> () : "a" { () }; assert!( ast_validity - .unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS) - .next() + .iter_prefix_keys("lrpar") + .find(|(key, _)| !LRPAR_KEYS.contains(key)) .is_none() ); } @@ -1165,8 +1157,8 @@ start: "a" { () }; ); assert!( ast_validity - .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) - .next() + .iter_prefix_keys("cfgrammar") + .find(|(key, _)| !CFGRAMMAR_KEYS.contains(key)) .is_none() ); } @@ -1196,8 +1188,8 @@ start: "a" { () }; ); assert!( ast_validity - .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) - .next() + .iter_prefix_keys("cfgrammar") + .find(|(key, _)| !CFGRAMMAR_KEYS.contains(key)) .is_none() ); } @@ -1227,8 +1219,8 @@ start: "a" { () }; ); assert!( ast_validity - .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) - .next() + .iter_prefix_keys("cfgrammar") + .find(|(key, _)| !CFGRAMMAR_KEYS.contains(key)) .is_none() ); } diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 270388ff5..54c04350a 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -463,12 +463,14 @@ where let mut unrecognized_keys = Vec::new(); unrecognized_keys.extend( self.ast_with_validity_info() - .unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS) + .iter_prefix_keys("cfgrammar") + .filter(|(key, _)| !CFGRAMMAR_KEYS.contains(key)) .map(|(s, span)| (s.to_string(), span)), ); unrecognized_keys.extend( self.ast_with_validity_info() - .unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS) + .iter_prefix_keys("lrpar") + .filter(|(key, _)| !LRPAR_KEYS.contains(key)) .map(|(s, span)| (s.to_string(), span)), ); if !unrecognized_keys.is_empty() { diff --git a/nimbleparse/src/main.rs b/nimbleparse/src/main.rs index 4e6a5765f..0295c501d 100644 --- a/nimbleparse/src/main.rs +++ b/nimbleparse/src/main.rs @@ -246,8 +246,17 @@ fn main() { let yacc_kind = yk_arg.unwrap_or(yk_header_val.unwrap_or(YaccKind::Grmtools)); let ast_validation = ASTWithValidityInfo::new(yacc_kind, &yacc_src); let mut unused_keys = Vec::new(); - unused_keys.extend(ast_validation.unrecognized_keys_for_prefix("cfgrammar", &CFGRAMMAR_KEYS)); - unused_keys.extend(ast_validation.unrecognized_keys_for_prefix("lrpar", &LRPAR_KEYS)); + // FIXME stop using hidden API + unused_keys.extend( + ast_validation + .iter_prefix_keys("cfgrammar") + .filter(|(key, _)| !CFGRAMMAR_KEYS.contains(key)), + ); + unused_keys.extend( + ast_validation + .iter_prefix_keys("lrpar") + .filter(|(key, _)| !LRPAR_KEYS.contains(key)), + ); if !unused_keys.is_empty() { eprintln!( From 19289b52ea5894dc674b8c0eca903b08d57b46fd Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 1 Oct 2026 20:50:19 -0700 Subject: [PATCH 32/38] First stab at 'exhaustive' checking --- cfgrammar/src/lib/yacc/ast.rs | 21 ++++++++++++++++++--- 1 file changed, 18 insertions(+), 3 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 974f38b08..679ca312a 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -14,7 +14,10 @@ use super::{ use crate::{ Span, - header::{GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, Value}, + header::{ + GrmtoolsSectionParser, Header, HeaderError, HeaderErrorKind, HeaderValue, RE_CRATE_DOT, + Value, + }, yacc::YaccOriginalActionKind, }; @@ -126,7 +129,6 @@ impl ASTWithValidityInfo { .into_iter() .filter_map(move |(key, val)| { let HeaderValue(span, _) = val; - eprintln!("prefix: {prefix:?} key: {key:?}"); if key.starts_with(&prefix) { Some((key.as_str(), *span)) } else { @@ -135,6 +137,16 @@ impl ASTWithValidityInfo { }) } + pub fn iter_prefixes(&self) -> impl Iterator { + let mut prefixes = HashSet::new(); + for (key, _) in &self.grmtools_section { + if let Some(prefix) = RE_CRATE_DOT.find(key) { + prefixes.insert(prefix.as_str().strip_suffix('.').expect("Regex ends in dot")); + } + } + prefixes.into_iter() + } + #[doc(hidden)] pub fn header(&self) -> &Header { &self.grmtools_section @@ -1094,7 +1106,6 @@ start -> () : "a" { () }; let value = ast_validity.header_value_get(&key); assert_eq!(value, Some((expected_span, &expected_value))); } - eprintln!("umm {src}"); assert_eq!( ast_validity .iter_prefix_keys("test") @@ -1130,6 +1141,10 @@ start -> () : "a" { () }; .find(|(key, _)| !LRPAR_KEYS.contains(key)) .is_none() ); + + let found_prefixes: HashSet<&str> = HashSet::from_iter(ast_validity.iter_prefixes()); + let expected_prefixes: HashSet<&str> = HashSet::from_iter(["test", "cfgrammar", "lrpar"]); + assert_eq!(found_prefixes, expected_prefixes); } #[test] From a8065ff643ad6ce9fcb9d982523c0c997f7c6216 Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 1 Oct 2026 21:21:46 -0700 Subject: [PATCH 33/38] add registration error, and method to retreive registered prefixes --- cfgrammar/src/lib/yacc/ast.rs | 7 ++++- lrpar/src/lib/codegen.rs | 51 ++++++++++++++++++++++++----------- 2 files changed, 41 insertions(+), 17 deletions(-) diff --git a/cfgrammar/src/lib/yacc/ast.rs b/cfgrammar/src/lib/yacc/ast.rs index 679ca312a..677a79047 100644 --- a/cfgrammar/src/lib/yacc/ast.rs +++ b/cfgrammar/src/lib/yacc/ast.rs @@ -141,7 +141,12 @@ impl ASTWithValidityInfo { let mut prefixes = HashSet::new(); for (key, _) in &self.grmtools_section { if let Some(prefix) = RE_CRATE_DOT.find(key) { - prefixes.insert(prefix.as_str().strip_suffix('.').expect("Regex ends in dot")); + prefixes.insert( + prefix + .as_str() + .strip_suffix('.') + .expect("Regex ends in dot"), + ); } } prefixes.into_iter() diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 54c04350a..1371bdecd 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -51,6 +51,7 @@ where StateTableError(StateTableError), YaccGrammarErrors(Vec), GrmtoolsSectionUnusedKeys(Vec<(String, Span)>), + DuplicatePrefixRegistrationError(String), } #[derive(Debug)] @@ -138,6 +139,9 @@ where let keys = keys.iter().cloned().map(|(s, _)| s).collect::>(); format!("Unused keys in %grmtools section: {}", keys.join(", ")) } + Self::DuplicatePrefixRegistrationError(prefix) => { + format!("Registered duplicate prefix: \"{prefix}\"") + } }) } } @@ -192,7 +196,7 @@ where phantom_storaget: PhantomData, mod_name: String, grammar_path: Option, - crates_to_check: HashSet, + registered_header_prefixes: HashSet, } pub(crate) struct ParserCodegen @@ -382,22 +386,19 @@ where let recoverer = self.resolve_recoverer(&header)?; let serialisation_format = self.resolve_serialisation_format(&header)?; let mod_name = self.resolve_mod_name(&args)?; - let grammar_path = self.grammar_path_cache_entry; - let crates_to_check = vec![ - "cfgrammar".to_string(), - "lrpar".to_string(), - "lrlex".to_string(), - ]; - Ok(ParserBuildEnv { + let registered_header_prefixes = + HashSet::from_iter(["cfgrammar".to_string(), "lrpar".to_string()]); + let build_env = ParserBuildEnv { ast_with_validity_info, cache_args: args, - crates_to_check: HashSet::from_iter(crates_to_check), + registered_header_prefixes, recoverer, serialisation_format, mod_name, - grammar_path, + grammar_path: self.grammar_path_cache_entry, phantom_storaget: PhantomData, - }) + }; + Ok(build_env) } } @@ -450,11 +451,29 @@ where self.ast_with_validity_info.yacc_kind() } - /// Causes the `code_generator()` function to check for unused entries in the grmtools section - /// starting for entries starting with `crate_prefix`. - #[allow(unused)] - pub(crate) fn register_header_value_prefix(&mut self, crate_prefix: &str) { - self.crates_to_check.insert(crate_prefix.to_string()); + /// Adds the `crate_prefix` to the list of registered prefixes. + /// The intent is that the user can check the registered keys against `ast_with_validity_info.iter_prefixes()` + /// Such that `assert_eq!(ast_with_validity_info.iter_prefixes(), self.registered_header_prefixes())`. + #[expect(unused)] + pub(crate) fn register_header_prefix( + &mut self, + crate_prefix: &str, + ) -> Result<(), ParserBuildEnvError> { + if !self + .registered_header_prefixes + .insert(crate_prefix.to_string()) + { + Err(ParserBuildEnvError::DuplicatePrefixRegistrationError( + crate_prefix.to_string(), + )) + } else { + Ok(()) + } + } + + #[expect(unused)] + pub(crate) fn registered_header_prefixes(&self) -> impl Iterator { + self.registered_header_prefixes.iter().map(|s| s.as_str()) } /// Checks all the keys staring with `crate_prefixes`. If any of them are `unused`, return an error. From 16d66fd955e3131890b49c2fd826009918631542 Mon Sep 17 00:00:00 2001 From: matt rice Date: Thu, 1 Oct 2026 21:32:37 -0700 Subject: [PATCH 34/38] Add test for prefix registration --- lrpar/src/lib/codegen.rs | 27 +++++++++++++++++++++++++-- 1 file changed, 25 insertions(+), 2 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 1371bdecd..1b907f665 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -454,7 +454,7 @@ where /// Adds the `crate_prefix` to the list of registered prefixes. /// The intent is that the user can check the registered keys against `ast_with_validity_info.iter_prefixes()` /// Such that `assert_eq!(ast_with_validity_info.iter_prefixes(), self.registered_header_prefixes())`. - #[expect(unused)] + #[allow(unused)] pub(crate) fn register_header_prefix( &mut self, crate_prefix: &str, @@ -471,7 +471,7 @@ where } } - #[expect(unused)] + #[allow(unused)] pub(crate) fn registered_header_prefixes(&self) -> impl Iterator { self.registered_header_prefixes.iter().map(|s| s.as_str()) } @@ -1379,4 +1379,27 @@ mod test { _ => panic!("Unexpected error result"), } } + + #[test] + fn test_unregistered_grmtools_prefix() { + let src = r#" + %grmtools{ + yacckind: Grmtools, + unregistered.key: "unknown prefix should be unregistered", + registered.key: "should be registered", + } + %% + start -> () : "A" { () }; + "#; + let src_env = ParserSrcEnv::::new(src, None); + let mut build_env = src_env + .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) + .unwrap(); + build_env.register_header_prefix("registered").unwrap(); + let found_prefixes = build_env.ast_with_validity_info().iter_prefixes().collect::>(); + let registered_prefixes = build_env.registered_header_prefixes().collect::>(); + let expected_unregistered: HashSet<&str> = HashSet::from_iter(["unregistered"]); + let unregistered_prefixes: HashSet<&str> = found_prefixes.difference(®istered_prefixes).copied().collect::>(); + assert_eq!(expected_unregistered, unregistered_prefixes); + } } From 5c9ac9825272c8f123b67567e05ef5a416d70d96 Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 2 Oct 2026 03:27:50 -0700 Subject: [PATCH 35/38] Fix docs example --- lrpar/src/lib/codegen.rs | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 1b907f665..49c19d2ae 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -453,7 +453,14 @@ where /// Adds the `crate_prefix` to the list of registered prefixes. /// The intent is that the user can check the registered keys against `ast_with_validity_info.iter_prefixes()` - /// Such that `assert_eq!(ast_with_validity_info.iter_prefixes(), self.registered_header_prefixes())`. + /// So that all the values in `ast_with_validity_info.iter_prefixes()` are also in `self.registered_header_prefixes()` + /// for example + /// + /// ``` + /// let prefixes_set = ast_with_validity_info.iter_prefixes().collect::>(); + /// let registered_set = self.registered_header_prefixes().collect>(); + /// assert!(prefixes_set.difference(registered_set).next().is_none()) + /// ``` #[allow(unused)] pub(crate) fn register_header_prefix( &mut self, From 8f90a95bc1af93138d83da3145a52042c1da47db Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 2 Oct 2026 03:27:57 -0700 Subject: [PATCH 36/38] rustfmt --- lrpar/src/lib/codegen.rs | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 49c19d2ae..828dd7b76 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -1387,7 +1387,7 @@ mod test { } } - #[test] + #[test] fn test_unregistered_grmtools_prefix() { let src = r#" %grmtools{ @@ -1403,10 +1403,18 @@ mod test { .build_env(ParserBuildEnvArgs::new().mod_name(Some("test_module"))) .unwrap(); build_env.register_header_prefix("registered").unwrap(); - let found_prefixes = build_env.ast_with_validity_info().iter_prefixes().collect::>(); - let registered_prefixes = build_env.registered_header_prefixes().collect::>(); + let found_prefixes = build_env + .ast_with_validity_info() + .iter_prefixes() + .collect::>(); + let registered_prefixes = build_env + .registered_header_prefixes() + .collect::>(); let expected_unregistered: HashSet<&str> = HashSet::from_iter(["unregistered"]); - let unregistered_prefixes: HashSet<&str> = found_prefixes.difference(®istered_prefixes).copied().collect::>(); + let unregistered_prefixes: HashSet<&str> = found_prefixes + .difference(®istered_prefixes) + .copied() + .collect::>(); assert_eq!(expected_unregistered, unregistered_prefixes); } } From 642f94ee652dda599a11dab36ebc09db570a92c4 Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 2 Oct 2026 03:42:18 -0700 Subject: [PATCH 37/38] small example improvement --- lrpar/src/lib/codegen.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 828dd7b76..279e23a00 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -457,8 +457,8 @@ where /// for example /// /// ``` - /// let prefixes_set = ast_with_validity_info.iter_prefixes().collect::>(); - /// let registered_set = self.registered_header_prefixes().collect>(); + /// let prefixes_set = build_env.ast_with_validity_info().iter_prefixes().collect::>(); + /// let registered_set = build_env.registered_header_prefixes().collect>(); /// assert!(prefixes_set.difference(registered_set).next().is_none()) /// ``` #[allow(unused)] From d8d00bf0046a73973c9477577e0541dfd3674b66 Mon Sep 17 00:00:00 2001 From: matt rice Date: Fri, 2 Oct 2026 04:01:03 -0700 Subject: [PATCH 38/38] ignore docstring example --- lrpar/src/lib/codegen.rs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/lrpar/src/lib/codegen.rs b/lrpar/src/lib/codegen.rs index 279e23a00..fd97fd187 100644 --- a/lrpar/src/lib/codegen.rs +++ b/lrpar/src/lib/codegen.rs @@ -456,9 +456,9 @@ where /// So that all the values in `ast_with_validity_info.iter_prefixes()` are also in `self.registered_header_prefixes()` /// for example /// - /// ``` - /// let prefixes_set = build_env.ast_with_validity_info().iter_prefixes().collect::>(); - /// let registered_set = build_env.registered_header_prefixes().collect>(); + /// ```ignore + /// let prefixes_set = HashSet::from_iter(build_env.ast_with_validity_info().iter_prefixes()); + /// let registered_set = HashSet::from_iter(build_env.registered_header_prefixes()); /// assert!(prefixes_set.difference(registered_set).next().is_none()) /// ``` #[allow(unused)]