From 4b81b3f0a5f48652d66d0bc8bf5689c340bb76e2 Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Sun, 26 Sep 2021 15:03:59 -0700 Subject: [PATCH 1/7] Changed to use struct SpeechRulesWithContext but having Rust issues with lifetimes and mutable/immutable conflicts. Saving so I can try something else. --- src/interface.rs | 17 +- src/speech.rs | 503 +++++++++++++++++++++++------------------ src/tts.rs | 6 +- src/xpath_functions.rs | 75 +++--- 4 files changed, 334 insertions(+), 267 deletions(-) diff --git a/src/interface.rs b/src/interface.rs index 423b992dc..a1c90fa36 100755 --- a/src/interface.rs +++ b/src/interface.rs @@ -47,7 +47,7 @@ pub fn speak_mathml(mathml_str: &str) -> String { let package = package.unwrap(); cleanup_mathml(&package); let mathml = get_element(&package); - return crate::speech::speak_mathml(&mathml); + return crate::speech::speak_mathml(mathml); } /// Given MathML, a Unicode braille string is return. @@ -64,7 +64,7 @@ pub fn braille_mathml(mathml_str: &str) -> String { let package = package.unwrap(); cleanup_mathml(&package); let mathml = get_element(&package); - return crate::speech::braille_mathml(&mathml); + return crate::speech::braille_mathml(mathml); } thread_local!{ @@ -108,7 +108,7 @@ pub fn GetSpokenText(_py: Python) -> PyResult { return MATHML_INSTANCE.with(|package_instance| { let package_instance = package_instance.borrow(); let mathml = get_element(&*package_instance); - let speech = crate::speech::speak_mathml(&mathml); + let speech = crate::speech::speak_mathml(mathml); eprintln!("Time taken: {}ms", instant.elapsed().as_millis()); return Ok( speech ); }); @@ -196,8 +196,15 @@ fn set_speech_tags(pref_manager: &mut PreferenceManager, speech_tags: String ) - /// Get the braille associated with the MathML that was set by [`SetMathML`]. /// The braille returned depends upon the preference for braille output. pub fn GetBraille(_py: Python) -> PyResult { - // FIX: not yet implemented (basically what the braille says) - return Ok("⠠⠃⠗⠁⠊⠇⠇⠑ ⠛⠑⠝⠑⠗⠁⠞⠊⠕⠝ ⠝⠕⠞ ⠽⠑⠞ ⠊⠍⠏⠇⠑⠍⠑⠝⠞".to_string()); + use std::time::{Instant}; + let instant = Instant::now(); + return MATHML_INSTANCE.with(|package_instance| { + let package_instance = package_instance.borrow(); + let mathml = get_element(&*package_instance); + let braille = crate::speech::braille_mathml(mathml); + eprintln!("Time taken: {}ms", instant.elapsed().as_millis()); + return Ok( braille ); + }); } #[pyfunction] diff --git a/src/speech.rs b/src/speech.rs index 3e16f3834..0aa6c476c 100755 --- a/src/speech.rs +++ b/src/speech.rs @@ -9,8 +9,9 @@ use std::{collections::HashMap, process::exit}; use std::cell::RefCell; -use std::rc::Rc; use sxd_document::dom::Element; +use sxd_document::QName; +use sxd_xpath::context::Evaluation; use sxd_xpath::{Context, Factory, Value, XPath, nodeset}; use sxd_xpath::nodeset::Node; use std::fmt; @@ -35,15 +36,12 @@ use phf::phf_map; /// /// A string is returned in call cases. /// If there is an error, the speech string will indicate an error. -pub fn speak_mathml(mathml: &Element) -> String { +pub fn speak_mathml(mathml: Element) -> String { SPEECH_RULES.with(|rules| { let mut rules = rules.borrow_mut(); rules.update(true); - CONTEXT_STACK.with(|cs| { - let mut cs = cs.borrow_mut(); - cs.init(&rules.pref_manager) - }); - match rules.match_pattern(mathml) { + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + match rules_with_context.match_pattern(mathml) { Ok(speech_string) => { return rules.pref_manager.get_tts() .merge_pauses(remove_optional_indicators( @@ -60,15 +58,12 @@ pub fn speak_mathml(mathml: &Element) -> String { }) } -pub fn braille_mathml(mathml: &Element) -> String { +pub fn braille_mathml(mathml: Element) -> String { BRAILLE_RULES.with(|rules| { let mut rules = rules.borrow_mut(); rules.update(false); - CONTEXT_STACK.with(|cs| { - let mut cs = cs.borrow_mut(); - cs.init(&rules.pref_manager) - }); - match rules.match_pattern(mathml) { + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + match rules_with_context.match_pattern(mathml) { // FIX: need to set name of speech rules so test Nemeth/UEB clean for Ok(speech_string) => { return nemeth_cleanup(speech_string.replace(" ", "")); @@ -85,7 +80,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Typeface: S: sans-serif, B: bold, T: script/blackboard, I: italic, R: Roman // Language: E: English, D: German, G: Greek, V: Greek variants, H: Hebrew, U: Russian // Indicators: C: capital, N: number, P: punctuation, M: multipurpose - // Others: W -- whitespace that should be kept (e.g, in a numeral) + // Others: + // W -- whitespace that should be kept (e.g, in a numeral) + // 𝑁 -- hack for special case of a lone decimal pt -- not considered a number but follows rules mostly // SRE doesn't have H: Hebrew or U: Russian, so not encoded (yet) // Note: some "positive" patterns find cases to keep the char and transform them to the lower case version static INDICATOR_REPLACEMENTS: phf::Map<&str, &str> = phf_map! { @@ -106,6 +103,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { "m" => "⠐", "N" => "", "n" => "⠼", + "𝑁" => "", "W" => "⠀" }; @@ -128,15 +126,15 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Multipurpose indicator insertion // 177.2 -- add after a letter and before a digit (or decimal pt) -- these will start with N static ref MULTI_177_2: Regex = - Regex::new(r"([⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵])N").unwrap(); + Regex::new(r"([⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵])[N𝑁]").unwrap(); // keep between numeric subscript and digit ('M' added by subscript rule) static ref MULTI_177_3: Regex = - Regex::new(r"(N.)M(N.)").unwrap(); + Regex::new(r"([N𝑁].)M([N𝑁].)").unwrap(); // add after decimal pt for non-digits except for comma and punctuation static ref MULTI_177_5: Regex = - Regex::new(r"N⠨([^N⠠P])").unwrap(); + Regex::new(r"([N𝑁]⠨)([^N𝑁⠠P])").unwrap(); // Pattern for rule II.9a (add numeric indicator at start of line or after a space) and 9a (add after typeface) @@ -171,10 +169,12 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { static ref REMOVE_PUNCT_IND: Regex = Regex::new(r"(^|⠀|\w)P(.)").unwrap(); - static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNnW])").unwrap(); + static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNn𝑁W])").unwrap(); static ref REMOVE_LEVEL_IND_BEFORE_BASELINE: Regex = Regex::new(r"(?:[⠘⠰]+⠐)").unwrap(); - static ref REMOVE_LEVEL_IND_BEFORE_SPACE: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)(⠀|$)").unwrap(); + + // Before 79b (punctuation) + static ref REMOVE_LEVEL_IND_BEFORE_SPACE_OR_PUNCT: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)([P⠠⠀]|$)").unwrap(); static ref COLLAPSE_SPACES: Regex = Regex::new(r"⠀⠀+").unwrap(); } @@ -193,9 +193,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { println!("spaces: \"{}\"", result); // Multipurpose indicator - let result = MULTI_177_2.replace_all(&result, "${1}mN"); + let result = MULTI_177_2.replace_all(&result, "${1}m${2}"); let result = MULTI_177_3.replace_all(&result, "${1}m$2"); - let result = MULTI_177_5.replace_all(&result, "N⠨m$1"); + let result = MULTI_177_5.replace_all(&result, "${1}m$2"); println!("MULTI: \"{}\"", result); let result = NUM_IND_9A.replace_all(&result, "$start$minus${face}n"); @@ -210,9 +210,15 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { let result = NUM_IND_AFTER_PUNCT.replace_all(&result, "$punct${minus}n"); println!("A PUNCT: \"{}\"", &result); + // strip level indicators + // checks for punctuation char, so needs to before punctuation is stripped. + let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_SPACE_OR_PUNCT, "$1"); + let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_BASELINE, "⠐"); + let result = REMOVE_PUNCT_IND.replace_all(&result, "$1$2"); println!("Punct38: \"{}\"", &result); + let result = REPLACE_INDICATORS.replace_all(&result, |cap: &Captures| { match INDICATOR_REPLACEMENTS.get(&cap[0]) { None => panic!("REPLACE_INDICATORS and INDICATOR_REPLACEMENTS are not in sync"), @@ -220,10 +226,6 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { } }); - // strip level indicators before a space - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_SPACE, "$1"); - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_BASELINE, "⠐"); - let result = COLLAPSE_SPACES.replace_all(&result, "⠀"); return result.to_string(); @@ -481,8 +483,8 @@ impl InsertChildren { // The solution adopted is to find out the number of nodes and build up MyXPaths with each node selected (e.g, "*" => "*[3]") // and put those nodes into a flat ReplacementArray and then do a standard replace on that. // This is slower than the alternatives, but reuses a bunch of code and hence is less complicated. - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { - let result = self.xpath.evaluate(mathml) + fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + let result = self.xpath.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; match result { Value::Nodeset(nodes) => { @@ -505,11 +507,11 @@ impl InsertChildren { ); } let replacements = ReplacementArray{ replacements: expanded_result }; - return replacements.replace(rules, mathml); + return replacements.replace(rules_with_context, mathml); }, // FIX: should the options be errors??? - Value::String(t) => { return rules.replace_chars(&t, mathml); }, + Value::String(t) => { return rules_with_context.replace_chars(&t, mathml); }, Value::Number(num) => { return Ok( num.to_string() ); }, Value::Boolean(b) => { return Ok( b.to_string() ); }, // FIX: is this right??? } @@ -558,13 +560,13 @@ impl ReplacementArray { } /// Do all the replacements in `mathml` using `rules`. - pub fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + pub fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { // do the replacements // remove the empty strings (the later 'join' would add extraneous spaces) // collect the strings together into an array let mut replacement_strings = self.replacements.iter() - .map(|group| rules.replace(group, mathml)) + .map(|group| rules_with_context.replace(group, mathml)) .filter(|result| if let Ok(str) = result {!str.is_empty()} else {true}) .collect::>>()?; @@ -594,7 +596,7 @@ impl ReplacementArray { let after = if i+1 == replacement_strings.len() {""} else {&replacement_strings[i+1]}; replacement_strings[i] = replacement_strings[i].replace( PAUSE_AUTO_STR, - &rules.pref_manager.get_tts().compute_auto_pause(&rules.pref_manager, before, after)); + &rules_with_context.speech_rules.pref_manager.get_tts().compute_auto_pause(&rules_with_context.speech_rules.pref_manager, before, after)); } } @@ -817,10 +819,10 @@ impl MyXPath { } } - fn is_true(&self, mathml: &Element) -> Result { + fn is_true(&self, context: &Context, mathml: Element) -> Result { // return true if there is no condition or if the condition evaluates to true return Ok( - match self.evaluate(mathml)? { + match self.evaluate(context, mathml)? { Value::Boolean(b) => b, Value::Nodeset(nodes) => nodes.size() > 0, _ => false, @@ -828,84 +830,39 @@ impl MyXPath { ) } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { - let result = self.evaluate(mathml) + fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + let result = self.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; - let answer; - match result { - Value::Nodeset(nodes) => { - if nodes.size() == 0 { - bail!("During replacement, no matching element found"); - } - return rules.replace_nodes(nodes, mathml); - }, - // Value::String(t) => { return rules.replace_chars(&t, mathml); }, - Value::String(t) => { answer = t; }, - Value::Number(num) => { answer = num.to_string(); }, - Value::Boolean(b) => { answer = b.to_string(); }, // FIX: is this right??? + return unsafe { + match result { + Value::Nodeset(nodes) => { + if nodes.size() == 0 { + bail!("During replacement, no matching element found"); + } + rules_with_context.replace_nodes(nodes, mathml) + }, + Value::String(t) => Ok( t ), + Value::Number(num) => Ok( num.to_string() ), + Value::Boolean(b) => Ok( b.to_string() ), // FIX: is this right??? + } } - return Ok( answer ); } - fn evaluate<'a,'c>(&'a self, mathml: &'c Element) -> Result> { - return CONTEXT_STACK.with(|context_stack| { - let context_stack = context_stack.borrow(); - let context = context_stack.top(); - // println!("evaluate: {}", self); - let result = self.xpath.evaluate(&context, *mathml); - return match result { - Ok(val) => Ok( val ), - Err(e) => { - bail!( "{}\n\n", - e.to_string() // remove confusing parts of error message from xpath - .replace("OwnedPrefixedName { prefix: None, local_part:", "") - .replace(" }", "") ); - } - }; - }); - } -} - -// Used for speech rules with "variables: ..." -#[derive(Debug)] -struct VariableDefinition { - name: String, // name of variable - value: MyXPath, // value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes -} - -impl fmt::Display for VariableDefinition { - fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { - return write!(f, "{{name: {}, value: {}}}", self.name, self.value); - } -} - -#[derive(Debug)] -struct VariableDefinitions { - defs: Vec -} - -impl VariableDefinitions { - fn new() -> VariableDefinitions { - return VariableDefinitions{ defs: Vec::new() }; - } - - fn push(&mut self, var_def: VariableDefinition) { - self.defs.push(var_def); - } - - fn len(&self) -> usize { - return self.defs.len(); - } - - fn evaluate_to_yaml(&self, mathml: &Element) -> Result { - let mut new_prefs = HashMap::with_capacity(self.defs.len()); - for var_def in &self.defs { - new_prefs.insert( - var_def.name.clone(), - value_to_yaml(&var_def.value.evaluate(mathml)?) - .chain_err(|| format!("while evaluating variable '{}'", var_def.name))?); + fn evaluate<'a, 'd>(&'a self, context: &'d Context, mathml: Element<'d>) -> Result> { + // println!("evaluate: {}", self); + let result = self.xpath.evaluate(context, mathml); + return match result { + Ok(val) => { + // println!(" result: '{:?}'", val); + Ok( val ) + }, + Err(e) => { + bail!( "{}\n\n", + e.to_string() // remove confusing parts of error message from xpath + .replace("OwnedPrefixedName { prefix: None, local_part:", "") + .replace(" }", "") ); + } }; - return Ok( new_prefs ); } } @@ -920,11 +877,11 @@ struct SpeechPattern { tag_name: String, file_name: String, pattern: MyXPath, // the xpath expr to attempt to match - var_defs: VariableDefinitions, // any variable definitions [can be and probably is an empty vector most of the time] + var_defs: VariableDefinitions, // any variable definitions [can be and probably is an empty vector most of the time] replacements: ReplacementArray, // the replacements in case there is a match } -impl fmt::Display for SpeechPattern { +impl<'a> fmt::Display for SpeechPattern { fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { return write!(f, "{{name: {}, tag: {}, variables: {:?}, pattern: {}, replacement: {}}}", self.pattern_name, self.tag_name, self.var_defs, self.pattern, @@ -1006,7 +963,7 @@ impl SpeechPattern { format!("value for 'match' in rule ({}: {}):\n{}", tag_name, pattern_name, yaml_to_string(dict, 1)) })?, - var_defs: ContextStack::build(&dict["variables"]) + var_defs: VariableDefinitions::build(&dict["variables"]) .chain_err(|| { format!("value for 'variables' in rule ({}: {}):\n{}", tag_name, pattern_name, yaml_to_string(dict, 1)) @@ -1035,7 +992,7 @@ impl SpeechPattern { return Ok( () ); } - fn is_match(&self, mathml: &Element) -> Result { + fn is_match(&self, context: &Context, mathml: Element) -> Result { if self.tag_name != mathml.name().local_part() && self.tag_name != "unknown" { return Ok( false ); } @@ -1044,7 +1001,7 @@ impl SpeechPattern { // println!(" pattern_expr {:?}", self.pattern_expr); // print!("is_match: mathml is\n{}", crate::pretty_print::mml_to_string(mathml)); return Ok( - match self.pattern.evaluate(mathml)? { + match self.pattern.evaluate(context, mathml)? { Value::Boolean(b) => b, Value::Nodeset(nodes) => nodes.size() > 0, _ => false, @@ -1103,7 +1060,7 @@ impl TestArray { let else_part = TestOrReplacements::build(&test, "else", "else_test", false)?; let n_keys = if else_part.is_none() {2} else {3}; if test.as_hash().unwrap().len() > n_keys { - bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in an else clause of 'test'"); + bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in the 'then' clause of 'test'"); }; test_array.push( Test { condition, then_part, else_part } @@ -1112,7 +1069,7 @@ impl TestArray { // second case: should be else/else_test let else_part = TestOrReplacements::build(&test, "else", "else_test", true)?; if test.as_hash().unwrap().len() > 1 { - bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in an else clause of 'test'"); + bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found the 'else' clause of 'test'"); }; test_array.push( Test { condition: None, then_part: None, else_part } @@ -1132,13 +1089,13 @@ impl TestArray { return Ok( TestArray { tests: test_array } ); } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { for test in &self.tests { - if test.is_true(mathml)? { + if test.is_true(&rules_with_context.context_stack.base, mathml)? { assert!(test.then_part.is_some()); - return test.then_part.as_ref().unwrap().replace(rules, mathml); + return test.then_part.as_ref().unwrap().replace(rules_with_context, mathml); } else if let Some(else_part) = test.else_part.as_ref() { - return else_part.replace(rules, mathml); + return else_part.replace(rules_with_context, mathml); } } return Ok( "".to_string() ); @@ -1191,10 +1148,10 @@ impl TestOrReplacements { } } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { return match self { - TestOrReplacements::Replacements(r) => r.replace(rules, mathml), - TestOrReplacements::Test(t) => t.replace(rules, mathml), + TestOrReplacements::Replacements(r) => r.replace(rules_with_context, mathml), + TestOrReplacements::Test(t) => t.replace(rules_with_context, mathml), } } } @@ -1222,49 +1179,47 @@ impl fmt::Display for Test { } impl Test { - fn is_true(&self, mathml: &Element) -> Result { + fn is_true(&self, context: &Context, mathml: Element) -> Result { return match self.condition.as_ref() { None => Ok( false ), // trivially false -- want to do else part - Some(condition) => condition.is_true(mathml) + Some(condition) => condition.is_true(context, mathml) .chain_err(|| "Failure in conditional test"), } } } +// Used for speech rules with "variables: ..." +#[derive(Debug)] +struct VariableDefinition { + name: String, // name of variable + value: MyXPath, // xpath value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes +} -struct ContextStack<'c>{ - // FIX: this really should just clone the top of the stack and add on new vars. - // However, Context does not support 'clone' because the functions it stores are traits, not types - // Instead, we recreate the base each time and we store the vars to add in a stack (yuck!) - // Fortunately, adding new vars is rare - new_defs: Vec, - contexts: Vec>>, +impl fmt::Display for VariableDefinition { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + return write!(f, "{{name: {}, value: {}}}", self.name, self.value); + } } -impl<'c> ContextStack<'static>{ - fn init<'d>(&mut self, pref_manager: &'d PreferenceManager) { - let prefs = pref_manager.merge_prefs(); - let context = ContextStack::base_context(&prefs); - self.new_defs.push(prefs); - self.contexts.push(context); - } +// Used for speech rules with "variables: ..." +#[derive(Debug)] +struct VariableValue<'v> { + name: &'v str, // name of variable + value: Option>, // xpath value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes +} - fn base_context<'a>(var_defs: &'a PreferenceHashMap) -> Rc> { - let mut context = Context::new(); - context.set_namespace("m", "http://www.w3.org/1998/Math/MathML"); - crate::xpath_functions::add_builtin_functions(&mut context); - for (key, value) in var_defs { - context.set_variable(key.as_str(), yaml_to_value(value)); - // if let Some(str_value) = value.as_str() { - // if str_value != "Auto" { - // println!("Set {}='{}'", key.as_str(), str_value); - // } - // } +impl<'v> fmt::Display for VariableValue<'v> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + let value = match &self.value { + None => "unset".to_string(), + Some(val) => format!("{:?}", val) }; - return Rc::new( context ); - } + return write!(f, "{{name: {}, value: {}}}", self.name, value); + } +} - fn new_def(name_value_def: &Yaml) -> Result<(String, MyXPath)> { +impl VariableDefinition { + fn build(name_value_def: &Yaml) -> Result { match name_value_def.as_hash() { Some(map) => { if map.len() != 1 { @@ -1280,23 +1235,46 @@ impl<'c> ContextStack<'static>{ _ => bail!("definition value is not a string, boolean, or number. Found {}", yaml_to_string(value, 1) ) }; - return Ok( (name, MyXPath::build(value)? ) ); + return Ok( + VariableDefinition{ + name, + value: MyXPath::build(value)? + } + ); }, None => bail!("definition is not a key/value pair. Found {}", yaml_to_string(name_value_def, 1) ) } } +} + + +#[derive(Debug)] +struct VariableDefinitions { + defs: Vec +} + +#[derive(Debug)] +struct VariableValues<'v> { + defs: Vec> +} + +impl VariableDefinitions { + fn new(len: usize) -> VariableDefinitions { + return VariableDefinitions{ defs: Vec::with_capacity(len) }; + } fn build(defs: &Yaml) -> Result { if defs.is_badvalue() { - return Ok( VariableDefinitions::new() ); + return Ok( VariableDefinitions::new(0) ); }; if defs.is_array() { - let mut definitions = VariableDefinitions::new(); - for def in defs.as_vec().unwrap() { - let (name, value) = ContextStack::new_def(def) + let defs = defs.as_vec().unwrap(); + let mut definitions = VariableDefinitions::new(defs.len()); + for def in defs { + let variable_def = VariableDefinition::build(def) .chain_err(|| "definition of 'variables'")?; - definitions.push( VariableDefinition{ name, value} ); + definitions.push( variable_def); }; return Ok (definitions ); } @@ -1304,27 +1282,86 @@ impl<'c> ContextStack<'static>{ yaml_to_string(defs, 1) ); } - fn push(&mut self, new_prefs: &PreferenceHashMap) { - // evaluate the XPath's into - self.new_defs.push(new_prefs.clone()); // do first so they are included in context + fn push(&mut self, var_def: VariableDefinition) { + self.defs.push(var_def); + } + + fn len(&self) -> usize { + return self.defs.len(); + } +} + +struct ContextStack<'c> { + // Note: values are generated by calling value_of on an Evaluation -- that makes the two lifetimes the same + old_values: Vec>, // store old values so they can be set on pop + base: Context<'c> // initial context -- contains all the function defs and pref variables +} + +impl<'c> fmt::Display for ContextStack<'c> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + return writeln!(f, " {} old_values", self.old_values.len()); + } +} + +impl<'c> ContextStack<'c> { + fn new<'a>(pref_manager: &'a PreferenceManager) -> ContextStack<'c> { + let prefs = pref_manager.merge_prefs(); + return ContextStack { + base: ContextStack::base_context(prefs), + old_values: Vec::with_capacity(31) // should avoid allocations + } + } + + fn base_context(var_defs: PreferenceHashMap) -> Context<'c> { let mut context = Context::new(); context.set_namespace("m", "http://www.w3.org/1998/Math/MathML"); crate::xpath_functions::add_builtin_functions(&mut context); - for defs in &self.new_defs { - for (name, value) in defs { - context.set_variable(name.as_str(), yaml_to_value(&value)); - }; + for (key, value) in var_defs { + context.set_variable(key.as_str(), yaml_to_value(&value)); + // if let Some(str_value) = value.as_str() { + // if str_value != "Auto" { + // println!("Set {}='{}'", key.as_str(), str_value); + // } + // } + }; + return context; + } + + fn push(&'c mut self, new_vars: &'c VariableDefinitions, mathml: Element<'c>) -> Result<()> { + // store the old value and set the new one + let mut old_values = VariableValues {defs: Vec::with_capacity(new_vars.defs.len()) }; + let evaluation = Evaluation::new(&self.base, Node::Element(mathml)); + for def in &new_vars.defs { + // get the old value (might not be defined) + let qname = QName::new(def.name.as_str()); + let old_value = match evaluation.value_of(qname) { + Some(val) => Some( val.clone() ), + None => None, + }; + old_values.defs.push( VariableValue{ name: &def.name, value: old_value} ); + + // set the new value + let new_value = match def.value.evaluate(&self.base, mathml) { + Ok(val) => val, + Err(_) => bail!(format!("Can't evaluate variable def for {}", def)), + }; + self.base.set_variable(qname, new_value); } - self.contexts.push(Rc::new( context )); + self.old_values.push(old_values); + return Ok( () ); } fn pop(&mut self) { - self.new_defs.pop(); - self.contexts.pop(); - } - - fn top<'a>(&self) -> Rc>{ - return Rc::clone(&self.contexts[self.contexts.len()-1] ); + const MISSING_VALUE: &str = "-- unset value --"; // can't remove a variable from context, so use this value + let old_values = self.old_values.pop().unwrap(); + for variable in old_values.defs { + let qname = QName::new(variable.name); + let old_value = match variable.value { + None => Value::String(MISSING_VALUE.to_string()), + Some(val) => val, + }; + self.base.set_variable(qname, old_value); + } } } @@ -1367,7 +1404,7 @@ impl UnicodeDef { fn build(unicode_def: &Yaml, file_name: &Path, rules: &mut SpeechRules) -> Result<()> { if let Ok(include_file_name) = find_str(unicode_def, "include") { let do_include_fn = |new_file: &Path| { - rules.read_unicode(&[Some(new_file.to_path_buf()), None, None]); + rules.read_unicode(&[Some(new_file.to_path_buf()), None, None]); }; return process_include(file_name, include_file_name, do_include_fn); @@ -1488,17 +1525,17 @@ pub fn print_errors(e:&Error) { type RuleTable = HashMap>>; type UnicodeTable = HashMap>; -/// `SpeechRules` encapsulates a named group of speech rules (e.g, "ClearSpeak") +/// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") /// along with the preferences to be used for speech. -pub struct SpeechRules{ +struct SpeechRules { name: String, pub pref_manager: Box, - rules: RuleTable, // the speech rules used (partitioned into MathML tags in hashmap, then linearly searched) - translate_single_chars_only: bool, // strings like "half" don't want 'a's translated, but braille does - unicode: UnicodeTable, // the speech rules used for Unicode characters + rules: RuleTable, // the speech rules used (partitioned into MathML tags in hashmap, then linearly searched) + translate_single_chars_only: bool, // strings like "half" don't want 'a's translated, but braille does + unicode: UnicodeTable, // the speech rules used for Unicode characters } -impl fmt::Display for SpeechRules{ +impl fmt::Display for SpeechRules { fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { writeln!(f, "SpeechRules '{}'\n{})", self.name, self.pref_manager)?; let mut rules_vec: Vec<(&String, &Vec>)> = self.rules.iter().collect(); @@ -1510,6 +1547,22 @@ impl fmt::Display for SpeechRules{ } } + +/// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") +/// along with the preferences to be used for speech. +/// Because speech rules can define variables, there is also a context that is carried with them +pub struct SpeechRulesWithContext<'c> { + speech_rules: &'c SpeechRules, + context_stack: ContextStack<'c>, // current value of (context) variables +} + +impl<'c> fmt::Display for SpeechRulesWithContext<'c> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + writeln!(f, "SpeechRulesWithContext \n{})", self.speech_rules)?; + return writeln!(f, " {} context entries", &self.context_stack); + } +} + thread_local!{ /// The current set of speech rules // maybe this should be a small cache of rules in case people switch rules/prefs? @@ -1519,10 +1572,9 @@ thread_local!{ pub static BRAILLE_RULES: RefCell = RefCell::new( SpeechRules::new("braille", false) ); - static CONTEXT_STACK: RefCell> = RefCell::new( ContextStack{ new_defs: vec![], contexts: vec![] } ); + // static CONTEXT_STACK: RefCell> = RefCell::new( ContextStack{ new_defs: vec![], contexts: vec![] } ); } -use crate::prefs::FilesChanged; impl SpeechRules { fn new(name: &str, translate_single_chars_only: bool) -> SpeechRules { let pref_manager = PreferenceManager::new(); @@ -1531,7 +1583,7 @@ impl SpeechRules { exit(1); }; - let rules = SpeechRules{ + let rules = SpeechRules { name: String::from(name), rules: HashMap::with_capacity(31), // lazy load them unicode: HashMap::with_capacity(6997), // lazy load them @@ -1638,37 +1690,37 @@ impl SpeechRules { } } - fn match_pattern(&self, mathml: &Element) -> Result { +} + +use crate::prefs::FilesChanged; +impl<'c> SpeechRulesWithContext<'c> { + fn new(speech_rules: &SpeechRules) -> SpeechRulesWithContext { + return SpeechRulesWithContext { + speech_rules, + context_stack: ContextStack::new(&speech_rules.pref_manager) + } + } + + fn match_pattern(&'c mut self, mathml: Element<'c>) -> Result { // println!("Looking for a match for: \n{}", crate::pretty_print::mml_to_string(mathml)); let mut tag_name = mathml.name().local_part(); - if !self.rules.contains_key(tag_name) { + let rules = &self.speech_rules.rules; + if !rules.contains_key(tag_name) { tag_name = "unknown"; // should be rules for 'unknown' } - let rule_value = self.rules.get(tag_name); + let rule_value = rules.get(tag_name); if let Some(rule_vector) = rule_value { for pattern in rule_vector { // println!("Pattern: {}", pattern); - if pattern.is_match(mathml) + if pattern.is_match(&self.context_stack.base, mathml) .chain_err(|| error_string(pattern, mathml) )? { if pattern.var_defs.len() > 0 { - CONTEXT_STACK.with(|context_stack| { - match pattern.var_defs.evaluate_to_yaml(mathml).chain_err(|| error_string(pattern, mathml)) { - Err(e) => Err(e), - Ok(prefs) => { - let mut context = context_stack.borrow_mut(); - context.push(&prefs); - Ok( () ) - }, - } - })? + self.context_stack.push(&pattern.var_defs, mathml); } let result = pattern.replacements.replace(self, mathml); if pattern.var_defs.len() > 0 { - CONTEXT_STACK.with(|context_stack| { - let mut context = context_stack.borrow_mut(); - context.pop(); - }); + self.context_stack.pop(); } return result.chain_err(|| format!( @@ -1678,7 +1730,7 @@ impl SpeechRules { The patterns are in {}.\n", pattern.pattern_name, pattern.tag_name, pattern.replacements.pretty_print_replacements(),pattern.pattern, - pretty_print::mml_to_string(mathml), + pretty_print::mml_to_string(&mathml), pattern.file_name ) ); @@ -1688,17 +1740,18 @@ impl SpeechRules { // unknown element -- should have rules to handle this -- let flow through to default error let mut file_name = "unknown"; - if let Some(path) = &self.pref_manager.get_style_file()[0] { + if let Some(path) = &self.speech_rules.pref_manager.get_style_file()[0] { file_name= path.to_str().unwrap(); } - if self.rules.get("math").is_none() { + if rules.get("math").is_none() { bail!("No rules found for any matches!!! See the error log."); } else { // FIX: handle error appropriately - bail!("\nNo match found!\nMissing patterns in {} or bad MathML.\n{}", file_name, crate::pretty_print::mml_to_string(mathml)); + bail!("\nNo match found!\nMissing patterns in {} or bad MathML.\n{}", + file_name, crate::pretty_print::mml_to_string(&mathml)); } - fn error_string(pattern: &SpeechPattern, mathml: &Element) -> String { + fn error_string(pattern: &SpeechPattern, mathml: Element) -> String { return format!( "error during pattern match using: \"{}\" for \"{}\".\n\ Pattern is \n{}\nMathML for the match:\n\ @@ -1706,19 +1759,19 @@ impl SpeechRules { The patterns are in {}.\n", pattern.pattern_name, pattern.tag_name, pattern.pattern, - pretty_print::mml_to_string(mathml), + pretty_print::mml_to_string(&mathml), pattern.file_name ); } } - fn replace(&self, replacement: &Replacement, mathml: &Element) -> Result { + fn replace(&'c mut self, replacement: &Replacement, mathml: Element<'c>) -> Result { return Ok( match &*replacement { Replacement::Text(t) => t.clone(), - Replacement::XPath(path) => path.replace(&self, mathml)?, + Replacement::XPath(path) => path.replace(self, mathml)?, Replacement::TTS(tts) => { - self.pref_manager.get_tts().replace(&tts, &self.pref_manager, self, mathml)? + self.speech_rules.pref_manager.get_tts().replace(&tts, &self.speech_rules.pref_manager, self, mathml)? }, Replacement::Test(test) => { test.replace(self, mathml)? @@ -1730,13 +1783,13 @@ impl SpeechRules { ) } - fn replace_nodes(&self, nodes: nodeset::Nodeset, mathml: &Element) -> Result { + fn replace_nodes(&'c mut self, nodes: nodeset::Nodeset<'c>, mathml: Element<'c>) -> Result { //println!("replace_nodes: working on {} nodes", nodes.size()); let result = nodes.document_order() .iter() - .map(|node| + .map(|node: &'c Node<'c>| match node { - Node::Element(mathml) => self.match_pattern(&mathml), + Node::Element(n) => self.match_pattern(*n), Node::Text(t) => self.replace_chars(&t.text(), mathml), _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); Ok( "".to_string() )} }) @@ -1745,15 +1798,16 @@ impl SpeechRules { return Ok( result ); } - pub fn replace_chars(&self, str: &str, mathml: & Element) -> Result { + fn replace_chars(&self, str: &str, mathml: Element) -> Result { // Lookup unicode "pronunciation" of char // Note: TTS is not supported here (not needed and a little less efficient) + let rules = self.speech_rules; let mut chars = str.chars(); - if self.translate_single_chars_only { + if rules.translate_single_chars_only { let ch = chars.next().unwrap_or(' '); if chars.next().is_none() { // single char - return replace_single_char(&self, ch, mathml) + return replace_single_char(self, ch, mathml) } else { // more than one char (don't use str.len() since that is bytes, not chars) return Ok(String::from(str)); @@ -1761,14 +1815,14 @@ impl SpeechRules { }; let result = chars - .map(|ch| replace_single_char(&self, ch, mathml)) + .map(|ch| replace_single_char(self, ch, mathml)) .collect::>>()? .join(""); return Ok( result ); - fn replace_single_char(rules: &SpeechRules, ch: char, mathml: & Element) -> Result { + fn replace_single_char<'c>(rules_with_context: &'c SpeechRulesWithContext<'c>, ch: char, mathml: Element<'c>) -> Result { let ch_as_u32 = ch as u32; - let replacements = rules.unicode.get( &ch_as_u32 ); + let replacements = rules_with_context.speech_rules.unicode.get( &ch_as_u32 ); if replacements.is_none() { return Ok(String::from(ch)); // no replacement, so just return the char and hope for the best }; @@ -1778,7 +1832,7 @@ impl SpeechRules { replacements.unwrap() .iter() .map(|replacement| - rules.replace(replacement, mathml) + rules_with_context.replace(replacement, mathml) .chain_err(|| format!("Unicode replacement error: {}", replacement)) ) .collect::>>()? .join(" ") @@ -1787,6 +1841,19 @@ impl SpeechRules { } } +// Hack to allow replacement of `str` with braille chars. +pub fn braille_replace_chars(str: &str, mathml: Element) -> Result { + return BRAILLE_RULES.with(|rules| { + // this is called while BRAILLE_SPEECH_RULES is mutably borrowed, so we can't borrow it again + // instead hack to get around borrow rules because we know no changes happen during call + unsafe { + let rules_with_context = SpeechRulesWithContext::new(rules.as_ptr().as_ref().unwrap()); + return rules_with_context.replace_chars(str, mathml); + } + }) +} + + #[cfg(test)] mod tests { diff --git a/src/tts.rs b/src/tts.rs index d05984e53..1a8f6e3b1 100755 --- a/src/tts.rs +++ b/src/tts.rs @@ -65,7 +65,7 @@ use sxd_document::dom::Element; use yaml_rust::Yaml; use std::{fmt}; -use crate::speech::{SpeechRules}; +use crate::speech::{SpeechRulesWithContext}; use strum_macros::IntoStaticStr; use regex::Regex; @@ -221,7 +221,7 @@ impl TTS { /// A string is returned for the speech engine. /// /// `auto` pausing is handled at a later phase and a special char is used for it - pub fn replace(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules: &SpeechRules, mathml: &Element) -> Result { + pub fn replace<'c>(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { // The general idea is we handle the begin tag, the contents, and then the end tag // For the begin/end tag, we dispatch off to specialized code for each TTS engine let mut result = String::with_capacity(255); @@ -236,7 +236,7 @@ impl TTS { if result.is_empty() { result += " "; } - result += &command.replacements.replace(rules, mathml)?; + result += &command.replacements.replace(rules_with_context, mathml)?; } let end_tag = match self { diff --git a/src/xpath_functions.rs b/src/xpath_functions.rs index b35afa321..222d0b5ca 100755 --- a/src/xpath_functions.rs +++ b/src/xpath_functions.rs @@ -745,50 +745,43 @@ impl NemethChars { _ => "R", // normal and unknown }, }; - return crate::speech::BRAILLE_RULES.with(|rules| { - let text = as_text(*node); - // start of pattern matching mutably borrows BRAILLE_RULES, so we can't use borrow. - // instead hack to get around borrow rules because we know no changes happen during call - let braille_chars = unsafe { - rules.as_ptr().as_ref().unwrap() - .replace_chars(text, node).unwrap_or("".to_string()) - }; - println!("braille_chars: '{}'", braille_chars); - - // we want to pull the prefix (typeface, language) out to the front until a change happens - // the same is true for number indicator - // also true (sort of) for capitalization -- if all caps, use double cap in front (assume abbr or Roman Numeral) - let is_in_enclosed_list = name(node) == "mn" && NemethChars::is_in_enclosed_list(*node); - let mut typeface = "".to_string(); // illegal value to force first value - let mut is_all_caps = true; - let result = PICK_APART_CHAR.replace_all(&braille_chars, |caps: &Captures| { - println!(" face: {:?}, lang: {:?}, num {:?}, cap: {:?}, char: {:?}", - &caps["face"], &caps["lang"], &caps["num"], &caps["cap"], &caps["char"]); - let mut nemeth_chars = "".to_string(); - let typeface_changed = &typeface != &caps["face"]; - if typeface_changed { - typeface = caps["face"].to_string(); // needs to outlast this instance of the loop - nemeth_chars += if typeface.is_empty() {attr_typeface} else {&typeface}; - nemeth_chars += &caps["lang"]; - } else { - nemeth_chars += &caps["lang"]; - } - println!("is_in_list: {}; num: {}", is_in_enclosed_list, caps["num"].is_empty()); - if !caps["num"].is_empty() && (typeface_changed || !is_in_enclosed_list) { - nemeth_chars += "N"; - } - is_all_caps &= !&caps["cap"].is_empty(); - nemeth_chars += &caps["cap"]; // will be stripped later if all caps - nemeth_chars += &caps["char"]; - return nemeth_chars; - }); - let mut text_chars = text.chars(); // see if more than one char - if is_all_caps && text_chars.next().is_some() && text_chars.next().is_some() { - return Ok( "CC".to_string() + &result.replace("C", "")); + let text = as_text(*node); + let braille_chars = crate::speech::braille_replace_chars(text, *node).unwrap_or("".to_string()); + println!("braille_chars: '{}'", braille_chars); + + // we want to pull the prefix (typeface, language) out to the front until a change happens + // the same is true for number indicator + // also true (sort of) for capitalization -- if all caps, use double cap in front (assume abbr or Roman Numeral) + let is_in_enclosed_list = name(node) == "mn" && NemethChars::is_in_enclosed_list(*node); + let mut typeface = "".to_string(); // illegal value to force first value + let mut is_all_caps = true; + let result = PICK_APART_CHAR.replace_all(&braille_chars, |caps: &Captures| { + println!(" face: {:?}, lang: {:?}, num {:?}, cap: {:?}, char: {:?}", + &caps["face"], &caps["lang"], &caps["num"], &caps["cap"], &caps["char"]); + let mut nemeth_chars = "".to_string(); + let typeface_changed = &typeface != &caps["face"]; + if typeface_changed { + typeface = caps["face"].to_string(); // needs to outlast this instance of the loop + nemeth_chars += if typeface.is_empty() {attr_typeface} else {&typeface}; + nemeth_chars += &caps["lang"]; } else { - return Ok( result.to_string() ); + nemeth_chars += &caps["lang"]; + } + println!("is_in_list: {}; num: {}", is_in_enclosed_list, caps["num"].is_empty()); + if !caps["num"].is_empty() && (typeface_changed || !is_in_enclosed_list) { + nemeth_chars += "N"; } + is_all_caps &= !&caps["cap"].is_empty(); + nemeth_chars += &caps["cap"]; // will be stripped later if all caps + nemeth_chars += &caps["char"]; + return nemeth_chars; }); + let mut text_chars = text.chars(); // see if more than one char + if is_all_caps && text_chars.next().is_some() && text_chars.next().is_some() { + return Ok( "CC".to_string() + &result.replace("C", "")); + } else { + return Ok( result.to_string() ); + } } fn is_in_enclosed_list(node: Element) -> bool { From 53ef4850c7acf2afb05ce637cfd0e5cea5e43059 Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 11:22:47 -0700 Subject: [PATCH 2/7] Add test for "|" as a comparison operator ("such that"). This works for the tests in rule 145, not sure it will work for "divides" or other infix vertical bar rules. Remove some DEBUG() helpers --- Rules/Nemeth/unicode.yaml | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/Rules/Nemeth/unicode.yaml b/Rules/Nemeth/unicode.yaml index d50d16587..1b90cf056 100755 --- a/Rules/Nemeth/unicode.yaml +++ b/Rules/Nemeth/unicode.yaml @@ -2179,9 +2179,9 @@ # test if first ancestor that isn't an mrow is a script tag (rule 78) - if: "self::m:mn" then: [t: "⠠"] - - else_if: "ancestor-or-self::*[not(parent::m:mrow)][1][parent::m:msub or DEBUG(parent::m:msup )or parent::m:msubsup][preceding-sibling::*]" + - else_if: "ancestor-or-self::*[not(parent::m:mrow)][1][parent::m:msub or parent::m:msup or parent::m:msubsup][preceding-sibling::*]" then: [t: "⠪"] # Rule 78 - else: [t: "⠠⠀"] + else: [t: ","] # ',' used for matching purposes, converted to "⠠⠀" - "-": # 0x002D (Hyphen) (0x2212 normalized to here) - test: if: "self::m:mtext" @@ -2190,9 +2190,9 @@ - ".": # 0x002E (Full stop or decimal pt) - test: - if: "DEBUG(self::m:mn)" + if: "self::m:mn" then_test: - if: "DEBUG(string-length(text()) = 1)" + if: "string-length(text()) = 1" then: [t: "𝑁⠨"] # example 8c(4) -- lone "." is not considered numeric, but not a period else: [t: "N⠨"] else: [t: "P⠲"] # period @@ -2210,7 +2210,11 @@ - "^": [t: "⠸⠣"] # 0x005E (Circumflex accent) - "_": [t: ""] # 0x005F (Low line) - "{": [t: "⠨⠷"] # 0x007B (Left curly bracket) - - "|": [t: "⠳"] # 0x007C (Vertical line) + - "|": # 0x007C (Vertical line) + - test: + if: "preceding-sibling::* and following-sibling::*" + then: [t: "⠀⠳⠀"] # comparison (e.g., such that -- rule 145) + else: [t: "⠳"] # absolute value, others??? - "}": [t: "⠨⠾"] # 0x007D (Right curly bracket) - "~": [t: "⠈⠱"] # 0x007E (Tilde) - "¢": [t: "⠈⠉"] # 0x00A2 (Cent sign) From cc3c18f3a439b2d8c081c77d9a1ff943fbb1b58f Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 11:25:43 -0700 Subject: [PATCH 3/7] tweak matrix rules for matrix with just one row to avoid 'large' paren version of chars --- Rules/Nemeth/Nemeth_Rules.yaml | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/Rules/Nemeth/Nemeth_Rules.yaml b/Rules/Nemeth/Nemeth_Rules.yaml index 03a051ccf..17df9092d 100755 --- a/Rules/Nemeth/Nemeth_Rules.yaml +++ b/Rules/Nemeth/Nemeth_Rules.yaml @@ -214,7 +214,7 @@ tag: mrow variables: - RowStart: "*[1]" - - RowEnd: "DEBUG(*[3])" + - RowEnd: "*[3]" match: - "*[2][self::m:mtable] and" - (IsBracketed(., '(', ')') or IsBracketed(., '[', ']') or IsBracketed(., '|', '|')) @@ -230,9 +230,9 @@ match: "." replace: - test: - if: "preceding-sibling::*" - then: [t: "⠀"] - - t: "⠠" + if: "count(parent::*) > 1" + then: [t: "⠠"] + - t: "" - x: $RowStart - test: if: .[self::m:mlabeledtr] @@ -243,7 +243,9 @@ if: .[self::m:mlabeledtr] then: [x: "*[position()>1]"] else: {x: "*"} - - t: "⠠" + - test: + if: "count(parent::*) > 1" + then: [t: "⠠"] - x: $RowEnd - name: default From de2a9fbdcad21ade9847e75655c00a5284d27fe7 Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 11:48:44 -0700 Subject: [PATCH 4/7] improve/fix cleanup for numbers --- src/canonicalize.rs | 60 +++++++++++++++++++++++++++++++++++++++------ 1 file changed, 53 insertions(+), 7 deletions(-) diff --git a/src/canonicalize.rs b/src/canonicalize.rs index ac3da0752..5f3a2581a 100755 --- a/src/canonicalize.rs +++ b/src/canonicalize.rs @@ -309,6 +309,14 @@ pub fn canonicalize(mathml: Element) -> Element { struct CanonicalizeContext { } +#[derive(PartialEq)] +#[allow(non_camel_case_types)] +enum DigitBlockType { + None, + DecimalBlock_3, + BinaryBlock_4, +} + impl CanonicalizeContext { fn new() -> CanonicalizeContext { return CanonicalizeContext{} @@ -443,12 +451,22 @@ impl CanonicalizeContext { return mathml; } - fn is_digit_block(mathml: Element) -> bool { + fn is_digit_block(mathml: Element) -> DigitBlockType { // returns true if an 'mn' with exactly three digits lazy_static! { - static ref IS_DIGIT_BLOCK: Regex = Regex::new(r"^\d\d\d$").unwrap(); // only Unicode whitespace + static ref IS_DIGIT_BLOCK: Regex = Regex::new(r"^\d\d\d$").unwrap(); + static ref IS_BINARY_DIGIT_BLOCK: Regex = Regex::new(r"^[01]{4}$").unwrap(); + } + if name(&mathml) == "mn" { + let text = as_text(mathml); + if IS_DIGIT_BLOCK.is_match(text) { + return DigitBlockType::DecimalBlock_3; + } + if IS_BINARY_DIGIT_BLOCK.is_match(text) { + return DigitBlockType::BinaryBlock_4; + } } - return name(&mathml) == "mn" && IS_DIGIT_BLOCK.is_match(as_text(mathml)); + return DigitBlockType::None; } fn merge_number_blocks<'a>(mrow: Element<'a>) -> Element<'a> { @@ -459,6 +477,8 @@ impl CanonicalizeContext { let mut i = 0; while i < children.len() { let child = as_element(children[i]); + let mut is_comma = false; + let mut is_decimal_pt = false; if name(&child) == "mn" { let mut looking_for_separator = true; let mut end = children.len() - i; @@ -467,14 +487,28 @@ impl CanonicalizeContext { let sibling_name = name(&sibling); // FIX: generalize to more types of spacing (e.g., mtext with space) // FIX: generalize to include locale ("." vs ",") + if looking_for_separator { + if sibling_name == "mo" { + let leaf_text = as_text(sibling); + is_comma = leaf_text == ","; + is_decimal_pt = leaf_text == "."; + } else { + is_comma = false; + is_decimal_pt = false; + } + }; + println!("j/name={}/{}, looking={}, is ',' {}, '.' {}, ", + i+j, sibling_name, looking_for_separator, is_comma, is_decimal_pt); if !(looking_for_separator && - (sibling_name == "mspace" || (sibling_name == "mo" && as_text(sibling) == ","))) && - (looking_for_separator || !is_digit_block(sibling)) { - end = j; + (sibling_name == "mspace" || is_comma || is_decimal_pt)) && + ( looking_for_separator || + !(is_decimal_pt || is_digit_block(sibling) != DigitBlockType::None)) { + end = j+1; break; } looking_for_separator = !looking_for_separator; } + println!("start={}, end={}", i, i+end); if is_likely_a_number(mrow, i, i+end) { merge_block(mrow, i, i+end); children = mrow.children(); // mrow has changed, so we need a new children array @@ -497,6 +531,18 @@ impl CanonicalizeContext { } if name(&as_element(children[start+1])) == "mspace" || IS_WHITESPACE.is_match(as_text(as_element(children[start+1]))) { + // make sure all the digit blocks are of the same type + let mut digit_block = DigitBlockType::None; // initial "illegal" value (we know it is not NONE) + for child in children { + let child = as_element(child); + if name(&child) == "mn" { + if digit_block == DigitBlockType::None { + digit_block = is_digit_block(child); + } else if is_digit_block(child) != digit_block { + return false; // differing digit block types + } + } + } return true; // digit block separated by whitespace } @@ -514,7 +560,7 @@ impl CanonicalizeContext { } let parent = parent.element().unwrap(); if name(&parent) != "mrow" { - // if parent is not an mrow, then there aren't parens around this + // if parent is not an mrow, then there aren't parens around this return true; } let preceding = parent.preceding_siblings(); From e2317c84e4162fbf5c42a8908e243c14f6ed1f39 Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 11:50:39 -0700 Subject: [PATCH 5/7] Fixed Rust memory checker problem by adding a lifetime for references SpeechRules can now be borrowed immutably, so cleaned that up Changed "comma" to use "," which is cleaned up later. Passes all current tests in Rules. --- src/main.rs | 68 ++++++++-------- src/speech.rs | 213 +++++++++++++++++++++++++++----------------------- src/tts.rs | 2 +- 3 files changed, 151 insertions(+), 132 deletions(-) diff --git a/src/main.rs b/src/main.rs index db1e4d5dc..1d6c7149d 100755 --- a/src/main.rs +++ b/src/main.rs @@ -46,37 +46,37 @@ fn main() { // let logger = builder.build().unwrap(); // info!(logger, "Hello World!"); -// let expr = " -// -// -// [ -// -// -// -// 3 -// -// -// 1 -// -// -// 4 -// -// -// -// -// 0 -// -// -// 2 -// -// -// 6 -// -// -// -// ] -// -// "; + let expr = " + + + [ + + + + 3 + + + 1 + + + 4 + + + + + 0 + + + 2 + + + 6 + + + + ] + +"; // let expr = " // @@ -84,9 +84,9 @@ fn main() { // + // f⁡(x+y) // ""; -let expr = "c=4598 - 037 - 234"; +// let expr = "c=4598 +// 037 +// 234"; // let expr = "𝟏𝟐𝟑"; // let expr = "A=00,B= // 01,…,Z=25"; diff --git a/src/speech.rs b/src/speech.rs index 0aa6c476c..27a423ad9 100755 --- a/src/speech.rs +++ b/src/speech.rs @@ -15,7 +15,6 @@ use sxd_xpath::context::Evaluation; use sxd_xpath::{Context, Factory, Value, XPath, nodeset}; use sxd_xpath::nodeset::Node; use std::fmt; -use std::borrow::{Cow}; use crate::{errors::*, pretty_print}; use crate::prefs::*; use yaml_rust::{YamlLoader, Yaml, yaml::Hash}; @@ -38,10 +37,13 @@ use phf::phf_map; /// If there is an error, the speech string will indicate an error. pub fn speak_mathml(mathml: Element) -> String { SPEECH_RULES.with(|rules| { - let mut rules = rules.borrow_mut(); - rules.update(true); + { + let mut mut_rules = rules.borrow_mut(); + mut_rules.update(true); + } + let rules = rules.borrow(); let mut rules_with_context = SpeechRulesWithContext::new(&rules); - match rules_with_context.match_pattern(mathml) { + match rules_with_context.match_pattern(&mathml) { Ok(speech_string) => { return rules.pref_manager.get_tts() .merge_pauses(remove_optional_indicators( @@ -60,10 +62,13 @@ pub fn speak_mathml(mathml: Element) -> String { pub fn braille_mathml(mathml: Element) -> String { BRAILLE_RULES.with(|rules| { - let mut rules = rules.borrow_mut(); - rules.update(false); + { + let mut mut_rules = rules.borrow_mut(); + mut_rules.update(false); + } + let rules = rules.borrow(); let mut rules_with_context = SpeechRulesWithContext::new(&rules); - match rules_with_context.match_pattern(mathml) { + match rules_with_context.match_pattern(&&mathml) { // FIX: need to set name of speech rules so test Nemeth/UEB clean for Ok(speech_string) => { return nemeth_cleanup(speech_string.replace(" ", "")); @@ -104,7 +109,8 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { "N" => "", "n" => "⠼", "𝑁" => "", - "W" => "⠀" + "W" => "⠀", + "," => "⠠⠀" }; lazy_static! { @@ -134,7 +140,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // add after decimal pt for non-digits except for comma and punctuation static ref MULTI_177_5: Regex = - Regex::new(r"([N𝑁]⠨)([^N𝑁⠠P])").unwrap(); + Regex::new(r"([N𝑁]⠨)([^N𝑁,P])").unwrap(); // Pattern for rule II.9a (add numeric indicator at start of line or after a space) and 9a (add after typeface) @@ -143,7 +149,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // 3. optional typeface indicator // 4. number (N) static ref NUM_IND_9A: Regex = - Regex::new(r"(?P^|⠀)(?P⠤?)(?P[SBTIR]*?)N").unwrap(); + Regex::new(r"(?P^|[,⠀])(?P⠤?)(?P[SBTIR]*?)N").unwrap(); // FIX add rule 9d after section mark, etc @@ -154,7 +160,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Never use punctuation indicator before these (38-6) // "…": "⠀⠄⠄⠄" // "-": "⠸⠤" (hyphen and dash) - // ",": "⠠⠀" -- spacing already add + // ",": "⠠⠀" -- spacing already added // Rule II.9b (add numeric indicator after punctuation [optional minus[optional .][digit] // because this is run after the above rule, some cases are already caught, so don't // match if there is already a numeric indicator @@ -169,12 +175,12 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { static ref REMOVE_PUNCT_IND: Regex = Regex::new(r"(^|⠀|\w)P(.)").unwrap(); - static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNn𝑁W])").unwrap(); + static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNn𝑁W,])").unwrap(); static ref REMOVE_LEVEL_IND_BEFORE_BASELINE: Regex = Regex::new(r"(?:[⠘⠰]+⠐)").unwrap(); // Before 79b (punctuation) - static ref REMOVE_LEVEL_IND_BEFORE_SPACE_OR_PUNCT: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)([P⠠⠀]|$)").unwrap(); + static ref REMOVE_LEVEL_IND_BEFORE_SPACE_COMMA_PUNCT: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)([⠀,P]|$)").unwrap(); static ref COLLAPSE_SPACES: Regex = Regex::new(r"⠀⠀+").unwrap(); } @@ -212,8 +218,11 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // strip level indicators // checks for punctuation char, so needs to before punctuation is stripped. - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_SPACE_OR_PUNCT, "$1"); - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_BASELINE, "⠐"); + + let result = REMOVE_LEVEL_IND_BEFORE_SPACE_COMMA_PUNCT.replace_all(&result, "$1"); + println!("Punct : \"{}\"", &result); + let result = REMOVE_LEVEL_IND_BEFORE_BASELINE.replace_all(&result, "⠐"); + println!("Bseline: \"{}\"", &result); let result = REMOVE_PUNCT_IND.replace_all(&result, "$1$2"); println!("Punct38: \"{}\"", &result); @@ -229,11 +238,6 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { let result = COLLAPSE_SPACES.replace_all(&result, "⠀"); return result.to_string(); - - fn do_replace_all<'a>(old: &'a str, regex: &Regex, replace: &str) -> Cow<'a, str> { - let replace = replace.to_string() + "$1"; - return regex.replace_all(old, replace.as_str()); - } } @@ -452,7 +456,7 @@ impl fmt::Display for InsertChildren { } } -impl InsertChildren { +impl<'r> InsertChildren { fn build(insert: &Yaml) -> Result> { // 'insert:' -- 'nodes': xxx 'replace': xxx if insert.as_hash().is_none() { @@ -483,7 +487,7 @@ impl InsertChildren { // The solution adopted is to find out the number of nodes and build up MyXPaths with each node selected (e.g, "*" => "*[3]") // and put those nodes into a flat ReplacementArray and then do a standard replace on that. // This is slower than the alternatives, but reuses a bunch of code and hence is less complicated. - fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { let result = self.xpath.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; match result { @@ -531,7 +535,7 @@ impl fmt::Display for ReplacementArray { } } -impl ReplacementArray { +impl<'r> ReplacementArray { /// Return an empty `ReplacementArray` pub fn build_empty() -> ReplacementArray { return ReplacementArray { @@ -560,15 +564,23 @@ impl ReplacementArray { } /// Do all the replacements in `mathml` using `rules`. - pub fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + pub fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { // do the replacements // remove the empty strings (the later 'join' would add extraneous spaces) // collect the strings together into an array - let mut replacement_strings = - self.replacements.iter() - .map(|group| rules_with_context.replace(group, mathml)) - .filter(|result| if let Ok(str) = result {!str.is_empty()} else {true}) - .collect::>>()?; + let mut replacement_strings = Vec::with_capacity(32); // probably conservative guess + for replacement in self.replacements.iter() { + let string = rules_with_context.replace(replacement, mathml)?; + if !string.is_empty() { + replacement_strings.push(string); + } + } + + // let mut replacement_strings = + // self.replacements.iter() + // .map(|group| rules_with_context.replace(group, mathml)) + // .filter(|result| if let Ok(str) = result {!str.is_empty()} else {true}) + // .collect::>>()?; if replacement_strings.is_empty() { return Ok( "".to_string() ); @@ -680,7 +692,7 @@ impl Clone for MyXPath { } } -impl MyXPath { +impl<'r> MyXPath { fn new(xpath: String) -> Result { return Ok ( MyXPath { @@ -819,7 +831,7 @@ impl MyXPath { } } - fn is_true(&self, context: &Context, mathml: Element) -> Result { + fn is_true(&self, context: &Context, mathml: &Element) -> Result { // return true if there is no condition or if the condition evaluates to true return Ok( match self.evaluate(context, mathml)? { @@ -830,11 +842,10 @@ impl MyXPath { ) } - fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { let result = self.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; - return unsafe { - match result { + return match result { Value::Nodeset(nodes) => { if nodes.size() == 0 { bail!("During replacement, no matching element found"); @@ -844,13 +855,12 @@ impl MyXPath { Value::String(t) => Ok( t ), Value::Number(num) => Ok( num.to_string() ), Value::Boolean(b) => Ok( b.to_string() ), // FIX: is this right??? - } } } - fn evaluate<'a, 'd>(&'a self, context: &'d Context, mathml: Element<'d>) -> Result> { + fn evaluate<'a, 'c>(&'a self, context: &'r Context<'c>, mathml: &'r Element<'c>) -> Result> { // println!("evaluate: {}", self); - let result = self.xpath.evaluate(context, mathml); + let result = self.xpath.evaluate(context, *mathml); return match result { Ok(val) => { // println!(" result: '{:?}'", val); @@ -992,7 +1002,7 @@ impl SpeechPattern { return Ok( () ); } - fn is_match(&self, context: &Context, mathml: Element) -> Result { + fn is_match(&self, context: &Context, mathml: &Element) -> Result { if self.tag_name != mathml.name().local_part() && self.tag_name != "unknown" { return Ok( false ); } @@ -1028,7 +1038,7 @@ impl fmt::Display for TestArray { } } -impl TestArray { +impl<'r> TestArray { fn build(test: &Yaml) -> Result { // 'test:' for convenience takes either a dictionary with keys if/else_if/then/then_test/else/else_test or // or an array of those values (there should be at most one else/else_test) @@ -1089,7 +1099,7 @@ impl TestArray { return Ok( TestArray { tests: test_array } ); } - fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { for test in &self.tests { if test.is_true(&rules_with_context.context_stack.base, mathml)? { assert!(test.then_part.is_some()); @@ -1122,7 +1132,7 @@ impl fmt::Display for TestOrReplacements { } } -impl TestOrReplacements { +impl<'r> TestOrReplacements { fn build(test: &Yaml, replace_key: &str, test_key: &str, key_required: bool) -> Result> { let part = &test[replace_key]; let test_part = &test[test_key]; @@ -1148,7 +1158,7 @@ impl TestOrReplacements { } } - fn replace<'c>(&self, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { return match self { TestOrReplacements::Replacements(r) => r.replace(rules_with_context, mathml), TestOrReplacements::Test(t) => t.replace(rules_with_context, mathml), @@ -1179,7 +1189,7 @@ impl fmt::Display for Test { } impl Test { - fn is_true(&self, context: &Context, mathml: Element) -> Result { + fn is_true(&self, context: &Context, mathml: &Element) -> Result { return match self.condition.as_ref() { None => Ok( false ), // trivially false -- want to do else part Some(condition) => condition.is_true(context, mathml) @@ -1303,7 +1313,7 @@ impl<'c> fmt::Display for ContextStack<'c> { } } -impl<'c> ContextStack<'c> { +impl<'c, 'r> ContextStack<'c> { fn new<'a>(pref_manager: &'a PreferenceManager) -> ContextStack<'c> { let prefs = pref_manager.merge_prefs(); return ContextStack { @@ -1327,24 +1337,28 @@ impl<'c> ContextStack<'c> { return context; } - fn push(&'c mut self, new_vars: &'c VariableDefinitions, mathml: Element<'c>) -> Result<()> { + fn push<'a:'c>(&'r mut self, new_vars: &'a VariableDefinitions, mathml: &'r Element<'c>) -> Result<()> { // store the old value and set the new one let mut old_values = VariableValues {defs: Vec::with_capacity(new_vars.defs.len()) }; - let evaluation = Evaluation::new(&self.base, Node::Element(mathml)); + let evaluation = Evaluation::new(&self.base, Node::Element(*mathml)); for def in &new_vars.defs { // get the old value (might not be defined) let qname = QName::new(def.name.as_str()); let old_value = match evaluation.value_of(qname) { - Some(val) => Some( val.clone() ), + Some(val) => Some( val.clone() ), // &Value -> Value None => None, }; old_values.defs.push( VariableValue{ name: &def.name, value: old_value} ); + } + // use a second loop because of borrow problem with self.base and 'evaluation' + for def in &new_vars.defs { // set the new value let new_value = match def.value.evaluate(&self.base, mathml) { Ok(val) => val, Err(_) => bail!(format!("Can't evaluate variable def for {}", def)), }; + let qname = QName::new(def.name.as_str()); self.base.set_variable(qname, new_value); } self.old_values.push(old_values); @@ -1379,14 +1393,6 @@ fn yaml_to_value<'a, 'b>(yaml: &'a Yaml) -> Value<'b> { } } -fn value_to_yaml<'a>(value: &'a Value) -> Result { - return Ok( match value { - Value::String(s) => Yaml::String(s.clone()), - Value::Boolean(b) => Yaml::Boolean(*b), - Value::Number(n) => Yaml::Real(n.to_string()), - Value::Nodeset(nodes) => Yaml::Boolean(nodes.size() > 0), }) -} - // Information for matching a Unicode char (defined in unicode.yaml) and building its replacement struct UnicodeDef { @@ -1527,7 +1533,7 @@ pub fn print_errors(e:&Error) { /// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") /// along with the preferences to be used for speech. -struct SpeechRules { +pub struct SpeechRules { name: String, pub pref_manager: Box, rules: RuleTable, // the speech rules used (partitioned into MathML tags in hashmap, then linearly searched) @@ -1551,12 +1557,12 @@ impl fmt::Display for SpeechRules { /// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") /// along with the preferences to be used for speech. /// Because speech rules can define variables, there is also a context that is carried with them -pub struct SpeechRulesWithContext<'c> { - speech_rules: &'c SpeechRules, +pub struct SpeechRulesWithContext<'c, 's> { + speech_rules: &'s SpeechRules, context_stack: ContextStack<'c>, // current value of (context) variables } -impl<'c> fmt::Display for SpeechRulesWithContext<'c> { +impl<'c, 's> fmt::Display for SpeechRulesWithContext<'c, 's> { fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { writeln!(f, "SpeechRulesWithContext \n{})", self.speech_rules)?; return writeln!(f, " {} context entries", &self.context_stack); @@ -1571,8 +1577,6 @@ thread_local!{ pub static BRAILLE_RULES: RefCell = RefCell::new( SpeechRules::new("braille", false) ); - - // static CONTEXT_STACK: RefCell> = RefCell::new( ContextStack{ new_defs: vec![], contexts: vec![] } ); } impl SpeechRules { @@ -1693,15 +1697,19 @@ impl SpeechRules { } use crate::prefs::FilesChanged; -impl<'c> SpeechRulesWithContext<'c> { - fn new(speech_rules: &SpeechRules) -> SpeechRulesWithContext { +/// We track three different lifetimes: +/// 'c -- the lifetime of the context and mathml +/// 's -- the lifetime of the speech rules (which is static) +/// 'r -- the lifetime of the reference (this seems to be key to keep the rust memory checker happy) +impl<'c, 's:'c, 'r> SpeechRulesWithContext<'c, 's> { + fn new(speech_rules: &'s SpeechRules) -> SpeechRulesWithContext { return SpeechRulesWithContext { speech_rules, context_stack: ContextStack::new(&speech_rules.pref_manager) } } - fn match_pattern(&'c mut self, mathml: Element<'c>) -> Result { + fn match_pattern(&'r mut self, mathml: &'r Element<'c>) -> Result { // println!("Looking for a match for: \n{}", crate::pretty_print::mml_to_string(mathml)); let mut tag_name = mathml.name().local_part(); let rules = &self.speech_rules.rules; @@ -1714,26 +1722,26 @@ impl<'c> SpeechRulesWithContext<'c> { for pattern in rule_vector { // println!("Pattern: {}", pattern); if pattern.is_match(&self.context_stack.base, mathml) - .chain_err(|| error_string(pattern, mathml) )? { + .chain_err(|| error_string(pattern, *mathml) )? { if pattern.var_defs.len() > 0 { - self.context_stack.push(&pattern.var_defs, mathml); + self.context_stack.push(&pattern.var_defs, mathml)?; } let result = pattern.replacements.replace(self, mathml); if pattern.var_defs.len() > 0 { self.context_stack.pop(); } return result.chain_err(|| - format!( - "attempting replacement pattern: \"{}\" for \"{}\".\n\ - Replacement \"{}\" due to matching the following MathML with the pattern \"{}\".\n\ - {}\ - The patterns are in {}.\n", - pattern.pattern_name, pattern.tag_name, - pattern.replacements.pretty_print_replacements(),pattern.pattern, - pretty_print::mml_to_string(&mathml), - pattern.file_name - ) - ); + format!( + "attempting replacement pattern: \"{}\" for \"{}\".\n\ + Replacement \"{}\" due to matching the following MathML with the pattern \"{}\".\n\ + {}\ + The patterns are in {}.\n", + pattern.pattern_name, pattern.tag_name, + pattern.replacements.pretty_print_replacements(),pattern.pattern, + pretty_print::mml_to_string(&mathml), + pattern.file_name + ) + ); } } } @@ -1765,7 +1773,7 @@ impl<'c> SpeechRulesWithContext<'c> { } } - fn replace(&'c mut self, replacement: &Replacement, mathml: Element<'c>) -> Result { + fn replace(&'r mut self, replacement: &Replacement, mathml: &'r Element<'c>) -> Result { return Ok( match &*replacement { Replacement::Text(t) => t.clone(), @@ -1783,22 +1791,36 @@ impl<'c> SpeechRulesWithContext<'c> { ) } - fn replace_nodes(&'c mut self, nodes: nodeset::Nodeset<'c>, mathml: Element<'c>) -> Result { + fn replace_nodes(&'r mut self, nodes: nodeset::Nodeset<'c>, mathml: &'r Element<'c>) -> Result { //println!("replace_nodes: working on {} nodes", nodes.size()); - let result = nodes.document_order() - .iter() - .map(|node: &'c Node<'c>| - match node { - Node::Element(n) => self.match_pattern(*n), - Node::Text(t) => self.replace_chars(&t.text(), mathml), - _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); Ok( "".to_string() )} - }) - .collect::>>()? - .join(" "); + let mut result = String::with_capacity(3*nodes.size()); // guess (2 chars/node + space) + let mut not_first_time = false; + for node in nodes.document_order() { + if not_first_time { + result.push(' '); + not_first_time = true; + } + let matched = match node { + Node::Element(n) => self.match_pattern(&n)?, + Node::Text(t) => self.replace_chars(&t.text(), mathml)?, + _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); "".to_string() } + }; + result += &matched; + } + // let result = nodes.document_order() + // .iter() + // .map(|node: &'r Node<'c>| + // match node { + // Node::Element(n) => self.match_pattern(n), + // Node::Text(t) => self.replace_chars(&t.text(), mathml), + // _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); Ok( "".to_string() )} + // }) + // .collect::>>()? + // .join(" "); return Ok( result ); } - fn replace_chars(&self, str: &str, mathml: Element) -> Result { + fn replace_chars(&'r mut self, str: &str, mathml: &'r Element<'c>) -> Result { // Lookup unicode "pronunciation" of char // Note: TTS is not supported here (not needed and a little less efficient) let rules = self.speech_rules; @@ -1820,7 +1842,7 @@ impl<'c> SpeechRulesWithContext<'c> { .join(""); return Ok( result ); - fn replace_single_char<'c>(rules_with_context: &'c SpeechRulesWithContext<'c>, ch: char, mathml: Element<'c>) -> Result { + fn replace_single_char<'c, 's:'c, 'r>(rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, ch: char, mathml: &'r Element<'c>) -> Result { let ch_as_u32 = ch as u32; let replacements = rules_with_context.speech_rules.unicode.get( &ch_as_u32 ); if replacements.is_none() { @@ -1844,12 +1866,9 @@ impl<'c> SpeechRulesWithContext<'c> { // Hack to allow replacement of `str` with braille chars. pub fn braille_replace_chars(str: &str, mathml: Element) -> Result { return BRAILLE_RULES.with(|rules| { - // this is called while BRAILLE_SPEECH_RULES is mutably borrowed, so we can't borrow it again - // instead hack to get around borrow rules because we know no changes happen during call - unsafe { - let rules_with_context = SpeechRulesWithContext::new(rules.as_ptr().as_ref().unwrap()); - return rules_with_context.replace_chars(str, mathml); - } + let rules = rules.borrow(); + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + return rules_with_context.replace_chars(str, &mathml); }) } diff --git a/src/tts.rs b/src/tts.rs index 1a8f6e3b1..b48d16623 100755 --- a/src/tts.rs +++ b/src/tts.rs @@ -221,7 +221,7 @@ impl TTS { /// A string is returned for the speech engine. /// /// `auto` pausing is handled at a later phase and a special char is used for it - pub fn replace<'c>(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules_with_context: &'c mut SpeechRulesWithContext<'c>, mathml: Element<'c>) -> Result { + pub fn replace<'c, 's:'c, 'r>(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { // The general idea is we handle the begin tag, the contents, and then the end tag // For the begin/end tag, we dispatch off to specialized code for each TTS engine let mut result = String::with_capacity(255); From 8d5966dc6a210d8bdd9eb726e1bfbb374fb0efb2 Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 17:07:05 -0700 Subject: [PATCH 6/7] removed spaces from divides/does not divide as they are not comparison per https://nfb.org/sites/www.nfb.org/files/files-pdf/braille-certification/lesson-6--provisional-5-22-19.pdf --- Rules/Nemeth/unicode.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Rules/Nemeth/unicode.yaml b/Rules/Nemeth/unicode.yaml index 1b90cf056..93c8f0f4e 100755 --- a/Rules/Nemeth/unicode.yaml +++ b/Rules/Nemeth/unicode.yaml @@ -2450,8 +2450,8 @@ - "∟": [t: "⠫⠪⠨⠗⠻"] # 0x221F (Right angle) - "∠": [t: "⠫⠪"] # 0x2220 (Angle) - "∡": [t: "⠫⠪⠈⠫⠁⠻"] # 0x2221 (Measured angle) - - "∣": [t: "⠀⠳⠀"] # 0x2223 (Divides) - - "∤": [t: "⠀⠌⠳⠀"] # 0x2224 (Does not divide) + - "∣": [t: "⠳"] # 0x2223 (Divides) + - "∤": [t: "⠌⠳"] # 0x2224 (Does not divide) - "∥": [t: "⠀⠫⠇⠀"] # 0x2225 (Parallel to) - "∦": [t: "⠀⠌⠫⠇⠀"] # 0x2226 (Not parallel to) - "∧": [t: "⠈⠩"] # 0x2227 (Logical AND) From 65fe76124f43548ac48c21994fe33ecc3ed35e8d Mon Sep 17 00:00:00 2001 From: Neil Soiffer Date: Wed, 29 Sep 2021 17:07:45 -0700 Subject: [PATCH 7/7] correct some tests --- tests/braille/Nemeth/AataNemeth.rs | 15 ++++++++++----- tests/braille/Nemeth/rules.rs | 4 ++-- 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/tests/braille/Nemeth/AataNemeth.rs b/tests/braille/Nemeth/AataNemeth.rs index 53d552f20..fe34d446d 100755 --- a/tests/braille/Nemeth/AataNemeth.rs +++ b/tests/braille/Nemeth/AataNemeth.rs @@ -338,7 +338,8 @@ fn test_045() { #[test] fn test_046() { let expr = "(01000101)"; - test_braille("Nemeth", expr, "⠷⠼⠴⠂⠴⠴⠀⠼⠴⠂⠴⠂⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠴⠂⠴⠴⠀⠴⠂⠴⠂⠾"); } #[test] @@ -493,7 +494,8 @@ fn test_069() { #[test] fn test_070() { let expr = "(10011000)"; - test_braille("Nemeth", expr, "⠷⠼⠂⠴⠴⠂⠀⠼⠂⠴⠴⠴⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠂⠴⠴⠂⠀⠂⠴⠴⠴⠾"); } #[test] @@ -634,7 +636,8 @@ fn test_090() { #[test] fn test_091() { let expr = "(011011100110)"; - test_braille("Nemeth", expr, "⠷⠼⠴⠂⠂⠴⠀⠼⠂⠂⠂⠴⠀⠼⠴⠂⠂⠴⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠴⠂⠂⠴⠀⠂⠂⠂⠴⠀⠴⠂⠂⠴⠾"); } #[test] @@ -993,7 +996,8 @@ fn test_147() { #[test] fn test_148() { let expr = "(n,E)=(451,231)"; - test_braille("Nemeth", expr, "⠷⠝⠠⠀⠠⠑⠾⠀⠨⠅⠀⠷⠲⠢⠂⠠⠀⠆⠒⠂⠾"); + // correct: 8b(1) -- no space after the comma + test_braille("Nemeth", expr, "⠷⠝⠠⠀⠠⠑⠾⠀⠨⠅⠀⠷⠲⠢⠂⠠⠆⠒⠂⠾"); } #[test] @@ -2084,7 +2088,8 @@ fn test_307() { 3i]→ N∪{0}"; - test_braille("Nemeth", expr, "⠨⠝⠸⠒⠀⠈⠰⠠⠵⠈⠷⠜⠒⠻⠊⠈⠾⠀⠫⠕⠀⠈⠰⠠⠝⠨⠬⠨⠷⠴⠨⠾"); + // corrected: removed extra space after "⠸⠒" + test_braille("Nemeth", expr, "⠨⠝⠸⠒⠈⠰⠠⠵⠈⠷⠜⠒⠻⠊⠈⠾⠀⠫⠕⠀⠈⠰⠠⠝⠨⠬⠨⠷⠴⠨⠾"); } #[test] diff --git a/tests/braille/Nemeth/rules.rs b/tests/braille/Nemeth/rules.rs index ffe0cdba3..218e67be0 100755 --- a/tests/braille/Nemeth/rules.rs +++ b/tests/braille/Nemeth/rules.rs @@ -184,8 +184,8 @@ fn comma_in_sup_79_b_4() { #[test] fn text_after_sup_79_c_3() { - // bad mn from Wiris - let expr = "6.696×108mph"; + // bad mn from Wiris; also &ao; + let expr = "6.696×108 mph"; test_braille("Nemeth", expr, "⠼⠖⠨⠖⠔⠖⠈⠡⠂⠴⠘⠦⠀⠍⠏⠓"); }