diff --git a/Rules/Nemeth/Nemeth_Rules.yaml b/Rules/Nemeth/Nemeth_Rules.yaml index 03a051ccf..17df9092d 100755 --- a/Rules/Nemeth/Nemeth_Rules.yaml +++ b/Rules/Nemeth/Nemeth_Rules.yaml @@ -214,7 +214,7 @@ tag: mrow variables: - RowStart: "*[1]" - - RowEnd: "DEBUG(*[3])" + - RowEnd: "*[3]" match: - "*[2][self::m:mtable] and" - (IsBracketed(., '(', ')') or IsBracketed(., '[', ']') or IsBracketed(., '|', '|')) @@ -230,9 +230,9 @@ match: "." replace: - test: - if: "preceding-sibling::*" - then: [t: "⠀"] - - t: "⠠" + if: "count(parent::*) > 1" + then: [t: "⠠"] + - t: "" - x: $RowStart - test: if: .[self::m:mlabeledtr] @@ -243,7 +243,9 @@ if: .[self::m:mlabeledtr] then: [x: "*[position()>1]"] else: {x: "*"} - - t: "⠠" + - test: + if: "count(parent::*) > 1" + then: [t: "⠠"] - x: $RowEnd - name: default diff --git a/Rules/Nemeth/unicode.yaml b/Rules/Nemeth/unicode.yaml index d50d16587..93c8f0f4e 100755 --- a/Rules/Nemeth/unicode.yaml +++ b/Rules/Nemeth/unicode.yaml @@ -2179,9 +2179,9 @@ # test if first ancestor that isn't an mrow is a script tag (rule 78) - if: "self::m:mn" then: [t: "⠠"] - - else_if: "ancestor-or-self::*[not(parent::m:mrow)][1][parent::m:msub or DEBUG(parent::m:msup )or parent::m:msubsup][preceding-sibling::*]" + - else_if: "ancestor-or-self::*[not(parent::m:mrow)][1][parent::m:msub or parent::m:msup or parent::m:msubsup][preceding-sibling::*]" then: [t: "⠪"] # Rule 78 - else: [t: "⠠⠀"] + else: [t: ","] # ',' used for matching purposes, converted to "⠠⠀" - "-": # 0x002D (Hyphen) (0x2212 normalized to here) - test: if: "self::m:mtext" @@ -2190,9 +2190,9 @@ - ".": # 0x002E (Full stop or decimal pt) - test: - if: "DEBUG(self::m:mn)" + if: "self::m:mn" then_test: - if: "DEBUG(string-length(text()) = 1)" + if: "string-length(text()) = 1" then: [t: "𝑁⠨"] # example 8c(4) -- lone "." is not considered numeric, but not a period else: [t: "N⠨"] else: [t: "P⠲"] # period @@ -2210,7 +2210,11 @@ - "^": [t: "⠸⠣"] # 0x005E (Circumflex accent) - "_": [t: ""] # 0x005F (Low line) - "{": [t: "⠨⠷"] # 0x007B (Left curly bracket) - - "|": [t: "⠳"] # 0x007C (Vertical line) + - "|": # 0x007C (Vertical line) + - test: + if: "preceding-sibling::* and following-sibling::*" + then: [t: "⠀⠳⠀"] # comparison (e.g., such that -- rule 145) + else: [t: "⠳"] # absolute value, others??? - "}": [t: "⠨⠾"] # 0x007D (Right curly bracket) - "~": [t: "⠈⠱"] # 0x007E (Tilde) - "¢": [t: "⠈⠉"] # 0x00A2 (Cent sign) @@ -2446,8 +2450,8 @@ - "∟": [t: "⠫⠪⠨⠗⠻"] # 0x221F (Right angle) - "∠": [t: "⠫⠪"] # 0x2220 (Angle) - "∡": [t: "⠫⠪⠈⠫⠁⠻"] # 0x2221 (Measured angle) - - "∣": [t: "⠀⠳⠀"] # 0x2223 (Divides) - - "∤": [t: "⠀⠌⠳⠀"] # 0x2224 (Does not divide) + - "∣": [t: "⠳"] # 0x2223 (Divides) + - "∤": [t: "⠌⠳"] # 0x2224 (Does not divide) - "∥": [t: "⠀⠫⠇⠀"] # 0x2225 (Parallel to) - "∦": [t: "⠀⠌⠫⠇⠀"] # 0x2226 (Not parallel to) - "∧": [t: "⠈⠩"] # 0x2227 (Logical AND) diff --git a/src/canonicalize.rs b/src/canonicalize.rs index ac3da0752..5f3a2581a 100755 --- a/src/canonicalize.rs +++ b/src/canonicalize.rs @@ -309,6 +309,14 @@ pub fn canonicalize(mathml: Element) -> Element { struct CanonicalizeContext { } +#[derive(PartialEq)] +#[allow(non_camel_case_types)] +enum DigitBlockType { + None, + DecimalBlock_3, + BinaryBlock_4, +} + impl CanonicalizeContext { fn new() -> CanonicalizeContext { return CanonicalizeContext{} @@ -443,12 +451,22 @@ impl CanonicalizeContext { return mathml; } - fn is_digit_block(mathml: Element) -> bool { + fn is_digit_block(mathml: Element) -> DigitBlockType { // returns true if an 'mn' with exactly three digits lazy_static! { - static ref IS_DIGIT_BLOCK: Regex = Regex::new(r"^\d\d\d$").unwrap(); // only Unicode whitespace + static ref IS_DIGIT_BLOCK: Regex = Regex::new(r"^\d\d\d$").unwrap(); + static ref IS_BINARY_DIGIT_BLOCK: Regex = Regex::new(r"^[01]{4}$").unwrap(); + } + if name(&mathml) == "mn" { + let text = as_text(mathml); + if IS_DIGIT_BLOCK.is_match(text) { + return DigitBlockType::DecimalBlock_3; + } + if IS_BINARY_DIGIT_BLOCK.is_match(text) { + return DigitBlockType::BinaryBlock_4; + } } - return name(&mathml) == "mn" && IS_DIGIT_BLOCK.is_match(as_text(mathml)); + return DigitBlockType::None; } fn merge_number_blocks<'a>(mrow: Element<'a>) -> Element<'a> { @@ -459,6 +477,8 @@ impl CanonicalizeContext { let mut i = 0; while i < children.len() { let child = as_element(children[i]); + let mut is_comma = false; + let mut is_decimal_pt = false; if name(&child) == "mn" { let mut looking_for_separator = true; let mut end = children.len() - i; @@ -467,14 +487,28 @@ impl CanonicalizeContext { let sibling_name = name(&sibling); // FIX: generalize to more types of spacing (e.g., mtext with space) // FIX: generalize to include locale ("." vs ",") + if looking_for_separator { + if sibling_name == "mo" { + let leaf_text = as_text(sibling); + is_comma = leaf_text == ","; + is_decimal_pt = leaf_text == "."; + } else { + is_comma = false; + is_decimal_pt = false; + } + }; + println!("j/name={}/{}, looking={}, is ',' {}, '.' {}, ", + i+j, sibling_name, looking_for_separator, is_comma, is_decimal_pt); if !(looking_for_separator && - (sibling_name == "mspace" || (sibling_name == "mo" && as_text(sibling) == ","))) && - (looking_for_separator || !is_digit_block(sibling)) { - end = j; + (sibling_name == "mspace" || is_comma || is_decimal_pt)) && + ( looking_for_separator || + !(is_decimal_pt || is_digit_block(sibling) != DigitBlockType::None)) { + end = j+1; break; } looking_for_separator = !looking_for_separator; } + println!("start={}, end={}", i, i+end); if is_likely_a_number(mrow, i, i+end) { merge_block(mrow, i, i+end); children = mrow.children(); // mrow has changed, so we need a new children array @@ -497,6 +531,18 @@ impl CanonicalizeContext { } if name(&as_element(children[start+1])) == "mspace" || IS_WHITESPACE.is_match(as_text(as_element(children[start+1]))) { + // make sure all the digit blocks are of the same type + let mut digit_block = DigitBlockType::None; // initial "illegal" value (we know it is not NONE) + for child in children { + let child = as_element(child); + if name(&child) == "mn" { + if digit_block == DigitBlockType::None { + digit_block = is_digit_block(child); + } else if is_digit_block(child) != digit_block { + return false; // differing digit block types + } + } + } return true; // digit block separated by whitespace } @@ -514,7 +560,7 @@ impl CanonicalizeContext { } let parent = parent.element().unwrap(); if name(&parent) != "mrow" { - // if parent is not an mrow, then there aren't parens around this + // if parent is not an mrow, then there aren't parens around this return true; } let preceding = parent.preceding_siblings(); diff --git a/src/interface.rs b/src/interface.rs index 423b992dc..a1c90fa36 100755 --- a/src/interface.rs +++ b/src/interface.rs @@ -47,7 +47,7 @@ pub fn speak_mathml(mathml_str: &str) -> String { let package = package.unwrap(); cleanup_mathml(&package); let mathml = get_element(&package); - return crate::speech::speak_mathml(&mathml); + return crate::speech::speak_mathml(mathml); } /// Given MathML, a Unicode braille string is return. @@ -64,7 +64,7 @@ pub fn braille_mathml(mathml_str: &str) -> String { let package = package.unwrap(); cleanup_mathml(&package); let mathml = get_element(&package); - return crate::speech::braille_mathml(&mathml); + return crate::speech::braille_mathml(mathml); } thread_local!{ @@ -108,7 +108,7 @@ pub fn GetSpokenText(_py: Python) -> PyResult { return MATHML_INSTANCE.with(|package_instance| { let package_instance = package_instance.borrow(); let mathml = get_element(&*package_instance); - let speech = crate::speech::speak_mathml(&mathml); + let speech = crate::speech::speak_mathml(mathml); eprintln!("Time taken: {}ms", instant.elapsed().as_millis()); return Ok( speech ); }); @@ -196,8 +196,15 @@ fn set_speech_tags(pref_manager: &mut PreferenceManager, speech_tags: String ) - /// Get the braille associated with the MathML that was set by [`SetMathML`]. /// The braille returned depends upon the preference for braille output. pub fn GetBraille(_py: Python) -> PyResult { - // FIX: not yet implemented (basically what the braille says) - return Ok("⠠⠃⠗⠁⠊⠇⠇⠑ ⠛⠑⠝⠑⠗⠁⠞⠊⠕⠝ ⠝⠕⠞ ⠽⠑⠞ ⠊⠍⠏⠇⠑⠍⠑⠝⠞".to_string()); + use std::time::{Instant}; + let instant = Instant::now(); + return MATHML_INSTANCE.with(|package_instance| { + let package_instance = package_instance.borrow(); + let mathml = get_element(&*package_instance); + let braille = crate::speech::braille_mathml(mathml); + eprintln!("Time taken: {}ms", instant.elapsed().as_millis()); + return Ok( braille ); + }); } #[pyfunction] diff --git a/src/main.rs b/src/main.rs index db1e4d5dc..1d6c7149d 100755 --- a/src/main.rs +++ b/src/main.rs @@ -46,37 +46,37 @@ fn main() { // let logger = builder.build().unwrap(); // info!(logger, "Hello World!"); -// let expr = " -// -// -// [ -// -// -// -// 3 -// -// -// 1 -// -// -// 4 -// -// -// -// -// 0 -// -// -// 2 -// -// -// 6 -// -// -// -// ] -// -// "; + let expr = " + + + [ + + + + 3 + + + 1 + + + 4 + + + + + 0 + + + 2 + + + 6 + + + + ] + +"; // let expr = " // @@ -84,9 +84,9 @@ fn main() { // + // f⁡(x+y) // ""; -let expr = "c=4598 - 037 - 234"; +// let expr = "c=4598 +// 037 +// 234"; // let expr = "𝟏𝟐𝟑"; // let expr = "A=00,B= // 01,…,Z=25"; diff --git a/src/speech.rs b/src/speech.rs index 3e16f3834..27a423ad9 100755 --- a/src/speech.rs +++ b/src/speech.rs @@ -9,12 +9,12 @@ use std::{collections::HashMap, process::exit}; use std::cell::RefCell; -use std::rc::Rc; use sxd_document::dom::Element; +use sxd_document::QName; +use sxd_xpath::context::Evaluation; use sxd_xpath::{Context, Factory, Value, XPath, nodeset}; use sxd_xpath::nodeset::Node; use std::fmt; -use std::borrow::{Cow}; use crate::{errors::*, pretty_print}; use crate::prefs::*; use yaml_rust::{YamlLoader, Yaml, yaml::Hash}; @@ -35,15 +35,15 @@ use phf::phf_map; /// /// A string is returned in call cases. /// If there is an error, the speech string will indicate an error. -pub fn speak_mathml(mathml: &Element) -> String { +pub fn speak_mathml(mathml: Element) -> String { SPEECH_RULES.with(|rules| { - let mut rules = rules.borrow_mut(); - rules.update(true); - CONTEXT_STACK.with(|cs| { - let mut cs = cs.borrow_mut(); - cs.init(&rules.pref_manager) - }); - match rules.match_pattern(mathml) { + { + let mut mut_rules = rules.borrow_mut(); + mut_rules.update(true); + } + let rules = rules.borrow(); + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + match rules_with_context.match_pattern(&mathml) { Ok(speech_string) => { return rules.pref_manager.get_tts() .merge_pauses(remove_optional_indicators( @@ -60,15 +60,15 @@ pub fn speak_mathml(mathml: &Element) -> String { }) } -pub fn braille_mathml(mathml: &Element) -> String { +pub fn braille_mathml(mathml: Element) -> String { BRAILLE_RULES.with(|rules| { - let mut rules = rules.borrow_mut(); - rules.update(false); - CONTEXT_STACK.with(|cs| { - let mut cs = cs.borrow_mut(); - cs.init(&rules.pref_manager) - }); - match rules.match_pattern(mathml) { + { + let mut mut_rules = rules.borrow_mut(); + mut_rules.update(false); + } + let rules = rules.borrow(); + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + match rules_with_context.match_pattern(&&mathml) { // FIX: need to set name of speech rules so test Nemeth/UEB clean for Ok(speech_string) => { return nemeth_cleanup(speech_string.replace(" ", "")); @@ -85,7 +85,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Typeface: S: sans-serif, B: bold, T: script/blackboard, I: italic, R: Roman // Language: E: English, D: German, G: Greek, V: Greek variants, H: Hebrew, U: Russian // Indicators: C: capital, N: number, P: punctuation, M: multipurpose - // Others: W -- whitespace that should be kept (e.g, in a numeral) + // Others: + // W -- whitespace that should be kept (e.g, in a numeral) + // 𝑁 -- hack for special case of a lone decimal pt -- not considered a number but follows rules mostly // SRE doesn't have H: Hebrew or U: Russian, so not encoded (yet) // Note: some "positive" patterns find cases to keep the char and transform them to the lower case version static INDICATOR_REPLACEMENTS: phf::Map<&str, &str> = phf_map! { @@ -106,7 +108,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { "m" => "⠐", "N" => "", "n" => "⠼", - "W" => "⠀" + "𝑁" => "", + "W" => "⠀", + "," => "⠠⠀" }; lazy_static! { @@ -128,15 +132,15 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Multipurpose indicator insertion // 177.2 -- add after a letter and before a digit (or decimal pt) -- these will start with N static ref MULTI_177_2: Regex = - Regex::new(r"([⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵])N").unwrap(); + Regex::new(r"([⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵])[N𝑁]").unwrap(); // keep between numeric subscript and digit ('M' added by subscript rule) static ref MULTI_177_3: Regex = - Regex::new(r"(N.)M(N.)").unwrap(); + Regex::new(r"([N𝑁].)M([N𝑁].)").unwrap(); // add after decimal pt for non-digits except for comma and punctuation static ref MULTI_177_5: Regex = - Regex::new(r"N⠨([^N⠠P])").unwrap(); + Regex::new(r"([N𝑁]⠨)([^N𝑁,P])").unwrap(); // Pattern for rule II.9a (add numeric indicator at start of line or after a space) and 9a (add after typeface) @@ -145,7 +149,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // 3. optional typeface indicator // 4. number (N) static ref NUM_IND_9A: Regex = - Regex::new(r"(?P^|⠀)(?P⠤?)(?P[SBTIR]*?)N").unwrap(); + Regex::new(r"(?P^|[,⠀])(?P⠤?)(?P[SBTIR]*?)N").unwrap(); // FIX add rule 9d after section mark, etc @@ -156,7 +160,7 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { // Never use punctuation indicator before these (38-6) // "…": "⠀⠄⠄⠄" // "-": "⠸⠤" (hyphen and dash) - // ",": "⠠⠀" -- spacing already add + // ",": "⠠⠀" -- spacing already added // Rule II.9b (add numeric indicator after punctuation [optional minus[optional .][digit] // because this is run after the above rule, some cases are already caught, so don't // match if there is already a numeric indicator @@ -171,10 +175,12 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { static ref REMOVE_PUNCT_IND: Regex = Regex::new(r"(^|⠀|\w)P(.)").unwrap(); - static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNnW])").unwrap(); + static ref REPLACE_INDICATORS: Regex =Regex::new(r"([SBTIREDGVHPCMmNn𝑁W,])").unwrap(); static ref REMOVE_LEVEL_IND_BEFORE_BASELINE: Regex = Regex::new(r"(?:[⠘⠰]+⠐)").unwrap(); - static ref REMOVE_LEVEL_IND_BEFORE_SPACE: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)(⠀|$)").unwrap(); + + // Before 79b (punctuation) + static ref REMOVE_LEVEL_IND_BEFORE_SPACE_COMMA_PUNCT: Regex = Regex::new(r"(?:[⠘⠰]+⠐?|⠐)([⠀,P]|$)").unwrap(); static ref COLLAPSE_SPACES: Regex = Regex::new(r"⠀⠀+").unwrap(); } @@ -193,9 +199,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { println!("spaces: \"{}\"", result); // Multipurpose indicator - let result = MULTI_177_2.replace_all(&result, "${1}mN"); + let result = MULTI_177_2.replace_all(&result, "${1}m${2}"); let result = MULTI_177_3.replace_all(&result, "${1}m$2"); - let result = MULTI_177_5.replace_all(&result, "N⠨m$1"); + let result = MULTI_177_5.replace_all(&result, "${1}m$2"); println!("MULTI: \"{}\"", result); let result = NUM_IND_9A.replace_all(&result, "$start$minus${face}n"); @@ -210,9 +216,18 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { let result = NUM_IND_AFTER_PUNCT.replace_all(&result, "$punct${minus}n"); println!("A PUNCT: \"{}\"", &result); + // strip level indicators + // checks for punctuation char, so needs to before punctuation is stripped. + + let result = REMOVE_LEVEL_IND_BEFORE_SPACE_COMMA_PUNCT.replace_all(&result, "$1"); + println!("Punct : \"{}\"", &result); + let result = REMOVE_LEVEL_IND_BEFORE_BASELINE.replace_all(&result, "⠐"); + println!("Bseline: \"{}\"", &result); + let result = REMOVE_PUNCT_IND.replace_all(&result, "$1$2"); println!("Punct38: \"{}\"", &result); + let result = REPLACE_INDICATORS.replace_all(&result, |cap: &Captures| { match INDICATOR_REPLACEMENTS.get(&cap[0]) { None => panic!("REPLACE_INDICATORS and INDICATOR_REPLACEMENTS are not in sync"), @@ -220,18 +235,9 @@ fn nemeth_cleanup(raw_nemeth: String) -> String { } }); - // strip level indicators before a space - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_SPACE, "$1"); - let result = do_replace_all(&result, &REMOVE_LEVEL_IND_BEFORE_BASELINE, "⠐"); - let result = COLLAPSE_SPACES.replace_all(&result, "⠀"); return result.to_string(); - - fn do_replace_all<'a>(old: &'a str, regex: &Regex, replace: &str) -> Cow<'a, str> { - let replace = replace.to_string() + "$1"; - return regex.replace_all(old, replace.as_str()); - } } @@ -450,7 +456,7 @@ impl fmt::Display for InsertChildren { } } -impl InsertChildren { +impl<'r> InsertChildren { fn build(insert: &Yaml) -> Result> { // 'insert:' -- 'nodes': xxx 'replace': xxx if insert.as_hash().is_none() { @@ -481,8 +487,8 @@ impl InsertChildren { // The solution adopted is to find out the number of nodes and build up MyXPaths with each node selected (e.g, "*" => "*[3]") // and put those nodes into a flat ReplacementArray and then do a standard replace on that. // This is slower than the alternatives, but reuses a bunch of code and hence is less complicated. - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { - let result = self.xpath.evaluate(mathml) + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { + let result = self.xpath.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; match result { Value::Nodeset(nodes) => { @@ -505,11 +511,11 @@ impl InsertChildren { ); } let replacements = ReplacementArray{ replacements: expanded_result }; - return replacements.replace(rules, mathml); + return replacements.replace(rules_with_context, mathml); }, // FIX: should the options be errors??? - Value::String(t) => { return rules.replace_chars(&t, mathml); }, + Value::String(t) => { return rules_with_context.replace_chars(&t, mathml); }, Value::Number(num) => { return Ok( num.to_string() ); }, Value::Boolean(b) => { return Ok( b.to_string() ); }, // FIX: is this right??? } @@ -529,7 +535,7 @@ impl fmt::Display for ReplacementArray { } } -impl ReplacementArray { +impl<'r> ReplacementArray { /// Return an empty `ReplacementArray` pub fn build_empty() -> ReplacementArray { return ReplacementArray { @@ -558,15 +564,23 @@ impl ReplacementArray { } /// Do all the replacements in `mathml` using `rules`. - pub fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + pub fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { // do the replacements // remove the empty strings (the later 'join' would add extraneous spaces) // collect the strings together into an array - let mut replacement_strings = - self.replacements.iter() - .map(|group| rules.replace(group, mathml)) - .filter(|result| if let Ok(str) = result {!str.is_empty()} else {true}) - .collect::>>()?; + let mut replacement_strings = Vec::with_capacity(32); // probably conservative guess + for replacement in self.replacements.iter() { + let string = rules_with_context.replace(replacement, mathml)?; + if !string.is_empty() { + replacement_strings.push(string); + } + } + + // let mut replacement_strings = + // self.replacements.iter() + // .map(|group| rules_with_context.replace(group, mathml)) + // .filter(|result| if let Ok(str) = result {!str.is_empty()} else {true}) + // .collect::>>()?; if replacement_strings.is_empty() { return Ok( "".to_string() ); @@ -594,7 +608,7 @@ impl ReplacementArray { let after = if i+1 == replacement_strings.len() {""} else {&replacement_strings[i+1]}; replacement_strings[i] = replacement_strings[i].replace( PAUSE_AUTO_STR, - &rules.pref_manager.get_tts().compute_auto_pause(&rules.pref_manager, before, after)); + &rules_with_context.speech_rules.pref_manager.get_tts().compute_auto_pause(&rules_with_context.speech_rules.pref_manager, before, after)); } } @@ -678,7 +692,7 @@ impl Clone for MyXPath { } } -impl MyXPath { +impl<'r> MyXPath { fn new(xpath: String) -> Result { return Ok ( MyXPath { @@ -817,10 +831,10 @@ impl MyXPath { } } - fn is_true(&self, mathml: &Element) -> Result { + fn is_true(&self, context: &Context, mathml: &Element) -> Result { // return true if there is no condition or if the condition evaluates to true return Ok( - match self.evaluate(mathml)? { + match self.evaluate(context, mathml)? { Value::Boolean(b) => b, Value::Nodeset(nodes) => nodes.size() > 0, _ => false, @@ -828,84 +842,37 @@ impl MyXPath { ) } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { - let result = self.evaluate(mathml) + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { + let result = self.evaluate(&rules_with_context.context_stack.base, mathml) .chain_err(||"replacing after pattern match" )?; - let answer; - match result { - Value::Nodeset(nodes) => { - if nodes.size() == 0 { - bail!("During replacement, no matching element found"); - } - return rules.replace_nodes(nodes, mathml); - }, - // Value::String(t) => { return rules.replace_chars(&t, mathml); }, - Value::String(t) => { answer = t; }, - Value::Number(num) => { answer = num.to_string(); }, - Value::Boolean(b) => { answer = b.to_string(); }, // FIX: is this right??? + return match result { + Value::Nodeset(nodes) => { + if nodes.size() == 0 { + bail!("During replacement, no matching element found"); + } + rules_with_context.replace_nodes(nodes, mathml) + }, + Value::String(t) => Ok( t ), + Value::Number(num) => Ok( num.to_string() ), + Value::Boolean(b) => Ok( b.to_string() ), // FIX: is this right??? } - return Ok( answer ); } - fn evaluate<'a,'c>(&'a self, mathml: &'c Element) -> Result> { - return CONTEXT_STACK.with(|context_stack| { - let context_stack = context_stack.borrow(); - let context = context_stack.top(); - // println!("evaluate: {}", self); - let result = self.xpath.evaluate(&context, *mathml); - return match result { - Ok(val) => Ok( val ), - Err(e) => { - bail!( "{}\n\n", - e.to_string() // remove confusing parts of error message from xpath - .replace("OwnedPrefixedName { prefix: None, local_part:", "") - .replace(" }", "") ); - } - }; - }); - } -} - -// Used for speech rules with "variables: ..." -#[derive(Debug)] -struct VariableDefinition { - name: String, // name of variable - value: MyXPath, // value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes -} - -impl fmt::Display for VariableDefinition { - fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { - return write!(f, "{{name: {}, value: {}}}", self.name, self.value); - } -} - -#[derive(Debug)] -struct VariableDefinitions { - defs: Vec -} - -impl VariableDefinitions { - fn new() -> VariableDefinitions { - return VariableDefinitions{ defs: Vec::new() }; - } - - fn push(&mut self, var_def: VariableDefinition) { - self.defs.push(var_def); - } - - fn len(&self) -> usize { - return self.defs.len(); - } - - fn evaluate_to_yaml(&self, mathml: &Element) -> Result { - let mut new_prefs = HashMap::with_capacity(self.defs.len()); - for var_def in &self.defs { - new_prefs.insert( - var_def.name.clone(), - value_to_yaml(&var_def.value.evaluate(mathml)?) - .chain_err(|| format!("while evaluating variable '{}'", var_def.name))?); + fn evaluate<'a, 'c>(&'a self, context: &'r Context<'c>, mathml: &'r Element<'c>) -> Result> { + // println!("evaluate: {}", self); + let result = self.xpath.evaluate(context, *mathml); + return match result { + Ok(val) => { + // println!(" result: '{:?}'", val); + Ok( val ) + }, + Err(e) => { + bail!( "{}\n\n", + e.to_string() // remove confusing parts of error message from xpath + .replace("OwnedPrefixedName { prefix: None, local_part:", "") + .replace(" }", "") ); + } }; - return Ok( new_prefs ); } } @@ -920,11 +887,11 @@ struct SpeechPattern { tag_name: String, file_name: String, pattern: MyXPath, // the xpath expr to attempt to match - var_defs: VariableDefinitions, // any variable definitions [can be and probably is an empty vector most of the time] + var_defs: VariableDefinitions, // any variable definitions [can be and probably is an empty vector most of the time] replacements: ReplacementArray, // the replacements in case there is a match } -impl fmt::Display for SpeechPattern { +impl<'a> fmt::Display for SpeechPattern { fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { return write!(f, "{{name: {}, tag: {}, variables: {:?}, pattern: {}, replacement: {}}}", self.pattern_name, self.tag_name, self.var_defs, self.pattern, @@ -1006,7 +973,7 @@ impl SpeechPattern { format!("value for 'match' in rule ({}: {}):\n{}", tag_name, pattern_name, yaml_to_string(dict, 1)) })?, - var_defs: ContextStack::build(&dict["variables"]) + var_defs: VariableDefinitions::build(&dict["variables"]) .chain_err(|| { format!("value for 'variables' in rule ({}: {}):\n{}", tag_name, pattern_name, yaml_to_string(dict, 1)) @@ -1035,7 +1002,7 @@ impl SpeechPattern { return Ok( () ); } - fn is_match(&self, mathml: &Element) -> Result { + fn is_match(&self, context: &Context, mathml: &Element) -> Result { if self.tag_name != mathml.name().local_part() && self.tag_name != "unknown" { return Ok( false ); } @@ -1044,7 +1011,7 @@ impl SpeechPattern { // println!(" pattern_expr {:?}", self.pattern_expr); // print!("is_match: mathml is\n{}", crate::pretty_print::mml_to_string(mathml)); return Ok( - match self.pattern.evaluate(mathml)? { + match self.pattern.evaluate(context, mathml)? { Value::Boolean(b) => b, Value::Nodeset(nodes) => nodes.size() > 0, _ => false, @@ -1071,7 +1038,7 @@ impl fmt::Display for TestArray { } } -impl TestArray { +impl<'r> TestArray { fn build(test: &Yaml) -> Result { // 'test:' for convenience takes either a dictionary with keys if/else_if/then/then_test/else/else_test or // or an array of those values (there should be at most one else/else_test) @@ -1103,7 +1070,7 @@ impl TestArray { let else_part = TestOrReplacements::build(&test, "else", "else_test", false)?; let n_keys = if else_part.is_none() {2} else {3}; if test.as_hash().unwrap().len() > n_keys { - bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in an else clause of 'test'"); + bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in the 'then' clause of 'test'"); }; test_array.push( Test { condition, then_part, else_part } @@ -1112,7 +1079,7 @@ impl TestArray { // second case: should be else/else_test let else_part = TestOrReplacements::build(&test, "else", "else_test", true)?; if test.as_hash().unwrap().len() > 1 { - bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found in an else clause of 'test'"); + bail!("A key other than 'if', 'else_if', 'then', 'then_test', 'else', or 'else_test' was found the 'else' clause of 'test'"); }; test_array.push( Test { condition: None, then_part: None, else_part } @@ -1132,13 +1099,13 @@ impl TestArray { return Ok( TestArray { tests: test_array } ); } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { for test in &self.tests { - if test.is_true(mathml)? { + if test.is_true(&rules_with_context.context_stack.base, mathml)? { assert!(test.then_part.is_some()); - return test.then_part.as_ref().unwrap().replace(rules, mathml); + return test.then_part.as_ref().unwrap().replace(rules_with_context, mathml); } else if let Some(else_part) = test.else_part.as_ref() { - return else_part.replace(rules, mathml); + return else_part.replace(rules_with_context, mathml); } } return Ok( "".to_string() ); @@ -1165,7 +1132,7 @@ impl fmt::Display for TestOrReplacements { } } -impl TestOrReplacements { +impl<'r> TestOrReplacements { fn build(test: &Yaml, replace_key: &str, test_key: &str, key_required: bool) -> Result> { let part = &test[replace_key]; let test_part = &test[test_key]; @@ -1191,10 +1158,10 @@ impl TestOrReplacements { } } - fn replace(&self, rules: &SpeechRules, mathml: &Element) -> Result { + fn replace<'c, 's:'c>(&self, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { return match self { - TestOrReplacements::Replacements(r) => r.replace(rules, mathml), - TestOrReplacements::Test(t) => t.replace(rules, mathml), + TestOrReplacements::Replacements(r) => r.replace(rules_with_context, mathml), + TestOrReplacements::Test(t) => t.replace(rules_with_context, mathml), } } } @@ -1222,49 +1189,47 @@ impl fmt::Display for Test { } impl Test { - fn is_true(&self, mathml: &Element) -> Result { + fn is_true(&self, context: &Context, mathml: &Element) -> Result { return match self.condition.as_ref() { None => Ok( false ), // trivially false -- want to do else part - Some(condition) => condition.is_true(mathml) + Some(condition) => condition.is_true(context, mathml) .chain_err(|| "Failure in conditional test"), } } } +// Used for speech rules with "variables: ..." +#[derive(Debug)] +struct VariableDefinition { + name: String, // name of variable + value: MyXPath, // xpath value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes +} -struct ContextStack<'c>{ - // FIX: this really should just clone the top of the stack and add on new vars. - // However, Context does not support 'clone' because the functions it stores are traits, not types - // Instead, we recreate the base each time and we store the vars to add in a stack (yuck!) - // Fortunately, adding new vars is rare - new_defs: Vec, - contexts: Vec>>, +impl fmt::Display for VariableDefinition { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + return write!(f, "{{name: {}, value: {}}}", self.name, self.value); + } } -impl<'c> ContextStack<'static>{ - fn init<'d>(&mut self, pref_manager: &'d PreferenceManager) { - let prefs = pref_manager.merge_prefs(); - let context = ContextStack::base_context(&prefs); - self.new_defs.push(prefs); - self.contexts.push(context); - } +// Used for speech rules with "variables: ..." +#[derive(Debug)] +struct VariableValue<'v> { + name: &'v str, // name of variable + value: Option>, // xpath value, typically a constant like "true" or "0", but could be "*/*[1]" to store some nodes +} - fn base_context<'a>(var_defs: &'a PreferenceHashMap) -> Rc> { - let mut context = Context::new(); - context.set_namespace("m", "http://www.w3.org/1998/Math/MathML"); - crate::xpath_functions::add_builtin_functions(&mut context); - for (key, value) in var_defs { - context.set_variable(key.as_str(), yaml_to_value(value)); - // if let Some(str_value) = value.as_str() { - // if str_value != "Auto" { - // println!("Set {}='{}'", key.as_str(), str_value); - // } - // } +impl<'v> fmt::Display for VariableValue<'v> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + let value = match &self.value { + None => "unset".to_string(), + Some(val) => format!("{:?}", val) }; - return Rc::new( context ); - } + return write!(f, "{{name: {}, value: {}}}", self.name, value); + } +} - fn new_def(name_value_def: &Yaml) -> Result<(String, MyXPath)> { +impl VariableDefinition { + fn build(name_value_def: &Yaml) -> Result { match name_value_def.as_hash() { Some(map) => { if map.len() != 1 { @@ -1280,23 +1245,46 @@ impl<'c> ContextStack<'static>{ _ => bail!("definition value is not a string, boolean, or number. Found {}", yaml_to_string(value, 1) ) }; - return Ok( (name, MyXPath::build(value)? ) ); + return Ok( + VariableDefinition{ + name, + value: MyXPath::build(value)? + } + ); }, None => bail!("definition is not a key/value pair. Found {}", yaml_to_string(name_value_def, 1) ) } } +} + + +#[derive(Debug)] +struct VariableDefinitions { + defs: Vec +} + +#[derive(Debug)] +struct VariableValues<'v> { + defs: Vec> +} + +impl VariableDefinitions { + fn new(len: usize) -> VariableDefinitions { + return VariableDefinitions{ defs: Vec::with_capacity(len) }; + } fn build(defs: &Yaml) -> Result { if defs.is_badvalue() { - return Ok( VariableDefinitions::new() ); + return Ok( VariableDefinitions::new(0) ); }; if defs.is_array() { - let mut definitions = VariableDefinitions::new(); - for def in defs.as_vec().unwrap() { - let (name, value) = ContextStack::new_def(def) + let defs = defs.as_vec().unwrap(); + let mut definitions = VariableDefinitions::new(defs.len()); + for def in defs { + let variable_def = VariableDefinition::build(def) .chain_err(|| "definition of 'variables'")?; - definitions.push( VariableDefinition{ name, value} ); + definitions.push( variable_def); }; return Ok (definitions ); } @@ -1304,27 +1292,90 @@ impl<'c> ContextStack<'static>{ yaml_to_string(defs, 1) ); } - fn push(&mut self, new_prefs: &PreferenceHashMap) { - // evaluate the XPath's into - self.new_defs.push(new_prefs.clone()); // do first so they are included in context + fn push(&mut self, var_def: VariableDefinition) { + self.defs.push(var_def); + } + + fn len(&self) -> usize { + return self.defs.len(); + } +} + +struct ContextStack<'c> { + // Note: values are generated by calling value_of on an Evaluation -- that makes the two lifetimes the same + old_values: Vec>, // store old values so they can be set on pop + base: Context<'c> // initial context -- contains all the function defs and pref variables +} + +impl<'c> fmt::Display for ContextStack<'c> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + return writeln!(f, " {} old_values", self.old_values.len()); + } +} + +impl<'c, 'r> ContextStack<'c> { + fn new<'a>(pref_manager: &'a PreferenceManager) -> ContextStack<'c> { + let prefs = pref_manager.merge_prefs(); + return ContextStack { + base: ContextStack::base_context(prefs), + old_values: Vec::with_capacity(31) // should avoid allocations + } + } + + fn base_context(var_defs: PreferenceHashMap) -> Context<'c> { let mut context = Context::new(); context.set_namespace("m", "http://www.w3.org/1998/Math/MathML"); crate::xpath_functions::add_builtin_functions(&mut context); - for defs in &self.new_defs { - for (name, value) in defs { - context.set_variable(name.as_str(), yaml_to_value(&value)); - }; + for (key, value) in var_defs { + context.set_variable(key.as_str(), yaml_to_value(&value)); + // if let Some(str_value) = value.as_str() { + // if str_value != "Auto" { + // println!("Set {}='{}'", key.as_str(), str_value); + // } + // } + }; + return context; + } + + fn push<'a:'c>(&'r mut self, new_vars: &'a VariableDefinitions, mathml: &'r Element<'c>) -> Result<()> { + // store the old value and set the new one + let mut old_values = VariableValues {defs: Vec::with_capacity(new_vars.defs.len()) }; + let evaluation = Evaluation::new(&self.base, Node::Element(*mathml)); + for def in &new_vars.defs { + // get the old value (might not be defined) + let qname = QName::new(def.name.as_str()); + let old_value = match evaluation.value_of(qname) { + Some(val) => Some( val.clone() ), // &Value -> Value + None => None, + }; + old_values.defs.push( VariableValue{ name: &def.name, value: old_value} ); } - self.contexts.push(Rc::new( context )); - } - fn pop(&mut self) { - self.new_defs.pop(); - self.contexts.pop(); + // use a second loop because of borrow problem with self.base and 'evaluation' + for def in &new_vars.defs { + // set the new value + let new_value = match def.value.evaluate(&self.base, mathml) { + Ok(val) => val, + Err(_) => bail!(format!("Can't evaluate variable def for {}", def)), + }; + let qname = QName::new(def.name.as_str()); + self.base.set_variable(qname, new_value); + } + self.old_values.push(old_values); + return Ok( () ); } - fn top<'a>(&self) -> Rc>{ - return Rc::clone(&self.contexts[self.contexts.len()-1] ); + fn pop(&mut self) { + const MISSING_VALUE: &str = "-- unset value --"; // can't remove a variable from context, so use this value + let old_values = self.old_values.pop().unwrap(); + for variable in old_values.defs { + let qname = QName::new(variable.name); + let old_value = match variable.value { + None => Value::String(MISSING_VALUE.to_string()), + Some(val) => val, + }; + self.base.set_variable(qname, old_value); + } } } @@ -1342,14 +1393,6 @@ fn yaml_to_value<'a, 'b>(yaml: &'a Yaml) -> Value<'b> { } } -fn value_to_yaml<'a>(value: &'a Value) -> Result { - return Ok( match value { - Value::String(s) => Yaml::String(s.clone()), - Value::Boolean(b) => Yaml::Boolean(*b), - Value::Number(n) => Yaml::Real(n.to_string()), - Value::Nodeset(nodes) => Yaml::Boolean(nodes.size() > 0), }) -} - // Information for matching a Unicode char (defined in unicode.yaml) and building its replacement struct UnicodeDef { @@ -1367,7 +1410,7 @@ impl UnicodeDef { fn build(unicode_def: &Yaml, file_name: &Path, rules: &mut SpeechRules) -> Result<()> { if let Ok(include_file_name) = find_str(unicode_def, "include") { let do_include_fn = |new_file: &Path| { - rules.read_unicode(&[Some(new_file.to_path_buf()), None, None]); + rules.read_unicode(&[Some(new_file.to_path_buf()), None, None]); }; return process_include(file_name, include_file_name, do_include_fn); @@ -1488,17 +1531,17 @@ pub fn print_errors(e:&Error) { type RuleTable = HashMap>>; type UnicodeTable = HashMap>; -/// `SpeechRules` encapsulates a named group of speech rules (e.g, "ClearSpeak") +/// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") /// along with the preferences to be used for speech. -pub struct SpeechRules{ +pub struct SpeechRules { name: String, pub pref_manager: Box, - rules: RuleTable, // the speech rules used (partitioned into MathML tags in hashmap, then linearly searched) - translate_single_chars_only: bool, // strings like "half" don't want 'a's translated, but braille does - unicode: UnicodeTable, // the speech rules used for Unicode characters + rules: RuleTable, // the speech rules used (partitioned into MathML tags in hashmap, then linearly searched) + translate_single_chars_only: bool, // strings like "half" don't want 'a's translated, but braille does + unicode: UnicodeTable, // the speech rules used for Unicode characters } -impl fmt::Display for SpeechRules{ +impl fmt::Display for SpeechRules { fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { writeln!(f, "SpeechRules '{}'\n{})", self.name, self.pref_manager)?; let mut rules_vec: Vec<(&String, &Vec>)> = self.rules.iter().collect(); @@ -1510,6 +1553,22 @@ impl fmt::Display for SpeechRules{ } } + +/// `SpeechRulesWithContext` encapsulates a named group of speech rules (e.g, "ClearSpeak") +/// along with the preferences to be used for speech. +/// Because speech rules can define variables, there is also a context that is carried with them +pub struct SpeechRulesWithContext<'c, 's> { + speech_rules: &'s SpeechRules, + context_stack: ContextStack<'c>, // current value of (context) variables +} + +impl<'c, 's> fmt::Display for SpeechRulesWithContext<'c, 's> { + fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result { + writeln!(f, "SpeechRulesWithContext \n{})", self.speech_rules)?; + return writeln!(f, " {} context entries", &self.context_stack); + } +} + thread_local!{ /// The current set of speech rules // maybe this should be a small cache of rules in case people switch rules/prefs? @@ -1518,11 +1577,8 @@ thread_local!{ pub static BRAILLE_RULES: RefCell = RefCell::new( SpeechRules::new("braille", false) ); - - static CONTEXT_STACK: RefCell> = RefCell::new( ContextStack{ new_defs: vec![], contexts: vec![] } ); } -use crate::prefs::FilesChanged; impl SpeechRules { fn new(name: &str, translate_single_chars_only: bool) -> SpeechRules { let pref_manager = PreferenceManager::new(); @@ -1531,7 +1587,7 @@ impl SpeechRules { exit(1); }; - let rules = SpeechRules{ + let rules = SpeechRules { name: String::from(name), rules: HashMap::with_capacity(31), // lazy load them unicode: HashMap::with_capacity(6997), // lazy load them @@ -1638,67 +1694,72 @@ impl SpeechRules { } } - fn match_pattern(&self, mathml: &Element) -> Result { +} + +use crate::prefs::FilesChanged; +/// We track three different lifetimes: +/// 'c -- the lifetime of the context and mathml +/// 's -- the lifetime of the speech rules (which is static) +/// 'r -- the lifetime of the reference (this seems to be key to keep the rust memory checker happy) +impl<'c, 's:'c, 'r> SpeechRulesWithContext<'c, 's> { + fn new(speech_rules: &'s SpeechRules) -> SpeechRulesWithContext { + return SpeechRulesWithContext { + speech_rules, + context_stack: ContextStack::new(&speech_rules.pref_manager) + } + } + + fn match_pattern(&'r mut self, mathml: &'r Element<'c>) -> Result { // println!("Looking for a match for: \n{}", crate::pretty_print::mml_to_string(mathml)); let mut tag_name = mathml.name().local_part(); - if !self.rules.contains_key(tag_name) { + let rules = &self.speech_rules.rules; + if !rules.contains_key(tag_name) { tag_name = "unknown"; // should be rules for 'unknown' } - let rule_value = self.rules.get(tag_name); + let rule_value = rules.get(tag_name); if let Some(rule_vector) = rule_value { for pattern in rule_vector { // println!("Pattern: {}", pattern); - if pattern.is_match(mathml) - .chain_err(|| error_string(pattern, mathml) )? { + if pattern.is_match(&self.context_stack.base, mathml) + .chain_err(|| error_string(pattern, *mathml) )? { if pattern.var_defs.len() > 0 { - CONTEXT_STACK.with(|context_stack| { - match pattern.var_defs.evaluate_to_yaml(mathml).chain_err(|| error_string(pattern, mathml)) { - Err(e) => Err(e), - Ok(prefs) => { - let mut context = context_stack.borrow_mut(); - context.push(&prefs); - Ok( () ) - }, - } - })? + self.context_stack.push(&pattern.var_defs, mathml)?; } let result = pattern.replacements.replace(self, mathml); if pattern.var_defs.len() > 0 { - CONTEXT_STACK.with(|context_stack| { - let mut context = context_stack.borrow_mut(); - context.pop(); - }); + self.context_stack.pop(); } return result.chain_err(|| - format!( - "attempting replacement pattern: \"{}\" for \"{}\".\n\ - Replacement \"{}\" due to matching the following MathML with the pattern \"{}\".\n\ - {}\ - The patterns are in {}.\n", - pattern.pattern_name, pattern.tag_name, - pattern.replacements.pretty_print_replacements(),pattern.pattern, - pretty_print::mml_to_string(mathml), - pattern.file_name - ) - ); + format!( + "attempting replacement pattern: \"{}\" for \"{}\".\n\ + Replacement \"{}\" due to matching the following MathML with the pattern \"{}\".\n\ + {}\ + The patterns are in {}.\n", + pattern.pattern_name, pattern.tag_name, + pattern.replacements.pretty_print_replacements(),pattern.pattern, + pretty_print::mml_to_string(&mathml), + pattern.file_name + ) + ); } } } // unknown element -- should have rules to handle this -- let flow through to default error let mut file_name = "unknown"; - if let Some(path) = &self.pref_manager.get_style_file()[0] { + if let Some(path) = &self.speech_rules.pref_manager.get_style_file()[0] { file_name= path.to_str().unwrap(); } - if self.rules.get("math").is_none() { + if rules.get("math").is_none() { bail!("No rules found for any matches!!! See the error log."); } else { // FIX: handle error appropriately - bail!("\nNo match found!\nMissing patterns in {} or bad MathML.\n{}", file_name, crate::pretty_print::mml_to_string(mathml)); + bail!("\nNo match found!\nMissing patterns in {} or bad MathML.\n{}", + file_name, crate::pretty_print::mml_to_string(&mathml)); } - fn error_string(pattern: &SpeechPattern, mathml: &Element) -> String { + fn error_string(pattern: &SpeechPattern, mathml: Element) -> String { return format!( "error during pattern match using: \"{}\" for \"{}\".\n\ Pattern is \n{}\nMathML for the match:\n\ @@ -1706,19 +1767,19 @@ impl SpeechRules { The patterns are in {}.\n", pattern.pattern_name, pattern.tag_name, pattern.pattern, - pretty_print::mml_to_string(mathml), + pretty_print::mml_to_string(&mathml), pattern.file_name ); } } - fn replace(&self, replacement: &Replacement, mathml: &Element) -> Result { + fn replace(&'r mut self, replacement: &Replacement, mathml: &'r Element<'c>) -> Result { return Ok( match &*replacement { Replacement::Text(t) => t.clone(), - Replacement::XPath(path) => path.replace(&self, mathml)?, + Replacement::XPath(path) => path.replace(self, mathml)?, Replacement::TTS(tts) => { - self.pref_manager.get_tts().replace(&tts, &self.pref_manager, self, mathml)? + self.speech_rules.pref_manager.get_tts().replace(&tts, &self.speech_rules.pref_manager, self, mathml)? }, Replacement::Test(test) => { test.replace(self, mathml)? @@ -1730,30 +1791,45 @@ impl SpeechRules { ) } - fn replace_nodes(&self, nodes: nodeset::Nodeset, mathml: &Element) -> Result { + fn replace_nodes(&'r mut self, nodes: nodeset::Nodeset<'c>, mathml: &'r Element<'c>) -> Result { //println!("replace_nodes: working on {} nodes", nodes.size()); - let result = nodes.document_order() - .iter() - .map(|node| - match node { - Node::Element(mathml) => self.match_pattern(&mathml), - Node::Text(t) => self.replace_chars(&t.text(), mathml), - _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); Ok( "".to_string() )} - }) - .collect::>>()? - .join(" "); + let mut result = String::with_capacity(3*nodes.size()); // guess (2 chars/node + space) + let mut not_first_time = false; + for node in nodes.document_order() { + if not_first_time { + result.push(' '); + not_first_time = true; + } + let matched = match node { + Node::Element(n) => self.match_pattern(&n)?, + Node::Text(t) => self.replace_chars(&t.text(), mathml)?, + _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); "".to_string() } + }; + result += &matched; + } + // let result = nodes.document_order() + // .iter() + // .map(|node: &'r Node<'c>| + // match node { + // Node::Element(n) => self.match_pattern(n), + // Node::Text(t) => self.replace_chars(&t.text(), mathml), + // _ => {eprintln!("replace_nodes: found unexpected node type!!! (ignored)"); Ok( "".to_string() )} + // }) + // .collect::>>()? + // .join(" "); return Ok( result ); } - pub fn replace_chars(&self, str: &str, mathml: & Element) -> Result { + fn replace_chars(&'r mut self, str: &str, mathml: &'r Element<'c>) -> Result { // Lookup unicode "pronunciation" of char // Note: TTS is not supported here (not needed and a little less efficient) + let rules = self.speech_rules; let mut chars = str.chars(); - if self.translate_single_chars_only { + if rules.translate_single_chars_only { let ch = chars.next().unwrap_or(' '); if chars.next().is_none() { // single char - return replace_single_char(&self, ch, mathml) + return replace_single_char(self, ch, mathml) } else { // more than one char (don't use str.len() since that is bytes, not chars) return Ok(String::from(str)); @@ -1761,14 +1837,14 @@ impl SpeechRules { }; let result = chars - .map(|ch| replace_single_char(&self, ch, mathml)) + .map(|ch| replace_single_char(self, ch, mathml)) .collect::>>()? .join(""); return Ok( result ); - fn replace_single_char(rules: &SpeechRules, ch: char, mathml: & Element) -> Result { + fn replace_single_char<'c, 's:'c, 'r>(rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, ch: char, mathml: &'r Element<'c>) -> Result { let ch_as_u32 = ch as u32; - let replacements = rules.unicode.get( &ch_as_u32 ); + let replacements = rules_with_context.speech_rules.unicode.get( &ch_as_u32 ); if replacements.is_none() { return Ok(String::from(ch)); // no replacement, so just return the char and hope for the best }; @@ -1778,7 +1854,7 @@ impl SpeechRules { replacements.unwrap() .iter() .map(|replacement| - rules.replace(replacement, mathml) + rules_with_context.replace(replacement, mathml) .chain_err(|| format!("Unicode replacement error: {}", replacement)) ) .collect::>>()? .join(" ") @@ -1787,6 +1863,16 @@ impl SpeechRules { } } +// Hack to allow replacement of `str` with braille chars. +pub fn braille_replace_chars(str: &str, mathml: Element) -> Result { + return BRAILLE_RULES.with(|rules| { + let rules = rules.borrow(); + let mut rules_with_context = SpeechRulesWithContext::new(&rules); + return rules_with_context.replace_chars(str, &mathml); + }) +} + + #[cfg(test)] mod tests { diff --git a/src/tts.rs b/src/tts.rs index d05984e53..b48d16623 100755 --- a/src/tts.rs +++ b/src/tts.rs @@ -65,7 +65,7 @@ use sxd_document::dom::Element; use yaml_rust::Yaml; use std::{fmt}; -use crate::speech::{SpeechRules}; +use crate::speech::{SpeechRulesWithContext}; use strum_macros::IntoStaticStr; use regex::Regex; @@ -221,7 +221,7 @@ impl TTS { /// A string is returned for the speech engine. /// /// `auto` pausing is handled at a later phase and a special char is used for it - pub fn replace(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules: &SpeechRules, mathml: &Element) -> Result { + pub fn replace<'c, 's:'c, 'r>(&self, command: &TTSCommandRule, prefs: &PreferenceManager, rules_with_context: &'r mut SpeechRulesWithContext<'c, 's>, mathml: &'r Element<'c>) -> Result { // The general idea is we handle the begin tag, the contents, and then the end tag // For the begin/end tag, we dispatch off to specialized code for each TTS engine let mut result = String::with_capacity(255); @@ -236,7 +236,7 @@ impl TTS { if result.is_empty() { result += " "; } - result += &command.replacements.replace(rules, mathml)?; + result += &command.replacements.replace(rules_with_context, mathml)?; } let end_tag = match self { diff --git a/src/xpath_functions.rs b/src/xpath_functions.rs index b35afa321..222d0b5ca 100755 --- a/src/xpath_functions.rs +++ b/src/xpath_functions.rs @@ -745,50 +745,43 @@ impl NemethChars { _ => "R", // normal and unknown }, }; - return crate::speech::BRAILLE_RULES.with(|rules| { - let text = as_text(*node); - // start of pattern matching mutably borrows BRAILLE_RULES, so we can't use borrow. - // instead hack to get around borrow rules because we know no changes happen during call - let braille_chars = unsafe { - rules.as_ptr().as_ref().unwrap() - .replace_chars(text, node).unwrap_or("".to_string()) - }; - println!("braille_chars: '{}'", braille_chars); - - // we want to pull the prefix (typeface, language) out to the front until a change happens - // the same is true for number indicator - // also true (sort of) for capitalization -- if all caps, use double cap in front (assume abbr or Roman Numeral) - let is_in_enclosed_list = name(node) == "mn" && NemethChars::is_in_enclosed_list(*node); - let mut typeface = "".to_string(); // illegal value to force first value - let mut is_all_caps = true; - let result = PICK_APART_CHAR.replace_all(&braille_chars, |caps: &Captures| { - println!(" face: {:?}, lang: {:?}, num {:?}, cap: {:?}, char: {:?}", - &caps["face"], &caps["lang"], &caps["num"], &caps["cap"], &caps["char"]); - let mut nemeth_chars = "".to_string(); - let typeface_changed = &typeface != &caps["face"]; - if typeface_changed { - typeface = caps["face"].to_string(); // needs to outlast this instance of the loop - nemeth_chars += if typeface.is_empty() {attr_typeface} else {&typeface}; - nemeth_chars += &caps["lang"]; - } else { - nemeth_chars += &caps["lang"]; - } - println!("is_in_list: {}; num: {}", is_in_enclosed_list, caps["num"].is_empty()); - if !caps["num"].is_empty() && (typeface_changed || !is_in_enclosed_list) { - nemeth_chars += "N"; - } - is_all_caps &= !&caps["cap"].is_empty(); - nemeth_chars += &caps["cap"]; // will be stripped later if all caps - nemeth_chars += &caps["char"]; - return nemeth_chars; - }); - let mut text_chars = text.chars(); // see if more than one char - if is_all_caps && text_chars.next().is_some() && text_chars.next().is_some() { - return Ok( "CC".to_string() + &result.replace("C", "")); + let text = as_text(*node); + let braille_chars = crate::speech::braille_replace_chars(text, *node).unwrap_or("".to_string()); + println!("braille_chars: '{}'", braille_chars); + + // we want to pull the prefix (typeface, language) out to the front until a change happens + // the same is true for number indicator + // also true (sort of) for capitalization -- if all caps, use double cap in front (assume abbr or Roman Numeral) + let is_in_enclosed_list = name(node) == "mn" && NemethChars::is_in_enclosed_list(*node); + let mut typeface = "".to_string(); // illegal value to force first value + let mut is_all_caps = true; + let result = PICK_APART_CHAR.replace_all(&braille_chars, |caps: &Captures| { + println!(" face: {:?}, lang: {:?}, num {:?}, cap: {:?}, char: {:?}", + &caps["face"], &caps["lang"], &caps["num"], &caps["cap"], &caps["char"]); + let mut nemeth_chars = "".to_string(); + let typeface_changed = &typeface != &caps["face"]; + if typeface_changed { + typeface = caps["face"].to_string(); // needs to outlast this instance of the loop + nemeth_chars += if typeface.is_empty() {attr_typeface} else {&typeface}; + nemeth_chars += &caps["lang"]; } else { - return Ok( result.to_string() ); + nemeth_chars += &caps["lang"]; + } + println!("is_in_list: {}; num: {}", is_in_enclosed_list, caps["num"].is_empty()); + if !caps["num"].is_empty() && (typeface_changed || !is_in_enclosed_list) { + nemeth_chars += "N"; } + is_all_caps &= !&caps["cap"].is_empty(); + nemeth_chars += &caps["cap"]; // will be stripped later if all caps + nemeth_chars += &caps["char"]; + return nemeth_chars; }); + let mut text_chars = text.chars(); // see if more than one char + if is_all_caps && text_chars.next().is_some() && text_chars.next().is_some() { + return Ok( "CC".to_string() + &result.replace("C", "")); + } else { + return Ok( result.to_string() ); + } } fn is_in_enclosed_list(node: Element) -> bool { diff --git a/tests/braille/Nemeth/AataNemeth.rs b/tests/braille/Nemeth/AataNemeth.rs index 53d552f20..fe34d446d 100755 --- a/tests/braille/Nemeth/AataNemeth.rs +++ b/tests/braille/Nemeth/AataNemeth.rs @@ -338,7 +338,8 @@ fn test_045() { #[test] fn test_046() { let expr = "(01000101)"; - test_braille("Nemeth", expr, "⠷⠼⠴⠂⠴⠴⠀⠼⠴⠂⠴⠂⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠴⠂⠴⠴⠀⠴⠂⠴⠂⠾"); } #[test] @@ -493,7 +494,8 @@ fn test_069() { #[test] fn test_070() { let expr = "(10011000)"; - test_braille("Nemeth", expr, "⠷⠼⠂⠴⠴⠂⠀⠼⠂⠴⠴⠴⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠂⠴⠴⠂⠀⠂⠴⠴⠴⠾"); } #[test] @@ -634,7 +636,8 @@ fn test_090() { #[test] fn test_091() { let expr = "(011011100110)"; - test_braille("Nemeth", expr, "⠷⠼⠴⠂⠂⠴⠀⠼⠂⠂⠂⠴⠀⠼⠴⠂⠂⠴⠾"); + // Corrected: no numeric indicators should be used after space as this is a single number; also none after paren + test_braille("Nemeth", expr, "⠷⠴⠂⠂⠴⠀⠂⠂⠂⠴⠀⠴⠂⠂⠴⠾"); } #[test] @@ -993,7 +996,8 @@ fn test_147() { #[test] fn test_148() { let expr = "(n,E)=(451,231)"; - test_braille("Nemeth", expr, "⠷⠝⠠⠀⠠⠑⠾⠀⠨⠅⠀⠷⠲⠢⠂⠠⠀⠆⠒⠂⠾"); + // correct: 8b(1) -- no space after the comma + test_braille("Nemeth", expr, "⠷⠝⠠⠀⠠⠑⠾⠀⠨⠅⠀⠷⠲⠢⠂⠠⠆⠒⠂⠾"); } #[test] @@ -2084,7 +2088,8 @@ fn test_307() { 3i]→ N∪{0}"; - test_braille("Nemeth", expr, "⠨⠝⠸⠒⠀⠈⠰⠠⠵⠈⠷⠜⠒⠻⠊⠈⠾⠀⠫⠕⠀⠈⠰⠠⠝⠨⠬⠨⠷⠴⠨⠾"); + // corrected: removed extra space after "⠸⠒" + test_braille("Nemeth", expr, "⠨⠝⠸⠒⠈⠰⠠⠵⠈⠷⠜⠒⠻⠊⠈⠾⠀⠫⠕⠀⠈⠰⠠⠝⠨⠬⠨⠷⠴⠨⠾"); } #[test] diff --git a/tests/braille/Nemeth/rules.rs b/tests/braille/Nemeth/rules.rs index ffe0cdba3..218e67be0 100755 --- a/tests/braille/Nemeth/rules.rs +++ b/tests/braille/Nemeth/rules.rs @@ -184,8 +184,8 @@ fn comma_in_sup_79_b_4() { #[test] fn text_after_sup_79_c_3() { - // bad mn from Wiris - let expr = "6.696×108mph"; + // bad mn from Wiris; also &ao; + let expr = "6.696×108 mph"; test_braille("Nemeth", expr, "⠼⠖⠨⠖⠔⠖⠈⠡⠂⠴⠘⠦⠀⠍⠏⠓"); }