Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 17 additions & 11 deletions .vscode/settings.json
Original file line number Diff line number Diff line change
Expand Up @@ -2,18 +2,11 @@
"cSpell.words": [
"ABEFHILM",
"ACHIRZ",
"CHNPQRZ",
"Coord",
"MATHML",
"ML's",
"Mult",
"Nemeth",
"PUNCT",
"Regow",
"airity",
"axisname",
"backtrace",
"badvalue",
"ℬℰℱℋℐℒℳ",
"bigop",
"bitand",
"bitflags",
Expand All @@ -22,17 +15,21 @@
"canonicalize",
"canonicalizing",
"chenyh",
"CHNPQRZ",
"clear",
"clippy",
"codegen",
"coef",
"columnspacing",
"concat",
"Coord",
"cosech",
"cotanh",
"displaystyle",
"downarrow",
"downdiagonalstrike",
"EDGVHR",
"EDGVHU",
"envlogger",
"eval",
"fullwidth",
Expand All @@ -49,12 +46,14 @@
"longdiv",
"madrub",
"mathcat",
"MATHML",
"mathsize",
"mathvariant",
"member",
"mfenced",
"mfrac",
"mglyph",
"ML's",
"mlabeledtr",
"mlongdiv",
"mmultiscripts",
Expand All @@ -77,10 +76,12 @@
"msupsup",
"mtable",
"mtext",
"Mult",
"munder",
"munderover",
"nbsp",
"neils",
"Nemeth",
"nodeset",
"nodetest",
"northeastarrow",
Expand All @@ -96,18 +97,25 @@
"phasor",
"phasorangle",
"prefs",
"PUNCT",
"pyfunction",
"pymodule",
"qname",
"rchunks",
"ℛℯℊℴ",
"Regow",
"repr",
"rfind",
"rightarrow",
"roundedbox",
"rowspacing",
"rustc",
"sapi",
"SBTIR",
"SBTIREDGVHP",
"SBTIREDGVHPC",
"scriptlevel",
"SEMIDIRECT",
"set",
"sheck",
"sinch",
Expand Down Expand Up @@ -135,9 +143,7 @@
"xlviii",
"xpath's",
"Ωαωϝ",
"Ωαωϰϕϱϖ",
"ℛℯℊℴ",
"ℬℰℱℋℐℒℳ"
"Ωαωϰϕϱϖ"
],
"cSpell.ignoreWords": [
"eigh",
Expand Down
94 changes: 89 additions & 5 deletions PythonScripts/alphanumerics.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,9 +2,9 @@
from bs4 import BeautifulSoup

def create_unicode_from_html(in_file: str, out_file):
with open(in_file, encoding='utf8') as _in_stream:
with open(in_file, encoding='utf8') as in_stream:
with open(out_file, 'a', encoding='utf8') as out_stream:
file_contents = BeautifulSoup(_in_stream, features="html.parser")
file_contents = BeautifulSoup(in_stream, features="html.parser")
for row in file_contents.find_all('tr'):
cols = row.find_all('td')
if len(cols) > 0:
Expand All @@ -25,6 +25,80 @@ def generate_digits(out_file, prefix: str, hex: str, description: str):
i += 1



# get the JSON versions -- the HTML versions seem wrong in many places (e.g., bold or numbers)
import json
def create_unicode_from_json(in_file: str, out_file):
with open(in_file, encoding='utf8') as in_stream:
with open(out_file, 'a', encoding='utf8') as out_stream:
file_contents = json.load(in_stream)
info = file_contents["information"].replace("tests", "chars")
out_stream.write('\n # --- {} ---\n'.format(info))
for char, braille in file_contents['tests'].items():
if braille["expected"] != "":
generate_char_line(out_stream, char, braille["expected"])

import re
MATCH_FACE_LANG_CHAR = re.compile(
("("
"(?P<prefix>^)"
"(?P<sans>⠠⠨)?"
"(?P<bold>⠸(?=[^⠠⠔⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵⠯⠫⠱⠹]))?"
"(?P<script>⠈(?=[^⠠⠔⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵⠯⠫⠱⠹]))?"
"(?P<italic>⠨(?=[^⠠⠔⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵⠯⠫⠱⠹]))?"
"((((?P<en>⠰)|(?P<de>⠸)|(?P<el>⠨)|(?P<el_var>⠨⠈))?"
"(?P<cap>⠠)?(?P<char>[⠁⠃⠉⠙⠑⠋⠛⠓⠊⠚⠅⠇⠍⠝⠕⠏⠟⠗⠎⠞⠥⠧⠺⠭⠽⠵⠯⠫⠱⠹]))|"
"((?P<num>⠼?)(?P<digit>[⠴⠂⠆⠒⠲⠢⠖⠶⠦⠔])))"
"(?P<postfix>$)"
")")
)

def generate_char_line(out_stream, char: str, braille: str):
# format to generate
# - "⬟": [t: "⠫⠸⠢"] # 0x2B1F
# escape quotes and backslashes
if (char == '"' or char == '\\'):
char = "\\" + char

# pick the braille apart, looking for one or more typefaces, alphabet, (cap? letter) | (numeric? digit) | char

result = ""
matched = MATCH_FACE_LANG_CHAR.match(braille)
if matched:
dict = matched.groupdict()
if dict["prefix"]:
result += dict["prefix"]
if dict["sans"]:
result += "S"
if dict["bold"]:
result += "B"
if dict["script"]:
result += "T"
if dict["italic"]:
result += "I"
if dict["en"]:
result += "E"
if dict["de"]:
result += "D"
if dict["el"]:
result += "G"
if dict["el_var"]:
result += "V"
if dict["char"]:
if dict["cap"]:
result += "C"
result += dict["char"]
if dict["digit"]:
result += "N" + dict["digit"]
if dict["postfix"]:
result += dict["postfix"]
else:
result = braille
first_part = ' - "{}": [t: "{}"]'.format(char, result)
out_stream.write('{:32}# {}\n'.format(first_part, hex(ord(char[-1])) ))



import os
# if os.path.exists("alphanumerics.txt"):
# os.remove("alphanumerics.txt")
Expand All @@ -34,6 +108,16 @@ def generate_digits(out_file, prefix: str, hex: str, description: str):
if os.path.exists("out"):
os.remove("out")

generate_digits("out", "⠸", "1D7CE", "bold")
generate_digits("out", "⠈", "1D7D8", "double struck (as script)")
generate_digits("out", "⠠", "1D7E2", "sans-serif")
path = "C:/Dev/speech-rule-engine/sre-tests/expected/nemeth/symbols/"
# file = "default_alphabet_bold.json"
if os.path.exists("alphanumerics.txt"):
os.remove("alphanumerics.txt")
for filename in os.listdir(path):
# these have multi char entries and aren't character definitions
if not(filename == 'default_functions.json' or filename == 'default_si_units.json' or filename == 'default_units.json'):
create_unicode_from_json(path+filename, "alphanumerics.txt")

# with open("out", 'a', encoding='utf8') as out_stream:
# generate_char_line(out_stream, "𝐅", "⠸⠰⠠⠋")
# generate_digits("out", "⠈", "1D7D8", "double struck (as script)")
# generate_digits("out", "⠠", "1D7E2", "sans-serif")
Loading