mirror of
https://github.com/lhns/steam-frame-nix.git
synced 2026-10-06 07:00:27 +02:00
Opt-in with steamFrame.keyboard.vr.enable; swipe, autocorrect and completion suggestions, Backspace drag and haptics are on under it. - vr-keyboard/patch.js (Steam UI): swipe decoder, a model of what the keyboard typed, suggestions that only replace what they typed, Backspace drag with word detents and retyping; checks its Steam internals by signature (signatures.json: vr-keyboard) and stays stock otherwise. - vr-keyboard/panel.js (SteamVR systemui) + relay.mjs (user service vr-keyboard-relay): the suggestion strip as a dashboard panel below or above the keyboard (signatures: vr-keyboard-panel). - Dictionary built from nixpkgs' wordfreq and hunspellDicts; default: the keyboard.layout language (de, fr, es, it, nl, pt, sv) plus English. - Tests: keyboard.vr.checks and the flake check vr-keyboard.
181 lines
8.2 KiB
Python
181 lines
8.2 KiB
Python
# Builds the dictionary at build time: the most frequent words per language
|
|
# (wordfreq), kept if Hunspell accepts them, in Hunspell's spelling and casing
|
|
# (wordfreq lowercases and folds ß: "strasse" is tried as "straße"; lowercase
|
|
# only if accepted without compounding, else Capitalised; "essen"/"Essen"
|
|
# both, the second slightly less frequent). Contractions Hunspell rejects
|
|
# ("geht's") are kept by frequency, other rejected words if their zipf is
|
|
# >= keepFrequentAbove (3+ letters, or zipf >= 5: "ok").
|
|
# usage: python3 gen-dict.py <out.js> <hunspell> <config.json>
|
|
# config: { languages: [{ lang, words, offset (zipf), dict (path or null),
|
|
# keepFrequentAbove (zipf or null) }],
|
|
# extra: [[word, zipf]], extraFiles: [path], extraZipf, exclude: [word],
|
|
# contractions: bool }
|
|
# Output: a JS string literal "word<TAB>zipf*10\n..." (most frequent first).
|
|
import itertools, json, re, shutil, subprocess, sys
|
|
import wordfreq
|
|
|
|
out, hunspell, config = sys.argv[1:]
|
|
cfg = json.load(open(config))
|
|
# Word letters: German/English a-z ä ö ü ß; other languages any lowercase
|
|
# letter (accents). Contractions ("couldn't", "geht's"): letters around
|
|
# apostrophes (wordfreq splits words at hyphens: no hyphenated words).
|
|
def word_patterns(lang):
|
|
L = "a-zäöüß" if lang in ("de", "en") else r"^\W\d_A-Z"
|
|
return re.compile(rf"^[{L}]{{2,24}}$"), re.compile(rf"^[{L}]+(?:'[{L}]+)+$")
|
|
SECONDARY = 5 # frequency penalty (zipf/10) of the other casing
|
|
cap = lambda w: w[0].upper() + w[1:]
|
|
|
|
|
|
def spellings(w):
|
|
"""w, then the variants with some "ss" written as "ß" (wordfreq folds ß)."""
|
|
idx = [m.start() for m in re.finditer("ss", w)][:3]
|
|
out = [w]
|
|
for r in range(1, len(idx) + 1):
|
|
for combo in itertools.combinations(idx, r):
|
|
s = w
|
|
for i in sorted(combo, reverse=True):
|
|
s = s[:i] + "ß" + s[i + 2:]
|
|
out.append(s)
|
|
return out
|
|
|
|
|
|
def check(dic, words):
|
|
"""The words Hunspell accepts (-G) with dictionary dic."""
|
|
res = subprocess.run([hunspell, "-i", "utf-8", "-d", dic, "-G"],
|
|
input="\n".join(words) + "\n", capture_output=True, text=True, check=True)
|
|
return set(res.stdout.split("\n"))
|
|
|
|
|
|
def strict_dict(dic, tmp):
|
|
"""dic without compounding: German .dic files also list nouns in lowercase
|
|
as compound parts ("wetter" for "Regenwetter"), which Hunspell then accepts
|
|
on their own; the strict copy decides the casing. Also returns the
|
|
capitalised stems (nouns, names)."""
|
|
aff = open(dic + ".aff", encoding="latin-1").read()
|
|
aff = "\n".join(l for l in aff.split("\n") if not l.startswith("COMPOUND"))
|
|
open(tmp + ".aff", "w", encoding="latin-1").write(aff) # latin-1: bytes unchanged
|
|
shutil.copyfile(dic + ".dic", tmp + ".dic")
|
|
enc = re.search(r"^SET\s+(\S+)", aff, re.M)
|
|
text = open(dic + ".dic", "rb").read().decode(enc.group(1) if enc else "latin-1", "replace")
|
|
caps = {l.split("/")[0].strip() for l in text.split("\n")[1:] if l[:1].isupper()}
|
|
return tmp, caps
|
|
|
|
|
|
best = {}
|
|
frequent_all = set() # kept by frequency only (keepFrequentAbove)
|
|
verified = set() # accepted by some language's Hunspell
|
|
unverified = set() # contractions Hunspell rejects ("geht's"), kept by frequency
|
|
def add(form, z):
|
|
if z > best.get(form, -1):
|
|
best[form] = z
|
|
|
|
contractions = cfg.get("contractions", True)
|
|
def top_words(lang, n):
|
|
"""The n most frequent plain words, plus the contractions among them."""
|
|
LETTERS, CONTRACTION = word_patterns(lang)
|
|
out, plain = [], 0
|
|
for w in wordfreq.iter_wordlist(lang):
|
|
w = w.replace("\u2019", "'")
|
|
if LETTERS.match(w):
|
|
out.append(w); plain += 1
|
|
if plain >= n:
|
|
break
|
|
elif contractions and CONTRACTION.match(w) and len(w) <= 24:
|
|
out.append(w)
|
|
return out
|
|
|
|
available = set(wordfreq.available_languages())
|
|
for spec in cfg["languages"]:
|
|
lang, n, dic = spec["lang"], int(spec["words"]), spec.get("dict")
|
|
keep_above = spec.get("keepFrequentAbove")
|
|
frequent = []
|
|
bias = round(float(spec.get("offset", 0)) * 10)
|
|
if lang not in available:
|
|
sys.exit(f"wordfreq has no language {lang!r}; available: {' '.join(sorted(available))}")
|
|
words = top_words(lang, n)
|
|
if not dic:
|
|
for w in words:
|
|
add(w, round(wordfreq.zipf_frequency(w, lang) * 10) + bias)
|
|
print(f"{lang}: {len(words)} words (no Hunspell filter)", file=sys.stderr)
|
|
continue
|
|
variants = {w: spellings(w) for w in words}
|
|
query = [f for vs in variants.values() for v in vs for f in (v, cap(v))]
|
|
ok = check(dic, query)
|
|
sdic, caps = strict_dict(dic, f"strict-{lang}")
|
|
strict = check(sdic, [f for f in query if f in ok])
|
|
kept = 0
|
|
for w in words:
|
|
z = round(wordfreq.zipf_frequency(w, lang) * 10) + bias
|
|
if not any(v in ok or cap(v) in ok for v in variants[w]):
|
|
zipf = wordfreq.zipf_frequency(w, lang)
|
|
if "'" in w: # Hunspell splits "geht's": keep by frequency
|
|
add(w, z)
|
|
unverified.add(w)
|
|
kept += 1
|
|
elif keep_above is not None and zipf >= keep_above and (len(w) >= 3 or zipf >= 5):
|
|
add(w, z) # frequent, but not in Hunspell ("ok", "lol", "colour")
|
|
frequent.append(w)
|
|
frequent_all.add(w)
|
|
continue
|
|
for v in variants[w]:
|
|
lo, up = v in ok, cap(v) in ok
|
|
if not (lo or up):
|
|
continue
|
|
kept += 1
|
|
verified.update(f for f in (v, cap(v)) if f in ok)
|
|
slo, sup = v in strict, cap(v) in strict
|
|
if slo:
|
|
add(v, z)
|
|
if cap(v) in caps: # "essen" / "Essen"
|
|
add(cap(v), z - SECONDARY)
|
|
elif sup or not lo: # "Wetter", "Haus", "Weihnachtsmarkt"
|
|
add(cap(v), z)
|
|
else:
|
|
add(v, z)
|
|
break
|
|
print(f"{lang}: {kept} of {len(words)} words kept, {len(frequent)} more by frequency: {' '.join(frequent[:30])}", file=sys.stderr)
|
|
|
|
# An unverified contraction that another language has in a checked casing
|
|
# ("i'm" from the German list, "I'm" from English): keep only the latter.
|
|
checked = {}
|
|
for w in best:
|
|
if "'" in w and w not in unverified:
|
|
checked.setdefault(w.lower(), w)
|
|
for w in [w for w in unverified if w in best and w.lower() in checked and checked[w.lower()] != w]:
|
|
k = checked[w.lower()]
|
|
best[k] = max(best[k], best.pop(w))
|
|
print(f"contractions: {sum(1 for w in best if chr(39) in w)} ({len(unverified)} not in Hunspell)", file=sys.stderr)
|
|
|
|
# A frequency-only word that is a contraction without its apostrophe ("dont",
|
|
# "thats") would only compete with the real one: dropped.
|
|
# Only words no Hunspell accepts ("is" stays although "i's" exists), and only
|
|
# if the contraction is more frequent.
|
|
contraction_freq = {}
|
|
for w, z in best.items():
|
|
if "'" in w:
|
|
k = w.replace("'", "").lower()
|
|
contraction_freq[k] = max(contraction_freq.get(k, -1), z)
|
|
dropped = [w for w in frequent_all if w in best and w not in verified and best[w] < contraction_freq.get(w, -1)]
|
|
for w in dropped:
|
|
del best[w]
|
|
print(f"frequency-only words dropped as contractions without apostrophe: {len(dropped)}", file=sys.stderr)
|
|
|
|
extra = [(w, z) for w, z in cfg.get("extra", [])]
|
|
for path in cfg.get("extraFiles", []):
|
|
for line in open(path, encoding="utf-8"):
|
|
parts = line.strip().split("\t")
|
|
if parts[0]:
|
|
extra.append((parts[0], float(parts[1]) if len(parts) > 1 else None))
|
|
for w, z in extra:
|
|
z = cfg.get("extraZipf", 5.0) if z is None else z
|
|
best[w] = max(best.get(w, -1), round(float(z) * 10))
|
|
exclude = {w.lower() for w in cfg.get("exclude", [])}
|
|
for w in [w for w in best if w.lower() in exclude]:
|
|
del best[w]
|
|
print(f"extra: {len(extra)}, excluded: {len(exclude)}", file=sys.stderr)
|
|
|
|
items = sorted(best.items(), key=lambda kv: (-kv[1], kv[0]))
|
|
with open(out, "w", encoding="utf-8") as f:
|
|
f.write(json.dumps("\n".join(f"{w}\t{z}" for w, z in items), ensure_ascii=False))
|
|
print(f"total: {len(items)} entries", file=sys.stderr)
|