Files
strans/map/mkemoji
Hojun-Cho fcdce7dc16 data(emoji): every Japanese alias in both kana
A query typed in the Hiragana mode is hiragana; CLDR's Japanese
keywords are mostly katakana (スマイル, ハート), so the one could not
find the other.  mkemoji now writes each alias that has kana in both
scripts.
2026-08-17 01:44:29 +09:00

85 lines
2.6 KiB
Python
Executable File

#!/usr/bin/env python3
"""Write emoji.dict to stdout from result-first UTF-8 TSV files."""
import sys
import unicodedata
from pathlib import Path
SOURCES = [Path(__file__).with_name(name)
for name in ("symbol.src", "emoji.src")]
MAXRUNES = 64
MAXCANDIDATES = 128
HIRA = {c: c + 0x60 for c in range(0x3041, 0x3097)}
KATA = {c: c - 0x60 for c in range(0x30A1, 0x30F7)}
def fold(s):
s = "".join(chr(ord(c) + 32) if "A" <= c <= "Z" else c for c in s)
return unicodedata.normalize("NFC", s)
def kana(alias):
"""The alias, and in the other kana where it has any: a query typed
in either Japanese mode finds it."""
return {alias, alias.translate(HIRA), alias.translate(KATA)}
def hascontrol(s):
return any(unicodedata.category(c) == "Cc" for c in s)
def read(path):
entries = []
with path.open(encoding="utf-8") as src:
for lineno, raw in enumerate(src, 1):
line = raw.rstrip("\r\n")
if not line.strip():
continue
if "\t" not in line and line.lstrip().startswith("#"):
continue
fields = line.split("\t")
if len(fields) < 2:
raise ValueError(f"{path}:{lineno}: need a result and an alias")
result = unicodedata.normalize("NFC", fields[0])
if (not result or len(result) > MAXRUNES or hascontrol(result)
or any(c.isspace() for c in result)):
raise ValueError(f"{path}:{lineno}: bad result")
for field in fields[1:]:
alias = fold(field)
if (not alias or len(alias) > MAXRUNES or hascontrol(alias)
or alias != alias.strip() or alias.startswith(";")):
raise ValueError(f"{path}:{lineno}: bad alias")
entries.extend((result, a) for a in sorted(kana(alias)))
return entries
def build(entries):
"""One row per alias, in order of first appearance; the engine searches
the dictionary by prefix, so no prefix rows are needed."""
table = {}
for result, alias in entries:
values = table.setdefault(alias, [])
if result not in values:
values.append(result)
for alias, values in table.items():
yield f"{alias}\t{' '.join(values[:MAXCANDIDATES])}"
def main():
paths = [Path(arg) for arg in sys.argv[1:]] or SOURCES
try:
entries = [entry for path in paths for entry in read(path)]
for line in build(entries):
print(line)
except (OSError, UnicodeError, ValueError) as error:
print(error, file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
sys.exit(main())