Source code for tetrak_translit.fold
"""The canonical fold: one index form for every spelling of a word.
This is the function a search index calls, at index time and at query
time, so that a record catalogued as ``Պօղոսեան`` is found by a reader who
types *Poghosyan*, *Boghossian*, *Pōghosean* or ``Պողոսյան``, and one
catalogued as ``ჭავჭავაძე`` by *Chavchavadze*, *Čavčavaje* or
*Ch'avch'avadze*. It is for **finding**, not display: the key it returns
is a deliberately coarse Latin skeleton that nobody should ever show to a
reader. The schemes are for display; this is the other half of the job,
and the two must not be confused.
Each script owns its rules -- :mod:`tetrak_translit.hy.fold` and
:mod:`tetrak_translit.ka.fold` -- because what should be merged differs
between scripts. This module only decides which rules a token gets: a
token in a registered script gets that script's; a Latin token gets the
rules of the script named by ``script=``, because *Tbilisi* folded under
Armenian rules would never find ``თბილისი``.
"""
from __future__ import annotations
import re
import unicodedata
from . import engine
from .registry import ScriptPackage, get_scheme, get_script, scripts_in
from .script import Scheme, Script
_TOKEN = re.compile(r"\S+")
class ScriptRequiredError(ValueError):
"""Raised when Latin text is folded with no script to fold it for."""
def _package_for(token: str, default: ScriptPackage | None) -> ScriptPackage:
present = scripts_in(token)
if len(present) == 1:
return present[0]
if len(present) > 1:
# One token in two scripts is a typo or a homoglyph; the first
# registered script wins, and the key is still deterministic.
return present[0]
if default is None:
raise ScriptRequiredError(
"fold() needs to know which script's rules to apply to Latin text: "
"pass script='hy' (Armenian) or script='ka' (Georgian)"
)
return default
def fold_token(token: str, script: str | Script | None = None) -> str:
"""Fold one whitespace-free token. Empty when nothing survives."""
default = get_script(script) if script is not None else None
return _package_for(token, default).fold_token(token)
[docs]
def fold(
text: str, *, script: str | Script | None = None, scheme: str | Scheme | None = None
) -> str:
"""Fold *text* onto its canonical index form.
Args:
text: Text in a registered script, Latin in any of its
romanisations, or a mixture. Tokens are folded one at a time
and joined with single spaces.
script: Which script's rules Latin tokens get: a key (``"hy"``,
``"ka"``) or name. Tokens written in a registered script are
recognised regardless. Required if any token is Latin.
scheme: Optional. When the Latin in *text* is known to be written
in a reversible scheme (``"iso_9985"``, ``"hubschmann_meillet"``,
``"ka/iso_9984"``), name it and the text is read back into
the script exactly before folding, which removes the
ambiguities the heuristic fold has (Armenian ``j``). For a
lossy scheme the heuristic fold is used. Either way the
scheme's script becomes the default for Latin tokens.
Returns:
A lowercase ASCII key: for comparing and indexing, never for
display.
Raises:
ScriptRequiredError: a Latin token and no *script*.
UnknownScriptError, UnknownSchemeError: the names resolve to nothing.
"""
text = unicodedata.normalize("NFC", text)
default = get_script(script) if script is not None else None
if scheme is not None:
table = get_scheme(scheme, default.script if default else None)
if table.reversible and not table.script.has_letters(text):
text = engine.read(text, table)
default = get_script(table.script)
keys = (
_package_for(match.group(), default).fold_token(match.group())
for match in _TOKEN.finditer(text)
)
return " ".join(key for key in keys if key)
__all__ = ["ScriptRequiredError", "fold", "fold_token"]