Skip to content

Languages

rigour.langs

Language code handling

This library helps to normalise the ISO 639 codes used to describe languages from two-letter codes to three letters, and vice versa.

import rigour.langs as languagecodes

assert 'eng' == languagecodes.iso_639_alpha3('en')
assert 'eng' == languagecodes.iso_639_alpha3('ENG ')
assert 'en' == languagecodes.iso_639_alpha2('ENG ')

Uses data from: https://iso639-3.sil.org/ See also: https://www.loc.gov/standards/iso639-2/php/code_list.php

LangStr

Bases: str

A string carrying an optional language tag.

Use this to keep track of which language a piece of multilingual content is written in while still passing it around as a str.

The language tag is part of the value's identity:

  • With no tag (lang is None), a LangStr is indistinguishable from its content string: it compares equal to the plain str, hashes the same, and deduplicates against it in sets and dict keys.
  • With a tag, a LangStr is a distinct value identified by the pair (content, lang). It is not equal to the bare content string, nor to a LangStr with a different or missing tag, and it never deduplicates against them.

Ordinary str methods (.upper(), slicing, concatenation, …) return plain str and drop the tag.

Parameters:

Name Type Description Default
content str

The text.

required
lang str | None

An ISO 639-3 language code, or None for untagged text.

None

Raises:

Type Description
ValueError

If lang is not a known ISO 639-3 code.

Source code in rigour/langs/text.py
class LangStr(str):
    """A string carrying an optional language tag.

    Use this to keep track of which language a piece of multilingual
    content is written in while still passing it around as a `str`.

    The language tag is part of the value's identity:

    * With no tag (`lang is None`), a `LangStr` is indistinguishable from
      its content string: it compares equal to the plain `str`, hashes the
      same, and deduplicates against it in sets and dict keys.
    * With a tag, a `LangStr` is a distinct value identified by the pair
      `(content, lang)`. It is not equal to the bare content string, nor
      to a `LangStr` with a different or missing tag, and it never
      deduplicates against them.

    Ordinary `str` methods (`.upper()`, slicing, concatenation, …) return
    plain `str` and drop the tag.

    Args:
        content: The text.
        lang: An ISO 639-3 language code, or `None` for untagged text.

    Raises:
        ValueError: If `lang` is not a known ISO 639-3 code.
    """

    __slots__ = ("lang",)

    def __new__(
        cls: type[LangStrT], content: str, lang: str | None = None
    ) -> LangStrT:
        return str.__new__(cls, content)

    def __init__(self, content: str, lang: str | None = None) -> None:
        if lang is not None and lang not in ISO3_ALL:
            raise ValueError(f"Invalid ISO 639-3 language code: {lang}")
        self.lang = lang

    def __repr__(self) -> str:
        if self.lang is not None:
            return f'"{super().__str__()}"@{self.lang}'
        return super().__repr__()

    def __hash__(self) -> int:
        if self.lang is None:
            return super().__hash__()
        return hash((super().__str__(), self.lang))

    def __eq__(self, value: object) -> bool:
        if not isinstance(value, str):
            return NotImplemented
        other_lang = value.lang if isinstance(value, LangStr) else None
        return super().__eq__(value) and self.lang == other_lang

    def __ne__(self, value: object) -> bool:
        equal = self.__eq__(value)
        if equal is NotImplemented:
            return NotImplemented
        return not equal

is_lang_better(candidate, baseline)

Decide if the candidate language code is 'better' than the baseline language code, according to the preferred languages list.

is_lang_better('eng', 'deu') True is_lang_better('fra', 'eng') False

Parameters:

Name Type Description Default
candidate str

The candidate language code.

required
baseline str

The baseline language code.

required

Returns:

Name Type Description
bool bool

True if the candidate is better than the baseline.

Source code in rigour/langs/__init__.py
def is_lang_better(candidate: str, baseline: str) -> bool:
    """Decide if the candidate language code is 'better' than the baseline
    language code, according to the preferred languages list.

     >>> is_lang_better('eng', 'deu')
     True
     >>> is_lang_better('fra', 'eng')
     False

    Args:
        candidate (str): The candidate language code.
        baseline (str): The baseline language code.

    Returns:
        bool: True if the candidate is better than the baseline.
    """
    try:
        candidate_index = PREFERRED_LANGS.index(candidate)
    except ValueError:
        candidate_index = len(PREFERRED_LANGS) + 1
    try:
        baseline_index = PREFERRED_LANGS.index(baseline)
    except ValueError:
        baseline_index = len(PREFERRED_LANGS) + 1
    return candidate_index < baseline_index

iso_639_alpha2(code)

Convert a language identifier to an ISO 639 Part 1 code, such as "en" or "de". For languages which do not have a two-letter identifier, or invalid language codes, None will be returned.

Source code in rigour/langs/__init__.py
def iso_639_alpha2(code: str) -> str | None:
    """Convert a language identifier to an ISO 639 Part 1 code, such as "en"
    or "de". For languages which do not have a two-letter identifier, or
    invalid language codes, ``None`` will be returned.
    """
    alpha3 = iso_639_alpha3(code)
    if alpha3 is None:
        return None
    return ISO2_MAP.get(alpha3)

iso_639_alpha3(code)

Convert a given language identifier into an ISO 639 Part 2 code, such as "eng" or "deu". This will accept language codes in the two- or three- letter format, some language names, and IETF/BCP 47-style tags with a script or region subtag (e.g. "zh-Hans", "pt-BR"), which resolve to the code of their primary subtag. If the given string cannot be converted, None will be returned.

iso_639_alpha3('en') 'eng'

Source code in rigour/langs/__init__.py
def iso_639_alpha3(code: str) -> str | None:
    """Convert a given language identifier into an ISO 639 Part 2 code, such
    as "eng" or "deu". This will accept language codes in the two- or three-
    letter format, some language names, and IETF/BCP 47-style tags with a
    script or region subtag (e.g. "zh-Hans", "pt-BR"), which resolve to the
    code of their primary subtag. If the given string cannot be converted,
    ``None`` will be returned.

    >>> iso_639_alpha3('en')
    'eng'
    """
    norm = normalize_code(code)
    if norm is None:
        return None
    resolved = ISO3_MAP.get(norm, norm)
    if resolved in ISO3_ALL and resolved not in NON_LANGS:
        return resolved
    # The full tag takes precedence (the synonym table has entries like
    # "chi_sim"); only then fall back to the primary subtag.
    for sep in ("-", "_"):
        if sep in norm:
            return iso_639_alpha3(norm.split(sep, 1)[0])
    return None

list_to_alpha3(languages, synonyms=True)

Parse all the language codes in a given list into ISO 639 Part 2 codes and optionally expand them with synonyms (i.e. other names for the same language).

Synonym groups mix in ISO 639-2/B and Tesseract-style codes (e.g. ger, chi) which aid input matching but are not valid ISO 639-3 identifiers; they are filtered out so the returned set only contains canonical codes.

Source code in rigour/langs/__init__.py
def list_to_alpha3(languages: Iterable[str], synonyms: bool = True) -> set[str]:
    """Parse all the language codes in a given list into ISO 639 Part 2 codes
    and optionally expand them with synonyms (i.e. other names for the same
    language).

    Synonym groups mix in ISO 639-2/B and Tesseract-style codes (e.g. ``ger``,
    ``chi``) which aid input matching but are not valid ISO 639-3 identifiers;
    they are filtered out so the returned set only contains canonical codes."""
    codes: set[str] = set()
    for language in languages:
        code = iso_639_alpha3(language)
        if code is None:
            continue
        codes.add(code)
        if synonyms:
            for synonym in expand_synonyms(code):
                if synonym in ISO3_ALL and synonym not in NON_LANGS:
                    codes.add(synonym)
    return codes