Source code for dacy.resources.dictionaries
"""Helper functions for loading Danish dictionaries."""
import csv
import zipfile
from pathlib import Path
import pandas as pd
from ..download import DEFAULT_CACHE_DIR, download_url
from .constants import RESOURCES
[docs]def load_ods_fullforms(redownload: bool = False) -> pd.DataFrame:
"""Loads the full-form list from ODS, a historical Danish dictionary.
`Ordbog over det danske Sprog (ODS)
<https://korpus.dsl.dk/resources/details/ods-fullforms.html>`_ covers Danish
from approximately 1700 to 1950. The August 2020 full-form list contains
about 1.3 million entries, including inflected forms. Original spelling
and capitalization are preserved.
`DSL's terms of use <https://korpus.dsl.dk/resources/licences/dsl-open.html>`_
allow reuse, modification and redistribution, including commercial use,
but prohibit publishing a dictionary or a product competing with DSL's
products. Attribution to DSL is requested. Downloading accepts these terms.
Args:
redownload: Download again even if the file is cached. Defaults to False.
Returns:
A DataFrame with five string columns.
- form: The word form, including inflected forms.
- headword: The dictionary headword associated with the form.
- homograph: The number distinguishing dictionary entries with the same
headword spelling; an empty string when absent.
- pos: The part of speech, using ODS abbreviations.
- id: The identifier of the dictionary entry in ODS.
Multiple entries for the same form are retained.
Example:
Look up the inflected form "Hesten" ("the horse") and its headword
"Hest" ("horse"):
>>> from dacy.resources import load_ods_fullforms
>>> entries = load_ods_fullforms()
>>> print(entries.loc[entries["form"] == "Hesten"].to_string(index=False))
form headword homograph pos id
Hesten Hest sb. 60137181
>>> wordforms = set(entries["form"])
"""
save_path = Path(DEFAULT_CACHE_DIR) / "resources" / "ods"
dl_path = save_path / "ods-fullform.zip"
if redownload or not dl_path.exists():
save_path.mkdir(parents=True, exist_ok=True)
download_url(RESOURCES["ods"], dl_path)
with zipfile.ZipFile(dl_path) as archive:
filename = next(
name
for name in archive.namelist()
if name.startswith("ods_fullforms_") and name.endswith(".csv")
)
with archive.open(filename) as source:
return pd.read_csv(
source,
sep="\t",
names=["form", "headword", "homograph", "pos", "id"],
dtype=str,
keep_default_na=False,
quoting=csv.QUOTE_NONE,
encoding="utf-8",
)