Coverage for paperless/utils.py: 12%
32 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 09:07 +0000
1import logging
3from dateparser.languages.loader import LocaleDataLoader
5logger = logging.getLogger("paperless.utils")
7OCR_TO_DATEPARSER_LANGUAGES = {
8 """
9 Translation map from languages supported by Tesseract OCR
10 to languages supported by dateparser.
11 To add a language, make sure it is supported by both libraries.
12 The ISO 639-2 will help you link a 3-char to 2-char language code.
13 Links:
14 - Tesseract languages: https://tesseract-ocr.github.io/tessdoc/Data-Files-in-different-versions.html
15 - Python dateparser languages: https://dateparser.readthedocs.io/en/latest/supported_locales.html
16 - ISO 639-2: https://www.loc.gov/standards/iso639-2/php/code_list.php
17 """
18 # TODO check these Dateparser languages as they are not referenced on the ISO639-2 standard,
19 # so we didn't find the equivalent in Tesseract:
20 # agq, asa, bez, brx, cgg, ckb, dav, dje, dyo, ebu, guz, jgo, jmc, kde, kea, khq, kln,
21 # ksb, ksf, ksh, lag, lkt, lrc, luy, mer, mfe, mgh, mgo, mua, mzn, naq, nmg, nnh, nus,
22 # rof, rwk, saq, sbp, she, ses, shi, teo, twq, tzm, vun, wae, xog, yav, yue
23 "afr": "af",
24 "amh": "am",
25 "ara": "ar",
26 "asm": "as",
27 "ast": "ast",
28 "aze": "az",
29 "bel": "be",
30 "bul": "bg",
31 "ben": "bn",
32 "bod": "bo",
33 "bre": "br",
34 "bos": "bs",
35 "cat": "ca",
36 "cher": "chr",
37 "ces": "cs",
38 "cym": "cy",
39 "dan": "da",
40 "deu": "de",
41 "dzo": "dz",
42 "ell": "el",
43 "eng": "en",
44 "epo": "eo",
45 "spa": "es",
46 "est": "et",
47 "eus": "eu",
48 "fas": "fa",
49 "fin": "fi",
50 "fil": "fil",
51 "fao": "fo", # codespell:ignore
52 "fra": "fr",
53 "fry": "fy",
54 "gle": "ga",
55 "gla": "gd",
56 "glg": "gl",
57 "guj": "gu",
58 "heb": "he",
59 "hin": "hi",
60 "hrv": "hr",
61 "hun": "hu",
62 "hye": "hy",
63 "ind": "id",
64 "isl": "is",
65 "ita": "it",
66 "jpn": "ja",
67 "kat": "ka",
68 "kaz": "kk",
69 "khm": "km",
70 "knda": "kn",
71 "kor": "ko",
72 "kir": "ky",
73 "ltz": "lb",
74 "lao": "lo",
75 "lit": "lt",
76 "lav": "lv",
77 "mal": "ml",
78 "mon": "mn",
79 "mar": "mr",
80 "msa": "ms",
81 "mlt": "mt",
82 "mya": "my",
83 "nep": "ne",
84 "nld": "nl",
85 "ori": "or",
86 "pan": "pa",
87 "pol": "pl",
88 "pus": "ps",
89 "por": "pt",
90 "que": "qu",
91 "ron": "ro",
92 "rus": "ru",
93 "sin": "si",
94 "slk": "sk",
95 "slv": "sl",
96 "sqi": "sq",
97 "srp": "sr",
98 "swe": "sv",
99 "swa": "sw",
100 "tam": "ta",
101 "tel": "te", # codespell:ignore
102 "tha": "th", # codespell:ignore
103 "tir": "ti",
104 "tgl": "tl",
105 "ton": "to",
106 "tur": "tr",
107 "uig": "ug",
108 "ukr": "uk",
109 "urd": "ur",
110 "uzb": "uz",
111 "via": "vi",
112 "yid": "yi",
113 "yor": "yo",
114 "chi": "zh",
115}
118def ocr_to_dateparser_languages(ocr_languages: str) -> list[str]:
119 """
120 Convert Tesseract OCR_LANGUAGE codes (ISO 639-2, e.g. "eng+fra", with optional scripts like "aze_Cyrl")
121 into a list of locales compatible with the `dateparser` library.
123 - If a script is provided (e.g., "aze_Cyrl"), attempts to use the full locale (e.g., "az-Cyrl").
124 Falls back to the base language (e.g., "az") if needed.
125 - If a language cannot be mapped or validated, it is skipped with a warning.
126 - Returns a list of valid locales, or an empty list if none could be converted.
127 """
128 loader = LocaleDataLoader()
129 result = []
130 try:
131 for ocr_language in ocr_languages.split("+"):
132 # Split into language and optional script
133 ocr_lang_part, *script = ocr_language.split("_")
134 ocr_script_part = script[0] if script else None
136 language_part = OCR_TO_DATEPARSER_LANGUAGES.get(ocr_lang_part)
137 if language_part is None:
138 logger.debug(
139 f'Unable to map OCR language "{ocr_lang_part}" to dateparser locale. ',
140 )
141 continue
143 # Ensure base language is supported by dateparser
144 loader.get_locale_map(locales=[language_part])
146 # Try to add the script part if it's supported by dateparser
147 if ocr_script_part:
148 dateparser_language = f"{language_part}-{ocr_script_part.title()}"
149 try:
150 loader.get_locale_map(locales=[dateparser_language])
151 except Exception:
152 logger.info(
153 f"Language variant '{dateparser_language}' not supported by dateparser; falling back to base language '{language_part}'. You can manually set PAPERLESS_DATE_PARSER_LANGUAGES if needed.",
154 )
155 dateparser_language = language_part
156 else:
157 dateparser_language = language_part
158 if dateparser_language not in result:
159 result.append(dateparser_language)
160 except Exception as e:
161 logger.warning(
162 f"Error auto-configuring dateparser languages. Set PAPERLESS_DATE_PARSER_LANGUAGES parameter to avoid this. Detail: {e}",
163 )
164 return []
165 if not result:
166 logger.info(
167 "Unable to automatically determine dateparser languages from OCR_LANGUAGE, falling back to multi-language support.",
168 )
169 return result