Coverage for paperless/utils.py: 12%

32 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 09:07 +0000

1import logging 

2 

3from dateparser.languages.loader import LocaleDataLoader 

4 

5logger = logging.getLogger("paperless.utils") 

6 

7OCR_TO_DATEPARSER_LANGUAGES = { 

8 """ 

9 Translation map from languages supported by Tesseract OCR 

10 to languages supported by dateparser. 

11 To add a language, make sure it is supported by both libraries. 

12 The ISO 639-2 will help you link a 3-char to 2-char language code. 

13 Links: 

14 - Tesseract languages: https://tesseract-ocr.github.io/tessdoc/Data-Files-in-different-versions.html 

15 - Python dateparser languages: https://dateparser.readthedocs.io/en/latest/supported_locales.html 

16 - ISO 639-2: https://www.loc.gov/standards/iso639-2/php/code_list.php 

17 """ 

18 # TODO check these Dateparser languages as they are not referenced on the ISO639-2 standard, 

19 # so we didn't find the equivalent in Tesseract: 

20 # agq, asa, bez, brx, cgg, ckb, dav, dje, dyo, ebu, guz, jgo, jmc, kde, kea, khq, kln, 

21 # ksb, ksf, ksh, lag, lkt, lrc, luy, mer, mfe, mgh, mgo, mua, mzn, naq, nmg, nnh, nus, 

22 # rof, rwk, saq, sbp, she, ses, shi, teo, twq, tzm, vun, wae, xog, yav, yue 

23 "afr": "af", 

24 "amh": "am", 

25 "ara": "ar", 

26 "asm": "as", 

27 "ast": "ast", 

28 "aze": "az", 

29 "bel": "be", 

30 "bul": "bg", 

31 "ben": "bn", 

32 "bod": "bo", 

33 "bre": "br", 

34 "bos": "bs", 

35 "cat": "ca", 

36 "cher": "chr", 

37 "ces": "cs", 

38 "cym": "cy", 

39 "dan": "da", 

40 "deu": "de", 

41 "dzo": "dz", 

42 "ell": "el", 

43 "eng": "en", 

44 "epo": "eo", 

45 "spa": "es", 

46 "est": "et", 

47 "eus": "eu", 

48 "fas": "fa", 

49 "fin": "fi", 

50 "fil": "fil", 

51 "fao": "fo", # codespell:ignore 

52 "fra": "fr", 

53 "fry": "fy", 

54 "gle": "ga", 

55 "gla": "gd", 

56 "glg": "gl", 

57 "guj": "gu", 

58 "heb": "he", 

59 "hin": "hi", 

60 "hrv": "hr", 

61 "hun": "hu", 

62 "hye": "hy", 

63 "ind": "id", 

64 "isl": "is", 

65 "ita": "it", 

66 "jpn": "ja", 

67 "kat": "ka", 

68 "kaz": "kk", 

69 "khm": "km", 

70 "knda": "kn", 

71 "kor": "ko", 

72 "kir": "ky", 

73 "ltz": "lb", 

74 "lao": "lo", 

75 "lit": "lt", 

76 "lav": "lv", 

77 "mal": "ml", 

78 "mon": "mn", 

79 "mar": "mr", 

80 "msa": "ms", 

81 "mlt": "mt", 

82 "mya": "my", 

83 "nep": "ne", 

84 "nld": "nl", 

85 "ori": "or", 

86 "pan": "pa", 

87 "pol": "pl", 

88 "pus": "ps", 

89 "por": "pt", 

90 "que": "qu", 

91 "ron": "ro", 

92 "rus": "ru", 

93 "sin": "si", 

94 "slk": "sk", 

95 "slv": "sl", 

96 "sqi": "sq", 

97 "srp": "sr", 

98 "swe": "sv", 

99 "swa": "sw", 

100 "tam": "ta", 

101 "tel": "te", # codespell:ignore 

102 "tha": "th", # codespell:ignore 

103 "tir": "ti", 

104 "tgl": "tl", 

105 "ton": "to", 

106 "tur": "tr", 

107 "uig": "ug", 

108 "ukr": "uk", 

109 "urd": "ur", 

110 "uzb": "uz", 

111 "via": "vi", 

112 "yid": "yi", 

113 "yor": "yo", 

114 "chi": "zh", 

115} 

116 

117 

118def ocr_to_dateparser_languages(ocr_languages: str) -> list[str]: 

119 """ 

120 Convert Tesseract OCR_LANGUAGE codes (ISO 639-2, e.g. "eng+fra", with optional scripts like "aze_Cyrl") 

121 into a list of locales compatible with the `dateparser` library. 

122 

123 - If a script is provided (e.g., "aze_Cyrl"), attempts to use the full locale (e.g., "az-Cyrl"). 

124 Falls back to the base language (e.g., "az") if needed. 

125 - If a language cannot be mapped or validated, it is skipped with a warning. 

126 - Returns a list of valid locales, or an empty list if none could be converted. 

127 """ 

128 loader = LocaleDataLoader() 

129 result = [] 

130 try: 

131 for ocr_language in ocr_languages.split("+"): 

132 # Split into language and optional script 

133 ocr_lang_part, *script = ocr_language.split("_") 

134 ocr_script_part = script[0] if script else None 

135 

136 language_part = OCR_TO_DATEPARSER_LANGUAGES.get(ocr_lang_part) 

137 if language_part is None: 

138 logger.debug( 

139 f'Unable to map OCR language "{ocr_lang_part}" to dateparser locale. ', 

140 ) 

141 continue 

142 

143 # Ensure base language is supported by dateparser 

144 loader.get_locale_map(locales=[language_part]) 

145 

146 # Try to add the script part if it's supported by dateparser 

147 if ocr_script_part: 

148 dateparser_language = f"{language_part}-{ocr_script_part.title()}" 

149 try: 

150 loader.get_locale_map(locales=[dateparser_language]) 

151 except Exception: 

152 logger.info( 

153 f"Language variant '{dateparser_language}' not supported by dateparser; falling back to base language '{language_part}'. You can manually set PAPERLESS_DATE_PARSER_LANGUAGES if needed.", 

154 ) 

155 dateparser_language = language_part 

156 else: 

157 dateparser_language = language_part 

158 if dateparser_language not in result: 

159 result.append(dateparser_language) 

160 except Exception as e: 

161 logger.warning( 

162 f"Error auto-configuring dateparser languages. Set PAPERLESS_DATE_PARSER_LANGUAGES parameter to avoid this. Detail: {e}", 

163 ) 

164 return [] 

165 if not result: 

166 logger.info( 

167 "Unable to automatically determine dateparser languages from OCR_LANGUAGE, falling back to multi-language support.", 

168 ) 

169 return result