2023-10-10 20:47:56 -05:00
|
|
|
import pytest
|
|
|
|
|
2023-10-19 23:15:28 -05:00
|
|
|
from unstructured.documents.elements import (
|
|
|
|
NarrativeText,
|
|
|
|
PageBreak,
|
|
|
|
)
|
|
|
|
from unstructured.partition.lang import (
|
2024-01-19 13:59:08 -06:00
|
|
|
_clean_ocr_languages_arg,
|
2024-01-16 11:51:03 -06:00
|
|
|
_convert_language_code_to_pytesseract_lang_code,
|
2023-10-19 23:15:28 -05:00
|
|
|
apply_lang_metadata,
|
|
|
|
detect_languages,
|
|
|
|
prepare_languages_for_tesseract,
|
|
|
|
)
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_with_one_language():
|
|
|
|
languages = ["en"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "eng"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
2023-11-06 20:30:12 -05:00
|
|
|
def test_prepare_languages_for_tesseract_with_duplicated_languages():
|
|
|
|
languages = ["en", "eng"]
|
|
|
|
assert prepare_languages_for_tesseract(languages) == "eng"
|
|
|
|
|
|
|
|
|
2023-09-18 11:42:02 -04:00
|
|
|
def test_prepare_languages_for_tesseract_special_case():
|
|
|
|
languages = ["osd"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "osd"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
languages = ["equ"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "equ"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_removes_empty_inputs():
|
|
|
|
languages = ["kbd", "es"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "spa+spa_old"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_includes_variants():
|
|
|
|
languages = ["chi"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "chi_sim+chi_sim_vert+chi_tra+chi_tra_vert"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_with_multiple_languages():
|
|
|
|
languages = ["ja", "afr", "en", "equ"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "jpn+jpn_vert+afr+eng+equ"
|
2023-09-18 11:42:02 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_warns_nonstandard_language(caplog):
|
|
|
|
languages = ["zzz", "chi"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "chi_sim+chi_sim_vert+chi_tra+chi_tra_vert"
|
2023-09-18 11:42:02 -04:00
|
|
|
assert "not a valid standard language code" in caplog.text
|
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_warns_non_tesseract_language(caplog):
|
|
|
|
languages = ["kbd", "eng"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert prepare_languages_for_tesseract(languages) == "eng"
|
2023-09-18 11:42:02 -04:00
|
|
|
assert "not a language supported by Tesseract" in caplog.text
|
2023-09-26 14:09:27 -04:00
|
|
|
|
|
|
|
|
2023-11-06 20:30:12 -05:00
|
|
|
def test_prepare_languages_for_tesseract_None_languages():
|
|
|
|
with pytest.raises(ValueError, match="`languages` can not be `None`"):
|
|
|
|
languages = None
|
|
|
|
prepare_languages_for_tesseract(languages)
|
|
|
|
|
|
|
|
|
|
|
|
def test_prepare_languages_for_tesseract_no_valid_languages(caplog):
|
|
|
|
languages = [""]
|
|
|
|
assert prepare_languages_for_tesseract(languages) == "eng"
|
|
|
|
assert "Failed to find any valid standard language code from languages" in caplog.text
|
|
|
|
|
|
|
|
|
2023-09-26 14:09:27 -04:00
|
|
|
def test_detect_languages_english_auto():
|
|
|
|
text = "This is a short sentence."
|
2023-10-19 23:15:28 -05:00
|
|
|
assert detect_languages(text) == ["eng"]
|
2023-09-26 14:09:27 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_detect_languages_english_provided():
|
|
|
|
text = "This is another short sentence."
|
|
|
|
languages = ["en"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert detect_languages(text, languages) == ["eng"]
|
2023-09-26 14:09:27 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_detect_languages_korean_auto():
|
|
|
|
text = "안녕하세요"
|
2023-10-19 23:15:28 -05:00
|
|
|
assert detect_languages(text) == ["kor"]
|
2023-09-26 14:09:27 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_detect_languages_gets_multiple_languages():
|
|
|
|
text = "My lubimy mleko i chleb."
|
2023-10-19 23:15:28 -05:00
|
|
|
assert detect_languages(text) == ["ces", "pol", "slk"]
|
2023-09-26 14:09:27 -04:00
|
|
|
|
|
|
|
|
|
|
|
def test_detect_languages_warns_for_auto_and_other_input(caplog):
|
|
|
|
text = "This is another short sentence."
|
|
|
|
languages = ["en", "auto", "rus"]
|
2023-10-19 23:15:28 -05:00
|
|
|
assert detect_languages(text, languages) == ["eng"]
|
2023-09-26 14:09:27 -04:00
|
|
|
assert "rest of the inputted languages will be ignored" in caplog.text
|
2023-10-10 20:47:56 -05:00
|
|
|
|
|
|
|
|
|
|
|
def test_detect_languages_raises_TypeError_for_invalid_languages():
|
|
|
|
with pytest.raises(TypeError):
|
|
|
|
text = "This is a short sentence."
|
2023-10-19 23:15:28 -05:00
|
|
|
detect_languages(text, languages="eng") == ["eng"]
|
|
|
|
|
|
|
|
|
|
|
|
def test_apply_lang_metadata_has_no_warning_for_PageBreak(caplog):
|
|
|
|
elements = [NarrativeText("Sample text."), PageBreak("")]
|
|
|
|
elements = list(
|
|
|
|
apply_lang_metadata(
|
|
|
|
elements=elements,
|
|
|
|
languages=["auto"],
|
|
|
|
detect_language_per_element=True,
|
|
|
|
),
|
|
|
|
)
|
|
|
|
assert "No features in text." not in [rec.message for rec in caplog.records]
|
2024-01-10 18:34:13 -06:00
|
|
|
|
|
|
|
|
2024-01-16 11:51:03 -06:00
|
|
|
@pytest.mark.parametrize(
|
|
|
|
("lang_in", "expected_lang"),
|
|
|
|
[
|
|
|
|
("en", "eng"),
|
|
|
|
("fr", "fra"),
|
|
|
|
],
|
|
|
|
)
|
|
|
|
def test_convert_language_code_to_pytesseract_lang_code(lang_in, expected_lang):
|
|
|
|
assert expected_lang == _convert_language_code_to_pytesseract_lang_code(lang_in)
|
|
|
|
|
|
|
|
|
2024-01-19 13:59:08 -06:00
|
|
|
@pytest.mark.parametrize(
|
|
|
|
("input_ocr_langs", "expected"),
|
|
|
|
[
|
|
|
|
(["eng"], "eng"), # list
|
|
|
|
('"deu"', "deu"), # extra quotation marks
|
|
|
|
("[deu]", "deu"), # brackets
|
|
|
|
("['deu']", "deu"), # brackets and quotation marks
|
|
|
|
(["[deu]"], "deu"), # list, brackets and quotation marks
|
|
|
|
(['"deu"'], "deu"), # list and quotation marks
|
|
|
|
("deu+spa", "deu+spa"), # correct input
|
|
|
|
],
|
|
|
|
)
|
|
|
|
def test_clean_ocr_languages_arg(input_ocr_langs, expected):
|
|
|
|
assert _clean_ocr_languages_arg(input_ocr_langs) == expected
|
|
|
|
|
|
|
|
|
2024-01-18 18:14:45 -06:00
|
|
|
def test_detect_languages_handles_spelled_out_languages():
|
|
|
|
languages = detect_languages(text="Sample text longer than 5 words.", languages=["Spanish"])
|
|
|
|
assert languages == ["spa"]
|