Coverage for src/wiktextract/extractor/cs/translation.py: 96%

48 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from wikitextprocessor import HTMLNode, NodeKind, TemplateNode, WikiNode 

2 

3from ...page import clean_node 

4from ...wxr_context import WiktextractContext 

5from .models import Translation, WordEntry 

6from .tags import translate_raw_tags 

7 

8 

9def extract_translation_section( 

10 wxr: WiktextractContext, word_entry: WordEntry, level_node: WikiNode 

11): 

12 sense_index = 0 

13 for list_node in level_node.find_child(NodeKind.LIST): 

14 for list_item in list_node.find_child(NodeKind.LIST_ITEM): 

15 sense_index += 1 

16 for t_node in list_item.find_child(NodeKind.TEMPLATE): 

17 if ( 17 ↛ 16line 17 didn't jump to line 16 because the condition on line 17 was always true

18 t_node.template_name == "Překlady" 

19 and len(t_node.template_parameters) > 0 

20 ): 

21 extract_překlady_template( 

22 wxr, word_entry, t_node, sense_index 

23 ) 

24 

25 

26def extract_překlady_template( 

27 wxr: WiktextractContext, 

28 word_entry: WordEntry, 

29 t_node: TemplateNode, 

30 sense_index: int, 

31): 

32 # https://cs.wiktionary.org/wiki/Šablona:Překlady 

33 expanded_node = wxr.wtp.parse( 

34 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

35 ) 

36 sense = "" 

37 translations = [] 

38 for dfn_tag in expanded_node.find_html_recursively("dfn"): 

39 sense = clean_node(wxr, None, dfn_tag) 

40 for li_tag in expanded_node.find_html_recursively("li"): 

41 lang_name = "unknown" 

42 # A multi-word translation is often written as one `P` template per 

43 # word, separated by a space: "{{P|fr|le}} {{P|fr|meilleur}}" is 

44 # "le meilleur", while a comma separates two translations. 

45 joinable = False 

46 joined = False 

47 for node in li_tag.children: 

48 if ( 

49 isinstance(node, HTMLNode) 

50 and node.tag == "span" 

51 and "translation-item" in node.attrs.get("class", "").split() 

52 ): 

53 word = clean_node(wxr, None, node) 

54 if word == "": 54 ↛ 55line 54 didn't jump to line 55 because the condition on line 54 was never true

55 continue 

56 lang_code = node.attrs.get("lang", "unknown") 

57 if joinable and translations[-1].lang_code == lang_code: 

58 # no space after an elision: "{{P|fr|d’}}{{P|fr|avion}}" 

59 if not translations[-1].word.endswith(("'", "’")): 

60 translations[-1].word += " " 

61 translations[-1].word += word 

62 joined = True 

63 else: 

64 translations.append( 

65 Translation( 

66 word=word, 

67 lang=lang_name, 

68 lang_code=lang_code, 

69 sense=sense, 

70 sense_index=sense_index, 

71 ) 

72 ) 

73 joined = False 

74 joinable = True 

75 elif ( 

76 isinstance(node, HTMLNode) 

77 and node.tag == "abbr" 

78 and "genus" in node.attrs.get("class", "").split() 

79 ): 

80 # The gender after the first word is the head noun's; after 

81 # a later word it belongs to an inner noun ("une fois") 

82 raw_tag = node.attrs.get("title", "") 

83 if raw_tag != "" and len(translations) > 0 and not joined: 

84 translations[-1].raw_tags.append(raw_tag) 

85 translate_raw_tags(translations[-1]) 

86 else: 

87 if ( 

88 isinstance(node, str) 

89 and lang_name == "unknown" 

90 and node.strip().endswith(":") 

91 ): 

92 lang_name = node.strip().removesuffix(":") or "unknown" 

93 # category links expand to nothing and don't break a phrase 

94 if clean_node(wxr, None, node) != "": 

95 joinable = False 

96 

97 word_entry.translations.extend(translations) 

98 clean_node(wxr, word_entry, expanded_node)