Coverage for src/wiktextract/extractor/cs/translation.py: 96%
48 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor import HTMLNode, NodeKind, TemplateNode, WikiNode
3from ...page import clean_node
4from ...wxr_context import WiktextractContext
5from .models import Translation, WordEntry
6from .tags import translate_raw_tags
9def extract_translation_section(
10 wxr: WiktextractContext, word_entry: WordEntry, level_node: WikiNode
11):
12 sense_index = 0
13 for list_node in level_node.find_child(NodeKind.LIST):
14 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
15 sense_index += 1
16 for t_node in list_item.find_child(NodeKind.TEMPLATE):
17 if ( 17 ↛ 16line 17 didn't jump to line 16 because the condition on line 17 was always true
18 t_node.template_name == "Překlady"
19 and len(t_node.template_parameters) > 0
20 ):
21 extract_překlady_template(
22 wxr, word_entry, t_node, sense_index
23 )
26def extract_překlady_template(
27 wxr: WiktextractContext,
28 word_entry: WordEntry,
29 t_node: TemplateNode,
30 sense_index: int,
31):
32 # https://cs.wiktionary.org/wiki/Šablona:Překlady
33 expanded_node = wxr.wtp.parse(
34 wxr.wtp.node_to_wikitext(t_node), expand_all=True
35 )
36 sense = ""
37 translations = []
38 for dfn_tag in expanded_node.find_html_recursively("dfn"):
39 sense = clean_node(wxr, None, dfn_tag)
40 for li_tag in expanded_node.find_html_recursively("li"):
41 lang_name = "unknown"
42 # A multi-word translation is often written as one `P` template per
43 # word, separated by a space: "{{P|fr|le}} {{P|fr|meilleur}}" is
44 # "le meilleur", while a comma separates two translations.
45 joinable = False
46 joined = False
47 for node in li_tag.children:
48 if (
49 isinstance(node, HTMLNode)
50 and node.tag == "span"
51 and "translation-item" in node.attrs.get("class", "").split()
52 ):
53 word = clean_node(wxr, None, node)
54 if word == "": 54 ↛ 55line 54 didn't jump to line 55 because the condition on line 54 was never true
55 continue
56 lang_code = node.attrs.get("lang", "unknown")
57 if joinable and translations[-1].lang_code == lang_code:
58 # no space after an elision: "{{P|fr|d’}}{{P|fr|avion}}"
59 if not translations[-1].word.endswith(("'", "’")):
60 translations[-1].word += " "
61 translations[-1].word += word
62 joined = True
63 else:
64 translations.append(
65 Translation(
66 word=word,
67 lang=lang_name,
68 lang_code=lang_code,
69 sense=sense,
70 sense_index=sense_index,
71 )
72 )
73 joined = False
74 joinable = True
75 elif (
76 isinstance(node, HTMLNode)
77 and node.tag == "abbr"
78 and "genus" in node.attrs.get("class", "").split()
79 ):
80 # The gender after the first word is the head noun's; after
81 # a later word it belongs to an inner noun ("une fois")
82 raw_tag = node.attrs.get("title", "")
83 if raw_tag != "" and len(translations) > 0 and not joined:
84 translations[-1].raw_tags.append(raw_tag)
85 translate_raw_tags(translations[-1])
86 else:
87 if (
88 isinstance(node, str)
89 and lang_name == "unknown"
90 and node.strip().endswith(":")
91 ):
92 lang_name = node.strip().removesuffix(":") or "unknown"
93 # category links expand to nothing and don't break a phrase
94 if clean_node(wxr, None, node) != "":
95 joinable = False
97 word_entry.translations.extend(translations)
98 clean_node(wxr, word_entry, expanded_node)