Coverage for src/wiktextract/extractor/en/descendant.py: 85%
174 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from copy import deepcopy
3from mediawiki_langcodes import name_to_code
4from wikitextprocessor import (
5 HTMLNode,
6 LevelNode,
7 NodeKind,
8 TemplateNode,
9 WikiNode,
10)
12from ...clean import clean_template_args
13from ...datautils import data_append, data_extend
14from ...page import clean_node
15from ...tags import valid_tags
16from ...wxr_context import WiktextractContext
17from ..ruby import extract_ruby
18from .type_utils import DescendantData, TemplateArgs, TemplateData, WordData
20# Annoying templates that should be in etymology sections, but sometimes
21# are thrown in heads because the etymology section is missing, like at
22# the oldest level of a reconstruction: see wiktextract#1658
23ETYMOLOGY_TEMPLATES_IN_HEADS = {
24 "ety",
25 "etymon",
26}
29def extract_descendant_section(
30 wxr: WiktextractContext,
31 word_entry: WordData,
32 level_node: LevelNode,
33 is_derived: bool,
34):
35 desc_list = []
36 for t_node in level_node.find_child(NodeKind.TEMPLATE):
37 if (
38 isinstance(t_node, TemplateNode)
39 and t_node.template_name.lower() == "cjkv"
40 ):
41 desc_list.extend(extract_cjkv_template(wxr, t_node))
43 seen_lists = set()
44 # get around unnecessarily pre-expanded "top" template
45 for list_node in level_node.find_child_recursively(NodeKind.LIST):
46 if list_node in seen_lists:
47 continue
48 seen_lists.add(list_node)
49 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
50 desc_list.extend(
51 extract_desc_list_item(wxr, list_item, [], seen_lists, [])[0]
52 )
54 if is_derived:
55 for data in desc_list:
56 if "derived" not in data.get("tags", []): 56 ↛ 55line 56 didn't jump to line 55 because the condition on line 56 was always true
57 data_append(data, "tags", "derived")
58 if len(desc_list) > 0:
59 data_extend(word_entry, "descendants", desc_list)
62def extract_cjkv_template(
63 wxr: WiktextractContext, t_node: TemplateNode
64) -> list[DescendantData]:
65 expanded_template = wxr.wtp.parse(
66 wxr.wtp.node_to_wikitext(t_node), expand_all=True
67 )
68 seen_lists = set()
69 desc_list = []
70 for list_node in expanded_template.find_child_recursively(NodeKind.LIST): 70 ↛ 71line 70 didn't jump to line 71 because the loop on line 70 never started
71 if list_node in seen_lists:
72 continue
73 seen_lists.add(list_node)
74 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
75 desc_list.extend(
76 extract_desc_list_item(wxr, list_item, [], seen_lists, [])[0]
77 )
78 return desc_list
81def extract_desc_list_item(
82 wxr: WiktextractContext,
83 list_item: WikiNode,
84 parent_descendant_datas: list[DescendantData],
85 seen_lists: set[WikiNode],
86 raw_tags: list[str],
87 lang_code: str = "unknown",
88 lang_name: str = "unknown",
89 etym_templates: list[TemplateNode] | None = None,
90) -> tuple[list[DescendantData], str, str]:
91 # process list item node and <li> tag
92 data_list = []
93 before_word_raw_tags = []
94 if etym_templates is None:
95 etym_templates = []
96 after_word = False
97 for child in list_item.children:
98 if isinstance(child, str):
99 child = child.strip()
100 if child == ",":
101 after_word = False
102 elif child.endswith(":"):
103 lang_name = child.strip(": \n") or "unknown"
104 lang_code = (
105 choose_more_specific_langcode(
106 name_to_code(lang_name, "en"), lang_code
107 )
108 or "unknown"
109 )
110 elif lcode := name_to_code(child): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true
111 lang_name = child
112 lang_code = lcode
113 lang_code = (
114 choose_more_specific_langcode(lcode, lang_code) or "unknown"
115 )
116 elif lname := does_text_look_like_language_name(child):
117 lang_name = lname
118 lang_code = (
119 choose_more_specific_langcode(
120 name_to_code(lang_name, "en"), lang_code
121 )
122 or "unknown"
123 )
124 elif isinstance(child, HTMLNode) and child.tag == "span":
125 after_word = extract_desc_span_tag(
126 wxr,
127 child,
128 data_list,
129 lang_code,
130 lang_name,
131 raw_tags,
132 before_word_raw_tags,
133 after_word,
134 etym_templates,
135 )
136 elif ( 136 ↛ 141line 136 didn't jump to line 141 because the condition on line 136 was never true
137 isinstance(child, HTMLNode)
138 and child.tag == "i"
139 and len(data_list) > 0
140 ):
141 for span_tag in child.find_html(
142 "span", attr_name="class", attr_value="Latn"
143 ):
144 roman = clean_node(wxr, None, span_tag)
145 if roman != "":
146 data_list[-1]["roman"] = roman
147 if len(
148 data_list
149 ) > 1 and "Traditional-Chinese" in data_list[-2].get(
150 "tags", []
151 ):
152 data_list[-2]["roman"] = roman
153 elif isinstance(child, TemplateNode) and child.template_name in [
154 "desctree",
155 "descendants tree",
156 "desc",
157 "descendant",
158 "ja-r",
159 "zh-l",
160 "zh-m",
161 "link", # used in Reconstruction pages
162 "l",
163 ]:
164 if child.template_name.startswith("desc"):
165 lang_code = child.template_parameters.get(1, "") or "unknown"
166 expanded_template = wxr.wtp.parse(
167 wxr.wtp.node_to_wikitext(child), expand_all=True
168 )
169 new_data, new_l_code, new_l_name = extract_desc_list_item(
170 wxr,
171 expanded_template,
172 [], # avoid add twice
173 seen_lists,
174 raw_tags,
175 lang_code,
176 lang_name,
177 etym_templates,
178 )
179 data_list.extend(new_data)
180 # save lang data from desc template
181 lang_code = new_l_code
182 lang_name = new_l_name
183 elif (
184 isinstance(child, TemplateNode)
185 and child.template_name in ETYMOLOGY_TEMPLATES_IN_HEADS
186 ):
187 etym_templates.append(child)
189 if len(data_list) == 0 and (
190 lang_code != "unknown" or lang_name != "unknown"
191 ):
192 data = DescendantData(lang_code=lang_code, lang=lang_name)
193 if len(raw_tags) > 0:
194 data["raw_tags"] = raw_tags
195 if len(etym_templates) > 0:
196 etymology_nodes_append(wxr, data, etym_templates)
197 etym_templates.clear()
198 data_list.append(data)
200 for ul_tag in list_item.find_html("ul"):
201 for li_tag in ul_tag.find_html("li"):
202 extract_desc_list_item(
203 wxr,
204 li_tag,
205 data_list,
206 seen_lists,
207 [],
208 )
209 for next_list in list_item.find_child(NodeKind.LIST):
210 if next_list in seen_lists: 210 ↛ 211line 210 didn't jump to line 211 because the condition on line 210 was never true
211 continue
212 seen_lists.add(next_list)
213 for next_list_item in next_list.find_child(NodeKind.LIST_ITEM):
214 extract_desc_list_item(
215 wxr,
216 next_list_item,
217 data_list,
218 seen_lists,
219 [],
220 )
222 for p_data in parent_descendant_datas:
223 data_extend(p_data, "descendants", data_list)
224 return data_list, lang_code, lang_name
227def extract_desc_span_tag(
228 wxr: WiktextractContext,
229 span_tag: HTMLNode,
230 desc_lists: list[DescendantData],
231 lang_code: str,
232 lang_name: str,
233 raw_tags: list[str],
234 before_word_raw_tags: list[str],
235 after_word: bool,
236 etym_templates: list[TemplateNode],
237) -> bool:
238 class_names = span_tag.attrs.get("class", "").split()
239 span_lang = span_tag.attrs.get("lang", "")
240 span_title = span_tag.attrs.get("title", "")
241 if ("tr" in class_names or span_lang.endswith("-Latn")) and len(
242 desc_lists
243 ) > 0:
244 roman = clean_node(wxr, None, span_tag)
245 if roman != "": 245 ↛ 308line 245 didn't jump to line 308 because the condition on line 245 was always true
246 desc_lists[-1]["roman"] = clean_node(wxr, None, span_tag)
247 if len(desc_lists) > 1 and "Traditional-Chinese" in desc_lists[ 247 ↛ 250line 247 didn't jump to line 250 because the condition on line 247 was never true
248 -2
249 ].get("tags", []):
250 desc_lists[-2]["roman"] = roman
251 elif (
252 "qualifier-content" in class_names
253 or "gender" in class_names
254 or "label-content" in class_names
255 ) and len(desc_lists) > 0:
256 for raw_tag in clean_node(wxr, None, span_tag).split(","):
257 raw_tag = raw_tag.strip()
258 if raw_tag != "": 258 ↛ 256line 258 didn't jump to line 256 because the condition on line 258 was always true
259 if after_word:
260 data_append(
261 desc_lists[-1],
262 "tags" if raw_tag in valid_tags else "raw_tags",
263 raw_tag,
264 )
265 else:
266 before_word_raw_tags.append(raw_tag)
267 elif span_lang != "":
268 ruby_data, nodes_without_ruby = extract_ruby(wxr, span_tag)
269 desc_data = DescendantData(
270 lang=lang_name,
271 lang_code=lang_code,
272 word=clean_node(wxr, None, nodes_without_ruby),
273 )
274 for raw_tag_list in [before_word_raw_tags, raw_tags]:
275 for raw_tag in raw_tag_list:
276 data_append(
277 desc_data,
278 "tags" if raw_tag in valid_tags else "raw_tags",
279 raw_tag,
280 )
281 before_word_raw_tags.clear()
282 if len(ruby_data) > 0: 282 ↛ 283line 282 didn't jump to line 283 because the condition on line 282 was never true
283 desc_data["ruby"] = ruby_data
284 if len(etym_templates) > 0:
285 etymology_nodes_append(wxr, desc_data, etym_templates)
286 etym_templates.clear()
287 if desc_data["lang_code"] == "unknown":
288 desc_data["lang_code"] = span_lang
289 if "Hant" in class_names: 289 ↛ 290line 289 didn't jump to line 290 because the condition on line 289 was never true
290 data_append(desc_data, "tags", "Traditional-Chinese")
291 elif "Hans" in class_names: 291 ↛ 292line 291 didn't jump to line 292 because the condition on line 291 was never true
292 data_append(desc_data, "tags", "Simplified-Chinese")
293 if desc_data["word"] not in ["", "/"]: 293 ↛ 295line 293 didn't jump to line 295 because the condition on line 293 was always true
294 desc_lists.append(deepcopy(desc_data))
295 after_word = True
296 elif span_title != "" and clean_node(wxr, None, span_tag) in [
297 "→",
298 "⇒",
299 ">",
300 "?",
301 ]:
302 raw_tags.append(span_title)
303 elif "mention-gloss" in class_names and len(desc_lists) > 0:
304 sense = clean_node(wxr, None, span_tag)
305 if sense != "": 305 ↛ 308line 305 didn't jump to line 308 because the condition on line 305 was always true
306 desc_lists[-1]["sense"] = sense
308 return after_word
311def does_text_look_like_language_name(text: str) -> str | None:
312 text = text.strip()
313 if not text:
314 return None
315 split_text = text.replace("-", " ").split()
316 if any(name_to_code(s.strip(), "en") for s in split_text): 316 ↛ 317line 316 didn't jump to line 317 because the condition on line 316 was never true
317 return text
318 if len(split_text) >= 2:
319 if all(s != "" and s[0].isupper() for s in split_text):
320 return text
321 # len(text) == 1
322 elif text.endswith(("ic", "ish", "an")):
323 return text
324 return None
327def choose_more_specific_langcode(new: str | None, old: str) -> str | None:
328 if old == "unknown":
329 return new
330 if new is None or new == "":
331 return old
332 if old.startswith(new + "-"): 332 ↛ 334line 332 didn't jump to line 334 because the condition on line 332 was never true
333 # "fa-cls" or "fa" -> "fa-cls"
334 return old
335 return new
338def etymology_template_append(
339 data: WordData | DescendantData,
340 name: str,
341 args_ht: TemplateArgs,
342 expansion: str,
343):
344 dt: TemplateData = {
345 "name": name,
346 "args": args_ht,
347 "expansion": expansion,
348 }
349 data_append(data, "etymology_templates", dt)
352def etymology_nodes_append(
353 wxr: WiktextractContext,
354 desc_data: DescendantData,
355 etym_templates: list[TemplateNode],
356) -> None:
357 for etemp in etym_templates:
358 args_ht = clean_template_args(wxr, etemp.template_parameters)
359 expansion = clean_node(wxr, None, etemp)
360 etymology_template_append(
361 desc_data, etemp.template_name, args_ht, expansion
362 )