Coverage for src/wiktextract/extractor/en/descendant.py: 85%

174 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from copy import deepcopy 

2 

3from mediawiki_langcodes import name_to_code 

4from wikitextprocessor import ( 

5 HTMLNode, 

6 LevelNode, 

7 NodeKind, 

8 TemplateNode, 

9 WikiNode, 

10) 

11 

12from ...clean import clean_template_args 

13from ...datautils import data_append, data_extend 

14from ...page import clean_node 

15from ...tags import valid_tags 

16from ...wxr_context import WiktextractContext 

17from ..ruby import extract_ruby 

18from .type_utils import DescendantData, TemplateArgs, TemplateData, WordData 

19 

20# Annoying templates that should be in etymology sections, but sometimes 

21# are thrown in heads because the etymology section is missing, like at 

22# the oldest level of a reconstruction: see wiktextract#1658 

23ETYMOLOGY_TEMPLATES_IN_HEADS = { 

24 "ety", 

25 "etymon", 

26} 

27 

28 

29def extract_descendant_section( 

30 wxr: WiktextractContext, 

31 word_entry: WordData, 

32 level_node: LevelNode, 

33 is_derived: bool, 

34): 

35 desc_list = [] 

36 for t_node in level_node.find_child(NodeKind.TEMPLATE): 

37 if ( 

38 isinstance(t_node, TemplateNode) 

39 and t_node.template_name.lower() == "cjkv" 

40 ): 

41 desc_list.extend(extract_cjkv_template(wxr, t_node)) 

42 

43 seen_lists = set() 

44 # get around unnecessarily pre-expanded "top" template 

45 for list_node in level_node.find_child_recursively(NodeKind.LIST): 

46 if list_node in seen_lists: 

47 continue 

48 seen_lists.add(list_node) 

49 for list_item in list_node.find_child(NodeKind.LIST_ITEM): 

50 desc_list.extend( 

51 extract_desc_list_item(wxr, list_item, [], seen_lists, [])[0] 

52 ) 

53 

54 if is_derived: 

55 for data in desc_list: 

56 if "derived" not in data.get("tags", []): 56 ↛ 55line 56 didn't jump to line 55 because the condition on line 56 was always true

57 data_append(data, "tags", "derived") 

58 if len(desc_list) > 0: 

59 data_extend(word_entry, "descendants", desc_list) 

60 

61 

62def extract_cjkv_template( 

63 wxr: WiktextractContext, t_node: TemplateNode 

64) -> list[DescendantData]: 

65 expanded_template = wxr.wtp.parse( 

66 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

67 ) 

68 seen_lists = set() 

69 desc_list = [] 

70 for list_node in expanded_template.find_child_recursively(NodeKind.LIST): 70 ↛ 71line 70 didn't jump to line 71 because the loop on line 70 never started

71 if list_node in seen_lists: 

72 continue 

73 seen_lists.add(list_node) 

74 for list_item in list_node.find_child(NodeKind.LIST_ITEM): 

75 desc_list.extend( 

76 extract_desc_list_item(wxr, list_item, [], seen_lists, [])[0] 

77 ) 

78 return desc_list 

79 

80 

81def extract_desc_list_item( 

82 wxr: WiktextractContext, 

83 list_item: WikiNode, 

84 parent_descendant_datas: list[DescendantData], 

85 seen_lists: set[WikiNode], 

86 raw_tags: list[str], 

87 lang_code: str = "unknown", 

88 lang_name: str = "unknown", 

89 etym_templates: list[TemplateNode] | None = None, 

90) -> tuple[list[DescendantData], str, str]: 

91 # process list item node and <li> tag 

92 data_list = [] 

93 before_word_raw_tags = [] 

94 if etym_templates is None: 

95 etym_templates = [] 

96 after_word = False 

97 for child in list_item.children: 

98 if isinstance(child, str): 

99 child = child.strip() 

100 if child == ",": 

101 after_word = False 

102 elif child.endswith(":"): 

103 lang_name = child.strip(": \n") or "unknown" 

104 lang_code = ( 

105 choose_more_specific_langcode( 

106 name_to_code(lang_name, "en"), lang_code 

107 ) 

108 or "unknown" 

109 ) 

110 elif lcode := name_to_code(child): 110 ↛ 111line 110 didn't jump to line 111 because the condition on line 110 was never true

111 lang_name = child 

112 lang_code = lcode 

113 lang_code = ( 

114 choose_more_specific_langcode(lcode, lang_code) or "unknown" 

115 ) 

116 elif lname := does_text_look_like_language_name(child): 

117 lang_name = lname 

118 lang_code = ( 

119 choose_more_specific_langcode( 

120 name_to_code(lang_name, "en"), lang_code 

121 ) 

122 or "unknown" 

123 ) 

124 elif isinstance(child, HTMLNode) and child.tag == "span": 

125 after_word = extract_desc_span_tag( 

126 wxr, 

127 child, 

128 data_list, 

129 lang_code, 

130 lang_name, 

131 raw_tags, 

132 before_word_raw_tags, 

133 after_word, 

134 etym_templates, 

135 ) 

136 elif ( 136 ↛ 141line 136 didn't jump to line 141 because the condition on line 136 was never true

137 isinstance(child, HTMLNode) 

138 and child.tag == "i" 

139 and len(data_list) > 0 

140 ): 

141 for span_tag in child.find_html( 

142 "span", attr_name="class", attr_value="Latn" 

143 ): 

144 roman = clean_node(wxr, None, span_tag) 

145 if roman != "": 

146 data_list[-1]["roman"] = roman 

147 if len( 

148 data_list 

149 ) > 1 and "Traditional-Chinese" in data_list[-2].get( 

150 "tags", [] 

151 ): 

152 data_list[-2]["roman"] = roman 

153 elif isinstance(child, TemplateNode) and child.template_name in [ 

154 "desctree", 

155 "descendants tree", 

156 "desc", 

157 "descendant", 

158 "ja-r", 

159 "zh-l", 

160 "zh-m", 

161 "link", # used in Reconstruction pages 

162 "l", 

163 ]: 

164 if child.template_name.startswith("desc"): 

165 lang_code = child.template_parameters.get(1, "") or "unknown" 

166 expanded_template = wxr.wtp.parse( 

167 wxr.wtp.node_to_wikitext(child), expand_all=True 

168 ) 

169 new_data, new_l_code, new_l_name = extract_desc_list_item( 

170 wxr, 

171 expanded_template, 

172 [], # avoid add twice 

173 seen_lists, 

174 raw_tags, 

175 lang_code, 

176 lang_name, 

177 etym_templates, 

178 ) 

179 data_list.extend(new_data) 

180 # save lang data from desc template 

181 lang_code = new_l_code 

182 lang_name = new_l_name 

183 elif ( 

184 isinstance(child, TemplateNode) 

185 and child.template_name in ETYMOLOGY_TEMPLATES_IN_HEADS 

186 ): 

187 etym_templates.append(child) 

188 

189 if len(data_list) == 0 and ( 

190 lang_code != "unknown" or lang_name != "unknown" 

191 ): 

192 data = DescendantData(lang_code=lang_code, lang=lang_name) 

193 if len(raw_tags) > 0: 

194 data["raw_tags"] = raw_tags 

195 if len(etym_templates) > 0: 

196 etymology_nodes_append(wxr, data, etym_templates) 

197 etym_templates.clear() 

198 data_list.append(data) 

199 

200 for ul_tag in list_item.find_html("ul"): 

201 for li_tag in ul_tag.find_html("li"): 

202 extract_desc_list_item( 

203 wxr, 

204 li_tag, 

205 data_list, 

206 seen_lists, 

207 [], 

208 ) 

209 for next_list in list_item.find_child(NodeKind.LIST): 

210 if next_list in seen_lists: 210 ↛ 211line 210 didn't jump to line 211 because the condition on line 210 was never true

211 continue 

212 seen_lists.add(next_list) 

213 for next_list_item in next_list.find_child(NodeKind.LIST_ITEM): 

214 extract_desc_list_item( 

215 wxr, 

216 next_list_item, 

217 data_list, 

218 seen_lists, 

219 [], 

220 ) 

221 

222 for p_data in parent_descendant_datas: 

223 data_extend(p_data, "descendants", data_list) 

224 return data_list, lang_code, lang_name 

225 

226 

227def extract_desc_span_tag( 

228 wxr: WiktextractContext, 

229 span_tag: HTMLNode, 

230 desc_lists: list[DescendantData], 

231 lang_code: str, 

232 lang_name: str, 

233 raw_tags: list[str], 

234 before_word_raw_tags: list[str], 

235 after_word: bool, 

236 etym_templates: list[TemplateNode], 

237) -> bool: 

238 class_names = span_tag.attrs.get("class", "").split() 

239 span_lang = span_tag.attrs.get("lang", "") 

240 span_title = span_tag.attrs.get("title", "") 

241 if ("tr" in class_names or span_lang.endswith("-Latn")) and len( 

242 desc_lists 

243 ) > 0: 

244 roman = clean_node(wxr, None, span_tag) 

245 if roman != "": 245 ↛ 308line 245 didn't jump to line 308 because the condition on line 245 was always true

246 desc_lists[-1]["roman"] = clean_node(wxr, None, span_tag) 

247 if len(desc_lists) > 1 and "Traditional-Chinese" in desc_lists[ 247 ↛ 250line 247 didn't jump to line 250 because the condition on line 247 was never true

248 -2 

249 ].get("tags", []): 

250 desc_lists[-2]["roman"] = roman 

251 elif ( 

252 "qualifier-content" in class_names 

253 or "gender" in class_names 

254 or "label-content" in class_names 

255 ) and len(desc_lists) > 0: 

256 for raw_tag in clean_node(wxr, None, span_tag).split(","): 

257 raw_tag = raw_tag.strip() 

258 if raw_tag != "": 258 ↛ 256line 258 didn't jump to line 256 because the condition on line 258 was always true

259 if after_word: 

260 data_append( 

261 desc_lists[-1], 

262 "tags" if raw_tag in valid_tags else "raw_tags", 

263 raw_tag, 

264 ) 

265 else: 

266 before_word_raw_tags.append(raw_tag) 

267 elif span_lang != "": 

268 ruby_data, nodes_without_ruby = extract_ruby(wxr, span_tag) 

269 desc_data = DescendantData( 

270 lang=lang_name, 

271 lang_code=lang_code, 

272 word=clean_node(wxr, None, nodes_without_ruby), 

273 ) 

274 for raw_tag_list in [before_word_raw_tags, raw_tags]: 

275 for raw_tag in raw_tag_list: 

276 data_append( 

277 desc_data, 

278 "tags" if raw_tag in valid_tags else "raw_tags", 

279 raw_tag, 

280 ) 

281 before_word_raw_tags.clear() 

282 if len(ruby_data) > 0: 282 ↛ 283line 282 didn't jump to line 283 because the condition on line 282 was never true

283 desc_data["ruby"] = ruby_data 

284 if len(etym_templates) > 0: 

285 etymology_nodes_append(wxr, desc_data, etym_templates) 

286 etym_templates.clear() 

287 if desc_data["lang_code"] == "unknown": 

288 desc_data["lang_code"] = span_lang 

289 if "Hant" in class_names: 289 ↛ 290line 289 didn't jump to line 290 because the condition on line 289 was never true

290 data_append(desc_data, "tags", "Traditional-Chinese") 

291 elif "Hans" in class_names: 291 ↛ 292line 291 didn't jump to line 292 because the condition on line 291 was never true

292 data_append(desc_data, "tags", "Simplified-Chinese") 

293 if desc_data["word"] not in ["", "/"]: 293 ↛ 295line 293 didn't jump to line 295 because the condition on line 293 was always true

294 desc_lists.append(deepcopy(desc_data)) 

295 after_word = True 

296 elif span_title != "" and clean_node(wxr, None, span_tag) in [ 

297 "→", 

298 "⇒", 

299 ">", 

300 "?", 

301 ]: 

302 raw_tags.append(span_title) 

303 elif "mention-gloss" in class_names and len(desc_lists) > 0: 

304 sense = clean_node(wxr, None, span_tag) 

305 if sense != "": 305 ↛ 308line 305 didn't jump to line 308 because the condition on line 305 was always true

306 desc_lists[-1]["sense"] = sense 

307 

308 return after_word 

309 

310 

311def does_text_look_like_language_name(text: str) -> str | None: 

312 text = text.strip() 

313 if not text: 

314 return None 

315 split_text = text.replace("-", " ").split() 

316 if any(name_to_code(s.strip(), "en") for s in split_text): 316 ↛ 317line 316 didn't jump to line 317 because the condition on line 316 was never true

317 return text 

318 if len(split_text) >= 2: 

319 if all(s != "" and s[0].isupper() for s in split_text): 

320 return text 

321 # len(text) == 1 

322 elif text.endswith(("ic", "ish", "an")): 

323 return text 

324 return None 

325 

326 

327def choose_more_specific_langcode(new: str | None, old: str) -> str | None: 

328 if old == "unknown": 

329 return new 

330 if new is None or new == "": 

331 return old 

332 if old.startswith(new + "-"): 332 ↛ 334line 332 didn't jump to line 334 because the condition on line 332 was never true

333 # "fa-cls" or "fa" -> "fa-cls" 

334 return old 

335 return new 

336 

337 

338def etymology_template_append( 

339 data: WordData | DescendantData, 

340 name: str, 

341 args_ht: TemplateArgs, 

342 expansion: str, 

343): 

344 dt: TemplateData = { 

345 "name": name, 

346 "args": args_ht, 

347 "expansion": expansion, 

348 } 

349 data_append(data, "etymology_templates", dt) 

350 

351 

352def etymology_nodes_append( 

353 wxr: WiktextractContext, 

354 desc_data: DescendantData, 

355 etym_templates: list[TemplateNode], 

356) -> None: 

357 for etemp in etym_templates: 

358 args_ht = clean_template_args(wxr, etemp.template_parameters) 

359 expansion = clean_node(wxr, None, etemp) 

360 etymology_template_append( 

361 desc_data, etemp.template_name, args_ht, expansion 

362 )