Coverage for src/wiktextract/extractor/ko/pos.py: 72%

207 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1import re 

2 

3from wikitextprocessor import ( 

4 HTMLNode, 

5 LevelNode, 

6 NodeKind, 

7 TemplateNode, 

8 WikiNode, 

9) 

10 

11from ...page import clean_node 

12from ...wxr_context import WiktextractContext 

13from ..ruby import extract_ruby 

14from .example import extract_example_list_item 

15from .linkage import ( 

16 LINKAGE_TEMPLATES, 

17 extract_linkage_list_item, 

18 extract_linkage_template, 

19) 

20from .models import AltForm, Classifier, Form, Sense, WordEntry 

21from .section_titles import LINKAGE_SECTIONS, POS_DATA 

22from .sound import SOUND_TEMPLATES, extract_sound_template 

23from .tags import translate_raw_tags 

24from .translation import extract_translation_template 

25 

26 

27def extract_pos_section( 

28 wxr: WiktextractContext, 

29 page_data: list[WordEntry], 

30 base_data: WordEntry, 

31 level_node: LevelNode, 

32 pos_title: str, 

33) -> None: 

34 page_data.append(base_data.model_copy(deep=True)) 

35 orig_title = pos_title 

36 pos_title = pos_title.removeprefix("보조 ").strip() 

37 if pos_title in POS_DATA: 

38 page_data[-1].pos_title = orig_title 

39 pos_data = POS_DATA[pos_title] 

40 page_data[-1].pos = pos_data["pos"] 

41 page_data[-1].tags.extend(pos_data.get("tags", [])) 

42 if ( 42 ↛ 46line 42 didn't jump to line 46 because the condition on line 42 was never true

43 orig_title.startswith("보조 ") 

44 and "auxiliary" not in page_data[-1].tags 

45 ): 

46 page_data[-1].tags.append("auxiliary") 

47 

48 has_linkage = False 

49 for node in level_node.find_child(NodeKind.LIST | NodeKind.TEMPLATE): 

50 if isinstance(node, TemplateNode): 

51 if node.template_name in SOUND_TEMPLATES: 

52 extract_sound_template(wxr, page_data[-1], node) 

53 elif node.template_name in LINKAGE_TEMPLATES: 

54 has_linkage = extract_linkage_template( 

55 wxr, page_data[-1], node, "derived" 

56 ) 

57 elif node.template_name == "외국어": 

58 extract_translation_template( 

59 wxr, 

60 page_data[-1], 

61 node, 

62 page_data[-1].senses[-1].glosses[-1] 

63 if len(page_data[-1].senses) > 0 

64 else "", 

65 ) 

66 elif node.template_name.startswith( 66 ↛ 49line 66 didn't jump to line 49 because the condition on line 66 was always true

67 base_data.lang_code + "-" 

68 ) or node.template_name.endswith((" 동사", " 명사", " 고유명사")): 

69 extract_headword_line_template(wxr, page_data[-1], node) 

70 elif node.kind == NodeKind.LIST: 70 ↛ 49line 70 didn't jump to line 49 because the condition on line 70 was always true

71 for list_item in node.find_child(NodeKind.LIST_ITEM): 

72 if node.sarg.startswith("#") and node.sarg.endswith("#"): 

73 extract_gloss_list_item( 

74 wxr, 

75 page_data[-1], 

76 list_item, 

77 Sense(pattern=page_data[-1].pattern), 

78 ) 

79 else: 

80 extract_unordered_list_item(wxr, page_data[-1], list_item) 

81 

82 if not ( 

83 len(page_data[-1].senses) > 0 

84 or len(page_data[-1].sounds) > len(base_data.sounds) 

85 or len(page_data[-1].translations) > len(base_data.translations) 

86 or has_linkage 

87 ): 

88 page_data.pop() 

89 

90 

91def extract_gloss_list_item( 

92 wxr: WiktextractContext, 

93 word_entry: WordEntry, 

94 list_item: WikiNode, 

95 parent_sense: Sense, 

96) -> None: 

97 gloss_nodes = [] 

98 sense = parent_sense.model_copy(deep=True) 

99 for node in list_item.children: 

100 if isinstance(node, WikiNode) and node.kind == NodeKind.LIST: 

101 gloss_text = clean_node(wxr, sense, gloss_nodes) 

102 if len(gloss_text) > 0: 102 ↛ 107line 102 didn't jump to line 107 because the condition on line 102 was always true

103 sense.glosses.append(gloss_text) 

104 translate_raw_tags(sense) 

105 word_entry.senses.append(sense) 

106 gloss_nodes.clear() 

107 for nested_list_item in node.find_child(NodeKind.LIST_ITEM): 

108 if node.sarg.startswith("#") and node.sarg.endswith("#"): 

109 extract_gloss_list_item( 

110 wxr, word_entry, nested_list_item, sense 

111 ) 

112 else: 

113 extract_unordered_list_item( 

114 wxr, word_entry, nested_list_item 

115 ) 

116 continue 

117 elif isinstance(node, TemplateNode) and node.template_name.endswith( 

118 " of" 

119 ): 

120 extract_form_of_template(wxr, sense, node) 

121 gloss_nodes.append(node) 

122 elif isinstance(node, TemplateNode) and node.template_name == "라벨": 

123 sense.raw_tags.extend( 

124 [ 

125 raw_tag.strip() 

126 for raw_tag in clean_node(wxr, sense, node) 

127 .strip("()") 

128 .split(",") 

129 ] 

130 ) 

131 elif isinstance(node, TemplateNode) and node.template_name == "zh-mw": 131 ↛ 132line 131 didn't jump to line 132 because the condition on line 131 was never true

132 extract_zh_mw_template(wxr, node, sense) 

133 else: 

134 gloss_nodes.append(node) 

135 

136 gloss_text = clean_node(wxr, sense, gloss_nodes) 

137 if len(gloss_text) > 0: 

138 sense.glosses.append(gloss_text) 

139 translate_raw_tags(sense) 

140 word_entry.senses.append(sense) 

141 

142 

143def extract_unordered_list_item( 

144 wxr: WiktextractContext, word_entry: WordEntry, list_item: WikiNode 

145) -> None: 

146 is_first_bold = True 

147 for index, node in enumerate(list_item.children): 

148 if ( 

149 isinstance(node, WikiNode) 

150 and node.kind == NodeKind.BOLD 

151 and is_first_bold 

152 ): 

153 # `* '''1.''' gloss text`, terrible obsolete layout 

154 is_first_bold = False 

155 bold_text = clean_node(wxr, None, node) 

156 if re.fullmatch(r"\d+(?:-\d+)?\.?", bold_text): 

157 new_list_item = WikiNode(NodeKind.LIST_ITEM, 0) 

158 new_list_item.children = list_item.children[index + 1 :] 

159 extract_gloss_list_item(wxr, word_entry, new_list_item, Sense()) 

160 break 

161 elif isinstance(node, str) and "어원:" in node: 

162 etymology_nodes = [] 

163 etymology_nodes.append(node[node.index(":") + 1 :]) 

164 etymology_nodes.extend(list_item.children[index + 1 :]) 

165 e_links: list[tuple[str, str]] = [] 

166 e_text = clean_node( 

167 wxr, None, etymology_nodes, link_collector=e_links 

168 ) 

169 if len(e_text) > 0: 169 ↛ 172line 169 didn't jump to line 172 because the condition on line 169 was always true

170 word_entry.etymology_texts.append(e_text) 

171 word_entry.etymology_links.extend(e_links) 

172 break 

173 elif ( 

174 isinstance(node, str) 

175 and re.search(r"(?:참고|참조|활용):", node) is not None 

176 ): 

177 note_str = node[node.index(":") + 1 :].strip() 

178 note_str += clean_node( 

179 wxr, 

180 word_entry.senses[-1] 

181 if len(word_entry.senses) > 0 

182 else word_entry, 

183 list_item.children[index + 1 :], 

184 ) 

185 if len(word_entry.senses) > 0: 

186 word_entry.senses[-1].note = note_str 

187 else: 

188 word_entry.note = note_str 

189 break 

190 elif ( 

191 isinstance(node, str) 

192 and ":" in node 

193 and node[: node.index(":")].strip() in LINKAGE_SECTIONS 

194 ): 

195 extract_linkage_list_item(wxr, word_entry, list_item, "", False) 

196 break 

197 elif isinstance(node, str) and "문형:" in node: 

198 word_entry.pattern = node[node.index(":") + 1 :].strip() 

199 word_entry.pattern += clean_node( 

200 wxr, None, list_item.children[index + 1 :] 

201 ) 

202 break 

203 else: 

204 if len(word_entry.senses) > 0: 

205 extract_example_list_item( 

206 wxr, word_entry.senses[-1], list_item, word_entry.lang_code 

207 ) 

208 

209 

210def extract_form_of_template( 

211 wxr: WiktextractContext, sense: Sense, t_node: TemplateNode 

212) -> None: 

213 if "form-of" not in sense.tags: 213 ↛ 215line 213 didn't jump to line 215 because the condition on line 213 was always true

214 sense.tags.append("form-of") 

215 word_arg = 1 if t_node.template_name == "ko-hanja form of" else 2 

216 word = clean_node(wxr, None, t_node.template_parameters.get(word_arg, "")) 

217 if len(word) > 0: 217 ↛ exitline 217 didn't return from function 'extract_form_of_template' because the condition on line 217 was always true

218 sense.form_of.append(AltForm(word=word)) 

219 

220 

221def extract_grammar_note_section( 

222 wxr: WiktextractContext, word_entry: WordEntry, level_node: LevelNode 

223) -> None: 

224 for list_item in level_node.find_child_recursively(NodeKind.LIST_ITEM): 

225 word_entry.note = clean_node(wxr, None, list_item.children) 

226 

227 

228def extract_zh_mw_template( 

229 wxr: WiktextractContext, t_node: TemplateNode, sense: Sense 

230) -> None: 

231 # Chinese inline classifier template 

232 # copied from zh edition code 

233 expanded_node = wxr.wtp.parse( 

234 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

235 ) 

236 classifiers = [] 

237 last_word = "" 

238 for span_tag in expanded_node.find_html_recursively("span"): 

239 span_class = span_tag.attrs.get("class", "") 

240 if span_class in ["Hani", "Hant", "Hans"]: 

241 word = clean_node(wxr, None, span_tag) 

242 if word != "/": 

243 classifier = Classifier(classifier=word) 

244 if span_class == "Hant": 

245 classifier.tags.append("Traditional-Chinese") 

246 elif span_class == "Hans": 

247 classifier.tags.append("Simplified-Chinese") 

248 

249 if len(classifiers) > 0 and last_word != "/": 

250 sense.classifiers.extend(classifiers) 

251 classifiers.clear() 

252 classifiers.append(classifier) 

253 last_word = word 

254 elif "title" in span_tag.attrs: 

255 raw_tag = clean_node(wxr, None, span_tag.attrs["title"]) 

256 if len(raw_tag) > 0: 

257 for classifier in classifiers: 

258 classifier.raw_tags.append(raw_tag) 

259 sense.classifiers.extend(classifiers) 

260 for classifier in sense.classifiers: 

261 translate_raw_tags(classifier) 

262 

263 

264def extract_headword_line_template( 

265 wxr: WiktextractContext, word_entry: WordEntry, t_node: TemplateNode 

266): 

267 forms = [] 

268 expanded_node = wxr.wtp.parse( 

269 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

270 ) 

271 for main_span_tag in expanded_node.find_html( 

272 "span", attr_name="class", attr_value="headword-line" 

273 ): 

274 i_tags = [] 

275 for html_node in main_span_tag.find_child(NodeKind.HTML): 

276 class_names = html_node.attrs.get("class", "").split() 

277 if html_node.tag == "strong" and "headword" in class_names: 

278 ruby, no_ruby = extract_ruby(wxr, html_node) 

279 strong_str = clean_node(wxr, None, no_ruby) 

280 if strong_str not in ["", wxr.wtp.title] or len(ruby) > 0: 

281 forms.append( 

282 Form(form=strong_str, tags=["canonical"], ruby=ruby) 

283 ) 

284 elif html_node.tag == "span": 

285 if "headword-tr" in class_names or "tr" in class_names: 

286 roman = clean_node(wxr, None, html_node) 

287 if ( 

288 len(forms) > 0 

289 and "canonical" not in forms[-1].tags 

290 and "romanization" not in forms[-1].tags 

291 ): 

292 forms[-1].roman = roman 

293 elif roman != "": 293 ↛ 275line 293 didn't jump to line 275 because the condition on line 293 was always true

294 forms.append(Form(form=roman, tags=["romanization"])) 

295 elif "gender" in class_names: 295 ↛ 296line 295 didn't jump to line 296 because the condition on line 295 was never true

296 for abbr_tag in html_node.find_html("abbr"): 

297 gender_tag = clean_node(wxr, None, abbr_tag) 

298 if ( 

299 len(forms) > 0 

300 and "canonical" not in forms[-1].tags 

301 and "romanization" not in forms[-1].tags 

302 ): 

303 forms[-1].raw_tags.append(gender_tag) 

304 else: 

305 word_entry.raw_tags.append(gender_tag) 

306 elif "ib-content" in class_names: 306 ↛ 307line 306 didn't jump to line 307 because the condition on line 306 was never true

307 raw_tag = clean_node(wxr, None, html_node) 

308 if raw_tag != "": 

309 word_entry.raw_tags.append(raw_tag) 

310 elif html_node.tag == "sup" and word_entry.lang_code == "ja": 310 ↛ 311line 310 didn't jump to line 311 because the condition on line 310 was never true

311 forms.append(extract_historical_kana(wxr, html_node)) 

312 elif html_node.tag == "i": 

313 if len(i_tags) > 0: 

314 word_entry.raw_tags.extend(i_tags) 

315 i_tags.clear() 

316 for i_child in html_node.children: 

317 raw_tag = ( 

318 clean_node(wxr, None, i_child) 

319 .removeprefix("^†") 

320 .strip() 

321 ) 

322 if raw_tag != "": 322 ↛ 316line 322 didn't jump to line 316 because the condition on line 322 was always true

323 i_tags.append(raw_tag) 

324 elif html_node.tag == "b": 324 ↛ 275line 324 didn't jump to line 275 because the condition on line 324 was always true

325 ruby, no_ruby = extract_ruby(wxr, html_node) 

326 for form_str in filter( 

327 None, 

328 map(str.strip, clean_node(wxr, None, no_ruby).split(",")), 

329 ): 

330 form = Form(form=form_str, ruby=ruby) 

331 if i_tags == ["또는"]: 331 ↛ 332line 331 didn't jump to line 332 because the condition on line 331 was never true

332 if len(forms) > 0: 

333 form.raw_tags.extend(forms[-1].raw_tags) 

334 else: 

335 form.raw_tags.extend(i_tags) 

336 forms.append(form) 

337 i_tags.clear() 

338 

339 if len(i_tags) > 0: 339 ↛ 340line 339 didn't jump to line 340 because the condition on line 339 was never true

340 word_entry.raw_tags.extend(i_tags) 

341 for form in forms: 

342 translate_raw_tags(form) 

343 word_entry.forms.extend(forms) 

344 clean_node(wxr, word_entry, expanded_node) 

345 translate_raw_tags(word_entry) 

346 

347 

348def extract_historical_kana( 

349 wxr: WiktextractContext, sup_node: HTMLNode 

350) -> Form: 

351 form = Form(form="", tags=["archaic"]) 

352 for strong_node in sup_node.find_html("strong"): 

353 form.form = clean_node(wxr, None, strong_node) 

354 for span_node in sup_node.find_html( 

355 "span", attr_name="class", attr_value="tr" 

356 ): 

357 form.roman = clean_node(wxr, None, span_node) 

358 return form