Coverage for src/wiktextract/extractor/ko/pos.py: 72%
207 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1import re
3from wikitextprocessor import (
4 HTMLNode,
5 LevelNode,
6 NodeKind,
7 TemplateNode,
8 WikiNode,
9)
11from ...page import clean_node
12from ...wxr_context import WiktextractContext
13from ..ruby import extract_ruby
14from .example import extract_example_list_item
15from .linkage import (
16 LINKAGE_TEMPLATES,
17 extract_linkage_list_item,
18 extract_linkage_template,
19)
20from .models import AltForm, Classifier, Form, Sense, WordEntry
21from .section_titles import LINKAGE_SECTIONS, POS_DATA
22from .sound import SOUND_TEMPLATES, extract_sound_template
23from .tags import translate_raw_tags
24from .translation import extract_translation_template
27def extract_pos_section(
28 wxr: WiktextractContext,
29 page_data: list[WordEntry],
30 base_data: WordEntry,
31 level_node: LevelNode,
32 pos_title: str,
33) -> None:
34 page_data.append(base_data.model_copy(deep=True))
35 orig_title = pos_title
36 pos_title = pos_title.removeprefix("보조 ").strip()
37 if pos_title in POS_DATA:
38 page_data[-1].pos_title = orig_title
39 pos_data = POS_DATA[pos_title]
40 page_data[-1].pos = pos_data["pos"]
41 page_data[-1].tags.extend(pos_data.get("tags", []))
42 if ( 42 ↛ 46line 42 didn't jump to line 46 because the condition on line 42 was never true
43 orig_title.startswith("보조 ")
44 and "auxiliary" not in page_data[-1].tags
45 ):
46 page_data[-1].tags.append("auxiliary")
48 has_linkage = False
49 for node in level_node.find_child(NodeKind.LIST | NodeKind.TEMPLATE):
50 if isinstance(node, TemplateNode):
51 if node.template_name in SOUND_TEMPLATES:
52 extract_sound_template(wxr, page_data[-1], node)
53 elif node.template_name in LINKAGE_TEMPLATES:
54 has_linkage = extract_linkage_template(
55 wxr, page_data[-1], node, "derived"
56 )
57 elif node.template_name == "외국어":
58 extract_translation_template(
59 wxr,
60 page_data[-1],
61 node,
62 page_data[-1].senses[-1].glosses[-1]
63 if len(page_data[-1].senses) > 0
64 else "",
65 )
66 elif node.template_name.startswith( 66 ↛ 49line 66 didn't jump to line 49 because the condition on line 66 was always true
67 base_data.lang_code + "-"
68 ) or node.template_name.endswith((" 동사", " 명사", " 고유명사")):
69 extract_headword_line_template(wxr, page_data[-1], node)
70 elif node.kind == NodeKind.LIST: 70 ↛ 49line 70 didn't jump to line 49 because the condition on line 70 was always true
71 for list_item in node.find_child(NodeKind.LIST_ITEM):
72 if node.sarg.startswith("#") and node.sarg.endswith("#"):
73 extract_gloss_list_item(
74 wxr,
75 page_data[-1],
76 list_item,
77 Sense(pattern=page_data[-1].pattern),
78 )
79 else:
80 extract_unordered_list_item(wxr, page_data[-1], list_item)
82 if not (
83 len(page_data[-1].senses) > 0
84 or len(page_data[-1].sounds) > len(base_data.sounds)
85 or len(page_data[-1].translations) > len(base_data.translations)
86 or has_linkage
87 ):
88 page_data.pop()
91def extract_gloss_list_item(
92 wxr: WiktextractContext,
93 word_entry: WordEntry,
94 list_item: WikiNode,
95 parent_sense: Sense,
96) -> None:
97 gloss_nodes = []
98 sense = parent_sense.model_copy(deep=True)
99 for node in list_item.children:
100 if isinstance(node, WikiNode) and node.kind == NodeKind.LIST:
101 gloss_text = clean_node(wxr, sense, gloss_nodes)
102 if len(gloss_text) > 0: 102 ↛ 107line 102 didn't jump to line 107 because the condition on line 102 was always true
103 sense.glosses.append(gloss_text)
104 translate_raw_tags(sense)
105 word_entry.senses.append(sense)
106 gloss_nodes.clear()
107 for nested_list_item in node.find_child(NodeKind.LIST_ITEM):
108 if node.sarg.startswith("#") and node.sarg.endswith("#"):
109 extract_gloss_list_item(
110 wxr, word_entry, nested_list_item, sense
111 )
112 else:
113 extract_unordered_list_item(
114 wxr, word_entry, nested_list_item
115 )
116 continue
117 elif isinstance(node, TemplateNode) and node.template_name.endswith(
118 " of"
119 ):
120 extract_form_of_template(wxr, sense, node)
121 gloss_nodes.append(node)
122 elif isinstance(node, TemplateNode) and node.template_name == "라벨":
123 sense.raw_tags.extend(
124 [
125 raw_tag.strip()
126 for raw_tag in clean_node(wxr, sense, node)
127 .strip("()")
128 .split(",")
129 ]
130 )
131 elif isinstance(node, TemplateNode) and node.template_name == "zh-mw": 131 ↛ 132line 131 didn't jump to line 132 because the condition on line 131 was never true
132 extract_zh_mw_template(wxr, node, sense)
133 else:
134 gloss_nodes.append(node)
136 gloss_text = clean_node(wxr, sense, gloss_nodes)
137 if len(gloss_text) > 0:
138 sense.glosses.append(gloss_text)
139 translate_raw_tags(sense)
140 word_entry.senses.append(sense)
143def extract_unordered_list_item(
144 wxr: WiktextractContext, word_entry: WordEntry, list_item: WikiNode
145) -> None:
146 is_first_bold = True
147 for index, node in enumerate(list_item.children):
148 if (
149 isinstance(node, WikiNode)
150 and node.kind == NodeKind.BOLD
151 and is_first_bold
152 ):
153 # `* '''1.''' gloss text`, terrible obsolete layout
154 is_first_bold = False
155 bold_text = clean_node(wxr, None, node)
156 if re.fullmatch(r"\d+(?:-\d+)?\.?", bold_text):
157 new_list_item = WikiNode(NodeKind.LIST_ITEM, 0)
158 new_list_item.children = list_item.children[index + 1 :]
159 extract_gloss_list_item(wxr, word_entry, new_list_item, Sense())
160 break
161 elif isinstance(node, str) and "어원:" in node:
162 etymology_nodes = []
163 etymology_nodes.append(node[node.index(":") + 1 :])
164 etymology_nodes.extend(list_item.children[index + 1 :])
165 e_links: list[tuple[str, str]] = []
166 e_text = clean_node(
167 wxr, None, etymology_nodes, link_collector=e_links
168 )
169 if len(e_text) > 0: 169 ↛ 172line 169 didn't jump to line 172 because the condition on line 169 was always true
170 word_entry.etymology_texts.append(e_text)
171 word_entry.etymology_links.extend(e_links)
172 break
173 elif (
174 isinstance(node, str)
175 and re.search(r"(?:참고|참조|활용):", node) is not None
176 ):
177 note_str = node[node.index(":") + 1 :].strip()
178 note_str += clean_node(
179 wxr,
180 word_entry.senses[-1]
181 if len(word_entry.senses) > 0
182 else word_entry,
183 list_item.children[index + 1 :],
184 )
185 if len(word_entry.senses) > 0:
186 word_entry.senses[-1].note = note_str
187 else:
188 word_entry.note = note_str
189 break
190 elif (
191 isinstance(node, str)
192 and ":" in node
193 and node[: node.index(":")].strip() in LINKAGE_SECTIONS
194 ):
195 extract_linkage_list_item(wxr, word_entry, list_item, "", False)
196 break
197 elif isinstance(node, str) and "문형:" in node:
198 word_entry.pattern = node[node.index(":") + 1 :].strip()
199 word_entry.pattern += clean_node(
200 wxr, None, list_item.children[index + 1 :]
201 )
202 break
203 else:
204 if len(word_entry.senses) > 0:
205 extract_example_list_item(
206 wxr, word_entry.senses[-1], list_item, word_entry.lang_code
207 )
210def extract_form_of_template(
211 wxr: WiktextractContext, sense: Sense, t_node: TemplateNode
212) -> None:
213 if "form-of" not in sense.tags: 213 ↛ 215line 213 didn't jump to line 215 because the condition on line 213 was always true
214 sense.tags.append("form-of")
215 word_arg = 1 if t_node.template_name == "ko-hanja form of" else 2
216 word = clean_node(wxr, None, t_node.template_parameters.get(word_arg, ""))
217 if len(word) > 0: 217 ↛ exitline 217 didn't return from function 'extract_form_of_template' because the condition on line 217 was always true
218 sense.form_of.append(AltForm(word=word))
221def extract_grammar_note_section(
222 wxr: WiktextractContext, word_entry: WordEntry, level_node: LevelNode
223) -> None:
224 for list_item in level_node.find_child_recursively(NodeKind.LIST_ITEM):
225 word_entry.note = clean_node(wxr, None, list_item.children)
228def extract_zh_mw_template(
229 wxr: WiktextractContext, t_node: TemplateNode, sense: Sense
230) -> None:
231 # Chinese inline classifier template
232 # copied from zh edition code
233 expanded_node = wxr.wtp.parse(
234 wxr.wtp.node_to_wikitext(t_node), expand_all=True
235 )
236 classifiers = []
237 last_word = ""
238 for span_tag in expanded_node.find_html_recursively("span"):
239 span_class = span_tag.attrs.get("class", "")
240 if span_class in ["Hani", "Hant", "Hans"]:
241 word = clean_node(wxr, None, span_tag)
242 if word != "/":
243 classifier = Classifier(classifier=word)
244 if span_class == "Hant":
245 classifier.tags.append("Traditional-Chinese")
246 elif span_class == "Hans":
247 classifier.tags.append("Simplified-Chinese")
249 if len(classifiers) > 0 and last_word != "/":
250 sense.classifiers.extend(classifiers)
251 classifiers.clear()
252 classifiers.append(classifier)
253 last_word = word
254 elif "title" in span_tag.attrs:
255 raw_tag = clean_node(wxr, None, span_tag.attrs["title"])
256 if len(raw_tag) > 0:
257 for classifier in classifiers:
258 classifier.raw_tags.append(raw_tag)
259 sense.classifiers.extend(classifiers)
260 for classifier in sense.classifiers:
261 translate_raw_tags(classifier)
264def extract_headword_line_template(
265 wxr: WiktextractContext, word_entry: WordEntry, t_node: TemplateNode
266):
267 forms = []
268 expanded_node = wxr.wtp.parse(
269 wxr.wtp.node_to_wikitext(t_node), expand_all=True
270 )
271 for main_span_tag in expanded_node.find_html(
272 "span", attr_name="class", attr_value="headword-line"
273 ):
274 i_tags = []
275 for html_node in main_span_tag.find_child(NodeKind.HTML):
276 class_names = html_node.attrs.get("class", "").split()
277 if html_node.tag == "strong" and "headword" in class_names:
278 ruby, no_ruby = extract_ruby(wxr, html_node)
279 strong_str = clean_node(wxr, None, no_ruby)
280 if strong_str not in ["", wxr.wtp.title] or len(ruby) > 0:
281 forms.append(
282 Form(form=strong_str, tags=["canonical"], ruby=ruby)
283 )
284 elif html_node.tag == "span":
285 if "headword-tr" in class_names or "tr" in class_names:
286 roman = clean_node(wxr, None, html_node)
287 if (
288 len(forms) > 0
289 and "canonical" not in forms[-1].tags
290 and "romanization" not in forms[-1].tags
291 ):
292 forms[-1].roman = roman
293 elif roman != "": 293 ↛ 275line 293 didn't jump to line 275 because the condition on line 293 was always true
294 forms.append(Form(form=roman, tags=["romanization"]))
295 elif "gender" in class_names: 295 ↛ 296line 295 didn't jump to line 296 because the condition on line 295 was never true
296 for abbr_tag in html_node.find_html("abbr"):
297 gender_tag = clean_node(wxr, None, abbr_tag)
298 if (
299 len(forms) > 0
300 and "canonical" not in forms[-1].tags
301 and "romanization" not in forms[-1].tags
302 ):
303 forms[-1].raw_tags.append(gender_tag)
304 else:
305 word_entry.raw_tags.append(gender_tag)
306 elif "ib-content" in class_names: 306 ↛ 307line 306 didn't jump to line 307 because the condition on line 306 was never true
307 raw_tag = clean_node(wxr, None, html_node)
308 if raw_tag != "":
309 word_entry.raw_tags.append(raw_tag)
310 elif html_node.tag == "sup" and word_entry.lang_code == "ja": 310 ↛ 311line 310 didn't jump to line 311 because the condition on line 310 was never true
311 forms.append(extract_historical_kana(wxr, html_node))
312 elif html_node.tag == "i":
313 if len(i_tags) > 0:
314 word_entry.raw_tags.extend(i_tags)
315 i_tags.clear()
316 for i_child in html_node.children:
317 raw_tag = (
318 clean_node(wxr, None, i_child)
319 .removeprefix("^†")
320 .strip()
321 )
322 if raw_tag != "": 322 ↛ 316line 322 didn't jump to line 316 because the condition on line 322 was always true
323 i_tags.append(raw_tag)
324 elif html_node.tag == "b": 324 ↛ 275line 324 didn't jump to line 275 because the condition on line 324 was always true
325 ruby, no_ruby = extract_ruby(wxr, html_node)
326 for form_str in filter(
327 None,
328 map(str.strip, clean_node(wxr, None, no_ruby).split(",")),
329 ):
330 form = Form(form=form_str, ruby=ruby)
331 if i_tags == ["또는"]: 331 ↛ 332line 331 didn't jump to line 332 because the condition on line 331 was never true
332 if len(forms) > 0:
333 form.raw_tags.extend(forms[-1].raw_tags)
334 else:
335 form.raw_tags.extend(i_tags)
336 forms.append(form)
337 i_tags.clear()
339 if len(i_tags) > 0: 339 ↛ 340line 339 didn't jump to line 340 because the condition on line 339 was never true
340 word_entry.raw_tags.extend(i_tags)
341 for form in forms:
342 translate_raw_tags(form)
343 word_entry.forms.extend(forms)
344 clean_node(wxr, word_entry, expanded_node)
345 translate_raw_tags(word_entry)
348def extract_historical_kana(
349 wxr: WiktextractContext, sup_node: HTMLNode
350) -> Form:
351 form = Form(form="", tags=["archaic"])
352 for strong_node in sup_node.find_html("strong"):
353 form.form = clean_node(wxr, None, strong_node)
354 for span_node in sup_node.find_html(
355 "span", attr_name="class", attr_value="tr"
356 ):
357 form.roman = clean_node(wxr, None, span_node)
358 return form