Coverage for src/wiktextract/extractor/ko/etymology.py: 54%
60 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor import HTMLNode, LevelNode, NodeKind, TemplateNode
3from ...page import clean_node
4from ...wxr_context import WiktextractContext
5from .models import Form, WordEntry
6from .tags import translate_raw_tags
9def extract_etymology_section(
10 wxr: WiktextractContext, word_entry: WordEntry, level_node: LevelNode
11) -> None:
12 if len(word_entry.etymology_texts) > 0:
13 word_entry.etymology_texts.clear()
14 word_entry.etymology_links.clear()
15 word_entry.categories.clear()
17 has_list = False
18 for list_node in level_node.find_child(NodeKind.LIST):
19 has_list = True
20 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
21 e_links: list[tuple[str, str]] = []
22 text = clean_node(
23 wxr, word_entry, list_item.children, link_collector=e_links
24 )
25 if len(text) > 0: 25 ↛ 20line 25 didn't jump to line 20 because the condition on line 25 was always true
26 word_entry.etymology_texts.append(text)
27 word_entry.etymology_links.extend(e_links)
29 if not has_list:
30 e_nodes = []
31 for node in level_node.children:
32 if isinstance(node, TemplateNode) and (
33 node.template_name.endswith("-kanjitab")
34 or node.template_name == "ja-kt"
35 ):
36 extract_ja_kanjitab_template(wxr, node, word_entry)
37 elif isinstance(node, LevelNode):
38 break
39 else:
40 e_nodes.append(node)
42 e_links: list[tuple[str, str]] = []
43 text = clean_node(wxr, word_entry, e_nodes, link_collector=e_links)
44 if len(text) > 0:
45 word_entry.etymology_texts.append(text)
46 word_entry.etymology_links.extend(e_links)
49def extract_ja_kanjitab_template(
50 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry
51):
52 expanded_node = wxr.wtp.parse(
53 wxr.wtp.node_to_wikitext(t_node), expand_all=True
54 )
55 for table in expanded_node.find_child(NodeKind.TABLE): 55 ↛ 56line 55 didn't jump to line 56 because the loop on line 55 never started
56 is_alt_form_table = False
57 for row in table.find_child(NodeKind.TABLE_ROW):
58 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL):
59 header_text = clean_node(wxr, None, header_node)
60 if header_text == "다른 표기":
61 is_alt_form_table = True
62 if not is_alt_form_table:
63 continue
64 forms = []
65 for row in table.find_child(NodeKind.TABLE_ROW):
66 for cell_node in row.find_child(NodeKind.TABLE_CELL):
67 for child_node in cell_node.children:
68 if isinstance(child_node, HTMLNode):
69 if child_node.tag == "span":
70 word = clean_node(wxr, None, child_node)
71 if word != "":
72 forms.append(
73 Form(
74 form=word, tags=["alternative", "kanji"]
75 )
76 )
77 elif child_node.tag == "small":
78 raw_tag = clean_node(wxr, None, child_node).strip(
79 "()"
80 )
81 if raw_tag != "" and len(forms) > 0:
82 forms[-1].raw_tags.append(raw_tag)
83 translate_raw_tags(forms[-1])
84 base_data.forms.extend(forms)
85 for link_node in expanded_node.find_child(NodeKind.LINK):
86 clean_node(wxr, base_data, link_node)