Coverage for src/wiktextract/extractor/zh/etymology.py: 58%
73 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor import (
2 HTMLNode,
3 LevelNode,
4 NodeKind,
5 TemplateNode,
6 WikiNode,
7)
9from ...page import clean_node
10from ...wxr_context import WiktextractContext
11from .models import Example, Form, WordEntry
12from .tags import translate_raw_tags
15def extract_etymology_section(
16 wxr: WiktextractContext,
17 page_data: list[WordEntry],
18 base_data: WordEntry,
19 level_node: WikiNode,
20):
21 from .example import extract_template_zh_x
23 e_nodes = []
24 for node in level_node.children:
25 if isinstance(node, TemplateNode) and node.template_name in [
26 "zh-x",
27 "zh-q",
28 ]:
29 for example_data in extract_template_zh_x(
30 wxr, node, Example(text="")
31 ):
32 base_data.etymology_examples.append(example_data)
33 clean_node(wxr, base_data, node)
34 elif isinstance(node, TemplateNode) and node.template_name.lower() in [ 34 ↛ 41line 34 didn't jump to line 41 because the condition on line 34 was never true
35 "rfe", # missing etymology
36 "zh-forms",
37 "zh-wp",
38 "wp",
39 "wikipedia",
40 ]:
41 continue
42 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST:
43 has_zh_x = False
44 for template_node in node.find_child_recursively(NodeKind.TEMPLATE):
45 if template_node.template_name in ["zh-x", "zh-q"]:
46 has_zh_x = True
47 for example_data in extract_template_zh_x(
48 wxr, template_node, Example(text="")
49 ):
50 base_data.etymology_examples.append(example_data)
51 clean_node(wxr, base_data, template_node)
52 if not has_zh_x:
53 for list_item in node.find_child(NodeKind.LIST_ITEM):
54 e_links: list[tuple[str, str]] = []
55 e_text = clean_node(
56 wxr, None, list_item.children, link_collector=e_links
57 )
58 if len(e_text) > 0: 58 ↛ 53line 58 didn't jump to line 53 because the condition on line 58 was always true
59 base_data.etymology_texts.append(e_text)
60 base_data.etymology_links.extend(e_links)
61 elif isinstance(node, TemplateNode) and node.template_name in [ 61 ↛ 66line 61 didn't jump to line 66 because the condition on line 61 was never true
62 "ja-see",
63 "ja-see-kango",
64 "zh-see",
65 ]:
66 from .page import process_soft_redirect_template
68 page_data.append(base_data.model_copy(deep=True))
69 process_soft_redirect_template(wxr, node, page_data[-1])
70 elif isinstance(node, TemplateNode) and (
71 node.template_name.endswith("-kanjitab")
72 or node.template_name == "ja-kt"
73 ):
74 extract_ja_kanjitab_template(wxr, node, base_data)
75 elif isinstance(node, LevelNode):
76 break
77 else:
78 e_nodes.append(node)
80 if len(e_nodes) > 0: 80 ↛ exitline 80 didn't return from function 'extract_etymology_section' because the condition on line 80 was always true
81 e_links: list[tuple[str, str]] = []
82 etymology_text = clean_node(
83 wxr, base_data, e_nodes, link_collector=e_links
84 )
85 if len(etymology_text) > 0:
86 base_data.etymology_texts.append(etymology_text)
87 base_data.etymology_links.extend(e_links)
90def extract_ja_kanjitab_template(
91 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry
92):
93 # https://zh.wiktionary.org/wiki/Template:ja-kanjitab
94 expanded_node = wxr.wtp.parse(
95 wxr.wtp.node_to_wikitext(t_node), expand_all=True
96 )
97 for table in expanded_node.find_child(NodeKind.TABLE): 97 ↛ 98line 97 didn't jump to line 98 because the loop on line 97 never started
98 is_alt_form_table = False
99 for row in table.find_child(NodeKind.TABLE_ROW):
100 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL):
101 header_text = clean_node(wxr, None, header_node)
102 if header_text == "其他表記":
103 is_alt_form_table = True
104 if not is_alt_form_table:
105 continue
106 forms = []
107 for row in table.find_child(NodeKind.TABLE_ROW):
108 for cell_node in row.find_child(NodeKind.TABLE_CELL):
109 for child_node in cell_node.children:
110 if isinstance(child_node, HTMLNode):
111 if child_node.tag == "span":
112 word = clean_node(wxr, None, child_node)
113 if word != "":
114 forms.append(
115 Form(
116 form=word, tags=["alternative", "kanji"]
117 )
118 )
119 elif child_node.tag == "small":
120 raw_tag = clean_node(wxr, None, child_node).strip(
121 "()"
122 )
123 if raw_tag != "" and len(forms) > 0:
124 forms[-1].raw_tags.append(raw_tag)
125 translate_raw_tags(forms[-1])
126 base_data.forms.extend(forms)
127 for link_node in expanded_node.find_child(NodeKind.LINK):
128 clean_node(wxr, base_data, link_node)