Coverage for src/wiktextract/extractor/th/etymology.py: 48%
53 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor import (
2 HTMLNode,
3 LevelNode,
4 NodeKind,
5 TemplateNode,
6 WikiNode,
7)
9from ...page import clean_node
10from ...wxr_context import WiktextractContext
11from .models import Form, WordEntry
12from .tags import translate_raw_tags
15def extract_etymology_section(
16 wxr: WiktextractContext, base_data: WordEntry, level_node: LevelNode
17):
18 e_nodes = []
19 for node in level_node.children:
20 if isinstance(node, TemplateNode) and (
21 node.template_name.endswith("-kanjitab")
22 or node.template_name == "ja-kt"
23 ):
24 extract_ja_kanjitab_template(wxr, node, base_data)
25 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST:
26 for list_item in node.find_child(NodeKind.LIST_ITEM):
27 e_links: list[tuple[str, str]] = []
28 e_text = clean_node(
29 wxr, base_data, list_item.children, link_collector=e_links
30 )
31 if e_text != "": 31 ↛ 26line 31 didn't jump to line 26 because the condition on line 31 was always true
32 base_data.etymology_texts.append(e_text)
33 base_data.etymology_links.extend(e_links)
34 elif not (
35 isinstance(node, LevelNode)
36 or (
37 isinstance(node, TemplateNode)
38 and node.template_name in ["ja-see", "ja-see-kango"]
39 )
40 ):
41 e_nodes.append(node)
43 if len(e_nodes) > 0: 43 ↛ exitline 43 didn't return from function 'extract_etymology_section' because the condition on line 43 was always true
44 e_links: list[tuple[str, str]] = []
45 e_str = clean_node(wxr, base_data, e_nodes, link_collector=e_links)
46 if e_str != "":
47 base_data.etymology_texts.append(e_str)
48 base_data.etymology_links.extend(e_links)
51def extract_ja_kanjitab_template(
52 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry
53):
54 # https://th.wiktionary.org/wiki/Template:ja-kanjitab
55 expanded_node = wxr.wtp.parse(
56 wxr.wtp.node_to_wikitext(t_node), expand_all=True
57 )
58 for table in expanded_node.find_child(NodeKind.TABLE): 58 ↛ 59line 58 didn't jump to line 59 because the loop on line 58 never started
59 is_alt_form_table = False
60 for row in table.find_child(NodeKind.TABLE_ROW):
61 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL):
62 header_text = clean_node(wxr, None, header_node)
63 if header_text.startswith("การสะกดแบบอื่น"):
64 is_alt_form_table = True
65 if not is_alt_form_table:
66 continue
67 forms = []
68 for row in table.find_child(NodeKind.TABLE_ROW):
69 for cell_node in row.find_child(NodeKind.TABLE_CELL):
70 for child_node in cell_node.children:
71 if isinstance(child_node, HTMLNode):
72 if child_node.tag == "span":
73 word = clean_node(wxr, None, child_node)
74 if word != "":
75 forms.append(
76 Form(
77 form=word, tags=["alternative", "kanji"]
78 )
79 )
80 elif child_node.tag == "small":
81 raw_tag = clean_node(wxr, None, child_node).strip(
82 "()"
83 )
84 if raw_tag != "" and len(forms) > 0:
85 forms[-1].raw_tags.append(raw_tag)
86 translate_raw_tags(forms[-1])
87 base_data.forms.extend(forms)
88 for link_node in expanded_node.find_child(NodeKind.LINK):
89 clean_node(wxr, base_data, link_node)