Coverage for src/wiktextract/extractor/ms/page.py: 78%
108 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1import string
2from typing import Any
4from mediawiki_langcodes import name_to_code
5from wikitextprocessor.parser import (
6 LEVEL_KIND_FLAGS,
7 LevelNode,
8 NodeKind,
9 WikiNode,
10)
12from ...page import clean_node
13from ...wxr_context import WiktextractContext
14from .linkage import extract_form_section, extract_linkage_section
15from .models import Sense, WordEntry
16from .pos import extract_pos_section
17from .section_titles import FORM_SECTIONS, LINKAGE_SECTIONS, POS_DATA
18from .sound import extract_sound_section
19from .translation import extract_translation_section
22def parse_section(
23 wxr: WiktextractContext,
24 page_data: list[WordEntry],
25 base_data: WordEntry,
26 level_node: LevelNode,
27) -> None:
28 title_text = clean_node(wxr, None, level_node.largs)
29 wxr.wtp.start_subsection(title_text)
30 title_text = title_text.rstrip(string.digits + string.whitespace + "IVX")
31 lower_title = title_text.lower()
32 if lower_title in POS_DATA:
33 old_data_len = len(page_data)
34 extract_pos_section(wxr, page_data, base_data, level_node, title_text)
35 if len(page_data) == old_data_len and lower_title in LINKAGE_SECTIONS:
36 extract_linkage_section(wxr, page_data, base_data, level_node)
37 elif lower_title == "etimologi":
38 extract_etymology_section(wxr, page_data, base_data, level_node)
39 elif lower_title in FORM_SECTIONS:
40 extract_form_section(
41 wxr,
42 page_data[-1] if len(page_data) > 0 else base_data,
43 level_node,
44 FORM_SECTIONS[lower_title],
45 )
46 elif lower_title == "tesaurus" or lower_title in LINKAGE_SECTIONS:
47 extract_linkage_section(wxr, page_data, base_data, level_node)
48 elif lower_title == "terjemahan":
49 extract_translation_section(wxr, page_data, base_data, level_node)
50 elif lower_title == "sebutan": 50 ↛ 52line 50 didn't jump to line 52 because the condition on line 50 was always true
51 extract_sound_section(wxr, page_data, base_data, level_node)
52 elif lower_title in ["nota penggunaan", "penggunaan"]:
53 extract_note_section(
54 wxr, page_data[-1] if len(page_data) > 0 else base_data, level_node
55 )
56 elif lower_title not in [
57 "pautan luar",
58 "rujukan",
59 "bacaan lanjut",
60 "lihat juga",
61 ]:
62 wxr.wtp.debug(f"Unknown section: {title_text}", sortid="ms/page/44")
64 for next_level in level_node.find_child(LEVEL_KIND_FLAGS):
65 parse_section(wxr, page_data, base_data, next_level)
66 for link_node in level_node.find_child(NodeKind.LINK):
67 clean_node(
68 wxr, page_data[-1] if len(page_data) > 0 else base_data, link_node
69 )
70 for t_node in level_node.find_child(NodeKind.TEMPLATE):
71 if t_node.template_name in ["topik", "C", "topics"]: 71 ↛ 72line 71 didn't jump to line 72 because the condition on line 71 was never true
72 clean_node(
73 wxr, page_data[-1] if len(page_data) > 0 else base_data, t_node
74 )
77def parse_page(
78 wxr: WiktextractContext, page_title: str, page_text: str
79) -> list[dict[str, Any]]:
80 # Page format
81 # https://ms.wiktionary.org/wiki/Wikikamus:Memulakan_laman_baru#Format_laman
82 if page_title.startswith(("Portal:", "Reconstruction:")): 82 ↛ 83line 82 didn't jump to line 83 because the condition on line 82 was never true
83 return []
84 wxr.wtp.start_page(page_title)
85 tree = wxr.wtp.parse(page_text, pre_expand=True)
86 page_data: list[WordEntry] = []
88 for level2_node in tree.find_child(NodeKind.LEVEL2):
89 pre_data_len = len(page_data)
90 lang_name = clean_node(wxr, None, level2_node.largs)
91 lang_code = (
92 name_to_code(lang_name.removeprefix("Bahasa "), "ms") or "unknown"
93 )
94 wxr.wtp.start_section(lang_name)
95 base_data = WordEntry(
96 word=wxr.wtp.title,
97 lang_code=lang_code,
98 lang=lang_name,
99 pos="unknown",
100 )
101 for next_level_node in level2_node.find_child(LEVEL_KIND_FLAGS):
102 parse_section(wxr, page_data, base_data, next_level_node)
103 if len(page_data) == pre_data_len:
104 page_data.append(base_data.model_copy(deep=True))
106 for data in page_data:
107 if len(data.senses) == 0:
108 data.senses.append(Sense(tags=["no-gloss"]))
109 return [m.model_dump(exclude_defaults=True) for m in page_data]
112def extract_etymology_section(
113 wxr: WiktextractContext,
114 page_data: list[WordEntry],
115 base_data: WordEntry,
116 level_node: LevelNode,
117):
118 cats = {}
119 e_nodes = []
120 e_texts = []
121 links: list[tuple[str, str]] = []
122 for node in level_node.children:
123 if isinstance(node, LevelNode): 123 ↛ 124line 123 didn't jump to line 124 because the condition on line 123 was never true
124 break
125 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST:
126 for list_item in node.find_child(NodeKind.LIST_ITEM):
127 e_text = clean_node(
128 wxr, cats, list_item.children, link_collector=links
129 )
130 if e_text != "": 130 ↛ 126line 130 didn't jump to line 126 because the condition on line 130 was always true
131 e_texts.append(e_text)
132 else:
133 e_nodes.append(node)
134 if len(e_nodes) > 0: 134 ↛ 138line 134 didn't jump to line 138 because the condition on line 134 was always true
135 e_text = clean_node(wxr, cats, e_nodes, link_collector=links)
136 if e_text != "":
137 e_texts.append(e_text)
138 if len(e_texts) == 0: 138 ↛ 139line 138 didn't jump to line 139 because the condition on line 138 was never true
139 return
140 if len(page_data) == 0 or page_data[-1].lang_code != base_data.lang_code:
141 base_data.etymology_texts = e_texts
142 base_data.etymology_links = links.copy()
143 base_data.categories.extend(cats.get("categories", []))
144 elif level_node.kind == NodeKind.LEVEL3:
145 for data in page_data:
146 if data.lang_code == page_data[-1].lang_code:
147 data.etymology_texts = e_texts
148 data.etymology_links = links.copy()
149 data.categories.extend(cats.get("categories", []))
150 else:
151 page_data[-1].etymology_texts = e_texts
152 page_data[-1].etymology_links = links.copy()
153 page_data[-1].categories.extend(cats.get("categories", []))
156def extract_note_section(
157 wxr: WiktextractContext, word_entry: WordEntry, level_node: LevelNode
158) -> None:
159 has_list = False
160 for list_node in level_node.find_child(NodeKind.LIST):
161 has_list = True
162 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
163 note = clean_node(wxr, None, list_item.children)
164 if note != "":
165 word_entry.notes.append(note)
166 if not has_list:
167 note = clean_node(wxr, None, level_node.children)
168 if note != "":
169 word_entry.notes.append(note)