Coverage for src/wiktextract/extractor/simple/page.py: 79%
36 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor.parser import LEVEL_KIND_FLAGS # , print_tree
3from wiktextract.page import clean_node
4from wiktextract.wxr_context import WiktextractContext
5from wiktextract.wxr_logging import logger
7from .etymology import process_etym
8from .models import WordEntry
9from .pos import process_pos
10from .pronunciation import process_pron
11from .section_titles import POS_DATA
12from .text_utils import POS_ENDING_NUMBER_RE
14# =========================
15# Simple English Wiktionary
16# =========================
18# Every Wiktionary is different from others, and Simple English Wiktionary
19# is no different; but it is usually pretty simple and reasonably regular
20# in its structure, small enough so that its extractor doesn't get completely
21# out of hand, and in English so that you can follow along much more easily,
22# which is why it works well as an example case.
24# Every extractor has a subfolder in src/wiktextract/extractor that contains
25# meta-data, config files and source code for that particular extractor. The
26# main Wiktextract code calls these modules by loading them dynamically using
27# importlib, then calls parse_page(); this is the main entry point for your
28# extractor data.
30# WordEntry is a Pydantic model (or just a dict in the original English
31# extractor) that will eventually be outputted as a json object; each
32# entry is, in general, either a redirect page (which you do not have to
33# worry about) or a Part of Speech section's data. If a word has a `Noun`
34# section and an `Adjective` section, they're processed into different entries;
35# different Etymologies are also separate and can have several entries within
36# them.
38# WordEntries are composed of smaller data sections, like lists of Senses,
39# Sounds and Etymology data. Your extractor does not need to handle everything,
40# and there is rarely any need to add new fields (because often they already
41# exist in the English extractor), but it is possible to add them if needed;
42# please check first with an issue to keep everyone in synch.
45def parse_page(
46 wxr: WiktextractContext, page_title: str, page_text: str
47) -> list[dict[str, WordEntry]]:
48 """Parse Simple English Wiktionary page."""
49 # Unlike other wiktionaries, Simple Wikt. has a limited scope: only English
50 # words. The pages also tend to be much shorter and simpler in structure,
51 # which makes parsing them much easier.
52 # https://simple.wiktionary.org/wiki/Wiktionary:Entry_layout_explained
54 # Usually things like this are handled by checking a page's namespace
55 # code; in SEW's case, Main Page and the Appendices share the same
56 # namespace as the word articles.
57 if page_title == "Main Page" or page_title.startswith("Appendix:"): 57 ↛ 58line 57 didn't jump to line 58 because the condition on line 57 was never true
58 return []
60 # "Parsing page: " is the only place wxr.config.verbose appears.
61 # XXX use wxr.config.verbose more often!
62 if wxr.config.verbose: 62 ↛ 63line 62 didn't jump to line 63 because the condition on line 62 was never true
63 logger.info(f"Parsing page: {page_title}")
65 # In a larger edition, we might need to handle more complex titles that have
66 # been changed due to the restrictions of WikiMedia article names.
67 wxr.config.word = page_title
68 wxr.wtp.start_page(page_title)
70 ##### Debug printing #####
71 # Temporary debug print stuff without page parsing went into debug_bypass.
72 # If you're running a full extraction to find data, multiprocessing
73 # *will* sometimes mess up your prints by overlaying some prints
74 # with others. For quick and dirty stuff this might not matter, but if
75 # you really want to be sure to get good prints you need to use
76 # something like the logging package, which here is represented by
77 # `logger`; the only annoyance is that it's not easy to get rid of
78 # the datetime string at the start on the fly, so I just do it crudely
79 # by inserting a `\n` in the message after `f"{wxr.wtp.title}..."`.
81 # from .debug_bypass import debug_bypass
82 # return debug_bypass(wxr, page_title, page_text)
84 #####
86 ##### Main page parse #####
87 # `Wtp.parse()` (Wtp being the main context class of Wikitextprocessor
88 # imbedded into the Wiktextract context object) takes wikitext and
89 # returns a tree of parsed nodes: strings and WikiNodes.
91 # `Wtp.parse()` returns a `NodeKind.ROOT` node, which has as its children
92 # everything else that was parsed from the string.
93 page_root = wxr.wtp.parse(
94 page_text,
95 )
97 # print_tree(page_root)
99 # If this was a normal wiktionary edition, this is where you would split
100 # the page into different language entries; "English", "Gaelic", "Swahili",
101 # etc.
102 # Each section would be a Level 2 node (`NodeKind.LEVEL2`) child of the
103 # page's root node. Wiktionaries do not seem to use LEVEL1 for anything,
104 # although there might be (there definitely is...) an exception somewhere.
106 # Some data that is collected is shared among several entries; things like
107 # pronunciation data and etymological data is collected and updated in this
108 # base_data object that can be copied for more specific entries.
109 # Simple English Wiktionary has these sections under Level 3 ("===")
110 # headers, which put them in an awkward hierarchy with the main page and
111 # Part-of-Speech sections, which are higher; they usually appear before
112 # POS sections, but a few pages have a more complex structure. This
113 # is handled by simply flattening the parse tree later, and handling
114 # sections in a linear order.
115 # ║ Noun Section
116 # ╚═ Etymology Data For Next Section
117 # ║ Next section
118 # ╚═Pronunciation data for third section
119 # ║ Third section.
120 base_data = WordEntry(
121 word=page_title,
122 # Simple English wiktionary entries are only for English words,
123 # so it's atypical in that way.
124 lang_code="en",
125 lang="English",
126 pos="ERROR_UNKNOWN_POS",
127 )
129 # This is our return list.
130 word_data: list[WordEntry] = []
132 for level in page_root.find_child_recursively(LEVEL_KIND_FLAGS):
133 # Ignore everything outside of a section with a heading; there shouldn't
134 # be anything there. Previous version of this code looked through that
135 # stuff and would spit out warnings, which can be useful when figuring
136 # out what kind of stuff there can be found in an article.
138 # .find_child_recursively() (in contrast with find_child()) yields
139 # a flattened tree, not just direct children of root
141 # clean_node() is the general purpose WikiNode/string -> string
142 # implementation. Things like formatting are stripped; it mimics
143 # the output of wikitext when possible.
144 # WikiNodes have a list of lists of anything as their "arguments",
145 # if applicable. Some WikiNode subclasses have .sarg (string argument)
146 # instead to make things easier. Mainly this stems from stuff like
147 # Template nodes having their arguments parsed as nodes themselves;
148 # each `|` separated parameter is a list of nodes.
149 heading_title = clean_node(wxr, None, level.largs[0]).lower()
150 # print(f"=== {heading_title=}")
152 # Sometimes headings in SEW have a number at the end ("Noun 2")
153 if m := POS_ENDING_NUMBER_RE.search(heading_title): 153 ↛ 154line 153 didn't jump to line 154 because the condition on line 153 was never true
154 pos_num = int(m.group(0).strip())
155 heading_title = heading_title[: m.start()]
156 else:
157 pos_num = -1 # default: see models.py/Sense
158 if heading_title in POS_DATA:
159 pos_data = base_data.model_copy(deep=True)
160 new_data = process_pos(wxr, level, pos_data, heading_title, pos_num)
161 if new_data is not None: 161 ↛ 132line 161 didn't jump to line 132 because the condition on line 161 was always true
162 # new_data would be one WordEntry object, for one Part of
163 # Speech section ("Noun", "Verb"); this is generally how we
164 # want it.
165 word_data.append(new_data)
166 else:
167 # Process pronunciation and etym sections.
168 # Ignore other sections, like 'Description'
169 # On Simple Wiktionary, these appear as level-3 nodes under the
170 # **previous** POS node; that's why we flatten everything with the
171 # recursive iterator. At least these are in "order".
172 if heading_title.startswith("pronunciation"): 172 ↛ 174line 172 didn't jump to line 174 because the condition on line 172 was never true
173 # Replace sound data in target_data with new data, if applicable
174 process_pron(wxr, level, base_data)
175 elif heading_title.startswith(("etymology", "word parts")): 175 ↛ 132line 175 didn't jump to line 132 because the condition on line 175 was always true
176 # Replace etymology data in target_data with new data, if
177 # applicable
178 process_etym(wxr, level, base_data)
180 # Transform pydantic objects to normal dicts so that the old code can
181 # handle them.
182 return [wd.model_dump(exclude_defaults=True) for wd in word_data]
183 # return [base_data.model_dump(exclude_defaults=True)]