Coverage for src/wiktextract/extractor/simple/page.py: 79%

36 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from wikitextprocessor.parser import LEVEL_KIND_FLAGS # , print_tree 

2 

3from wiktextract.page import clean_node 

4from wiktextract.wxr_context import WiktextractContext 

5from wiktextract.wxr_logging import logger 

6 

7from .etymology import process_etym 

8from .models import WordEntry 

9from .pos import process_pos 

10from .pronunciation import process_pron 

11from .section_titles import POS_DATA 

12from .text_utils import POS_ENDING_NUMBER_RE 

13 

14# ========================= 

15# Simple English Wiktionary 

16# ========================= 

17 

18# Every Wiktionary is different from others, and Simple English Wiktionary 

19# is no different; but it is usually pretty simple and reasonably regular 

20# in its structure, small enough so that its extractor doesn't get completely 

21# out of hand, and in English so that you can follow along much more easily, 

22# which is why it works well as an example case. 

23 

24# Every extractor has a subfolder in src/wiktextract/extractor that contains 

25# meta-data, config files and source code for that particular extractor. The 

26# main Wiktextract code calls these modules by loading them dynamically using 

27# importlib, then calls parse_page(); this is the main entry point for your 

28# extractor data. 

29 

30# WordEntry is a Pydantic model (or just a dict in the original English 

31# extractor) that will eventually be outputted as a json object; each 

32# entry is, in general, either a redirect page (which you do not have to 

33# worry about) or a Part of Speech section's data. If a word has a `Noun` 

34# section and an `Adjective` section, they're processed into different entries; 

35# different Etymologies are also separate and can have several entries within 

36# them. 

37 

38# WordEntries are composed of smaller data sections, like lists of Senses, 

39# Sounds and Etymology data. Your extractor does not need to handle everything, 

40# and there is rarely any need to add new fields (because often they already 

41# exist in the English extractor), but it is possible to add them if needed; 

42# please check first with an issue to keep everyone in synch. 

43 

44 

45def parse_page( 

46 wxr: WiktextractContext, page_title: str, page_text: str 

47) -> list[dict[str, WordEntry]]: 

48 """Parse Simple English Wiktionary page.""" 

49 # Unlike other wiktionaries, Simple Wikt. has a limited scope: only English 

50 # words. The pages also tend to be much shorter and simpler in structure, 

51 # which makes parsing them much easier. 

52 # https://simple.wiktionary.org/wiki/Wiktionary:Entry_layout_explained 

53 

54 # Usually things like this are handled by checking a page's namespace 

55 # code; in SEW's case, Main Page and the Appendices share the same 

56 # namespace as the word articles. 

57 if page_title == "Main Page" or page_title.startswith("Appendix:"): 57 ↛ 58line 57 didn't jump to line 58 because the condition on line 57 was never true

58 return [] 

59 

60 # "Parsing page: " is the only place wxr.config.verbose appears. 

61 # XXX use wxr.config.verbose more often! 

62 if wxr.config.verbose: 62 ↛ 63line 62 didn't jump to line 63 because the condition on line 62 was never true

63 logger.info(f"Parsing page: {page_title}") 

64 

65 # In a larger edition, we might need to handle more complex titles that have 

66 # been changed due to the restrictions of WikiMedia article names. 

67 wxr.config.word = page_title 

68 wxr.wtp.start_page(page_title) 

69 

70 ##### Debug printing ##### 

71 # Temporary debug print stuff without page parsing went into debug_bypass. 

72 # If you're running a full extraction to find data, multiprocessing 

73 # *will* sometimes mess up your prints by overlaying some prints 

74 # with others. For quick and dirty stuff this might not matter, but if 

75 # you really want to be sure to get good prints you need to use 

76 # something like the logging package, which here is represented by 

77 # `logger`; the only annoyance is that it's not easy to get rid of 

78 # the datetime string at the start on the fly, so I just do it crudely 

79 # by inserting a `\n` in the message after `f"{wxr.wtp.title}..."`. 

80 

81 # from .debug_bypass import debug_bypass 

82 # return debug_bypass(wxr, page_title, page_text) 

83 

84 ##### 

85 

86 ##### Main page parse ##### 

87 # `Wtp.parse()` (Wtp being the main context class of Wikitextprocessor 

88 # imbedded into the Wiktextract context object) takes wikitext and 

89 # returns a tree of parsed nodes: strings and WikiNodes. 

90 

91 # `Wtp.parse()` returns a `NodeKind.ROOT` node, which has as its children 

92 # everything else that was parsed from the string. 

93 page_root = wxr.wtp.parse( 

94 page_text, 

95 ) 

96 

97 # print_tree(page_root) 

98 

99 # If this was a normal wiktionary edition, this is where you would split 

100 # the page into different language entries; "English", "Gaelic", "Swahili", 

101 # etc. 

102 # Each section would be a Level 2 node (`NodeKind.LEVEL2`) child of the 

103 # page's root node. Wiktionaries do not seem to use LEVEL1 for anything, 

104 # although there might be (there definitely is...) an exception somewhere. 

105 

106 # Some data that is collected is shared among several entries; things like 

107 # pronunciation data and etymological data is collected and updated in this 

108 # base_data object that can be copied for more specific entries. 

109 # Simple English Wiktionary has these sections under Level 3 ("===") 

110 # headers, which put them in an awkward hierarchy with the main page and 

111 # Part-of-Speech sections, which are higher; they usually appear before 

112 # POS sections, but a few pages have a more complex structure. This 

113 # is handled by simply flattening the parse tree later, and handling 

114 # sections in a linear order. 

115 # ║ Noun Section 

116 # ╚═ Etymology Data For Next Section 

117 # ║ Next section 

118 # ╚═Pronunciation data for third section 

119 # ║ Third section. 

120 base_data = WordEntry( 

121 word=page_title, 

122 # Simple English wiktionary entries are only for English words, 

123 # so it's atypical in that way. 

124 lang_code="en", 

125 lang="English", 

126 pos="ERROR_UNKNOWN_POS", 

127 ) 

128 

129 # This is our return list. 

130 word_data: list[WordEntry] = [] 

131 

132 for level in page_root.find_child_recursively(LEVEL_KIND_FLAGS): 

133 # Ignore everything outside of a section with a heading; there shouldn't 

134 # be anything there. Previous version of this code looked through that 

135 # stuff and would spit out warnings, which can be useful when figuring 

136 # out what kind of stuff there can be found in an article. 

137 

138 # .find_child_recursively() (in contrast with find_child()) yields 

139 # a flattened tree, not just direct children of root 

140 

141 # clean_node() is the general purpose WikiNode/string -> string 

142 # implementation. Things like formatting are stripped; it mimics 

143 # the output of wikitext when possible. 

144 # WikiNodes have a list of lists of anything as their "arguments", 

145 # if applicable. Some WikiNode subclasses have .sarg (string argument) 

146 # instead to make things easier. Mainly this stems from stuff like 

147 # Template nodes having their arguments parsed as nodes themselves; 

148 # each `|` separated parameter is a list of nodes. 

149 heading_title = clean_node(wxr, None, level.largs[0]).lower() 

150 # print(f"=== {heading_title=}") 

151 

152 # Sometimes headings in SEW have a number at the end ("Noun 2") 

153 if m := POS_ENDING_NUMBER_RE.search(heading_title): 153 ↛ 154line 153 didn't jump to line 154 because the condition on line 153 was never true

154 pos_num = int(m.group(0).strip()) 

155 heading_title = heading_title[: m.start()] 

156 else: 

157 pos_num = -1 # default: see models.py/Sense 

158 if heading_title in POS_DATA: 

159 pos_data = base_data.model_copy(deep=True) 

160 new_data = process_pos(wxr, level, pos_data, heading_title, pos_num) 

161 if new_data is not None: 161 ↛ 132line 161 didn't jump to line 132 because the condition on line 161 was always true

162 # new_data would be one WordEntry object, for one Part of 

163 # Speech section ("Noun", "Verb"); this is generally how we 

164 # want it. 

165 word_data.append(new_data) 

166 else: 

167 # Process pronunciation and etym sections. 

168 # Ignore other sections, like 'Description' 

169 # On Simple Wiktionary, these appear as level-3 nodes under the 

170 # **previous** POS node; that's why we flatten everything with the 

171 # recursive iterator. At least these are in "order". 

172 if heading_title.startswith("pronunciation"): 172 ↛ 174line 172 didn't jump to line 174 because the condition on line 172 was never true

173 # Replace sound data in target_data with new data, if applicable 

174 process_pron(wxr, level, base_data) 

175 elif heading_title.startswith(("etymology", "word parts")): 175 ↛ 132line 175 didn't jump to line 132 because the condition on line 175 was always true

176 # Replace etymology data in target_data with new data, if 

177 # applicable 

178 process_etym(wxr, level, base_data) 

179 

180 # Transform pydantic objects to normal dicts so that the old code can 

181 # handle them. 

182 return [wd.model_dump(exclude_defaults=True) for wd in word_data] 

183 # return [base_data.model_dump(exclude_defaults=True)]