Coverage for src/wiktextract/extractor/en/page.py: 79%
1845 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1# Code for parsing information from a single Wiktionary page.
2#
3# Copyright (c) 2018-2022 Tatu Ylonen. See file LICENSE and https://ylonen.org
5import copy
6import html
7import re
8from collections import defaultdict
9from functools import partial
10from typing import (
11 TYPE_CHECKING,
12 Any,
13 Iterable,
14 Literal,
15 Optional,
16 Set,
17 Union,
18 cast,
19)
21from mediawiki_langcodes import get_all_names, name_to_code
22from wikitextprocessor.core import TemplateArgs, TemplateFnCallable
23from wikitextprocessor.parser import (
24 LEVEL_KIND_FLAGS,
25 GeneralNode,
26 HTMLNode,
27 LevelNode,
28 NodeKind,
29 TemplateNode,
30 WikiNode,
31)
33from ...clean import clean_template_args, clean_value
34from ...datautils import (
35 data_append,
36 data_extend,
37 ns_title_prefix_tuple,
38)
39from ...page import (
40 LEVEL_KINDS,
41 clean_node,
42 is_panel_template,
43 recursively_extract,
44)
45from ...tags import valid_tags
46from ...wxr_context import WiktextractContext
47from ...wxr_logging import logger
48from ..ruby import extract_ruby, parse_ruby
49from ..share import strip_nodes
50from .descendant import (
51 ETYMOLOGY_TEMPLATES_IN_HEADS,
52 etymology_template_append,
53 extract_descendant_section,
54)
55from .example import extract_example_list_item, extract_template_zh_x
56from .form_descriptions import (
57 classify_desc,
58 decode_tags,
59 distw,
60 parse_alt_or_inflection_of,
61 parse_sense_qualifier,
62 parse_word_head,
63)
64from .inflection import TableContext, parse_inflection_section
65from .info_templates import (
66 INFO_TEMPLATE_FUNCS,
67 parse_info_template_arguments,
68 parse_info_template_node,
69)
70from .linkages import (
71 extract_alt_form_section,
72 parse_linkage,
73)
74from .parts_of_speech import PARTS_OF_SPEECH
75from .section_titles import (
76 COMPOUNDS_TITLE,
77 DESCENDANTS_TITLE,
78 ETYMOLOGY_TITLES,
79 IGNORED_TITLES,
80 INFLECTION_TITLES,
81 LINKAGE_TITLES,
82 POS_TITLES,
83 PRONUNCIATION_TITLE,
84 PROTO_ROOT_DERIVED_TITLES,
85 TRANSLATIONS_TITLE,
86)
87from .translations import parse_translation_item_text
88from .type_utils import (
89 AttestationData,
90 ExampleData,
91 FormData,
92 LinkageData,
93 ReferenceData,
94 SenseData,
95 SoundData,
96 TemplateData,
97 WordData,
98)
99from .unsupported_titles import unsupported_title_map
101# When determining whether a string is 'english', classify_desc
102# might return 'taxonomic' which is English text 99% of the time.
103ENGLISH_TEXTS = ("english", "taxonomic")
105# Matches head tag
106HEAD_TAG_RE = re.compile(
107 r"^(head|Han char|arabic-noun|arabic-noun-form|"
108 r"hangul-symbol|syllable-hangul)$|"
109 + r"^(latin|"
110 + "|".join(lang_code for lang_code, *_ in get_all_names("en"))
111 + r")-("
112 + "|".join(
113 [
114 "abbr",
115 "adj",
116 "adjective",
117 "adjective form",
118 "adjective-form",
119 "adv",
120 "adverb",
121 "affix",
122 "animal command",
123 "art",
124 "article",
125 "aux",
126 "bound pronoun",
127 "bound-pronoun",
128 "Buyla",
129 "card num",
130 "card-num",
131 "cardinal",
132 "chunom",
133 "classifier",
134 "clitic",
135 "cls",
136 "cmene",
137 "cmavo",
138 "colloq-verb",
139 "colverbform",
140 "combining form",
141 "combining-form",
142 "comparative",
143 "con",
144 "concord",
145 "conj",
146 "conjunction",
147 "conjug",
148 "cont",
149 "contr",
150 "converb",
151 "daybox",
152 "decl",
153 "decl noun",
154 "def",
155 "dem",
156 "det",
157 "determ",
158 "Deva",
159 "ending",
160 "entry",
161 "form",
162 "fuhivla",
163 "gerund",
164 "gismu",
165 "hanja",
166 "hantu",
167 "hanzi",
168 "head",
169 "ideophone",
170 "idiom",
171 "inf",
172 "indef",
173 "infixed pronoun",
174 "infixed-pronoun",
175 "infl",
176 "inflection",
177 "initialism",
178 "int",
179 "interfix",
180 "interj",
181 "interjection",
182 "jyut",
183 "latin",
184 "letter",
185 "locative",
186 "lujvo",
187 "monthbox",
188 "mutverb",
189 "name",
190 "nisba",
191 "nom",
192 "noun",
193 "noun form",
194 "noun-form",
195 "noun plural",
196 "noun-plural",
197 "nounprefix",
198 "num",
199 "number",
200 "numeral",
201 "ord",
202 "ordinal",
203 "par",
204 "part",
205 "part form",
206 "part-form",
207 "participle",
208 "particle",
209 "past",
210 "past neg",
211 "past-neg",
212 "past participle",
213 "past-participle",
214 "perfect participle",
215 "perfect-participle",
216 "personal pronoun",
217 "personal-pronoun",
218 "pref",
219 "prefix",
220 "phrase",
221 "pinyin",
222 "plural noun",
223 "plural-noun",
224 "pos",
225 "poss-noun",
226 "post",
227 "postp",
228 "postposition",
229 "PP",
230 "pp",
231 "ppron",
232 "pred",
233 "predicative",
234 "prep",
235 "prep phrase",
236 "prep-phrase",
237 "preposition",
238 "present participle",
239 "present-participle",
240 "pron",
241 "prondem",
242 "pronindef",
243 "pronoun",
244 "prop",
245 "proper noun",
246 "proper-noun",
247 "proper noun form",
248 "proper-noun form",
249 "proper noun-form",
250 "proper-noun-form",
251 "prov",
252 "proverb",
253 "prpn",
254 "prpr",
255 "punctuation mark",
256 "punctuation-mark",
257 "regnoun",
258 "rel",
259 "rom",
260 "romanji",
261 "root",
262 "sign",
263 "suff",
264 "suffix",
265 "syllable",
266 "symbol",
267 "verb",
268 "verb form",
269 "verb-form",
270 "verbal noun",
271 "verbal-noun",
272 "verbnec",
273 "vform",
274 ]
275 )
276 + r")(-|/|\+|$)"
277)
279# Head-templates causing problems (like newlines) that can be squashed into
280# an empty string in the template handler while saving their template
281# data for later.
282WORD_LEVEL_HEAD_TEMPLATES = {"term-label", "tlb"}
285PROBLEMATIC_TEMPLATES_CLUMP = (
286 WORD_LEVEL_HEAD_TEMPLATES | ETYMOLOGY_TEMPLATES_IN_HEADS
287)
289FLOATING_TABLE_TEMPLATES: set[str] = {
290 # az-suffix-form creates a style=floatright div that is otherwise
291 # deleted; if it is not pre-expanded, we can intercept the template
292 # so we add this set into do_not_pre_expand, and intercept the
293 # templates in parse_part_of_speech
294 "az-suffix-forms",
295 "az-inf-p",
296 "kk-suffix-forms",
297 "ky-suffix-forms",
298 "tr-inf-p",
299 "tr-suffix-forms",
300 "tt-suffix-forms",
301 "uz-suffix-forms",
302}
303# These two should contain template names that should always be
304# pre-expanded when *first* processing the tree, or not pre-expanded
305# so that the template are left in place with their identifying
306# name intact for later filtering.
308DO_NOT_PRE_EXPAND_TEMPLATES: set[str] = set()
309DO_NOT_PRE_EXPAND_TEMPLATES.update(FLOATING_TABLE_TEMPLATES)
311# Additional templates to be expanded in the pre-expand phase
312ADDITIONAL_EXPAND_TEMPLATES: set[str] = {
313 "multitrans",
314 "multitrans-nowiki",
315 "trans-top",
316 "trans-top-also",
317 "trans-bottom",
318 "checktrans-top",
319 "checktrans-bottom",
320 "col",
321 "col1",
322 "col2",
323 "col3",
324 "col4",
325 "col5",
326 "col1-u",
327 "col2-u",
328 "col3-u",
329 "col4-u",
330 "col5-u",
331 "check deprecated lang param usage",
332 "deprecated code",
333 "ru-verb-alt-ё",
334 "ru-noun-alt-ё",
335 "ru-adj-alt-ё",
336 "ru-proper noun-alt-ё",
337 "ru-pos-alt-ё",
338 "ru-alt-ё",
339 "inflection of",
340 "no deprecated lang param usage",
341 "transclude", # these produce sense entries (or other lists)
342 "tcl",
343}
345# Inverse linkage for those that have them
346linkage_inverses: dict[str, str] = {
347 # XXX this is not currently used, move to post-processing
348 "synonyms": "synonyms",
349 "hypernyms": "hyponyms",
350 "hyponyms": "hypernyms",
351 "holonyms": "meronyms",
352 "meronyms": "holonyms",
353 "derived": "derived_from",
354 "coordinate_terms": "coordinate_terms",
355 "troponyms": "hypernyms",
356 "antonyms": "antonyms",
357 "instances": "instance_of",
358 "related": "related",
359}
361# Templates that are used to form panels on pages and that
362# should be ignored in various positions
363PANEL_TEMPLATES: set[str] = {
364 "Character info",
365 "CJKV",
366 "French personal pronouns",
367 "French possessive adjectives",
368 "French possessive pronouns",
369 "Han etym",
370 "Han etyl", # this redirects to Han etym and would cause Lua errors,
371 # and I don't know why, but I'm putting it here because
372 # we should be ignoring it anyhow.
373 "Japanese demonstratives",
374 "Latn-script",
375 "LDL",
376 "MW1913Abbr",
377 "Number-encoding",
378 "Nuttall",
379 "Spanish possessive adjectives",
380 "Spanish possessive pronouns",
381 "USRegionDisputed",
382 "Webster 1913",
383 "ase-rfr",
384 "attention",
385 "attn",
386 "beer",
387 "broken ref",
388 "ca-compass",
389 "character info",
390 "character info/var",
391 "checksense",
392 "compass-fi",
393 "copyvio suspected",
394 "delete",
395 "dial syn", # Currently ignore these, but could be useful in Chinese/Korean
396 "etystub",
397 "examples",
398 "hu-corr",
399 "hu-suff-pron",
400 "interwiktionary",
401 "ja-kanjitab",
402 "ja-kt",
403 "ko-hanja-search",
404 "look",
405 "maintenance box",
406 "maintenance line",
407 "mediagenic terms",
408 "merge",
409 "missing template",
410 "morse links",
411 "move",
412 "multiple images",
413 "no inline",
414 "picdic",
415 "picdicimg",
416 "picdiclabel",
417 "polyominoes",
418 "predidential nomics",
419 "punctuation", # This actually gets pre-expanded
420 "reconstructed",
421 "request box",
422 "rf-sound example",
423 "rfaccents",
424 "rfap",
425 "rfaspect",
426 "rfc",
427 "rfc-auto",
428 "rfc-header",
429 "rfc-level",
430 "rfc-pron-n",
431 "rfc-sense",
432 "rfclarify",
433 "rfd",
434 "rfd-redundant",
435 "rfd-sense",
436 "rfdate",
437 "rfdatek",
438 "rfdef",
439 "rfe",
440 "rfe/dowork",
441 "rfex",
442 "rfexp",
443 "rfform",
444 "rfgender",
445 "rfi",
446 "rfinfl",
447 "rfm",
448 "rfm-sense",
449 "rfp",
450 "rfp-old",
451 "rfquote",
452 "rfquote-sense",
453 "rfquotek",
454 "rfref",
455 "rfscript",
456 "rft2",
457 "rftaxon",
458 "rftone",
459 "rftranslit",
460 "rfv",
461 "rfv-etym",
462 "rfv-pron",
463 "rfv-quote",
464 "rfv-sense",
465 "selfref",
466 "split",
467 "stroke order", # XXX consider capturing this?
468 "stub entry",
469 "t-needed",
470 "tbot entry",
471 "tea room",
472 "tea room sense",
473 # "ttbc", - XXX needed in at least on/Preposition/Translation page
474 "unblock",
475 "unsupportedpage",
476 "video frames",
477 "was wotd",
478 "wrongtitle",
479 "zh-forms",
480 "zh-hanzi-box",
481 "no entry",
482}
484# Template name prefixes used for language-specific panel templates (i.e.,
485# templates that create side boxes or notice boxes or that should generally
486# be ignored).
487PANEL_PREFIXES: set[str] = {
488 "list:compass points/",
489 "list:Gregorian calendar months/",
490 "RQ:",
491}
493# Templates used for wikipedia links.
494wikipedia_templates: set[str] = {
495 "wikipedia",
496 "slim-wikipedia",
497 "w",
498 "W",
499 "swp",
500 "wiki",
501 "Wikipedia",
502 "wtorw",
503}
504for x in PANEL_PREFIXES & wikipedia_templates: 504 ↛ 505line 504 didn't jump to line 505 because the loop on line 504 never started
505 print(
506 "WARNING: {!r} in both panel_templates and wikipedia_templates".format(
507 x
508 )
509 )
511# Mapping from a template name (without language prefix) for the main word
512# (e.g., fi-noun, fi-adj, en-verb) to permitted parts-of-speech in which
513# it could validly occur. This is used as just a sanity check to give
514# warnings about probably incorrect coding in Wiktionary.
515template_allowed_pos_map: dict[str, list[str]] = {
516 "abbr": ["abbrev"],
517 "noun": ["noun", "abbrev", "pron", "name", "num", "adj_noun"],
518 "plural noun": ["noun", "name"],
519 "plural-noun": ["noun", "name"],
520 "proper noun": ["noun", "name"],
521 "proper-noun": ["name", "noun"],
522 "prop": ["name", "noun"],
523 "verb": ["verb", "phrase"],
524 "gerund": ["verb"],
525 "particle": ["adv", "particle"],
526 "adj": ["adj", "adj_noun"],
527 "pron": ["pron", "noun"],
528 "name": ["name", "noun"],
529 "adv": ["adv", "intj", "conj", "particle"],
530 "phrase": ["phrase", "prep_phrase"],
531 "noun phrase": ["phrase"],
532 "ordinal": ["num"],
533 "number": ["num"],
534 "pos": ["affix", "name", "num"],
535 "suffix": ["suffix", "affix"],
536 "character": ["character"],
537 "letter": ["character"],
538 "kanji": ["character"],
539 "cont": ["abbrev"],
540 "interj": ["intj"],
541 "con": ["conj"],
542 "part": ["particle"],
543 "prep": ["prep", "postp"],
544 "postp": ["postp"],
545 "misspelling": ["noun", "adj", "verb", "adv"],
546 "part-form": ["verb"],
547}
548for k, v in template_allowed_pos_map.items():
549 for x in v:
550 if x not in PARTS_OF_SPEECH: 550 ↛ 551line 550 didn't jump to line 551 because the condition on line 550 was never true
551 print(
552 "BAD PART OF SPEECH {!r} IN template_allowed_pos_map: {}={}"
553 "".format(x, k, v)
554 )
555 assert False
558# Templates ignored during etymology extraction, i.e., these will not be listed
559# in the extracted etymology templates.
560ignored_etymology_templates: list[str] = [
561 "...",
562 "IPAchar",
563 "ipachar",
564 "ISBN",
565 "isValidPageName",
566 "redlink category",
567 "deprecated code",
568 "check deprecated lang param usage",
569 "para",
570 "p",
571 "cite",
572 "Cite news",
573 "Cite newsgroup",
574 "cite paper",
575 "cite MLLM 1976",
576 "cite journal",
577 "cite news/documentation",
578 "cite paper/documentation",
579 "cite video game",
580 "cite video game/documentation",
581 "cite newsgroup",
582 "cite newsgroup/documentation",
583 "cite web/documentation",
584 "cite news",
585 "Cite book",
586 "Cite-book",
587 "cite book",
588 "cite web",
589 "cite-usenet",
590 "cite-video/documentation",
591 "Cite-journal",
592 "rfe",
593 "catlangname",
594 "cln",
595 "langname-lite",
596 "no deprecated lang param usage",
597 "mention",
598 "m",
599 "m-self",
600 "link",
601 "l",
602 "ll",
603 "l-self",
604]
605# Regexp for matching ignored etymology template names. This adds certain
606# prefixes to the names listed above.
607ignored_etymology_templates_re = re.compile(
608 r"^((cite-|R:|RQ:).*|"
609 + r"|".join(re.escape(x) for x in ignored_etymology_templates)
610 + r")$"
611)
613# Regexp for matching ignored descendants template names. Right now we just
614# copy the ignored etymology templates
615ignored_descendants_templates_re = ignored_etymology_templates_re
617# Set of template names that are used to define usage examples. If the usage
618# example contains one of these templates, then it its type is set to
619# "example"
620usex_templates: set[str] = {
621 "afex",
622 "affixusex",
623 "co", # {{collocation}} acts like a example template, specifically for
624 # pairs of combinations of words that are more common than you'd
625 # except would be randomly; hlavní#Czech
626 "coi",
627 "collocation",
628 "el-example",
629 "el-x",
630 "example",
631 "examples",
632 "he-usex",
633 "he-x",
634 "hi-usex",
635 "hi-x",
636 "ja-usex-inline",
637 "ja-usex",
638 "ja-x",
639 "jbo-example",
640 "jbo-x",
641 "km-usex",
642 "km-x",
643 "ko-usex",
644 "ko-x",
645 "lo-usex",
646 "lo-x",
647 "ne-x",
648 "ne-usex",
649 "prefixusex",
650 "ryu-usex",
651 "ryu-x",
652 "shn-usex",
653 "shn-x",
654 "suffixusex",
655 "th-usex",
656 "th-x",
657 "ur-usex",
658 "ur-x",
659 "usex",
660 "usex-suffix",
661 "ux",
662 "uxi",
663}
665stop_head_at_these_templates: set[str] = {
666 "category",
667 "cat",
668 "topics",
669 "catlangname",
670 "c",
671 "C",
672 "top",
673 "cln",
674}
676# Set of template names that are used to define quotation examples. If the
677# usage example contains one of these templates, then its type is set to
678# "quotation".
679quotation_templates: set[str] = {
680 "collapse-quote",
681 "quote-av",
682 "quote-book",
683 "quote-GYLD",
684 "quote-hansard",
685 "quotei",
686 "quote-journal",
687 "quotelite",
688 "quote-mailing list",
689 "quote-meta",
690 "quote-newsgroup",
691 "quote-song",
692 "quote-text",
693 "quote",
694 "quote-us-patent",
695 "quote-video game",
696 "quote-web",
697 "quote-wikipedia",
698 "wikiquote",
699 "Wikiquote",
700 "Q",
701}
703taxonomy_templates = {
704 # argument 1 should be the taxonomic name, frex. "Lupus lupus"
705 "taxfmt",
706 "taxlink",
707 "taxlink2",
708 "taxlinknew",
709 "taxlook",
710}
712# Template names, this was exctracted from template_linkage_mappings,
713# because the code using template_linkage_mappings was actually not used
714# (but not removed).
715template_linkages_to_ignore_in_examples: set[str] = {
716 "syn",
717 "synonyms",
718 "ant",
719 "antonyms",
720 "hyp",
721 "hyponyms",
722 "der",
723 "derived terms",
724 "coordinate terms",
725 "cot",
726 "rel",
727 "col",
728 "inline alt forms",
729 "alti",
730 "comeronyms",
731 "holonyms",
732 "holo",
733 "hypernyms",
734 "hyper",
735 "meronyms",
736 "mero",
737 "troponyms",
738 "perfectives",
739 "pf",
740 "imperfectives",
741 "impf",
742 "syndiff",
743 "synsee",
744 # not linkage nor example templates
745 "sense",
746 "s",
747 "color panel",
748 "colour panel",
749}
751# Maps template name used in a word sense to a linkage field that it adds.
752sense_linkage_templates: dict[str, str] = {
753 "syn": "synonyms",
754 "synonyms": "synonyms",
755 "synsee": "synonyms",
756 "syndiff": "synonyms",
757 "hyp": "hyponyms",
758 "hyponyms": "hyponyms",
759 "ant": "antonyms",
760 "antonyms": "antonyms",
761 "alti": "related",
762 "inline alt forms": "related",
763 "coordinate terms": "coordinate_terms",
764 "cot": "coordinate_terms",
765 "comeronyms": "related",
766 "holonyms": "holonyms",
767 "holo": "holonyms",
768 "hypernyms": "hypernyms",
769 "hyper": "hypernyms",
770 "meronyms": "meronyms",
771 "mero": "meronyms",
772 "troponyms": "troponyms",
773 "perfectives": "related",
774 "pf": "related",
775 "imperfectives": "related",
776 "impf": "related",
777 "parasynonyms": "synonyms",
778 "par": "synonyms",
779 "parasyn": "synonyms",
780 "nearsyn": "synonyms",
781 "near-syn": "synonyms",
782}
784sense_linkage_templates_tags: dict[str, list[str]] = {
785 "alti": ["alternative"],
786 "inline alt forms": ["alternative"],
787 "comeronyms": ["comeronym"],
788 "perfectives": ["perfective"],
789 "pf": ["perfective"],
790 "imperfectives": ["imperfective"],
791 "impf": ["imperfective"],
792}
795def decode_html_entities(v: Union[str, int]) -> str:
796 """Decodes HTML entities from a value, converting them to the respective
797 Unicode characters/strings."""
798 if isinstance(v, int):
799 # I changed this to return str(v) instead of v = str(v),
800 # but there might have been the intention to have more logic
801 # here. html.unescape would not do anything special with an integer,
802 # it needs html escape symbols (&xx;).
803 return str(v)
804 return html.unescape(v)
807def parse_sense_linkage(
808 wxr: WiktextractContext,
809 data: SenseData,
810 name: str,
811 ht: TemplateArgs,
812 pos: str,
813) -> None:
814 """Parses a linkage (synonym, etc) specified in a word sense."""
815 assert isinstance(wxr, WiktextractContext)
816 assert isinstance(data, dict)
817 assert isinstance(name, str)
818 assert isinstance(ht, dict)
819 field = sense_linkage_templates[name]
820 field_tags = sense_linkage_templates_tags.get(name, [])
821 for i in range(2, 20):
822 if i not in ht:
823 break
824 w = clean_node(wxr, data, ht[i])
825 if "#" in w:
826 w = w[: w.index("#")]
827 if w in ["", "<"]: # `<` used in "hypernyms" template
828 continue
829 if ( 829 ↛ 834line 829 didn't jump to line 834 because the condition on line 829 was never true
830 i > 2
831 and w in (",", "or", ";")
832 or w.startswith(("see also", "See also"))
833 ):
834 continue
835 is_thesaurus = False
836 for alias in ns_title_prefix_tuple(wxr, "Thesaurus"):
837 if w.startswith(alias):
838 is_thesaurus = True
839 w = w[len(alias) :]
840 if w != wxr.wtp.title: 840 ↛ 860line 840 didn't jump to line 860 because the condition on line 840 was always true
841 from ...thesaurus import search_thesaurus
843 lang_code = clean_node(wxr, None, ht.get(1, ""))
844 for t_data in search_thesaurus(
845 wxr.thesaurus_db_conn, # type: ignore
846 w,
847 lang_code,
848 pos,
849 "synonyms", # GH issue #1570
850 ):
851 l_data: LinkageData = {
852 "word": t_data.term,
853 "source": "Thesaurus:" + w,
854 }
855 if len(t_data.tags) > 0: 855 ↛ 856line 855 didn't jump to line 856 because the condition on line 855 was never true
856 l_data["tags"] = t_data.tags
857 if len(t_data.raw_tags) > 0: 857 ↛ 858line 857 didn't jump to line 858 because the condition on line 857 was never true
858 l_data["raw_tags"] = t_data.raw_tags
859 data_append(data, field, l_data)
860 break
861 if is_thesaurus:
862 continue
863 tags: list[str] = []
864 topics: list[str] = []
865 english: Optional[str] = None
866 # Try to find qualifiers for this synonym
867 q = ht.get("q{}".format(i - 1))
868 if q:
869 cls = classify_desc(q)
870 if cls == "tags":
871 tagsets1, topics1 = decode_tags(q)
872 for ts in tagsets1:
873 tags.extend(ts)
874 topics.extend(topics1)
875 elif cls == "english": 875 ↛ 881line 875 didn't jump to line 881 because the condition on line 875 was always true
876 if english: 876 ↛ 877line 876 didn't jump to line 877 because the condition on line 876 was never true
877 english += "; " + q
878 else:
879 english = q
880 # Try to find English translation for this synonym
881 t = ht.get("t{}".format(i - 1))
882 if t: 882 ↛ 883line 882 didn't jump to line 883 because the condition on line 882 was never true
883 if english:
884 english += "; " + t
885 else:
886 english = t
888 # See if the linkage contains a parenthesized alt
889 alt = None
890 m = re.search(r"\(([^)]+)\)$", w)
891 if m: 891 ↛ 892line 891 didn't jump to line 892 because the condition on line 891 was never true
892 w = w[: m.start()].strip()
893 alt = m.group(1)
895 dt = {"word": w}
896 if field_tags: 896 ↛ 897line 896 didn't jump to line 897 because the condition on line 896 was never true
897 data_extend(dt, "tags", field_tags)
898 if tags:
899 data_extend(dt, "tags", tags)
900 if topics: 900 ↛ 901line 900 didn't jump to line 901 because the condition on line 900 was never true
901 data_extend(dt, "topics", topics)
902 if english:
903 dt["english"] = english # DEPRECATED for "translation"
904 dt["translation"] = english
905 if alt: 905 ↛ 906line 905 didn't jump to line 906 because the condition on line 905 was never true
906 dt["alt"] = alt
907 data_append(data, field, dt)
910EXAMPLE_SPLITTERS = r"\s*[―—]+\s*"
911example_splitter_re = re.compile(EXAMPLE_SPLITTERS)
912captured_splitters_re = re.compile(r"(" + EXAMPLE_SPLITTERS + r")")
915def synch_splits_with_args(
916 line: str, targs: TemplateArgs
917) -> Optional[list[str]]:
918 """If it looks like there's something weird with how a line of example
919 text has been split, this function will do the splitting after counting
920 occurences of the splitting regex inside the two main template arguments
921 containing the string data for the original language example and the
922 English translations.
923 """
924 # Previously, we split without capturing groups, but here we want to
925 # keep the original splitting hyphen regex intact.
926 fparts = captured_splitters_re.split(line)
927 new_parts = []
928 # ["First", " – ", "second", " – ", "third..."] from OL argument
929 first = 1 + (2 * len(example_splitter_re.findall(targs.get(2, ""))))
930 new_parts.append("".join(fparts[:first]))
931 # Translation argument
932 tr_arg = targs.get(3) or targs.get("translation") or targs.get("t", "")
933 # +2 = + 1 to skip the "expected" hyphen, + 1 as the `1 +` above.
934 second = first + 2 + (2 * len(example_splitter_re.findall(tr_arg)))
935 new_parts.append("".join(fparts[first + 1 : second]))
937 if all(new_parts): # no empty strings from the above spaghetti
938 new_parts.extend(fparts[second + 1 :: 2]) # skip rest of hyphens
939 return new_parts
940 else:
941 return None
944QUALIFIERS = r"^\((([^()]|\([^()]*\))*)\):?\s*"
945QUALIFIERS_RE = re.compile(QUALIFIERS)
946# (...): ... or (...(...)...): ...
949def parse_language(
950 wxr: WiktextractContext, langnode: WikiNode, language: str, lang_code: str
951) -> list[WordData]:
952 """Iterates over the text of the page, returning words (parts-of-speech)
953 defined on the page one at a time. (Individual word senses for the
954 same part-of-speech are typically encoded in the same entry.)"""
955 # imported here to avoid circular import
956 from .pronunciation import parse_pronunciation
958 assert isinstance(wxr, WiktextractContext)
959 assert isinstance(langnode, WikiNode)
960 assert isinstance(language, str)
961 assert isinstance(lang_code, str)
962 # print("parse_language", language)
964 is_reconstruction = False
965 word: str = wxr.wtp.title # type: ignore[assignment]
966 unsupported_prefix = "Unsupported titles/"
967 if word.startswith(unsupported_prefix):
968 w = word[len(unsupported_prefix) :]
969 if w in unsupported_title_map: 969 ↛ 972line 969 didn't jump to line 972 because the condition on line 969 was always true
970 word = unsupported_title_map[w]
971 else:
972 wxr.wtp.error(
973 "Unimplemented unsupported title: {}".format(word),
974 sortid="page/870",
975 )
976 word = w
977 elif word.startswith("Reconstruction:"):
978 word = word[word.find("/") + 1 :]
979 is_reconstruction = True
980 elif word.startswith("a/languages"): 980 ↛ 982line 980 didn't jump to line 982 because the condition on line 980 was never true
981 # ATM there's only one "mammoth page" in English wiktionary, 'a'
982 word = "a"
984 base_data: WordData = {
985 "word": word,
986 "lang": language,
987 "lang_code": lang_code,
988 }
989 if is_reconstruction:
990 data_append(base_data, "tags", "reconstruction")
991 sense_data: SenseData = {}
992 pos_data: WordData = {} # For a current part-of-speech
993 level_four_data: WordData = {} # Chinese Pronunciation-sections in-between
994 etym_data: WordData = {} # For one etymology
995 sense_datas: list[SenseData] = []
996 sense_ordinal = 0 # The recursive sense parsing messes up the ordering
997 # Never reset, do not use as data
998 level_four_datas: list[WordData] = []
999 etym_datas: list[WordData] = []
1000 page_datas: list[WordData] = []
1001 have_etym = False
1002 inside_level_four = False # This is for checking if the etymology section
1003 # or article has a Pronunciation section, for Chinese mostly; because
1004 # Chinese articles can have three level three sections (two etymology
1005 # sections and pronunciation sections) one after another, we need a kludge
1006 # to better keep track of whether we're in a normal "etym" or inside a
1007 # "level four" (which is what we've turned the level three Pron sections
1008 # into in the fix_subtitle_hierarchy(); all other sections are demoted by
1009 # a step.
1010 stack: list[str] = [] # names of items on the "stack"
1012 def merge_base(data: WordData, base: WordData) -> None:
1013 for k, v in base.items():
1014 # Copy the value to ensure that we don't share lists or
1015 # dicts between structures (even nested ones).
1016 v = copy.deepcopy(v)
1017 if k not in data:
1018 # The list was copied above, so this will not create shared ref
1019 data[k] = v # type: ignore[literal-required]
1020 continue
1021 if data[k] == v: # type: ignore[literal-required]
1022 continue
1023 if ( 1023 ↛ 1031line 1023 didn't jump to line 1031 because the condition on line 1023 was always true
1024 isinstance(data[k], (list, tuple)) # type: ignore[literal-required]
1025 or isinstance(
1026 v,
1027 (list, tuple), # Should this be "and"?
1028 )
1029 ):
1030 data[k] = list(data[k]) + list(v) # type: ignore
1031 elif data[k] != v: # type: ignore[literal-required]
1032 wxr.wtp.warning(
1033 "conflicting values for {} in merge_base: "
1034 "{!r} vs {!r}".format(k, data[k], v), # type: ignore[literal-required]
1035 sortid="page/904",
1036 )
1038 def complementary_pop(pron: SoundData, key: str) -> SoundData:
1039 """Remove unnecessary keys from dict values
1040 in a list comprehension..."""
1041 if key in pron:
1042 pron.pop(key) # type: ignore
1043 return pron
1045 def sound_matches_pos(sound: SoundData, pos: str) -> bool:
1046 if "pos" not in sound:
1047 return True
1048 sound_pos = sound["pos"] # type: ignore[typeddict-item]
1049 return pos in sound_pos
1051 def strip_sound_pos(sound: SoundData) -> SoundData:
1052 complementary_pop(sound, "pos")
1053 return sound
1055 # If the result has sounds, eliminate sounds that have a prefix that
1056 # does not match "word" or one of "forms"
1057 if "sounds" in data and "word" in data:
1058 accepted = [data["word"]]
1059 accepted.extend(f["form"] for f in data.get("forms", dict()))
1060 data["sounds"] = list(
1061 s
1062 for s in data["sounds"]
1063 if "form" not in s or s["form"] in accepted
1064 )
1065 # If the result has sounds, eliminate sounds that have a pos that
1066 # does not match "pos"
1067 if "sounds" in data and "pos" in data:
1068 data["sounds"] = list(
1069 strip_sound_pos(s)
1070 for s in data["sounds"]
1071 # "pos" is not a field of SoundData, correctly, so we're
1072 # removing it here. It's a kludge on a kludge on a kludge.
1073 if sound_matches_pos(s, data["pos"])
1074 )
1075 elif "sounds" in data: 1075 ↛ 1076line 1075 didn't jump to line 1076 because the condition on line 1075 was never true
1076 data["sounds"] = [strip_sound_pos(s) for s in data["sounds"]]
1078 def push_sense(sorting_ordinal: int | None = None) -> bool:
1079 """Starts collecting data for a new word sense. This returns True
1080 if a sense was added."""
1081 nonlocal sense_data
1082 if sorting_ordinal is None:
1083 sorting_ordinal = sense_ordinal
1084 tags = sense_data.get("tags", ())
1085 if (
1086 not sense_data.get("glosses")
1087 and "translation-hub" not in tags
1088 and "no-gloss" not in tags
1089 ):
1090 return False
1092 if ( 1092 ↛ 1102line 1092 didn't jump to line 1102 because the condition on line 1092 was never true
1093 (
1094 "participle" in sense_data.get("tags", ())
1095 or "infinitive" in sense_data.get("tags", ())
1096 )
1097 and "alt_of" not in sense_data
1098 and "form_of" not in sense_data
1099 and "etymology_text" in etym_data
1100 and etym_data["etymology_text"] != ""
1101 ):
1102 etym = etym_data["etymology_text"]
1103 etym = etym.split(". ")[0]
1104 ret = parse_alt_or_inflection_of(wxr, etym, set())
1105 if ret is not None:
1106 tags, lst = ret
1107 assert isinstance(lst, (list, tuple))
1108 if "form-of" in tags:
1109 data_extend(sense_data, "form_of", lst)
1110 data_extend(sense_data, "tags", tags)
1111 elif "alt-of" in tags:
1112 data_extend(sense_data, "alt_of", lst)
1113 data_extend(sense_data, "tags", tags)
1115 if not sense_data.get("glosses") and "no-gloss" not in sense_data.get( 1115 ↛ 1118line 1115 didn't jump to line 1118 because the condition on line 1115 was never true
1116 "tags", ()
1117 ):
1118 data_append(sense_data, "tags", "no-gloss")
1120 sense_data["__temp_sense_sorting_ordinal"] = sorting_ordinal # type: ignore
1121 sense_datas.append(sense_data)
1122 sense_data = {}
1123 return True
1125 def push_pos(sorting_ordinal: int | None = None) -> None:
1126 """Starts collecting data for a new part-of-speech."""
1127 nonlocal pos_data
1128 nonlocal sense_datas
1129 push_sense(sorting_ordinal)
1130 if wxr.wtp.subsection:
1131 data: WordData = {"senses": sense_datas}
1132 merge_base(data, pos_data)
1133 level_four_datas.append(data)
1134 pos_data = {}
1135 sense_datas = []
1136 wxr.wtp.start_subsection(None)
1138 def push_level_four_section(clear_sound_data: bool) -> None:
1139 """Starts collecting data for a new level four sections, which
1140 is usually virtual and empty, unless the article has Chinese
1141 'Pronunciation' sections that are etymology-section-like but
1142 under etymology, and at the same level in the source. We modify
1143 the source to demote Pronunciation sections like that to level
1144 4, and other sections one step lower."""
1145 nonlocal level_four_data
1146 nonlocal level_four_datas
1147 nonlocal etym_datas
1148 push_pos()
1149 # print(f"======\n{etym_data=}")
1150 # print(f"======\n{etym_datas=}")
1151 # print(f"======\n{level_four_data=}")
1152 # print(f"======\n{level_four_datas=}")
1153 for data in level_four_datas:
1154 merge_base(data, level_four_data)
1155 etym_datas.append(data)
1156 for data in etym_datas:
1157 merge_base(data, etym_data)
1158 page_datas.append(data)
1159 if clear_sound_data:
1160 level_four_data = {}
1161 level_four_datas = []
1162 etym_datas = []
1164 def push_etym() -> None:
1165 """Starts collecting data for a new etymology."""
1166 nonlocal etym_data
1167 nonlocal etym_datas
1168 nonlocal have_etym
1169 nonlocal inside_level_four
1170 have_etym = True
1171 push_level_four_section(False)
1172 inside_level_four = False
1173 # etymology section could under pronunciation section
1174 etym_data = (
1175 copy.deepcopy(level_four_data) if len(level_four_data) > 0 else {}
1176 )
1178 def select_data() -> WordData:
1179 """Selects where to store data (pos or etym) based on whether we
1180 are inside a pos (part-of-speech)."""
1181 # print(f"{wxr.wtp.subsection=}")
1182 # print(f"{stack=}")
1183 if wxr.wtp.subsection is not None:
1184 return pos_data
1185 if inside_level_four:
1186 return level_four_data
1187 if stack[-1] == language:
1188 return base_data
1189 return etym_data
1191 def parse_part_of_speech(posnode: WikiNode, pos: str) -> None:
1192 """Parses the subsection for a part-of-speech under a language on
1193 a page."""
1194 assert isinstance(posnode, WikiNode)
1195 assert isinstance(pos, str)
1196 # print("parse_part_of_speech", pos)
1197 pos_data["pos"] = pos
1198 pre: list[list[Union[str, WikiNode]]] = [[]] # list of lists
1199 lists: list[list[WikiNode]] = [[]] # list of lists
1200 first_para = True
1201 first_head_tmplt = True
1202 collecting_head = True
1203 start_of_paragraph = True
1205 # XXX extract templates from posnode with recursively_extract
1206 # that break stuff, like ja-kanji or az-suffix-form.
1207 # Do the extraction with a list of template names, combined from
1208 # different lists, then separate out them into different lists
1209 # that are handled at different points of the POS section.
1210 # First, extract az-suffix-form, put it in `inflection`,
1211 # and parse `inflection`'s content when appropriate later.
1212 # The contents of az-suffix-form (and ja-kanji) that generate
1213 # divs with "floatright" in their style gets deleted by
1214 # clean_value, so templates that slip through from here won't
1215 # break anything.
1216 # XXX bookmark
1217 # print("===================")
1218 # print(posnode.children)
1220 floaters, poschildren = recursively_extract(
1221 posnode.children,
1222 lambda x: (
1223 isinstance(x, WikiNode)
1224 and (
1225 (
1226 isinstance(x, TemplateNode)
1227 and x.template_name in FLOATING_TABLE_TEMPLATES
1228 )
1229 or (
1230 x.kind == NodeKind.LINK
1231 # Need to check for stringiness because some links are
1232 # broken; for example, if a template is missing an
1233 # argument, a link might look like `[[{{{1}}}...]]`
1234 and len(x.largs) > 0
1235 and len(x.largs[0]) > 0
1236 and isinstance(x.largs[0][0], str)
1237 and x.largs[0][0].lower().startswith("file:") # type:ignore[union-attr]
1238 )
1239 )
1240 ),
1241 )
1242 tempnode = WikiNode(NodeKind.LEVEL6, 0)
1243 tempnode.largs = [["Inflection"]]
1244 tempnode.children = floaters
1245 parse_inflection(tempnode, "Floating Div", pos)
1246 # print(poschildren)
1247 # XXX new above
1249 if not poschildren: 1249 ↛ 1250line 1249 didn't jump to line 1250 because the condition on line 1249 was never true
1250 if not floaters:
1251 wxr.wtp.debug(
1252 "PoS section without contents",
1253 sortid="en/page/1051/20230612",
1254 )
1255 else:
1256 wxr.wtp.debug(
1257 "PoS section without contents except for a floating table",
1258 sortid="en/page/1056/20230612",
1259 )
1260 return
1262 for node in poschildren:
1263 if isinstance(node, str):
1264 for m in re.finditer(r"\n+|[^\n]+", node):
1265 p = m.group(0)
1266 if p.startswith("\n\n") and pre:
1267 first_para = False
1268 start_of_paragraph = True
1269 break
1270 if p and collecting_head:
1271 pre[-1].append(p)
1272 continue
1273 assert isinstance(node, WikiNode)
1274 kind = node.kind
1275 if kind == NodeKind.LIST:
1276 lists[-1].append(node)
1277 collecting_head = False
1278 start_of_paragraph = True
1279 continue
1280 elif kind in LEVEL_KINDS:
1281 # Stop parsing section if encountering any kind of
1282 # level header (like ===Noun=== or ====Further Reading====).
1283 # At a quick glance, this should be the default behavior,
1284 # but if some kinds of source articles have sub-sub-sections
1285 # that should be parsed XXX it should be handled by changing
1286 # this break.
1287 break
1288 elif collecting_head and kind == NodeKind.LINK:
1289 # We might collect relevant links as they are often pictures
1290 # relating to the word
1291 if len(node.largs[0]) >= 1 and isinstance( 1291 ↛ 1306line 1291 didn't jump to line 1306 because the condition on line 1291 was always true
1292 node.largs[0][0], str
1293 ):
1294 if node.largs[0][0].startswith( 1294 ↛ 1300line 1294 didn't jump to line 1300 because the condition on line 1294 was never true
1295 ns_title_prefix_tuple(wxr, "Category")
1296 ):
1297 # [[Category:...]]
1298 # We're at the end of the file, probably, so stop
1299 # here. Otherwise the head will get garbage.
1300 break
1301 if node.largs[0][0].startswith(
1302 ns_title_prefix_tuple(wxr, "File")
1303 ):
1304 # Skips file links
1305 continue
1306 start_of_paragraph = False
1307 pre[-1].append(node)
1308 elif kind == NodeKind.HTML:
1309 if node.sarg == "br":
1310 if pre[-1]: 1310 ↛ 1262line 1310 didn't jump to line 1262 because the condition on line 1310 was always true
1311 pre.append([]) # Switch to next head
1312 lists.append([]) # Lists parallels pre
1313 collecting_head = True
1314 start_of_paragraph = True
1315 elif collecting_head and node.sarg not in ( 1315 ↛ 1321line 1315 didn't jump to line 1321 because the condition on line 1315 was never true
1316 "gallery",
1317 "ref",
1318 "cite",
1319 "caption",
1320 ):
1321 start_of_paragraph = False
1322 pre[-1].append(node)
1323 else:
1324 start_of_paragraph = False
1325 elif isinstance(node, TemplateNode):
1326 # XXX Insert code here that disambiguates between
1327 # templates that generate word heads and templates
1328 # that don't.
1329 # There's head_tag_re that seems like a regex meant
1330 # to identify head templates. Too bad it's None.
1332 # ignore {{category}}, {{cat}}... etc.
1333 if node.template_name in stop_head_at_these_templates:
1334 # we've reached a template that should be at the end,
1335 continue
1337 # skip these templates; panel_templates is already used
1338 # to skip certain templates else, but it also applies to
1339 # head parsing quite well.
1340 # node.largs[0][0] should always be str, but can't type-check
1341 # that.
1342 if is_panel_template(wxr, node.template_name):
1343 continue
1344 # skip these templates
1345 # if node.largs[0][0] in skip_these_templates_in_head:
1346 # first_head_tmplt = False # no first_head_tmplt at all
1347 # start_of_paragraph = False
1348 # continue
1350 if first_head_tmplt and pre[-1]:
1351 first_head_tmplt = False
1352 start_of_paragraph = False
1353 pre[-1].append(node)
1354 elif pre[-1] and start_of_paragraph:
1355 pre.append([]) # Switch to the next head
1356 lists.append([]) # lists parallel pre
1357 collecting_head = True
1358 start_of_paragraph = False
1359 pre[-1].append(node)
1360 else:
1361 pre[-1].append(node)
1362 elif first_para:
1363 start_of_paragraph = False
1364 if collecting_head: 1364 ↛ 1262line 1364 didn't jump to line 1262 because the condition on line 1364 was always true
1365 pre[-1].append(node)
1366 # XXX use template_fn in clean_node to check that the head macro
1367 # is compatible with the current part-of-speech and generate warning
1368 # if not. Use template_allowed_pos_map.
1370 # Clean up empty pairs, and fix messes with extra newlines that
1371 # separate templates that are followed by lists wiktextract issue #314
1373 cleaned_pre: list[list[Union[str, WikiNode]]] = []
1374 cleaned_lists: list[list[WikiNode]] = []
1375 pairless_pre_index = None
1377 for pre1, ls in zip(pre, lists):
1378 if pre1 and not ls:
1379 pairless_pre_index = len(cleaned_pre)
1380 if not pre1 and not ls: 1380 ↛ 1382line 1380 didn't jump to line 1382 because the condition on line 1380 was never true
1381 # skip [] + []
1382 continue
1383 if not ls and all(
1384 (isinstance(x, str) and not x.strip()) for x in pre1
1385 ):
1386 # skip ["\n", " "] + []
1387 continue
1388 if ls and not pre1:
1389 if pairless_pre_index is not None: 1389 ↛ 1390line 1389 didn't jump to line 1390 because the condition on line 1389 was never true
1390 cleaned_lists[pairless_pre_index] = ls
1391 pairless_pre_index = None
1392 continue
1393 cleaned_pre.append(pre1)
1394 cleaned_lists.append(ls)
1396 pre = cleaned_pre
1397 lists = cleaned_lists
1399 there_are_many_heads = len(pre) > 1
1400 header_tags: list[str] = []
1401 header_topics: list[str] = []
1402 previous_head_had_list = False
1404 if not any(g for g in lists):
1405 process_gloss_without_list(
1406 poschildren, pos, pos_data, header_tags, header_topics
1407 )
1408 else:
1409 for i, (pre1, ls) in enumerate(zip(pre, lists)):
1410 # if len(ls) == 0:
1411 # # don't have gloss list
1412 # # XXX add code here to filter out 'garbage', like text
1413 # # that isn't a head template or head.
1414 # continue
1416 if all(not sl for sl in lists[i:]):
1417 if i == 0: 1417 ↛ 1418line 1417 didn't jump to line 1418 because the condition on line 1417 was never true
1418 if isinstance(node, str):
1419 wxr.wtp.debug(
1420 "first head without list of senses,"
1421 "string: '{}[...]', {}/{}".format(
1422 node[:20], word, language
1423 ),
1424 sortid="page/1689/20221215",
1425 )
1426 if isinstance(node, WikiNode):
1427 if node.largs and node.largs[0][0] in [
1428 "Han char",
1429 ]:
1430 # just ignore these templates
1431 pass
1432 else:
1433 wxr.wtp.debug(
1434 "first head without "
1435 "list of senses, "
1436 "template node "
1437 "{}, {}/{}".format(
1438 node.largs, word, language
1439 ),
1440 sortid="page/1694/20221215",
1441 )
1442 else:
1443 wxr.wtp.debug(
1444 "first head without list of senses, "
1445 "{}/{}".format(word, language),
1446 sortid="page/1700/20221215",
1447 )
1448 # no break here so that the first head always
1449 # gets processed.
1450 else:
1451 if isinstance(node, str): 1451 ↛ 1452line 1451 didn't jump to line 1452 because the condition on line 1451 was never true
1452 wxr.wtp.debug(
1453 "later head without list of senses,"
1454 "string: '{}[...]', {}/{}".format(
1455 node[:20], word, language
1456 ),
1457 sortid="page/1708/20221215",
1458 )
1459 if isinstance(node, WikiNode): 1459 ↛ 1471line 1459 didn't jump to line 1471 because the condition on line 1459 was always true
1460 wxr.wtp.debug(
1461 "later head without list of senses,"
1462 "template node "
1463 "{}, {}/{}".format(
1464 node.sarg if node.sarg else node.largs,
1465 word,
1466 language,
1467 ),
1468 sortid="page/1713/20221215",
1469 )
1470 else:
1471 wxr.wtp.debug(
1472 "later head without list of senses, "
1473 "{}/{}".format(word, language),
1474 sortid="page/1719/20221215",
1475 )
1476 break
1477 head_group = i + 1 if there_are_many_heads else None
1478 # print("parse_part_of_speech: {}: {}: pre={}"
1479 # .format(wxr.wtp.section, wxr.wtp.subsection, pre1))
1481 if previous_head_had_list:
1482 # We use a boolean flag here because we want to be able
1483 # let the header_tags data pass through after the loop
1484 # is over without accidentally emptying it, if there are
1485 # no pos_datas and we need a dummy data.
1486 header_tags.clear()
1487 header_topics.clear()
1489 # print(f"{pre1=}")
1490 process_gloss_header(
1491 pre1, pos, head_group, pos_data, header_tags, header_topics
1492 )
1493 for ln in ls:
1494 # Parse each list associated with this head.
1495 for node in ln.children:
1496 # Parse nodes in l.children recursively.
1497 # The recursion function uses push_sense() to
1498 # add stuff into sense_datas, and returns True or
1499 # False if something is added, which bubbles upward.
1500 # If the bubble is "True", then higher levels of
1501 # the recursion will not push_sense(), because
1502 # the data is already pushed into a sub-gloss
1503 # downstream, unless the higher level has examples
1504 # that need to be put somewhere.
1505 common_data: SenseData = {
1506 "tags": list(header_tags),
1507 "topics": list(header_topics),
1508 }
1509 if head_group:
1510 common_data["head_nr"] = head_group
1511 parse_sense_node(node, common_data, pos) # type: ignore[arg-type]
1513 if len(ls) > 0:
1514 previous_head_had_list = True
1515 else:
1516 previous_head_had_list = False
1518 # If there are no senses extracted, add a dummy sense. We want to
1519 # keep tags extracted from the head for the dummy sense.
1520 push_sense() # Make sure unfinished data pushed, and start clean sense
1521 if len(sense_datas) == 0:
1522 data_extend(sense_data, "tags", header_tags)
1523 data_extend(sense_data, "topics", header_topics)
1524 data_append(sense_data, "tags", "no-gloss")
1525 push_sense()
1527 sense_datas.sort(key=lambda x: x.get("__temp_sense_sorting_ordinal", 0)) # type: ignore
1529 for sd in sense_datas:
1530 if "__temp_sense_sorting_ordinal" in sd: 1530 ↛ 1529line 1530 didn't jump to line 1529 because the condition on line 1530 was always true
1531 del sd["__temp_sense_sorting_ordinal"] # type: ignore
1533 term_label_templates: list[TemplateData] = []
1534 normal_label_templates: list[TemplateData] = []
1536 def head_post_template_fn(
1537 name: str, ht: TemplateArgs, expansion: str
1538 ) -> Optional[str]:
1539 """Handles special templates in the head section of a word. Head
1540 section is the text after part-of-speech subtitle and before word
1541 sense list. Typically it generates the bold line for the word, but
1542 may also contain other useful information that often ends in
1543 side boxes. We want to capture some of that additional information."""
1544 # print("HEAD_POST_TEMPLATE_FN", name, ht)
1545 if is_panel_template(wxr, name): 1545 ↛ 1548line 1545 didn't jump to line 1548 because the condition on line 1545 was never true
1546 # Completely ignore these templates (not even recorded in
1547 # head_templates)
1548 return ""
1549 if name == "head":
1550 # XXX are these also captured in forms? Should this special case
1551 # be removed?
1552 t = ht.get(2, "")
1553 if t == "pinyin": 1553 ↛ 1554line 1553 didn't jump to line 1554 because the condition on line 1553 was never true
1554 data_append(pos_data, "tags", "Pinyin")
1555 elif t == "romanization": 1555 ↛ 1556line 1555 didn't jump to line 1556 because the condition on line 1555 was never true
1556 data_append(pos_data, "tags", "romanization")
1557 if (
1558 HEAD_TAG_RE.search(name) is not None
1559 or name in PROBLEMATIC_TEMPLATES_CLUMP
1560 ):
1561 args_ht = clean_template_args(wxr, ht)
1562 cleaned_expansion = clean_node(wxr, None, expansion)
1563 dt: TemplateData = {
1564 "name": name,
1565 "args": args_ht,
1566 "expansion": cleaned_expansion,
1567 }
1568 if name in ETYMOLOGY_TEMPLATES_IN_HEADS:
1569 etymology_template_append(
1570 pos_data, name, args_ht, cleaned_expansion
1571 )
1572 else:
1573 data_append(pos_data, "head_templates", dt)
1574 if name in WORD_LEVEL_HEAD_TEMPLATES:
1575 term_label_templates.append(dt)
1576 # Squash these, their tags are applied to the whole word,
1577 # and some cause problems like "term-label"
1578 return ""
1580 # The following are both captured in head_templates and parsed
1581 # separately
1583 if name in wikipedia_templates:
1584 # Note: various places expect to have content from wikipedia
1585 # templates, so cannot convert this to empty
1586 parse_wikipedia_template(wxr, pos_data, ht)
1587 return None
1589 if name == "number box": 1589 ↛ 1591line 1589 didn't jump to line 1591 because the condition on line 1589 was never true
1590 # XXX extract numeric value?
1591 return ""
1592 if name == "enum":
1593 # XXX extract?
1594 return ""
1595 if name == "cardinalbox": 1595 ↛ 1598line 1595 didn't jump to line 1598 because the condition on line 1595 was never true
1596 # XXX extract similar to enum?
1597 # XXX this can also occur in top-level under language
1598 return ""
1599 if name == "Han simplified forms": 1599 ↛ 1601line 1599 didn't jump to line 1601 because the condition on line 1599 was never true
1600 # XXX extract?
1601 return ""
1602 # if name == "ja-kanji forms":
1603 # # XXX extract?
1604 # return ""
1605 # if name == "vi-readings":
1606 # # XXX extract?
1607 # return ""
1608 # if name == "ja-kanji":
1609 # # XXX extract?
1610 # return ""
1611 if name == "picdic" or name == "picdicimg" or name == "picdiclabel": 1611 ↛ 1613line 1611 didn't jump to line 1613 because the condition on line 1611 was never true
1612 # XXX extract?
1613 return ""
1614 if name == "defdate": 1614 ↛ 1616line 1614 didn't jump to line 1616 because the condition on line 1614 was never true
1615 # the one exampe I saw of this in a head was weird.
1616 return ""
1617 if name in ("lb", "lbl", "label"):
1618 args_ht = clean_template_args(wxr, ht)
1619 cleaned_expansion = clean_node(wxr, None, expansion).strip("()")
1620 dt = {
1621 "name": name,
1622 "args": args_ht,
1623 "expansion": cleaned_expansion,
1624 }
1625 normal_label_templates.append(dt)
1626 # The parens around __LABEL... below is meaningful: label
1627 # templates generate text with parens, so if we add the magical
1628 # phrase here with parens, it will look like a normal label that
1629 # will be handled as a parenthetical text; only when handling
1630 # parenthetical text do we need to actually actually access
1631 # the contents of the label.
1632 return f"(__LABEL_TEMPLATE_{len(normal_label_templates) - 1}__)"
1634 return None
1636 def process_gloss_header(
1637 header_nodes: list[Union[WikiNode, str]],
1638 pos_type: str,
1639 header_group: Optional[int],
1640 pos_data: WordData,
1641 header_tags: list[str],
1642 header_topics: list[str],
1643 ) -> None:
1644 ruby = []
1646 # process template parse nodes here
1647 new_nodes = []
1648 info_template_data = []
1649 for node in header_nodes:
1650 # print(f"{node=}")
1651 info_data, info_out = parse_info_template_node(wxr, node, "head")
1652 if info_data or info_out:
1653 if info_data: 1653 ↛ 1655line 1653 didn't jump to line 1655 because the condition on line 1653 was always true
1654 info_template_data.append(info_data)
1655 if info_out: # including just the original node 1655 ↛ 1656line 1655 didn't jump to line 1656 because the condition on line 1655 was never true
1656 new_nodes.append(info_out)
1657 else:
1658 new_nodes.append(node)
1659 header_nodes = new_nodes
1661 if info_template_data:
1662 if "info_templates" not in pos_data: 1662 ↛ 1665line 1662 didn't jump to line 1665 because the condition on line 1662 was always true
1663 pos_data["info_templates"] = info_template_data
1664 else:
1665 pos_data["info_templates"].extend(info_template_data)
1667 if lang_code == "ja":
1668 exp = wxr.wtp.parse(
1669 wxr.wtp.node_to_wikitext(header_nodes), expand_all=True
1670 )
1671 rub, _ = recursively_extract(
1672 exp.children,
1673 lambda x: (
1674 isinstance(x, WikiNode)
1675 and x.kind == NodeKind.HTML
1676 and x.sarg == "ruby"
1677 ),
1678 )
1679 if rub is not None: 1679 ↛ 1723line 1679 didn't jump to line 1723 because the condition on line 1679 was always true
1680 for r in rub:
1681 if TYPE_CHECKING:
1682 # we know the lambda above in recursively_extract
1683 # returns only WikiNodes in rub
1684 assert isinstance(r, WikiNode)
1685 rt = parse_ruby(wxr, r)
1686 if rt is not None: 1686 ↛ 1680line 1686 didn't jump to line 1680 because the condition on line 1686 was always true
1687 ruby.append(rt)
1688 elif lang_code == "vi":
1689 # Handle vi-readings templates that have a weird structures for
1690 # Chu Nom vietnamese characters heads
1691 # https://en.wiktionary.org/wiki/Template:vi-readings
1692 new_header_nodes = []
1693 related_readings: list[LinkageData] = []
1694 for node in header_nodes:
1695 if ( 1695 ↛ 1718line 1695 didn't jump to line 1718 because the condition on line 1695 was always true
1696 isinstance(node, TemplateNode)
1697 and node.template_name == "vi-readings"
1698 ):
1699 for parameter, tag in (
1700 ("hanviet", "han-viet-reading"),
1701 ("nom", "nom-reading"),
1702 # we ignore the fanqie parameter "phienthiet"
1703 ):
1704 arg = node.template_parameters.get(parameter)
1705 if arg is not None: 1705 ↛ 1699line 1705 didn't jump to line 1699 because the condition on line 1705 was always true
1706 text = clean_node(wxr, None, arg)
1707 for w in text.split(","):
1708 # ignore - separated references
1709 if "-" in w:
1710 w = w[: w.index("-")]
1711 w = w.strip()
1712 related_readings.append(
1713 LinkageData(word=w, tags=[tag])
1714 )
1715 continue
1717 # Skip the vi-reading template for the rest of the head parsing
1718 new_header_nodes.append(node)
1719 if len(related_readings) > 0: 1719 ↛ 1723line 1719 didn't jump to line 1723 because the condition on line 1719 was always true
1720 data_extend(pos_data, "related", related_readings)
1721 header_nodes = new_header_nodes
1723 header_text = clean_node(
1724 wxr,
1725 pos_data,
1726 header_nodes,
1727 post_template_fn=head_post_template_fn,
1728 collect_links=True,
1729 remove_anchors_from_links=True,
1730 )
1731 if "links" in pos_data:
1732 # WordData doesn't use `links`, so we can use `collect_links=True`
1733 # above without special handling and smuggle link data.
1734 extracted_links = pos_data["links"] # type: ignore
1735 del pos_data["links"] # type: ignore
1736 else:
1737 extracted_links = None
1738 # print(f"{header_text=}, {extracted_links=}")
1740 header_text = re.sub(r"\s+", " ", header_text).strip()
1742 if not header_text:
1743 return
1745 term_label_tags: list[str] = []
1746 term_label_topics: list[str] = []
1747 if len(term_label_templates) > 0:
1748 # parse term label templates; if there are other similar kinds
1749 # of templates in headers that you want to squash and apply as
1750 # tags, you can add them to WORD_LEVEL_HEAD_TEMPLATES
1751 for templ_data in term_label_templates:
1752 # print(templ_data)
1753 expan = templ_data.get("expansion", "").strip("().,; ")
1754 if not expan: 1754 ↛ 1755line 1754 didn't jump to line 1755 because the condition on line 1754 was never true
1755 continue
1756 tlb_tagsets, tlb_topics = decode_tags(expan)
1757 for tlb_tags in tlb_tagsets:
1758 if len(tlb_tags) > 0 and not any(
1759 t.startswith("error-") for t in tlb_tags
1760 ):
1761 term_label_tags.extend(tlb_tags)
1762 term_label_topics.extend(tlb_topics)
1763 # print(f"{tlb_tagsets=}, {tlb_topicsets=}")
1765 # print(f"{header_text=}")
1766 parse_word_head(
1767 wxr,
1768 word,
1769 pos_type,
1770 header_text,
1771 pos_data,
1772 is_reconstruction,
1773 header_group,
1774 header_nodes,
1775 ruby=ruby,
1776 links=extracted_links,
1777 label_templates=normal_label_templates,
1778 )
1779 if "tags" in pos_data:
1780 # pos_data can get "tags" data from some source; type-checkers
1781 # doesn't like it, so let's ignore it.
1782 header_tags.extend(pos_data["tags"]) # type: ignore[typeddict-item]
1783 del pos_data["tags"] # type: ignore[typeddict-item]
1784 if len(term_label_tags) > 0:
1785 header_tags.extend(term_label_tags)
1786 if len(term_label_topics) > 0:
1787 header_topics.extend(term_label_topics)
1789 def process_gloss_without_list(
1790 nodes: list[Union[WikiNode, str]],
1791 pos_type: str,
1792 pos_data: WordData,
1793 header_tags: list[str],
1794 header_topics: list[str],
1795 ) -> None:
1796 # gloss text might not inside a list
1797 header_nodes: list[Union[str, WikiNode]] = []
1798 gloss_nodes: list[Union[str, WikiNode]] = []
1799 for node in strip_nodes(nodes):
1800 if isinstance(node, WikiNode):
1801 if isinstance(node, TemplateNode):
1802 if node.template_name in (
1803 "zh-see",
1804 "ja-see",
1805 "ja-see-kango",
1806 ):
1807 continue # soft redirect
1808 elif (
1809 node.template_name == "head"
1810 or node.template_name.startswith(f"{lang_code}-")
1811 ):
1812 header_nodes.append(node)
1813 continue
1814 elif node.kind in LEVEL_KINDS: # following nodes are not gloss 1814 ↛ 1816line 1814 didn't jump to line 1816 because the condition on line 1814 was always true
1815 break
1816 gloss_nodes.append(node)
1818 if len(header_nodes) > 0:
1819 process_gloss_header(
1820 header_nodes,
1821 pos_type,
1822 None,
1823 pos_data,
1824 header_tags,
1825 header_topics,
1826 )
1827 if len(gloss_nodes) > 0:
1828 process_gloss_contents(
1829 gloss_nodes,
1830 pos_type,
1831 {"tags": list(header_tags), "topics": list(header_topics)},
1832 )
1834 def parse_sense_node(
1835 node: Union[str, WikiNode], # never receives str
1836 sense_base: SenseData,
1837 pos: str,
1838 ) -> bool:
1839 """Recursively (depth first) parse LIST_ITEM nodes for sense data.
1840 Uses push_sense() to attempt adding data to pos_data in the scope
1841 of parse_language() when it reaches deep in the recursion. push_sense()
1842 returns True if it succeeds, and that is bubbled up the stack; if
1843 a sense was added downstream, the higher levels (whose shared data
1844 was already added by a subsense) do not push_sense(), unless it
1845 has examples that need to be put somewhere.
1846 """
1847 assert isinstance(sense_base, dict) # Added to every sense deeper in
1849 nonlocal sense_ordinal
1850 my_ordinal = sense_ordinal # copies, not a reference
1851 sense_ordinal += 1 # only use for sorting
1853 if not isinstance(node, WikiNode): 1853 ↛ 1855line 1853 didn't jump to line 1855 because the condition on line 1853 was never true
1854 # This doesn't seem to ever happen in practice.
1855 wxr.wtp.debug(
1856 "{}: parse_sense_node called with"
1857 "something that isn't a WikiNode".format(pos),
1858 sortid="page/1287/20230119",
1859 )
1860 return False
1862 if node.kind != NodeKind.LIST_ITEM: 1862 ↛ 1863line 1862 didn't jump to line 1863 because the condition on line 1862 was never true
1863 wxr.wtp.debug(
1864 "{}: non-list-item inside list".format(pos), sortid="page/1678"
1865 )
1866 return False
1868 if node.sarg == ":":
1869 # Skip example entries at the highest level, ones without
1870 # a sense ("...#") above them.
1871 # If node.sarg is exactly and only ":", then it's at
1872 # the highest level; lower levels would have more
1873 # "indentation", like "#:" or "##:"
1874 return False
1876 # If a recursion call succeeds in push_sense(), bubble it up with
1877 # `added`.
1878 # added |= push_sense() or added |= parse_sense_node(...) to OR.
1879 added = False
1881 gloss_template_args: set[str] = set()
1883 # For LISTs and LIST_ITEMS, their argument is something like
1884 # "##" or "##:", and using that we can rudimentally determine
1885 # list 'depth' if need be, and also what kind of list or
1886 # entry it is; # is for normal glosses, : for examples (indent)
1887 # and * is used for quotations on wiktionary.
1888 current_depth = node.sarg
1890 children = node.children
1892 # subentries, (presumably) a list
1893 # of subglosses below this. The list's
1894 # argument ends with #, and its depth should
1895 # be bigger than parent node.
1896 subentries = [
1897 x
1898 for x in children
1899 if isinstance(x, WikiNode)
1900 and x.kind == NodeKind.LIST
1901 and x.sarg == current_depth + "#"
1902 ]
1904 # sublists of examples and quotations. .sarg
1905 # does not end with "#".
1906 others = [
1907 x
1908 for x in children
1909 if isinstance(x, WikiNode)
1910 and x.kind == NodeKind.LIST
1911 and x.sarg != current_depth + "#"
1912 ]
1914 # the actual contents of this particular node.
1915 # can be a gloss (or a template that expands into
1916 # many glosses which we can't easily pre-expand)
1917 # or could be an "outer gloss" with more specific
1918 # subglosses, or could be a qualfier for the subglosses.
1919 contents = [
1920 x
1921 for x in children
1922 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST
1923 ]
1924 # If this entry has sublists of entries, we should combine
1925 # gloss information from both the "outer" and sublist content.
1926 # Sometimes the outer gloss
1927 # is more non-gloss or tags, sometimes it is a coarse sense
1928 # and the inner glosses are more specific. The outer one
1929 # does not seem to have qualifiers.
1931 # If we have one sublist with one element, treat it
1932 # specially as it may be a Wiktionary error; raise
1933 # that nested element to the same level.
1934 # XXX If need be, this block can be easily removed in
1935 # the current recursive logicand the result is one sense entry
1936 # with both glosses in the glosses list, as you would
1937 # expect. If the higher entry has examples, there will
1938 # be a higher entry with some duplicated data.
1939 if len(subentries) == 1:
1940 slc = subentries[0].children
1941 if len(slc) == 1:
1942 # copy current node and modify it so it doesn't
1943 # loop infinitely.
1944 cropped_node = copy.copy(node)
1945 cropped_node.children = [
1946 x
1947 for x in children
1948 if not (
1949 isinstance(x, WikiNode)
1950 and x.kind == NodeKind.LIST
1951 and x.sarg == current_depth + "#"
1952 )
1953 ]
1954 added |= parse_sense_node(cropped_node, sense_base, pos)
1955 nonlocal sense_data # this kludge causes duplicated raw_
1956 # glosses data if this is not done;
1957 # if the top-level (cropped_node)
1958 # does not push_sense() properly or
1959 # parse_sense_node() returns early,
1960 # sense_data is not reset. This happens
1961 # for example when you have a no-gloss
1962 # string like "(intransitive)":
1963 # no gloss, push_sense() returns early
1964 # and sense_data has duplicate data with
1965 # sense_base
1966 sense_data = {}
1967 added |= parse_sense_node(slc[0], sense_base, pos)
1968 return added
1970 return process_gloss_contents(
1971 contents,
1972 pos,
1973 sense_base,
1974 subentries,
1975 others,
1976 gloss_template_args,
1977 added,
1978 my_ordinal,
1979 )
1981 def process_gloss_contents(
1982 contents: list[Union[str, WikiNode]],
1983 pos: str,
1984 sense_base: SenseData,
1985 subentries: list[WikiNode] = [],
1986 others: list[WikiNode] = [],
1987 gloss_template_args: Set[str] = set(),
1988 added: bool = False,
1989 sorting_ordinal: int | None = None,
1990 ) -> bool:
1991 def sense_template_fn(
1992 name: str, ht: TemplateArgs, is_gloss: bool = False
1993 ) -> Optional[str]:
1994 # print(f"sense_template_fn: {name}, {ht}")
1995 if name in wikipedia_templates:
1996 # parse_wikipedia_template(wxr, pos_data, ht)
1997 return None
1998 if is_panel_template(wxr, name):
1999 return ""
2000 if name in INFO_TEMPLATE_FUNCS:
2001 info_data, info_exp = parse_info_template_arguments(
2002 wxr, name, ht, "sense"
2003 )
2004 if info_data or info_exp: 2004 ↛ 2010line 2004 didn't jump to line 2010 because the condition on line 2004 was always true
2005 if info_data: 2005 ↛ 2007line 2005 didn't jump to line 2007 because the condition on line 2005 was always true
2006 data_append(sense_base, "info_templates", info_data)
2007 if info_exp and isinstance(info_exp, str): 2007 ↛ 2009line 2007 didn't jump to line 2009 because the condition on line 2007 was always true
2008 return info_exp
2009 return ""
2010 if name in ("defdate",):
2011 date = clean_node(wxr, None, ht.get(1, ()))
2012 if part_two := ht.get(2): 2012 ↛ 2014line 2012 didn't jump to line 2014 because the condition on line 2012 was never true
2013 # Unicode mdash, not '-'
2014 date += "–" + clean_node(wxr, None, part_two)
2015 refs: dict[str, ReferenceData] = {}
2016 # ref, refn, ref2, ref2n, ref3, ref3n
2017 # ref1 not valid
2018 for k, v in sorted(
2019 (k, v) for k, v in ht.items() if isinstance(k, str)
2020 ):
2021 if m := re.match(r"ref(\d?)(n?)", k): 2021 ↛ 2018line 2021 didn't jump to line 2018 because the condition on line 2021 was always true
2022 ref_v = clean_node(wxr, None, v)
2023 if m.group(1) not in refs: # empty string or digit
2024 refs[m.group(1)] = ReferenceData()
2025 if m.group(2):
2026 refs[m.group(1)]["refn"] = ref_v
2027 else:
2028 refs[m.group(1)]["text"] = ref_v
2029 data_append(
2030 sense_base,
2031 "attestations",
2032 AttestationData(date=date, references=list(refs.values())),
2033 )
2034 return ""
2035 if name == "senseid":
2036 langid = clean_node(wxr, None, ht.get(1, ()))
2037 arg = clean_node(wxr, sense_base, ht.get(2, ()))
2038 if re.match(r"Q\d+$", arg):
2039 data_append(sense_base, "wikidata", arg)
2040 data_append(sense_base, "senseid", langid + ":" + arg)
2041 if name in sense_linkage_templates:
2042 # print(f"SENSE_TEMPLATE_FN: {name}")
2043 parse_sense_linkage(wxr, sense_base, name, ht, pos)
2044 return ""
2045 if name == "†" or name == "zh-obsolete":
2046 data_append(sense_base, "tags", "obsolete")
2047 return ""
2048 if name in {
2049 "ux",
2050 "uxi",
2051 "usex",
2052 "afex",
2053 "prefixusex",
2054 "ko-usex",
2055 "ko-x",
2056 "hi-x",
2057 "ja-usex-inline",
2058 "ja-x",
2059 "quotei",
2060 "he-x",
2061 "hi-x",
2062 "km-x",
2063 "ne-x",
2064 "shn-x",
2065 "th-x",
2066 "ur-x",
2067 }:
2068 # Usage examples are captured separately below. We don't
2069 # want to expand them into glosses even when unusual coding
2070 # is used in the entry.
2071 # These templates may slip through inside another item, but
2072 # currently we're separating out example entries (..#:)
2073 # well enough that there seems to very little contamination.
2074 if is_gloss:
2075 wxr.wtp.wiki_notice(
2076 "Example template is used for gloss text",
2077 sortid="extractor.en.page.sense_template_fn/1415",
2078 )
2079 else:
2080 return ""
2081 if name == "w": 2081 ↛ 2082line 2081 didn't jump to line 2082 because the condition on line 2081 was never true
2082 if ht.get(2) == "Wp":
2083 return ""
2084 for v in ht.values():
2085 v = v.strip()
2086 if v and "<" not in v:
2087 gloss_template_args.add(v)
2088 return None
2090 def extract_link_texts(item: GeneralNode) -> None:
2091 """Recursively extracts link texts from the gloss source. This
2092 information is used to select whether to remove final "." from
2093 form_of/alt_of (e.g., ihm/Hunsrik)."""
2094 if isinstance(item, (list, tuple)):
2095 for x in item:
2096 extract_link_texts(x)
2097 return
2098 if isinstance(item, str):
2099 # There seem to be HTML sections that may futher contain
2100 # unparsed links.
2101 for m in re.finditer(r"\[\[([^]]*)\]\]", item): 2101 ↛ 2102line 2101 didn't jump to line 2102 because the loop on line 2101 never started
2102 print("ITER:", m.group(0))
2103 v = m.group(1).split("|")[-1].strip()
2104 if v:
2105 gloss_template_args.add(v)
2106 return
2107 if not isinstance(item, WikiNode): 2107 ↛ 2108line 2107 didn't jump to line 2108 because the condition on line 2107 was never true
2108 return
2109 if item.kind == NodeKind.LINK:
2110 v = item.largs[-1]
2111 if ( 2111 ↛ 2117line 2111 didn't jump to line 2117 because the condition on line 2111 was always true
2112 isinstance(v, list)
2113 and len(v) == 1
2114 and isinstance(v[0], str)
2115 ):
2116 gloss_template_args.add(v[0].strip())
2117 for x in item.children:
2118 extract_link_texts(x)
2120 extract_link_texts(contents)
2122 # get the raw text of non-list contents of this node, and other stuff
2123 # like tag and category data added to sense_base
2124 # cast = no-op type-setter for the type-checker
2125 partial_template_fn = cast(
2126 TemplateFnCallable,
2127 partial(sense_template_fn, is_gloss=True),
2128 )
2129 rawgloss = clean_node(
2130 wxr,
2131 sense_base,
2132 contents,
2133 template_fn=partial_template_fn,
2134 collect_links=True,
2135 )
2137 if not rawgloss: 2137 ↛ 2138line 2137 didn't jump to line 2138 because the condition on line 2137 was never true
2138 return False
2140 # remove manually typed ordered list text at the start("1. ")
2141 rawgloss = re.sub(r"^\d+\.\s+", "", rawgloss).strip()
2143 # get stuff like synonyms and categories from "others",
2144 # maybe examples and quotations
2145 clean_node(wxr, sense_base, others, template_fn=sense_template_fn)
2147 # The gloss could contain templates that produce more list items.
2148 # This happens commonly with, e.g., {{inflection of|...}}. Split
2149 # to parts. However, e.g. Interlingua generates multiple glosses
2150 # in HTML directly without Wikitext markup, so we must also split
2151 # by just newlines.
2152 subglosses = rawgloss.splitlines()
2154 if len(subglosses) == 0: 2154 ↛ 2155line 2154 didn't jump to line 2155 because the condition on line 2154 was never true
2155 return False
2157 if any(s.startswith("#") for s in subglosses):
2158 subtree = wxr.wtp.parse(rawgloss)
2159 # from wikitextprocessor.parser import print_tree
2160 # print("SUBTREE GENERATED BY TEMPLATE:")
2161 # print_tree(subtree)
2162 new_subentries = [
2163 x
2164 for x in subtree.children
2165 if isinstance(x, WikiNode) and x.kind == NodeKind.LIST
2166 ]
2168 new_others = [
2169 x
2170 for x in subtree.children
2171 if isinstance(x, WikiNode)
2172 and x.kind == NodeKind.LIST
2173 and not x.sarg.endswith("#")
2174 ]
2176 new_contents = [
2177 clean_node(wxr, [], x)
2178 for x in subtree.children
2179 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST
2180 ]
2182 subentries = subentries or new_subentries
2183 others = others or new_others
2184 subglosses = new_contents
2185 rawgloss = "".join(subglosses)
2186 # Generate no gloss for translation hub pages, but add the
2187 # "translation-hub" tag for them
2188 if rawgloss == "(This entry is a translation hub.)": 2188 ↛ 2189line 2188 didn't jump to line 2189 because the condition on line 2188 was never true
2189 data_append(sense_data, "tags", "translation-hub")
2190 return push_sense(sorting_ordinal)
2192 # Remove certain substrings specific to outer glosses
2193 strip_ends = [", particularly:"]
2194 for x in strip_ends:
2195 if rawgloss.endswith(x):
2196 rawgloss = rawgloss[: -len(x)].strip()
2197 break
2199 # A single gloss, or possibly an outer gloss.
2200 # Check if the possible outer gloss starts with
2201 # parenthesized tags/topics
2203 if rawgloss and rawgloss not in sense_base.get("raw_glosses", ()):
2204 data_append(sense_base, "raw_glosses", subglosses[0].strip())
2205 m = QUALIFIERS_RE.match(rawgloss)
2206 # (...): ... or (...(...)...): ...
2207 if m:
2208 q = m.group(1)
2209 rawgloss = rawgloss[m.end() :].strip()
2210 parse_sense_qualifier(wxr, q, sense_base)
2211 if rawgloss == "A pejorative:": 2211 ↛ 2212line 2211 didn't jump to line 2212 because the condition on line 2211 was never true
2212 data_append(sense_base, "tags", "pejorative")
2213 rawgloss = ""
2214 elif rawgloss == "Short forms.": 2214 ↛ 2215line 2214 didn't jump to line 2215 because the condition on line 2214 was never true
2215 data_append(sense_base, "tags", "abbreviation")
2216 rawgloss = ""
2217 elif rawgloss == "Technical or specialized senses.": 2217 ↛ 2218line 2217 didn't jump to line 2218 because the condition on line 2217 was never true
2218 rawgloss = ""
2219 elif rawgloss.startswith("inflection of "):
2220 parsed = parse_alt_or_inflection_of(wxr, rawgloss, set())
2221 if parsed is not None: 2221 ↛ 2230line 2221 didn't jump to line 2230 because the condition on line 2221 was always true
2222 tags, origins = parsed
2223 if origins is not None: 2223 ↛ 2225line 2223 didn't jump to line 2225 because the condition on line 2223 was always true
2224 data_extend(sense_base, "form_of", origins)
2225 if tags is not None: 2225 ↛ 2228line 2225 didn't jump to line 2228 because the condition on line 2225 was always true
2226 data_extend(sense_base, "tags", tags)
2227 else:
2228 data_append(sense_base, "tags", "form-of")
2229 else:
2230 data_append(sense_base, "tags", "form-of")
2231 if rawgloss: 2231 ↛ 2262line 2231 didn't jump to line 2262 because the condition on line 2231 was always true
2232 # Code duplicating a lot of clean-up operations from later in
2233 # this block. We want to clean up the "supergloss" as much as
2234 # possible, in almost the same way as a normal gloss.
2235 supergloss = rawgloss
2237 if supergloss.startswith("; "): 2237 ↛ 2238line 2237 didn't jump to line 2238 because the condition on line 2237 was never true
2238 supergloss = supergloss[1:].strip()
2240 if supergloss.startswith(("^†", "†")):
2241 data_append(sense_base, "tags", "obsolete")
2242 supergloss = supergloss[2:].strip()
2243 elif supergloss.startswith("^‡"): 2243 ↛ 2244line 2243 didn't jump to line 2244 because the condition on line 2243 was never true
2244 data_extend(sense_base, "tags", ["obsolete", "historical"])
2245 supergloss = supergloss[2:].strip()
2247 # remove [14th century...] style brackets at the end
2248 supergloss = re.sub(r"\s\[[^]]*\]\s*$", "", supergloss)
2250 if supergloss.startswith((",", ":")):
2251 supergloss = supergloss[1:]
2252 supergloss = supergloss.strip()
2253 if supergloss.startswith("N. of "): 2253 ↛ 2254line 2253 didn't jump to line 2254 because the condition on line 2253 was never true
2254 supergloss = "Name of " + supergloss[6:]
2255 supergloss = supergloss[2:]
2256 data_append(sense_base, "glosses", supergloss)
2257 if supergloss in ("A person:",):
2258 data_append(sense_base, "tags", "g-person")
2260 # The main recursive call (except for the exceptions at the
2261 # start of this function).
2262 for sublist in subentries:
2263 if not ( 2263 ↛ 2266line 2263 didn't jump to line 2266 because the condition on line 2263 was never true
2264 isinstance(sublist, WikiNode) and sublist.kind == NodeKind.LIST
2265 ):
2266 wxr.wtp.debug(
2267 f"'{repr(rawgloss[:20])}.' gloss has `subentries`"
2268 f"with items that are not LISTs",
2269 sortid="page/1511/20230119",
2270 )
2271 continue
2272 for item in sublist.children:
2273 if not ( 2273 ↛ 2277line 2273 didn't jump to line 2277 because the condition on line 2273 was never true
2274 isinstance(item, WikiNode)
2275 and item.kind == NodeKind.LIST_ITEM
2276 ):
2277 continue
2278 # copy sense_base to prevent cross-contamination between
2279 # subglosses and other subglosses and superglosses
2280 sense_base2 = copy.deepcopy(sense_base)
2281 if parse_sense_node(item, sense_base2, pos): 2281 ↛ 2272line 2281 didn't jump to line 2272 because the condition on line 2281 was always true
2282 added = True
2284 # Capture examples.
2285 # This is called after the recursive calls above so that
2286 # sense_base is not contaminated with meta-data from
2287 # example entries for *this* gloss.
2288 examples = []
2289 if wxr.config.capture_examples: 2289 ↛ 2293line 2289 didn't jump to line 2293 because the condition on line 2289 was always true
2290 examples = extract_examples(others, sense_base)
2292 # push_sense() succeeded somewhere down-river, so skip this level
2293 if added:
2294 if examples:
2295 # this higher-up gloss has examples that we do not want to skip
2296 wxr.wtp.debug(
2297 "'{}[...]' gloss has examples we want to keep, "
2298 "but there are subglosses.".format(repr(rawgloss[:30])),
2299 sortid="page/1498/20230118",
2300 )
2301 else:
2302 return True
2304 # Some entries, e.g., "iacebam", have weird sentences in quotes
2305 # after the gloss, but these sentences don't seem to be intended
2306 # as glosses. Skip them.
2307 indexed_subglosses = list(
2308 (i, gl)
2309 for i, gl in enumerate(subglosses)
2310 if gl.strip() and not re.match(r'\s*(\([^)]*\)\s*)?"[^"]*"\s*$', gl)
2311 )
2313 if len(indexed_subglosses) > 1 and "form_of" not in sense_base: 2313 ↛ 2314line 2313 didn't jump to line 2314 because the condition on line 2313 was never true
2314 gl = indexed_subglosses[0][1].strip()
2315 if gl.endswith(":"):
2316 gl = gl[:-1].strip()
2317 parsed = parse_alt_or_inflection_of(wxr, gl, gloss_template_args)
2318 if parsed is not None:
2319 infl_tags, infl_dts = parsed
2320 if infl_dts and "form-of" in infl_tags and len(infl_tags) == 1:
2321 # Interpret others as a particular form under
2322 # "inflection of"
2323 data_extend(sense_base, "tags", infl_tags)
2324 data_extend(sense_base, "form_of", infl_dts)
2325 indexed_subglosses = indexed_subglosses[1:]
2326 elif not infl_dts:
2327 data_extend(sense_base, "tags", infl_tags)
2328 indexed_subglosses = indexed_subglosses[1:]
2330 # Create senses for remaining subglosses
2331 for i, (gloss_i, gloss) in enumerate(indexed_subglosses):
2332 gloss = gloss.strip()
2333 if not gloss and len(indexed_subglosses) > 1: 2333 ↛ 2334line 2333 didn't jump to line 2334 because the condition on line 2333 was never true
2334 continue
2335 # Push a new sense (if the last one is not empty)
2336 if push_sense(sorting_ordinal): 2336 ↛ 2337line 2336 didn't jump to line 2337 because the condition on line 2336 was never true
2337 added = True
2338 # if gloss not in sense_data.get("raw_glosses", ()):
2339 # data_append(sense_data, "raw_glosses", gloss)
2340 if i == 0 and examples:
2341 # In a multi-line gloss, associate examples
2342 # with only one of them.
2343 # XXX or you could use gloss_i == len(indexed_subglosses)
2344 # to associate examples with the *last* one.
2345 data_extend(sense_data, "examples", examples)
2346 if gloss.startswith("; ") and gloss_i > 0: 2346 ↛ 2347line 2346 didn't jump to line 2347 because the condition on line 2346 was never true
2347 gloss = gloss[1:].strip()
2348 # If the gloss starts with †, mark as obsolete
2349 if gloss.startswith("^†"): 2349 ↛ 2350line 2349 didn't jump to line 2350 because the condition on line 2349 was never true
2350 data_append(sense_data, "tags", "obsolete")
2351 gloss = gloss[2:].strip()
2352 elif gloss.startswith("^‡"): 2352 ↛ 2353line 2352 didn't jump to line 2353 because the condition on line 2352 was never true
2353 data_extend(sense_data, "tags", ["obsolete", "historical"])
2354 gloss = gloss[2:].strip()
2355 # Copy data for all senses to this sense
2356 for k, v in sense_base.items():
2357 if isinstance(v, (list, tuple)):
2358 if k != "tags":
2359 # Tags handled below (countable/uncountable special)
2360 data_extend(sense_data, k, v)
2361 else:
2362 assert k not in ("tags", "categories", "topics")
2363 sense_data[k] = v # type:ignore[literal-required]
2364 # Parse the gloss for this particular sense
2365 m = QUALIFIERS_RE.match(gloss)
2366 # (...): ... or (...(...)...): ...
2367 if m:
2368 parse_sense_qualifier(wxr, m.group(1), sense_data)
2369 gloss = gloss[m.end() :].strip()
2371 # Remove common suffix "[from 14th c.]" and similar
2372 gloss = re.sub(r"\s\[[^]]*\]\s*$", "", gloss)
2374 # Check to make sure we don't have unhandled list items in gloss
2375 ofs = max(gloss.find("#"), gloss.find("* "))
2376 if ofs > 10 and "(#)" not in gloss:
2377 wxr.wtp.debug(
2378 "gloss may contain unhandled list items: {}".format(gloss),
2379 sortid="page/1412",
2380 )
2381 elif "\n" in gloss: 2381 ↛ 2382line 2381 didn't jump to line 2382 because the condition on line 2381 was never true
2382 wxr.wtp.debug(
2383 "gloss contains newline: {}".format(gloss),
2384 sortid="page/1416",
2385 )
2387 # Kludge, some glosses have a comma after initial qualifiers in
2388 # parentheses
2389 if gloss.startswith((",", ":")):
2390 gloss = gloss[1:]
2391 gloss = gloss.strip()
2392 if gloss.endswith(":"):
2393 gloss = gloss[:-1].strip()
2394 if gloss.startswith("N. of "): 2394 ↛ 2395line 2394 didn't jump to line 2395 because the condition on line 2394 was never true
2395 gloss = "Name of " + gloss[6:]
2396 if gloss.startswith("†"): 2396 ↛ 2397line 2396 didn't jump to line 2397 because the condition on line 2396 was never true
2397 data_append(sense_data, "tags", "obsolete")
2398 gloss = gloss[1:]
2399 elif gloss.startswith("^†"): 2399 ↛ 2400line 2399 didn't jump to line 2400 because the condition on line 2399 was never true
2400 data_append(sense_data, "tags", "obsolete")
2401 gloss = gloss[2:]
2403 # Copy tags from sense_base if any. This will not copy
2404 # countable/uncountable if either was specified in the sense,
2405 # as sometimes both are specified in word head but only one
2406 # in individual senses.
2407 countability_tags = []
2408 base_tags = sense_base.get("tags", ())
2409 sense_tags = sense_data.get("tags", ())
2410 for tag in base_tags:
2411 if tag in ("countable", "uncountable"):
2412 if tag not in countability_tags: 2412 ↛ 2414line 2412 didn't jump to line 2414 because the condition on line 2412 was always true
2413 countability_tags.append(tag)
2414 continue
2415 if tag not in sense_tags:
2416 data_append(sense_data, "tags", tag)
2417 if countability_tags:
2418 if ( 2418 ↛ 2427line 2418 didn't jump to line 2427 because the condition on line 2418 was always true
2419 "countable" not in sense_tags
2420 and "uncountable" not in sense_tags
2421 ):
2422 data_extend(sense_data, "tags", countability_tags)
2424 # If outer gloss specifies a form-of ("inflection of", see
2425 # aquamarine/German), try to parse the inner glosses as
2426 # tags for an inflected form.
2427 if "form-of" in sense_base.get("tags", ()):
2428 parsed = parse_alt_or_inflection_of(
2429 wxr, gloss, gloss_template_args
2430 )
2431 if parsed is not None: 2431 ↛ 2437line 2431 didn't jump to line 2437 because the condition on line 2431 was always true
2432 infl_tags, infl_dts = parsed
2433 if not infl_dts and infl_tags: 2433 ↛ 2437line 2433 didn't jump to line 2437 because the condition on line 2433 was always true
2434 # Interpret as a particular form under "inflection of"
2435 data_extend(sense_data, "tags", infl_tags)
2437 if not gloss: 2437 ↛ 2438line 2437 didn't jump to line 2438 because the condition on line 2437 was never true
2438 data_append(sense_data, "tags", "empty-gloss")
2439 elif gloss != "-" and gloss not in sense_data.get("glosses", []):
2440 if ( 2440 ↛ 2451line 2440 didn't jump to line 2451 because the condition on line 2440 was always true
2441 gloss_i == 0
2442 and len(sense_data.get("glosses", tuple())) >= 1
2443 ):
2444 # If we added a "high-level gloss" from rawgloss, but this
2445 # is that same gloss_i, add this instead of the raw_gloss
2446 # from before if they're different: the rawgloss was not
2447 # cleaned exactly the same as this later gloss
2448 sense_data["glosses"][-1] = gloss
2449 else:
2450 # Add the gloss for the sense.
2451 data_append(sense_data, "glosses", gloss)
2453 # Kludge: there are cases (e.g., etc./Swedish) where there are
2454 # two abbreviations in the same sense, both generated by the
2455 # {{abbreviation of|...}} template. Handle these with some magic.
2456 position = 0
2457 split_glosses = []
2458 for m in re.finditer(r"Abbreviation of ", gloss):
2459 if m.start() != position: 2459 ↛ 2458line 2459 didn't jump to line 2458 because the condition on line 2459 was always true
2460 split_glosses.append(gloss[position : m.start()])
2461 position = m.start()
2462 split_glosses.append(gloss[position:])
2463 for gloss in split_glosses:
2464 # Check if this gloss describes an alt-of or inflection-of
2465 if (
2466 lang_code != "en"
2467 and " " not in gloss
2468 and distw([word], gloss) < 0.3
2469 ):
2470 # Don't try to parse gloss if it is one word
2471 # that is close to the word itself for non-English words
2472 # (probable translations of a tag/form name)
2473 continue
2474 parsed = parse_alt_or_inflection_of(
2475 wxr, gloss, gloss_template_args
2476 )
2477 if parsed is None:
2478 continue
2479 tags, dts = parsed
2480 if not dts and tags:
2481 data_extend(sense_data, "tags", tags)
2482 continue
2483 for dt in dts: # type:ignore[union-attr]
2484 ftags = list(tag for tag in tags if tag != "form-of")
2485 if "alt-of" in tags:
2486 data_extend(sense_data, "tags", ftags)
2487 data_append(sense_data, "alt_of", dt)
2488 elif "compound-of" in tags: 2488 ↛ 2489line 2488 didn't jump to line 2489 because the condition on line 2488 was never true
2489 data_extend(sense_data, "tags", ftags)
2490 data_append(sense_data, "compound_of", dt)
2491 elif "synonym-of" in tags: 2491 ↛ 2492line 2491 didn't jump to line 2492 because the condition on line 2491 was never true
2492 data_extend(dt, "tags", ftags)
2493 data_append(sense_data, "synonyms", dt)
2494 elif tags and dt.get("word", "").startswith("of "): 2494 ↛ 2495line 2494 didn't jump to line 2495 because the condition on line 2494 was never true
2495 dt["word"] = dt["word"][3:]
2496 data_append(sense_data, "tags", "form-of")
2497 data_extend(sense_data, "tags", ftags)
2498 data_append(sense_data, "form_of", dt)
2499 elif "form-of" in tags: 2499 ↛ 2483line 2499 didn't jump to line 2483 because the condition on line 2499 was always true
2500 data_extend(sense_data, "tags", tags)
2501 data_append(sense_data, "form_of", dt)
2503 if len(sense_data) == 0:
2504 if len(sense_base.get("tags", [])) == 0: 2504 ↛ 2506line 2504 didn't jump to line 2506 because the condition on line 2504 was always true
2505 del sense_base["tags"]
2506 sense_data.update(sense_base)
2507 if push_sense(sorting_ordinal): 2507 ↛ 2511line 2507 didn't jump to line 2511 because the condition on line 2507 was always true
2508 # push_sense succeded in adding a sense to pos_data
2509 added = True
2510 # print("PARSE_SENSE DONE:", pos_datas[-1])
2511 return added
2513 def parse_inflection(
2514 node: WikiNode, section: str, pos: Optional[str]
2515 ) -> None:
2516 """Parses inflection data (declension, conjugation) from the given
2517 page. This retrieves the actual inflection template
2518 parameters, which are very useful for applications that need
2519 to learn the inflection classes and generate inflected
2520 forms."""
2521 assert isinstance(node, WikiNode)
2522 assert isinstance(section, str)
2523 assert pos is None or isinstance(pos, str)
2524 # print("parse_inflection:", node)
2526 if pos is None: 2526 ↛ 2527line 2526 didn't jump to line 2527 because the condition on line 2526 was never true
2527 wxr.wtp.debug(
2528 "inflection table outside part-of-speech", sortid="page/1812"
2529 )
2530 return
2532 def inflection_template_fn(
2533 name: str, ht: TemplateArgs
2534 ) -> Optional[str]:
2535 # print("decl_conj_template_fn", name, ht)
2536 if is_panel_template(wxr, name): 2536 ↛ 2537line 2536 didn't jump to line 2537 because the condition on line 2536 was never true
2537 return ""
2538 if name in ("is-u-mutation",): 2538 ↛ 2541line 2538 didn't jump to line 2541 because the condition on line 2538 was never true
2539 # These are not to be captured as an exception to the
2540 # generic code below
2541 return None
2542 m = re.search(
2543 r"-(conj|decl|ndecl|adecl|infl|conjugation|"
2544 r"declension|inflection|mut|mutation)($|-)",
2545 name,
2546 )
2547 if m:
2548 args_ht = clean_template_args(wxr, ht)
2549 dt = {"name": name, "args": args_ht}
2550 data_append(pos_data, "inflection_templates", dt)
2552 return None
2554 # Convert the subtree back to Wikitext, then expand all and parse,
2555 # capturing templates in the process
2556 text = wxr.wtp.node_to_wikitext(node.children)
2558 # Split text into separate sections for each to-level template
2559 brace_matches = re.split(r"((?:^|\n)\s*{\||\n\s*\|}|{{+|}}+)", text)
2560 # ["{{", "template", "}}"] or ["^{|", "table contents", "\n|}"]
2561 # The (?:...) creates a non-capturing regex group; if it was capturing,
2562 # like the group around it, it would create elements in brace_matches,
2563 # including None if it doesn't match.
2564 # 20250114: Added {| and |} into the regex because tables were being
2565 # cut into pieces by this code. Issue #973, introduction of two-part
2566 # book-end templates similar to trans-top and tran-bottom.
2567 template_sections = []
2568 template_nesting = 0 # depth of SINGLE BRACES { { nesting } }
2569 # Because there is the possibility of triple curly braces
2570 # ("{{{", "}}}") in addition to normal ("{{ }}"), we do not
2571 # count nesting depth using pairs of two brackets, but
2572 # instead use singular braces ("{ }").
2573 # Because template delimiters should be balanced, regardless
2574 # of whether {{ or {{{ is used, and because we only care
2575 # about the outer-most delimiters (the highest level template)
2576 # we can just count the single braces when those single
2577 # braces are part of a group.
2578 table_nesting = 0
2579 # However, if we have a stray table ({| ... |}) that should always
2580 # be its own section, and should prevent templates from cutting it
2581 # into sections.
2583 # print(f"Parse inflection: {text=}")
2584 # print(f"Brace matches: {repr('///'.join(brace_matches))}")
2585 if len(brace_matches) > 1:
2586 tsection: list[str] = []
2587 after_templates = False # kludge to keep any text
2588 # before first template
2589 # with the first template;
2590 # otherwise, text
2591 # goes with preceding template
2592 for m in brace_matches:
2593 if m.startswith("\n; ") and after_templates: 2593 ↛ 2594line 2593 didn't jump to line 2594 because the condition on line 2593 was never true
2594 after_templates = False
2595 template_sections.append(tsection)
2596 tsection = []
2597 tsection.append(m)
2598 elif m.startswith("{{") or m.endswith("{|"):
2599 if (
2600 template_nesting == 0
2601 and after_templates
2602 and table_nesting == 0
2603 ):
2604 template_sections.append(tsection)
2605 tsection = []
2606 # start new section
2607 after_templates = True
2608 if m.startswith("{{"):
2609 template_nesting += 1
2610 else:
2611 # m.endswith("{|")
2612 table_nesting += 1
2613 tsection.append(m)
2614 elif m.startswith("}}") or m.endswith("|}"):
2615 if m.startswith("}}"):
2616 template_nesting -= 1
2617 if template_nesting < 0: 2617 ↛ 2618line 2617 didn't jump to line 2618 because the condition on line 2617 was never true
2618 wxr.wtp.error(
2619 "Negatively nested braces, "
2620 "couldn't split inflection templates, "
2621 "{}/{} section {}".format(
2622 word, language, section
2623 ),
2624 sortid="page/1871",
2625 )
2626 template_sections = [] # use whole text
2627 break
2628 else:
2629 table_nesting -= 1
2630 if table_nesting < 0: 2630 ↛ 2631line 2630 didn't jump to line 2631 because the condition on line 2630 was never true
2631 wxr.wtp.error(
2632 "Negatively nested table braces, "
2633 "couldn't split inflection section, "
2634 "{}/{} section {}".format(
2635 word, language, section
2636 ),
2637 sortid="page/20250114",
2638 )
2639 template_sections = [] # use whole text
2640 break
2641 tsection.append(m)
2642 else:
2643 tsection.append(m)
2644 if tsection: # dangling tsection 2644 ↛ 2652line 2644 didn't jump to line 2652 because the condition on line 2644 was always true
2645 template_sections.append(tsection)
2646 # Why do it this way around? The parser has a preference
2647 # to associate bits outside of tables with the preceding
2648 # table (`after`-variable), so a new tsection begins
2649 # at {{ and everything before it belongs to the previous
2650 # template.
2652 texts = []
2653 if not template_sections:
2654 texts = [text]
2655 else:
2656 for tsection in template_sections:
2657 texts.append("".join(tsection))
2658 if template_nesting != 0: 2658 ↛ 2659line 2658 didn't jump to line 2659 because the condition on line 2658 was never true
2659 wxr.wtp.error(
2660 "Template nesting error: "
2661 "template_nesting = {} "
2662 "couldn't split inflection templates, "
2663 "{}/{} section {}".format(
2664 template_nesting, word, language, section
2665 ),
2666 sortid="page/1896",
2667 )
2668 texts = [text]
2669 for text in texts:
2670 tree = wxr.wtp.parse(
2671 text, expand_all=True, template_fn=inflection_template_fn
2672 )
2674 if not text.strip():
2675 continue
2677 # Parse inflection tables from the section. The data is stored
2678 # under "forms".
2679 if wxr.config.capture_inflections: 2679 ↛ 2669line 2679 didn't jump to line 2669 because the condition on line 2679 was always true
2680 tablecontext = None
2681 m = re.search(r"{{([^}{|]+)\|?", text)
2682 if m:
2683 template_name = m.group(1).strip()
2684 tablecontext = TableContext(template_name)
2686 parse_inflection_section(
2687 wxr,
2688 pos_data,
2689 word,
2690 language,
2691 pos,
2692 section,
2693 tree,
2694 tablecontext=tablecontext,
2695 )
2697 def get_subpage_section(
2698 title: str, subtitle: str, seqs: list[Union[list[str], tuple[str, ...]]]
2699 ) -> Optional[Union[WikiNode, str]]:
2700 """Loads a subpage of the given page, and finds the section
2701 for the given language, part-of-speech, and section title. This
2702 is used for finding translations and other sections on subpages."""
2703 assert isinstance(language, str)
2704 assert isinstance(title, str)
2705 assert isinstance(subtitle, str)
2706 assert isinstance(seqs, (list, tuple))
2707 for seq in seqs:
2708 for x in seq:
2709 assert isinstance(x, str)
2710 subpage_title = word + "/" + subtitle
2711 subpage_content = wxr.wtp.get_page_body(subpage_title, 0)
2712 if subpage_content is None:
2713 wxr.wtp.error(
2714 "/translations not found despite "
2715 "{{see translation subpage|...}}",
2716 sortid="page/1934",
2717 )
2718 return None
2720 def recurse(
2721 node: Union[str, WikiNode], seq: Union[list[str], tuple[str, ...]]
2722 ) -> Optional[Union[str, WikiNode]]:
2723 # print(f"seq: {seq}")
2724 if not seq:
2725 return node
2726 if not isinstance(node, WikiNode):
2727 return None
2728 # print(f"node.kind: {node.kind}")
2729 if node.kind in LEVEL_KINDS:
2730 t = clean_node(wxr, None, node.largs[0])
2731 # print(f"t: {t} == seq[0]: {seq[0]}?")
2732 if t.lower() == seq[0].lower():
2733 seq = seq[1:]
2734 if not seq:
2735 return node
2736 for n in node.children:
2737 ret = recurse(n, seq)
2738 if ret is not None:
2739 return ret
2740 return None
2742 tree = wxr.wtp.parse(
2743 subpage_content,
2744 pre_expand=True,
2745 additional_expand=ADDITIONAL_EXPAND_TEMPLATES,
2746 do_not_pre_expand=DO_NOT_PRE_EXPAND_TEMPLATES,
2747 )
2748 assert tree.kind == NodeKind.ROOT
2749 for seq in seqs:
2750 ret = recurse(tree, seq)
2751 if ret is None:
2752 wxr.wtp.debug(
2753 "Failed to find subpage section {}/{} seq {}".format(
2754 title, subtitle, seq
2755 ),
2756 sortid="page/1963",
2757 )
2758 return ret
2760 def parse_translations(data: WordData, xlatnode: WikiNode) -> None:
2761 """Parses translations for a word. This may also pull in translations
2762 from separate translation subpages."""
2763 assert isinstance(data, dict)
2764 assert isinstance(xlatnode, WikiNode)
2765 # print("===== PARSE_TRANSLATIONS {} {} {}"
2766 # .format(wxr.wtp.title, wxr.wtp.section, wxr.wtp.subsection))
2767 # print("parse_translations xlatnode={}".format(xlatnode))
2768 if not wxr.config.capture_translations: 2768 ↛ 2769line 2768 didn't jump to line 2769 because the condition on line 2768 was never true
2769 return
2770 sense_parts: list[Union[WikiNode, str]] = []
2771 sense: Optional[str] = None
2773 def parse_translation_item(
2774 contents: list[Union[WikiNode, str]], lang: Optional[str] = None
2775 ) -> None:
2776 nonlocal sense
2777 assert isinstance(contents, list)
2778 assert lang is None or isinstance(lang, str)
2779 # print("PARSE_TRANSLATION_ITEM:", contents)
2781 langcode: Optional[str] = None
2782 if sense is None:
2783 sense = clean_node(wxr, data, sense_parts).strip()
2784 # print("sense <- clean_node: ", sense)
2785 idx = sense.find("See also translations at")
2786 if idx > 0: 2786 ↛ 2787line 2786 didn't jump to line 2787 because the condition on line 2786 was never true
2787 wxr.wtp.debug(
2788 "Skipping translation see also: {}".format(sense),
2789 sortid="page/2361",
2790 )
2791 sense = sense[:idx].strip()
2792 if sense.endswith(":"): 2792 ↛ 2793line 2792 didn't jump to line 2793 because the condition on line 2792 was never true
2793 sense = sense[:-1].strip()
2794 if sense.endswith("—"): 2794 ↛ 2795line 2794 didn't jump to line 2795 because the condition on line 2794 was never true
2795 sense = sense[:-1].strip()
2796 translations_from_template: list[str] = []
2798 def translation_item_template_fn(
2799 name: str, ht: TemplateArgs
2800 ) -> Optional[str]:
2801 nonlocal langcode
2802 # print("TRANSLATION_ITEM_TEMPLATE_FN:", name, ht)
2803 if is_panel_template(wxr, name):
2804 return ""
2805 if name in ("t+check", "t-check", "t-needed"):
2806 # We ignore these templates. They seem to have outright
2807 # garbage in some entries, and very varying formatting in
2808 # others. These should be transitory and unreliable
2809 # anyway.
2810 return "__IGNORE__"
2811 if name in ("t", "t+", "t-simple", "tt", "tt+"):
2812 code = ht.get(1)
2813 if code: 2813 ↛ 2823line 2813 didn't jump to line 2823 because the condition on line 2813 was always true
2814 if langcode and code != langcode:
2815 wxr.wtp.debug(
2816 "inconsistent language codes {} vs "
2817 "{} in translation item: {!r} {}".format(
2818 langcode, code, name, ht
2819 ),
2820 sortid="page/2386",
2821 )
2822 langcode = code
2823 tr = ht.get(2)
2824 if tr:
2825 tr = clean_node(wxr, None, [tr])
2826 translations_from_template.append(tr)
2827 return None
2828 if name == "t-egy":
2829 langcode = "egy"
2830 return None
2831 if name == "ttbc":
2832 code = ht.get(1)
2833 if code: 2833 ↛ 2835line 2833 didn't jump to line 2835 because the condition on line 2833 was always true
2834 langcode = code
2835 return None
2836 if name == "trans-see": 2836 ↛ 2837line 2836 didn't jump to line 2837 because the condition on line 2836 was never true
2837 wxr.wtp.error(
2838 "UNIMPLEMENTED trans-see template", sortid="page/2405"
2839 )
2840 return ""
2841 if name.endswith("-top"): 2841 ↛ 2842line 2841 didn't jump to line 2842 because the condition on line 2841 was never true
2842 return ""
2843 if name.endswith("-bottom"): 2843 ↛ 2844line 2843 didn't jump to line 2844 because the condition on line 2843 was never true
2844 return ""
2845 if name.endswith("-mid"): 2845 ↛ 2846line 2845 didn't jump to line 2846 because the condition on line 2845 was never true
2846 return ""
2847 # wxr.wtp.debug("UNHANDLED TRANSLATION ITEM TEMPLATE: {!r}"
2848 # .format(name),
2849 # sortid="page/2414")
2850 return None
2852 sublists = list(
2853 x
2854 for x in contents
2855 if isinstance(x, WikiNode) and x.kind == NodeKind.LIST
2856 )
2857 contents = list(
2858 x
2859 for x in contents
2860 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST
2861 )
2863 item = clean_node(
2864 wxr, data, contents, template_fn=translation_item_template_fn
2865 )
2866 # print(" TRANSLATION ITEM: {!r} [{}]".format(item, sense))
2868 # Parse the translation item.
2869 if item: 2869 ↛ exitline 2869 didn't return from function 'parse_translation_item' because the condition on line 2869 was always true
2870 lang = parse_translation_item_text(
2871 wxr,
2872 word,
2873 data,
2874 item,
2875 sense,
2876 lang,
2877 langcode,
2878 translations_from_template,
2879 is_reconstruction,
2880 )
2882 # Handle sublists. They are frequently used for different
2883 # scripts for the language and different variants of the
2884 # language. We will include the lower-level header as a
2885 # tag in those cases.
2886 for listnode in sublists:
2887 assert listnode.kind == NodeKind.LIST
2888 for node in listnode.children:
2889 if not isinstance(node, WikiNode): 2889 ↛ 2890line 2889 didn't jump to line 2890 because the condition on line 2889 was never true
2890 continue
2891 if node.kind == NodeKind.LIST_ITEM: 2891 ↛ 2888line 2891 didn't jump to line 2888 because the condition on line 2891 was always true
2892 parse_translation_item(node.children, lang=lang)
2894 def parse_translation_template(node: WikiNode) -> None:
2895 assert isinstance(node, WikiNode)
2897 def template_fn(name: str, ht: TemplateArgs) -> Optional[str]:
2898 nonlocal sense_parts
2899 nonlocal sense
2900 if is_panel_template(wxr, name):
2901 return ""
2902 if name == "see also":
2903 # XXX capture
2904 # XXX for example, "/" has top-level list containing
2905 # see also items. So also should parse those.
2906 return ""
2907 if name == "trans-see":
2908 # XXX capture
2909 return ""
2910 if name == "see translation subpage": 2910 ↛ 2911line 2910 didn't jump to line 2911 because the condition on line 2910 was never true
2911 sense_parts = []
2912 sense = None
2913 sub = ht.get(1, "")
2914 if sub:
2915 m = re.match(
2916 r"\s*(([^:\d]*)\s*\d*)\s*:\s*([^:]*)\s*", sub
2917 )
2918 else:
2919 m = None
2920 etym = ""
2921 etym_numbered = ""
2922 pos = ""
2923 if m:
2924 etym_numbered = m.group(1)
2925 etym = m.group(2)
2926 pos = m.group(3)
2927 if not sub:
2928 wxr.wtp.debug(
2929 "no part-of-speech in "
2930 "{{see translation subpage|...}}, "
2931 "defaulting to just wxr.wtp.section "
2932 "(= language)",
2933 sortid="page/2468",
2934 )
2935 # seq sent to get_subpage_section without sub and pos
2936 seq = [
2937 language,
2938 TRANSLATIONS_TITLE,
2939 ]
2940 elif (
2941 m
2942 and etym.lower().strip() in ETYMOLOGY_TITLES
2943 and pos.lower() in POS_TITLES
2944 ):
2945 seq = [
2946 language,
2947 etym_numbered,
2948 pos,
2949 TRANSLATIONS_TITLE,
2950 ]
2951 elif sub.lower() in POS_TITLES:
2952 # seq with sub but not pos
2953 seq = [
2954 language,
2955 sub,
2956 TRANSLATIONS_TITLE,
2957 ]
2958 else:
2959 # seq with sub and pos
2960 pos = wxr.wtp.subsection or "MISSING_SUBSECTION"
2961 if pos.lower() not in POS_TITLES:
2962 wxr.wtp.debug(
2963 "unhandled see translation subpage: "
2964 "language={} sub={} "
2965 "wxr.wtp.subsection={}".format(
2966 language, sub, wxr.wtp.subsection
2967 ),
2968 sortid="page/2478",
2969 )
2970 seq = [language, sub, pos, TRANSLATIONS_TITLE]
2971 subnode = get_subpage_section(
2972 wxr.wtp.title or "MISSING_TITLE",
2973 TRANSLATIONS_TITLE,
2974 [seq],
2975 )
2976 if subnode is None or not isinstance(subnode, WikiNode):
2977 # Failed to find the normal subpage section
2978 # seq with sub and pos
2979 pos = wxr.wtp.subsection or "MISSING_SUBSECTION"
2980 # print(f"{language=}, {pos=}, {TRANSLATIONS_TITLE=}")
2981 seqs: list[list[str] | tuple[str, ...]] = [
2982 [TRANSLATIONS_TITLE],
2983 [language, pos],
2984 ]
2985 subnode = get_subpage_section(
2986 wxr.wtp.title or "MISSING_TITLE",
2987 TRANSLATIONS_TITLE,
2988 seqs,
2989 )
2990 if subnode is not None and isinstance(subnode, WikiNode):
2991 parse_translations(data, subnode)
2992 return ""
2993 if name in (
2994 "c",
2995 "C",
2996 "categorize",
2997 "cat",
2998 "catlangname",
2999 "topics",
3000 "top",
3001 "qualifier",
3002 "cln",
3003 ):
3004 # These are expanded in the default way
3005 return None
3006 if name in (
3007 "trans-top",
3008 "trans-top-see",
3009 ):
3010 # XXX capture id from trans-top? Capture sense here
3011 # instead of trying to parse it from expanded content?
3012 if ht.get(1):
3013 sense_parts = []
3014 sense = ht.get(1)
3015 else:
3016 sense_parts = []
3017 sense = None
3018 return None
3019 if name in (
3020 "trans-bottom",
3021 "trans-mid",
3022 "checktrans-mid",
3023 "checktrans-bottom",
3024 ):
3025 return None
3026 if name == "checktrans-top":
3027 sense_parts = []
3028 sense = None
3029 return ""
3030 if name == "trans-top-also":
3031 # XXX capture?
3032 sense_parts = []
3033 sense = None
3034 return ""
3035 wxr.wtp.error(
3036 "UNIMPLEMENTED parse_translation_template: {} {}".format(
3037 name, ht
3038 ),
3039 sortid="page/2517",
3040 )
3041 return ""
3043 wxr.wtp.expand(
3044 wxr.wtp.node_to_wikitext(node), template_fn=template_fn
3045 )
3047 def parse_translation_recurse(xlatnode: WikiNode) -> None:
3048 nonlocal sense
3049 nonlocal sense_parts
3050 for node in xlatnode.children:
3051 # print(node)
3052 if isinstance(node, str):
3053 if sense:
3054 if not node.isspace():
3055 wxr.wtp.debug(
3056 "skipping string in the middle of "
3057 "translations: {}".format(node),
3058 sortid="page/2530",
3059 )
3060 continue
3061 # Add a part to the sense
3062 sense_parts.append(node)
3063 sense = None
3064 continue
3065 assert isinstance(node, WikiNode)
3066 kind = node.kind
3067 if kind == NodeKind.LIST:
3068 for item in node.children:
3069 if not isinstance(item, WikiNode): 3069 ↛ 3070line 3069 didn't jump to line 3070 because the condition on line 3069 was never true
3070 continue
3071 if item.kind != NodeKind.LIST_ITEM: 3071 ↛ 3072line 3071 didn't jump to line 3072 because the condition on line 3071 was never true
3072 continue
3073 if item.sarg == ":": 3073 ↛ 3074line 3073 didn't jump to line 3074 because the condition on line 3073 was never true
3074 continue
3075 parse_translation_item(item.children)
3076 elif kind == NodeKind.LIST_ITEM and node.sarg == ":": 3076 ↛ 3080line 3076 didn't jump to line 3080 because the condition on line 3076 was never true
3077 # Silently skip list items that are just indented; these
3078 # are used for text between translations, such as indicating
3079 # translations that need to be checked.
3080 pass
3081 elif kind == NodeKind.TEMPLATE:
3082 parse_translation_template(node)
3083 elif kind in ( 3083 ↛ 3088line 3083 didn't jump to line 3088 because the condition on line 3083 was never true
3084 NodeKind.TABLE,
3085 NodeKind.TABLE_ROW,
3086 NodeKind.TABLE_CELL,
3087 ):
3088 parse_translation_recurse(node)
3089 elif kind == NodeKind.HTML:
3090 if node.attrs.get("class") == "NavFrame": 3090 ↛ 3096line 3090 didn't jump to line 3096 because the condition on line 3090 was never true
3091 # Reset ``sense_parts`` (and force recomputing
3092 # by clearing ``sense``) as each NavFrame specifies
3093 # its own sense. This helps eliminate garbage coming
3094 # from text at the beginning at the translations
3095 # section.
3096 sense_parts = []
3097 sense = None
3098 # for item in node.children:
3099 # if not isinstance(item, WikiNode):
3100 # continue
3101 # parse_translation_recurse(item)
3102 parse_translation_recurse(node)
3103 elif kind in LEVEL_KINDS: 3103 ↛ 3105line 3103 didn't jump to line 3105 because the condition on line 3103 was never true
3104 # Sub-levels will be recursed elsewhere
3105 pass
3106 elif kind in (NodeKind.ITALIC, NodeKind.BOLD):
3107 parse_translation_recurse(node)
3108 elif kind == NodeKind.PREFORMATTED: 3108 ↛ 3109line 3108 didn't jump to line 3109 because the condition on line 3108 was never true
3109 print("parse_translation_recurse: PREFORMATTED:", node)
3110 elif kind == NodeKind.LINK: 3110 ↛ 3164line 3110 didn't jump to line 3164 because the condition on line 3110 was always true
3111 arg0 = node.largs[0]
3112 # Kludge: I've seen occasional normal links to translation
3113 # subpages from main pages (e.g., language/English/Noun
3114 # in July 2021) instead of the normal
3115 # {{see translation subpage|...}} template. This should
3116 # handle them. Note: must be careful not to read other
3117 # links, particularly things like in "human being":
3118 # "a human being -- see [[man/translations]]" (group title)
3119 if ( 3119 ↛ 3127line 3119 didn't jump to line 3127 because the condition on line 3119 was never true
3120 isinstance(arg0, (list, tuple))
3121 and arg0
3122 and isinstance(arg0[0], str)
3123 and arg0[0].endswith("/" + TRANSLATIONS_TITLE)
3124 and arg0[0][: -(1 + len(TRANSLATIONS_TITLE))]
3125 == wxr.wtp.title
3126 ):
3127 wxr.wtp.debug(
3128 "translations subpage link found on main "
3129 "page instead "
3130 "of normal {{see translation subpage|...}}",
3131 sortid="page/2595",
3132 )
3133 sub = wxr.wtp.subsection or "MISSING_SUBSECTION"
3134 if sub.lower() in POS_TITLES:
3135 seq = [
3136 language,
3137 sub,
3138 TRANSLATIONS_TITLE,
3139 ]
3140 subnode = get_subpage_section(
3141 wxr.wtp.title,
3142 TRANSLATIONS_TITLE,
3143 [seq],
3144 )
3145 if subnode is not None and isinstance(
3146 subnode, WikiNode
3147 ):
3148 parse_translations(data, subnode)
3149 else:
3150 wxr.wtp.error(
3151 "/translations link outside part-of-speech"
3152 )
3154 if (
3155 len(arg0) >= 1
3156 and isinstance(arg0[0], str)
3157 and not arg0[0].lower().startswith("category:")
3158 ):
3159 for x in node.largs[-1]:
3160 if isinstance(x, str): 3160 ↛ 3163line 3160 didn't jump to line 3163 because the condition on line 3160 was always true
3161 sense_parts.append(x)
3162 else:
3163 parse_translation_recurse(x)
3164 elif not sense:
3165 sense_parts.append(node)
3166 else:
3167 wxr.wtp.debug(
3168 "skipping text between translation items/senses: "
3169 "{}".format(node),
3170 sortid="page/2621",
3171 )
3173 # Main code of parse_translation(). We want ``sense`` to be assigned
3174 # regardless of recursion levels, and thus the code is structured
3175 # to define at this level and recurse in parse_translation_recurse().
3176 parse_translation_recurse(xlatnode)
3178 def parse_etymology(data: WordData, node: LevelNode) -> None:
3179 """Parses an etymology section."""
3180 assert isinstance(data, dict)
3181 assert isinstance(node, WikiNode)
3183 templates: list[TemplateData] = []
3185 # Counter for preventing the capture of etymology templates
3186 # when we are inside templates that we want to ignore (i.e.,
3187 # not capture).
3188 ignore_count = 0
3190 def etym_template_fn(name: str, ht: TemplateArgs) -> Optional[str]:
3191 nonlocal ignore_count
3192 if is_panel_template(wxr, name) or name in ["zh-x", "zh-q"]:
3193 return ""
3194 if re.match(ignored_etymology_templates_re, name):
3195 ignore_count += 1
3196 return None
3198 def etym_post_template_fn(
3199 name: str, ht: TemplateArgs, expansion: str
3200 ) -> None:
3201 nonlocal ignore_count
3202 if name in wikipedia_templates:
3203 parse_wikipedia_template(wxr, data, ht)
3204 return None
3205 if re.match(ignored_etymology_templates_re, name):
3206 ignore_count -= 1
3207 return None
3208 if ignore_count == 0: 3208 ↛ 3214line 3208 didn't jump to line 3214 because the condition on line 3208 was always true
3209 ht = clean_template_args(wxr, ht)
3210 expansion = clean_node(wxr, None, expansion)
3211 templates.append(
3212 {"name": name, "args": ht, "expansion": expansion}
3213 )
3214 return None
3216 # Remove any subsections
3217 contents = list(
3218 x
3219 for x in node.children
3220 if not isinstance(x, WikiNode) or x.kind not in LEVEL_KINDS
3221 )
3222 # Collect expanded links separately from templates. Generic linking
3223 # templates such as m/l remain ignored in etymology_templates, but
3224 # their destinations (and ordinary wikilinks) are still useful.
3225 links: list[tuple[str, str]] = []
3226 # Convert to text, also capturing templates using post_template_fn
3227 text = clean_node(
3228 wxr,
3229 None,
3230 contents,
3231 template_fn=etym_template_fn,
3232 post_template_fn=etym_post_template_fn,
3233 link_collector=links,
3234 ).strip(": \n") # remove ":" indent wikitext before zh-x template
3235 # Save the collected information.
3236 if len(text) > 0:
3237 data["etymology_text"] = text
3238 if links:
3239 data["etymology_links"] = links
3240 if len(templates) > 0:
3241 # Some etymology templates, like Template:root do not generate
3242 # text, so they should be added here. Elsewhere, we check
3243 # for Template:root and add some text to the expansion to please
3244 # the validation.
3245 data["etymology_templates"] = templates
3247 for child_node in node.find_child_recursively( 3247 ↛ exitline 3247 didn't return from function 'parse_etymology' because the loop on line 3247 didn't complete
3248 LEVEL_KIND_FLAGS | NodeKind.TEMPLATE
3249 ):
3250 if child_node.kind in LEVEL_KIND_FLAGS:
3251 break
3252 elif isinstance( 3252 ↛ 3255line 3252 didn't jump to line 3255 because the condition on line 3252 was never true
3253 child_node, TemplateNode
3254 ) and child_node.template_name in ["zh-x", "zh-q"]:
3255 if "etymology_examples" not in data:
3256 data["etymology_examples"] = []
3257 data["etymology_examples"].extend(
3258 extract_template_zh_x(
3259 wxr, child_node, None, ExampleData(raw_tags=[], tags=[])
3260 )
3261 )
3263 def process_children(treenode: WikiNode, pos: Optional[str]) -> None:
3264 """This recurses into a subtree in the parse tree for a page."""
3265 nonlocal etym_data
3266 nonlocal pos_data
3267 nonlocal inside_level_four
3269 redirect_list: list[str] = [] # for `zh-see` template
3271 def skip_template_fn(name: str, ht: TemplateArgs) -> Optional[str]:
3272 """This is called for otherwise unprocessed parts of the page.
3273 We still expand them so that e.g. Category links get captured."""
3274 if name in wikipedia_templates:
3275 data = select_data()
3276 parse_wikipedia_template(wxr, data, ht)
3277 return None
3278 if is_panel_template(wxr, name):
3279 return ""
3280 return None
3282 for node in treenode.children:
3283 if not isinstance(node, WikiNode):
3284 # print(" X{}".format(repr(node)[:40]))
3285 continue
3286 if isinstance(node, TemplateNode):
3287 if process_soft_redirect_template(wxr, node, redirect_list):
3288 continue
3289 elif node.template_name == "zh-forms":
3290 extract_zh_forms_template(wxr, node, select_data())
3291 elif (
3292 node.template_name.endswith("-kanjitab")
3293 or node.template_name == "ja-kt"
3294 ):
3295 extract_ja_kanjitab_template(wxr, node, select_data())
3296 elif node.template_name in ETYMOLOGY_TEMPLATES_IN_HEADS:
3297 args_ht = clean_template_args(wxr, node.template_parameters)
3298 expansion = clean_node(wxr, etym_data, node)
3299 etymology_template_append(
3300 etym_data, node.template_name, args_ht, expansion
3301 )
3303 if not isinstance(node, LevelNode):
3304 # XXX handle e.g. wikipedia links at the top of a language
3305 # XXX should at least capture "also" at top of page
3306 if node.kind in (
3307 NodeKind.HLINE,
3308 NodeKind.LIST,
3309 NodeKind.LIST_ITEM,
3310 ):
3311 continue
3312 # print(" UNEXPECTED: {}".format(node))
3313 # Clean the node to collect category links
3314 clean_node(wxr, etym_data, node, template_fn=skip_template_fn)
3315 continue
3316 t = clean_node(
3317 wxr, etym_data, node.sarg if node.sarg else node.largs
3318 )
3319 t = t.lower()
3320 # XXX these counts were never implemented fully, and even this
3321 # gets discarded: Search STATISTICS_IMPLEMENTATION
3322 wxr.config.section_counts[t] += 1
3323 # print("PROCESS_CHILDREN: T:", repr(t))
3324 if t in IGNORED_TITLES:
3325 pass
3326 elif t.startswith(PRONUNCIATION_TITLE):
3327 # Chinese Pronunciation section kludge; we demote these to
3328 # be level 4 instead of 3 so that they're part of a larger
3329 # etymology hierarchy; usually the data here is empty and
3330 # acts as an inbetween between POS and Etymology data
3331 if lang_code in ("zh",):
3332 inside_level_four = True
3333 if t.startswith(PRONUNCIATION_TITLE + " "):
3334 # Pronunciation 1, etc, are used in Chinese Glyphs,
3335 # and each of them may have senses under Definition
3336 push_level_four_section(True)
3337 wxr.wtp.start_subsection(None)
3338 if wxr.config.capture_pronunciation: 3338 ↛ 3446line 3338 didn't jump to line 3446 because the condition on line 3338 was always true
3339 data = select_data()
3340 parse_pronunciation(
3341 wxr,
3342 node,
3343 data,
3344 etym_data,
3345 have_etym,
3346 base_data,
3347 lang_code,
3348 )
3349 elif t.startswith(tuple(ETYMOLOGY_TITLES)):
3350 push_etym()
3351 wxr.wtp.start_subsection(None)
3352 if wxr.config.capture_etymologies: 3352 ↛ 3446line 3352 didn't jump to line 3446 because the condition on line 3352 was always true
3353 m = re.search(r"\s(\d+(\.\d+)?)$", t)
3354 if m:
3355 etym_data["etymology_number"] = m.group(1)
3356 parse_etymology(etym_data, node)
3357 elif t == DESCENDANTS_TITLE and wxr.config.capture_descendants:
3358 data = select_data()
3359 extract_descendant_section(wxr, data, node, False)
3360 elif (
3361 t in PROTO_ROOT_DERIVED_TITLES
3362 and pos == "root"
3363 and is_reconstruction
3364 and wxr.config.capture_descendants
3365 ):
3366 data = select_data()
3367 extract_descendant_section(wxr, data, node, True)
3368 elif t == TRANSLATIONS_TITLE:
3369 data = select_data()
3370 parse_translations(data, node)
3371 elif t in INFLECTION_TITLES:
3372 parse_inflection(node, t, pos)
3373 elif t == "alternative forms":
3374 extract_alt_form_section(wxr, select_data(), node)
3375 else:
3376 lst = t.split()
3377 while len(lst) > 1 and lst[-1].isdigit():
3378 lst = lst[:-1]
3379 t_no_number = " ".join(lst).lower()
3380 if t_no_number in POS_TITLES:
3381 push_pos()
3382 dt = POS_TITLES[t_no_number] # type:ignore[literal-required]
3383 pos = dt["pos"] or "MISSING_POS"
3384 wxr.wtp.start_subsection(t)
3385 if "debug" in dt:
3386 wxr.wtp.debug(
3387 "{} in section {}".format(dt["debug"], t),
3388 sortid="page/2755",
3389 )
3390 if "warning" in dt: 3390 ↛ 3391line 3390 didn't jump to line 3391 because the condition on line 3390 was never true
3391 wxr.wtp.wiki_notice(
3392 "{} in section {}".format(dt["warning"], t),
3393 sortid="page/2759",
3394 )
3395 if "error" in dt: 3395 ↛ 3396line 3395 didn't jump to line 3396 because the condition on line 3395 was never true
3396 wxr.wtp.error(
3397 "{} in section {}".format(dt["error"], t),
3398 sortid="page/2763",
3399 )
3400 if "note" in dt: 3400 ↛ 3401line 3400 didn't jump to line 3401 because the condition on line 3400 was never true
3401 wxr.wtp.note(
3402 "{} in section {}".format(dt["note"], t),
3403 sortid="page/20251017a",
3404 )
3405 if "wiki_notice" in dt: 3405 ↛ 3406line 3405 didn't jump to line 3406 because the condition on line 3405 was never true
3406 wxr.wtp.wiki_notice(
3407 "{} in section {}".format(dt["wiki_notices"], t),
3408 sortid="page/20251017b",
3409 )
3410 # Parse word senses for the part-of-speech
3411 parse_part_of_speech(node, pos)
3412 if "tags" in dt:
3413 for pdata in sense_datas:
3414 data_extend(pdata, "tags", dt["tags"])
3415 elif t_no_number in LINKAGE_TITLES:
3416 # print(f"LINKAGE_TITLES NODE {node=}")
3417 rel = LINKAGE_TITLES[t_no_number]
3418 data = select_data()
3419 parse_linkage(
3420 wxr,
3421 data,
3422 rel,
3423 node,
3424 word,
3425 sense_datas,
3426 is_reconstruction,
3427 )
3428 elif t_no_number == COMPOUNDS_TITLE:
3429 data = select_data()
3430 if wxr.config.capture_compounds: 3430 ↛ 3446line 3430 didn't jump to line 3446 because the condition on line 3430 was always true
3431 parse_linkage(
3432 wxr,
3433 data,
3434 "derived",
3435 node,
3436 word,
3437 sense_datas,
3438 is_reconstruction,
3439 )
3441 # XXX parse interesting templates also from other sections. E.g.,
3442 # {{Letter|...}} in ===See also===
3443 # Also <gallery>
3445 # Recurse to children of this node, processing subtitles therein
3446 stack.append(t)
3447 process_children(node, pos)
3448 stack.pop()
3450 if len(redirect_list) > 0:
3451 if len(pos_data) > 0:
3452 pos_data["redirects"] = redirect_list
3453 if "pos" not in pos_data: 3453 ↛ 3454line 3453 didn't jump to line 3454 because the condition on line 3453 was never true
3454 pos_data["pos"] = "soft-redirect"
3455 else:
3456 new_page_data = copy.deepcopy(base_data)
3457 new_page_data["redirects"] = redirect_list
3458 if "pos" not in new_page_data: 3458 ↛ 3460line 3458 didn't jump to line 3460 because the condition on line 3458 was always true
3459 new_page_data["pos"] = "soft-redirect"
3460 new_page_data["senses"] = [{"tags": ["no-gloss"]}]
3461 page_datas.append(new_page_data)
3463 def extract_examples(
3464 others: list[WikiNode], sense_base: SenseData
3465 ) -> list[ExampleData]:
3466 """Parses through a list of definitions and quotes to find examples.
3467 Returns a list of example dicts to be added to sense data. Adds
3468 meta-data, mostly categories, into sense_base."""
3469 assert isinstance(others, list)
3470 examples: list[ExampleData] = []
3472 for sub in others:
3473 if not sub.sarg.endswith((":", "*")): 3473 ↛ 3474line 3473 didn't jump to line 3474 because the condition on line 3473 was never true
3474 continue
3475 for item in sub.children:
3476 if not isinstance(item, WikiNode): 3476 ↛ 3477line 3476 didn't jump to line 3477 because the condition on line 3476 was never true
3477 continue
3478 if item.kind != NodeKind.LIST_ITEM: 3478 ↛ 3479line 3478 didn't jump to line 3479 because the condition on line 3478 was never true
3479 continue
3480 usex_type = None
3481 example_template_args = []
3482 example_template_names = []
3483 taxons = set()
3485 # Bypass this function when parsing Chinese, Japanese and
3486 # quotation templates.
3487 new_example_lists = extract_example_list_item(
3488 wxr, item, sense_base, ExampleData(raw_tags=[], tags=[])
3489 )
3490 if len(new_example_lists) > 0:
3491 examples.extend(new_example_lists)
3492 continue
3494 def usex_template_fn(
3495 name: str, ht: TemplateArgs
3496 ) -> Optional[str]:
3497 nonlocal usex_type
3498 if is_panel_template(wxr, name):
3499 return ""
3500 if name in usex_templates:
3501 usex_type = "example"
3502 example_template_args.append(ht)
3503 example_template_names.append(name)
3504 elif name in quotation_templates:
3505 usex_type = "quotation"
3506 elif name in taxonomy_templates: 3506 ↛ 3507line 3506 didn't jump to line 3507 because the condition on line 3506 was never true
3507 taxons.update(ht.get(1, "").split())
3508 for prefix in template_linkages_to_ignore_in_examples:
3509 if re.search(
3510 r"(^|[-/\s]){}($|\b|[0-9])".format(prefix), name
3511 ):
3512 return ""
3513 return None
3515 # bookmark
3516 ruby: list[tuple[str, str]] = []
3517 contents = item.children
3518 if lang_code == "ja":
3519 # Capture ruby contents if this is a Japanese language
3520 # example.
3521 # print(contents)
3522 if ( 3522 ↛ 3527line 3522 didn't jump to line 3527 because the condition on line 3522 was never true
3523 contents
3524 and isinstance(contents, str)
3525 and re.match(r"\s*$", contents[0])
3526 ):
3527 contents = contents[1:]
3528 exp = wxr.wtp.parse(
3529 wxr.wtp.node_to_wikitext(contents),
3530 # post_template_fn=head_post_template_fn,
3531 expand_all=True,
3532 )
3533 rub, rest = extract_ruby(wxr, exp.children)
3534 if rub:
3535 for rtup in rub:
3536 ruby.append(rtup)
3537 contents = rest
3538 subtext = clean_node(
3539 wxr, sense_base, contents, template_fn=usex_template_fn
3540 )
3542 frozen_taxons = frozenset(taxons)
3543 classify_desc2 = partial(classify_desc, accepted=frozen_taxons)
3545 # print(f"{subtext=}")
3546 subtext = re.sub(
3547 r"\s*\(please add an English "
3548 r"translation of this "
3549 r"(example|usage example|quote)\)",
3550 "",
3551 subtext,
3552 ).strip()
3553 subtext = re.sub(r"\^\([^)]*\)", "", subtext)
3554 subtext = re.sub(r"\s*[―—]+$", "", subtext)
3555 # print("subtext:", repr(subtext))
3557 lines = subtext.splitlines()
3558 # print(lines)
3560 lines = list(re.sub(r"^[#:*]*", "", x).strip() for x in lines)
3561 lines = list(
3562 x
3563 for x in lines
3564 if not re.match(
3565 r"(Synonyms: |Antonyms: |Hyponyms: |"
3566 r"Synonym: |Antonym: |Hyponym: |"
3567 r"Hypernyms: |Derived terms: |"
3568 r"Related terms: |"
3569 r"Hypernym: |Derived term: |"
3570 r"Coordinate terms:|"
3571 r"Related term: |"
3572 r"For more quotations using )",
3573 x,
3574 )
3575 )
3576 tr = ""
3577 ref = ""
3578 roman = ""
3579 # for line in lines:
3580 # print("LINE:", repr(line))
3581 # print(classify_desc(line))
3582 if len(lines) == 1 and lang_code != "en":
3583 parts = example_splitter_re.split(lines[0])
3584 if ( 3584 ↛ 3592line 3584 didn't jump to line 3592 because the condition on line 3584 was never true
3585 len(parts) > 2
3586 and len(example_template_args) == 1
3587 and any(
3588 ("―" in s) or ("—" in s)
3589 for s in example_template_args[0].values()
3590 )
3591 ):
3592 if nparts := synch_splits_with_args(
3593 lines[0], example_template_args[0]
3594 ):
3595 parts = nparts
3596 if ( 3596 ↛ 3601line 3596 didn't jump to line 3601 because the condition on line 3596 was never true
3597 len(example_template_args) == 1
3598 and "lit" in example_template_args[0]
3599 ):
3600 # ugly brute-force kludge in case there's a lit= arg
3601 literally = example_template_args[0].get("lit", "")
3602 if literally:
3603 literally = (
3604 " (literally, “"
3605 + clean_value(wxr, literally)
3606 + "”)"
3607 )
3608 else:
3609 literally = ""
3610 if ( 3610 ↛ 3649line 3610 didn't jump to line 3649 because the condition on line 3610 was never true
3611 len(example_template_args) == 1
3612 and len(parts) == 2
3613 and len(example_template_args[0])
3614 - (
3615 # horrible kludge to ignore these arguments
3616 # when calculating how many there are
3617 sum(
3618 s in example_template_args[0]
3619 for s in (
3620 "lit", # generates text, but we handle it
3621 "inline",
3622 "noenum",
3623 "nocat",
3624 "sort",
3625 )
3626 )
3627 )
3628 == 3
3629 and clean_value(
3630 wxr, example_template_args[0].get(2, "")
3631 )
3632 == parts[0].strip()
3633 and clean_value(
3634 wxr,
3635 (
3636 example_template_args[0].get(3)
3637 or example_template_args[0].get("translation")
3638 or example_template_args[0].get("t", "")
3639 )
3640 + literally, # in case there's a lit= argument
3641 )
3642 == parts[1].strip()
3643 ):
3644 # {{exampletemplate|ex|Foo bar baz|English translation}}
3645 # is a pretty reliable 'heuristic', so we use it here
3646 # before the others. To be extra sure the template
3647 # doesn't do anything weird, we compare the arguments
3648 # and the output to each other.
3649 lines = [parts[0].strip()]
3650 tr = parts[1].strip()
3651 elif (
3652 len(parts) == 2
3653 and classify_desc2(parts[1]) in ENGLISH_TEXTS
3654 ):
3655 # These other branches just do some simple heuristics w/
3656 # the expanded output of the template (if applicable).
3657 lines = [parts[0].strip()]
3658 tr = parts[1].strip()
3659 elif ( 3659 ↛ 3665line 3659 didn't jump to line 3665 because the condition on line 3659 was never true
3660 len(parts) == 3
3661 and classify_desc2(parts[1])
3662 in ("romanization", "english")
3663 and classify_desc2(parts[2]) in ENGLISH_TEXTS
3664 ):
3665 lines = [parts[0].strip()]
3666 roman = parts[1].strip()
3667 tr = parts[2].strip()
3668 else:
3669 parts = re.split(r"\s+-\s+", lines[0])
3670 if ( 3670 ↛ 3674line 3670 didn't jump to line 3674 because the condition on line 3670 was never true
3671 len(parts) == 2
3672 and classify_desc2(parts[1]) in ENGLISH_TEXTS
3673 ):
3674 lines = [parts[0].strip()]
3675 tr = parts[1].strip()
3676 elif len(lines) > 1:
3677 if any(
3678 re.search(r"[]\d:)]\s*$", x) for x in lines[:-1]
3679 ) and not (len(example_template_names) == 1):
3680 refs: list[str] = []
3681 for i in range(len(lines)): 3681 ↛ 3687line 3681 didn't jump to line 3687 because the loop on line 3681 didn't complete
3682 if re.match(r"^[#*]*:+(\s*$|\s+)", lines[i]): 3682 ↛ 3683line 3682 didn't jump to line 3683 because the condition on line 3682 was never true
3683 break
3684 refs.append(lines[i].strip())
3685 if re.search(r"[]\d:)]\s*$", lines[i]):
3686 break
3687 ref = " ".join(refs)
3688 lines = lines[i + 1 :]
3689 if (
3690 lang_code != "en"
3691 and len(lines) >= 2
3692 and classify_desc2(lines[-1]) in ENGLISH_TEXTS
3693 ):
3694 i = len(lines) - 1
3695 while ( 3695 ↛ 3700line 3695 didn't jump to line 3700 because the condition on line 3695 was never true
3696 i > 1
3697 and classify_desc2(lines[i - 1])
3698 in ENGLISH_TEXTS
3699 ):
3700 i -= 1
3701 tr = "\n".join(lines[i:])
3702 lines = lines[:i]
3703 if len(lines) >= 2:
3704 if classify_desc2(lines[-1]) == "romanization":
3705 roman = lines[-1].strip()
3706 lines = lines[:-1]
3708 elif lang_code == "en" and re.match(r"^[#*]*:+", lines[1]):
3709 ref = lines[0]
3710 lines = lines[1:]
3711 elif lang_code != "en" and len(lines) == 2:
3712 cls1 = classify_desc2(lines[0])
3713 cls2 = classify_desc2(lines[1])
3714 if cls2 in ENGLISH_TEXTS and cls1 != "english":
3715 tr = lines[1]
3716 lines = [lines[0]]
3717 elif cls1 in ENGLISH_TEXTS and cls2 != "english": 3717 ↛ 3718line 3717 didn't jump to line 3718 because the condition on line 3717 was never true
3718 tr = lines[0]
3719 lines = [lines[1]]
3720 elif ( 3720 ↛ 3727line 3720 didn't jump to line 3727 because the condition on line 3720 was never true
3721 re.match(r"^[#*]*:+", lines[1])
3722 and classify_desc2(
3723 re.sub(r"^[#*:]+\s*", "", lines[1])
3724 )
3725 in ENGLISH_TEXTS
3726 ):
3727 tr = re.sub(r"^[#*:]+\s*", "", lines[1])
3728 lines = [lines[0]]
3729 elif cls1 == "english" and cls2 in ENGLISH_TEXTS:
3730 # Both were classified as English, but
3731 # presumably one is not. Assume first is
3732 # non-English, as that seems more common.
3733 tr = lines[1]
3734 lines = [lines[0]]
3735 elif (
3736 usex_type != "quotation"
3737 and lang_code != "en"
3738 and len(lines) == 3
3739 ):
3740 cls1 = classify_desc2(lines[0])
3741 cls2 = classify_desc2(lines[1])
3742 cls3 = classify_desc2(lines[2])
3743 if (
3744 cls3 == "english"
3745 and cls2 in ("english", "romanization")
3746 and cls1 != "english"
3747 ):
3748 tr = lines[2].strip()
3749 roman = lines[1].strip()
3750 lines = [lines[0].strip()]
3751 elif ( 3751 ↛ 3759line 3751 didn't jump to line 3759 because the condition on line 3751 was never true
3752 usex_type == "quotation"
3753 and lang_code != "en"
3754 and len(lines) > 2
3755 ):
3756 # for x in lines:
3757 # print(" LINE: {}: {}"
3758 # .format(classify_desc2(x), x))
3759 if re.match(r"^[#*]*:+\s*$", lines[1]):
3760 ref = lines[0]
3761 lines = lines[2:]
3762 cls1 = classify_desc2(lines[-1])
3763 if cls1 == "english":
3764 i = len(lines) - 1
3765 while (
3766 i > 1
3767 and classify_desc2(lines[i - 1])
3768 == ENGLISH_TEXTS
3769 ):
3770 i -= 1
3771 tr = "\n".join(lines[i:])
3772 lines = lines[:i]
3774 roman = re.sub(r"[ \t\r]+", " ", roman).strip()
3775 roman = re.sub(r"\[\s*…\s*\]", "[…]", roman)
3776 tr = re.sub(r"^[#*:]+\s*", "", tr)
3777 tr = re.sub(r"[ \t\r]+", " ", tr).strip()
3778 tr = re.sub(r"\[\s*…\s*\]", "[…]", tr)
3779 ref = re.sub(r"^[#*:]+\s*", "", ref)
3780 ref = re.sub(
3781 r", (volume |number |page )?“?"
3782 r"\(please specify ([^)]|\(s\))*\)”?|"
3783 ", text here$",
3784 "",
3785 ref,
3786 )
3787 ref = re.sub(r"\[\s*…\s*\]", "[…]", ref)
3788 lines = list(re.sub(r"^[#*:]+\s*", "", x) for x in lines)
3789 subtext = "\n".join(x for x in lines if x)
3790 if not tr and lang_code != "en":
3791 m = re.search(r"([.!?])\s+\(([^)]+)\)\s*$", subtext)
3792 if m and classify_desc2(m.group(2)) in ENGLISH_TEXTS: 3792 ↛ 3793line 3792 didn't jump to line 3793 because the condition on line 3792 was never true
3793 tr = m.group(2)
3794 subtext = subtext[: m.start()] + m.group(1)
3795 elif lines:
3796 parts = re.split(r"\s*[―—]+\s*", lines[0])
3797 if ( 3797 ↛ 3801line 3797 didn't jump to line 3801 because the condition on line 3797 was never true
3798 len(parts) == 2
3799 and classify_desc2(parts[1]) in ENGLISH_TEXTS
3800 ):
3801 subtext = parts[0].strip()
3802 tr = parts[1].strip()
3803 subtext = re.sub(r'^[“"`]([^“"`”\']*)[”"\']$', r"\1", subtext)
3804 subtext = re.sub(
3805 r"(please add an English translation of "
3806 r"this (quote|usage example))",
3807 "",
3808 subtext,
3809 )
3810 subtext = re.sub(
3811 r"\s*→New International Version " "translation$",
3812 "",
3813 subtext,
3814 ) # e.g. pis/Tok Pisin (Bible)
3815 subtext = re.sub(r"[ \t\r]+", " ", subtext).strip()
3816 subtext = re.sub(r"\[\s*…\s*\]", "[…]", subtext)
3817 note = None
3818 m = re.match(r"^\(([^)]*)\):\s+", subtext)
3819 if ( 3819 ↛ 3827line 3819 didn't jump to line 3827 because the condition on line 3819 was never true
3820 m is not None
3821 and lang_code != "en"
3822 and (
3823 m.group(1).startswith("with ")
3824 or classify_desc2(m.group(1)) == "english"
3825 )
3826 ):
3827 note = m.group(1)
3828 subtext = subtext[m.end() :]
3829 ref = re.sub(r"\s*\(→ISBN\)", "", ref)
3830 ref = re.sub(r",\s*→ISBN", "", ref)
3831 ref = ref.strip()
3832 if ref.endswith(":") or ref.endswith(","):
3833 ref = ref[:-1].strip()
3834 ref = re.sub(r"\s+,\s+", ", ", ref)
3835 ref = re.sub(r"\s+", " ", ref)
3836 if ref and not subtext: 3836 ↛ 3837line 3836 didn't jump to line 3837 because the condition on line 3836 was never true
3837 subtext = ref
3838 ref = ""
3839 if subtext:
3840 dt: ExampleData = {"text": subtext}
3841 if ref:
3842 dt["ref"] = ref
3843 if tr:
3844 dt["english"] = tr # DEPRECATED for "translation"
3845 dt["translation"] = tr
3846 if usex_type:
3847 dt["type"] = usex_type
3848 if note: 3848 ↛ 3849line 3848 didn't jump to line 3849 because the condition on line 3848 was never true
3849 dt["note"] = note
3850 if roman:
3851 dt["roman"] = roman
3852 if ruby:
3853 dt["ruby"] = ruby
3854 examples.append(dt)
3856 return examples
3858 # Main code of parse_language()
3859 # Process the section
3860 stack.append(language)
3861 process_children(langnode, None)
3862 stack.pop()
3864 # Finalize word entires
3865 push_etym()
3866 ret = []
3867 for data in page_datas:
3868 merge_base(data, base_data)
3869 ret.append(data)
3871 # Copy all tags to word senses
3872 for data in ret:
3873 if "senses" not in data: 3873 ↛ 3874line 3873 didn't jump to line 3874 because the condition on line 3873 was never true
3874 continue
3875 # WordData should not have a 'tags' field, but if it does, it's
3876 # deleted and its contents removed and placed in each sense;
3877 # that's why the type ignores.
3878 tags: Iterable = data.get("tags", ()) # type: ignore[assignment]
3879 if "tags" in data:
3880 del data["tags"] # type: ignore[typeddict-item]
3881 for sense in data["senses"]:
3882 data_extend(sense, "tags", tags)
3884 return ret
3887def parse_wikipedia_template(
3888 wxr: WiktextractContext, data: WordData, ht: TemplateArgs
3889) -> None:
3890 """Helper function for parsing {{wikipedia|...}} and related templates."""
3891 assert isinstance(wxr, WiktextractContext)
3892 assert isinstance(data, dict)
3893 assert isinstance(ht, dict)
3894 langid = clean_node(wxr, data, ht.get("lang", ()))
3895 pagename = (
3896 clean_node(wxr, data, ht.get(1, ()))
3897 or wxr.wtp.title
3898 or "MISSING_PAGE_TITLE"
3899 )
3900 if langid:
3901 data_append(data, "wikipedia", langid + ":" + pagename)
3902 else:
3903 data_append(data, "wikipedia", pagename)
3906def parse_top_template(
3907 wxr: WiktextractContext, node: WikiNode, data: WordData
3908) -> None:
3909 """Parses a template that occurs on the top-level in a page, before any
3910 language subtitles."""
3911 assert isinstance(wxr, WiktextractContext)
3912 assert isinstance(node, WikiNode)
3913 assert isinstance(data, dict)
3915 def top_template_fn(name: str, ht: TemplateArgs) -> Optional[str]:
3916 if name in wikipedia_templates:
3917 parse_wikipedia_template(wxr, data, ht)
3918 return None
3919 if is_panel_template(wxr, name):
3920 return ""
3921 if name in ("reconstruction",): 3921 ↛ 3922line 3921 didn't jump to line 3922 because the condition on line 3921 was never true
3922 return ""
3923 if name.lower() == "also" or name.lower().startswith("also/"):
3924 # XXX shows related words that might really have been the intended
3925 # word, capture them
3926 return ""
3927 if name == "see also": 3927 ↛ 3929line 3927 didn't jump to line 3929 because the condition on line 3927 was never true
3928 # XXX capture
3929 return ""
3930 if name == "cardinalbox": 3930 ↛ 3932line 3930 didn't jump to line 3932 because the condition on line 3930 was never true
3931 # XXX capture
3932 return ""
3933 if name == "character info": 3933 ↛ 3935line 3933 didn't jump to line 3935 because the condition on line 3933 was never true
3934 # XXX capture
3935 return ""
3936 if name == "commonscat": 3936 ↛ 3938line 3936 didn't jump to line 3938 because the condition on line 3936 was never true
3937 # XXX capture link to Wikimedia commons
3938 return ""
3939 if name == "wrongtitle": 3939 ↛ 3942line 3939 didn't jump to line 3942 because the condition on line 3939 was never true
3940 # XXX this should be captured to replace page title with the
3941 # correct title. E.g. ⿰亻革家
3942 return ""
3943 if name == "wikidata": 3943 ↛ 3944line 3943 didn't jump to line 3944 because the condition on line 3943 was never true
3944 arg = clean_node(wxr, data, ht.get(1, ()))
3945 if arg.startswith("Q") or arg.startswith("Lexeme:L"):
3946 data_append(data, "wikidata", arg)
3947 return ""
3948 wxr.wtp.debug(
3949 "UNIMPLEMENTED top-level template: {} {}".format(name, ht),
3950 sortid="page/2870",
3951 )
3952 return ""
3954 clean_node(wxr, None, [node], template_fn=top_template_fn)
3957def fix_subtitle_hierarchy(wxr: WiktextractContext, text: str) -> str:
3958 """Fix subtitle hierarchy to be strict Language -> Etymology ->
3959 Part-of-Speech -> Translation/Linkage. Also merge Etymology sections
3960 that are next to each other."""
3962 # Wiktextract issue #620, Chinese Glyph Origin before an etymology
3963 # section get overwritten. In this case, let's just combine the two.
3965 # In Chinese entries, Pronunciation can be preceded on the
3966 # same level 3 by its Etymology *and* Glyph Origin sections:
3967 # ===Glyph Origin===
3968 # ===Etymology===
3969 # ===Pronunciation===
3970 # Tatu suggested adding a new 'level' between 3 and 4, so Pronunciation
3971 # is now Level 4, POS is shifted to Level 5 and the rest (incl. 'default')
3972 # are now level 6
3974 # Known lowercase PoS names are in part_of_speech_map
3975 # Known lowercase linkage section names are in linkage_map
3977 old = re.split(
3978 r"(?m)^(==+)[ \t]*([^= \t]([^=\n]|=[^=])*?)" r"[ \t]*(==+)[ \t]*$", text
3979 )
3981 parts = []
3982 npar = 4 # Number of parentheses in above expression
3983 parts.append(old[0])
3984 prev_level = None
3985 level = None
3986 skip_level_title = False # When combining etymology sections
3987 for i in range(1, len(old), npar + 1):
3988 left = old[i]
3989 right = old[i + npar - 1]
3990 # remove Wikilinks in title
3991 title = re.sub(r"^\[\[", "", old[i + 1])
3992 title = re.sub(r"\]\]$", "", title)
3993 prev_level = level
3994 level = len(left)
3995 part = old[i + npar]
3996 if level != len(right): 3996 ↛ 3997line 3996 didn't jump to line 3997 because the condition on line 3996 was never true
3997 wxr.wtp.debug(
3998 "subtitle has unbalanced levels: "
3999 "{!r} has {} on the left and {} on the right".format(
4000 title, left, right
4001 ),
4002 sortid="page/2904",
4003 )
4004 lc = title.lower()
4005 if name_to_code(title, "en") != "":
4006 if level > 2: 4006 ↛ 4007line 4006 didn't jump to line 4007 because the condition on line 4006 was never true
4007 wxr.wtp.debug(
4008 "subtitle has language name {} at level {}".format(
4009 title, level
4010 ),
4011 sortid="page/2911",
4012 )
4013 level = 2
4014 elif lc.startswith(tuple(ETYMOLOGY_TITLES)):
4015 if level > 3: 4015 ↛ 4016line 4015 didn't jump to line 4016 because the condition on line 4015 was never true
4016 wxr.wtp.debug(
4017 "etymology section {} at level {}".format(title, level),
4018 sortid="page/2917",
4019 )
4020 if prev_level == 3: # Two etymology (Glyph Origin + Etymology)
4021 # sections cheek-to-cheek
4022 skip_level_title = True
4023 # Modify the title of previous ("Glyph Origin") section, in
4024 # case we have a meaningful title like "Etymology 1"
4025 parts[-2] = "{}{}{}".format("=" * level, title, "=" * level)
4026 level = 3
4027 elif lc.startswith(PRONUNCIATION_TITLE):
4028 # Pronunciation is now a level between POS and Etymology, so
4029 # we need to shift everything down by one
4030 level = 4
4031 elif lc in POS_TITLES:
4032 level = 5
4033 elif lc == TRANSLATIONS_TITLE:
4034 level = 6
4035 elif lc in LINKAGE_TITLES or lc == COMPOUNDS_TITLE:
4036 level = 6
4037 elif lc in INFLECTION_TITLES:
4038 level = 6
4039 elif lc == DESCENDANTS_TITLE:
4040 level = 6
4041 elif title in PROTO_ROOT_DERIVED_TITLES: 4041 ↛ 4042line 4041 didn't jump to line 4042 because the condition on line 4041 was never true
4042 level = 6
4043 elif lc in IGNORED_TITLES:
4044 level = 6
4045 else:
4046 level = 6
4047 if skip_level_title:
4048 skip_level_title = False
4049 parts.append(part)
4050 else:
4051 parts.append("{}{}{}".format("=" * level, title, "=" * level))
4052 parts.append(part)
4053 # print("=" * level, title)
4054 # if level != len(left):
4055 # print(" FIXED LEVEL OF {} {} -> {}"
4056 # .format(title, len(left), level))
4058 text = "".join(parts)
4059 # print(text)
4060 return text
4063def parse_page(wxr: WiktextractContext, word: str, text: str) -> list[WordData]:
4064 # Skip translation pages
4065 if word.endswith("/" + TRANSLATIONS_TITLE): 4065 ↛ 4066line 4065 didn't jump to line 4066 because the condition on line 4065 was never true
4066 return []
4068 if wxr.config.verbose: 4068 ↛ 4069line 4068 didn't jump to line 4069 because the condition on line 4068 was never true
4069 logger.info(f"Parsing page: {word}")
4071 wxr.config.word = word
4072 wxr.wtp.start_page(word)
4074 # Remove <noinclude> and similar tags from main pages. They
4075 # should not appear there, but at least net/Elfdala has one and it
4076 # is probably not the only one.
4077 text = re.sub(r"(?si)<(/)?noinclude\s*>", "", text)
4078 text = re.sub(r"(?si)<(/)?onlyinclude\s*>", "", text)
4079 text = re.sub(r"(?si)<(/)?includeonly\s*>", "", text)
4081 # Fix up the subtitle hierarchy. There are hundreds if not thousands of
4082 # pages that have, for example, Translations section under Linkage, or
4083 # Translations section on the same level as Noun. Enforce a proper
4084 # hierarchy by manipulating the subtitle levels in certain cases.
4085 text = fix_subtitle_hierarchy(wxr, text)
4087 # Parse the page, pre-expanding those templates that are likely to
4088 # influence parsing
4089 tree = wxr.wtp.parse(
4090 text,
4091 pre_expand=True,
4092 additional_expand=ADDITIONAL_EXPAND_TEMPLATES,
4093 do_not_pre_expand=DO_NOT_PRE_EXPAND_TEMPLATES,
4094 )
4095 # from wikitextprocessor.parser import print_tree
4096 # print("PAGE PARSE:", print_tree(tree))
4098 top_data: WordData = {}
4100 # Iterate over top-level titles, which should be languages for normal
4101 # pages
4102 by_lang = defaultdict(list)
4103 for langnode in tree.children:
4104 if not isinstance(langnode, WikiNode):
4105 continue
4106 if langnode.kind == NodeKind.TEMPLATE:
4107 parse_top_template(wxr, langnode, top_data)
4108 continue
4109 if langnode.kind == NodeKind.LINK:
4110 # Some pages have links at top level, e.g., "trees" in Wiktionary
4111 continue
4112 if langnode.kind != NodeKind.LEVEL2: 4112 ↛ 4113line 4112 didn't jump to line 4113 because the condition on line 4112 was never true
4113 wxr.wtp.debug(
4114 f"unexpected top-level node: {langnode}", sortid="page/3014"
4115 )
4116 continue
4117 lang = clean_node(
4118 wxr, None, langnode.sarg if langnode.sarg else langnode.largs
4119 )
4120 lang_code = name_to_code(lang, "en")
4121 if lang_code == "": 4121 ↛ 4122line 4121 didn't jump to line 4122 because the condition on line 4121 was never true
4122 wxr.wtp.debug(
4123 f"unrecognized language name: {lang}", sortid="page/3019"
4124 )
4125 if (
4126 wxr.config.capture_language_codes
4127 and lang_code not in wxr.config.capture_language_codes
4128 ):
4129 continue
4130 wxr.wtp.start_section(lang)
4132 # Collect all words from the page.
4133 # print(f"{langnode=}")
4134 datas = parse_language(wxr, langnode, lang, lang_code)
4136 # Propagate fields resulting from top-level templates to this
4137 # part-of-speech.
4138 for data in datas:
4139 if "lang" not in data: 4139 ↛ 4140line 4139 didn't jump to line 4140 because the condition on line 4139 was never true
4140 wxr.wtp.debug(
4141 "internal error -- no lang in data: {}".format(data),
4142 sortid="page/3034",
4143 )
4144 continue
4145 for k, v in top_data.items():
4146 assert isinstance(v, (list, tuple))
4147 data_extend(data, k, v)
4148 by_lang[data["lang"]].append(data)
4150 # XXX this code is clearly out of date. There is no longer a "conjugation"
4151 # field. FIX OR REMOVE.
4152 # Do some post-processing on the words. For example, we may distribute
4153 # conjugation information to all the words.
4154 ret = []
4155 for lang, lang_datas in by_lang.items():
4156 ret.extend(lang_datas)
4158 for x in ret:
4159 if x["word"] != word:
4160 if word.startswith("Unsupported titles/"):
4161 wxr.wtp.debug(
4162 f"UNSUPPORTED TITLE: '{word}' -> '{x['word']}'",
4163 sortid="20231101/3578page.py",
4164 )
4165 else:
4166 wxr.wtp.debug(
4167 f"DIFFERENT ORIGINAL TITLE: '{word}' -> '{x['word']}'",
4168 sortid="20231101/3582page.py",
4169 )
4170 x["original_title"] = word
4171 # validate tag data
4172 recursively_separate_raw_tags(wxr, x) # type:ignore[arg-type]
4173 return ret
4176def recursively_separate_raw_tags(
4177 wxr: WiktextractContext, data: dict[str, Any]
4178) -> None:
4179 if not isinstance(data, dict): 4179 ↛ 4180line 4179 didn't jump to line 4180 because the condition on line 4179 was never true
4180 wxr.wtp.error(
4181 "'data' is not dict; most probably "
4182 "data has a list that contains at least one dict and "
4183 "at least one non-dict item",
4184 sortid="en/page-4016/20240419",
4185 )
4186 return
4187 new_tags: list[str] = []
4188 raw_tags: list[str] = data.get("raw_tags", [])
4189 for field, val in data.items():
4190 if field == "tags":
4191 for tag in val:
4192 if tag not in valid_tags:
4193 raw_tags.append(tag)
4194 else:
4195 new_tags.append(tag)
4196 if isinstance(val, list):
4197 if len(val) > 0 and isinstance(val[0], dict):
4198 for d in val:
4199 recursively_separate_raw_tags(wxr, d)
4200 if "tags" in data and not new_tags:
4201 del data["tags"]
4202 elif new_tags:
4203 data["tags"] = new_tags
4204 if raw_tags:
4205 data["raw_tags"] = raw_tags
4208def process_soft_redirect_template(
4209 wxr: WiktextractContext,
4210 template_node: TemplateNode,
4211 redirect_pages: list[str],
4212) -> bool:
4213 # return `True` if the template is soft redirect template
4214 if template_node.template_name == "zh-see":
4215 # https://en.wiktionary.org/wiki/Template:zh-see
4216 title = clean_node(
4217 wxr, None, template_node.template_parameters.get(1, "")
4218 )
4219 if title != "": 4219 ↛ 4221line 4219 didn't jump to line 4221 because the condition on line 4219 was always true
4220 redirect_pages.append(title)
4221 return True
4222 elif template_node.template_name in ["ja-see", "ja-see-kango"]:
4223 # https://en.wiktionary.org/wiki/Template:ja-see
4224 for key, value in template_node.template_parameters.items():
4225 if isinstance(key, int): 4225 ↛ 4224line 4225 didn't jump to line 4224 because the condition on line 4225 was always true
4226 title = clean_node(wxr, None, value)
4227 if title != "": 4227 ↛ 4224line 4227 didn't jump to line 4224 because the condition on line 4227 was always true
4228 redirect_pages.append(title)
4229 return True
4230 return False
4233ZH_FORMS_TAGS = {
4234 "trad.": "Traditional-Chinese",
4235 "simp.": "Simplified-Chinese",
4236 "alternative forms": "alternative",
4237 "2nd round simp.": "Second-Round-Simplified-Chinese",
4238}
4241def extract_zh_forms_template(
4242 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordData
4243):
4244 # https://en.wiktionary.org/wiki/Template:zh-forms
4245 lit_meaning = clean_node(
4246 wxr, None, t_node.template_parameters.get("lit", "")
4247 )
4248 if lit_meaning != "":
4249 base_data["literal_meaning"] = lit_meaning
4250 expanded_node = wxr.wtp.parse(
4251 wxr.wtp.node_to_wikitext(t_node), expand_all=True
4252 )
4253 for table in expanded_node.find_child(NodeKind.TABLE):
4254 for row in table.find_child(NodeKind.TABLE_ROW):
4255 row_header = ""
4256 row_header_tags: list[str] = []
4257 header_has_span = False
4258 for cell in row.find_child(
4259 NodeKind.TABLE_HEADER_CELL | NodeKind.TABLE_CELL
4260 ):
4261 if cell.kind == NodeKind.TABLE_HEADER_CELL:
4262 row_header, row_header_tags, header_has_span = (
4263 extract_zh_forms_header_cell(wxr, base_data, cell)
4264 )
4265 elif not header_has_span:
4266 extract_zh_forms_data_cell(
4267 wxr, base_data, cell, row_header, row_header_tags
4268 )
4270 if "forms" in base_data and len(base_data["forms"]) == 0: 4270 ↛ 4271line 4270 didn't jump to line 4271 because the condition on line 4270 was never true
4271 del base_data["forms"]
4274def extract_zh_forms_header_cell(
4275 wxr: WiktextractContext, base_data: WordData, header_cell: WikiNode
4276) -> tuple[str, list[str], bool]:
4277 row_header = ""
4278 row_header_tags = []
4279 header_has_span = False
4280 first_span_index = len(header_cell.children)
4281 for index, span_tag in header_cell.find_html("span", with_index=True):
4282 if index < first_span_index: 4282 ↛ 4284line 4282 didn't jump to line 4284 because the condition on line 4282 was always true
4283 first_span_index = index
4284 header_has_span = True
4285 row_header = clean_node(wxr, None, header_cell.children[:first_span_index])
4286 for raw_tag in row_header.split(" and "):
4287 raw_tag = raw_tag.strip()
4288 if raw_tag != "":
4289 row_header_tags.append(raw_tag)
4290 for span_tag in header_cell.find_html_recursively("span"):
4291 span_lang = span_tag.attrs.get("lang", "")
4292 form_nodes = []
4293 sup_title = ""
4294 for node in span_tag.children:
4295 if isinstance(node, HTMLNode) and node.tag == "sup": 4295 ↛ 4296line 4295 didn't jump to line 4296 because the condition on line 4295 was never true
4296 for sup_span in node.find_html("span"):
4297 sup_title = sup_span.attrs.get("title", "")
4298 else:
4299 form_nodes.append(node)
4300 if span_lang in ["zh-Hant", "zh-Hans"]:
4301 for word in clean_node(wxr, None, form_nodes).split("/"):
4302 if word not in [wxr.wtp.title, ""]:
4303 form = {"form": word}
4304 for raw_tag in row_header_tags:
4305 if raw_tag in ZH_FORMS_TAGS: 4305 ↛ 4308line 4305 didn't jump to line 4308 because the condition on line 4305 was always true
4306 data_append(form, "tags", ZH_FORMS_TAGS[raw_tag])
4307 else:
4308 data_append(form, "raw_tags", raw_tag)
4309 if sup_title != "": 4309 ↛ 4310line 4309 didn't jump to line 4310 because the condition on line 4309 was never true
4310 data_append(form, "raw_tags", sup_title)
4311 data_append(base_data, "forms", form)
4312 return row_header, row_header_tags, header_has_span
4315TagLiteral = Literal["tags", "raw_tags"]
4316TAG_LITERALS_TUPLE: tuple[TagLiteral, ...] = ("tags", "raw_tags")
4319def extract_zh_forms_data_cell(
4320 wxr: WiktextractContext,
4321 base_data: WordData,
4322 cell: WikiNode,
4323 row_header: str,
4324 row_header_tags: list[str],
4325) -> None:
4326 from .zh_pron_tags import ZH_PRON_TAGS
4328 forms: list[FormData] = []
4329 for top_span_tag in cell.find_html("span"):
4330 span_style = top_span_tag.attrs.get("style", "")
4331 span_lang = top_span_tag.attrs.get("lang", "")
4332 if span_style == "white-space:nowrap;":
4333 extract_zh_forms_data_cell(
4334 wxr, base_data, top_span_tag, row_header, row_header_tags
4335 )
4336 elif "font-size:80%" in span_style:
4337 raw_tag = clean_node(wxr, None, top_span_tag)
4338 if raw_tag != "": 4338 ↛ 4329line 4338 didn't jump to line 4329 because the condition on line 4338 was always true
4339 for form in forms:
4340 if raw_tag in ZH_PRON_TAGS: 4340 ↛ 4346line 4340 didn't jump to line 4346 because the condition on line 4340 was always true
4341 tr_tag = ZH_PRON_TAGS[raw_tag]
4342 if isinstance(tr_tag, list): 4342 ↛ 4343line 4342 didn't jump to line 4343 because the condition on line 4342 was never true
4343 data_extend(form, "tags", tr_tag)
4344 elif isinstance(tr_tag, str): 4344 ↛ 4339line 4344 didn't jump to line 4339 because the condition on line 4344 was always true
4345 data_append(form, "tags", tr_tag)
4346 elif raw_tag in valid_tags:
4347 data_append(form, "tags", raw_tag)
4348 else:
4349 data_append(form, "raw_tags", raw_tag)
4350 elif span_lang in ["zh-Hant", "zh-Hans", "zh"]: 4350 ↛ 4329line 4350 didn't jump to line 4329 because the condition on line 4350 was always true
4351 word = clean_node(wxr, None, top_span_tag)
4352 if word not in ["", "/", wxr.wtp.title]:
4353 form = {"form": word}
4354 if row_header != "anagram": 4354 ↛ 4360line 4354 didn't jump to line 4360 because the condition on line 4354 was always true
4355 for raw_tag in row_header_tags:
4356 if raw_tag in ZH_FORMS_TAGS: 4356 ↛ 4359line 4356 didn't jump to line 4359 because the condition on line 4356 was always true
4357 data_append(form, "tags", ZH_FORMS_TAGS[raw_tag])
4358 else:
4359 data_append(form, "raw_tags", raw_tag)
4360 if span_lang == "zh-Hant":
4361 data_append(form, "tags", "Traditional-Chinese")
4362 elif span_lang == "zh-Hans":
4363 data_append(form, "tags", "Simplified-Chinese")
4364 forms.append(form)
4366 if row_header == "anagram": 4366 ↛ 4367line 4366 didn't jump to line 4367 because the condition on line 4366 was never true
4367 for form in forms:
4368 l_data: LinkageData = {"word": form["form"]}
4369 for key in TAG_LITERALS_TUPLE:
4370 if key in form:
4371 l_data[key] = form[key]
4372 data_append(base_data, "anagrams", l_data)
4373 else:
4374 data_extend(base_data, "forms", forms)
4377def extract_ja_kanjitab_template(
4378 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordData
4379):
4380 # https://en.wiktionary.org/wiki/Template:ja-kanjitab
4381 expanded_node = wxr.wtp.parse(
4382 wxr.wtp.node_to_wikitext(t_node), expand_all=True
4383 )
4384 for table in expanded_node.find_child(NodeKind.TABLE):
4385 is_alt_form_table = False
4386 for row in table.find_child(NodeKind.TABLE_ROW):
4387 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL):
4388 header_text = clean_node(wxr, None, header_node)
4389 if header_text.startswith("Alternative spelling"):
4390 is_alt_form_table = True
4391 if not is_alt_form_table:
4392 continue
4393 forms = []
4394 for row in table.find_child(NodeKind.TABLE_ROW):
4395 for cell_node in row.find_child(NodeKind.TABLE_CELL):
4396 for child_node in cell_node.children:
4397 if isinstance(child_node, HTMLNode):
4398 if child_node.tag == "span":
4399 word = clean_node(wxr, None, child_node)
4400 if word != "": 4400 ↛ 4396line 4400 didn't jump to line 4396 because the condition on line 4400 was always true
4401 forms.append(
4402 {
4403 "form": word,
4404 "tags": ["alternative", "kanji"],
4405 }
4406 )
4407 elif child_node.tag == "small":
4408 raw_tag = clean_node(wxr, None, child_node).strip(
4409 "()"
4410 )
4411 if raw_tag != "" and len(forms) > 0: 4411 ↛ 4396line 4411 didn't jump to line 4396 because the condition on line 4411 was always true
4412 data_append(
4413 forms[-1],
4414 "tags"
4415 if raw_tag in valid_tags
4416 else "raw_tags",
4417 raw_tag,
4418 )
4419 data_extend(base_data, "forms", forms)
4420 for link_node in expanded_node.find_child(NodeKind.LINK):
4421 clean_node(wxr, base_data, link_node)