Coverage for src/wiktextract/extractor/en/page.py: 79%

1845 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1# Code for parsing information from a single Wiktionary page. 

2# 

3# Copyright (c) 2018-2022 Tatu Ylonen. See file LICENSE and https://ylonen.org 

4 

5import copy 

6import html 

7import re 

8from collections import defaultdict 

9from functools import partial 

10from typing import ( 

11 TYPE_CHECKING, 

12 Any, 

13 Iterable, 

14 Literal, 

15 Optional, 

16 Set, 

17 Union, 

18 cast, 

19) 

20 

21from mediawiki_langcodes import get_all_names, name_to_code 

22from wikitextprocessor.core import TemplateArgs, TemplateFnCallable 

23from wikitextprocessor.parser import ( 

24 LEVEL_KIND_FLAGS, 

25 GeneralNode, 

26 HTMLNode, 

27 LevelNode, 

28 NodeKind, 

29 TemplateNode, 

30 WikiNode, 

31) 

32 

33from ...clean import clean_template_args, clean_value 

34from ...datautils import ( 

35 data_append, 

36 data_extend, 

37 ns_title_prefix_tuple, 

38) 

39from ...page import ( 

40 LEVEL_KINDS, 

41 clean_node, 

42 is_panel_template, 

43 recursively_extract, 

44) 

45from ...tags import valid_tags 

46from ...wxr_context import WiktextractContext 

47from ...wxr_logging import logger 

48from ..ruby import extract_ruby, parse_ruby 

49from ..share import strip_nodes 

50from .descendant import ( 

51 ETYMOLOGY_TEMPLATES_IN_HEADS, 

52 etymology_template_append, 

53 extract_descendant_section, 

54) 

55from .example import extract_example_list_item, extract_template_zh_x 

56from .form_descriptions import ( 

57 classify_desc, 

58 decode_tags, 

59 distw, 

60 parse_alt_or_inflection_of, 

61 parse_sense_qualifier, 

62 parse_word_head, 

63) 

64from .inflection import TableContext, parse_inflection_section 

65from .info_templates import ( 

66 INFO_TEMPLATE_FUNCS, 

67 parse_info_template_arguments, 

68 parse_info_template_node, 

69) 

70from .linkages import ( 

71 extract_alt_form_section, 

72 parse_linkage, 

73) 

74from .parts_of_speech import PARTS_OF_SPEECH 

75from .section_titles import ( 

76 COMPOUNDS_TITLE, 

77 DESCENDANTS_TITLE, 

78 ETYMOLOGY_TITLES, 

79 IGNORED_TITLES, 

80 INFLECTION_TITLES, 

81 LINKAGE_TITLES, 

82 POS_TITLES, 

83 PRONUNCIATION_TITLE, 

84 PROTO_ROOT_DERIVED_TITLES, 

85 TRANSLATIONS_TITLE, 

86) 

87from .translations import parse_translation_item_text 

88from .type_utils import ( 

89 AttestationData, 

90 ExampleData, 

91 FormData, 

92 LinkageData, 

93 ReferenceData, 

94 SenseData, 

95 SoundData, 

96 TemplateData, 

97 WordData, 

98) 

99from .unsupported_titles import unsupported_title_map 

100 

101# When determining whether a string is 'english', classify_desc 

102# might return 'taxonomic' which is English text 99% of the time. 

103ENGLISH_TEXTS = ("english", "taxonomic") 

104 

105# Matches head tag 

106HEAD_TAG_RE = re.compile( 

107 r"^(head|Han char|arabic-noun|arabic-noun-form|" 

108 r"hangul-symbol|syllable-hangul)$|" 

109 + r"^(latin|" 

110 + "|".join(lang_code for lang_code, *_ in get_all_names("en")) 

111 + r")-(" 

112 + "|".join( 

113 [ 

114 "abbr", 

115 "adj", 

116 "adjective", 

117 "adjective form", 

118 "adjective-form", 

119 "adv", 

120 "adverb", 

121 "affix", 

122 "animal command", 

123 "art", 

124 "article", 

125 "aux", 

126 "bound pronoun", 

127 "bound-pronoun", 

128 "Buyla", 

129 "card num", 

130 "card-num", 

131 "cardinal", 

132 "chunom", 

133 "classifier", 

134 "clitic", 

135 "cls", 

136 "cmene", 

137 "cmavo", 

138 "colloq-verb", 

139 "colverbform", 

140 "combining form", 

141 "combining-form", 

142 "comparative", 

143 "con", 

144 "concord", 

145 "conj", 

146 "conjunction", 

147 "conjug", 

148 "cont", 

149 "contr", 

150 "converb", 

151 "daybox", 

152 "decl", 

153 "decl noun", 

154 "def", 

155 "dem", 

156 "det", 

157 "determ", 

158 "Deva", 

159 "ending", 

160 "entry", 

161 "form", 

162 "fuhivla", 

163 "gerund", 

164 "gismu", 

165 "hanja", 

166 "hantu", 

167 "hanzi", 

168 "head", 

169 "ideophone", 

170 "idiom", 

171 "inf", 

172 "indef", 

173 "infixed pronoun", 

174 "infixed-pronoun", 

175 "infl", 

176 "inflection", 

177 "initialism", 

178 "int", 

179 "interfix", 

180 "interj", 

181 "interjection", 

182 "jyut", 

183 "latin", 

184 "letter", 

185 "locative", 

186 "lujvo", 

187 "monthbox", 

188 "mutverb", 

189 "name", 

190 "nisba", 

191 "nom", 

192 "noun", 

193 "noun form", 

194 "noun-form", 

195 "noun plural", 

196 "noun-plural", 

197 "nounprefix", 

198 "num", 

199 "number", 

200 "numeral", 

201 "ord", 

202 "ordinal", 

203 "par", 

204 "part", 

205 "part form", 

206 "part-form", 

207 "participle", 

208 "particle", 

209 "past", 

210 "past neg", 

211 "past-neg", 

212 "past participle", 

213 "past-participle", 

214 "perfect participle", 

215 "perfect-participle", 

216 "personal pronoun", 

217 "personal-pronoun", 

218 "pref", 

219 "prefix", 

220 "phrase", 

221 "pinyin", 

222 "plural noun", 

223 "plural-noun", 

224 "pos", 

225 "poss-noun", 

226 "post", 

227 "postp", 

228 "postposition", 

229 "PP", 

230 "pp", 

231 "ppron", 

232 "pred", 

233 "predicative", 

234 "prep", 

235 "prep phrase", 

236 "prep-phrase", 

237 "preposition", 

238 "present participle", 

239 "present-participle", 

240 "pron", 

241 "prondem", 

242 "pronindef", 

243 "pronoun", 

244 "prop", 

245 "proper noun", 

246 "proper-noun", 

247 "proper noun form", 

248 "proper-noun form", 

249 "proper noun-form", 

250 "proper-noun-form", 

251 "prov", 

252 "proverb", 

253 "prpn", 

254 "prpr", 

255 "punctuation mark", 

256 "punctuation-mark", 

257 "regnoun", 

258 "rel", 

259 "rom", 

260 "romanji", 

261 "root", 

262 "sign", 

263 "suff", 

264 "suffix", 

265 "syllable", 

266 "symbol", 

267 "verb", 

268 "verb form", 

269 "verb-form", 

270 "verbal noun", 

271 "verbal-noun", 

272 "verbnec", 

273 "vform", 

274 ] 

275 ) 

276 + r")(-|/|\+|$)" 

277) 

278 

279# Head-templates causing problems (like newlines) that can be squashed into 

280# an empty string in the template handler while saving their template 

281# data for later. 

282WORD_LEVEL_HEAD_TEMPLATES = {"term-label", "tlb"} 

283 

284 

285PROBLEMATIC_TEMPLATES_CLUMP = ( 

286 WORD_LEVEL_HEAD_TEMPLATES | ETYMOLOGY_TEMPLATES_IN_HEADS 

287) 

288 

289FLOATING_TABLE_TEMPLATES: set[str] = { 

290 # az-suffix-form creates a style=floatright div that is otherwise 

291 # deleted; if it is not pre-expanded, we can intercept the template 

292 # so we add this set into do_not_pre_expand, and intercept the 

293 # templates in parse_part_of_speech 

294 "az-suffix-forms", 

295 "az-inf-p", 

296 "kk-suffix-forms", 

297 "ky-suffix-forms", 

298 "tr-inf-p", 

299 "tr-suffix-forms", 

300 "tt-suffix-forms", 

301 "uz-suffix-forms", 

302} 

303# These two should contain template names that should always be 

304# pre-expanded when *first* processing the tree, or not pre-expanded 

305# so that the template are left in place with their identifying 

306# name intact for later filtering. 

307 

308DO_NOT_PRE_EXPAND_TEMPLATES: set[str] = set() 

309DO_NOT_PRE_EXPAND_TEMPLATES.update(FLOATING_TABLE_TEMPLATES) 

310 

311# Additional templates to be expanded in the pre-expand phase 

312ADDITIONAL_EXPAND_TEMPLATES: set[str] = { 

313 "multitrans", 

314 "multitrans-nowiki", 

315 "trans-top", 

316 "trans-top-also", 

317 "trans-bottom", 

318 "checktrans-top", 

319 "checktrans-bottom", 

320 "col", 

321 "col1", 

322 "col2", 

323 "col3", 

324 "col4", 

325 "col5", 

326 "col1-u", 

327 "col2-u", 

328 "col3-u", 

329 "col4-u", 

330 "col5-u", 

331 "check deprecated lang param usage", 

332 "deprecated code", 

333 "ru-verb-alt-ё", 

334 "ru-noun-alt-ё", 

335 "ru-adj-alt-ё", 

336 "ru-proper noun-alt-ё", 

337 "ru-pos-alt-ё", 

338 "ru-alt-ё", 

339 "inflection of", 

340 "no deprecated lang param usage", 

341 "transclude", # these produce sense entries (or other lists) 

342 "tcl", 

343} 

344 

345# Inverse linkage for those that have them 

346linkage_inverses: dict[str, str] = { 

347 # XXX this is not currently used, move to post-processing 

348 "synonyms": "synonyms", 

349 "hypernyms": "hyponyms", 

350 "hyponyms": "hypernyms", 

351 "holonyms": "meronyms", 

352 "meronyms": "holonyms", 

353 "derived": "derived_from", 

354 "coordinate_terms": "coordinate_terms", 

355 "troponyms": "hypernyms", 

356 "antonyms": "antonyms", 

357 "instances": "instance_of", 

358 "related": "related", 

359} 

360 

361# Templates that are used to form panels on pages and that 

362# should be ignored in various positions 

363PANEL_TEMPLATES: set[str] = { 

364 "Character info", 

365 "CJKV", 

366 "French personal pronouns", 

367 "French possessive adjectives", 

368 "French possessive pronouns", 

369 "Han etym", 

370 "Han etyl", # this redirects to Han etym and would cause Lua errors, 

371 # and I don't know why, but I'm putting it here because 

372 # we should be ignoring it anyhow. 

373 "Japanese demonstratives", 

374 "Latn-script", 

375 "LDL", 

376 "MW1913Abbr", 

377 "Number-encoding", 

378 "Nuttall", 

379 "Spanish possessive adjectives", 

380 "Spanish possessive pronouns", 

381 "USRegionDisputed", 

382 "Webster 1913", 

383 "ase-rfr", 

384 "attention", 

385 "attn", 

386 "beer", 

387 "broken ref", 

388 "ca-compass", 

389 "character info", 

390 "character info/var", 

391 "checksense", 

392 "compass-fi", 

393 "copyvio suspected", 

394 "delete", 

395 "dial syn", # Currently ignore these, but could be useful in Chinese/Korean 

396 "etystub", 

397 "examples", 

398 "hu-corr", 

399 "hu-suff-pron", 

400 "interwiktionary", 

401 "ja-kanjitab", 

402 "ja-kt", 

403 "ko-hanja-search", 

404 "look", 

405 "maintenance box", 

406 "maintenance line", 

407 "mediagenic terms", 

408 "merge", 

409 "missing template", 

410 "morse links", 

411 "move", 

412 "multiple images", 

413 "no inline", 

414 "picdic", 

415 "picdicimg", 

416 "picdiclabel", 

417 "polyominoes", 

418 "predidential nomics", 

419 "punctuation", # This actually gets pre-expanded 

420 "reconstructed", 

421 "request box", 

422 "rf-sound example", 

423 "rfaccents", 

424 "rfap", 

425 "rfaspect", 

426 "rfc", 

427 "rfc-auto", 

428 "rfc-header", 

429 "rfc-level", 

430 "rfc-pron-n", 

431 "rfc-sense", 

432 "rfclarify", 

433 "rfd", 

434 "rfd-redundant", 

435 "rfd-sense", 

436 "rfdate", 

437 "rfdatek", 

438 "rfdef", 

439 "rfe", 

440 "rfe/dowork", 

441 "rfex", 

442 "rfexp", 

443 "rfform", 

444 "rfgender", 

445 "rfi", 

446 "rfinfl", 

447 "rfm", 

448 "rfm-sense", 

449 "rfp", 

450 "rfp-old", 

451 "rfquote", 

452 "rfquote-sense", 

453 "rfquotek", 

454 "rfref", 

455 "rfscript", 

456 "rft2", 

457 "rftaxon", 

458 "rftone", 

459 "rftranslit", 

460 "rfv", 

461 "rfv-etym", 

462 "rfv-pron", 

463 "rfv-quote", 

464 "rfv-sense", 

465 "selfref", 

466 "split", 

467 "stroke order", # XXX consider capturing this? 

468 "stub entry", 

469 "t-needed", 

470 "tbot entry", 

471 "tea room", 

472 "tea room sense", 

473 # "ttbc", - XXX needed in at least on/Preposition/Translation page 

474 "unblock", 

475 "unsupportedpage", 

476 "video frames", 

477 "was wotd", 

478 "wrongtitle", 

479 "zh-forms", 

480 "zh-hanzi-box", 

481 "no entry", 

482} 

483 

484# Template name prefixes used for language-specific panel templates (i.e., 

485# templates that create side boxes or notice boxes or that should generally 

486# be ignored). 

487PANEL_PREFIXES: set[str] = { 

488 "list:compass points/", 

489 "list:Gregorian calendar months/", 

490 "RQ:", 

491} 

492 

493# Templates used for wikipedia links. 

494wikipedia_templates: set[str] = { 

495 "wikipedia", 

496 "slim-wikipedia", 

497 "w", 

498 "W", 

499 "swp", 

500 "wiki", 

501 "Wikipedia", 

502 "wtorw", 

503} 

504for x in PANEL_PREFIXES & wikipedia_templates: 504 ↛ 505line 504 didn't jump to line 505 because the loop on line 504 never started

505 print( 

506 "WARNING: {!r} in both panel_templates and wikipedia_templates".format( 

507 x 

508 ) 

509 ) 

510 

511# Mapping from a template name (without language prefix) for the main word 

512# (e.g., fi-noun, fi-adj, en-verb) to permitted parts-of-speech in which 

513# it could validly occur. This is used as just a sanity check to give 

514# warnings about probably incorrect coding in Wiktionary. 

515template_allowed_pos_map: dict[str, list[str]] = { 

516 "abbr": ["abbrev"], 

517 "noun": ["noun", "abbrev", "pron", "name", "num", "adj_noun"], 

518 "plural noun": ["noun", "name"], 

519 "plural-noun": ["noun", "name"], 

520 "proper noun": ["noun", "name"], 

521 "proper-noun": ["name", "noun"], 

522 "prop": ["name", "noun"], 

523 "verb": ["verb", "phrase"], 

524 "gerund": ["verb"], 

525 "particle": ["adv", "particle"], 

526 "adj": ["adj", "adj_noun"], 

527 "pron": ["pron", "noun"], 

528 "name": ["name", "noun"], 

529 "adv": ["adv", "intj", "conj", "particle"], 

530 "phrase": ["phrase", "prep_phrase"], 

531 "noun phrase": ["phrase"], 

532 "ordinal": ["num"], 

533 "number": ["num"], 

534 "pos": ["affix", "name", "num"], 

535 "suffix": ["suffix", "affix"], 

536 "character": ["character"], 

537 "letter": ["character"], 

538 "kanji": ["character"], 

539 "cont": ["abbrev"], 

540 "interj": ["intj"], 

541 "con": ["conj"], 

542 "part": ["particle"], 

543 "prep": ["prep", "postp"], 

544 "postp": ["postp"], 

545 "misspelling": ["noun", "adj", "verb", "adv"], 

546 "part-form": ["verb"], 

547} 

548for k, v in template_allowed_pos_map.items(): 

549 for x in v: 

550 if x not in PARTS_OF_SPEECH: 550 ↛ 551line 550 didn't jump to line 551 because the condition on line 550 was never true

551 print( 

552 "BAD PART OF SPEECH {!r} IN template_allowed_pos_map: {}={}" 

553 "".format(x, k, v) 

554 ) 

555 assert False 

556 

557 

558# Templates ignored during etymology extraction, i.e., these will not be listed 

559# in the extracted etymology templates. 

560ignored_etymology_templates: list[str] = [ 

561 "...", 

562 "IPAchar", 

563 "ipachar", 

564 "ISBN", 

565 "isValidPageName", 

566 "redlink category", 

567 "deprecated code", 

568 "check deprecated lang param usage", 

569 "para", 

570 "p", 

571 "cite", 

572 "Cite news", 

573 "Cite newsgroup", 

574 "cite paper", 

575 "cite MLLM 1976", 

576 "cite journal", 

577 "cite news/documentation", 

578 "cite paper/documentation", 

579 "cite video game", 

580 "cite video game/documentation", 

581 "cite newsgroup", 

582 "cite newsgroup/documentation", 

583 "cite web/documentation", 

584 "cite news", 

585 "Cite book", 

586 "Cite-book", 

587 "cite book", 

588 "cite web", 

589 "cite-usenet", 

590 "cite-video/documentation", 

591 "Cite-journal", 

592 "rfe", 

593 "catlangname", 

594 "cln", 

595 "langname-lite", 

596 "no deprecated lang param usage", 

597 "mention", 

598 "m", 

599 "m-self", 

600 "link", 

601 "l", 

602 "ll", 

603 "l-self", 

604] 

605# Regexp for matching ignored etymology template names. This adds certain 

606# prefixes to the names listed above. 

607ignored_etymology_templates_re = re.compile( 

608 r"^((cite-|R:|RQ:).*|" 

609 + r"|".join(re.escape(x) for x in ignored_etymology_templates) 

610 + r")$" 

611) 

612 

613# Regexp for matching ignored descendants template names. Right now we just 

614# copy the ignored etymology templates 

615ignored_descendants_templates_re = ignored_etymology_templates_re 

616 

617# Set of template names that are used to define usage examples. If the usage 

618# example contains one of these templates, then it its type is set to 

619# "example" 

620usex_templates: set[str] = { 

621 "afex", 

622 "affixusex", 

623 "co", # {{collocation}} acts like a example template, specifically for 

624 # pairs of combinations of words that are more common than you'd 

625 # except would be randomly; hlavní#Czech 

626 "coi", 

627 "collocation", 

628 "el-example", 

629 "el-x", 

630 "example", 

631 "examples", 

632 "he-usex", 

633 "he-x", 

634 "hi-usex", 

635 "hi-x", 

636 "ja-usex-inline", 

637 "ja-usex", 

638 "ja-x", 

639 "jbo-example", 

640 "jbo-x", 

641 "km-usex", 

642 "km-x", 

643 "ko-usex", 

644 "ko-x", 

645 "lo-usex", 

646 "lo-x", 

647 "ne-x", 

648 "ne-usex", 

649 "prefixusex", 

650 "ryu-usex", 

651 "ryu-x", 

652 "shn-usex", 

653 "shn-x", 

654 "suffixusex", 

655 "th-usex", 

656 "th-x", 

657 "ur-usex", 

658 "ur-x", 

659 "usex", 

660 "usex-suffix", 

661 "ux", 

662 "uxi", 

663} 

664 

665stop_head_at_these_templates: set[str] = { 

666 "category", 

667 "cat", 

668 "topics", 

669 "catlangname", 

670 "c", 

671 "C", 

672 "top", 

673 "cln", 

674} 

675 

676# Set of template names that are used to define quotation examples. If the 

677# usage example contains one of these templates, then its type is set to 

678# "quotation". 

679quotation_templates: set[str] = { 

680 "collapse-quote", 

681 "quote-av", 

682 "quote-book", 

683 "quote-GYLD", 

684 "quote-hansard", 

685 "quotei", 

686 "quote-journal", 

687 "quotelite", 

688 "quote-mailing list", 

689 "quote-meta", 

690 "quote-newsgroup", 

691 "quote-song", 

692 "quote-text", 

693 "quote", 

694 "quote-us-patent", 

695 "quote-video game", 

696 "quote-web", 

697 "quote-wikipedia", 

698 "wikiquote", 

699 "Wikiquote", 

700 "Q", 

701} 

702 

703taxonomy_templates = { 

704 # argument 1 should be the taxonomic name, frex. "Lupus lupus" 

705 "taxfmt", 

706 "taxlink", 

707 "taxlink2", 

708 "taxlinknew", 

709 "taxlook", 

710} 

711 

712# Template names, this was exctracted from template_linkage_mappings, 

713# because the code using template_linkage_mappings was actually not used 

714# (but not removed). 

715template_linkages_to_ignore_in_examples: set[str] = { 

716 "syn", 

717 "synonyms", 

718 "ant", 

719 "antonyms", 

720 "hyp", 

721 "hyponyms", 

722 "der", 

723 "derived terms", 

724 "coordinate terms", 

725 "cot", 

726 "rel", 

727 "col", 

728 "inline alt forms", 

729 "alti", 

730 "comeronyms", 

731 "holonyms", 

732 "holo", 

733 "hypernyms", 

734 "hyper", 

735 "meronyms", 

736 "mero", 

737 "troponyms", 

738 "perfectives", 

739 "pf", 

740 "imperfectives", 

741 "impf", 

742 "syndiff", 

743 "synsee", 

744 # not linkage nor example templates 

745 "sense", 

746 "s", 

747 "color panel", 

748 "colour panel", 

749} 

750 

751# Maps template name used in a word sense to a linkage field that it adds. 

752sense_linkage_templates: dict[str, str] = { 

753 "syn": "synonyms", 

754 "synonyms": "synonyms", 

755 "synsee": "synonyms", 

756 "syndiff": "synonyms", 

757 "hyp": "hyponyms", 

758 "hyponyms": "hyponyms", 

759 "ant": "antonyms", 

760 "antonyms": "antonyms", 

761 "alti": "related", 

762 "inline alt forms": "related", 

763 "coordinate terms": "coordinate_terms", 

764 "cot": "coordinate_terms", 

765 "comeronyms": "related", 

766 "holonyms": "holonyms", 

767 "holo": "holonyms", 

768 "hypernyms": "hypernyms", 

769 "hyper": "hypernyms", 

770 "meronyms": "meronyms", 

771 "mero": "meronyms", 

772 "troponyms": "troponyms", 

773 "perfectives": "related", 

774 "pf": "related", 

775 "imperfectives": "related", 

776 "impf": "related", 

777 "parasynonyms": "synonyms", 

778 "par": "synonyms", 

779 "parasyn": "synonyms", 

780 "nearsyn": "synonyms", 

781 "near-syn": "synonyms", 

782} 

783 

784sense_linkage_templates_tags: dict[str, list[str]] = { 

785 "alti": ["alternative"], 

786 "inline alt forms": ["alternative"], 

787 "comeronyms": ["comeronym"], 

788 "perfectives": ["perfective"], 

789 "pf": ["perfective"], 

790 "imperfectives": ["imperfective"], 

791 "impf": ["imperfective"], 

792} 

793 

794 

795def decode_html_entities(v: Union[str, int]) -> str: 

796 """Decodes HTML entities from a value, converting them to the respective 

797 Unicode characters/strings.""" 

798 if isinstance(v, int): 

799 # I changed this to return str(v) instead of v = str(v), 

800 # but there might have been the intention to have more logic 

801 # here. html.unescape would not do anything special with an integer, 

802 # it needs html escape symbols (&xx;). 

803 return str(v) 

804 return html.unescape(v) 

805 

806 

807def parse_sense_linkage( 

808 wxr: WiktextractContext, 

809 data: SenseData, 

810 name: str, 

811 ht: TemplateArgs, 

812 pos: str, 

813) -> None: 

814 """Parses a linkage (synonym, etc) specified in a word sense.""" 

815 assert isinstance(wxr, WiktextractContext) 

816 assert isinstance(data, dict) 

817 assert isinstance(name, str) 

818 assert isinstance(ht, dict) 

819 field = sense_linkage_templates[name] 

820 field_tags = sense_linkage_templates_tags.get(name, []) 

821 for i in range(2, 20): 

822 if i not in ht: 

823 break 

824 w = clean_node(wxr, data, ht[i]) 

825 if "#" in w: 

826 w = w[: w.index("#")] 

827 if w in ["", "<"]: # `<` used in "hypernyms" template 

828 continue 

829 if ( 829 ↛ 834line 829 didn't jump to line 834 because the condition on line 829 was never true

830 i > 2 

831 and w in (",", "or", ";") 

832 or w.startswith(("see also", "See also")) 

833 ): 

834 continue 

835 is_thesaurus = False 

836 for alias in ns_title_prefix_tuple(wxr, "Thesaurus"): 

837 if w.startswith(alias): 

838 is_thesaurus = True 

839 w = w[len(alias) :] 

840 if w != wxr.wtp.title: 840 ↛ 860line 840 didn't jump to line 860 because the condition on line 840 was always true

841 from ...thesaurus import search_thesaurus 

842 

843 lang_code = clean_node(wxr, None, ht.get(1, "")) 

844 for t_data in search_thesaurus( 

845 wxr.thesaurus_db_conn, # type: ignore 

846 w, 

847 lang_code, 

848 pos, 

849 "synonyms", # GH issue #1570 

850 ): 

851 l_data: LinkageData = { 

852 "word": t_data.term, 

853 "source": "Thesaurus:" + w, 

854 } 

855 if len(t_data.tags) > 0: 855 ↛ 856line 855 didn't jump to line 856 because the condition on line 855 was never true

856 l_data["tags"] = t_data.tags 

857 if len(t_data.raw_tags) > 0: 857 ↛ 858line 857 didn't jump to line 858 because the condition on line 857 was never true

858 l_data["raw_tags"] = t_data.raw_tags 

859 data_append(data, field, l_data) 

860 break 

861 if is_thesaurus: 

862 continue 

863 tags: list[str] = [] 

864 topics: list[str] = [] 

865 english: Optional[str] = None 

866 # Try to find qualifiers for this synonym 

867 q = ht.get("q{}".format(i - 1)) 

868 if q: 

869 cls = classify_desc(q) 

870 if cls == "tags": 

871 tagsets1, topics1 = decode_tags(q) 

872 for ts in tagsets1: 

873 tags.extend(ts) 

874 topics.extend(topics1) 

875 elif cls == "english": 875 ↛ 881line 875 didn't jump to line 881 because the condition on line 875 was always true

876 if english: 876 ↛ 877line 876 didn't jump to line 877 because the condition on line 876 was never true

877 english += "; " + q 

878 else: 

879 english = q 

880 # Try to find English translation for this synonym 

881 t = ht.get("t{}".format(i - 1)) 

882 if t: 882 ↛ 883line 882 didn't jump to line 883 because the condition on line 882 was never true

883 if english: 

884 english += "; " + t 

885 else: 

886 english = t 

887 

888 # See if the linkage contains a parenthesized alt 

889 alt = None 

890 m = re.search(r"\(([^)]+)\)$", w) 

891 if m: 891 ↛ 892line 891 didn't jump to line 892 because the condition on line 891 was never true

892 w = w[: m.start()].strip() 

893 alt = m.group(1) 

894 

895 dt = {"word": w} 

896 if field_tags: 896 ↛ 897line 896 didn't jump to line 897 because the condition on line 896 was never true

897 data_extend(dt, "tags", field_tags) 

898 if tags: 

899 data_extend(dt, "tags", tags) 

900 if topics: 900 ↛ 901line 900 didn't jump to line 901 because the condition on line 900 was never true

901 data_extend(dt, "topics", topics) 

902 if english: 

903 dt["english"] = english # DEPRECATED for "translation" 

904 dt["translation"] = english 

905 if alt: 905 ↛ 906line 905 didn't jump to line 906 because the condition on line 905 was never true

906 dt["alt"] = alt 

907 data_append(data, field, dt) 

908 

909 

910EXAMPLE_SPLITTERS = r"\s*[―—]+\s*" 

911example_splitter_re = re.compile(EXAMPLE_SPLITTERS) 

912captured_splitters_re = re.compile(r"(" + EXAMPLE_SPLITTERS + r")") 

913 

914 

915def synch_splits_with_args( 

916 line: str, targs: TemplateArgs 

917) -> Optional[list[str]]: 

918 """If it looks like there's something weird with how a line of example 

919 text has been split, this function will do the splitting after counting 

920 occurences of the splitting regex inside the two main template arguments 

921 containing the string data for the original language example and the 

922 English translations. 

923 """ 

924 # Previously, we split without capturing groups, but here we want to 

925 # keep the original splitting hyphen regex intact. 

926 fparts = captured_splitters_re.split(line) 

927 new_parts = [] 

928 # ["First", " – ", "second", " – ", "third..."] from OL argument 

929 first = 1 + (2 * len(example_splitter_re.findall(targs.get(2, "")))) 

930 new_parts.append("".join(fparts[:first])) 

931 # Translation argument 

932 tr_arg = targs.get(3) or targs.get("translation") or targs.get("t", "") 

933 # +2 = + 1 to skip the "expected" hyphen, + 1 as the `1 +` above. 

934 second = first + 2 + (2 * len(example_splitter_re.findall(tr_arg))) 

935 new_parts.append("".join(fparts[first + 1 : second])) 

936 

937 if all(new_parts): # no empty strings from the above spaghetti 

938 new_parts.extend(fparts[second + 1 :: 2]) # skip rest of hyphens 

939 return new_parts 

940 else: 

941 return None 

942 

943 

944QUALIFIERS = r"^\((([^()]|\([^()]*\))*)\):?\s*" 

945QUALIFIERS_RE = re.compile(QUALIFIERS) 

946# (...): ... or (...(...)...): ... 

947 

948 

949def parse_language( 

950 wxr: WiktextractContext, langnode: WikiNode, language: str, lang_code: str 

951) -> list[WordData]: 

952 """Iterates over the text of the page, returning words (parts-of-speech) 

953 defined on the page one at a time. (Individual word senses for the 

954 same part-of-speech are typically encoded in the same entry.)""" 

955 # imported here to avoid circular import 

956 from .pronunciation import parse_pronunciation 

957 

958 assert isinstance(wxr, WiktextractContext) 

959 assert isinstance(langnode, WikiNode) 

960 assert isinstance(language, str) 

961 assert isinstance(lang_code, str) 

962 # print("parse_language", language) 

963 

964 is_reconstruction = False 

965 word: str = wxr.wtp.title # type: ignore[assignment] 

966 unsupported_prefix = "Unsupported titles/" 

967 if word.startswith(unsupported_prefix): 

968 w = word[len(unsupported_prefix) :] 

969 if w in unsupported_title_map: 969 ↛ 972line 969 didn't jump to line 972 because the condition on line 969 was always true

970 word = unsupported_title_map[w] 

971 else: 

972 wxr.wtp.error( 

973 "Unimplemented unsupported title: {}".format(word), 

974 sortid="page/870", 

975 ) 

976 word = w 

977 elif word.startswith("Reconstruction:"): 

978 word = word[word.find("/") + 1 :] 

979 is_reconstruction = True 

980 elif word.startswith("a/languages"): 980 ↛ 982line 980 didn't jump to line 982 because the condition on line 980 was never true

981 # ATM there's only one "mammoth page" in English wiktionary, 'a' 

982 word = "a" 

983 

984 base_data: WordData = { 

985 "word": word, 

986 "lang": language, 

987 "lang_code": lang_code, 

988 } 

989 if is_reconstruction: 

990 data_append(base_data, "tags", "reconstruction") 

991 sense_data: SenseData = {} 

992 pos_data: WordData = {} # For a current part-of-speech 

993 level_four_data: WordData = {} # Chinese Pronunciation-sections in-between 

994 etym_data: WordData = {} # For one etymology 

995 sense_datas: list[SenseData] = [] 

996 sense_ordinal = 0 # The recursive sense parsing messes up the ordering 

997 # Never reset, do not use as data 

998 level_four_datas: list[WordData] = [] 

999 etym_datas: list[WordData] = [] 

1000 page_datas: list[WordData] = [] 

1001 have_etym = False 

1002 inside_level_four = False # This is for checking if the etymology section 

1003 # or article has a Pronunciation section, for Chinese mostly; because 

1004 # Chinese articles can have three level three sections (two etymology 

1005 # sections and pronunciation sections) one after another, we need a kludge 

1006 # to better keep track of whether we're in a normal "etym" or inside a 

1007 # "level four" (which is what we've turned the level three Pron sections 

1008 # into in the fix_subtitle_hierarchy(); all other sections are demoted by 

1009 # a step. 

1010 stack: list[str] = [] # names of items on the "stack" 

1011 

1012 def merge_base(data: WordData, base: WordData) -> None: 

1013 for k, v in base.items(): 

1014 # Copy the value to ensure that we don't share lists or 

1015 # dicts between structures (even nested ones). 

1016 v = copy.deepcopy(v) 

1017 if k not in data: 

1018 # The list was copied above, so this will not create shared ref 

1019 data[k] = v # type: ignore[literal-required] 

1020 continue 

1021 if data[k] == v: # type: ignore[literal-required] 

1022 continue 

1023 if ( 1023 ↛ 1031line 1023 didn't jump to line 1031 because the condition on line 1023 was always true

1024 isinstance(data[k], (list, tuple)) # type: ignore[literal-required] 

1025 or isinstance( 

1026 v, 

1027 (list, tuple), # Should this be "and"? 

1028 ) 

1029 ): 

1030 data[k] = list(data[k]) + list(v) # type: ignore 

1031 elif data[k] != v: # type: ignore[literal-required] 

1032 wxr.wtp.warning( 

1033 "conflicting values for {} in merge_base: " 

1034 "{!r} vs {!r}".format(k, data[k], v), # type: ignore[literal-required] 

1035 sortid="page/904", 

1036 ) 

1037 

1038 def complementary_pop(pron: SoundData, key: str) -> SoundData: 

1039 """Remove unnecessary keys from dict values 

1040 in a list comprehension...""" 

1041 if key in pron: 

1042 pron.pop(key) # type: ignore 

1043 return pron 

1044 

1045 def sound_matches_pos(sound: SoundData, pos: str) -> bool: 

1046 if "pos" not in sound: 

1047 return True 

1048 sound_pos = sound["pos"] # type: ignore[typeddict-item] 

1049 return pos in sound_pos 

1050 

1051 def strip_sound_pos(sound: SoundData) -> SoundData: 

1052 complementary_pop(sound, "pos") 

1053 return sound 

1054 

1055 # If the result has sounds, eliminate sounds that have a prefix that 

1056 # does not match "word" or one of "forms" 

1057 if "sounds" in data and "word" in data: 

1058 accepted = [data["word"]] 

1059 accepted.extend(f["form"] for f in data.get("forms", dict())) 

1060 data["sounds"] = list( 

1061 s 

1062 for s in data["sounds"] 

1063 if "form" not in s or s["form"] in accepted 

1064 ) 

1065 # If the result has sounds, eliminate sounds that have a pos that 

1066 # does not match "pos" 

1067 if "sounds" in data and "pos" in data: 

1068 data["sounds"] = list( 

1069 strip_sound_pos(s) 

1070 for s in data["sounds"] 

1071 # "pos" is not a field of SoundData, correctly, so we're 

1072 # removing it here. It's a kludge on a kludge on a kludge. 

1073 if sound_matches_pos(s, data["pos"]) 

1074 ) 

1075 elif "sounds" in data: 1075 ↛ 1076line 1075 didn't jump to line 1076 because the condition on line 1075 was never true

1076 data["sounds"] = [strip_sound_pos(s) for s in data["sounds"]] 

1077 

1078 def push_sense(sorting_ordinal: int | None = None) -> bool: 

1079 """Starts collecting data for a new word sense. This returns True 

1080 if a sense was added.""" 

1081 nonlocal sense_data 

1082 if sorting_ordinal is None: 

1083 sorting_ordinal = sense_ordinal 

1084 tags = sense_data.get("tags", ()) 

1085 if ( 

1086 not sense_data.get("glosses") 

1087 and "translation-hub" not in tags 

1088 and "no-gloss" not in tags 

1089 ): 

1090 return False 

1091 

1092 if ( 1092 ↛ 1102line 1092 didn't jump to line 1102 because the condition on line 1092 was never true

1093 ( 

1094 "participle" in sense_data.get("tags", ()) 

1095 or "infinitive" in sense_data.get("tags", ()) 

1096 ) 

1097 and "alt_of" not in sense_data 

1098 and "form_of" not in sense_data 

1099 and "etymology_text" in etym_data 

1100 and etym_data["etymology_text"] != "" 

1101 ): 

1102 etym = etym_data["etymology_text"] 

1103 etym = etym.split(". ")[0] 

1104 ret = parse_alt_or_inflection_of(wxr, etym, set()) 

1105 if ret is not None: 

1106 tags, lst = ret 

1107 assert isinstance(lst, (list, tuple)) 

1108 if "form-of" in tags: 

1109 data_extend(sense_data, "form_of", lst) 

1110 data_extend(sense_data, "tags", tags) 

1111 elif "alt-of" in tags: 

1112 data_extend(sense_data, "alt_of", lst) 

1113 data_extend(sense_data, "tags", tags) 

1114 

1115 if not sense_data.get("glosses") and "no-gloss" not in sense_data.get( 1115 ↛ 1118line 1115 didn't jump to line 1118 because the condition on line 1115 was never true

1116 "tags", () 

1117 ): 

1118 data_append(sense_data, "tags", "no-gloss") 

1119 

1120 sense_data["__temp_sense_sorting_ordinal"] = sorting_ordinal # type: ignore 

1121 sense_datas.append(sense_data) 

1122 sense_data = {} 

1123 return True 

1124 

1125 def push_pos(sorting_ordinal: int | None = None) -> None: 

1126 """Starts collecting data for a new part-of-speech.""" 

1127 nonlocal pos_data 

1128 nonlocal sense_datas 

1129 push_sense(sorting_ordinal) 

1130 if wxr.wtp.subsection: 

1131 data: WordData = {"senses": sense_datas} 

1132 merge_base(data, pos_data) 

1133 level_four_datas.append(data) 

1134 pos_data = {} 

1135 sense_datas = [] 

1136 wxr.wtp.start_subsection(None) 

1137 

1138 def push_level_four_section(clear_sound_data: bool) -> None: 

1139 """Starts collecting data for a new level four sections, which 

1140 is usually virtual and empty, unless the article has Chinese 

1141 'Pronunciation' sections that are etymology-section-like but 

1142 under etymology, and at the same level in the source. We modify 

1143 the source to demote Pronunciation sections like that to level 

1144 4, and other sections one step lower.""" 

1145 nonlocal level_four_data 

1146 nonlocal level_four_datas 

1147 nonlocal etym_datas 

1148 push_pos() 

1149 # print(f"======\n{etym_data=}") 

1150 # print(f"======\n{etym_datas=}") 

1151 # print(f"======\n{level_four_data=}") 

1152 # print(f"======\n{level_four_datas=}") 

1153 for data in level_four_datas: 

1154 merge_base(data, level_four_data) 

1155 etym_datas.append(data) 

1156 for data in etym_datas: 

1157 merge_base(data, etym_data) 

1158 page_datas.append(data) 

1159 if clear_sound_data: 

1160 level_four_data = {} 

1161 level_four_datas = [] 

1162 etym_datas = [] 

1163 

1164 def push_etym() -> None: 

1165 """Starts collecting data for a new etymology.""" 

1166 nonlocal etym_data 

1167 nonlocal etym_datas 

1168 nonlocal have_etym 

1169 nonlocal inside_level_four 

1170 have_etym = True 

1171 push_level_four_section(False) 

1172 inside_level_four = False 

1173 # etymology section could under pronunciation section 

1174 etym_data = ( 

1175 copy.deepcopy(level_four_data) if len(level_four_data) > 0 else {} 

1176 ) 

1177 

1178 def select_data() -> WordData: 

1179 """Selects where to store data (pos or etym) based on whether we 

1180 are inside a pos (part-of-speech).""" 

1181 # print(f"{wxr.wtp.subsection=}") 

1182 # print(f"{stack=}") 

1183 if wxr.wtp.subsection is not None: 

1184 return pos_data 

1185 if inside_level_four: 

1186 return level_four_data 

1187 if stack[-1] == language: 

1188 return base_data 

1189 return etym_data 

1190 

1191 def parse_part_of_speech(posnode: WikiNode, pos: str) -> None: 

1192 """Parses the subsection for a part-of-speech under a language on 

1193 a page.""" 

1194 assert isinstance(posnode, WikiNode) 

1195 assert isinstance(pos, str) 

1196 # print("parse_part_of_speech", pos) 

1197 pos_data["pos"] = pos 

1198 pre: list[list[Union[str, WikiNode]]] = [[]] # list of lists 

1199 lists: list[list[WikiNode]] = [[]] # list of lists 

1200 first_para = True 

1201 first_head_tmplt = True 

1202 collecting_head = True 

1203 start_of_paragraph = True 

1204 

1205 # XXX extract templates from posnode with recursively_extract 

1206 # that break stuff, like ja-kanji or az-suffix-form. 

1207 # Do the extraction with a list of template names, combined from 

1208 # different lists, then separate out them into different lists 

1209 # that are handled at different points of the POS section. 

1210 # First, extract az-suffix-form, put it in `inflection`, 

1211 # and parse `inflection`'s content when appropriate later. 

1212 # The contents of az-suffix-form (and ja-kanji) that generate 

1213 # divs with "floatright" in their style gets deleted by 

1214 # clean_value, so templates that slip through from here won't 

1215 # break anything. 

1216 # XXX bookmark 

1217 # print("===================") 

1218 # print(posnode.children) 

1219 

1220 floaters, poschildren = recursively_extract( 

1221 posnode.children, 

1222 lambda x: ( 

1223 isinstance(x, WikiNode) 

1224 and ( 

1225 ( 

1226 isinstance(x, TemplateNode) 

1227 and x.template_name in FLOATING_TABLE_TEMPLATES 

1228 ) 

1229 or ( 

1230 x.kind == NodeKind.LINK 

1231 # Need to check for stringiness because some links are 

1232 # broken; for example, if a template is missing an 

1233 # argument, a link might look like `[[{{{1}}}...]]` 

1234 and len(x.largs) > 0 

1235 and len(x.largs[0]) > 0 

1236 and isinstance(x.largs[0][0], str) 

1237 and x.largs[0][0].lower().startswith("file:") # type:ignore[union-attr] 

1238 ) 

1239 ) 

1240 ), 

1241 ) 

1242 tempnode = WikiNode(NodeKind.LEVEL6, 0) 

1243 tempnode.largs = [["Inflection"]] 

1244 tempnode.children = floaters 

1245 parse_inflection(tempnode, "Floating Div", pos) 

1246 # print(poschildren) 

1247 # XXX new above 

1248 

1249 if not poschildren: 1249 ↛ 1250line 1249 didn't jump to line 1250 because the condition on line 1249 was never true

1250 if not floaters: 

1251 wxr.wtp.debug( 

1252 "PoS section without contents", 

1253 sortid="en/page/1051/20230612", 

1254 ) 

1255 else: 

1256 wxr.wtp.debug( 

1257 "PoS section without contents except for a floating table", 

1258 sortid="en/page/1056/20230612", 

1259 ) 

1260 return 

1261 

1262 for node in poschildren: 

1263 if isinstance(node, str): 

1264 for m in re.finditer(r"\n+|[^\n]+", node): 

1265 p = m.group(0) 

1266 if p.startswith("\n\n") and pre: 

1267 first_para = False 

1268 start_of_paragraph = True 

1269 break 

1270 if p and collecting_head: 

1271 pre[-1].append(p) 

1272 continue 

1273 assert isinstance(node, WikiNode) 

1274 kind = node.kind 

1275 if kind == NodeKind.LIST: 

1276 lists[-1].append(node) 

1277 collecting_head = False 

1278 start_of_paragraph = True 

1279 continue 

1280 elif kind in LEVEL_KINDS: 

1281 # Stop parsing section if encountering any kind of 

1282 # level header (like ===Noun=== or ====Further Reading====). 

1283 # At a quick glance, this should be the default behavior, 

1284 # but if some kinds of source articles have sub-sub-sections 

1285 # that should be parsed XXX it should be handled by changing 

1286 # this break. 

1287 break 

1288 elif collecting_head and kind == NodeKind.LINK: 

1289 # We might collect relevant links as they are often pictures 

1290 # relating to the word 

1291 if len(node.largs[0]) >= 1 and isinstance( 1291 ↛ 1306line 1291 didn't jump to line 1306 because the condition on line 1291 was always true

1292 node.largs[0][0], str 

1293 ): 

1294 if node.largs[0][0].startswith( 1294 ↛ 1300line 1294 didn't jump to line 1300 because the condition on line 1294 was never true

1295 ns_title_prefix_tuple(wxr, "Category") 

1296 ): 

1297 # [[Category:...]] 

1298 # We're at the end of the file, probably, so stop 

1299 # here. Otherwise the head will get garbage. 

1300 break 

1301 if node.largs[0][0].startswith( 

1302 ns_title_prefix_tuple(wxr, "File") 

1303 ): 

1304 # Skips file links 

1305 continue 

1306 start_of_paragraph = False 

1307 pre[-1].append(node) 

1308 elif kind == NodeKind.HTML: 

1309 if node.sarg == "br": 

1310 if pre[-1]: 1310 ↛ 1262line 1310 didn't jump to line 1262 because the condition on line 1310 was always true

1311 pre.append([]) # Switch to next head 

1312 lists.append([]) # Lists parallels pre 

1313 collecting_head = True 

1314 start_of_paragraph = True 

1315 elif collecting_head and node.sarg not in ( 1315 ↛ 1321line 1315 didn't jump to line 1321 because the condition on line 1315 was never true

1316 "gallery", 

1317 "ref", 

1318 "cite", 

1319 "caption", 

1320 ): 

1321 start_of_paragraph = False 

1322 pre[-1].append(node) 

1323 else: 

1324 start_of_paragraph = False 

1325 elif isinstance(node, TemplateNode): 

1326 # XXX Insert code here that disambiguates between 

1327 # templates that generate word heads and templates 

1328 # that don't. 

1329 # There's head_tag_re that seems like a regex meant 

1330 # to identify head templates. Too bad it's None. 

1331 

1332 # ignore {{category}}, {{cat}}... etc. 

1333 if node.template_name in stop_head_at_these_templates: 

1334 # we've reached a template that should be at the end, 

1335 continue 

1336 

1337 # skip these templates; panel_templates is already used 

1338 # to skip certain templates else, but it also applies to 

1339 # head parsing quite well. 

1340 # node.largs[0][0] should always be str, but can't type-check 

1341 # that. 

1342 if is_panel_template(wxr, node.template_name): 

1343 continue 

1344 # skip these templates 

1345 # if node.largs[0][0] in skip_these_templates_in_head: 

1346 # first_head_tmplt = False # no first_head_tmplt at all 

1347 # start_of_paragraph = False 

1348 # continue 

1349 

1350 if first_head_tmplt and pre[-1]: 

1351 first_head_tmplt = False 

1352 start_of_paragraph = False 

1353 pre[-1].append(node) 

1354 elif pre[-1] and start_of_paragraph: 

1355 pre.append([]) # Switch to the next head 

1356 lists.append([]) # lists parallel pre 

1357 collecting_head = True 

1358 start_of_paragraph = False 

1359 pre[-1].append(node) 

1360 else: 

1361 pre[-1].append(node) 

1362 elif first_para: 

1363 start_of_paragraph = False 

1364 if collecting_head: 1364 ↛ 1262line 1364 didn't jump to line 1262 because the condition on line 1364 was always true

1365 pre[-1].append(node) 

1366 # XXX use template_fn in clean_node to check that the head macro 

1367 # is compatible with the current part-of-speech and generate warning 

1368 # if not. Use template_allowed_pos_map. 

1369 

1370 # Clean up empty pairs, and fix messes with extra newlines that 

1371 # separate templates that are followed by lists wiktextract issue #314 

1372 

1373 cleaned_pre: list[list[Union[str, WikiNode]]] = [] 

1374 cleaned_lists: list[list[WikiNode]] = [] 

1375 pairless_pre_index = None 

1376 

1377 for pre1, ls in zip(pre, lists): 

1378 if pre1 and not ls: 

1379 pairless_pre_index = len(cleaned_pre) 

1380 if not pre1 and not ls: 1380 ↛ 1382line 1380 didn't jump to line 1382 because the condition on line 1380 was never true

1381 # skip [] + [] 

1382 continue 

1383 if not ls and all( 

1384 (isinstance(x, str) and not x.strip()) for x in pre1 

1385 ): 

1386 # skip ["\n", " "] + [] 

1387 continue 

1388 if ls and not pre1: 

1389 if pairless_pre_index is not None: 1389 ↛ 1390line 1389 didn't jump to line 1390 because the condition on line 1389 was never true

1390 cleaned_lists[pairless_pre_index] = ls 

1391 pairless_pre_index = None 

1392 continue 

1393 cleaned_pre.append(pre1) 

1394 cleaned_lists.append(ls) 

1395 

1396 pre = cleaned_pre 

1397 lists = cleaned_lists 

1398 

1399 there_are_many_heads = len(pre) > 1 

1400 header_tags: list[str] = [] 

1401 header_topics: list[str] = [] 

1402 previous_head_had_list = False 

1403 

1404 if not any(g for g in lists): 

1405 process_gloss_without_list( 

1406 poschildren, pos, pos_data, header_tags, header_topics 

1407 ) 

1408 else: 

1409 for i, (pre1, ls) in enumerate(zip(pre, lists)): 

1410 # if len(ls) == 0: 

1411 # # don't have gloss list 

1412 # # XXX add code here to filter out 'garbage', like text 

1413 # # that isn't a head template or head. 

1414 # continue 

1415 

1416 if all(not sl for sl in lists[i:]): 

1417 if i == 0: 1417 ↛ 1418line 1417 didn't jump to line 1418 because the condition on line 1417 was never true

1418 if isinstance(node, str): 

1419 wxr.wtp.debug( 

1420 "first head without list of senses," 

1421 "string: '{}[...]', {}/{}".format( 

1422 node[:20], word, language 

1423 ), 

1424 sortid="page/1689/20221215", 

1425 ) 

1426 if isinstance(node, WikiNode): 

1427 if node.largs and node.largs[0][0] in [ 

1428 "Han char", 

1429 ]: 

1430 # just ignore these templates 

1431 pass 

1432 else: 

1433 wxr.wtp.debug( 

1434 "first head without " 

1435 "list of senses, " 

1436 "template node " 

1437 "{}, {}/{}".format( 

1438 node.largs, word, language 

1439 ), 

1440 sortid="page/1694/20221215", 

1441 ) 

1442 else: 

1443 wxr.wtp.debug( 

1444 "first head without list of senses, " 

1445 "{}/{}".format(word, language), 

1446 sortid="page/1700/20221215", 

1447 ) 

1448 # no break here so that the first head always 

1449 # gets processed. 

1450 else: 

1451 if isinstance(node, str): 1451 ↛ 1452line 1451 didn't jump to line 1452 because the condition on line 1451 was never true

1452 wxr.wtp.debug( 

1453 "later head without list of senses," 

1454 "string: '{}[...]', {}/{}".format( 

1455 node[:20], word, language 

1456 ), 

1457 sortid="page/1708/20221215", 

1458 ) 

1459 if isinstance(node, WikiNode): 1459 ↛ 1471line 1459 didn't jump to line 1471 because the condition on line 1459 was always true

1460 wxr.wtp.debug( 

1461 "later head without list of senses," 

1462 "template node " 

1463 "{}, {}/{}".format( 

1464 node.sarg if node.sarg else node.largs, 

1465 word, 

1466 language, 

1467 ), 

1468 sortid="page/1713/20221215", 

1469 ) 

1470 else: 

1471 wxr.wtp.debug( 

1472 "later head without list of senses, " 

1473 "{}/{}".format(word, language), 

1474 sortid="page/1719/20221215", 

1475 ) 

1476 break 

1477 head_group = i + 1 if there_are_many_heads else None 

1478 # print("parse_part_of_speech: {}: {}: pre={}" 

1479 # .format(wxr.wtp.section, wxr.wtp.subsection, pre1)) 

1480 

1481 if previous_head_had_list: 

1482 # We use a boolean flag here because we want to be able 

1483 # let the header_tags data pass through after the loop 

1484 # is over without accidentally emptying it, if there are 

1485 # no pos_datas and we need a dummy data. 

1486 header_tags.clear() 

1487 header_topics.clear() 

1488 

1489 # print(f"{pre1=}") 

1490 process_gloss_header( 

1491 pre1, pos, head_group, pos_data, header_tags, header_topics 

1492 ) 

1493 for ln in ls: 

1494 # Parse each list associated with this head. 

1495 for node in ln.children: 

1496 # Parse nodes in l.children recursively. 

1497 # The recursion function uses push_sense() to 

1498 # add stuff into sense_datas, and returns True or 

1499 # False if something is added, which bubbles upward. 

1500 # If the bubble is "True", then higher levels of 

1501 # the recursion will not push_sense(), because 

1502 # the data is already pushed into a sub-gloss 

1503 # downstream, unless the higher level has examples 

1504 # that need to be put somewhere. 

1505 common_data: SenseData = { 

1506 "tags": list(header_tags), 

1507 "topics": list(header_topics), 

1508 } 

1509 if head_group: 

1510 common_data["head_nr"] = head_group 

1511 parse_sense_node(node, common_data, pos) # type: ignore[arg-type] 

1512 

1513 if len(ls) > 0: 

1514 previous_head_had_list = True 

1515 else: 

1516 previous_head_had_list = False 

1517 

1518 # If there are no senses extracted, add a dummy sense. We want to 

1519 # keep tags extracted from the head for the dummy sense. 

1520 push_sense() # Make sure unfinished data pushed, and start clean sense 

1521 if len(sense_datas) == 0: 

1522 data_extend(sense_data, "tags", header_tags) 

1523 data_extend(sense_data, "topics", header_topics) 

1524 data_append(sense_data, "tags", "no-gloss") 

1525 push_sense() 

1526 

1527 sense_datas.sort(key=lambda x: x.get("__temp_sense_sorting_ordinal", 0)) # type: ignore 

1528 

1529 for sd in sense_datas: 

1530 if "__temp_sense_sorting_ordinal" in sd: 1530 ↛ 1529line 1530 didn't jump to line 1529 because the condition on line 1530 was always true

1531 del sd["__temp_sense_sorting_ordinal"] # type: ignore 

1532 

1533 term_label_templates: list[TemplateData] = [] 

1534 normal_label_templates: list[TemplateData] = [] 

1535 

1536 def head_post_template_fn( 

1537 name: str, ht: TemplateArgs, expansion: str 

1538 ) -> Optional[str]: 

1539 """Handles special templates in the head section of a word. Head 

1540 section is the text after part-of-speech subtitle and before word 

1541 sense list. Typically it generates the bold line for the word, but 

1542 may also contain other useful information that often ends in 

1543 side boxes. We want to capture some of that additional information.""" 

1544 # print("HEAD_POST_TEMPLATE_FN", name, ht) 

1545 if is_panel_template(wxr, name): 1545 ↛ 1548line 1545 didn't jump to line 1548 because the condition on line 1545 was never true

1546 # Completely ignore these templates (not even recorded in 

1547 # head_templates) 

1548 return "" 

1549 if name == "head": 

1550 # XXX are these also captured in forms? Should this special case 

1551 # be removed? 

1552 t = ht.get(2, "") 

1553 if t == "pinyin": 1553 ↛ 1554line 1553 didn't jump to line 1554 because the condition on line 1553 was never true

1554 data_append(pos_data, "tags", "Pinyin") 

1555 elif t == "romanization": 1555 ↛ 1556line 1555 didn't jump to line 1556 because the condition on line 1555 was never true

1556 data_append(pos_data, "tags", "romanization") 

1557 if ( 

1558 HEAD_TAG_RE.search(name) is not None 

1559 or name in PROBLEMATIC_TEMPLATES_CLUMP 

1560 ): 

1561 args_ht = clean_template_args(wxr, ht) 

1562 cleaned_expansion = clean_node(wxr, None, expansion) 

1563 dt: TemplateData = { 

1564 "name": name, 

1565 "args": args_ht, 

1566 "expansion": cleaned_expansion, 

1567 } 

1568 if name in ETYMOLOGY_TEMPLATES_IN_HEADS: 

1569 etymology_template_append( 

1570 pos_data, name, args_ht, cleaned_expansion 

1571 ) 

1572 else: 

1573 data_append(pos_data, "head_templates", dt) 

1574 if name in WORD_LEVEL_HEAD_TEMPLATES: 

1575 term_label_templates.append(dt) 

1576 # Squash these, their tags are applied to the whole word, 

1577 # and some cause problems like "term-label" 

1578 return "" 

1579 

1580 # The following are both captured in head_templates and parsed 

1581 # separately 

1582 

1583 if name in wikipedia_templates: 

1584 # Note: various places expect to have content from wikipedia 

1585 # templates, so cannot convert this to empty 

1586 parse_wikipedia_template(wxr, pos_data, ht) 

1587 return None 

1588 

1589 if name == "number box": 1589 ↛ 1591line 1589 didn't jump to line 1591 because the condition on line 1589 was never true

1590 # XXX extract numeric value? 

1591 return "" 

1592 if name == "enum": 

1593 # XXX extract? 

1594 return "" 

1595 if name == "cardinalbox": 1595 ↛ 1598line 1595 didn't jump to line 1598 because the condition on line 1595 was never true

1596 # XXX extract similar to enum? 

1597 # XXX this can also occur in top-level under language 

1598 return "" 

1599 if name == "Han simplified forms": 1599 ↛ 1601line 1599 didn't jump to line 1601 because the condition on line 1599 was never true

1600 # XXX extract? 

1601 return "" 

1602 # if name == "ja-kanji forms": 

1603 # # XXX extract? 

1604 # return "" 

1605 # if name == "vi-readings": 

1606 # # XXX extract? 

1607 # return "" 

1608 # if name == "ja-kanji": 

1609 # # XXX extract? 

1610 # return "" 

1611 if name == "picdic" or name == "picdicimg" or name == "picdiclabel": 1611 ↛ 1613line 1611 didn't jump to line 1613 because the condition on line 1611 was never true

1612 # XXX extract? 

1613 return "" 

1614 if name == "defdate": 1614 ↛ 1616line 1614 didn't jump to line 1616 because the condition on line 1614 was never true

1615 # the one exampe I saw of this in a head was weird. 

1616 return "" 

1617 if name in ("lb", "lbl", "label"): 

1618 args_ht = clean_template_args(wxr, ht) 

1619 cleaned_expansion = clean_node(wxr, None, expansion).strip("()") 

1620 dt = { 

1621 "name": name, 

1622 "args": args_ht, 

1623 "expansion": cleaned_expansion, 

1624 } 

1625 normal_label_templates.append(dt) 

1626 # The parens around __LABEL... below is meaningful: label 

1627 # templates generate text with parens, so if we add the magical 

1628 # phrase here with parens, it will look like a normal label that 

1629 # will be handled as a parenthetical text; only when handling 

1630 # parenthetical text do we need to actually actually access 

1631 # the contents of the label. 

1632 return f"(__LABEL_TEMPLATE_{len(normal_label_templates) - 1}__)" 

1633 

1634 return None 

1635 

1636 def process_gloss_header( 

1637 header_nodes: list[Union[WikiNode, str]], 

1638 pos_type: str, 

1639 header_group: Optional[int], 

1640 pos_data: WordData, 

1641 header_tags: list[str], 

1642 header_topics: list[str], 

1643 ) -> None: 

1644 ruby = [] 

1645 

1646 # process template parse nodes here 

1647 new_nodes = [] 

1648 info_template_data = [] 

1649 for node in header_nodes: 

1650 # print(f"{node=}") 

1651 info_data, info_out = parse_info_template_node(wxr, node, "head") 

1652 if info_data or info_out: 

1653 if info_data: 1653 ↛ 1655line 1653 didn't jump to line 1655 because the condition on line 1653 was always true

1654 info_template_data.append(info_data) 

1655 if info_out: # including just the original node 1655 ↛ 1656line 1655 didn't jump to line 1656 because the condition on line 1655 was never true

1656 new_nodes.append(info_out) 

1657 else: 

1658 new_nodes.append(node) 

1659 header_nodes = new_nodes 

1660 

1661 if info_template_data: 

1662 if "info_templates" not in pos_data: 1662 ↛ 1665line 1662 didn't jump to line 1665 because the condition on line 1662 was always true

1663 pos_data["info_templates"] = info_template_data 

1664 else: 

1665 pos_data["info_templates"].extend(info_template_data) 

1666 

1667 if lang_code == "ja": 

1668 exp = wxr.wtp.parse( 

1669 wxr.wtp.node_to_wikitext(header_nodes), expand_all=True 

1670 ) 

1671 rub, _ = recursively_extract( 

1672 exp.children, 

1673 lambda x: ( 

1674 isinstance(x, WikiNode) 

1675 and x.kind == NodeKind.HTML 

1676 and x.sarg == "ruby" 

1677 ), 

1678 ) 

1679 if rub is not None: 1679 ↛ 1723line 1679 didn't jump to line 1723 because the condition on line 1679 was always true

1680 for r in rub: 

1681 if TYPE_CHECKING: 

1682 # we know the lambda above in recursively_extract 

1683 # returns only WikiNodes in rub 

1684 assert isinstance(r, WikiNode) 

1685 rt = parse_ruby(wxr, r) 

1686 if rt is not None: 1686 ↛ 1680line 1686 didn't jump to line 1680 because the condition on line 1686 was always true

1687 ruby.append(rt) 

1688 elif lang_code == "vi": 

1689 # Handle vi-readings templates that have a weird structures for 

1690 # Chu Nom vietnamese characters heads 

1691 # https://en.wiktionary.org/wiki/Template:vi-readings 

1692 new_header_nodes = [] 

1693 related_readings: list[LinkageData] = [] 

1694 for node in header_nodes: 

1695 if ( 1695 ↛ 1718line 1695 didn't jump to line 1718 because the condition on line 1695 was always true

1696 isinstance(node, TemplateNode) 

1697 and node.template_name == "vi-readings" 

1698 ): 

1699 for parameter, tag in ( 

1700 ("hanviet", "han-viet-reading"), 

1701 ("nom", "nom-reading"), 

1702 # we ignore the fanqie parameter "phienthiet" 

1703 ): 

1704 arg = node.template_parameters.get(parameter) 

1705 if arg is not None: 1705 ↛ 1699line 1705 didn't jump to line 1699 because the condition on line 1705 was always true

1706 text = clean_node(wxr, None, arg) 

1707 for w in text.split(","): 

1708 # ignore - separated references 

1709 if "-" in w: 

1710 w = w[: w.index("-")] 

1711 w = w.strip() 

1712 related_readings.append( 

1713 LinkageData(word=w, tags=[tag]) 

1714 ) 

1715 continue 

1716 

1717 # Skip the vi-reading template for the rest of the head parsing 

1718 new_header_nodes.append(node) 

1719 if len(related_readings) > 0: 1719 ↛ 1723line 1719 didn't jump to line 1723 because the condition on line 1719 was always true

1720 data_extend(pos_data, "related", related_readings) 

1721 header_nodes = new_header_nodes 

1722 

1723 header_text = clean_node( 

1724 wxr, 

1725 pos_data, 

1726 header_nodes, 

1727 post_template_fn=head_post_template_fn, 

1728 collect_links=True, 

1729 remove_anchors_from_links=True, 

1730 ) 

1731 if "links" in pos_data: 

1732 # WordData doesn't use `links`, so we can use `collect_links=True` 

1733 # above without special handling and smuggle link data. 

1734 extracted_links = pos_data["links"] # type: ignore 

1735 del pos_data["links"] # type: ignore 

1736 else: 

1737 extracted_links = None 

1738 # print(f"{header_text=}, {extracted_links=}") 

1739 

1740 header_text = re.sub(r"\s+", " ", header_text).strip() 

1741 

1742 if not header_text: 

1743 return 

1744 

1745 term_label_tags: list[str] = [] 

1746 term_label_topics: list[str] = [] 

1747 if len(term_label_templates) > 0: 

1748 # parse term label templates; if there are other similar kinds 

1749 # of templates in headers that you want to squash and apply as 

1750 # tags, you can add them to WORD_LEVEL_HEAD_TEMPLATES 

1751 for templ_data in term_label_templates: 

1752 # print(templ_data) 

1753 expan = templ_data.get("expansion", "").strip("().,; ") 

1754 if not expan: 1754 ↛ 1755line 1754 didn't jump to line 1755 because the condition on line 1754 was never true

1755 continue 

1756 tlb_tagsets, tlb_topics = decode_tags(expan) 

1757 for tlb_tags in tlb_tagsets: 

1758 if len(tlb_tags) > 0 and not any( 

1759 t.startswith("error-") for t in tlb_tags 

1760 ): 

1761 term_label_tags.extend(tlb_tags) 

1762 term_label_topics.extend(tlb_topics) 

1763 # print(f"{tlb_tagsets=}, {tlb_topicsets=}") 

1764 

1765 # print(f"{header_text=}") 

1766 parse_word_head( 

1767 wxr, 

1768 word, 

1769 pos_type, 

1770 header_text, 

1771 pos_data, 

1772 is_reconstruction, 

1773 header_group, 

1774 header_nodes, 

1775 ruby=ruby, 

1776 links=extracted_links, 

1777 label_templates=normal_label_templates, 

1778 ) 

1779 if "tags" in pos_data: 

1780 # pos_data can get "tags" data from some source; type-checkers 

1781 # doesn't like it, so let's ignore it. 

1782 header_tags.extend(pos_data["tags"]) # type: ignore[typeddict-item] 

1783 del pos_data["tags"] # type: ignore[typeddict-item] 

1784 if len(term_label_tags) > 0: 

1785 header_tags.extend(term_label_tags) 

1786 if len(term_label_topics) > 0: 

1787 header_topics.extend(term_label_topics) 

1788 

1789 def process_gloss_without_list( 

1790 nodes: list[Union[WikiNode, str]], 

1791 pos_type: str, 

1792 pos_data: WordData, 

1793 header_tags: list[str], 

1794 header_topics: list[str], 

1795 ) -> None: 

1796 # gloss text might not inside a list 

1797 header_nodes: list[Union[str, WikiNode]] = [] 

1798 gloss_nodes: list[Union[str, WikiNode]] = [] 

1799 for node in strip_nodes(nodes): 

1800 if isinstance(node, WikiNode): 

1801 if isinstance(node, TemplateNode): 

1802 if node.template_name in ( 

1803 "zh-see", 

1804 "ja-see", 

1805 "ja-see-kango", 

1806 ): 

1807 continue # soft redirect 

1808 elif ( 

1809 node.template_name == "head" 

1810 or node.template_name.startswith(f"{lang_code}-") 

1811 ): 

1812 header_nodes.append(node) 

1813 continue 

1814 elif node.kind in LEVEL_KINDS: # following nodes are not gloss 1814 ↛ 1816line 1814 didn't jump to line 1816 because the condition on line 1814 was always true

1815 break 

1816 gloss_nodes.append(node) 

1817 

1818 if len(header_nodes) > 0: 

1819 process_gloss_header( 

1820 header_nodes, 

1821 pos_type, 

1822 None, 

1823 pos_data, 

1824 header_tags, 

1825 header_topics, 

1826 ) 

1827 if len(gloss_nodes) > 0: 

1828 process_gloss_contents( 

1829 gloss_nodes, 

1830 pos_type, 

1831 {"tags": list(header_tags), "topics": list(header_topics)}, 

1832 ) 

1833 

1834 def parse_sense_node( 

1835 node: Union[str, WikiNode], # never receives str 

1836 sense_base: SenseData, 

1837 pos: str, 

1838 ) -> bool: 

1839 """Recursively (depth first) parse LIST_ITEM nodes for sense data. 

1840 Uses push_sense() to attempt adding data to pos_data in the scope 

1841 of parse_language() when it reaches deep in the recursion. push_sense() 

1842 returns True if it succeeds, and that is bubbled up the stack; if 

1843 a sense was added downstream, the higher levels (whose shared data 

1844 was already added by a subsense) do not push_sense(), unless it 

1845 has examples that need to be put somewhere. 

1846 """ 

1847 assert isinstance(sense_base, dict) # Added to every sense deeper in 

1848 

1849 nonlocal sense_ordinal 

1850 my_ordinal = sense_ordinal # copies, not a reference 

1851 sense_ordinal += 1 # only use for sorting 

1852 

1853 if not isinstance(node, WikiNode): 1853 ↛ 1855line 1853 didn't jump to line 1855 because the condition on line 1853 was never true

1854 # This doesn't seem to ever happen in practice. 

1855 wxr.wtp.debug( 

1856 "{}: parse_sense_node called with" 

1857 "something that isn't a WikiNode".format(pos), 

1858 sortid="page/1287/20230119", 

1859 ) 

1860 return False 

1861 

1862 if node.kind != NodeKind.LIST_ITEM: 1862 ↛ 1863line 1862 didn't jump to line 1863 because the condition on line 1862 was never true

1863 wxr.wtp.debug( 

1864 "{}: non-list-item inside list".format(pos), sortid="page/1678" 

1865 ) 

1866 return False 

1867 

1868 if node.sarg == ":": 

1869 # Skip example entries at the highest level, ones without 

1870 # a sense ("...#") above them. 

1871 # If node.sarg is exactly and only ":", then it's at 

1872 # the highest level; lower levels would have more 

1873 # "indentation", like "#:" or "##:" 

1874 return False 

1875 

1876 # If a recursion call succeeds in push_sense(), bubble it up with 

1877 # `added`. 

1878 # added |= push_sense() or added |= parse_sense_node(...) to OR. 

1879 added = False 

1880 

1881 gloss_template_args: set[str] = set() 

1882 

1883 # For LISTs and LIST_ITEMS, their argument is something like 

1884 # "##" or "##:", and using that we can rudimentally determine 

1885 # list 'depth' if need be, and also what kind of list or 

1886 # entry it is; # is for normal glosses, : for examples (indent) 

1887 # and * is used for quotations on wiktionary. 

1888 current_depth = node.sarg 

1889 

1890 children = node.children 

1891 

1892 # subentries, (presumably) a list 

1893 # of subglosses below this. The list's 

1894 # argument ends with #, and its depth should 

1895 # be bigger than parent node. 

1896 subentries = [ 

1897 x 

1898 for x in children 

1899 if isinstance(x, WikiNode) 

1900 and x.kind == NodeKind.LIST 

1901 and x.sarg == current_depth + "#" 

1902 ] 

1903 

1904 # sublists of examples and quotations. .sarg 

1905 # does not end with "#". 

1906 others = [ 

1907 x 

1908 for x in children 

1909 if isinstance(x, WikiNode) 

1910 and x.kind == NodeKind.LIST 

1911 and x.sarg != current_depth + "#" 

1912 ] 

1913 

1914 # the actual contents of this particular node. 

1915 # can be a gloss (or a template that expands into 

1916 # many glosses which we can't easily pre-expand) 

1917 # or could be an "outer gloss" with more specific 

1918 # subglosses, or could be a qualfier for the subglosses. 

1919 contents = [ 

1920 x 

1921 for x in children 

1922 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST 

1923 ] 

1924 # If this entry has sublists of entries, we should combine 

1925 # gloss information from both the "outer" and sublist content. 

1926 # Sometimes the outer gloss 

1927 # is more non-gloss or tags, sometimes it is a coarse sense 

1928 # and the inner glosses are more specific. The outer one 

1929 # does not seem to have qualifiers. 

1930 

1931 # If we have one sublist with one element, treat it 

1932 # specially as it may be a Wiktionary error; raise 

1933 # that nested element to the same level. 

1934 # XXX If need be, this block can be easily removed in 

1935 # the current recursive logicand the result is one sense entry 

1936 # with both glosses in the glosses list, as you would 

1937 # expect. If the higher entry has examples, there will 

1938 # be a higher entry with some duplicated data. 

1939 if len(subentries) == 1: 

1940 slc = subentries[0].children 

1941 if len(slc) == 1: 

1942 # copy current node and modify it so it doesn't 

1943 # loop infinitely. 

1944 cropped_node = copy.copy(node) 

1945 cropped_node.children = [ 

1946 x 

1947 for x in children 

1948 if not ( 

1949 isinstance(x, WikiNode) 

1950 and x.kind == NodeKind.LIST 

1951 and x.sarg == current_depth + "#" 

1952 ) 

1953 ] 

1954 added |= parse_sense_node(cropped_node, sense_base, pos) 

1955 nonlocal sense_data # this kludge causes duplicated raw_ 

1956 # glosses data if this is not done; 

1957 # if the top-level (cropped_node) 

1958 # does not push_sense() properly or 

1959 # parse_sense_node() returns early, 

1960 # sense_data is not reset. This happens 

1961 # for example when you have a no-gloss 

1962 # string like "(intransitive)": 

1963 # no gloss, push_sense() returns early 

1964 # and sense_data has duplicate data with 

1965 # sense_base 

1966 sense_data = {} 

1967 added |= parse_sense_node(slc[0], sense_base, pos) 

1968 return added 

1969 

1970 return process_gloss_contents( 

1971 contents, 

1972 pos, 

1973 sense_base, 

1974 subentries, 

1975 others, 

1976 gloss_template_args, 

1977 added, 

1978 my_ordinal, 

1979 ) 

1980 

1981 def process_gloss_contents( 

1982 contents: list[Union[str, WikiNode]], 

1983 pos: str, 

1984 sense_base: SenseData, 

1985 subentries: list[WikiNode] = [], 

1986 others: list[WikiNode] = [], 

1987 gloss_template_args: Set[str] = set(), 

1988 added: bool = False, 

1989 sorting_ordinal: int | None = None, 

1990 ) -> bool: 

1991 def sense_template_fn( 

1992 name: str, ht: TemplateArgs, is_gloss: bool = False 

1993 ) -> Optional[str]: 

1994 # print(f"sense_template_fn: {name}, {ht}") 

1995 if name in wikipedia_templates: 

1996 # parse_wikipedia_template(wxr, pos_data, ht) 

1997 return None 

1998 if is_panel_template(wxr, name): 

1999 return "" 

2000 if name in INFO_TEMPLATE_FUNCS: 

2001 info_data, info_exp = parse_info_template_arguments( 

2002 wxr, name, ht, "sense" 

2003 ) 

2004 if info_data or info_exp: 2004 ↛ 2010line 2004 didn't jump to line 2010 because the condition on line 2004 was always true

2005 if info_data: 2005 ↛ 2007line 2005 didn't jump to line 2007 because the condition on line 2005 was always true

2006 data_append(sense_base, "info_templates", info_data) 

2007 if info_exp and isinstance(info_exp, str): 2007 ↛ 2009line 2007 didn't jump to line 2009 because the condition on line 2007 was always true

2008 return info_exp 

2009 return "" 

2010 if name in ("defdate",): 

2011 date = clean_node(wxr, None, ht.get(1, ())) 

2012 if part_two := ht.get(2): 2012 ↛ 2014line 2012 didn't jump to line 2014 because the condition on line 2012 was never true

2013 # Unicode mdash, not '-' 

2014 date += "–" + clean_node(wxr, None, part_two) 

2015 refs: dict[str, ReferenceData] = {} 

2016 # ref, refn, ref2, ref2n, ref3, ref3n 

2017 # ref1 not valid 

2018 for k, v in sorted( 

2019 (k, v) for k, v in ht.items() if isinstance(k, str) 

2020 ): 

2021 if m := re.match(r"ref(\d?)(n?)", k): 2021 ↛ 2018line 2021 didn't jump to line 2018 because the condition on line 2021 was always true

2022 ref_v = clean_node(wxr, None, v) 

2023 if m.group(1) not in refs: # empty string or digit 

2024 refs[m.group(1)] = ReferenceData() 

2025 if m.group(2): 

2026 refs[m.group(1)]["refn"] = ref_v 

2027 else: 

2028 refs[m.group(1)]["text"] = ref_v 

2029 data_append( 

2030 sense_base, 

2031 "attestations", 

2032 AttestationData(date=date, references=list(refs.values())), 

2033 ) 

2034 return "" 

2035 if name == "senseid": 

2036 langid = clean_node(wxr, None, ht.get(1, ())) 

2037 arg = clean_node(wxr, sense_base, ht.get(2, ())) 

2038 if re.match(r"Q\d+$", arg): 

2039 data_append(sense_base, "wikidata", arg) 

2040 data_append(sense_base, "senseid", langid + ":" + arg) 

2041 if name in sense_linkage_templates: 

2042 # print(f"SENSE_TEMPLATE_FN: {name}") 

2043 parse_sense_linkage(wxr, sense_base, name, ht, pos) 

2044 return "" 

2045 if name == "†" or name == "zh-obsolete": 

2046 data_append(sense_base, "tags", "obsolete") 

2047 return "" 

2048 if name in { 

2049 "ux", 

2050 "uxi", 

2051 "usex", 

2052 "afex", 

2053 "prefixusex", 

2054 "ko-usex", 

2055 "ko-x", 

2056 "hi-x", 

2057 "ja-usex-inline", 

2058 "ja-x", 

2059 "quotei", 

2060 "he-x", 

2061 "hi-x", 

2062 "km-x", 

2063 "ne-x", 

2064 "shn-x", 

2065 "th-x", 

2066 "ur-x", 

2067 }: 

2068 # Usage examples are captured separately below. We don't 

2069 # want to expand them into glosses even when unusual coding 

2070 # is used in the entry. 

2071 # These templates may slip through inside another item, but 

2072 # currently we're separating out example entries (..#:) 

2073 # well enough that there seems to very little contamination. 

2074 if is_gloss: 

2075 wxr.wtp.wiki_notice( 

2076 "Example template is used for gloss text", 

2077 sortid="extractor.en.page.sense_template_fn/1415", 

2078 ) 

2079 else: 

2080 return "" 

2081 if name == "w": 2081 ↛ 2082line 2081 didn't jump to line 2082 because the condition on line 2081 was never true

2082 if ht.get(2) == "Wp": 

2083 return "" 

2084 for v in ht.values(): 

2085 v = v.strip() 

2086 if v and "<" not in v: 

2087 gloss_template_args.add(v) 

2088 return None 

2089 

2090 def extract_link_texts(item: GeneralNode) -> None: 

2091 """Recursively extracts link texts from the gloss source. This 

2092 information is used to select whether to remove final "." from 

2093 form_of/alt_of (e.g., ihm/Hunsrik).""" 

2094 if isinstance(item, (list, tuple)): 

2095 for x in item: 

2096 extract_link_texts(x) 

2097 return 

2098 if isinstance(item, str): 

2099 # There seem to be HTML sections that may futher contain 

2100 # unparsed links. 

2101 for m in re.finditer(r"\[\[([^]]*)\]\]", item): 2101 ↛ 2102line 2101 didn't jump to line 2102 because the loop on line 2101 never started

2102 print("ITER:", m.group(0)) 

2103 v = m.group(1).split("|")[-1].strip() 

2104 if v: 

2105 gloss_template_args.add(v) 

2106 return 

2107 if not isinstance(item, WikiNode): 2107 ↛ 2108line 2107 didn't jump to line 2108 because the condition on line 2107 was never true

2108 return 

2109 if item.kind == NodeKind.LINK: 

2110 v = item.largs[-1] 

2111 if ( 2111 ↛ 2117line 2111 didn't jump to line 2117 because the condition on line 2111 was always true

2112 isinstance(v, list) 

2113 and len(v) == 1 

2114 and isinstance(v[0], str) 

2115 ): 

2116 gloss_template_args.add(v[0].strip()) 

2117 for x in item.children: 

2118 extract_link_texts(x) 

2119 

2120 extract_link_texts(contents) 

2121 

2122 # get the raw text of non-list contents of this node, and other stuff 

2123 # like tag and category data added to sense_base 

2124 # cast = no-op type-setter for the type-checker 

2125 partial_template_fn = cast( 

2126 TemplateFnCallable, 

2127 partial(sense_template_fn, is_gloss=True), 

2128 ) 

2129 rawgloss = clean_node( 

2130 wxr, 

2131 sense_base, 

2132 contents, 

2133 template_fn=partial_template_fn, 

2134 collect_links=True, 

2135 ) 

2136 

2137 if not rawgloss: 2137 ↛ 2138line 2137 didn't jump to line 2138 because the condition on line 2137 was never true

2138 return False 

2139 

2140 # remove manually typed ordered list text at the start("1. ") 

2141 rawgloss = re.sub(r"^\d+\.\s+", "", rawgloss).strip() 

2142 

2143 # get stuff like synonyms and categories from "others", 

2144 # maybe examples and quotations 

2145 clean_node(wxr, sense_base, others, template_fn=sense_template_fn) 

2146 

2147 # The gloss could contain templates that produce more list items. 

2148 # This happens commonly with, e.g., {{inflection of|...}}. Split 

2149 # to parts. However, e.g. Interlingua generates multiple glosses 

2150 # in HTML directly without Wikitext markup, so we must also split 

2151 # by just newlines. 

2152 subglosses = rawgloss.splitlines() 

2153 

2154 if len(subglosses) == 0: 2154 ↛ 2155line 2154 didn't jump to line 2155 because the condition on line 2154 was never true

2155 return False 

2156 

2157 if any(s.startswith("#") for s in subglosses): 

2158 subtree = wxr.wtp.parse(rawgloss) 

2159 # from wikitextprocessor.parser import print_tree 

2160 # print("SUBTREE GENERATED BY TEMPLATE:") 

2161 # print_tree(subtree) 

2162 new_subentries = [ 

2163 x 

2164 for x in subtree.children 

2165 if isinstance(x, WikiNode) and x.kind == NodeKind.LIST 

2166 ] 

2167 

2168 new_others = [ 

2169 x 

2170 for x in subtree.children 

2171 if isinstance(x, WikiNode) 

2172 and x.kind == NodeKind.LIST 

2173 and not x.sarg.endswith("#") 

2174 ] 

2175 

2176 new_contents = [ 

2177 clean_node(wxr, [], x) 

2178 for x in subtree.children 

2179 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST 

2180 ] 

2181 

2182 subentries = subentries or new_subentries 

2183 others = others or new_others 

2184 subglosses = new_contents 

2185 rawgloss = "".join(subglosses) 

2186 # Generate no gloss for translation hub pages, but add the 

2187 # "translation-hub" tag for them 

2188 if rawgloss == "(This entry is a translation hub.)": 2188 ↛ 2189line 2188 didn't jump to line 2189 because the condition on line 2188 was never true

2189 data_append(sense_data, "tags", "translation-hub") 

2190 return push_sense(sorting_ordinal) 

2191 

2192 # Remove certain substrings specific to outer glosses 

2193 strip_ends = [", particularly:"] 

2194 for x in strip_ends: 

2195 if rawgloss.endswith(x): 

2196 rawgloss = rawgloss[: -len(x)].strip() 

2197 break 

2198 

2199 # A single gloss, or possibly an outer gloss. 

2200 # Check if the possible outer gloss starts with 

2201 # parenthesized tags/topics 

2202 

2203 if rawgloss and rawgloss not in sense_base.get("raw_glosses", ()): 

2204 data_append(sense_base, "raw_glosses", subglosses[0].strip()) 

2205 m = QUALIFIERS_RE.match(rawgloss) 

2206 # (...): ... or (...(...)...): ... 

2207 if m: 

2208 q = m.group(1) 

2209 rawgloss = rawgloss[m.end() :].strip() 

2210 parse_sense_qualifier(wxr, q, sense_base) 

2211 if rawgloss == "A pejorative:": 2211 ↛ 2212line 2211 didn't jump to line 2212 because the condition on line 2211 was never true

2212 data_append(sense_base, "tags", "pejorative") 

2213 rawgloss = "" 

2214 elif rawgloss == "Short forms.": 2214 ↛ 2215line 2214 didn't jump to line 2215 because the condition on line 2214 was never true

2215 data_append(sense_base, "tags", "abbreviation") 

2216 rawgloss = "" 

2217 elif rawgloss == "Technical or specialized senses.": 2217 ↛ 2218line 2217 didn't jump to line 2218 because the condition on line 2217 was never true

2218 rawgloss = "" 

2219 elif rawgloss.startswith("inflection of "): 

2220 parsed = parse_alt_or_inflection_of(wxr, rawgloss, set()) 

2221 if parsed is not None: 2221 ↛ 2230line 2221 didn't jump to line 2230 because the condition on line 2221 was always true

2222 tags, origins = parsed 

2223 if origins is not None: 2223 ↛ 2225line 2223 didn't jump to line 2225 because the condition on line 2223 was always true

2224 data_extend(sense_base, "form_of", origins) 

2225 if tags is not None: 2225 ↛ 2228line 2225 didn't jump to line 2228 because the condition on line 2225 was always true

2226 data_extend(sense_base, "tags", tags) 

2227 else: 

2228 data_append(sense_base, "tags", "form-of") 

2229 else: 

2230 data_append(sense_base, "tags", "form-of") 

2231 if rawgloss: 2231 ↛ 2262line 2231 didn't jump to line 2262 because the condition on line 2231 was always true

2232 # Code duplicating a lot of clean-up operations from later in 

2233 # this block. We want to clean up the "supergloss" as much as 

2234 # possible, in almost the same way as a normal gloss. 

2235 supergloss = rawgloss 

2236 

2237 if supergloss.startswith("; "): 2237 ↛ 2238line 2237 didn't jump to line 2238 because the condition on line 2237 was never true

2238 supergloss = supergloss[1:].strip() 

2239 

2240 if supergloss.startswith(("^†", "†")): 

2241 data_append(sense_base, "tags", "obsolete") 

2242 supergloss = supergloss[2:].strip() 

2243 elif supergloss.startswith("^‡"): 2243 ↛ 2244line 2243 didn't jump to line 2244 because the condition on line 2243 was never true

2244 data_extend(sense_base, "tags", ["obsolete", "historical"]) 

2245 supergloss = supergloss[2:].strip() 

2246 

2247 # remove [14th century...] style brackets at the end 

2248 supergloss = re.sub(r"\s\[[^]]*\]\s*$", "", supergloss) 

2249 

2250 if supergloss.startswith((",", ":")): 

2251 supergloss = supergloss[1:] 

2252 supergloss = supergloss.strip() 

2253 if supergloss.startswith("N. of "): 2253 ↛ 2254line 2253 didn't jump to line 2254 because the condition on line 2253 was never true

2254 supergloss = "Name of " + supergloss[6:] 

2255 supergloss = supergloss[2:] 

2256 data_append(sense_base, "glosses", supergloss) 

2257 if supergloss in ("A person:",): 

2258 data_append(sense_base, "tags", "g-person") 

2259 

2260 # The main recursive call (except for the exceptions at the 

2261 # start of this function). 

2262 for sublist in subentries: 

2263 if not ( 2263 ↛ 2266line 2263 didn't jump to line 2266 because the condition on line 2263 was never true

2264 isinstance(sublist, WikiNode) and sublist.kind == NodeKind.LIST 

2265 ): 

2266 wxr.wtp.debug( 

2267 f"'{repr(rawgloss[:20])}.' gloss has `subentries`" 

2268 f"with items that are not LISTs", 

2269 sortid="page/1511/20230119", 

2270 ) 

2271 continue 

2272 for item in sublist.children: 

2273 if not ( 2273 ↛ 2277line 2273 didn't jump to line 2277 because the condition on line 2273 was never true

2274 isinstance(item, WikiNode) 

2275 and item.kind == NodeKind.LIST_ITEM 

2276 ): 

2277 continue 

2278 # copy sense_base to prevent cross-contamination between 

2279 # subglosses and other subglosses and superglosses 

2280 sense_base2 = copy.deepcopy(sense_base) 

2281 if parse_sense_node(item, sense_base2, pos): 2281 ↛ 2272line 2281 didn't jump to line 2272 because the condition on line 2281 was always true

2282 added = True 

2283 

2284 # Capture examples. 

2285 # This is called after the recursive calls above so that 

2286 # sense_base is not contaminated with meta-data from 

2287 # example entries for *this* gloss. 

2288 examples = [] 

2289 if wxr.config.capture_examples: 2289 ↛ 2293line 2289 didn't jump to line 2293 because the condition on line 2289 was always true

2290 examples = extract_examples(others, sense_base) 

2291 

2292 # push_sense() succeeded somewhere down-river, so skip this level 

2293 if added: 

2294 if examples: 

2295 # this higher-up gloss has examples that we do not want to skip 

2296 wxr.wtp.debug( 

2297 "'{}[...]' gloss has examples we want to keep, " 

2298 "but there are subglosses.".format(repr(rawgloss[:30])), 

2299 sortid="page/1498/20230118", 

2300 ) 

2301 else: 

2302 return True 

2303 

2304 # Some entries, e.g., "iacebam", have weird sentences in quotes 

2305 # after the gloss, but these sentences don't seem to be intended 

2306 # as glosses. Skip them. 

2307 indexed_subglosses = list( 

2308 (i, gl) 

2309 for i, gl in enumerate(subglosses) 

2310 if gl.strip() and not re.match(r'\s*(\([^)]*\)\s*)?"[^"]*"\s*$', gl) 

2311 ) 

2312 

2313 if len(indexed_subglosses) > 1 and "form_of" not in sense_base: 2313 ↛ 2314line 2313 didn't jump to line 2314 because the condition on line 2313 was never true

2314 gl = indexed_subglosses[0][1].strip() 

2315 if gl.endswith(":"): 

2316 gl = gl[:-1].strip() 

2317 parsed = parse_alt_or_inflection_of(wxr, gl, gloss_template_args) 

2318 if parsed is not None: 

2319 infl_tags, infl_dts = parsed 

2320 if infl_dts and "form-of" in infl_tags and len(infl_tags) == 1: 

2321 # Interpret others as a particular form under 

2322 # "inflection of" 

2323 data_extend(sense_base, "tags", infl_tags) 

2324 data_extend(sense_base, "form_of", infl_dts) 

2325 indexed_subglosses = indexed_subglosses[1:] 

2326 elif not infl_dts: 

2327 data_extend(sense_base, "tags", infl_tags) 

2328 indexed_subglosses = indexed_subglosses[1:] 

2329 

2330 # Create senses for remaining subglosses 

2331 for i, (gloss_i, gloss) in enumerate(indexed_subglosses): 

2332 gloss = gloss.strip() 

2333 if not gloss and len(indexed_subglosses) > 1: 2333 ↛ 2334line 2333 didn't jump to line 2334 because the condition on line 2333 was never true

2334 continue 

2335 # Push a new sense (if the last one is not empty) 

2336 if push_sense(sorting_ordinal): 2336 ↛ 2337line 2336 didn't jump to line 2337 because the condition on line 2336 was never true

2337 added = True 

2338 # if gloss not in sense_data.get("raw_glosses", ()): 

2339 # data_append(sense_data, "raw_glosses", gloss) 

2340 if i == 0 and examples: 

2341 # In a multi-line gloss, associate examples 

2342 # with only one of them. 

2343 # XXX or you could use gloss_i == len(indexed_subglosses) 

2344 # to associate examples with the *last* one. 

2345 data_extend(sense_data, "examples", examples) 

2346 if gloss.startswith("; ") and gloss_i > 0: 2346 ↛ 2347line 2346 didn't jump to line 2347 because the condition on line 2346 was never true

2347 gloss = gloss[1:].strip() 

2348 # If the gloss starts with †, mark as obsolete 

2349 if gloss.startswith("^†"): 2349 ↛ 2350line 2349 didn't jump to line 2350 because the condition on line 2349 was never true

2350 data_append(sense_data, "tags", "obsolete") 

2351 gloss = gloss[2:].strip() 

2352 elif gloss.startswith("^‡"): 2352 ↛ 2353line 2352 didn't jump to line 2353 because the condition on line 2352 was never true

2353 data_extend(sense_data, "tags", ["obsolete", "historical"]) 

2354 gloss = gloss[2:].strip() 

2355 # Copy data for all senses to this sense 

2356 for k, v in sense_base.items(): 

2357 if isinstance(v, (list, tuple)): 

2358 if k != "tags": 

2359 # Tags handled below (countable/uncountable special) 

2360 data_extend(sense_data, k, v) 

2361 else: 

2362 assert k not in ("tags", "categories", "topics") 

2363 sense_data[k] = v # type:ignore[literal-required] 

2364 # Parse the gloss for this particular sense 

2365 m = QUALIFIERS_RE.match(gloss) 

2366 # (...): ... or (...(...)...): ... 

2367 if m: 

2368 parse_sense_qualifier(wxr, m.group(1), sense_data) 

2369 gloss = gloss[m.end() :].strip() 

2370 

2371 # Remove common suffix "[from 14th c.]" and similar 

2372 gloss = re.sub(r"\s\[[^]]*\]\s*$", "", gloss) 

2373 

2374 # Check to make sure we don't have unhandled list items in gloss 

2375 ofs = max(gloss.find("#"), gloss.find("* ")) 

2376 if ofs > 10 and "(#)" not in gloss: 

2377 wxr.wtp.debug( 

2378 "gloss may contain unhandled list items: {}".format(gloss), 

2379 sortid="page/1412", 

2380 ) 

2381 elif "\n" in gloss: 2381 ↛ 2382line 2381 didn't jump to line 2382 because the condition on line 2381 was never true

2382 wxr.wtp.debug( 

2383 "gloss contains newline: {}".format(gloss), 

2384 sortid="page/1416", 

2385 ) 

2386 

2387 # Kludge, some glosses have a comma after initial qualifiers in 

2388 # parentheses 

2389 if gloss.startswith((",", ":")): 

2390 gloss = gloss[1:] 

2391 gloss = gloss.strip() 

2392 if gloss.endswith(":"): 

2393 gloss = gloss[:-1].strip() 

2394 if gloss.startswith("N. of "): 2394 ↛ 2395line 2394 didn't jump to line 2395 because the condition on line 2394 was never true

2395 gloss = "Name of " + gloss[6:] 

2396 if gloss.startswith("†"): 2396 ↛ 2397line 2396 didn't jump to line 2397 because the condition on line 2396 was never true

2397 data_append(sense_data, "tags", "obsolete") 

2398 gloss = gloss[1:] 

2399 elif gloss.startswith("^†"): 2399 ↛ 2400line 2399 didn't jump to line 2400 because the condition on line 2399 was never true

2400 data_append(sense_data, "tags", "obsolete") 

2401 gloss = gloss[2:] 

2402 

2403 # Copy tags from sense_base if any. This will not copy 

2404 # countable/uncountable if either was specified in the sense, 

2405 # as sometimes both are specified in word head but only one 

2406 # in individual senses. 

2407 countability_tags = [] 

2408 base_tags = sense_base.get("tags", ()) 

2409 sense_tags = sense_data.get("tags", ()) 

2410 for tag in base_tags: 

2411 if tag in ("countable", "uncountable"): 

2412 if tag not in countability_tags: 2412 ↛ 2414line 2412 didn't jump to line 2414 because the condition on line 2412 was always true

2413 countability_tags.append(tag) 

2414 continue 

2415 if tag not in sense_tags: 

2416 data_append(sense_data, "tags", tag) 

2417 if countability_tags: 

2418 if ( 2418 ↛ 2427line 2418 didn't jump to line 2427 because the condition on line 2418 was always true

2419 "countable" not in sense_tags 

2420 and "uncountable" not in sense_tags 

2421 ): 

2422 data_extend(sense_data, "tags", countability_tags) 

2423 

2424 # If outer gloss specifies a form-of ("inflection of", see 

2425 # aquamarine/German), try to parse the inner glosses as 

2426 # tags for an inflected form. 

2427 if "form-of" in sense_base.get("tags", ()): 

2428 parsed = parse_alt_or_inflection_of( 

2429 wxr, gloss, gloss_template_args 

2430 ) 

2431 if parsed is not None: 2431 ↛ 2437line 2431 didn't jump to line 2437 because the condition on line 2431 was always true

2432 infl_tags, infl_dts = parsed 

2433 if not infl_dts and infl_tags: 2433 ↛ 2437line 2433 didn't jump to line 2437 because the condition on line 2433 was always true

2434 # Interpret as a particular form under "inflection of" 

2435 data_extend(sense_data, "tags", infl_tags) 

2436 

2437 if not gloss: 2437 ↛ 2438line 2437 didn't jump to line 2438 because the condition on line 2437 was never true

2438 data_append(sense_data, "tags", "empty-gloss") 

2439 elif gloss != "-" and gloss not in sense_data.get("glosses", []): 

2440 if ( 2440 ↛ 2451line 2440 didn't jump to line 2451 because the condition on line 2440 was always true

2441 gloss_i == 0 

2442 and len(sense_data.get("glosses", tuple())) >= 1 

2443 ): 

2444 # If we added a "high-level gloss" from rawgloss, but this 

2445 # is that same gloss_i, add this instead of the raw_gloss 

2446 # from before if they're different: the rawgloss was not 

2447 # cleaned exactly the same as this later gloss 

2448 sense_data["glosses"][-1] = gloss 

2449 else: 

2450 # Add the gloss for the sense. 

2451 data_append(sense_data, "glosses", gloss) 

2452 

2453 # Kludge: there are cases (e.g., etc./Swedish) where there are 

2454 # two abbreviations in the same sense, both generated by the 

2455 # {{abbreviation of|...}} template. Handle these with some magic. 

2456 position = 0 

2457 split_glosses = [] 

2458 for m in re.finditer(r"Abbreviation of ", gloss): 

2459 if m.start() != position: 2459 ↛ 2458line 2459 didn't jump to line 2458 because the condition on line 2459 was always true

2460 split_glosses.append(gloss[position : m.start()]) 

2461 position = m.start() 

2462 split_glosses.append(gloss[position:]) 

2463 for gloss in split_glosses: 

2464 # Check if this gloss describes an alt-of or inflection-of 

2465 if ( 

2466 lang_code != "en" 

2467 and " " not in gloss 

2468 and distw([word], gloss) < 0.3 

2469 ): 

2470 # Don't try to parse gloss if it is one word 

2471 # that is close to the word itself for non-English words 

2472 # (probable translations of a tag/form name) 

2473 continue 

2474 parsed = parse_alt_or_inflection_of( 

2475 wxr, gloss, gloss_template_args 

2476 ) 

2477 if parsed is None: 

2478 continue 

2479 tags, dts = parsed 

2480 if not dts and tags: 

2481 data_extend(sense_data, "tags", tags) 

2482 continue 

2483 for dt in dts: # type:ignore[union-attr] 

2484 ftags = list(tag for tag in tags if tag != "form-of") 

2485 if "alt-of" in tags: 

2486 data_extend(sense_data, "tags", ftags) 

2487 data_append(sense_data, "alt_of", dt) 

2488 elif "compound-of" in tags: 2488 ↛ 2489line 2488 didn't jump to line 2489 because the condition on line 2488 was never true

2489 data_extend(sense_data, "tags", ftags) 

2490 data_append(sense_data, "compound_of", dt) 

2491 elif "synonym-of" in tags: 2491 ↛ 2492line 2491 didn't jump to line 2492 because the condition on line 2491 was never true

2492 data_extend(dt, "tags", ftags) 

2493 data_append(sense_data, "synonyms", dt) 

2494 elif tags and dt.get("word", "").startswith("of "): 2494 ↛ 2495line 2494 didn't jump to line 2495 because the condition on line 2494 was never true

2495 dt["word"] = dt["word"][3:] 

2496 data_append(sense_data, "tags", "form-of") 

2497 data_extend(sense_data, "tags", ftags) 

2498 data_append(sense_data, "form_of", dt) 

2499 elif "form-of" in tags: 2499 ↛ 2483line 2499 didn't jump to line 2483 because the condition on line 2499 was always true

2500 data_extend(sense_data, "tags", tags) 

2501 data_append(sense_data, "form_of", dt) 

2502 

2503 if len(sense_data) == 0: 

2504 if len(sense_base.get("tags", [])) == 0: 2504 ↛ 2506line 2504 didn't jump to line 2506 because the condition on line 2504 was always true

2505 del sense_base["tags"] 

2506 sense_data.update(sense_base) 

2507 if push_sense(sorting_ordinal): 2507 ↛ 2511line 2507 didn't jump to line 2511 because the condition on line 2507 was always true

2508 # push_sense succeded in adding a sense to pos_data 

2509 added = True 

2510 # print("PARSE_SENSE DONE:", pos_datas[-1]) 

2511 return added 

2512 

2513 def parse_inflection( 

2514 node: WikiNode, section: str, pos: Optional[str] 

2515 ) -> None: 

2516 """Parses inflection data (declension, conjugation) from the given 

2517 page. This retrieves the actual inflection template 

2518 parameters, which are very useful for applications that need 

2519 to learn the inflection classes and generate inflected 

2520 forms.""" 

2521 assert isinstance(node, WikiNode) 

2522 assert isinstance(section, str) 

2523 assert pos is None or isinstance(pos, str) 

2524 # print("parse_inflection:", node) 

2525 

2526 if pos is None: 2526 ↛ 2527line 2526 didn't jump to line 2527 because the condition on line 2526 was never true

2527 wxr.wtp.debug( 

2528 "inflection table outside part-of-speech", sortid="page/1812" 

2529 ) 

2530 return 

2531 

2532 def inflection_template_fn( 

2533 name: str, ht: TemplateArgs 

2534 ) -> Optional[str]: 

2535 # print("decl_conj_template_fn", name, ht) 

2536 if is_panel_template(wxr, name): 2536 ↛ 2537line 2536 didn't jump to line 2537 because the condition on line 2536 was never true

2537 return "" 

2538 if name in ("is-u-mutation",): 2538 ↛ 2541line 2538 didn't jump to line 2541 because the condition on line 2538 was never true

2539 # These are not to be captured as an exception to the 

2540 # generic code below 

2541 return None 

2542 m = re.search( 

2543 r"-(conj|decl|ndecl|adecl|infl|conjugation|" 

2544 r"declension|inflection|mut|mutation)($|-)", 

2545 name, 

2546 ) 

2547 if m: 

2548 args_ht = clean_template_args(wxr, ht) 

2549 dt = {"name": name, "args": args_ht} 

2550 data_append(pos_data, "inflection_templates", dt) 

2551 

2552 return None 

2553 

2554 # Convert the subtree back to Wikitext, then expand all and parse, 

2555 # capturing templates in the process 

2556 text = wxr.wtp.node_to_wikitext(node.children) 

2557 

2558 # Split text into separate sections for each to-level template 

2559 brace_matches = re.split(r"((?:^|\n)\s*{\||\n\s*\|}|{{+|}}+)", text) 

2560 # ["{{", "template", "}}"] or ["^{|", "table contents", "\n|}"] 

2561 # The (?:...) creates a non-capturing regex group; if it was capturing, 

2562 # like the group around it, it would create elements in brace_matches, 

2563 # including None if it doesn't match. 

2564 # 20250114: Added {| and |} into the regex because tables were being 

2565 # cut into pieces by this code. Issue #973, introduction of two-part 

2566 # book-end templates similar to trans-top and tran-bottom. 

2567 template_sections = [] 

2568 template_nesting = 0 # depth of SINGLE BRACES { { nesting } } 

2569 # Because there is the possibility of triple curly braces 

2570 # ("{{{", "}}}") in addition to normal ("{{ }}"), we do not 

2571 # count nesting depth using pairs of two brackets, but 

2572 # instead use singular braces ("{ }"). 

2573 # Because template delimiters should be balanced, regardless 

2574 # of whether {{ or {{{ is used, and because we only care 

2575 # about the outer-most delimiters (the highest level template) 

2576 # we can just count the single braces when those single 

2577 # braces are part of a group. 

2578 table_nesting = 0 

2579 # However, if we have a stray table ({| ... |}) that should always 

2580 # be its own section, and should prevent templates from cutting it 

2581 # into sections. 

2582 

2583 # print(f"Parse inflection: {text=}") 

2584 # print(f"Brace matches: {repr('///'.join(brace_matches))}") 

2585 if len(brace_matches) > 1: 

2586 tsection: list[str] = [] 

2587 after_templates = False # kludge to keep any text 

2588 # before first template 

2589 # with the first template; 

2590 # otherwise, text 

2591 # goes with preceding template 

2592 for m in brace_matches: 

2593 if m.startswith("\n; ") and after_templates: 2593 ↛ 2594line 2593 didn't jump to line 2594 because the condition on line 2593 was never true

2594 after_templates = False 

2595 template_sections.append(tsection) 

2596 tsection = [] 

2597 tsection.append(m) 

2598 elif m.startswith("{{") or m.endswith("{|"): 

2599 if ( 

2600 template_nesting == 0 

2601 and after_templates 

2602 and table_nesting == 0 

2603 ): 

2604 template_sections.append(tsection) 

2605 tsection = [] 

2606 # start new section 

2607 after_templates = True 

2608 if m.startswith("{{"): 

2609 template_nesting += 1 

2610 else: 

2611 # m.endswith("{|") 

2612 table_nesting += 1 

2613 tsection.append(m) 

2614 elif m.startswith("}}") or m.endswith("|}"): 

2615 if m.startswith("}}"): 

2616 template_nesting -= 1 

2617 if template_nesting < 0: 2617 ↛ 2618line 2617 didn't jump to line 2618 because the condition on line 2617 was never true

2618 wxr.wtp.error( 

2619 "Negatively nested braces, " 

2620 "couldn't split inflection templates, " 

2621 "{}/{} section {}".format( 

2622 word, language, section 

2623 ), 

2624 sortid="page/1871", 

2625 ) 

2626 template_sections = [] # use whole text 

2627 break 

2628 else: 

2629 table_nesting -= 1 

2630 if table_nesting < 0: 2630 ↛ 2631line 2630 didn't jump to line 2631 because the condition on line 2630 was never true

2631 wxr.wtp.error( 

2632 "Negatively nested table braces, " 

2633 "couldn't split inflection section, " 

2634 "{}/{} section {}".format( 

2635 word, language, section 

2636 ), 

2637 sortid="page/20250114", 

2638 ) 

2639 template_sections = [] # use whole text 

2640 break 

2641 tsection.append(m) 

2642 else: 

2643 tsection.append(m) 

2644 if tsection: # dangling tsection 2644 ↛ 2652line 2644 didn't jump to line 2652 because the condition on line 2644 was always true

2645 template_sections.append(tsection) 

2646 # Why do it this way around? The parser has a preference 

2647 # to associate bits outside of tables with the preceding 

2648 # table (`after`-variable), so a new tsection begins 

2649 # at {{ and everything before it belongs to the previous 

2650 # template. 

2651 

2652 texts = [] 

2653 if not template_sections: 

2654 texts = [text] 

2655 else: 

2656 for tsection in template_sections: 

2657 texts.append("".join(tsection)) 

2658 if template_nesting != 0: 2658 ↛ 2659line 2658 didn't jump to line 2659 because the condition on line 2658 was never true

2659 wxr.wtp.error( 

2660 "Template nesting error: " 

2661 "template_nesting = {} " 

2662 "couldn't split inflection templates, " 

2663 "{}/{} section {}".format( 

2664 template_nesting, word, language, section 

2665 ), 

2666 sortid="page/1896", 

2667 ) 

2668 texts = [text] 

2669 for text in texts: 

2670 tree = wxr.wtp.parse( 

2671 text, expand_all=True, template_fn=inflection_template_fn 

2672 ) 

2673 

2674 if not text.strip(): 

2675 continue 

2676 

2677 # Parse inflection tables from the section. The data is stored 

2678 # under "forms". 

2679 if wxr.config.capture_inflections: 2679 ↛ 2669line 2679 didn't jump to line 2669 because the condition on line 2679 was always true

2680 tablecontext = None 

2681 m = re.search(r"{{([^}{|]+)\|?", text) 

2682 if m: 

2683 template_name = m.group(1).strip() 

2684 tablecontext = TableContext(template_name) 

2685 

2686 parse_inflection_section( 

2687 wxr, 

2688 pos_data, 

2689 word, 

2690 language, 

2691 pos, 

2692 section, 

2693 tree, 

2694 tablecontext=tablecontext, 

2695 ) 

2696 

2697 def get_subpage_section( 

2698 title: str, subtitle: str, seqs: list[Union[list[str], tuple[str, ...]]] 

2699 ) -> Optional[Union[WikiNode, str]]: 

2700 """Loads a subpage of the given page, and finds the section 

2701 for the given language, part-of-speech, and section title. This 

2702 is used for finding translations and other sections on subpages.""" 

2703 assert isinstance(language, str) 

2704 assert isinstance(title, str) 

2705 assert isinstance(subtitle, str) 

2706 assert isinstance(seqs, (list, tuple)) 

2707 for seq in seqs: 

2708 for x in seq: 

2709 assert isinstance(x, str) 

2710 subpage_title = word + "/" + subtitle 

2711 subpage_content = wxr.wtp.get_page_body(subpage_title, 0) 

2712 if subpage_content is None: 

2713 wxr.wtp.error( 

2714 "/translations not found despite " 

2715 "{{see translation subpage|...}}", 

2716 sortid="page/1934", 

2717 ) 

2718 return None 

2719 

2720 def recurse( 

2721 node: Union[str, WikiNode], seq: Union[list[str], tuple[str, ...]] 

2722 ) -> Optional[Union[str, WikiNode]]: 

2723 # print(f"seq: {seq}") 

2724 if not seq: 

2725 return node 

2726 if not isinstance(node, WikiNode): 

2727 return None 

2728 # print(f"node.kind: {node.kind}") 

2729 if node.kind in LEVEL_KINDS: 

2730 t = clean_node(wxr, None, node.largs[0]) 

2731 # print(f"t: {t} == seq[0]: {seq[0]}?") 

2732 if t.lower() == seq[0].lower(): 

2733 seq = seq[1:] 

2734 if not seq: 

2735 return node 

2736 for n in node.children: 

2737 ret = recurse(n, seq) 

2738 if ret is not None: 

2739 return ret 

2740 return None 

2741 

2742 tree = wxr.wtp.parse( 

2743 subpage_content, 

2744 pre_expand=True, 

2745 additional_expand=ADDITIONAL_EXPAND_TEMPLATES, 

2746 do_not_pre_expand=DO_NOT_PRE_EXPAND_TEMPLATES, 

2747 ) 

2748 assert tree.kind == NodeKind.ROOT 

2749 for seq in seqs: 

2750 ret = recurse(tree, seq) 

2751 if ret is None: 

2752 wxr.wtp.debug( 

2753 "Failed to find subpage section {}/{} seq {}".format( 

2754 title, subtitle, seq 

2755 ), 

2756 sortid="page/1963", 

2757 ) 

2758 return ret 

2759 

2760 def parse_translations(data: WordData, xlatnode: WikiNode) -> None: 

2761 """Parses translations for a word. This may also pull in translations 

2762 from separate translation subpages.""" 

2763 assert isinstance(data, dict) 

2764 assert isinstance(xlatnode, WikiNode) 

2765 # print("===== PARSE_TRANSLATIONS {} {} {}" 

2766 # .format(wxr.wtp.title, wxr.wtp.section, wxr.wtp.subsection)) 

2767 # print("parse_translations xlatnode={}".format(xlatnode)) 

2768 if not wxr.config.capture_translations: 2768 ↛ 2769line 2768 didn't jump to line 2769 because the condition on line 2768 was never true

2769 return 

2770 sense_parts: list[Union[WikiNode, str]] = [] 

2771 sense: Optional[str] = None 

2772 

2773 def parse_translation_item( 

2774 contents: list[Union[WikiNode, str]], lang: Optional[str] = None 

2775 ) -> None: 

2776 nonlocal sense 

2777 assert isinstance(contents, list) 

2778 assert lang is None or isinstance(lang, str) 

2779 # print("PARSE_TRANSLATION_ITEM:", contents) 

2780 

2781 langcode: Optional[str] = None 

2782 if sense is None: 

2783 sense = clean_node(wxr, data, sense_parts).strip() 

2784 # print("sense <- clean_node: ", sense) 

2785 idx = sense.find("See also translations at") 

2786 if idx > 0: 2786 ↛ 2787line 2786 didn't jump to line 2787 because the condition on line 2786 was never true

2787 wxr.wtp.debug( 

2788 "Skipping translation see also: {}".format(sense), 

2789 sortid="page/2361", 

2790 ) 

2791 sense = sense[:idx].strip() 

2792 if sense.endswith(":"): 2792 ↛ 2793line 2792 didn't jump to line 2793 because the condition on line 2792 was never true

2793 sense = sense[:-1].strip() 

2794 if sense.endswith("—"): 2794 ↛ 2795line 2794 didn't jump to line 2795 because the condition on line 2794 was never true

2795 sense = sense[:-1].strip() 

2796 translations_from_template: list[str] = [] 

2797 

2798 def translation_item_template_fn( 

2799 name: str, ht: TemplateArgs 

2800 ) -> Optional[str]: 

2801 nonlocal langcode 

2802 # print("TRANSLATION_ITEM_TEMPLATE_FN:", name, ht) 

2803 if is_panel_template(wxr, name): 

2804 return "" 

2805 if name in ("t+check", "t-check", "t-needed"): 

2806 # We ignore these templates. They seem to have outright 

2807 # garbage in some entries, and very varying formatting in 

2808 # others. These should be transitory and unreliable 

2809 # anyway. 

2810 return "__IGNORE__" 

2811 if name in ("t", "t+", "t-simple", "tt", "tt+"): 

2812 code = ht.get(1) 

2813 if code: 2813 ↛ 2823line 2813 didn't jump to line 2823 because the condition on line 2813 was always true

2814 if langcode and code != langcode: 

2815 wxr.wtp.debug( 

2816 "inconsistent language codes {} vs " 

2817 "{} in translation item: {!r} {}".format( 

2818 langcode, code, name, ht 

2819 ), 

2820 sortid="page/2386", 

2821 ) 

2822 langcode = code 

2823 tr = ht.get(2) 

2824 if tr: 

2825 tr = clean_node(wxr, None, [tr]) 

2826 translations_from_template.append(tr) 

2827 return None 

2828 if name == "t-egy": 

2829 langcode = "egy" 

2830 return None 

2831 if name == "ttbc": 

2832 code = ht.get(1) 

2833 if code: 2833 ↛ 2835line 2833 didn't jump to line 2835 because the condition on line 2833 was always true

2834 langcode = code 

2835 return None 

2836 if name == "trans-see": 2836 ↛ 2837line 2836 didn't jump to line 2837 because the condition on line 2836 was never true

2837 wxr.wtp.error( 

2838 "UNIMPLEMENTED trans-see template", sortid="page/2405" 

2839 ) 

2840 return "" 

2841 if name.endswith("-top"): 2841 ↛ 2842line 2841 didn't jump to line 2842 because the condition on line 2841 was never true

2842 return "" 

2843 if name.endswith("-bottom"): 2843 ↛ 2844line 2843 didn't jump to line 2844 because the condition on line 2843 was never true

2844 return "" 

2845 if name.endswith("-mid"): 2845 ↛ 2846line 2845 didn't jump to line 2846 because the condition on line 2845 was never true

2846 return "" 

2847 # wxr.wtp.debug("UNHANDLED TRANSLATION ITEM TEMPLATE: {!r}" 

2848 # .format(name), 

2849 # sortid="page/2414") 

2850 return None 

2851 

2852 sublists = list( 

2853 x 

2854 for x in contents 

2855 if isinstance(x, WikiNode) and x.kind == NodeKind.LIST 

2856 ) 

2857 contents = list( 

2858 x 

2859 for x in contents 

2860 if not isinstance(x, WikiNode) or x.kind != NodeKind.LIST 

2861 ) 

2862 

2863 item = clean_node( 

2864 wxr, data, contents, template_fn=translation_item_template_fn 

2865 ) 

2866 # print(" TRANSLATION ITEM: {!r} [{}]".format(item, sense)) 

2867 

2868 # Parse the translation item. 

2869 if item: 2869 ↛ exitline 2869 didn't return from function 'parse_translation_item' because the condition on line 2869 was always true

2870 lang = parse_translation_item_text( 

2871 wxr, 

2872 word, 

2873 data, 

2874 item, 

2875 sense, 

2876 lang, 

2877 langcode, 

2878 translations_from_template, 

2879 is_reconstruction, 

2880 ) 

2881 

2882 # Handle sublists. They are frequently used for different 

2883 # scripts for the language and different variants of the 

2884 # language. We will include the lower-level header as a 

2885 # tag in those cases. 

2886 for listnode in sublists: 

2887 assert listnode.kind == NodeKind.LIST 

2888 for node in listnode.children: 

2889 if not isinstance(node, WikiNode): 2889 ↛ 2890line 2889 didn't jump to line 2890 because the condition on line 2889 was never true

2890 continue 

2891 if node.kind == NodeKind.LIST_ITEM: 2891 ↛ 2888line 2891 didn't jump to line 2888 because the condition on line 2891 was always true

2892 parse_translation_item(node.children, lang=lang) 

2893 

2894 def parse_translation_template(node: WikiNode) -> None: 

2895 assert isinstance(node, WikiNode) 

2896 

2897 def template_fn(name: str, ht: TemplateArgs) -> Optional[str]: 

2898 nonlocal sense_parts 

2899 nonlocal sense 

2900 if is_panel_template(wxr, name): 

2901 return "" 

2902 if name == "see also": 

2903 # XXX capture 

2904 # XXX for example, "/" has top-level list containing 

2905 # see also items. So also should parse those. 

2906 return "" 

2907 if name == "trans-see": 

2908 # XXX capture 

2909 return "" 

2910 if name == "see translation subpage": 2910 ↛ 2911line 2910 didn't jump to line 2911 because the condition on line 2910 was never true

2911 sense_parts = [] 

2912 sense = None 

2913 sub = ht.get(1, "") 

2914 if sub: 

2915 m = re.match( 

2916 r"\s*(([^:\d]*)\s*\d*)\s*:\s*([^:]*)\s*", sub 

2917 ) 

2918 else: 

2919 m = None 

2920 etym = "" 

2921 etym_numbered = "" 

2922 pos = "" 

2923 if m: 

2924 etym_numbered = m.group(1) 

2925 etym = m.group(2) 

2926 pos = m.group(3) 

2927 if not sub: 

2928 wxr.wtp.debug( 

2929 "no part-of-speech in " 

2930 "{{see translation subpage|...}}, " 

2931 "defaulting to just wxr.wtp.section " 

2932 "(= language)", 

2933 sortid="page/2468", 

2934 ) 

2935 # seq sent to get_subpage_section without sub and pos 

2936 seq = [ 

2937 language, 

2938 TRANSLATIONS_TITLE, 

2939 ] 

2940 elif ( 

2941 m 

2942 and etym.lower().strip() in ETYMOLOGY_TITLES 

2943 and pos.lower() in POS_TITLES 

2944 ): 

2945 seq = [ 

2946 language, 

2947 etym_numbered, 

2948 pos, 

2949 TRANSLATIONS_TITLE, 

2950 ] 

2951 elif sub.lower() in POS_TITLES: 

2952 # seq with sub but not pos 

2953 seq = [ 

2954 language, 

2955 sub, 

2956 TRANSLATIONS_TITLE, 

2957 ] 

2958 else: 

2959 # seq with sub and pos 

2960 pos = wxr.wtp.subsection or "MISSING_SUBSECTION" 

2961 if pos.lower() not in POS_TITLES: 

2962 wxr.wtp.debug( 

2963 "unhandled see translation subpage: " 

2964 "language={} sub={} " 

2965 "wxr.wtp.subsection={}".format( 

2966 language, sub, wxr.wtp.subsection 

2967 ), 

2968 sortid="page/2478", 

2969 ) 

2970 seq = [language, sub, pos, TRANSLATIONS_TITLE] 

2971 subnode = get_subpage_section( 

2972 wxr.wtp.title or "MISSING_TITLE", 

2973 TRANSLATIONS_TITLE, 

2974 [seq], 

2975 ) 

2976 if subnode is None or not isinstance(subnode, WikiNode): 

2977 # Failed to find the normal subpage section 

2978 # seq with sub and pos 

2979 pos = wxr.wtp.subsection or "MISSING_SUBSECTION" 

2980 # print(f"{language=}, {pos=}, {TRANSLATIONS_TITLE=}") 

2981 seqs: list[list[str] | tuple[str, ...]] = [ 

2982 [TRANSLATIONS_TITLE], 

2983 [language, pos], 

2984 ] 

2985 subnode = get_subpage_section( 

2986 wxr.wtp.title or "MISSING_TITLE", 

2987 TRANSLATIONS_TITLE, 

2988 seqs, 

2989 ) 

2990 if subnode is not None and isinstance(subnode, WikiNode): 

2991 parse_translations(data, subnode) 

2992 return "" 

2993 if name in ( 

2994 "c", 

2995 "C", 

2996 "categorize", 

2997 "cat", 

2998 "catlangname", 

2999 "topics", 

3000 "top", 

3001 "qualifier", 

3002 "cln", 

3003 ): 

3004 # These are expanded in the default way 

3005 return None 

3006 if name in ( 

3007 "trans-top", 

3008 "trans-top-see", 

3009 ): 

3010 # XXX capture id from trans-top? Capture sense here 

3011 # instead of trying to parse it from expanded content? 

3012 if ht.get(1): 

3013 sense_parts = [] 

3014 sense = ht.get(1) 

3015 else: 

3016 sense_parts = [] 

3017 sense = None 

3018 return None 

3019 if name in ( 

3020 "trans-bottom", 

3021 "trans-mid", 

3022 "checktrans-mid", 

3023 "checktrans-bottom", 

3024 ): 

3025 return None 

3026 if name == "checktrans-top": 

3027 sense_parts = [] 

3028 sense = None 

3029 return "" 

3030 if name == "trans-top-also": 

3031 # XXX capture? 

3032 sense_parts = [] 

3033 sense = None 

3034 return "" 

3035 wxr.wtp.error( 

3036 "UNIMPLEMENTED parse_translation_template: {} {}".format( 

3037 name, ht 

3038 ), 

3039 sortid="page/2517", 

3040 ) 

3041 return "" 

3042 

3043 wxr.wtp.expand( 

3044 wxr.wtp.node_to_wikitext(node), template_fn=template_fn 

3045 ) 

3046 

3047 def parse_translation_recurse(xlatnode: WikiNode) -> None: 

3048 nonlocal sense 

3049 nonlocal sense_parts 

3050 for node in xlatnode.children: 

3051 # print(node) 

3052 if isinstance(node, str): 

3053 if sense: 

3054 if not node.isspace(): 

3055 wxr.wtp.debug( 

3056 "skipping string in the middle of " 

3057 "translations: {}".format(node), 

3058 sortid="page/2530", 

3059 ) 

3060 continue 

3061 # Add a part to the sense 

3062 sense_parts.append(node) 

3063 sense = None 

3064 continue 

3065 assert isinstance(node, WikiNode) 

3066 kind = node.kind 

3067 if kind == NodeKind.LIST: 

3068 for item in node.children: 

3069 if not isinstance(item, WikiNode): 3069 ↛ 3070line 3069 didn't jump to line 3070 because the condition on line 3069 was never true

3070 continue 

3071 if item.kind != NodeKind.LIST_ITEM: 3071 ↛ 3072line 3071 didn't jump to line 3072 because the condition on line 3071 was never true

3072 continue 

3073 if item.sarg == ":": 3073 ↛ 3074line 3073 didn't jump to line 3074 because the condition on line 3073 was never true

3074 continue 

3075 parse_translation_item(item.children) 

3076 elif kind == NodeKind.LIST_ITEM and node.sarg == ":": 3076 ↛ 3080line 3076 didn't jump to line 3080 because the condition on line 3076 was never true

3077 # Silently skip list items that are just indented; these 

3078 # are used for text between translations, such as indicating 

3079 # translations that need to be checked. 

3080 pass 

3081 elif kind == NodeKind.TEMPLATE: 

3082 parse_translation_template(node) 

3083 elif kind in ( 3083 ↛ 3088line 3083 didn't jump to line 3088 because the condition on line 3083 was never true

3084 NodeKind.TABLE, 

3085 NodeKind.TABLE_ROW, 

3086 NodeKind.TABLE_CELL, 

3087 ): 

3088 parse_translation_recurse(node) 

3089 elif kind == NodeKind.HTML: 

3090 if node.attrs.get("class") == "NavFrame": 3090 ↛ 3096line 3090 didn't jump to line 3096 because the condition on line 3090 was never true

3091 # Reset ``sense_parts`` (and force recomputing 

3092 # by clearing ``sense``) as each NavFrame specifies 

3093 # its own sense. This helps eliminate garbage coming 

3094 # from text at the beginning at the translations 

3095 # section. 

3096 sense_parts = [] 

3097 sense = None 

3098 # for item in node.children: 

3099 # if not isinstance(item, WikiNode): 

3100 # continue 

3101 # parse_translation_recurse(item) 

3102 parse_translation_recurse(node) 

3103 elif kind in LEVEL_KINDS: 3103 ↛ 3105line 3103 didn't jump to line 3105 because the condition on line 3103 was never true

3104 # Sub-levels will be recursed elsewhere 

3105 pass 

3106 elif kind in (NodeKind.ITALIC, NodeKind.BOLD): 

3107 parse_translation_recurse(node) 

3108 elif kind == NodeKind.PREFORMATTED: 3108 ↛ 3109line 3108 didn't jump to line 3109 because the condition on line 3108 was never true

3109 print("parse_translation_recurse: PREFORMATTED:", node) 

3110 elif kind == NodeKind.LINK: 3110 ↛ 3164line 3110 didn't jump to line 3164 because the condition on line 3110 was always true

3111 arg0 = node.largs[0] 

3112 # Kludge: I've seen occasional normal links to translation 

3113 # subpages from main pages (e.g., language/English/Noun 

3114 # in July 2021) instead of the normal 

3115 # {{see translation subpage|...}} template. This should 

3116 # handle them. Note: must be careful not to read other 

3117 # links, particularly things like in "human being": 

3118 # "a human being -- see [[man/translations]]" (group title) 

3119 if ( 3119 ↛ 3127line 3119 didn't jump to line 3127 because the condition on line 3119 was never true

3120 isinstance(arg0, (list, tuple)) 

3121 and arg0 

3122 and isinstance(arg0[0], str) 

3123 and arg0[0].endswith("/" + TRANSLATIONS_TITLE) 

3124 and arg0[0][: -(1 + len(TRANSLATIONS_TITLE))] 

3125 == wxr.wtp.title 

3126 ): 

3127 wxr.wtp.debug( 

3128 "translations subpage link found on main " 

3129 "page instead " 

3130 "of normal {{see translation subpage|...}}", 

3131 sortid="page/2595", 

3132 ) 

3133 sub = wxr.wtp.subsection or "MISSING_SUBSECTION" 

3134 if sub.lower() in POS_TITLES: 

3135 seq = [ 

3136 language, 

3137 sub, 

3138 TRANSLATIONS_TITLE, 

3139 ] 

3140 subnode = get_subpage_section( 

3141 wxr.wtp.title, 

3142 TRANSLATIONS_TITLE, 

3143 [seq], 

3144 ) 

3145 if subnode is not None and isinstance( 

3146 subnode, WikiNode 

3147 ): 

3148 parse_translations(data, subnode) 

3149 else: 

3150 wxr.wtp.error( 

3151 "/translations link outside part-of-speech" 

3152 ) 

3153 

3154 if ( 

3155 len(arg0) >= 1 

3156 and isinstance(arg0[0], str) 

3157 and not arg0[0].lower().startswith("category:") 

3158 ): 

3159 for x in node.largs[-1]: 

3160 if isinstance(x, str): 3160 ↛ 3163line 3160 didn't jump to line 3163 because the condition on line 3160 was always true

3161 sense_parts.append(x) 

3162 else: 

3163 parse_translation_recurse(x) 

3164 elif not sense: 

3165 sense_parts.append(node) 

3166 else: 

3167 wxr.wtp.debug( 

3168 "skipping text between translation items/senses: " 

3169 "{}".format(node), 

3170 sortid="page/2621", 

3171 ) 

3172 

3173 # Main code of parse_translation(). We want ``sense`` to be assigned 

3174 # regardless of recursion levels, and thus the code is structured 

3175 # to define at this level and recurse in parse_translation_recurse(). 

3176 parse_translation_recurse(xlatnode) 

3177 

3178 def parse_etymology(data: WordData, node: LevelNode) -> None: 

3179 """Parses an etymology section.""" 

3180 assert isinstance(data, dict) 

3181 assert isinstance(node, WikiNode) 

3182 

3183 templates: list[TemplateData] = [] 

3184 

3185 # Counter for preventing the capture of etymology templates 

3186 # when we are inside templates that we want to ignore (i.e., 

3187 # not capture). 

3188 ignore_count = 0 

3189 

3190 def etym_template_fn(name: str, ht: TemplateArgs) -> Optional[str]: 

3191 nonlocal ignore_count 

3192 if is_panel_template(wxr, name) or name in ["zh-x", "zh-q"]: 

3193 return "" 

3194 if re.match(ignored_etymology_templates_re, name): 

3195 ignore_count += 1 

3196 return None 

3197 

3198 def etym_post_template_fn( 

3199 name: str, ht: TemplateArgs, expansion: str 

3200 ) -> None: 

3201 nonlocal ignore_count 

3202 if name in wikipedia_templates: 

3203 parse_wikipedia_template(wxr, data, ht) 

3204 return None 

3205 if re.match(ignored_etymology_templates_re, name): 

3206 ignore_count -= 1 

3207 return None 

3208 if ignore_count == 0: 3208 ↛ 3214line 3208 didn't jump to line 3214 because the condition on line 3208 was always true

3209 ht = clean_template_args(wxr, ht) 

3210 expansion = clean_node(wxr, None, expansion) 

3211 templates.append( 

3212 {"name": name, "args": ht, "expansion": expansion} 

3213 ) 

3214 return None 

3215 

3216 # Remove any subsections 

3217 contents = list( 

3218 x 

3219 for x in node.children 

3220 if not isinstance(x, WikiNode) or x.kind not in LEVEL_KINDS 

3221 ) 

3222 # Collect expanded links separately from templates. Generic linking 

3223 # templates such as m/l remain ignored in etymology_templates, but 

3224 # their destinations (and ordinary wikilinks) are still useful. 

3225 links: list[tuple[str, str]] = [] 

3226 # Convert to text, also capturing templates using post_template_fn 

3227 text = clean_node( 

3228 wxr, 

3229 None, 

3230 contents, 

3231 template_fn=etym_template_fn, 

3232 post_template_fn=etym_post_template_fn, 

3233 link_collector=links, 

3234 ).strip(": \n") # remove ":" indent wikitext before zh-x template 

3235 # Save the collected information. 

3236 if len(text) > 0: 

3237 data["etymology_text"] = text 

3238 if links: 

3239 data["etymology_links"] = links 

3240 if len(templates) > 0: 

3241 # Some etymology templates, like Template:root do not generate 

3242 # text, so they should be added here. Elsewhere, we check 

3243 # for Template:root and add some text to the expansion to please 

3244 # the validation. 

3245 data["etymology_templates"] = templates 

3246 

3247 for child_node in node.find_child_recursively( 3247 ↛ exitline 3247 didn't return from function 'parse_etymology' because the loop on line 3247 didn't complete

3248 LEVEL_KIND_FLAGS | NodeKind.TEMPLATE 

3249 ): 

3250 if child_node.kind in LEVEL_KIND_FLAGS: 

3251 break 

3252 elif isinstance( 3252 ↛ 3255line 3252 didn't jump to line 3255 because the condition on line 3252 was never true

3253 child_node, TemplateNode 

3254 ) and child_node.template_name in ["zh-x", "zh-q"]: 

3255 if "etymology_examples" not in data: 

3256 data["etymology_examples"] = [] 

3257 data["etymology_examples"].extend( 

3258 extract_template_zh_x( 

3259 wxr, child_node, None, ExampleData(raw_tags=[], tags=[]) 

3260 ) 

3261 ) 

3262 

3263 def process_children(treenode: WikiNode, pos: Optional[str]) -> None: 

3264 """This recurses into a subtree in the parse tree for a page.""" 

3265 nonlocal etym_data 

3266 nonlocal pos_data 

3267 nonlocal inside_level_four 

3268 

3269 redirect_list: list[str] = [] # for `zh-see` template 

3270 

3271 def skip_template_fn(name: str, ht: TemplateArgs) -> Optional[str]: 

3272 """This is called for otherwise unprocessed parts of the page. 

3273 We still expand them so that e.g. Category links get captured.""" 

3274 if name in wikipedia_templates: 

3275 data = select_data() 

3276 parse_wikipedia_template(wxr, data, ht) 

3277 return None 

3278 if is_panel_template(wxr, name): 

3279 return "" 

3280 return None 

3281 

3282 for node in treenode.children: 

3283 if not isinstance(node, WikiNode): 

3284 # print(" X{}".format(repr(node)[:40])) 

3285 continue 

3286 if isinstance(node, TemplateNode): 

3287 if process_soft_redirect_template(wxr, node, redirect_list): 

3288 continue 

3289 elif node.template_name == "zh-forms": 

3290 extract_zh_forms_template(wxr, node, select_data()) 

3291 elif ( 

3292 node.template_name.endswith("-kanjitab") 

3293 or node.template_name == "ja-kt" 

3294 ): 

3295 extract_ja_kanjitab_template(wxr, node, select_data()) 

3296 elif node.template_name in ETYMOLOGY_TEMPLATES_IN_HEADS: 

3297 args_ht = clean_template_args(wxr, node.template_parameters) 

3298 expansion = clean_node(wxr, etym_data, node) 

3299 etymology_template_append( 

3300 etym_data, node.template_name, args_ht, expansion 

3301 ) 

3302 

3303 if not isinstance(node, LevelNode): 

3304 # XXX handle e.g. wikipedia links at the top of a language 

3305 # XXX should at least capture "also" at top of page 

3306 if node.kind in ( 

3307 NodeKind.HLINE, 

3308 NodeKind.LIST, 

3309 NodeKind.LIST_ITEM, 

3310 ): 

3311 continue 

3312 # print(" UNEXPECTED: {}".format(node)) 

3313 # Clean the node to collect category links 

3314 clean_node(wxr, etym_data, node, template_fn=skip_template_fn) 

3315 continue 

3316 t = clean_node( 

3317 wxr, etym_data, node.sarg if node.sarg else node.largs 

3318 ) 

3319 t = t.lower() 

3320 # XXX these counts were never implemented fully, and even this 

3321 # gets discarded: Search STATISTICS_IMPLEMENTATION 

3322 wxr.config.section_counts[t] += 1 

3323 # print("PROCESS_CHILDREN: T:", repr(t)) 

3324 if t in IGNORED_TITLES: 

3325 pass 

3326 elif t.startswith(PRONUNCIATION_TITLE): 

3327 # Chinese Pronunciation section kludge; we demote these to 

3328 # be level 4 instead of 3 so that they're part of a larger 

3329 # etymology hierarchy; usually the data here is empty and 

3330 # acts as an inbetween between POS and Etymology data 

3331 if lang_code in ("zh",): 

3332 inside_level_four = True 

3333 if t.startswith(PRONUNCIATION_TITLE + " "): 

3334 # Pronunciation 1, etc, are used in Chinese Glyphs, 

3335 # and each of them may have senses under Definition 

3336 push_level_four_section(True) 

3337 wxr.wtp.start_subsection(None) 

3338 if wxr.config.capture_pronunciation: 3338 ↛ 3446line 3338 didn't jump to line 3446 because the condition on line 3338 was always true

3339 data = select_data() 

3340 parse_pronunciation( 

3341 wxr, 

3342 node, 

3343 data, 

3344 etym_data, 

3345 have_etym, 

3346 base_data, 

3347 lang_code, 

3348 ) 

3349 elif t.startswith(tuple(ETYMOLOGY_TITLES)): 

3350 push_etym() 

3351 wxr.wtp.start_subsection(None) 

3352 if wxr.config.capture_etymologies: 3352 ↛ 3446line 3352 didn't jump to line 3446 because the condition on line 3352 was always true

3353 m = re.search(r"\s(\d+(\.\d+)?)$", t) 

3354 if m: 

3355 etym_data["etymology_number"] = m.group(1) 

3356 parse_etymology(etym_data, node) 

3357 elif t == DESCENDANTS_TITLE and wxr.config.capture_descendants: 

3358 data = select_data() 

3359 extract_descendant_section(wxr, data, node, False) 

3360 elif ( 

3361 t in PROTO_ROOT_DERIVED_TITLES 

3362 and pos == "root" 

3363 and is_reconstruction 

3364 and wxr.config.capture_descendants 

3365 ): 

3366 data = select_data() 

3367 extract_descendant_section(wxr, data, node, True) 

3368 elif t == TRANSLATIONS_TITLE: 

3369 data = select_data() 

3370 parse_translations(data, node) 

3371 elif t in INFLECTION_TITLES: 

3372 parse_inflection(node, t, pos) 

3373 elif t == "alternative forms": 

3374 extract_alt_form_section(wxr, select_data(), node) 

3375 else: 

3376 lst = t.split() 

3377 while len(lst) > 1 and lst[-1].isdigit(): 

3378 lst = lst[:-1] 

3379 t_no_number = " ".join(lst).lower() 

3380 if t_no_number in POS_TITLES: 

3381 push_pos() 

3382 dt = POS_TITLES[t_no_number] # type:ignore[literal-required] 

3383 pos = dt["pos"] or "MISSING_POS" 

3384 wxr.wtp.start_subsection(t) 

3385 if "debug" in dt: 

3386 wxr.wtp.debug( 

3387 "{} in section {}".format(dt["debug"], t), 

3388 sortid="page/2755", 

3389 ) 

3390 if "warning" in dt: 3390 ↛ 3391line 3390 didn't jump to line 3391 because the condition on line 3390 was never true

3391 wxr.wtp.wiki_notice( 

3392 "{} in section {}".format(dt["warning"], t), 

3393 sortid="page/2759", 

3394 ) 

3395 if "error" in dt: 3395 ↛ 3396line 3395 didn't jump to line 3396 because the condition on line 3395 was never true

3396 wxr.wtp.error( 

3397 "{} in section {}".format(dt["error"], t), 

3398 sortid="page/2763", 

3399 ) 

3400 if "note" in dt: 3400 ↛ 3401line 3400 didn't jump to line 3401 because the condition on line 3400 was never true

3401 wxr.wtp.note( 

3402 "{} in section {}".format(dt["note"], t), 

3403 sortid="page/20251017a", 

3404 ) 

3405 if "wiki_notice" in dt: 3405 ↛ 3406line 3405 didn't jump to line 3406 because the condition on line 3405 was never true

3406 wxr.wtp.wiki_notice( 

3407 "{} in section {}".format(dt["wiki_notices"], t), 

3408 sortid="page/20251017b", 

3409 ) 

3410 # Parse word senses for the part-of-speech 

3411 parse_part_of_speech(node, pos) 

3412 if "tags" in dt: 

3413 for pdata in sense_datas: 

3414 data_extend(pdata, "tags", dt["tags"]) 

3415 elif t_no_number in LINKAGE_TITLES: 

3416 # print(f"LINKAGE_TITLES NODE {node=}") 

3417 rel = LINKAGE_TITLES[t_no_number] 

3418 data = select_data() 

3419 parse_linkage( 

3420 wxr, 

3421 data, 

3422 rel, 

3423 node, 

3424 word, 

3425 sense_datas, 

3426 is_reconstruction, 

3427 ) 

3428 elif t_no_number == COMPOUNDS_TITLE: 

3429 data = select_data() 

3430 if wxr.config.capture_compounds: 3430 ↛ 3446line 3430 didn't jump to line 3446 because the condition on line 3430 was always true

3431 parse_linkage( 

3432 wxr, 

3433 data, 

3434 "derived", 

3435 node, 

3436 word, 

3437 sense_datas, 

3438 is_reconstruction, 

3439 ) 

3440 

3441 # XXX parse interesting templates also from other sections. E.g., 

3442 # {{Letter|...}} in ===See also=== 

3443 # Also <gallery> 

3444 

3445 # Recurse to children of this node, processing subtitles therein 

3446 stack.append(t) 

3447 process_children(node, pos) 

3448 stack.pop() 

3449 

3450 if len(redirect_list) > 0: 

3451 if len(pos_data) > 0: 

3452 pos_data["redirects"] = redirect_list 

3453 if "pos" not in pos_data: 3453 ↛ 3454line 3453 didn't jump to line 3454 because the condition on line 3453 was never true

3454 pos_data["pos"] = "soft-redirect" 

3455 else: 

3456 new_page_data = copy.deepcopy(base_data) 

3457 new_page_data["redirects"] = redirect_list 

3458 if "pos" not in new_page_data: 3458 ↛ 3460line 3458 didn't jump to line 3460 because the condition on line 3458 was always true

3459 new_page_data["pos"] = "soft-redirect" 

3460 new_page_data["senses"] = [{"tags": ["no-gloss"]}] 

3461 page_datas.append(new_page_data) 

3462 

3463 def extract_examples( 

3464 others: list[WikiNode], sense_base: SenseData 

3465 ) -> list[ExampleData]: 

3466 """Parses through a list of definitions and quotes to find examples. 

3467 Returns a list of example dicts to be added to sense data. Adds 

3468 meta-data, mostly categories, into sense_base.""" 

3469 assert isinstance(others, list) 

3470 examples: list[ExampleData] = [] 

3471 

3472 for sub in others: 

3473 if not sub.sarg.endswith((":", "*")): 3473 ↛ 3474line 3473 didn't jump to line 3474 because the condition on line 3473 was never true

3474 continue 

3475 for item in sub.children: 

3476 if not isinstance(item, WikiNode): 3476 ↛ 3477line 3476 didn't jump to line 3477 because the condition on line 3476 was never true

3477 continue 

3478 if item.kind != NodeKind.LIST_ITEM: 3478 ↛ 3479line 3478 didn't jump to line 3479 because the condition on line 3478 was never true

3479 continue 

3480 usex_type = None 

3481 example_template_args = [] 

3482 example_template_names = [] 

3483 taxons = set() 

3484 

3485 # Bypass this function when parsing Chinese, Japanese and 

3486 # quotation templates. 

3487 new_example_lists = extract_example_list_item( 

3488 wxr, item, sense_base, ExampleData(raw_tags=[], tags=[]) 

3489 ) 

3490 if len(new_example_lists) > 0: 

3491 examples.extend(new_example_lists) 

3492 continue 

3493 

3494 def usex_template_fn( 

3495 name: str, ht: TemplateArgs 

3496 ) -> Optional[str]: 

3497 nonlocal usex_type 

3498 if is_panel_template(wxr, name): 

3499 return "" 

3500 if name in usex_templates: 

3501 usex_type = "example" 

3502 example_template_args.append(ht) 

3503 example_template_names.append(name) 

3504 elif name in quotation_templates: 

3505 usex_type = "quotation" 

3506 elif name in taxonomy_templates: 3506 ↛ 3507line 3506 didn't jump to line 3507 because the condition on line 3506 was never true

3507 taxons.update(ht.get(1, "").split()) 

3508 for prefix in template_linkages_to_ignore_in_examples: 

3509 if re.search( 

3510 r"(^|[-/\s]){}($|\b|[0-9])".format(prefix), name 

3511 ): 

3512 return "" 

3513 return None 

3514 

3515 # bookmark 

3516 ruby: list[tuple[str, str]] = [] 

3517 contents = item.children 

3518 if lang_code == "ja": 

3519 # Capture ruby contents if this is a Japanese language 

3520 # example. 

3521 # print(contents) 

3522 if ( 3522 ↛ 3527line 3522 didn't jump to line 3527 because the condition on line 3522 was never true

3523 contents 

3524 and isinstance(contents, str) 

3525 and re.match(r"\s*$", contents[0]) 

3526 ): 

3527 contents = contents[1:] 

3528 exp = wxr.wtp.parse( 

3529 wxr.wtp.node_to_wikitext(contents), 

3530 # post_template_fn=head_post_template_fn, 

3531 expand_all=True, 

3532 ) 

3533 rub, rest = extract_ruby(wxr, exp.children) 

3534 if rub: 

3535 for rtup in rub: 

3536 ruby.append(rtup) 

3537 contents = rest 

3538 subtext = clean_node( 

3539 wxr, sense_base, contents, template_fn=usex_template_fn 

3540 ) 

3541 

3542 frozen_taxons = frozenset(taxons) 

3543 classify_desc2 = partial(classify_desc, accepted=frozen_taxons) 

3544 

3545 # print(f"{subtext=}") 

3546 subtext = re.sub( 

3547 r"\s*\(please add an English " 

3548 r"translation of this " 

3549 r"(example|usage example|quote)\)", 

3550 "", 

3551 subtext, 

3552 ).strip() 

3553 subtext = re.sub(r"\^\([^)]*\)", "", subtext) 

3554 subtext = re.sub(r"\s*[―—]+$", "", subtext) 

3555 # print("subtext:", repr(subtext)) 

3556 

3557 lines = subtext.splitlines() 

3558 # print(lines) 

3559 

3560 lines = list(re.sub(r"^[#:*]*", "", x).strip() for x in lines) 

3561 lines = list( 

3562 x 

3563 for x in lines 

3564 if not re.match( 

3565 r"(Synonyms: |Antonyms: |Hyponyms: |" 

3566 r"Synonym: |Antonym: |Hyponym: |" 

3567 r"Hypernyms: |Derived terms: |" 

3568 r"Related terms: |" 

3569 r"Hypernym: |Derived term: |" 

3570 r"Coordinate terms:|" 

3571 r"Related term: |" 

3572 r"For more quotations using )", 

3573 x, 

3574 ) 

3575 ) 

3576 tr = "" 

3577 ref = "" 

3578 roman = "" 

3579 # for line in lines: 

3580 # print("LINE:", repr(line)) 

3581 # print(classify_desc(line)) 

3582 if len(lines) == 1 and lang_code != "en": 

3583 parts = example_splitter_re.split(lines[0]) 

3584 if ( 3584 ↛ 3592line 3584 didn't jump to line 3592 because the condition on line 3584 was never true

3585 len(parts) > 2 

3586 and len(example_template_args) == 1 

3587 and any( 

3588 ("―" in s) or ("—" in s) 

3589 for s in example_template_args[0].values() 

3590 ) 

3591 ): 

3592 if nparts := synch_splits_with_args( 

3593 lines[0], example_template_args[0] 

3594 ): 

3595 parts = nparts 

3596 if ( 3596 ↛ 3601line 3596 didn't jump to line 3601 because the condition on line 3596 was never true

3597 len(example_template_args) == 1 

3598 and "lit" in example_template_args[0] 

3599 ): 

3600 # ugly brute-force kludge in case there's a lit= arg 

3601 literally = example_template_args[0].get("lit", "") 

3602 if literally: 

3603 literally = ( 

3604 " (literally, “" 

3605 + clean_value(wxr, literally) 

3606 + "”)" 

3607 ) 

3608 else: 

3609 literally = "" 

3610 if ( 3610 ↛ 3649line 3610 didn't jump to line 3649 because the condition on line 3610 was never true

3611 len(example_template_args) == 1 

3612 and len(parts) == 2 

3613 and len(example_template_args[0]) 

3614 - ( 

3615 # horrible kludge to ignore these arguments 

3616 # when calculating how many there are 

3617 sum( 

3618 s in example_template_args[0] 

3619 for s in ( 

3620 "lit", # generates text, but we handle it 

3621 "inline", 

3622 "noenum", 

3623 "nocat", 

3624 "sort", 

3625 ) 

3626 ) 

3627 ) 

3628 == 3 

3629 and clean_value( 

3630 wxr, example_template_args[0].get(2, "") 

3631 ) 

3632 == parts[0].strip() 

3633 and clean_value( 

3634 wxr, 

3635 ( 

3636 example_template_args[0].get(3) 

3637 or example_template_args[0].get("translation") 

3638 or example_template_args[0].get("t", "") 

3639 ) 

3640 + literally, # in case there's a lit= argument 

3641 ) 

3642 == parts[1].strip() 

3643 ): 

3644 # {{exampletemplate|ex|Foo bar baz|English translation}} 

3645 # is a pretty reliable 'heuristic', so we use it here 

3646 # before the others. To be extra sure the template 

3647 # doesn't do anything weird, we compare the arguments 

3648 # and the output to each other. 

3649 lines = [parts[0].strip()] 

3650 tr = parts[1].strip() 

3651 elif ( 

3652 len(parts) == 2 

3653 and classify_desc2(parts[1]) in ENGLISH_TEXTS 

3654 ): 

3655 # These other branches just do some simple heuristics w/ 

3656 # the expanded output of the template (if applicable). 

3657 lines = [parts[0].strip()] 

3658 tr = parts[1].strip() 

3659 elif ( 3659 ↛ 3665line 3659 didn't jump to line 3665 because the condition on line 3659 was never true

3660 len(parts) == 3 

3661 and classify_desc2(parts[1]) 

3662 in ("romanization", "english") 

3663 and classify_desc2(parts[2]) in ENGLISH_TEXTS 

3664 ): 

3665 lines = [parts[0].strip()] 

3666 roman = parts[1].strip() 

3667 tr = parts[2].strip() 

3668 else: 

3669 parts = re.split(r"\s+-\s+", lines[0]) 

3670 if ( 3670 ↛ 3674line 3670 didn't jump to line 3674 because the condition on line 3670 was never true

3671 len(parts) == 2 

3672 and classify_desc2(parts[1]) in ENGLISH_TEXTS 

3673 ): 

3674 lines = [parts[0].strip()] 

3675 tr = parts[1].strip() 

3676 elif len(lines) > 1: 

3677 if any( 

3678 re.search(r"[]\d:)]\s*$", x) for x in lines[:-1] 

3679 ) and not (len(example_template_names) == 1): 

3680 refs: list[str] = [] 

3681 for i in range(len(lines)): 3681 ↛ 3687line 3681 didn't jump to line 3687 because the loop on line 3681 didn't complete

3682 if re.match(r"^[#*]*:+(\s*$|\s+)", lines[i]): 3682 ↛ 3683line 3682 didn't jump to line 3683 because the condition on line 3682 was never true

3683 break 

3684 refs.append(lines[i].strip()) 

3685 if re.search(r"[]\d:)]\s*$", lines[i]): 

3686 break 

3687 ref = " ".join(refs) 

3688 lines = lines[i + 1 :] 

3689 if ( 

3690 lang_code != "en" 

3691 and len(lines) >= 2 

3692 and classify_desc2(lines[-1]) in ENGLISH_TEXTS 

3693 ): 

3694 i = len(lines) - 1 

3695 while ( 3695 ↛ 3700line 3695 didn't jump to line 3700 because the condition on line 3695 was never true

3696 i > 1 

3697 and classify_desc2(lines[i - 1]) 

3698 in ENGLISH_TEXTS 

3699 ): 

3700 i -= 1 

3701 tr = "\n".join(lines[i:]) 

3702 lines = lines[:i] 

3703 if len(lines) >= 2: 

3704 if classify_desc2(lines[-1]) == "romanization": 

3705 roman = lines[-1].strip() 

3706 lines = lines[:-1] 

3707 

3708 elif lang_code == "en" and re.match(r"^[#*]*:+", lines[1]): 

3709 ref = lines[0] 

3710 lines = lines[1:] 

3711 elif lang_code != "en" and len(lines) == 2: 

3712 cls1 = classify_desc2(lines[0]) 

3713 cls2 = classify_desc2(lines[1]) 

3714 if cls2 in ENGLISH_TEXTS and cls1 != "english": 

3715 tr = lines[1] 

3716 lines = [lines[0]] 

3717 elif cls1 in ENGLISH_TEXTS and cls2 != "english": 3717 ↛ 3718line 3717 didn't jump to line 3718 because the condition on line 3717 was never true

3718 tr = lines[0] 

3719 lines = [lines[1]] 

3720 elif ( 3720 ↛ 3727line 3720 didn't jump to line 3727 because the condition on line 3720 was never true

3721 re.match(r"^[#*]*:+", lines[1]) 

3722 and classify_desc2( 

3723 re.sub(r"^[#*:]+\s*", "", lines[1]) 

3724 ) 

3725 in ENGLISH_TEXTS 

3726 ): 

3727 tr = re.sub(r"^[#*:]+\s*", "", lines[1]) 

3728 lines = [lines[0]] 

3729 elif cls1 == "english" and cls2 in ENGLISH_TEXTS: 

3730 # Both were classified as English, but 

3731 # presumably one is not. Assume first is 

3732 # non-English, as that seems more common. 

3733 tr = lines[1] 

3734 lines = [lines[0]] 

3735 elif ( 

3736 usex_type != "quotation" 

3737 and lang_code != "en" 

3738 and len(lines) == 3 

3739 ): 

3740 cls1 = classify_desc2(lines[0]) 

3741 cls2 = classify_desc2(lines[1]) 

3742 cls3 = classify_desc2(lines[2]) 

3743 if ( 

3744 cls3 == "english" 

3745 and cls2 in ("english", "romanization") 

3746 and cls1 != "english" 

3747 ): 

3748 tr = lines[2].strip() 

3749 roman = lines[1].strip() 

3750 lines = [lines[0].strip()] 

3751 elif ( 3751 ↛ 3759line 3751 didn't jump to line 3759 because the condition on line 3751 was never true

3752 usex_type == "quotation" 

3753 and lang_code != "en" 

3754 and len(lines) > 2 

3755 ): 

3756 # for x in lines: 

3757 # print(" LINE: {}: {}" 

3758 # .format(classify_desc2(x), x)) 

3759 if re.match(r"^[#*]*:+\s*$", lines[1]): 

3760 ref = lines[0] 

3761 lines = lines[2:] 

3762 cls1 = classify_desc2(lines[-1]) 

3763 if cls1 == "english": 

3764 i = len(lines) - 1 

3765 while ( 

3766 i > 1 

3767 and classify_desc2(lines[i - 1]) 

3768 == ENGLISH_TEXTS 

3769 ): 

3770 i -= 1 

3771 tr = "\n".join(lines[i:]) 

3772 lines = lines[:i] 

3773 

3774 roman = re.sub(r"[ \t\r]+", " ", roman).strip() 

3775 roman = re.sub(r"\[\s*…\s*\]", "[…]", roman) 

3776 tr = re.sub(r"^[#*:]+\s*", "", tr) 

3777 tr = re.sub(r"[ \t\r]+", " ", tr).strip() 

3778 tr = re.sub(r"\[\s*…\s*\]", "[…]", tr) 

3779 ref = re.sub(r"^[#*:]+\s*", "", ref) 

3780 ref = re.sub( 

3781 r", (volume |number |page )?“?" 

3782 r"\(please specify ([^)]|\(s\))*\)”?|" 

3783 ", text here$", 

3784 "", 

3785 ref, 

3786 ) 

3787 ref = re.sub(r"\[\s*…\s*\]", "[…]", ref) 

3788 lines = list(re.sub(r"^[#*:]+\s*", "", x) for x in lines) 

3789 subtext = "\n".join(x for x in lines if x) 

3790 if not tr and lang_code != "en": 

3791 m = re.search(r"([.!?])\s+\(([^)]+)\)\s*$", subtext) 

3792 if m and classify_desc2(m.group(2)) in ENGLISH_TEXTS: 3792 ↛ 3793line 3792 didn't jump to line 3793 because the condition on line 3792 was never true

3793 tr = m.group(2) 

3794 subtext = subtext[: m.start()] + m.group(1) 

3795 elif lines: 

3796 parts = re.split(r"\s*[―—]+\s*", lines[0]) 

3797 if ( 3797 ↛ 3801line 3797 didn't jump to line 3801 because the condition on line 3797 was never true

3798 len(parts) == 2 

3799 and classify_desc2(parts[1]) in ENGLISH_TEXTS 

3800 ): 

3801 subtext = parts[0].strip() 

3802 tr = parts[1].strip() 

3803 subtext = re.sub(r'^[“"`]([^“"`”\']*)[”"\']$', r"\1", subtext) 

3804 subtext = re.sub( 

3805 r"(please add an English translation of " 

3806 r"this (quote|usage example))", 

3807 "", 

3808 subtext, 

3809 ) 

3810 subtext = re.sub( 

3811 r"\s*→New International Version " "translation$", 

3812 "", 

3813 subtext, 

3814 ) # e.g. pis/Tok Pisin (Bible) 

3815 subtext = re.sub(r"[ \t\r]+", " ", subtext).strip() 

3816 subtext = re.sub(r"\[\s*…\s*\]", "[…]", subtext) 

3817 note = None 

3818 m = re.match(r"^\(([^)]*)\):\s+", subtext) 

3819 if ( 3819 ↛ 3827line 3819 didn't jump to line 3827 because the condition on line 3819 was never true

3820 m is not None 

3821 and lang_code != "en" 

3822 and ( 

3823 m.group(1).startswith("with ") 

3824 or classify_desc2(m.group(1)) == "english" 

3825 ) 

3826 ): 

3827 note = m.group(1) 

3828 subtext = subtext[m.end() :] 

3829 ref = re.sub(r"\s*\(→ISBN\)", "", ref) 

3830 ref = re.sub(r",\s*→ISBN", "", ref) 

3831 ref = ref.strip() 

3832 if ref.endswith(":") or ref.endswith(","): 

3833 ref = ref[:-1].strip() 

3834 ref = re.sub(r"\s+,\s+", ", ", ref) 

3835 ref = re.sub(r"\s+", " ", ref) 

3836 if ref and not subtext: 3836 ↛ 3837line 3836 didn't jump to line 3837 because the condition on line 3836 was never true

3837 subtext = ref 

3838 ref = "" 

3839 if subtext: 

3840 dt: ExampleData = {"text": subtext} 

3841 if ref: 

3842 dt["ref"] = ref 

3843 if tr: 

3844 dt["english"] = tr # DEPRECATED for "translation" 

3845 dt["translation"] = tr 

3846 if usex_type: 

3847 dt["type"] = usex_type 

3848 if note: 3848 ↛ 3849line 3848 didn't jump to line 3849 because the condition on line 3848 was never true

3849 dt["note"] = note 

3850 if roman: 

3851 dt["roman"] = roman 

3852 if ruby: 

3853 dt["ruby"] = ruby 

3854 examples.append(dt) 

3855 

3856 return examples 

3857 

3858 # Main code of parse_language() 

3859 # Process the section 

3860 stack.append(language) 

3861 process_children(langnode, None) 

3862 stack.pop() 

3863 

3864 # Finalize word entires 

3865 push_etym() 

3866 ret = [] 

3867 for data in page_datas: 

3868 merge_base(data, base_data) 

3869 ret.append(data) 

3870 

3871 # Copy all tags to word senses 

3872 for data in ret: 

3873 if "senses" not in data: 3873 ↛ 3874line 3873 didn't jump to line 3874 because the condition on line 3873 was never true

3874 continue 

3875 # WordData should not have a 'tags' field, but if it does, it's 

3876 # deleted and its contents removed and placed in each sense; 

3877 # that's why the type ignores. 

3878 tags: Iterable = data.get("tags", ()) # type: ignore[assignment] 

3879 if "tags" in data: 

3880 del data["tags"] # type: ignore[typeddict-item] 

3881 for sense in data["senses"]: 

3882 data_extend(sense, "tags", tags) 

3883 

3884 return ret 

3885 

3886 

3887def parse_wikipedia_template( 

3888 wxr: WiktextractContext, data: WordData, ht: TemplateArgs 

3889) -> None: 

3890 """Helper function for parsing {{wikipedia|...}} and related templates.""" 

3891 assert isinstance(wxr, WiktextractContext) 

3892 assert isinstance(data, dict) 

3893 assert isinstance(ht, dict) 

3894 langid = clean_node(wxr, data, ht.get("lang", ())) 

3895 pagename = ( 

3896 clean_node(wxr, data, ht.get(1, ())) 

3897 or wxr.wtp.title 

3898 or "MISSING_PAGE_TITLE" 

3899 ) 

3900 if langid: 

3901 data_append(data, "wikipedia", langid + ":" + pagename) 

3902 else: 

3903 data_append(data, "wikipedia", pagename) 

3904 

3905 

3906def parse_top_template( 

3907 wxr: WiktextractContext, node: WikiNode, data: WordData 

3908) -> None: 

3909 """Parses a template that occurs on the top-level in a page, before any 

3910 language subtitles.""" 

3911 assert isinstance(wxr, WiktextractContext) 

3912 assert isinstance(node, WikiNode) 

3913 assert isinstance(data, dict) 

3914 

3915 def top_template_fn(name: str, ht: TemplateArgs) -> Optional[str]: 

3916 if name in wikipedia_templates: 

3917 parse_wikipedia_template(wxr, data, ht) 

3918 return None 

3919 if is_panel_template(wxr, name): 

3920 return "" 

3921 if name in ("reconstruction",): 3921 ↛ 3922line 3921 didn't jump to line 3922 because the condition on line 3921 was never true

3922 return "" 

3923 if name.lower() == "also" or name.lower().startswith("also/"): 

3924 # XXX shows related words that might really have been the intended 

3925 # word, capture them 

3926 return "" 

3927 if name == "see also": 3927 ↛ 3929line 3927 didn't jump to line 3929 because the condition on line 3927 was never true

3928 # XXX capture 

3929 return "" 

3930 if name == "cardinalbox": 3930 ↛ 3932line 3930 didn't jump to line 3932 because the condition on line 3930 was never true

3931 # XXX capture 

3932 return "" 

3933 if name == "character info": 3933 ↛ 3935line 3933 didn't jump to line 3935 because the condition on line 3933 was never true

3934 # XXX capture 

3935 return "" 

3936 if name == "commonscat": 3936 ↛ 3938line 3936 didn't jump to line 3938 because the condition on line 3936 was never true

3937 # XXX capture link to Wikimedia commons 

3938 return "" 

3939 if name == "wrongtitle": 3939 ↛ 3942line 3939 didn't jump to line 3942 because the condition on line 3939 was never true

3940 # XXX this should be captured to replace page title with the 

3941 # correct title. E.g. ⿰亻革家 

3942 return "" 

3943 if name == "wikidata": 3943 ↛ 3944line 3943 didn't jump to line 3944 because the condition on line 3943 was never true

3944 arg = clean_node(wxr, data, ht.get(1, ())) 

3945 if arg.startswith("Q") or arg.startswith("Lexeme:L"): 

3946 data_append(data, "wikidata", arg) 

3947 return "" 

3948 wxr.wtp.debug( 

3949 "UNIMPLEMENTED top-level template: {} {}".format(name, ht), 

3950 sortid="page/2870", 

3951 ) 

3952 return "" 

3953 

3954 clean_node(wxr, None, [node], template_fn=top_template_fn) 

3955 

3956 

3957def fix_subtitle_hierarchy(wxr: WiktextractContext, text: str) -> str: 

3958 """Fix subtitle hierarchy to be strict Language -> Etymology -> 

3959 Part-of-Speech -> Translation/Linkage. Also merge Etymology sections 

3960 that are next to each other.""" 

3961 

3962 # Wiktextract issue #620, Chinese Glyph Origin before an etymology 

3963 # section get overwritten. In this case, let's just combine the two. 

3964 

3965 # In Chinese entries, Pronunciation can be preceded on the 

3966 # same level 3 by its Etymology *and* Glyph Origin sections: 

3967 # ===Glyph Origin=== 

3968 # ===Etymology=== 

3969 # ===Pronunciation=== 

3970 # Tatu suggested adding a new 'level' between 3 and 4, so Pronunciation 

3971 # is now Level 4, POS is shifted to Level 5 and the rest (incl. 'default') 

3972 # are now level 6 

3973 

3974 # Known lowercase PoS names are in part_of_speech_map 

3975 # Known lowercase linkage section names are in linkage_map 

3976 

3977 old = re.split( 

3978 r"(?m)^(==+)[ \t]*([^= \t]([^=\n]|=[^=])*?)" r"[ \t]*(==+)[ \t]*$", text 

3979 ) 

3980 

3981 parts = [] 

3982 npar = 4 # Number of parentheses in above expression 

3983 parts.append(old[0]) 

3984 prev_level = None 

3985 level = None 

3986 skip_level_title = False # When combining etymology sections 

3987 for i in range(1, len(old), npar + 1): 

3988 left = old[i] 

3989 right = old[i + npar - 1] 

3990 # remove Wikilinks in title 

3991 title = re.sub(r"^\[\[", "", old[i + 1]) 

3992 title = re.sub(r"\]\]$", "", title) 

3993 prev_level = level 

3994 level = len(left) 

3995 part = old[i + npar] 

3996 if level != len(right): 3996 ↛ 3997line 3996 didn't jump to line 3997 because the condition on line 3996 was never true

3997 wxr.wtp.debug( 

3998 "subtitle has unbalanced levels: " 

3999 "{!r} has {} on the left and {} on the right".format( 

4000 title, left, right 

4001 ), 

4002 sortid="page/2904", 

4003 ) 

4004 lc = title.lower() 

4005 if name_to_code(title, "en") != "": 

4006 if level > 2: 4006 ↛ 4007line 4006 didn't jump to line 4007 because the condition on line 4006 was never true

4007 wxr.wtp.debug( 

4008 "subtitle has language name {} at level {}".format( 

4009 title, level 

4010 ), 

4011 sortid="page/2911", 

4012 ) 

4013 level = 2 

4014 elif lc.startswith(tuple(ETYMOLOGY_TITLES)): 

4015 if level > 3: 4015 ↛ 4016line 4015 didn't jump to line 4016 because the condition on line 4015 was never true

4016 wxr.wtp.debug( 

4017 "etymology section {} at level {}".format(title, level), 

4018 sortid="page/2917", 

4019 ) 

4020 if prev_level == 3: # Two etymology (Glyph Origin + Etymology) 

4021 # sections cheek-to-cheek 

4022 skip_level_title = True 

4023 # Modify the title of previous ("Glyph Origin") section, in 

4024 # case we have a meaningful title like "Etymology 1" 

4025 parts[-2] = "{}{}{}".format("=" * level, title, "=" * level) 

4026 level = 3 

4027 elif lc.startswith(PRONUNCIATION_TITLE): 

4028 # Pronunciation is now a level between POS and Etymology, so 

4029 # we need to shift everything down by one 

4030 level = 4 

4031 elif lc in POS_TITLES: 

4032 level = 5 

4033 elif lc == TRANSLATIONS_TITLE: 

4034 level = 6 

4035 elif lc in LINKAGE_TITLES or lc == COMPOUNDS_TITLE: 

4036 level = 6 

4037 elif lc in INFLECTION_TITLES: 

4038 level = 6 

4039 elif lc == DESCENDANTS_TITLE: 

4040 level = 6 

4041 elif title in PROTO_ROOT_DERIVED_TITLES: 4041 ↛ 4042line 4041 didn't jump to line 4042 because the condition on line 4041 was never true

4042 level = 6 

4043 elif lc in IGNORED_TITLES: 

4044 level = 6 

4045 else: 

4046 level = 6 

4047 if skip_level_title: 

4048 skip_level_title = False 

4049 parts.append(part) 

4050 else: 

4051 parts.append("{}{}{}".format("=" * level, title, "=" * level)) 

4052 parts.append(part) 

4053 # print("=" * level, title) 

4054 # if level != len(left): 

4055 # print(" FIXED LEVEL OF {} {} -> {}" 

4056 # .format(title, len(left), level)) 

4057 

4058 text = "".join(parts) 

4059 # print(text) 

4060 return text 

4061 

4062 

4063def parse_page(wxr: WiktextractContext, word: str, text: str) -> list[WordData]: 

4064 # Skip translation pages 

4065 if word.endswith("/" + TRANSLATIONS_TITLE): 4065 ↛ 4066line 4065 didn't jump to line 4066 because the condition on line 4065 was never true

4066 return [] 

4067 

4068 if wxr.config.verbose: 4068 ↛ 4069line 4068 didn't jump to line 4069 because the condition on line 4068 was never true

4069 logger.info(f"Parsing page: {word}") 

4070 

4071 wxr.config.word = word 

4072 wxr.wtp.start_page(word) 

4073 

4074 # Remove <noinclude> and similar tags from main pages. They 

4075 # should not appear there, but at least net/Elfdala has one and it 

4076 # is probably not the only one. 

4077 text = re.sub(r"(?si)<(/)?noinclude\s*>", "", text) 

4078 text = re.sub(r"(?si)<(/)?onlyinclude\s*>", "", text) 

4079 text = re.sub(r"(?si)<(/)?includeonly\s*>", "", text) 

4080 

4081 # Fix up the subtitle hierarchy. There are hundreds if not thousands of 

4082 # pages that have, for example, Translations section under Linkage, or 

4083 # Translations section on the same level as Noun. Enforce a proper 

4084 # hierarchy by manipulating the subtitle levels in certain cases. 

4085 text = fix_subtitle_hierarchy(wxr, text) 

4086 

4087 # Parse the page, pre-expanding those templates that are likely to 

4088 # influence parsing 

4089 tree = wxr.wtp.parse( 

4090 text, 

4091 pre_expand=True, 

4092 additional_expand=ADDITIONAL_EXPAND_TEMPLATES, 

4093 do_not_pre_expand=DO_NOT_PRE_EXPAND_TEMPLATES, 

4094 ) 

4095 # from wikitextprocessor.parser import print_tree 

4096 # print("PAGE PARSE:", print_tree(tree)) 

4097 

4098 top_data: WordData = {} 

4099 

4100 # Iterate over top-level titles, which should be languages for normal 

4101 # pages 

4102 by_lang = defaultdict(list) 

4103 for langnode in tree.children: 

4104 if not isinstance(langnode, WikiNode): 

4105 continue 

4106 if langnode.kind == NodeKind.TEMPLATE: 

4107 parse_top_template(wxr, langnode, top_data) 

4108 continue 

4109 if langnode.kind == NodeKind.LINK: 

4110 # Some pages have links at top level, e.g., "trees" in Wiktionary 

4111 continue 

4112 if langnode.kind != NodeKind.LEVEL2: 4112 ↛ 4113line 4112 didn't jump to line 4113 because the condition on line 4112 was never true

4113 wxr.wtp.debug( 

4114 f"unexpected top-level node: {langnode}", sortid="page/3014" 

4115 ) 

4116 continue 

4117 lang = clean_node( 

4118 wxr, None, langnode.sarg if langnode.sarg else langnode.largs 

4119 ) 

4120 lang_code = name_to_code(lang, "en") 

4121 if lang_code == "": 4121 ↛ 4122line 4121 didn't jump to line 4122 because the condition on line 4121 was never true

4122 wxr.wtp.debug( 

4123 f"unrecognized language name: {lang}", sortid="page/3019" 

4124 ) 

4125 if ( 

4126 wxr.config.capture_language_codes 

4127 and lang_code not in wxr.config.capture_language_codes 

4128 ): 

4129 continue 

4130 wxr.wtp.start_section(lang) 

4131 

4132 # Collect all words from the page. 

4133 # print(f"{langnode=}") 

4134 datas = parse_language(wxr, langnode, lang, lang_code) 

4135 

4136 # Propagate fields resulting from top-level templates to this 

4137 # part-of-speech. 

4138 for data in datas: 

4139 if "lang" not in data: 4139 ↛ 4140line 4139 didn't jump to line 4140 because the condition on line 4139 was never true

4140 wxr.wtp.debug( 

4141 "internal error -- no lang in data: {}".format(data), 

4142 sortid="page/3034", 

4143 ) 

4144 continue 

4145 for k, v in top_data.items(): 

4146 assert isinstance(v, (list, tuple)) 

4147 data_extend(data, k, v) 

4148 by_lang[data["lang"]].append(data) 

4149 

4150 # XXX this code is clearly out of date. There is no longer a "conjugation" 

4151 # field. FIX OR REMOVE. 

4152 # Do some post-processing on the words. For example, we may distribute 

4153 # conjugation information to all the words. 

4154 ret = [] 

4155 for lang, lang_datas in by_lang.items(): 

4156 ret.extend(lang_datas) 

4157 

4158 for x in ret: 

4159 if x["word"] != word: 

4160 if word.startswith("Unsupported titles/"): 

4161 wxr.wtp.debug( 

4162 f"UNSUPPORTED TITLE: '{word}' -> '{x['word']}'", 

4163 sortid="20231101/3578page.py", 

4164 ) 

4165 else: 

4166 wxr.wtp.debug( 

4167 f"DIFFERENT ORIGINAL TITLE: '{word}' -> '{x['word']}'", 

4168 sortid="20231101/3582page.py", 

4169 ) 

4170 x["original_title"] = word 

4171 # validate tag data 

4172 recursively_separate_raw_tags(wxr, x) # type:ignore[arg-type] 

4173 return ret 

4174 

4175 

4176def recursively_separate_raw_tags( 

4177 wxr: WiktextractContext, data: dict[str, Any] 

4178) -> None: 

4179 if not isinstance(data, dict): 4179 ↛ 4180line 4179 didn't jump to line 4180 because the condition on line 4179 was never true

4180 wxr.wtp.error( 

4181 "'data' is not dict; most probably " 

4182 "data has a list that contains at least one dict and " 

4183 "at least one non-dict item", 

4184 sortid="en/page-4016/20240419", 

4185 ) 

4186 return 

4187 new_tags: list[str] = [] 

4188 raw_tags: list[str] = data.get("raw_tags", []) 

4189 for field, val in data.items(): 

4190 if field == "tags": 

4191 for tag in val: 

4192 if tag not in valid_tags: 

4193 raw_tags.append(tag) 

4194 else: 

4195 new_tags.append(tag) 

4196 if isinstance(val, list): 

4197 if len(val) > 0 and isinstance(val[0], dict): 

4198 for d in val: 

4199 recursively_separate_raw_tags(wxr, d) 

4200 if "tags" in data and not new_tags: 

4201 del data["tags"] 

4202 elif new_tags: 

4203 data["tags"] = new_tags 

4204 if raw_tags: 

4205 data["raw_tags"] = raw_tags 

4206 

4207 

4208def process_soft_redirect_template( 

4209 wxr: WiktextractContext, 

4210 template_node: TemplateNode, 

4211 redirect_pages: list[str], 

4212) -> bool: 

4213 # return `True` if the template is soft redirect template 

4214 if template_node.template_name == "zh-see": 

4215 # https://en.wiktionary.org/wiki/Template:zh-see 

4216 title = clean_node( 

4217 wxr, None, template_node.template_parameters.get(1, "") 

4218 ) 

4219 if title != "": 4219 ↛ 4221line 4219 didn't jump to line 4221 because the condition on line 4219 was always true

4220 redirect_pages.append(title) 

4221 return True 

4222 elif template_node.template_name in ["ja-see", "ja-see-kango"]: 

4223 # https://en.wiktionary.org/wiki/Template:ja-see 

4224 for key, value in template_node.template_parameters.items(): 

4225 if isinstance(key, int): 4225 ↛ 4224line 4225 didn't jump to line 4224 because the condition on line 4225 was always true

4226 title = clean_node(wxr, None, value) 

4227 if title != "": 4227 ↛ 4224line 4227 didn't jump to line 4224 because the condition on line 4227 was always true

4228 redirect_pages.append(title) 

4229 return True 

4230 return False 

4231 

4232 

4233ZH_FORMS_TAGS = { 

4234 "trad.": "Traditional-Chinese", 

4235 "simp.": "Simplified-Chinese", 

4236 "alternative forms": "alternative", 

4237 "2nd round simp.": "Second-Round-Simplified-Chinese", 

4238} 

4239 

4240 

4241def extract_zh_forms_template( 

4242 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordData 

4243): 

4244 # https://en.wiktionary.org/wiki/Template:zh-forms 

4245 lit_meaning = clean_node( 

4246 wxr, None, t_node.template_parameters.get("lit", "") 

4247 ) 

4248 if lit_meaning != "": 

4249 base_data["literal_meaning"] = lit_meaning 

4250 expanded_node = wxr.wtp.parse( 

4251 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

4252 ) 

4253 for table in expanded_node.find_child(NodeKind.TABLE): 

4254 for row in table.find_child(NodeKind.TABLE_ROW): 

4255 row_header = "" 

4256 row_header_tags: list[str] = [] 

4257 header_has_span = False 

4258 for cell in row.find_child( 

4259 NodeKind.TABLE_HEADER_CELL | NodeKind.TABLE_CELL 

4260 ): 

4261 if cell.kind == NodeKind.TABLE_HEADER_CELL: 

4262 row_header, row_header_tags, header_has_span = ( 

4263 extract_zh_forms_header_cell(wxr, base_data, cell) 

4264 ) 

4265 elif not header_has_span: 

4266 extract_zh_forms_data_cell( 

4267 wxr, base_data, cell, row_header, row_header_tags 

4268 ) 

4269 

4270 if "forms" in base_data and len(base_data["forms"]) == 0: 4270 ↛ 4271line 4270 didn't jump to line 4271 because the condition on line 4270 was never true

4271 del base_data["forms"] 

4272 

4273 

4274def extract_zh_forms_header_cell( 

4275 wxr: WiktextractContext, base_data: WordData, header_cell: WikiNode 

4276) -> tuple[str, list[str], bool]: 

4277 row_header = "" 

4278 row_header_tags = [] 

4279 header_has_span = False 

4280 first_span_index = len(header_cell.children) 

4281 for index, span_tag in header_cell.find_html("span", with_index=True): 

4282 if index < first_span_index: 4282 ↛ 4284line 4282 didn't jump to line 4284 because the condition on line 4282 was always true

4283 first_span_index = index 

4284 header_has_span = True 

4285 row_header = clean_node(wxr, None, header_cell.children[:first_span_index]) 

4286 for raw_tag in row_header.split(" and "): 

4287 raw_tag = raw_tag.strip() 

4288 if raw_tag != "": 

4289 row_header_tags.append(raw_tag) 

4290 for span_tag in header_cell.find_html_recursively("span"): 

4291 span_lang = span_tag.attrs.get("lang", "") 

4292 form_nodes = [] 

4293 sup_title = "" 

4294 for node in span_tag.children: 

4295 if isinstance(node, HTMLNode) and node.tag == "sup": 4295 ↛ 4296line 4295 didn't jump to line 4296 because the condition on line 4295 was never true

4296 for sup_span in node.find_html("span"): 

4297 sup_title = sup_span.attrs.get("title", "") 

4298 else: 

4299 form_nodes.append(node) 

4300 if span_lang in ["zh-Hant", "zh-Hans"]: 

4301 for word in clean_node(wxr, None, form_nodes).split("/"): 

4302 if word not in [wxr.wtp.title, ""]: 

4303 form = {"form": word} 

4304 for raw_tag in row_header_tags: 

4305 if raw_tag in ZH_FORMS_TAGS: 4305 ↛ 4308line 4305 didn't jump to line 4308 because the condition on line 4305 was always true

4306 data_append(form, "tags", ZH_FORMS_TAGS[raw_tag]) 

4307 else: 

4308 data_append(form, "raw_tags", raw_tag) 

4309 if sup_title != "": 4309 ↛ 4310line 4309 didn't jump to line 4310 because the condition on line 4309 was never true

4310 data_append(form, "raw_tags", sup_title) 

4311 data_append(base_data, "forms", form) 

4312 return row_header, row_header_tags, header_has_span 

4313 

4314 

4315TagLiteral = Literal["tags", "raw_tags"] 

4316TAG_LITERALS_TUPLE: tuple[TagLiteral, ...] = ("tags", "raw_tags") 

4317 

4318 

4319def extract_zh_forms_data_cell( 

4320 wxr: WiktextractContext, 

4321 base_data: WordData, 

4322 cell: WikiNode, 

4323 row_header: str, 

4324 row_header_tags: list[str], 

4325) -> None: 

4326 from .zh_pron_tags import ZH_PRON_TAGS 

4327 

4328 forms: list[FormData] = [] 

4329 for top_span_tag in cell.find_html("span"): 

4330 span_style = top_span_tag.attrs.get("style", "") 

4331 span_lang = top_span_tag.attrs.get("lang", "") 

4332 if span_style == "white-space:nowrap;": 

4333 extract_zh_forms_data_cell( 

4334 wxr, base_data, top_span_tag, row_header, row_header_tags 

4335 ) 

4336 elif "font-size:80%" in span_style: 

4337 raw_tag = clean_node(wxr, None, top_span_tag) 

4338 if raw_tag != "": 4338 ↛ 4329line 4338 didn't jump to line 4329 because the condition on line 4338 was always true

4339 for form in forms: 

4340 if raw_tag in ZH_PRON_TAGS: 4340 ↛ 4346line 4340 didn't jump to line 4346 because the condition on line 4340 was always true

4341 tr_tag = ZH_PRON_TAGS[raw_tag] 

4342 if isinstance(tr_tag, list): 4342 ↛ 4343line 4342 didn't jump to line 4343 because the condition on line 4342 was never true

4343 data_extend(form, "tags", tr_tag) 

4344 elif isinstance(tr_tag, str): 4344 ↛ 4339line 4344 didn't jump to line 4339 because the condition on line 4344 was always true

4345 data_append(form, "tags", tr_tag) 

4346 elif raw_tag in valid_tags: 

4347 data_append(form, "tags", raw_tag) 

4348 else: 

4349 data_append(form, "raw_tags", raw_tag) 

4350 elif span_lang in ["zh-Hant", "zh-Hans", "zh"]: 4350 ↛ 4329line 4350 didn't jump to line 4329 because the condition on line 4350 was always true

4351 word = clean_node(wxr, None, top_span_tag) 

4352 if word not in ["", "/", wxr.wtp.title]: 

4353 form = {"form": word} 

4354 if row_header != "anagram": 4354 ↛ 4360line 4354 didn't jump to line 4360 because the condition on line 4354 was always true

4355 for raw_tag in row_header_tags: 

4356 if raw_tag in ZH_FORMS_TAGS: 4356 ↛ 4359line 4356 didn't jump to line 4359 because the condition on line 4356 was always true

4357 data_append(form, "tags", ZH_FORMS_TAGS[raw_tag]) 

4358 else: 

4359 data_append(form, "raw_tags", raw_tag) 

4360 if span_lang == "zh-Hant": 

4361 data_append(form, "tags", "Traditional-Chinese") 

4362 elif span_lang == "zh-Hans": 

4363 data_append(form, "tags", "Simplified-Chinese") 

4364 forms.append(form) 

4365 

4366 if row_header == "anagram": 4366 ↛ 4367line 4366 didn't jump to line 4367 because the condition on line 4366 was never true

4367 for form in forms: 

4368 l_data: LinkageData = {"word": form["form"]} 

4369 for key in TAG_LITERALS_TUPLE: 

4370 if key in form: 

4371 l_data[key] = form[key] 

4372 data_append(base_data, "anagrams", l_data) 

4373 else: 

4374 data_extend(base_data, "forms", forms) 

4375 

4376 

4377def extract_ja_kanjitab_template( 

4378 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordData 

4379): 

4380 # https://en.wiktionary.org/wiki/Template:ja-kanjitab 

4381 expanded_node = wxr.wtp.parse( 

4382 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

4383 ) 

4384 for table in expanded_node.find_child(NodeKind.TABLE): 

4385 is_alt_form_table = False 

4386 for row in table.find_child(NodeKind.TABLE_ROW): 

4387 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL): 

4388 header_text = clean_node(wxr, None, header_node) 

4389 if header_text.startswith("Alternative spelling"): 

4390 is_alt_form_table = True 

4391 if not is_alt_form_table: 

4392 continue 

4393 forms = [] 

4394 for row in table.find_child(NodeKind.TABLE_ROW): 

4395 for cell_node in row.find_child(NodeKind.TABLE_CELL): 

4396 for child_node in cell_node.children: 

4397 if isinstance(child_node, HTMLNode): 

4398 if child_node.tag == "span": 

4399 word = clean_node(wxr, None, child_node) 

4400 if word != "": 4400 ↛ 4396line 4400 didn't jump to line 4396 because the condition on line 4400 was always true

4401 forms.append( 

4402 { 

4403 "form": word, 

4404 "tags": ["alternative", "kanji"], 

4405 } 

4406 ) 

4407 elif child_node.tag == "small": 

4408 raw_tag = clean_node(wxr, None, child_node).strip( 

4409 "()" 

4410 ) 

4411 if raw_tag != "" and len(forms) > 0: 4411 ↛ 4396line 4411 didn't jump to line 4396 because the condition on line 4411 was always true

4412 data_append( 

4413 forms[-1], 

4414 "tags" 

4415 if raw_tag in valid_tags 

4416 else "raw_tags", 

4417 raw_tag, 

4418 ) 

4419 data_extend(base_data, "forms", forms) 

4420 for link_node in expanded_node.find_child(NodeKind.LINK): 

4421 clean_node(wxr, base_data, link_node)