Coverage for src/wiktextract/extractor/el/etymology.py: 76%
39 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from typing import cast
3from wikitextprocessor import NodeKind, TemplateNode, WikiNode
4from wikitextprocessor.parser import LEVEL_KIND_FLAGS
6from wiktextract import WiktextractContext
7from wiktextract.page import clean_node
9from .models import WordEntry
10from .parse_utils import (
11 POSReturns,
12 find_sections,
13)
14from .pos import extract_alt_form_templates
15from .pronunciation import process_pron
16from .section_titles import Heading, POSName
19def process_etym(
20 wxr: WiktextractContext,
21 base_data: WordEntry,
22 node: WikiNode,
23 title: str,
24 num: int,
25) -> tuple[int, POSReturns]:
26 """Extract etymological data from section and process POS children."""
27 # Get everything except subsections, which we assume are POS nodes.
28 etym_contents = list(node.invert_find_child(LEVEL_KIND_FLAGS))
29 etym_sublevels = list(node.find_child(LEVEL_KIND_FLAGS))
30 ret_etym_sublevels: POSReturns = []
32 wxr.wtp.start_subsection(title)
34 section_num = num
36 # Extract form_of data
37 for i, t_node in enumerate(etym_contents):
38 if isinstance(t_node, TemplateNode):
39 extract_alt_form_templates(wxr, base_data, t_node, etym_contents, i)
40 if isinstance(t_node, WikiNode) and t_node.kind == NodeKind.LIST:
41 for l_item in t_node.find_child_recursively(NodeKind.LIST_ITEM):
42 for j, l_node in enumerate(l_item.children):
43 if isinstance(l_node, TemplateNode):
44 extract_alt_form_templates(
45 wxr, base_data, l_node, l_item.children, j
46 )
48 # Greek wiktionary doesn't seem to have etymology templates, or at
49 # least they're not used as much.
50 links: list[tuple[str, str]] = []
51 etym_text = (
52 clean_node(wxr, base_data, etym_contents, link_collector=links)
53 .lstrip(":#")
54 .strip()
55 )
57 if etym_text: 57 ↛ 61line 57 didn't jump to line 61 because the condition on line 57 was always true
58 base_data.etymology_text = etym_text
59 base_data.etymology_links = links
61 for heading_type, pos, title, tags, num, subnode in find_sections( 61 ↛ 64line 61 didn't jump to line 64 because the loop on line 61 never started
62 wxr, etym_sublevels
63 ):
64 if heading_type == Heading.POS:
65 section_num = num if num > section_num else section_num
66 # SAFETY: Since the heading_type is POS, find_sections
67 # "pos_or_section" is guaranteed to be a pos: POSName
68 pos = cast(POSName, pos)
69 ret_etym_sublevels.append(
70 (
71 pos,
72 title,
73 tags,
74 num,
75 subnode,
76 base_data.model_copy(deep=True),
77 )
78 )
79 elif heading_type == Heading.Pron:
80 section_num = num if num > section_num else section_num
82 num, pron_sublevels = process_pron(
83 wxr, subnode, base_data, title, section_num
84 )
86 ret_etym_sublevels.extend(pron_sublevels)
88 return section_num, ret_etym_sublevels