Coverage for src/wiktextract/extractor/el/etymology.py: 76%

39 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from typing import cast 

2 

3from wikitextprocessor import NodeKind, TemplateNode, WikiNode 

4from wikitextprocessor.parser import LEVEL_KIND_FLAGS 

5 

6from wiktextract import WiktextractContext 

7from wiktextract.page import clean_node 

8 

9from .models import WordEntry 

10from .parse_utils import ( 

11 POSReturns, 

12 find_sections, 

13) 

14from .pos import extract_alt_form_templates 

15from .pronunciation import process_pron 

16from .section_titles import Heading, POSName 

17 

18 

19def process_etym( 

20 wxr: WiktextractContext, 

21 base_data: WordEntry, 

22 node: WikiNode, 

23 title: str, 

24 num: int, 

25) -> tuple[int, POSReturns]: 

26 """Extract etymological data from section and process POS children.""" 

27 # Get everything except subsections, which we assume are POS nodes. 

28 etym_contents = list(node.invert_find_child(LEVEL_KIND_FLAGS)) 

29 etym_sublevels = list(node.find_child(LEVEL_KIND_FLAGS)) 

30 ret_etym_sublevels: POSReturns = [] 

31 

32 wxr.wtp.start_subsection(title) 

33 

34 section_num = num 

35 

36 # Extract form_of data 

37 for i, t_node in enumerate(etym_contents): 

38 if isinstance(t_node, TemplateNode): 

39 extract_alt_form_templates(wxr, base_data, t_node, etym_contents, i) 

40 if isinstance(t_node, WikiNode) and t_node.kind == NodeKind.LIST: 

41 for l_item in t_node.find_child_recursively(NodeKind.LIST_ITEM): 

42 for j, l_node in enumerate(l_item.children): 

43 if isinstance(l_node, TemplateNode): 

44 extract_alt_form_templates( 

45 wxr, base_data, l_node, l_item.children, j 

46 ) 

47 

48 # Greek wiktionary doesn't seem to have etymology templates, or at 

49 # least they're not used as much. 

50 links: list[tuple[str, str]] = [] 

51 etym_text = ( 

52 clean_node(wxr, base_data, etym_contents, link_collector=links) 

53 .lstrip(":#") 

54 .strip() 

55 ) 

56 

57 if etym_text: 57 ↛ 61line 57 didn't jump to line 61 because the condition on line 57 was always true

58 base_data.etymology_text = etym_text 

59 base_data.etymology_links = links 

60 

61 for heading_type, pos, title, tags, num, subnode in find_sections( 61 ↛ 64line 61 didn't jump to line 64 because the loop on line 61 never started

62 wxr, etym_sublevels 

63 ): 

64 if heading_type == Heading.POS: 

65 section_num = num if num > section_num else section_num 

66 # SAFETY: Since the heading_type is POS, find_sections 

67 # "pos_or_section" is guaranteed to be a pos: POSName 

68 pos = cast(POSName, pos) 

69 ret_etym_sublevels.append( 

70 ( 

71 pos, 

72 title, 

73 tags, 

74 num, 

75 subnode, 

76 base_data.model_copy(deep=True), 

77 ) 

78 ) 

79 elif heading_type == Heading.Pron: 

80 section_num = num if num > section_num else section_num 

81 

82 num, pron_sublevels = process_pron( 

83 wxr, subnode, base_data, title, section_num 

84 ) 

85 

86 ret_etym_sublevels.extend(pron_sublevels) 

87 

88 return section_num, ret_etym_sublevels