Coverage for src/wiktextract/extractor/zh/etymology.py: 58%

73 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from wikitextprocessor import ( 

2 HTMLNode, 

3 LevelNode, 

4 NodeKind, 

5 TemplateNode, 

6 WikiNode, 

7) 

8 

9from ...page import clean_node 

10from ...wxr_context import WiktextractContext 

11from .models import Example, Form, WordEntry 

12from .tags import translate_raw_tags 

13 

14 

15def extract_etymology_section( 

16 wxr: WiktextractContext, 

17 page_data: list[WordEntry], 

18 base_data: WordEntry, 

19 level_node: WikiNode, 

20): 

21 from .example import extract_template_zh_x 

22 

23 e_nodes = [] 

24 for node in level_node.children: 

25 if isinstance(node, TemplateNode) and node.template_name in [ 

26 "zh-x", 

27 "zh-q", 

28 ]: 

29 for example_data in extract_template_zh_x( 

30 wxr, node, Example(text="") 

31 ): 

32 base_data.etymology_examples.append(example_data) 

33 clean_node(wxr, base_data, node) 

34 elif isinstance(node, TemplateNode) and node.template_name.lower() in [ 34 ↛ 41line 34 didn't jump to line 41 because the condition on line 34 was never true

35 "rfe", # missing etymology 

36 "zh-forms", 

37 "zh-wp", 

38 "wp", 

39 "wikipedia", 

40 ]: 

41 continue 

42 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST: 

43 has_zh_x = False 

44 for template_node in node.find_child_recursively(NodeKind.TEMPLATE): 

45 if template_node.template_name in ["zh-x", "zh-q"]: 

46 has_zh_x = True 

47 for example_data in extract_template_zh_x( 

48 wxr, template_node, Example(text="") 

49 ): 

50 base_data.etymology_examples.append(example_data) 

51 clean_node(wxr, base_data, template_node) 

52 if not has_zh_x: 

53 for list_item in node.find_child(NodeKind.LIST_ITEM): 

54 e_links: list[tuple[str, str]] = [] 

55 e_text = clean_node( 

56 wxr, None, list_item.children, link_collector=e_links 

57 ) 

58 if len(e_text) > 0: 58 ↛ 53line 58 didn't jump to line 53 because the condition on line 58 was always true

59 base_data.etymology_texts.append(e_text) 

60 base_data.etymology_links.extend(e_links) 

61 elif isinstance(node, TemplateNode) and node.template_name in [ 61 ↛ 66line 61 didn't jump to line 66 because the condition on line 61 was never true

62 "ja-see", 

63 "ja-see-kango", 

64 "zh-see", 

65 ]: 

66 from .page import process_soft_redirect_template 

67 

68 page_data.append(base_data.model_copy(deep=True)) 

69 process_soft_redirect_template(wxr, node, page_data[-1]) 

70 elif isinstance(node, TemplateNode) and ( 

71 node.template_name.endswith("-kanjitab") 

72 or node.template_name == "ja-kt" 

73 ): 

74 extract_ja_kanjitab_template(wxr, node, base_data) 

75 elif isinstance(node, LevelNode): 

76 break 

77 else: 

78 e_nodes.append(node) 

79 

80 if len(e_nodes) > 0: 80 ↛ exitline 80 didn't return from function 'extract_etymology_section' because the condition on line 80 was always true

81 e_links: list[tuple[str, str]] = [] 

82 etymology_text = clean_node( 

83 wxr, base_data, e_nodes, link_collector=e_links 

84 ) 

85 if len(etymology_text) > 0: 

86 base_data.etymology_texts.append(etymology_text) 

87 base_data.etymology_links.extend(e_links) 

88 

89 

90def extract_ja_kanjitab_template( 

91 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry 

92): 

93 # https://zh.wiktionary.org/wiki/Template:ja-kanjitab 

94 expanded_node = wxr.wtp.parse( 

95 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

96 ) 

97 for table in expanded_node.find_child(NodeKind.TABLE): 97 ↛ 98line 97 didn't jump to line 98 because the loop on line 97 never started

98 is_alt_form_table = False 

99 for row in table.find_child(NodeKind.TABLE_ROW): 

100 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL): 

101 header_text = clean_node(wxr, None, header_node) 

102 if header_text == "其他表記": 

103 is_alt_form_table = True 

104 if not is_alt_form_table: 

105 continue 

106 forms = [] 

107 for row in table.find_child(NodeKind.TABLE_ROW): 

108 for cell_node in row.find_child(NodeKind.TABLE_CELL): 

109 for child_node in cell_node.children: 

110 if isinstance(child_node, HTMLNode): 

111 if child_node.tag == "span": 

112 word = clean_node(wxr, None, child_node) 

113 if word != "": 

114 forms.append( 

115 Form( 

116 form=word, tags=["alternative", "kanji"] 

117 ) 

118 ) 

119 elif child_node.tag == "small": 

120 raw_tag = clean_node(wxr, None, child_node).strip( 

121 "()" 

122 ) 

123 if raw_tag != "" and len(forms) > 0: 

124 forms[-1].raw_tags.append(raw_tag) 

125 translate_raw_tags(forms[-1]) 

126 base_data.forms.extend(forms) 

127 for link_node in expanded_node.find_child(NodeKind.LINK): 

128 clean_node(wxr, base_data, link_node)