Coverage for src/wiktextract/extractor/th/etymology.py: 48%

53 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from wikitextprocessor import ( 

2 HTMLNode, 

3 LevelNode, 

4 NodeKind, 

5 TemplateNode, 

6 WikiNode, 

7) 

8 

9from ...page import clean_node 

10from ...wxr_context import WiktextractContext 

11from .models import Form, WordEntry 

12from .tags import translate_raw_tags 

13 

14 

15def extract_etymology_section( 

16 wxr: WiktextractContext, base_data: WordEntry, level_node: LevelNode 

17): 

18 e_nodes = [] 

19 for node in level_node.children: 

20 if isinstance(node, TemplateNode) and ( 

21 node.template_name.endswith("-kanjitab") 

22 or node.template_name == "ja-kt" 

23 ): 

24 extract_ja_kanjitab_template(wxr, node, base_data) 

25 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST: 

26 for list_item in node.find_child(NodeKind.LIST_ITEM): 

27 e_links: list[tuple[str, str]] = [] 

28 e_text = clean_node( 

29 wxr, base_data, list_item.children, link_collector=e_links 

30 ) 

31 if e_text != "": 31 ↛ 26line 31 didn't jump to line 26 because the condition on line 31 was always true

32 base_data.etymology_texts.append(e_text) 

33 base_data.etymology_links.extend(e_links) 

34 elif not ( 

35 isinstance(node, LevelNode) 

36 or ( 

37 isinstance(node, TemplateNode) 

38 and node.template_name in ["ja-see", "ja-see-kango"] 

39 ) 

40 ): 

41 e_nodes.append(node) 

42 

43 if len(e_nodes) > 0: 43 ↛ exitline 43 didn't return from function 'extract_etymology_section' because the condition on line 43 was always true

44 e_links: list[tuple[str, str]] = [] 

45 e_str = clean_node(wxr, base_data, e_nodes, link_collector=e_links) 

46 if e_str != "": 

47 base_data.etymology_texts.append(e_str) 

48 base_data.etymology_links.extend(e_links) 

49 

50 

51def extract_ja_kanjitab_template( 

52 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry 

53): 

54 # https://th.wiktionary.org/wiki/Template:ja-kanjitab 

55 expanded_node = wxr.wtp.parse( 

56 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

57 ) 

58 for table in expanded_node.find_child(NodeKind.TABLE): 58 ↛ 59line 58 didn't jump to line 59 because the loop on line 58 never started

59 is_alt_form_table = False 

60 for row in table.find_child(NodeKind.TABLE_ROW): 

61 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL): 

62 header_text = clean_node(wxr, None, header_node) 

63 if header_text.startswith("การสะกดแบบอื่น"): 

64 is_alt_form_table = True 

65 if not is_alt_form_table: 

66 continue 

67 forms = [] 

68 for row in table.find_child(NodeKind.TABLE_ROW): 

69 for cell_node in row.find_child(NodeKind.TABLE_CELL): 

70 for child_node in cell_node.children: 

71 if isinstance(child_node, HTMLNode): 

72 if child_node.tag == "span": 

73 word = clean_node(wxr, None, child_node) 

74 if word != "": 

75 forms.append( 

76 Form( 

77 form=word, tags=["alternative", "kanji"] 

78 ) 

79 ) 

80 elif child_node.tag == "small": 

81 raw_tag = clean_node(wxr, None, child_node).strip( 

82 "()" 

83 ) 

84 if raw_tag != "" and len(forms) > 0: 

85 forms[-1].raw_tags.append(raw_tag) 

86 translate_raw_tags(forms[-1]) 

87 base_data.forms.extend(forms) 

88 for link_node in expanded_node.find_child(NodeKind.LINK): 

89 clean_node(wxr, base_data, link_node)