Coverage for src/wiktextract/extractor/ko/etymology.py: 54%

60 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1from wikitextprocessor import HTMLNode, LevelNode, NodeKind, TemplateNode 

2 

3from ...page import clean_node 

4from ...wxr_context import WiktextractContext 

5from .models import Form, WordEntry 

6from .tags import translate_raw_tags 

7 

8 

9def extract_etymology_section( 

10 wxr: WiktextractContext, word_entry: WordEntry, level_node: LevelNode 

11) -> None: 

12 if len(word_entry.etymology_texts) > 0: 

13 word_entry.etymology_texts.clear() 

14 word_entry.etymology_links.clear() 

15 word_entry.categories.clear() 

16 

17 has_list = False 

18 for list_node in level_node.find_child(NodeKind.LIST): 

19 has_list = True 

20 for list_item in list_node.find_child(NodeKind.LIST_ITEM): 

21 e_links: list[tuple[str, str]] = [] 

22 text = clean_node( 

23 wxr, word_entry, list_item.children, link_collector=e_links 

24 ) 

25 if len(text) > 0: 25 ↛ 20line 25 didn't jump to line 20 because the condition on line 25 was always true

26 word_entry.etymology_texts.append(text) 

27 word_entry.etymology_links.extend(e_links) 

28 

29 if not has_list: 

30 e_nodes = [] 

31 for node in level_node.children: 

32 if isinstance(node, TemplateNode) and ( 

33 node.template_name.endswith("-kanjitab") 

34 or node.template_name == "ja-kt" 

35 ): 

36 extract_ja_kanjitab_template(wxr, node, word_entry) 

37 elif isinstance(node, LevelNode): 

38 break 

39 else: 

40 e_nodes.append(node) 

41 

42 e_links: list[tuple[str, str]] = [] 

43 text = clean_node(wxr, word_entry, e_nodes, link_collector=e_links) 

44 if len(text) > 0: 

45 word_entry.etymology_texts.append(text) 

46 word_entry.etymology_links.extend(e_links) 

47 

48 

49def extract_ja_kanjitab_template( 

50 wxr: WiktextractContext, t_node: TemplateNode, base_data: WordEntry 

51): 

52 expanded_node = wxr.wtp.parse( 

53 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

54 ) 

55 for table in expanded_node.find_child(NodeKind.TABLE): 55 ↛ 56line 55 didn't jump to line 56 because the loop on line 55 never started

56 is_alt_form_table = False 

57 for row in table.find_child(NodeKind.TABLE_ROW): 

58 for header_node in row.find_child(NodeKind.TABLE_HEADER_CELL): 

59 header_text = clean_node(wxr, None, header_node) 

60 if header_text == "다른 표기": 

61 is_alt_form_table = True 

62 if not is_alt_form_table: 

63 continue 

64 forms = [] 

65 for row in table.find_child(NodeKind.TABLE_ROW): 

66 for cell_node in row.find_child(NodeKind.TABLE_CELL): 

67 for child_node in cell_node.children: 

68 if isinstance(child_node, HTMLNode): 

69 if child_node.tag == "span": 

70 word = clean_node(wxr, None, child_node) 

71 if word != "": 

72 forms.append( 

73 Form( 

74 form=word, tags=["alternative", "kanji"] 

75 ) 

76 ) 

77 elif child_node.tag == "small": 

78 raw_tag = clean_node(wxr, None, child_node).strip( 

79 "()" 

80 ) 

81 if raw_tag != "" and len(forms) > 0: 

82 forms[-1].raw_tags.append(raw_tag) 

83 translate_raw_tags(forms[-1]) 

84 base_data.forms.extend(forms) 

85 for link_node in expanded_node.find_child(NodeKind.LINK): 

86 clean_node(wxr, base_data, link_node)