Coverage for src/wiktextract/extractor/simple/etymology.py: 65%
30 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from wikitextprocessor import WikiNode
2from wikitextprocessor.core import TemplateArgs
3from wikitextprocessor.parser import LEVEL_KIND_FLAGS
5from wiktextract import WiktextractContext
6from wiktextract.clean import clean_value
7from wiktextract.page import clean_node
9from .models import TemplateData, WordEntry
10from .parse_utils import ETYMOLOGY_TEMPLATES, PANEL_TEMPLATES
13def process_etym(
14 wxr: WiktextractContext,
15 node: WikiNode,
16 target_data: WordEntry,
17) -> None:
18 """Extract etymological data from section."""
19 # Get everything except subsections.
20 etym_nodes = list(node.invert_find_child(LEVEL_KIND_FLAGS))
21 etym_templates = []
23 # A post-template_fn already has the expanded string from the template.
24 # If we'd want to suppress something without bothering to expand it,
25 # we can use a normal template_fn and return an empty string there.
26 # But returning an empty string here will also do the same thing.
27 # Returning None means "use whatever we expanded", default behavior.
28 def post_etym_template_fn(
29 name: str, ht: TemplateArgs, expanded: str
30 ) -> str | None:
31 lname = name.lower().strip()
32 if lname in PANEL_TEMPLATES: 32 ↛ 33line 32 didn't jump to line 33 because the condition on line 32 was never true
33 return ""
34 if lname in ETYMOLOGY_TEMPLATES: 34 ↛ 37line 34 didn't jump to line 37 because the condition on line 34 was never true
35 # there are a bunch of clean_ functions: clean_value is to remove
36 # italics and html tags and stuff like that from strings.
37 expanded = clean_value(wxr, expanded)
38 new_args = {}
39 for k, v in ht.items():
40 new_args[str(k)] = str(v)
41 tdata = TemplateData(name=name, args=new_args, expansion=expanded)
42 etym_templates.append(tdata)
43 return None
45 links: list[tuple[str, str]] = []
46 etym_text = clean_node(
47 wxr,
48 target_data,
49 etym_nodes,
50 post_template_fn=post_etym_template_fn,
51 link_collector=links,
52 )
54 if etym_text: 54 ↛ exitline 54 didn't return from function 'process_etym' because the condition on line 54 was always true
55 target_data.etymology_text = etym_text
56 target_data.etymology_links = links
57 if etym_templates: 57 ↛ 58line 57 didn't jump to line 58 because the condition on line 57 was never true
58 target_data.etymology_templates = etym_templates
60 # logprint = []
61 # for t in etym_templates:
62 # logprint.append(f"=== {t.name}")
63 # lp = '\n'.join(logprint)
64 # logger.info(f"{wxr.wtp.title}\n{lp}")