Coverage for src/wiktextract/extractor/en/pronunciation.py: 83%
878 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1import hashlib
2import re
3import urllib
4from copy import deepcopy
5from dataclasses import dataclass
6from typing import Iterator, Literal, NamedTuple
8from wikitextprocessor import (
9 HTMLNode,
10 LevelNode,
11 NodeKind,
12 TemplateNode,
13 WikiNode,
14)
16from ...clean import clean_value
17from ...datautils import data_append, data_extend, split_at_comma_semi
18from ...page import LEVEL_KINDS, clean_node, is_panel_template
19from ...tags import valid_tags
20from ...wxr_context import WiktextractContext
21from ..share import create_audio_url_dict
22from .form_descriptions import (
23 classify_desc,
24 decode_tags,
25 parse_pronunciation_tags,
26)
27from .parts_of_speech import part_of_speech_map
28from .type_utils import Hyphenation, SoundData, TemplateArgs, WordData
30PronunciationPoses = tuple[str, ...]
32# Prefixes, tags, and regexp for finding romanizations from the pronuncation
33# section
34pron_romanizations = {
35 " Revised Romanization ": "romanization revised",
36 " Revised Romanization (translit.) ": (
37 "romanization revised transliteration"
38 ),
39 " McCune-Reischauer ": "McCune-Reischauer romanization",
40 " McCune–Reischauer ": "McCune-Reischauer romanization",
41 " Yale Romanization ": "Yale romanization",
42}
43pron_romanization_re = re.compile(
44 "(?m)^("
45 + "|".join(
46 re.escape(x)
47 for x in sorted(pron_romanizations.keys(), key=len, reverse=True)
48 )
49 + ")([^\n]+)"
50)
52IPA_EXTRACT = r"^(\((.+)\) )?(IPA⁽ᵏᵉʸ⁾|enPR): ((.+?)( \(([^(]+)\))?\s*)$"
53IPA_EXTRACT_RE = re.compile(IPA_EXTRACT)
56class PronunciationPosMatch(NamedTuple):
57 pos_values: PronunciationPoses
58 residual: str
61class PronunciationPosPrefix(NamedTuple):
62 pos_values: PronunciationPoses
63 text: str
64 is_persistent: bool
67class FlattenedListNode(NamedTuple):
68 node: WikiNode | str
69 list_depth: int
72PRON_POS_TEMPLATE_NAMES = {
73 "q",
74 "qualifier",
75 "qual",
76 "i",
77 "sense",
78 "a",
79 "accent",
80 "lb",
81 "lbl",
82 "label",
83}
85PRON_POS_BY_LABEL = {
86 label: pos_data["pos"] for label, pos_data in part_of_speech_map.items()
87}
89PRON_POS_LABEL_RE = re.compile(
90 r"^(?:(?P<residual>.+?)\s+)?(?P<label>"
91 + "|".join(
92 re.escape(label)
93 for label in sorted(PRON_POS_BY_LABEL, key=len, reverse=True)
94 )
95 + r")$"
96)
99def normalize_pronunciation_pos_label(label: str) -> str:
100 label = label.strip().lower()
101 label = re.sub(r"\s+", " ", label)
102 # Drop explanatory suffixes such as "noun (barren areas)" before
103 # matching the label against part-of-speech names.
104 label = re.sub(r"\s*\([^)]*\)\s*$", "", label).strip()
105 label = label.strip(" \t\n\r():")
106 label = re.sub(r"\s+senses?$", "", label).strip()
107 return label
110def split_pronunciation_pos_text(text: str) -> PronunciationPosMatch:
111 pos_values: list[str] = []
112 residual: list[str] = []
113 for part in split_pronunciation_pos_parts(text):
114 pos, residual_part = pronunciation_pos_from_part(part)
115 if pos:
116 # POS-bearing qualifier text may also contain normal pronunciation
117 # tags before the POS label, e.g. "attributive adjective".
118 if pos not in pos_values: 118 ↛ 120line 118 didn't jump to line 120 because the condition on line 118 was always true
119 pos_values.append(pos)
120 if residual_part:
121 residual.append(residual_part)
122 elif residual_part: 122 ↛ 113line 122 didn't jump to line 113 because the condition on line 122 was always true
123 residual.append(residual_part)
124 if not pos_values:
125 # If nothing in the text was a POS label, preserve the original text
126 # for normal pronunciation tag/note parsing.
127 return PronunciationPosMatch((), text.strip())
128 return PronunciationPosMatch(tuple(pos_values), ", ".join(residual))
131def pronunciation_pos_from_part(part: str) -> tuple[str | None, str]:
132 normalized = normalize_pronunciation_pos_label(part)
133 if normalized in PRON_POS_BY_LABEL:
134 return PRON_POS_BY_LABEL[normalized], ""
135 # Match residual tag text followed by a POS label:
136 # "attributive adjective" -> ("adj", "attributive")
137 # "attributive proper noun" -> ("name", "attributive")
138 # The label alternation is sorted longest-first so multi-word POS labels
139 # such as "proper noun" win over their suffixes.
140 match = PRON_POS_LABEL_RE.match(normalized)
141 if match:
142 label = match.group("label")
143 residual = (match.group("residual") or "").rstrip(" ,;:")
144 if not residual or classify_desc(residual) == "tags":
145 return PRON_POS_BY_LABEL[label], residual
146 return None, part
149def split_pronunciation_pos_parts(text: str) -> list[str]:
150 parts: list[str] = []
151 for comma_part in re.split(r"[,;]", text):
152 comma_part = comma_part.strip()
153 if not comma_part:
154 continue
155 # Commas and semicolons reliably separate qualifier chunks. Only split
156 # "and"/"or" when at least one side is a POS label, so prose notes
157 # stay intact.
158 conjunction_parts = re.split(r"\s+(?:and|or)\s+", comma_part)
159 if len(conjunction_parts) > 1 and any(
160 pronunciation_pos_from_part(part)[0]
161 for part in conjunction_parts
162 ):
163 parts.extend(conjunction_parts)
164 else:
165 parts.append(comma_part)
166 return parts
169def set_sound_pos(
170 sound: SoundData, pos_values: PronunciationPoses | None
171) -> PronunciationPoses | None:
172 if pos_values:
173 sound["pos"] = pos_values # type: ignore[typeddict-unknown-key]
174 return pos_values
175 if "pos" in sound:
176 return sound["pos"] # type: ignore[typeddict-item]
177 return None
180def common_sound_pos(
181 pos_candidates: set[PronunciationPoses],
182) -> PronunciationPoses | None:
183 if len(pos_candidates) != 1:
184 return None
185 return next(iter(pos_candidates))
188def merge_pronunciation_tag_data(
189 sound: SoundData, tag_data: SoundData
190) -> None:
191 for value in tag_data.get("tags", []):
192 if value not in sound.get("tags", []): 192 ↛ 191line 192 didn't jump to line 191 because the condition on line 192 was always true
193 data_append(sound, "tags", value)
194 for value in tag_data.get("topics", []): 194 ↛ 195line 194 didn't jump to line 195 because the loop on line 194 never started
195 if value not in sound.get("topics", []):
196 data_append(sound, "topics", value)
197 if note := tag_data.get("note"):
198 existing_note = sound.get("note")
199 if not existing_note:
200 sound["note"] = note
201 elif note not in [n.strip() for n in existing_note.split(";")]: 201 ↛ exitline 201 didn't return from function 'merge_pronunciation_tag_data' because the condition on line 201 was always true
202 sound["note"] = f"{existing_note}; {note}"
205def inherit_pronunciation_tag_data(
206 sound: SoundData, parent_tag_data: SoundData
207) -> None:
208 """Add the tags and topics of a parent list item, such as
209 "* {{a|en|GA}}", to a nested pronunciation, and put its note before
210 the nested pronunciation's own."""
211 if not parent_tag_data:
212 return
213 tag_data: SoundData = {}
214 merge_pronunciation_tag_data(tag_data, parent_tag_data)
215 merge_pronunciation_tag_data(tag_data, sound)
216 if "tags" in tag_data:
217 tag_data["tags"] = sorted(tag_data["tags"])
218 if "topics" in tag_data: 218 ↛ 219line 218 didn't jump to line 219 because the condition on line 218 was never true
219 tag_data["topics"] = sorted(tag_data["topics"])
220 sound.update(tag_data)
223def parse_pronunciation_tags_with_pos(
224 wxr: WiktextractContext, text: str, sound: SoundData
225) -> PronunciationPoses:
226 match = split_pronunciation_pos_text(text)
227 set_sound_pos(sound, match.pos_values)
228 if match.residual:
229 tag_data: SoundData = {}
230 parse_pronunciation_tags(wxr, match.residual, tag_data)
231 merge_pronunciation_tag_data(sound, tag_data)
232 return match.pos_values
235def extract_pos_prefix(text: str) -> PronunciationPosPrefix | None:
236 stripped = text.strip()
237 if not (stripped.startswith("(") and stripped.endswith(")")):
238 bare_match = split_pronunciation_pos_text(text)
239 if bare_match.pos_values and not bare_match.residual:
240 return PronunciationPosPrefix(bare_match.pos_values, "", True)
242 colon_match = re.match(r"\s*([^:()]+?)\s*:\s*(.*)$", text)
243 if colon_match:
244 match = split_pronunciation_pos_text(colon_match.group(1))
245 if match.pos_values and not match.residual: 245 ↛ 246line 245 didn't jump to line 246 because the condition on line 245 was never true
246 return PronunciationPosPrefix(
247 match.pos_values, colon_match.group(2).strip(), True
248 )
250 paren_match = re.match(r"\s*\(([^()]*)\)\s*(.*)$", text)
251 if paren_match:
252 match = split_pronunciation_pos_text(paren_match.group(1))
253 if match.pos_values and not match.residual:
254 return PronunciationPosPrefix(
255 match.pos_values, paren_match.group(2).strip(), False
256 )
258 return None
261def extract_pronunciation_pos_template(
262 wxr: WiktextractContext,
263 name: str,
264 ht: TemplateArgs,
265 lang_code: str,
266) -> PronunciationPosMatch:
267 if name in {"a", "accent", "lb", "lbl", "label"}:
268 pos_args = [
269 value
270 for key, value in ht.items()
271 if isinstance(key, int) and key >= 2
272 ]
273 if not pos_args and ht.get(1) != lang_code:
274 pos_args = [ht.get(1, "")]
275 else:
276 pos_args = [
277 value
278 for key, value in ht.items()
279 if isinstance(key, int) and key >= 1
280 ]
282 pos_values: list[str] = []
283 residual: list[str] = []
284 for arg in pos_args:
285 text = clean_node(wxr, None, [arg])
286 match = split_pronunciation_pos_text(text)
287 for pos in match.pos_values:
288 if pos not in pos_values: 288 ↛ 287line 288 didn't jump to line 287 because the condition on line 288 was always true
289 pos_values.append(pos)
290 if match.residual:
291 residual.append(match.residual)
292 return PronunciationPosMatch(tuple(pos_values), ", ".join(residual))
295def extract_pron_template(
296 wxr: WiktextractContext, tname: str, targs: TemplateArgs, expanded: str
297) -> tuple[SoundData, list[SoundData]] | None:
298 """In post_template_fn, this is used to handle all enPR and IPA templates
299 so that we can leave breadcrumbs in the text that can later be handled
300 there. We return a `base_data` so that if there are two
301 or more templates on the same line, like this:
302 (Tags for the whole line, really) enPR: foo, IPA(keys): /foo/
303 then we can apply base_data fields to other templates, too, if needed.
304 """
305 cleaned = clean_value(wxr, expanded)
306 # print(f"extract_pron_template input: {tname=} {expanded=}-> {cleaned=}")
307 m = IPA_EXTRACT_RE.match(cleaned)
308 if not m:
309 wxr.wtp.error(
310 f"Text cannot match IPA_EXTRACT_RE regex: "
311 f"{cleaned=}, {tname=}, {targs=}",
312 sortid="en/pronunciation/54",
313 )
314 return None
315 # for i, group in enumerate(m.groups()):
316 # print(i + 1, repr(group))
317 main_qual = m.group(2) or ""
318 if "qq" in targs:
319 # If the template has been given a qualifier that applies to
320 # every entry, but which also happens to appear at the end
321 # which can be confused with the post-qualifier of a single
322 # entry in the style of "... /ipa3/ (foo) (bar)", where foo
323 # might not be present so the bar looks like it only might
324 # apply to `/ipa3/`
325 pron_body = m.group(5)
326 post_qual = m.group(7)
327 else:
328 pron_body = m.group(4)
329 post_qual = ""
331 if not pron_body: 331 ↛ 332line 331 didn't jump to line 332 because the condition on line 331 was never true
332 wxr.wtp.error(
333 f"Regex failed to find 'body' from {cleaned=}",
334 sortid="en/pronunciation/81",
335 )
336 return None
338 base_data: SoundData = {}
339 if main_qual:
340 parse_pronunciation_tags_with_pos(wxr, main_qual, base_data)
341 if post_qual:
342 parse_pronunciation_tags_with_pos(wxr, post_qual, base_data)
343 # This base_data is used as the base copy for all entries from this
344 # template, but it is also returned so that its contents may be applied
345 # to other templates on the same line.
346 # print(f"{base_data=}")
348 sound_datas: list[SoundData] = []
350 parts: list[list[str]] = [[]]
351 inside = 0
352 current: list[str] = []
353 for i, p in enumerate(re.split(r"(\s*,|;|\(|\)\s*)", pron_body)):
354 # Split the line on commas and semicolons outside of parens. This
355 # gives us lines with "(main-qualifier) /phon/ (post-qualifier, maybe)"
356 # print(f" {i=}, {p=}")
357 comp = p.strip()
358 if not p:
359 continue
360 if comp == "(":
361 if not inside and i > 0: 361 ↛ 364line 361 didn't jump to line 364 because the condition on line 361 was always true
362 if stripped := "".join(current).strip():
363 parts[-1].append("".join(current).strip()) # type:ignore[arg-type]
364 current = [p]
365 inside += 1
366 continue
367 if comp == ")":
368 inside -= 1
369 if not inside: 369 ↛ 374line 369 didn't jump to line 374 because the condition on line 369 was always true
370 if stripped := "".join(current).strip(): 370 ↛ 374line 370 didn't jump to line 374 because the condition on line 370 was always true
371 current.append(p)
372 parts[-1].append("".join(current).strip()) # type:ignore[arg-type]
373 current = []
374 continue
375 if not inside and comp in (",", ";"):
376 if stripped := "".join(current).strip():
377 parts[-1].append(stripped) # type:ignore[arg-type]
378 current = []
379 parts.append([])
380 continue
381 current.append(p)
382 if current:
383 parts[-1].append("".join(current).strip())
385 # print(f">>>>>> {parts=}")
386 new_parts: list[list[str]] = []
387 for entry in parts:
388 if not entry: 388 ↛ 389line 388 didn't jump to line 389 because the condition on line 388 was never true
389 continue
390 new_entry: list[str] = []
391 i1: int = entry[0].startswith("(") and entry[0].endswith(")")
392 if i1:
393 new_entry.append(entry[0][1:-1].strip())
394 else:
395 new_entry.append("")
396 i2: int = (
397 entry[-1].startswith("(")
398 and entry[-1].endswith(")")
399 and len(entry) > 1
400 )
401 if i2 == 0:
402 i2 = len(entry)
403 else:
404 i2 = -1
405 new_entry.append("".join(entry[i1:i2]).strip())
406 if not new_entry[-1]: 406 ↛ 407line 406 didn't jump to line 407 because the condition on line 406 was never true
407 wxr.wtp.error(
408 f"Missing IPA/enPRO sound data between qualifiers?{entry=}",
409 sortid="en/pronunciation/153",
410 )
411 if i2 == -1:
412 new_entry.append(entry[-1][1:-1].strip())
413 else:
414 new_entry.append("")
415 new_parts.append(new_entry)
417 # print(f">>>>> {new_parts=}")
419 for part in new_parts:
420 sd = deepcopy(base_data)
421 if part[0]:
422 parse_pronunciation_tags_with_pos(wxr, part[0], sd)
423 if part[2]:
424 parse_pronunciation_tags_with_pos(wxr, part[2], sd)
425 if tname == "enPR":
426 sd["enpr"] = part[1]
427 else:
428 sd["ipa"] = part[1]
429 sound_datas.append(sd)
431 # print(f"BASE_DATA: {base_data}")
432 # print(f"SOUND_DATAS: {sound_datas=}")
434 return base_data, sound_datas
437def parse_pronunciation(
438 wxr: WiktextractContext,
439 level_node: LevelNode,
440 data: WordData,
441 etym_data: WordData,
442 have_etym: bool,
443 base_data: WordData,
444 lang_code: str,
445) -> None:
446 """Parses the pronunciation section from a language section on a
447 page."""
448 if level_node.kind in LEVEL_KINDS: 448 ↛ 461line 448 didn't jump to line 461 because the condition on line 448 was always true
449 contents: list[str | WikiNode | TemplateNode] = []
450 for node in level_node.children:
451 if isinstance(node, TemplateNode):
452 if node.template_name == "th-pron":
453 extract_th_pron_template(wxr, data, node)
454 elif node.template_name == "zh-pron":
455 extract_zh_pron_template(wxr, data, node)
456 else:
457 contents.append(node)
458 else:
459 contents.append(node)
460 else:
461 contents = [level_node]
462 # Remove subsections, such as Usage notes. They may contain IPAchar
463 # templates in running text, and we do not want to extract IPAs from
464 # those.
465 # Filter out only LEVEL_KINDS; 'or' is doing heavy lifting here
466 # Slip through not-WikiNodes, then slip through WikiNodes that
467 # are not LEVEL_KINDS.
468 contents = [
469 x
470 for x in contents
471 if not isinstance(x, WikiNode) or x.kind not in LEVEL_KINDS
472 ]
473 if not any(
474 isinstance(x, WikiNode) and x.kind == NodeKind.LIST for x in contents
475 ):
476 # expand all templates
477 new_contents: list[str | WikiNode | TemplateNode] = []
478 for lst in contents:
479 if isinstance(lst, TemplateNode):
480 temp = wxr.wtp.node_to_wikitext(lst)
481 temp = wxr.wtp.expand(temp)
482 temp_parsed = wxr.wtp.parse(temp)
483 new_contents.extend(temp_parsed.children)
484 else:
485 new_contents.append(lst)
486 contents = new_contents
488 if have_etym and data is base_data: 488 ↛ 489line 488 didn't jump to line 489 because the condition on line 488 was never true
489 data = etym_data
490 pron_templates: list[tuple[SoundData, list[SoundData]]] = []
491 pron_pos_markers: list[PronunciationPoses] = []
492 hyphenations: list[Hyphenation] = []
493 audios: list[SoundData] = []
494 have_panel_templates = False
496 def parse_pronunciation_template_fn(
497 name: str, ht: TemplateArgs
498 ) -> str | None:
499 """Handle pronunciation and hyphenation templates"""
500 # _template_fn handles templates *before* they are expanded;
501 # this allows for special handling before all the work needed
502 # for expansion is done.
503 nonlocal have_panel_templates
504 if is_panel_template(wxr, name):
505 have_panel_templates = True
506 return ""
507 if name == "audio":
508 filename = ht.get(2) or ""
509 audio: SoundData = {"audio": filename.strip()}
510 dialect = ht.get("a", "")
511 if "aa" in ht: 511 ↛ 512line 511 didn't jump to line 512 because the condition on line 511 was never true
512 dialect += ", " + ht.get("aa", "")
513 if dialect:
514 dialect = dialect.replace("<", "").replace(">", "")
515 dialect = clean_node(wxr, None, [dialect])
516 for part in split_at_comma_semi(dialect):
517 if "(" not in part:
518 parse_pronunciation_tags(wxr, part, audio)
519 else:
520 for ppart in re.split(r"[][()]", part):
521 parse_pronunciation_tags(wxr, ppart, audio)
522 desc = ht.get(3) or ""
523 desc = clean_node(wxr, None, [desc])
524 if desc: 524 ↛ 525line 524 didn't jump to line 525 because the condition on line 524 was never true
525 audio["text"] = desc
526 m = re.search(r"\((([^()]|\([^()]*\))*)\)", desc)
527 skip = False
528 if m: 528 ↛ 529line 528 didn't jump to line 529 because the condition on line 528 was never true
529 par = m.group(1)
530 cls = classify_desc(par)
531 if cls == "tags":
532 parse_pronunciation_tags(wxr, par, audio)
533 else:
534 skip = True
535 if skip: 535 ↛ 536line 535 didn't jump to line 536 because the condition on line 535 was never true
536 return ""
537 audios.append(audio)
538 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__"
539 if name == "audio-IPA": 539 ↛ 540line 539 didn't jump to line 540 because the condition on line 539 was never true
540 filename = ht.get(2) or ""
541 ipa = ht.get(3) or ""
542 dial = ht.get("dial")
543 audio = {"audio": filename.strip()}
544 if dial:
545 dial = clean_node(wxr, None, [dial])
546 audio["text"] = dial
547 if ipa:
548 audio["audio-ipa"] = ipa
549 audios.append(audio)
550 # The problem with these IPAs is that they often just describe
551 # what's in the sound file, rather than giving the pronunciation
552 # of the word alone. It is common for audio files to contain
553 # multiple pronunciations or articles in the same file, and then
554 # this IPA often describes what is in the file.
555 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__"
556 if name == "audio-pron":
557 filename = ht.get(2) or ""
558 ipa = ht.get("ipa") or ""
559 dial = ht.get("dial")
560 country = ht.get("country")
561 audio = {"audio": filename.strip()}
562 if dial: 562 ↛ 566line 562 didn't jump to line 566 because the condition on line 562 was always true
563 dial = clean_node(wxr, None, [dial])
564 audio["text"] = dial
565 parse_pronunciation_tags(wxr, dial, audio)
566 if country: 566 ↛ 568line 566 didn't jump to line 568 because the condition on line 566 was always true
567 parse_pronunciation_tags(wxr, country, audio)
568 if ipa: 568 ↛ 570line 568 didn't jump to line 570 because the condition on line 568 was always true
569 audio["audio-ipa"] = ipa
570 audios.append(audio)
571 # XXX do we really want to extract pronunciations from these?
572 # Or are they spurious / just describing what is in the
573 # audio file?
574 # if ipa:
575 # pron = {"ipa": ipa}
576 # if dial:
577 # parse_pronunciation_tags(wxr, dial, pron)
578 # if country:
579 # parse_pronunciation_tags(wxr, country, pron)
580 # data_append(data, "sounds", pron)
581 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__"
582 if name in ("hyph", "hyphenation"):
583 # {{hyph|en|re|late|caption="Hyphenation UK:"}}
584 # {{hyphenation|it|quiè|to||qui|è|to||quié|to||qui|é|to}}
585 # and also nocaption=1
586 caption = clean_node(wxr, None, ht.get("caption", ""))
587 tagsets, _ = decode_tags(caption)
588 # flatten the tagsets into one; it would be really weird to have
589 # several tagsets for a hyphenation caption
590 tags = sorted(set(tag for tagset in tagsets for tag in tagset))
591 # We'll just ignore any errors from tags, it's not very important
592 # for hyphenation
593 tags = [tag for tag in tags if not tag.startswith("error")]
594 hyph_sequences: list[list[str]] = [[]]
595 for text in [
596 t for (k, t) in ht.items() if (isinstance(k, int) and k >= 2)
597 ]:
598 if not text:
599 hyph_sequences.append([])
600 else:
601 hyph_sequences[-1].append(clean_node(wxr, None, text))
602 for seq in hyph_sequences:
603 hyphenations.append(Hyphenation(parts=seq, tags=tags))
604 return ""
605 return None
607 may_be_duplicates = False
609 def parse_pron_post_template_fn(
610 name: str, ht: TemplateArgs, text: str
611 ) -> str | None:
612 # _post_template_fn handles templates *after* the work to expand
613 # them has been done; this is exactly the same as _template_fn,
614 # except with the additional expanded text as an input, and
615 # possible side-effects from the expansion and recursion (like
616 # calling other subtemplates that are handled in _template_fn.
617 nonlocal may_be_duplicates
618 if is_panel_template(wxr, name): 618 ↛ 619line 618 didn't jump to line 619 because the condition on line 618 was never true
619 return ""
620 if name in PRON_POS_TEMPLATE_NAMES:
621 pos_match = extract_pronunciation_pos_template(
622 wxr, name, ht, lang_code
623 )
624 if pos_match.pos_values:
625 pron_pos_markers.append(pos_match.pos_values)
626 marker = (
627 f"__PRON_POS_MARKER_{len(pron_pos_markers) - 1}__"
628 )
629 if pos_match.residual: 629 ↛ 630line 629 didn't jump to line 630 because the condition on line 629 was never true
630 return f"{marker} ({pos_match.residual})"
631 return marker
632 if name in {
633 *PRON_POS_TEMPLATE_NAMES,
634 "l",
635 "link",
636 }:
637 # Kludge: when these templates expand to /.../ or [...],
638 # replace the expansion by something safe. This is used
639 # to filter spurious IPA-looking expansions that aren't really
640 # IPAs. We probably don't care about these templates in the
641 # contexts where they expand to something containing these.
642 v = re.sub(r'href="[^"]*"', "", text) # Ignore URLs
643 v = re.sub(r'src="[^"]*"', "", v)
644 v = clean_value(wxr, v)
645 if re.search(r"/[^/,]+?/|\[[^]0-9,/][^],/]*?\]", v):
646 # Note: replacing by empty results in Lua errors that we
647 # would rather not have. For example, voi/Middle Vietnamese
648 # uses {{a|{{l{{vi|...}}}}, and the {{a|...}} will fail
649 # if {{l|...}} returns empty.
650 return "stripped-by-parse_pron_post_template_fn"
651 if name in ("IPA", "enPR"):
652 # Extract the data from IPA and enPR templates (same underlying
653 # template) and replace them in-text with magical cookie that
654 # can be later used to refer to the data's index inside
655 # pron_templates.
656 if pron_t := extract_pron_template(wxr, name, ht, text):
657 pron_templates.append(pron_t)
658 return f"__PRON_TEMPLATE_{len(pron_templates) - 1}__"
659 # Catch templates that generate duplicate sound data entries
660 # here; if the text produces a big, toggleable section, the
661 # "header" for that section might be duplicated. Add more conditions
662 # if necessary.
663 if text.startswith("<") and "vsToggleElement" in text: 663 ↛ 664line 663 didn't jump to line 664 because the condition on line 663 was never true
664 may_be_duplicates = True
665 return text
667 def flattened_tree(
668 lines: list[WikiNode | str],
669 ) -> Iterator[FlattenedListNode]:
670 assert isinstance(lines, list)
671 for line in lines:
672 yield from flattened_tree1(line, 0)
674 def flattened_tree1(
675 node: WikiNode | str, list_depth: int
676 ) -> Iterator[FlattenedListNode]:
677 assert isinstance(node, (WikiNode, str))
678 if isinstance(node, str):
679 yield FlattenedListNode(node, list_depth)
680 return
681 elif node.kind == NodeKind.LIST:
682 for item in node.children:
683 yield from flattened_tree1(item, list_depth)
684 elif node.kind == NodeKind.LIST_ITEM:
685 item_depth = (
686 len(node.sarg) if isinstance(node.sarg, str) else list_depth
687 )
688 new_children = []
689 # A list item can have several sublists, e.g. "**" lines
690 # followed by a "*:" line.
691 sublists = []
692 for child in node.children:
693 if isinstance(child, WikiNode) and child.kind == NodeKind.LIST:
694 sublists.append(child)
695 else:
696 new_children.append(child)
697 node.children = new_children
698 node.sarg = "*"
699 yield FlattenedListNode(node, item_depth)
700 for sublist in sublists:
701 yield from flattened_tree1(sublist, item_depth)
702 else:
703 yield FlattenedListNode(node, list_depth)
705 # XXX Do not use flattened_tree more than once here, for example for
706 # debug printing... The underlying data is changed, and the separated
707 # sublists disappear.
709 # Kludge for templates that generate several lines, but haven't
710 # been caught by earlier kludges...
711 def split_cleaned_node_on_newlines(
712 contents: list[WikiNode | str],
713 ) -> Iterator[tuple[str, int]]:
714 for flattened in flattened_tree(contents):
715 ipa_text = clean_node(
716 wxr,
717 data,
718 flattened.node,
719 template_fn=parse_pronunciation_template_fn,
720 post_template_fn=parse_pron_post_template_fn,
721 )
722 for line in ipa_text.splitlines():
723 yield line, flattened.list_depth
725 # have_pronunciations = False
726 active_pos: PronunciationPoses | None = None
727 # POS values from parent pronunciation lines by original list depth.
728 # Audio-only child lines can inherit from a parent pronunciation line,
729 # but same-depth audio lines must not inherit from a preceding IPA.
730 pronunciation_pos_stack: list[tuple[int, PronunciationPoses]] = []
732 def parent_pronunciation_pos(
733 list_depth: int,
734 ) -> PronunciationPoses | None:
735 if not pronunciation_pos_stack:
736 return None
737 parent_depth, parent_pos = pronunciation_pos_stack[-1]
738 return parent_pos if parent_depth < list_depth else None
740 # Tags from label-only lines by original list depth, e.g.
741 # "* {{a|en|GA}}" over "** {{IPA|en|...}}". Nested pronunciations
742 # start from their parent's tags and add their own.
743 pronunciation_tags_stack: list[tuple[int, SoundData]] = []
745 def parent_pronunciation_tags(list_depth: int) -> SoundData:
746 if not pronunciation_tags_stack:
747 return {}
748 parent_depth, parent_tags = pronunciation_tags_stack[-1]
749 return parent_tags if parent_depth < list_depth else {}
751 for line, list_depth in split_cleaned_node_on_newlines(contents):
752 prefix: str | None = None
753 earlier_base_data: SoundData | None = None
754 line_pos: PronunciationPoses | None = None
755 current_group_sounds: list[SoundData] = []
756 # POS values seen on sounds extracted from this physical line. A
757 # single candidate can seed adjacent audio-only child lines; multiple
758 # POS-marked sounds on one line are too ambiguous for inheritance.
759 line_sound_pos_candidates: set[PronunciationPoses] = set()
760 line_has_sound = False
761 if not line: 761 ↛ 762line 761 didn't jump to line 762 because the condition on line 761 was never true
762 continue
763 while (
764 pronunciation_pos_stack
765 and pronunciation_pos_stack[-1][0] >= list_depth
766 ):
767 pronunciation_pos_stack.pop()
768 while (
769 pronunciation_tags_stack
770 and pronunciation_tags_stack[-1][0] >= list_depth
771 ):
772 pronunciation_tags_stack.pop()
773 parent_tags = parent_pronunciation_tags(list_depth)
775 split_templates = re.split(r"__PRON_TEMPLATE_(\d+)__", line)
776 for i, text in enumerate(split_templates):
777 if not text:
778 continue
779 # clean up starts at the start of the line
780 text = re.sub(r"^\**\s*", "", text).strip()
781 if i == 0:
782 # At the start of a line, check for stuff like "Noun:"
783 # or "(verb)" for POS labels that apply to this line or
784 # structurally nested pronunciation lines.
785 # These labels feed the inheritance state that later sets the
786 # temporary sound["pos"] field used to route pronunciation
787 # data into matching POS sections.
788 if pos_prefix := extract_pos_prefix(text):
789 text = pos_prefix.text
790 line_pos = pos_prefix.pos_values
791 if pos_prefix.is_persistent:
792 active_pos = pos_prefix.pos_values
793 if not text:
794 continue
796 m = re.search(r"__PRON_POS_MARKER_(\d+)__", text)
797 while m:
798 if current_group_sounds and re.search(
799 r"[,;]", text[: m.start()]
800 ):
801 current_group_sounds = []
802 pos_values = pron_pos_markers[int(m.group(1))]
803 if current_group_sounds:
804 for sound in current_group_sounds:
805 set_sound_pos(sound, pos_values)
806 line_sound_pos_candidates.add(pos_values)
807 line_pos = pos_values
808 text = text[: m.start()] + text[m.end() :]
809 m = re.search(r"__PRON_POS_MARKER_(\d+)__", text)
810 text = text.strip()
811 if not text:
812 continue
813 # POS inheritance for normal pronunciation data:
814 # 1. line_pos: explicit POS marker on this line, e.g.
815 # "* {{q|noun}} {{IPA|...}}".
816 # 2. parent_pronunciation_pos: structurally inherited from a
817 # parent list item, e.g. "* {{q|noun}}" then "** {{IPA|...}}".
818 # 3. active_pos: support for "* Noun:" followed by
819 # "* {{IPA|...}}"; broad, so it stays after structural data.
820 inherited_pos = (
821 line_pos or parent_pronunciation_pos(list_depth) or active_pos
822 )
824 if i % 2 == 1:
825 # re.split (with capture groups) splits the lines so that
826 # every even entry is a captured splitter; odd lines are either
827 # empty strings or stuff around the splitters.
828 base_pron_data, first_prons = pron_templates[int(text)]
829 if base_pron_data:
830 earlier_base_data = base_pron_data
831 # print(f"Set {earlier_base_data=}")
832 elif earlier_base_data is not None:
833 # merge data from an earlier iteration of this loop
834 for pr in first_prons:
835 if "note" in pr and "note" in earlier_base_data: 835 ↛ 836line 835 didn't jump to line 836 because the condition on line 835 was never true
836 pr["note"] += ";" + earlier_base_data.get(
837 "note", ""
838 )
839 elif "note" in earlier_base_data: 839 ↛ 840line 839 didn't jump to line 840 because the condition on line 839 was never true
840 pr["note"] = earlier_base_data["note"]
841 if "topics" in earlier_base_data: 841 ↛ 842line 841 didn't jump to line 842 because the condition on line 841 was never true
842 data_extend(
843 pr, "topics", earlier_base_data["topics"]
844 )
845 if "tags" in pr and "tags" in earlier_base_data: 845 ↛ 846line 845 didn't jump to line 846 because the condition on line 845 was never true
846 pr["tags"].extend(earlier_base_data["tags"])
847 elif "tags" in earlier_base_data: 847 ↛ 834line 847 didn't jump to line 834 because the condition on line 847 was always true
848 pr["tags"] = sorted(set(earlier_base_data["tags"]))
849 for pr in first_prons:
850 inherit_pronunciation_tag_data(pr, parent_tags)
851 if sound_pos := set_sound_pos(
852 pr,
853 None if "pos" in pr else inherited_pos,
854 ):
855 line_sound_pos_candidates.add(sound_pos)
856 if pr not in data.get("sounds", ()): 856 ↛ 858line 856 didn't jump to line 858 because the condition on line 856 was always true
857 data_append(data, "sounds", pr)
858 current_group_sounds.append(pr)
859 line_has_sound = True
860 # This bit is handled
861 continue
863 if "IPA" in text:
864 field: Literal[
865 "audio",
866 "audio-ipa",
867 "enpr",
868 "form",
869 "hangeul",
870 "homophone",
871 "ipa",
872 "mp3_url",
873 "note",
874 "ogg_url",
875 "other",
876 "rhymes",
877 "tags",
878 "text",
879 "topics",
880 "zh-pron",
881 ] = "ipa"
882 else:
883 # This is used for Rhymes, Homophones, etc
884 field = "other"
886 # Check if it contains Japanese "Tokyo" pronunciation with
887 # special syntax
888 m = re.search(r"(?m)\(Tokyo\) +([^ ]+) +\[", text)
889 if m: 889 ↛ 890line 889 didn't jump to line 890 because the condition on line 889 was never true
890 pron: SoundData = {field: m.group(1)} # type: ignore[misc]
891 if sound_pos := set_sound_pos(pron, inherited_pos):
892 line_sound_pos_candidates.add(sound_pos)
893 data_append(data, "sounds", pron)
894 current_group_sounds.append(pron)
895 line_has_sound = True
896 # have_pronunciations = True
897 continue
899 # Check if it contains Rhymes
900 m = re.match(r"\s*Rhymes?: (.*)", text)
901 if m:
902 for ending in split_at_comma_semi(m.group(1)):
903 ending = ending.strip()
904 if ending: 904 ↛ 902line 904 didn't jump to line 902 because the condition on line 904 was always true
905 pron = {"rhymes": ending}
906 if sound_pos := set_sound_pos(pron, inherited_pos):
907 line_sound_pos_candidates.add(sound_pos)
908 data_append(data, "sounds", pron)
909 current_group_sounds.append(pron)
910 line_has_sound = True
911 # have_pronunciations = True
912 continue
914 # Check if it contains homophones
915 m = re.search(r"(?m)\bHomophones?: (.*)", text)
916 if m:
917 for w in split_at_comma_semi(m.group(1)):
918 w = w.strip()
919 if w: 919 ↛ 917line 919 didn't jump to line 917 because the condition on line 919 was always true
920 pron = {"homophone": w}
921 if sound_pos := set_sound_pos(pron, inherited_pos):
922 line_sound_pos_candidates.add(sound_pos)
923 data_append(data, "sounds", pron)
924 current_group_sounds.append(pron)
925 line_has_sound = True
926 # have_pronunciations = True
927 continue
929 # Check if it contains Phonetic hangeul
930 m = re.search(r"(?m)\bPhonetic hange?ul: \[([^]]+)\]", text)
931 if m: 931 ↛ 932line 931 didn't jump to line 932 because the condition on line 931 was never true
932 seen = set()
933 for w in m.group(1).split("/"):
934 w = w.strip()
935 if w and w not in seen:
936 seen.add(w)
937 pron = {"hangeul": w}
938 if sound_pos := set_sound_pos(pron, inherited_pos):
939 line_sound_pos_candidates.add(sound_pos)
940 data_append(data, "sounds", pron)
941 current_group_sounds.append(pron)
942 line_has_sound = True
943 # have_pronunciations = True
945 # This regex-based hyphenation detection left as backup
946 m = re.search(r"\b(Syllabification|Hyphenation): *([^\n.]*)", text)
947 if m:
948 data_append(data, "hyphenation", m.group(2))
949 commaseparated = m.group(2).split(",")
950 if len(commaseparated) > 1: 950 ↛ 961line 950 didn't jump to line 961 because the condition on line 950 was always true
951 for h in commaseparated:
952 # That second characters looks like a dash but it's
953 # actually unicode decimal code 8231, hyphenation dash
954 # Add more delimiters here if needed.
955 parts = re.split(r"-|‧", h.strip())
956 data_append(
957 data, "hyphenations", Hyphenation(parts=parts)
958 )
959 ...
960 else:
961 data_append(
962 data,
963 "hyphenations",
964 Hyphenation(parts=m.group(2).split(sep="-")),
965 )
966 # have_pronunciations = True
968 # See if it contains a word prefix restricting which forms the
969 # pronunciation applies to (see amica/Latin) and/or parenthesized
970 # tags.
971 m = re.match(
972 r"^[*#\s]*(([-\w]+):\s+)?\((([^()]|\([^()]*\))*?)\)", text
973 )
974 if m:
975 prefix = m.group(2) or ""
976 tagstext = m.group(3)
977 text = text[m.end() :]
978 else:
979 m = re.match(r"^[*#\s]*([-\w]+):\s+", text)
980 if m:
981 prefix = m.group(1)
982 tagstext = ""
983 text = text[m.end() :]
984 else:
985 # Spanish has tags before pronunciations, eg. aceite/Spanish
986 m = re.match(r".*:\s+\(([^)]*)\)\s+(.*)", text)
987 if m: 987 ↛ 988line 987 didn't jump to line 988 because the condition on line 987 was never true
988 tagstext = m.group(1)
989 text = m.group(2)
990 else:
991 # No prefix. In this case, we inherit prefix
992 # from previous entry. This particularly
993 # applies for nested Audio files.
994 tagstext = ""
995 if tagstext:
996 earlier_base_data = {}
997 parse_pronunciation_tags_with_pos(
998 wxr, tagstext, earlier_base_data
999 )
1001 # Find romanizations from the pronunciation section (routinely
1002 # produced for Korean by {{ko-IPA}})
1003 for m in re.finditer(pron_romanization_re, text): 1003 ↛ 1004line 1003 didn't jump to line 1004 because the loop on line 1003 never started
1004 prefix = m.group(1)
1005 w = m.group(2).strip()
1006 tag = pron_romanizations[prefix]
1007 form = {"form": w, "tags": tag.split()}
1008 data_append(data, "forms", form)
1010 # Find IPA pronunciations
1011 for m in re.finditer(
1012 r"(?m)/[^][\n/,]+?/" r"|" r"\[[^]\n0-9,/][^],/]*?\]", text
1013 ):
1014 v = m.group(0)
1015 # The regexp above can match file links. Skip them.
1016 if v.startswith("[[File:"): 1016 ↛ 1017line 1016 didn't jump to line 1017 because the condition on line 1016 was never true
1017 continue
1018 if v == "/wiki.local/": 1018 ↛ 1019line 1018 didn't jump to line 1019 because the condition on line 1018 was never true
1019 continue
1020 if field == "ipa" and "__AUDIO_IGNORE_THIS__" in text: 1020 ↛ 1021line 1020 didn't jump to line 1021 because the condition on line 1020 was never true
1021 m = re.search(r"__AUDIO_IGNORE_THIS__(\d+)__", text)
1022 assert m
1023 idx = int(m.group(1))
1024 if idx >= len(audios):
1025 continue
1026 if not audios[idx].get("audio-ipa"):
1027 audios[idx]["audio-ipa"] = v
1028 if prefix:
1029 audios[idx]["form"] = prefix
1030 else:
1031 if earlier_base_data:
1032 pron = deepcopy(earlier_base_data)
1033 pron[field] = v
1034 else:
1035 pron = {field: v} # type: ignore[misc]
1036 if prefix:
1037 pron["form"] = prefix
1038 inherit_pronunciation_tag_data(pron, parent_tags)
1039 if sound_pos := set_sound_pos(
1040 pron,
1041 None if "pos" in pron else inherited_pos,
1042 ):
1043 line_sound_pos_candidates.add(sound_pos)
1044 if may_be_duplicates is True: 1044 ↛ 1045line 1044 didn't jump to line 1045 because the condition on line 1044 was never true
1045 ok = True
1046 for comp_sound in data.get("sounds", []):
1047 # Python has dict comparison since 3.8
1048 if pron == comp_sound:
1049 ok = False
1050 break
1051 if ok:
1052 data_append(data, "sounds", pron)
1053 else:
1054 data_append(data, "sounds", pron)
1055 current_group_sounds.append(pron)
1056 line_has_sound = True
1057 # have_pronunciations = True
1058 if current_group_sounds and re.search(r"[,;]", text):
1059 current_group_sounds = []
1060 # XXX what about {{hyphenation|...}}, {{hyph|...}}
1061 # and those used to be stored under "hyphenation"
1063 # Add data that was collected in template_fn
1064 # POS inheritance for audio has one extra source:
1065 # common_sound_pos(line_sound_pos_candidates), from pronunciations
1066 # extracted earlier on the same physical line, e.g.
1067 # "* {{IPA|en|/foo/|a=verb}} {{audio|en|foo.wav}}".
1068 # Explicit line_pos still wins, then same-line sound agreement, then
1069 # parent-list structure, then active_pos.
1070 audio_inherited_pos = (
1071 line_pos
1072 or common_sound_pos(line_sound_pos_candidates)
1073 or parent_pronunciation_pos(list_depth)
1074 or active_pos
1075 )
1076 for audio in audios:
1077 if "audio" in audio: 1077 ↛ 1134line 1077 didn't jump to line 1134 because the condition on line 1077 was always true
1078 # Compute audio file URLs
1079 fn = audio["audio"]
1080 # Strip certain characters, e.g., left-to-right mark
1081 fn = re.sub(r"[\u200f\u200e]", "", fn)
1082 fn = fn.strip()
1083 fn = urllib.parse.unquote(fn)
1084 # First character is usually uppercased
1085 if re.match(r"^[a-z][a-z]+", fn):
1086 fn = fn[0].upper() + fn[1:]
1087 if fn in wxr.config.redirects: 1087 ↛ 1088line 1087 didn't jump to line 1088 because the condition on line 1087 was never true
1088 fn = wxr.config.redirects[fn]
1089 # File extension is lowercased
1090 # XXX some words seem to need this, some don't seem to
1091 # have this??? what is the exact rule?
1092 # fn = re.sub(r"\.[^.]*$", lambda m: m.group(0).lower(), fn)
1093 # Spaces are converted to underscores
1094 fn = re.sub(r"\s+", "_", fn)
1095 # Compute hash digest part
1096 h = hashlib.md5()
1097 hname = fn.encode("utf-8")
1098 h.update(hname)
1099 digest = h.hexdigest()
1100 # Quote filename for URL
1101 qfn = urllib.parse.quote(fn)
1102 # For safety when writing files
1103 qfn = qfn.replace("/", "__slash__")
1104 if re.search(r"(?i)\.(ogg|oga)$", fn):
1105 ogg = (
1106 "https://upload.wikimedia.org/wikipedia/"
1107 "commons/{}/{}/{}".format(digest[:1], digest[:2], qfn)
1108 )
1109 else:
1110 ogg = (
1111 "https://upload.wikimedia.org/wikipedia/"
1112 "commons/transcoded/"
1113 "{}/{}/{}/{}.ogg".format(
1114 digest[:1], digest[:2], qfn, qfn
1115 )
1116 )
1117 if re.search(r"(?i)\.(mp3)$", fn): 1117 ↛ 1118line 1117 didn't jump to line 1118 because the condition on line 1117 was never true
1118 mp3 = (
1119 "https://upload.wikimedia.org/wikipedia/"
1120 "commons/{}/{}/{}".format(digest[:1], digest[:2], qfn)
1121 )
1122 else:
1123 mp3 = (
1124 "https://upload.wikimedia.org/wikipedia/"
1125 "commons/transcoded/"
1126 "{}/{}/{}/{}.mp3".format(
1127 digest[:1], digest[:2], qfn, qfn
1128 )
1129 )
1130 audio["ogg_url"] = ogg
1131 audio["mp3_url"] = mp3
1132 if "pos" not in audio: 1132 ↛ 1134line 1132 didn't jump to line 1134 because the condition on line 1132 was always true
1133 set_sound_pos(audio, audio_inherited_pos)
1134 if audio not in data.get("sounds", ()):
1135 data_append(data, "sounds", audio)
1136 line_has_sound = True
1138 # if audios:
1139 # have_pronunciations = True
1140 audios = []
1142 data_extend(data, "hyphenations", hyphenations)
1143 hyphenations = []
1145 if line_pos and not line_has_sound:
1146 active_pos = line_pos
1147 pronunciation_pos_stack.append((list_depth, line_pos))
1148 elif line_pronunciation_pos := common_sound_pos(
1149 line_sound_pos_candidates
1150 ):
1151 pronunciation_pos_stack.append((list_depth, line_pronunciation_pos))
1153 if not line_has_sound and earlier_base_data:
1154 line_tags: SoundData = {}
1155 merge_pronunciation_tag_data(line_tags, earlier_base_data)
1156 if line_tags: 1156 ↛ 751line 1156 didn't jump to line 751 because the condition on line 1156 was always true
1157 inherit_pronunciation_tag_data(line_tags, parent_tags)
1158 pronunciation_tags_stack.append((list_depth, line_tags))
1160 ## I have commented out the otherwise unused have_pronunciation
1161 ## toggles; uncomment them to use this debug print
1162 # if not have_pronunciations and not have_panel_templates:
1163 # wxr.wtp.debug("no pronunciations found from pronunciation section",
1164 # sortid="pronunciations/533")
1167def extract_th_pron_template(
1168 wxr: WiktextractContext, word_entry: WordData, t_node: TemplateNode
1169):
1170 # https://en.wiktionary.org/wiki/Template:th-pron
1171 @dataclass
1172 class TableHeader:
1173 raw_tags: list[str]
1174 rowspan: int
1176 expanded_node = wxr.wtp.parse(
1177 wxr.wtp.node_to_wikitext(t_node), expand_all=True
1178 )
1179 sounds = []
1180 for table_tag in expanded_node.find_html("table"):
1181 row_headers = []
1182 for tr_tag in table_tag.find_html("tr"):
1183 field = "other"
1184 new_headers = []
1185 for header in row_headers:
1186 if header.rowspan > 1:
1187 header.rowspan -= 1
1188 new_headers.append(header)
1189 row_headers = new_headers
1190 for th_tag in tr_tag.find_html("th"):
1191 header_str = clean_node(wxr, None, th_tag)
1192 if header_str.startswith("(standard) IPA"):
1193 field = "ipa"
1194 elif header_str.startswith("Homophones"): 1194 ↛ 1195line 1194 didn't jump to line 1195 because the condition on line 1194 was never true
1195 field = "homophone"
1196 elif header_str == "Audio":
1197 field = "audio"
1198 elif header_str != "": 1198 ↛ 1190line 1198 didn't jump to line 1190 because the condition on line 1198 was always true
1199 rowspan = 1
1200 rowspan_str = th_tag.attrs.get("rowspan", "1")
1201 if re.fullmatch(r"\d+", rowspan_str): 1201 ↛ 1203line 1201 didn't jump to line 1203 because the condition on line 1201 was always true
1202 rowspan = int(rowspan_str)
1203 header = TableHeader([], rowspan)
1204 for line in header_str.splitlines():
1205 for raw_tag in line.strip("{}\n ").split(";"):
1206 raw_tag = raw_tag.strip()
1207 if raw_tag != "": 1207 ↛ 1205line 1207 didn't jump to line 1205 because the condition on line 1207 was always true
1208 header.raw_tags.append(raw_tag)
1209 row_headers.append(header)
1211 for td_tag in tr_tag.find_html("td"):
1212 if field == "audio":
1213 for link_node in td_tag.find_child(NodeKind.LINK):
1214 filename = clean_node(wxr, None, link_node.largs[0])
1215 if filename != "": 1215 ↛ 1213line 1215 didn't jump to line 1213 because the condition on line 1215 was always true
1216 sound = create_audio_url_dict(filename)
1217 sounds.append(sound)
1218 elif field == "homophone": 1218 ↛ 1219line 1218 didn't jump to line 1219 because the condition on line 1218 was never true
1219 for span_tag in td_tag.find_html_recursively(
1220 "span", attr_name="lang", attr_value="th"
1221 ):
1222 word = clean_node(wxr, None, span_tag)
1223 if word != "":
1224 sounds.append({"homophone": word})
1225 else:
1226 raw_tags = []
1227 for html_node in td_tag.find_child_recursively(
1228 NodeKind.HTML
1229 ):
1230 if html_node.tag == "small":
1231 node_str = clean_node(wxr, None, html_node)
1232 if node_str.startswith("[") and node_str.endswith(
1233 "]"
1234 ):
1235 for raw_tag in node_str.strip("[]").split(","):
1236 raw_tag = raw_tag.strip()
1237 if raw_tag != "": 1237 ↛ 1235line 1237 didn't jump to line 1235 because the condition on line 1237 was always true
1238 raw_tags.append(raw_tag)
1239 elif len(sounds) > 0: 1239 ↛ 1227line 1239 didn't jump to line 1227 because the condition on line 1239 was always true
1240 sounds[-1]["roman"] = node_str
1241 elif html_node.tag == "span":
1242 node_str = clean_node(wxr, None, html_node)
1243 span_lang = html_node.attrs.get("lang", "")
1244 span_class = html_node.attrs.get("class", "")
1245 if node_str != "" and (
1246 span_lang == "th" or span_class in ["IPA", "tr"]
1247 ):
1248 sound = {}
1249 for raw_tag in raw_tags:
1250 if raw_tag in valid_tags: 1250 ↛ 1253line 1250 didn't jump to line 1253 because the condition on line 1250 was always true
1251 data_append(sound, "tags", raw_tag)
1252 else:
1253 data_append(sound, "raw_tags", raw_tag)
1254 for header in row_headers:
1255 for raw_tag in header.raw_tags:
1256 if raw_tag.lower() in valid_tags:
1257 data_append(
1258 sound, "tags", raw_tag.lower()
1259 )
1260 else:
1261 data_append(
1262 sound, "raw_tags", raw_tag
1263 )
1264 if "romanization" in sound.get("tags", []):
1265 field = "roman"
1266 sound[field] = node_str
1267 sounds.append(sound)
1269 clean_node(wxr, word_entry, expanded_node)
1270 data_extend(word_entry, "sounds", sounds)
1273def extract_zh_pron_template(
1274 wxr: WiktextractContext, word_entry: WordData, t_node: TemplateNode
1275):
1276 # https://en.wiktionary.org/wiki/Template:zh-pron
1277 expanded_node = wxr.wtp.parse(
1278 wxr.wtp.node_to_wikitext(t_node), expand_all=True
1279 )
1280 seen_lists = set()
1281 sounds = []
1282 for list_node in expanded_node.find_child_recursively(NodeKind.LIST):
1283 if list_node not in seen_lists:
1284 for list_item in list_node.find_child(NodeKind.LIST_ITEM):
1285 sounds.extend(
1286 extract_zh_pron_list_item(wxr, list_item, [], seen_lists)
1287 )
1288 clean_node(wxr, word_entry, expanded_node)
1289 data_extend(word_entry, "sounds", sounds)
1292def extract_zh_pron_list_item(
1293 wxr: WiktextractContext,
1294 list_item: WikiNode,
1295 raw_tags: list[str],
1296 seen_lists: set[WikiNode],
1297) -> list[SoundData]:
1298 current_tags = raw_tags[:]
1299 sounds = []
1300 is_first_small_tag = True
1301 for node in list_item.children:
1302 if isinstance(node, WikiNode) and node.kind == NodeKind.LINK:
1303 link_str = clean_node(wxr, None, node.largs)
1304 node_str = clean_node(wxr, None, node)
1305 if link_str.startswith("File:"): 1305 ↛ 1306line 1305 didn't jump to line 1306 because the condition on line 1305 was never true
1306 sound = create_audio_url_dict(link_str.removeprefix("File:"))
1307 sound["raw_tags"] = current_tags[:]
1308 translate_zh_pron_raw_tags(sound)
1309 sounds.append(sound)
1310 elif node_str != "": 1310 ↛ 1301line 1310 didn't jump to line 1301 because the condition on line 1310 was always true
1311 current_tags.append(node_str)
1312 elif isinstance(node, HTMLNode):
1313 if node.tag == "small":
1314 if is_first_small_tag: 1314 ↛ 1325line 1314 didn't jump to line 1325 because the condition on line 1314 was always true
1315 raw_tag_text = clean_node(
1316 wxr,
1317 None,
1318 [
1319 n
1320 for n in node.children
1321 if not (isinstance(n, HTMLNode) and n.tag == "sup")
1322 ],
1323 )
1324 current_tags.extend(split_zh_pron_raw_tag(raw_tag_text))
1325 elif len(sounds) > 0:
1326 data_extend(
1327 sounds[-1],
1328 "raw_tags",
1329 split_zh_pron_raw_tag(clean_node(wxr, None, node)),
1330 )
1331 translate_zh_pron_raw_tags(sounds[-1])
1332 is_first_small_tag = False
1333 elif node.tag == "span":
1334 sounds.extend(extract_zh_pron_span(wxr, node, current_tags))
1335 elif ( 1335 ↛ 1340line 1335 didn't jump to line 1340 because the condition on line 1335 was never true
1336 node.tag == "table"
1337 and len(current_tags) > 0
1338 and current_tags[-1] == "Homophones"
1339 ):
1340 sounds.extend(
1341 extract_zh_pron_homophone_table(wxr, node, current_tags)
1342 )
1343 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST:
1344 seen_lists.add(node)
1345 for child_list_item in node.find_child(NodeKind.LIST_ITEM):
1346 sounds.extend(
1347 extract_zh_pron_list_item(
1348 wxr, child_list_item, current_tags, seen_lists
1349 )
1350 )
1352 return sounds
1355def extract_zh_pron_homophone_table(
1356 wxr: WiktextractContext, table: HTMLNode, raw_tags: list[str]
1357) -> list[SoundData]:
1358 sounds = []
1359 for td_tag in table.find_html_recursively("td"):
1360 for span_tag in td_tag.find_html("span"):
1361 span_class = span_tag.attrs.get("class", "")
1362 span_lang = span_tag.attrs.get("lang", "")
1363 span_str = clean_node(wxr, None, span_tag)
1364 if (
1365 span_str not in ["", "/"]
1366 and span_lang != ""
1367 and span_class in ["Hant", "Hans", "Hani"]
1368 ):
1369 sound = {"homophone": span_str, "raw_tags": raw_tags[:]}
1370 if span_class == "Hant":
1371 data_append(sound, "tags", "Traditional-Chinese")
1372 elif span_class == "Hans":
1373 data_append(sound, "tags", "Simplified-Chinese")
1374 translate_zh_pron_raw_tags(sound)
1375 sounds.append(sound)
1377 return sounds
1380def translate_zh_pron_raw_tags(sound: SoundData):
1381 from .zh_pron_tags import ZH_PRON_TAGS
1383 raw_tags = []
1384 for raw_tag in sound.get("raw_tags", []):
1385 if raw_tag in ZH_PRON_TAGS:
1386 tr_tag = ZH_PRON_TAGS[raw_tag]
1387 if isinstance(tr_tag, str):
1388 data_append(sound, "tags", tr_tag)
1389 elif isinstance(tr_tag, list) and tr_tag not in sound.get( 1389 ↛ 1384line 1389 didn't jump to line 1384 because the condition on line 1389 was always true
1390 "tags", []
1391 ):
1392 data_extend(sound, "tags", tr_tag)
1393 elif raw_tag in valid_tags:
1394 if raw_tag not in sound.get("tags", []): 1394 ↛ 1384line 1394 didn't jump to line 1384 because the condition on line 1394 was always true
1395 data_append(sound, "tags", raw_tag)
1396 elif raw_tag not in raw_tags: 1396 ↛ 1384line 1396 didn't jump to line 1384 because the condition on line 1396 was always true
1397 raw_tags.append(raw_tag)
1399 if len(raw_tags) > 0:
1400 sound["raw_tags"] = raw_tags
1401 elif "raw_tags" in sound: 1401 ↛ exitline 1401 didn't return from function 'translate_zh_pron_raw_tags' because the condition on line 1401 was always true
1402 del sound["raw_tags"]
1405def split_zh_pron_raw_tag(raw_tag_text: str) -> list[str]:
1406 raw_tags = []
1407 if "(" not in raw_tag_text:
1408 for raw_tag in re.split(r",|:|;| and ", raw_tag_text):
1409 raw_tag = raw_tag.strip().removeprefix("incl. ").strip()
1410 if raw_tag != "":
1411 raw_tags.append(raw_tag)
1412 else:
1413 processed_offsets = []
1414 for match in re.finditer(r"\([^()]+\)", raw_tag_text):
1415 processed_offsets.append((match.start(), match.end()))
1416 raw_tags.extend(
1417 split_zh_pron_raw_tag(
1418 raw_tag_text[match.start() + 1 : match.end() - 1]
1419 )
1420 )
1421 not_processed = ""
1422 last_end = 0
1423 for start, end in processed_offsets:
1424 not_processed += raw_tag_text[last_end:start]
1425 last_end = end
1426 not_processed += raw_tag_text[last_end:]
1427 if not_processed != raw_tag_text: 1427 ↛ 1430line 1427 didn't jump to line 1430 because the condition on line 1427 was always true
1428 raw_tags = split_zh_pron_raw_tag(not_processed) + raw_tags
1429 else:
1430 raw_tags.append(not_processed)
1432 return raw_tags
1435def extract_zh_pron_span(
1436 wxr: WiktextractContext, span_tag: HTMLNode, raw_tags: list[str]
1437) -> list[SoundData]:
1438 sounds = []
1439 small_tags = []
1440 pron_nodes = []
1441 roman = ""
1442 phonetic_pron = ""
1443 for index, node in enumerate(span_tag.children):
1444 if isinstance(node, HTMLNode) and node.tag == "small": 1444 ↛ 1445line 1444 didn't jump to line 1445 because the condition on line 1444 was never true
1445 small_tags = split_zh_pron_raw_tag(clean_node(wxr, None, node))
1446 elif ( 1446 ↛ 1451line 1446 didn't jump to line 1451 because the condition on line 1446 was never true
1447 isinstance(node, HTMLNode)
1448 and node.tag == "span"
1449 and "-Latn" in node.attrs.get("lang", "")
1450 ):
1451 roman = clean_node(wxr, None, node).strip("() ")
1452 elif isinstance(node, str) and node.strip() == "[Phonetic:": 1452 ↛ 1453line 1452 didn't jump to line 1453 because the condition on line 1452 was never true
1453 phonetic_pron = clean_node(
1454 wxr, None, span_tag.children[index + 1 :]
1455 ).strip("] ")
1456 break
1457 else:
1458 pron_nodes.append(node)
1459 for zh_pron in split_zh_pron(clean_node(wxr, None, pron_nodes)):
1460 zh_pron = zh_pron.strip("[]: ")
1461 if len(zh_pron) > 0: 1461 ↛ 1459line 1461 didn't jump to line 1459 because the condition on line 1461 was always true
1462 if "IPA" in span_tag.attrs.get("class", ""): 1462 ↛ 1463line 1462 didn't jump to line 1463 because the condition on line 1462 was never true
1463 sound = {"ipa": zh_pron, "raw_tags": raw_tags[:]}
1464 else:
1465 sound = {"zh_pron": zh_pron, "raw_tags": raw_tags[:]}
1466 if roman != "": 1466 ↛ 1467line 1466 didn't jump to line 1467 because the condition on line 1466 was never true
1467 sound["roman"] = roman
1468 sounds.append(sound)
1469 if len(sounds) > 0: 1469 ↛ 1471line 1469 didn't jump to line 1471 because the condition on line 1469 was always true
1470 data_extend(sounds[-1], "raw_tags", small_tags)
1471 if phonetic_pron != "": 1471 ↛ 1472line 1471 didn't jump to line 1472 because the condition on line 1471 was never true
1472 sound = {
1473 "zh_pron": phonetic_pron,
1474 "raw_tags": raw_tags[:] + ["Phonetic"],
1475 }
1476 if roman != "":
1477 sound["roman"] = roman
1478 sounds.append(sound)
1479 for sound in sounds:
1480 translate_zh_pron_raw_tags(sound)
1481 return sounds
1484def split_zh_pron(zh_pron: str) -> list[str]:
1485 # split by comma and other symbols that outside parentheses
1486 parentheses = 0
1487 pron_list = []
1488 pron = ""
1489 for c in zh_pron:
1490 if (
1491 (c in [",", ";", "→"] or (c == "/" and not zh_pron.startswith("/")))
1492 and parentheses == 0
1493 and len(pron.strip()) > 0
1494 ):
1495 pron_list.append(pron.strip())
1496 pron = ""
1497 elif c == "(":
1498 parentheses += 1
1499 pron += c
1500 elif c == ")":
1501 parentheses -= 1
1502 pron += c
1503 else:
1504 pron += c
1506 if pron.strip() != "": 1506 ↛ 1508line 1506 didn't jump to line 1508 because the condition on line 1506 was always true
1507 pron_list.append(pron)
1508 return pron_list