Coverage for src/wiktextract/extractor/en/pronunciation.py: 83%

878 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1import hashlib 

2import re 

3import urllib 

4from copy import deepcopy 

5from dataclasses import dataclass 

6from typing import Iterator, Literal, NamedTuple 

7 

8from wikitextprocessor import ( 

9 HTMLNode, 

10 LevelNode, 

11 NodeKind, 

12 TemplateNode, 

13 WikiNode, 

14) 

15 

16from ...clean import clean_value 

17from ...datautils import data_append, data_extend, split_at_comma_semi 

18from ...page import LEVEL_KINDS, clean_node, is_panel_template 

19from ...tags import valid_tags 

20from ...wxr_context import WiktextractContext 

21from ..share import create_audio_url_dict 

22from .form_descriptions import ( 

23 classify_desc, 

24 decode_tags, 

25 parse_pronunciation_tags, 

26) 

27from .parts_of_speech import part_of_speech_map 

28from .type_utils import Hyphenation, SoundData, TemplateArgs, WordData 

29 

30PronunciationPoses = tuple[str, ...] 

31 

32# Prefixes, tags, and regexp for finding romanizations from the pronuncation 

33# section 

34pron_romanizations = { 

35 " Revised Romanization ": "romanization revised", 

36 " Revised Romanization (translit.) ": ( 

37 "romanization revised transliteration" 

38 ), 

39 " McCune-Reischauer ": "McCune-Reischauer romanization", 

40 " McCune–Reischauer ": "McCune-Reischauer romanization", 

41 " Yale Romanization ": "Yale romanization", 

42} 

43pron_romanization_re = re.compile( 

44 "(?m)^(" 

45 + "|".join( 

46 re.escape(x) 

47 for x in sorted(pron_romanizations.keys(), key=len, reverse=True) 

48 ) 

49 + ")([^\n]+)" 

50) 

51 

52IPA_EXTRACT = r"^(\((.+)\) )?(IPA⁽ᵏᵉʸ⁾|enPR): ((.+?)( \(([^(]+)\))?\s*)$" 

53IPA_EXTRACT_RE = re.compile(IPA_EXTRACT) 

54 

55 

56class PronunciationPosMatch(NamedTuple): 

57 pos_values: PronunciationPoses 

58 residual: str 

59 

60 

61class PronunciationPosPrefix(NamedTuple): 

62 pos_values: PronunciationPoses 

63 text: str 

64 is_persistent: bool 

65 

66 

67class FlattenedListNode(NamedTuple): 

68 node: WikiNode | str 

69 list_depth: int 

70 

71 

72PRON_POS_TEMPLATE_NAMES = { 

73 "q", 

74 "qualifier", 

75 "qual", 

76 "i", 

77 "sense", 

78 "a", 

79 "accent", 

80 "lb", 

81 "lbl", 

82 "label", 

83} 

84 

85PRON_POS_BY_LABEL = { 

86 label: pos_data["pos"] for label, pos_data in part_of_speech_map.items() 

87} 

88 

89PRON_POS_LABEL_RE = re.compile( 

90 r"^(?:(?P<residual>.+?)\s+)?(?P<label>" 

91 + "|".join( 

92 re.escape(label) 

93 for label in sorted(PRON_POS_BY_LABEL, key=len, reverse=True) 

94 ) 

95 + r")$" 

96) 

97 

98 

99def normalize_pronunciation_pos_label(label: str) -> str: 

100 label = label.strip().lower() 

101 label = re.sub(r"\s+", " ", label) 

102 # Drop explanatory suffixes such as "noun (barren areas)" before 

103 # matching the label against part-of-speech names. 

104 label = re.sub(r"\s*\([^)]*\)\s*$", "", label).strip() 

105 label = label.strip(" \t\n\r():") 

106 label = re.sub(r"\s+senses?$", "", label).strip() 

107 return label 

108 

109 

110def split_pronunciation_pos_text(text: str) -> PronunciationPosMatch: 

111 pos_values: list[str] = [] 

112 residual: list[str] = [] 

113 for part in split_pronunciation_pos_parts(text): 

114 pos, residual_part = pronunciation_pos_from_part(part) 

115 if pos: 

116 # POS-bearing qualifier text may also contain normal pronunciation 

117 # tags before the POS label, e.g. "attributive adjective". 

118 if pos not in pos_values: 118 ↛ 120line 118 didn't jump to line 120 because the condition on line 118 was always true

119 pos_values.append(pos) 

120 if residual_part: 

121 residual.append(residual_part) 

122 elif residual_part: 122 ↛ 113line 122 didn't jump to line 113 because the condition on line 122 was always true

123 residual.append(residual_part) 

124 if not pos_values: 

125 # If nothing in the text was a POS label, preserve the original text 

126 # for normal pronunciation tag/note parsing. 

127 return PronunciationPosMatch((), text.strip()) 

128 return PronunciationPosMatch(tuple(pos_values), ", ".join(residual)) 

129 

130 

131def pronunciation_pos_from_part(part: str) -> tuple[str | None, str]: 

132 normalized = normalize_pronunciation_pos_label(part) 

133 if normalized in PRON_POS_BY_LABEL: 

134 return PRON_POS_BY_LABEL[normalized], "" 

135 # Match residual tag text followed by a POS label: 

136 # "attributive adjective" -> ("adj", "attributive") 

137 # "attributive proper noun" -> ("name", "attributive") 

138 # The label alternation is sorted longest-first so multi-word POS labels 

139 # such as "proper noun" win over their suffixes. 

140 match = PRON_POS_LABEL_RE.match(normalized) 

141 if match: 

142 label = match.group("label") 

143 residual = (match.group("residual") or "").rstrip(" ,;:") 

144 if not residual or classify_desc(residual) == "tags": 

145 return PRON_POS_BY_LABEL[label], residual 

146 return None, part 

147 

148 

149def split_pronunciation_pos_parts(text: str) -> list[str]: 

150 parts: list[str] = [] 

151 for comma_part in re.split(r"[,;]", text): 

152 comma_part = comma_part.strip() 

153 if not comma_part: 

154 continue 

155 # Commas and semicolons reliably separate qualifier chunks. Only split 

156 # "and"/"or" when at least one side is a POS label, so prose notes 

157 # stay intact. 

158 conjunction_parts = re.split(r"\s+(?:and|or)\s+", comma_part) 

159 if len(conjunction_parts) > 1 and any( 

160 pronunciation_pos_from_part(part)[0] 

161 for part in conjunction_parts 

162 ): 

163 parts.extend(conjunction_parts) 

164 else: 

165 parts.append(comma_part) 

166 return parts 

167 

168 

169def set_sound_pos( 

170 sound: SoundData, pos_values: PronunciationPoses | None 

171) -> PronunciationPoses | None: 

172 if pos_values: 

173 sound["pos"] = pos_values # type: ignore[typeddict-unknown-key] 

174 return pos_values 

175 if "pos" in sound: 

176 return sound["pos"] # type: ignore[typeddict-item] 

177 return None 

178 

179 

180def common_sound_pos( 

181 pos_candidates: set[PronunciationPoses], 

182) -> PronunciationPoses | None: 

183 if len(pos_candidates) != 1: 

184 return None 

185 return next(iter(pos_candidates)) 

186 

187 

188def merge_pronunciation_tag_data( 

189 sound: SoundData, tag_data: SoundData 

190) -> None: 

191 for value in tag_data.get("tags", []): 

192 if value not in sound.get("tags", []): 192 ↛ 191line 192 didn't jump to line 191 because the condition on line 192 was always true

193 data_append(sound, "tags", value) 

194 for value in tag_data.get("topics", []): 194 ↛ 195line 194 didn't jump to line 195 because the loop on line 194 never started

195 if value not in sound.get("topics", []): 

196 data_append(sound, "topics", value) 

197 if note := tag_data.get("note"): 

198 existing_note = sound.get("note") 

199 if not existing_note: 

200 sound["note"] = note 

201 elif note not in [n.strip() for n in existing_note.split(";")]: 201 ↛ exitline 201 didn't return from function 'merge_pronunciation_tag_data' because the condition on line 201 was always true

202 sound["note"] = f"{existing_note}; {note}" 

203 

204 

205def inherit_pronunciation_tag_data( 

206 sound: SoundData, parent_tag_data: SoundData 

207) -> None: 

208 """Add the tags and topics of a parent list item, such as 

209 "* {{a|en|GA}}", to a nested pronunciation, and put its note before 

210 the nested pronunciation's own.""" 

211 if not parent_tag_data: 

212 return 

213 tag_data: SoundData = {} 

214 merge_pronunciation_tag_data(tag_data, parent_tag_data) 

215 merge_pronunciation_tag_data(tag_data, sound) 

216 if "tags" in tag_data: 

217 tag_data["tags"] = sorted(tag_data["tags"]) 

218 if "topics" in tag_data: 218 ↛ 219line 218 didn't jump to line 219 because the condition on line 218 was never true

219 tag_data["topics"] = sorted(tag_data["topics"]) 

220 sound.update(tag_data) 

221 

222 

223def parse_pronunciation_tags_with_pos( 

224 wxr: WiktextractContext, text: str, sound: SoundData 

225) -> PronunciationPoses: 

226 match = split_pronunciation_pos_text(text) 

227 set_sound_pos(sound, match.pos_values) 

228 if match.residual: 

229 tag_data: SoundData = {} 

230 parse_pronunciation_tags(wxr, match.residual, tag_data) 

231 merge_pronunciation_tag_data(sound, tag_data) 

232 return match.pos_values 

233 

234 

235def extract_pos_prefix(text: str) -> PronunciationPosPrefix | None: 

236 stripped = text.strip() 

237 if not (stripped.startswith("(") and stripped.endswith(")")): 

238 bare_match = split_pronunciation_pos_text(text) 

239 if bare_match.pos_values and not bare_match.residual: 

240 return PronunciationPosPrefix(bare_match.pos_values, "", True) 

241 

242 colon_match = re.match(r"\s*([^:()]+?)\s*:\s*(.*)$", text) 

243 if colon_match: 

244 match = split_pronunciation_pos_text(colon_match.group(1)) 

245 if match.pos_values and not match.residual: 245 ↛ 246line 245 didn't jump to line 246 because the condition on line 245 was never true

246 return PronunciationPosPrefix( 

247 match.pos_values, colon_match.group(2).strip(), True 

248 ) 

249 

250 paren_match = re.match(r"\s*\(([^()]*)\)\s*(.*)$", text) 

251 if paren_match: 

252 match = split_pronunciation_pos_text(paren_match.group(1)) 

253 if match.pos_values and not match.residual: 

254 return PronunciationPosPrefix( 

255 match.pos_values, paren_match.group(2).strip(), False 

256 ) 

257 

258 return None 

259 

260 

261def extract_pronunciation_pos_template( 

262 wxr: WiktextractContext, 

263 name: str, 

264 ht: TemplateArgs, 

265 lang_code: str, 

266) -> PronunciationPosMatch: 

267 if name in {"a", "accent", "lb", "lbl", "label"}: 

268 pos_args = [ 

269 value 

270 for key, value in ht.items() 

271 if isinstance(key, int) and key >= 2 

272 ] 

273 if not pos_args and ht.get(1) != lang_code: 

274 pos_args = [ht.get(1, "")] 

275 else: 

276 pos_args = [ 

277 value 

278 for key, value in ht.items() 

279 if isinstance(key, int) and key >= 1 

280 ] 

281 

282 pos_values: list[str] = [] 

283 residual: list[str] = [] 

284 for arg in pos_args: 

285 text = clean_node(wxr, None, [arg]) 

286 match = split_pronunciation_pos_text(text) 

287 for pos in match.pos_values: 

288 if pos not in pos_values: 288 ↛ 287line 288 didn't jump to line 287 because the condition on line 288 was always true

289 pos_values.append(pos) 

290 if match.residual: 

291 residual.append(match.residual) 

292 return PronunciationPosMatch(tuple(pos_values), ", ".join(residual)) 

293 

294 

295def extract_pron_template( 

296 wxr: WiktextractContext, tname: str, targs: TemplateArgs, expanded: str 

297) -> tuple[SoundData, list[SoundData]] | None: 

298 """In post_template_fn, this is used to handle all enPR and IPA templates 

299 so that we can leave breadcrumbs in the text that can later be handled 

300 there. We return a `base_data` so that if there are two 

301 or more templates on the same line, like this: 

302 (Tags for the whole line, really) enPR: foo, IPA(keys): /foo/ 

303 then we can apply base_data fields to other templates, too, if needed. 

304 """ 

305 cleaned = clean_value(wxr, expanded) 

306 # print(f"extract_pron_template input: {tname=} {expanded=}-> {cleaned=}") 

307 m = IPA_EXTRACT_RE.match(cleaned) 

308 if not m: 

309 wxr.wtp.error( 

310 f"Text cannot match IPA_EXTRACT_RE regex: " 

311 f"{cleaned=}, {tname=}, {targs=}", 

312 sortid="en/pronunciation/54", 

313 ) 

314 return None 

315 # for i, group in enumerate(m.groups()): 

316 # print(i + 1, repr(group)) 

317 main_qual = m.group(2) or "" 

318 if "qq" in targs: 

319 # If the template has been given a qualifier that applies to 

320 # every entry, but which also happens to appear at the end 

321 # which can be confused with the post-qualifier of a single 

322 # entry in the style of "... /ipa3/ (foo) (bar)", where foo 

323 # might not be present so the bar looks like it only might 

324 # apply to `/ipa3/` 

325 pron_body = m.group(5) 

326 post_qual = m.group(7) 

327 else: 

328 pron_body = m.group(4) 

329 post_qual = "" 

330 

331 if not pron_body: 331 ↛ 332line 331 didn't jump to line 332 because the condition on line 331 was never true

332 wxr.wtp.error( 

333 f"Regex failed to find 'body' from {cleaned=}", 

334 sortid="en/pronunciation/81", 

335 ) 

336 return None 

337 

338 base_data: SoundData = {} 

339 if main_qual: 

340 parse_pronunciation_tags_with_pos(wxr, main_qual, base_data) 

341 if post_qual: 

342 parse_pronunciation_tags_with_pos(wxr, post_qual, base_data) 

343 # This base_data is used as the base copy for all entries from this 

344 # template, but it is also returned so that its contents may be applied 

345 # to other templates on the same line. 

346 # print(f"{base_data=}") 

347 

348 sound_datas: list[SoundData] = [] 

349 

350 parts: list[list[str]] = [[]] 

351 inside = 0 

352 current: list[str] = [] 

353 for i, p in enumerate(re.split(r"(\s*,|;|\(|\)\s*)", pron_body)): 

354 # Split the line on commas and semicolons outside of parens. This 

355 # gives us lines with "(main-qualifier) /phon/ (post-qualifier, maybe)" 

356 # print(f" {i=}, {p=}") 

357 comp = p.strip() 

358 if not p: 

359 continue 

360 if comp == "(": 

361 if not inside and i > 0: 361 ↛ 364line 361 didn't jump to line 364 because the condition on line 361 was always true

362 if stripped := "".join(current).strip(): 

363 parts[-1].append("".join(current).strip()) # type:ignore[arg-type] 

364 current = [p] 

365 inside += 1 

366 continue 

367 if comp == ")": 

368 inside -= 1 

369 if not inside: 369 ↛ 374line 369 didn't jump to line 374 because the condition on line 369 was always true

370 if stripped := "".join(current).strip(): 370 ↛ 374line 370 didn't jump to line 374 because the condition on line 370 was always true

371 current.append(p) 

372 parts[-1].append("".join(current).strip()) # type:ignore[arg-type] 

373 current = [] 

374 continue 

375 if not inside and comp in (",", ";"): 

376 if stripped := "".join(current).strip(): 

377 parts[-1].append(stripped) # type:ignore[arg-type] 

378 current = [] 

379 parts.append([]) 

380 continue 

381 current.append(p) 

382 if current: 

383 parts[-1].append("".join(current).strip()) 

384 

385 # print(f">>>>>> {parts=}") 

386 new_parts: list[list[str]] = [] 

387 for entry in parts: 

388 if not entry: 388 ↛ 389line 388 didn't jump to line 389 because the condition on line 388 was never true

389 continue 

390 new_entry: list[str] = [] 

391 i1: int = entry[0].startswith("(") and entry[0].endswith(")") 

392 if i1: 

393 new_entry.append(entry[0][1:-1].strip()) 

394 else: 

395 new_entry.append("") 

396 i2: int = ( 

397 entry[-1].startswith("(") 

398 and entry[-1].endswith(")") 

399 and len(entry) > 1 

400 ) 

401 if i2 == 0: 

402 i2 = len(entry) 

403 else: 

404 i2 = -1 

405 new_entry.append("".join(entry[i1:i2]).strip()) 

406 if not new_entry[-1]: 406 ↛ 407line 406 didn't jump to line 407 because the condition on line 406 was never true

407 wxr.wtp.error( 

408 f"Missing IPA/enPRO sound data between qualifiers?{entry=}", 

409 sortid="en/pronunciation/153", 

410 ) 

411 if i2 == -1: 

412 new_entry.append(entry[-1][1:-1].strip()) 

413 else: 

414 new_entry.append("") 

415 new_parts.append(new_entry) 

416 

417 # print(f">>>>> {new_parts=}") 

418 

419 for part in new_parts: 

420 sd = deepcopy(base_data) 

421 if part[0]: 

422 parse_pronunciation_tags_with_pos(wxr, part[0], sd) 

423 if part[2]: 

424 parse_pronunciation_tags_with_pos(wxr, part[2], sd) 

425 if tname == "enPR": 

426 sd["enpr"] = part[1] 

427 else: 

428 sd["ipa"] = part[1] 

429 sound_datas.append(sd) 

430 

431 # print(f"BASE_DATA: {base_data}") 

432 # print(f"SOUND_DATAS: {sound_datas=}") 

433 

434 return base_data, sound_datas 

435 

436 

437def parse_pronunciation( 

438 wxr: WiktextractContext, 

439 level_node: LevelNode, 

440 data: WordData, 

441 etym_data: WordData, 

442 have_etym: bool, 

443 base_data: WordData, 

444 lang_code: str, 

445) -> None: 

446 """Parses the pronunciation section from a language section on a 

447 page.""" 

448 if level_node.kind in LEVEL_KINDS: 448 ↛ 461line 448 didn't jump to line 461 because the condition on line 448 was always true

449 contents: list[str | WikiNode | TemplateNode] = [] 

450 for node in level_node.children: 

451 if isinstance(node, TemplateNode): 

452 if node.template_name == "th-pron": 

453 extract_th_pron_template(wxr, data, node) 

454 elif node.template_name == "zh-pron": 

455 extract_zh_pron_template(wxr, data, node) 

456 else: 

457 contents.append(node) 

458 else: 

459 contents.append(node) 

460 else: 

461 contents = [level_node] 

462 # Remove subsections, such as Usage notes. They may contain IPAchar 

463 # templates in running text, and we do not want to extract IPAs from 

464 # those. 

465 # Filter out only LEVEL_KINDS; 'or' is doing heavy lifting here 

466 # Slip through not-WikiNodes, then slip through WikiNodes that 

467 # are not LEVEL_KINDS. 

468 contents = [ 

469 x 

470 for x in contents 

471 if not isinstance(x, WikiNode) or x.kind not in LEVEL_KINDS 

472 ] 

473 if not any( 

474 isinstance(x, WikiNode) and x.kind == NodeKind.LIST for x in contents 

475 ): 

476 # expand all templates 

477 new_contents: list[str | WikiNode | TemplateNode] = [] 

478 for lst in contents: 

479 if isinstance(lst, TemplateNode): 

480 temp = wxr.wtp.node_to_wikitext(lst) 

481 temp = wxr.wtp.expand(temp) 

482 temp_parsed = wxr.wtp.parse(temp) 

483 new_contents.extend(temp_parsed.children) 

484 else: 

485 new_contents.append(lst) 

486 contents = new_contents 

487 

488 if have_etym and data is base_data: 488 ↛ 489line 488 didn't jump to line 489 because the condition on line 488 was never true

489 data = etym_data 

490 pron_templates: list[tuple[SoundData, list[SoundData]]] = [] 

491 pron_pos_markers: list[PronunciationPoses] = [] 

492 hyphenations: list[Hyphenation] = [] 

493 audios: list[SoundData] = [] 

494 have_panel_templates = False 

495 

496 def parse_pronunciation_template_fn( 

497 name: str, ht: TemplateArgs 

498 ) -> str | None: 

499 """Handle pronunciation and hyphenation templates""" 

500 # _template_fn handles templates *before* they are expanded; 

501 # this allows for special handling before all the work needed 

502 # for expansion is done. 

503 nonlocal have_panel_templates 

504 if is_panel_template(wxr, name): 

505 have_panel_templates = True 

506 return "" 

507 if name == "audio": 

508 filename = ht.get(2) or "" 

509 audio: SoundData = {"audio": filename.strip()} 

510 dialect = ht.get("a", "") 

511 if "aa" in ht: 511 ↛ 512line 511 didn't jump to line 512 because the condition on line 511 was never true

512 dialect += ", " + ht.get("aa", "") 

513 if dialect: 

514 dialect = dialect.replace("<", "").replace(">", "") 

515 dialect = clean_node(wxr, None, [dialect]) 

516 for part in split_at_comma_semi(dialect): 

517 if "(" not in part: 

518 parse_pronunciation_tags(wxr, part, audio) 

519 else: 

520 for ppart in re.split(r"[][()]", part): 

521 parse_pronunciation_tags(wxr, ppart, audio) 

522 desc = ht.get(3) or "" 

523 desc = clean_node(wxr, None, [desc]) 

524 if desc: 524 ↛ 525line 524 didn't jump to line 525 because the condition on line 524 was never true

525 audio["text"] = desc 

526 m = re.search(r"\((([^()]|\([^()]*\))*)\)", desc) 

527 skip = False 

528 if m: 528 ↛ 529line 528 didn't jump to line 529 because the condition on line 528 was never true

529 par = m.group(1) 

530 cls = classify_desc(par) 

531 if cls == "tags": 

532 parse_pronunciation_tags(wxr, par, audio) 

533 else: 

534 skip = True 

535 if skip: 535 ↛ 536line 535 didn't jump to line 536 because the condition on line 535 was never true

536 return "" 

537 audios.append(audio) 

538 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__" 

539 if name == "audio-IPA": 539 ↛ 540line 539 didn't jump to line 540 because the condition on line 539 was never true

540 filename = ht.get(2) or "" 

541 ipa = ht.get(3) or "" 

542 dial = ht.get("dial") 

543 audio = {"audio": filename.strip()} 

544 if dial: 

545 dial = clean_node(wxr, None, [dial]) 

546 audio["text"] = dial 

547 if ipa: 

548 audio["audio-ipa"] = ipa 

549 audios.append(audio) 

550 # The problem with these IPAs is that they often just describe 

551 # what's in the sound file, rather than giving the pronunciation 

552 # of the word alone. It is common for audio files to contain 

553 # multiple pronunciations or articles in the same file, and then 

554 # this IPA often describes what is in the file. 

555 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__" 

556 if name == "audio-pron": 

557 filename = ht.get(2) or "" 

558 ipa = ht.get("ipa") or "" 

559 dial = ht.get("dial") 

560 country = ht.get("country") 

561 audio = {"audio": filename.strip()} 

562 if dial: 562 ↛ 566line 562 didn't jump to line 566 because the condition on line 562 was always true

563 dial = clean_node(wxr, None, [dial]) 

564 audio["text"] = dial 

565 parse_pronunciation_tags(wxr, dial, audio) 

566 if country: 566 ↛ 568line 566 didn't jump to line 568 because the condition on line 566 was always true

567 parse_pronunciation_tags(wxr, country, audio) 

568 if ipa: 568 ↛ 570line 568 didn't jump to line 570 because the condition on line 568 was always true

569 audio["audio-ipa"] = ipa 

570 audios.append(audio) 

571 # XXX do we really want to extract pronunciations from these? 

572 # Or are they spurious / just describing what is in the 

573 # audio file? 

574 # if ipa: 

575 # pron = {"ipa": ipa} 

576 # if dial: 

577 # parse_pronunciation_tags(wxr, dial, pron) 

578 # if country: 

579 # parse_pronunciation_tags(wxr, country, pron) 

580 # data_append(data, "sounds", pron) 

581 return "__AUDIO_IGNORE_THIS__" + str(len(audios) - 1) + "__" 

582 if name in ("hyph", "hyphenation"): 

583 # {{hyph|en|re|late|caption="Hyphenation UK:"}} 

584 # {{hyphenation|it|quiè|to||qui|è|to||quié|to||qui|é|to}} 

585 # and also nocaption=1 

586 caption = clean_node(wxr, None, ht.get("caption", "")) 

587 tagsets, _ = decode_tags(caption) 

588 # flatten the tagsets into one; it would be really weird to have 

589 # several tagsets for a hyphenation caption 

590 tags = sorted(set(tag for tagset in tagsets for tag in tagset)) 

591 # We'll just ignore any errors from tags, it's not very important 

592 # for hyphenation 

593 tags = [tag for tag in tags if not tag.startswith("error")] 

594 hyph_sequences: list[list[str]] = [[]] 

595 for text in [ 

596 t for (k, t) in ht.items() if (isinstance(k, int) and k >= 2) 

597 ]: 

598 if not text: 

599 hyph_sequences.append([]) 

600 else: 

601 hyph_sequences[-1].append(clean_node(wxr, None, text)) 

602 for seq in hyph_sequences: 

603 hyphenations.append(Hyphenation(parts=seq, tags=tags)) 

604 return "" 

605 return None 

606 

607 may_be_duplicates = False 

608 

609 def parse_pron_post_template_fn( 

610 name: str, ht: TemplateArgs, text: str 

611 ) -> str | None: 

612 # _post_template_fn handles templates *after* the work to expand 

613 # them has been done; this is exactly the same as _template_fn, 

614 # except with the additional expanded text as an input, and 

615 # possible side-effects from the expansion and recursion (like 

616 # calling other subtemplates that are handled in _template_fn. 

617 nonlocal may_be_duplicates 

618 if is_panel_template(wxr, name): 618 ↛ 619line 618 didn't jump to line 619 because the condition on line 618 was never true

619 return "" 

620 if name in PRON_POS_TEMPLATE_NAMES: 

621 pos_match = extract_pronunciation_pos_template( 

622 wxr, name, ht, lang_code 

623 ) 

624 if pos_match.pos_values: 

625 pron_pos_markers.append(pos_match.pos_values) 

626 marker = ( 

627 f"__PRON_POS_MARKER_{len(pron_pos_markers) - 1}__" 

628 ) 

629 if pos_match.residual: 629 ↛ 630line 629 didn't jump to line 630 because the condition on line 629 was never true

630 return f"{marker} ({pos_match.residual})" 

631 return marker 

632 if name in { 

633 *PRON_POS_TEMPLATE_NAMES, 

634 "l", 

635 "link", 

636 }: 

637 # Kludge: when these templates expand to /.../ or [...], 

638 # replace the expansion by something safe. This is used 

639 # to filter spurious IPA-looking expansions that aren't really 

640 # IPAs. We probably don't care about these templates in the 

641 # contexts where they expand to something containing these. 

642 v = re.sub(r'href="[^"]*"', "", text) # Ignore URLs 

643 v = re.sub(r'src="[^"]*"', "", v) 

644 v = clean_value(wxr, v) 

645 if re.search(r"/[^/,]+?/|\[[^]0-9,/][^],/]*?\]", v): 

646 # Note: replacing by empty results in Lua errors that we 

647 # would rather not have. For example, voi/Middle Vietnamese 

648 # uses {{a|{{l{{vi|...}}}}, and the {{a|...}} will fail 

649 # if {{l|...}} returns empty. 

650 return "stripped-by-parse_pron_post_template_fn" 

651 if name in ("IPA", "enPR"): 

652 # Extract the data from IPA and enPR templates (same underlying 

653 # template) and replace them in-text with magical cookie that 

654 # can be later used to refer to the data's index inside 

655 # pron_templates. 

656 if pron_t := extract_pron_template(wxr, name, ht, text): 

657 pron_templates.append(pron_t) 

658 return f"__PRON_TEMPLATE_{len(pron_templates) - 1}__" 

659 # Catch templates that generate duplicate sound data entries 

660 # here; if the text produces a big, toggleable section, the 

661 # "header" for that section might be duplicated. Add more conditions 

662 # if necessary. 

663 if text.startswith("<") and "vsToggleElement" in text: 663 ↛ 664line 663 didn't jump to line 664 because the condition on line 663 was never true

664 may_be_duplicates = True 

665 return text 

666 

667 def flattened_tree( 

668 lines: list[WikiNode | str], 

669 ) -> Iterator[FlattenedListNode]: 

670 assert isinstance(lines, list) 

671 for line in lines: 

672 yield from flattened_tree1(line, 0) 

673 

674 def flattened_tree1( 

675 node: WikiNode | str, list_depth: int 

676 ) -> Iterator[FlattenedListNode]: 

677 assert isinstance(node, (WikiNode, str)) 

678 if isinstance(node, str): 

679 yield FlattenedListNode(node, list_depth) 

680 return 

681 elif node.kind == NodeKind.LIST: 

682 for item in node.children: 

683 yield from flattened_tree1(item, list_depth) 

684 elif node.kind == NodeKind.LIST_ITEM: 

685 item_depth = ( 

686 len(node.sarg) if isinstance(node.sarg, str) else list_depth 

687 ) 

688 new_children = [] 

689 # A list item can have several sublists, e.g. "**" lines 

690 # followed by a "*:" line. 

691 sublists = [] 

692 for child in node.children: 

693 if isinstance(child, WikiNode) and child.kind == NodeKind.LIST: 

694 sublists.append(child) 

695 else: 

696 new_children.append(child) 

697 node.children = new_children 

698 node.sarg = "*" 

699 yield FlattenedListNode(node, item_depth) 

700 for sublist in sublists: 

701 yield from flattened_tree1(sublist, item_depth) 

702 else: 

703 yield FlattenedListNode(node, list_depth) 

704 

705 # XXX Do not use flattened_tree more than once here, for example for 

706 # debug printing... The underlying data is changed, and the separated 

707 # sublists disappear. 

708 

709 # Kludge for templates that generate several lines, but haven't 

710 # been caught by earlier kludges... 

711 def split_cleaned_node_on_newlines( 

712 contents: list[WikiNode | str], 

713 ) -> Iterator[tuple[str, int]]: 

714 for flattened in flattened_tree(contents): 

715 ipa_text = clean_node( 

716 wxr, 

717 data, 

718 flattened.node, 

719 template_fn=parse_pronunciation_template_fn, 

720 post_template_fn=parse_pron_post_template_fn, 

721 ) 

722 for line in ipa_text.splitlines(): 

723 yield line, flattened.list_depth 

724 

725 # have_pronunciations = False 

726 active_pos: PronunciationPoses | None = None 

727 # POS values from parent pronunciation lines by original list depth. 

728 # Audio-only child lines can inherit from a parent pronunciation line, 

729 # but same-depth audio lines must not inherit from a preceding IPA. 

730 pronunciation_pos_stack: list[tuple[int, PronunciationPoses]] = [] 

731 

732 def parent_pronunciation_pos( 

733 list_depth: int, 

734 ) -> PronunciationPoses | None: 

735 if not pronunciation_pos_stack: 

736 return None 

737 parent_depth, parent_pos = pronunciation_pos_stack[-1] 

738 return parent_pos if parent_depth < list_depth else None 

739 

740 # Tags from label-only lines by original list depth, e.g. 

741 # "* {{a|en|GA}}" over "** {{IPA|en|...}}". Nested pronunciations 

742 # start from their parent's tags and add their own. 

743 pronunciation_tags_stack: list[tuple[int, SoundData]] = [] 

744 

745 def parent_pronunciation_tags(list_depth: int) -> SoundData: 

746 if not pronunciation_tags_stack: 

747 return {} 

748 parent_depth, parent_tags = pronunciation_tags_stack[-1] 

749 return parent_tags if parent_depth < list_depth else {} 

750 

751 for line, list_depth in split_cleaned_node_on_newlines(contents): 

752 prefix: str | None = None 

753 earlier_base_data: SoundData | None = None 

754 line_pos: PronunciationPoses | None = None 

755 current_group_sounds: list[SoundData] = [] 

756 # POS values seen on sounds extracted from this physical line. A 

757 # single candidate can seed adjacent audio-only child lines; multiple 

758 # POS-marked sounds on one line are too ambiguous for inheritance. 

759 line_sound_pos_candidates: set[PronunciationPoses] = set() 

760 line_has_sound = False 

761 if not line: 761 ↛ 762line 761 didn't jump to line 762 because the condition on line 761 was never true

762 continue 

763 while ( 

764 pronunciation_pos_stack 

765 and pronunciation_pos_stack[-1][0] >= list_depth 

766 ): 

767 pronunciation_pos_stack.pop() 

768 while ( 

769 pronunciation_tags_stack 

770 and pronunciation_tags_stack[-1][0] >= list_depth 

771 ): 

772 pronunciation_tags_stack.pop() 

773 parent_tags = parent_pronunciation_tags(list_depth) 

774 

775 split_templates = re.split(r"__PRON_TEMPLATE_(\d+)__", line) 

776 for i, text in enumerate(split_templates): 

777 if not text: 

778 continue 

779 # clean up starts at the start of the line 

780 text = re.sub(r"^\**\s*", "", text).strip() 

781 if i == 0: 

782 # At the start of a line, check for stuff like "Noun:" 

783 # or "(verb)" for POS labels that apply to this line or 

784 # structurally nested pronunciation lines. 

785 # These labels feed the inheritance state that later sets the 

786 # temporary sound["pos"] field used to route pronunciation 

787 # data into matching POS sections. 

788 if pos_prefix := extract_pos_prefix(text): 

789 text = pos_prefix.text 

790 line_pos = pos_prefix.pos_values 

791 if pos_prefix.is_persistent: 

792 active_pos = pos_prefix.pos_values 

793 if not text: 

794 continue 

795 

796 m = re.search(r"__PRON_POS_MARKER_(\d+)__", text) 

797 while m: 

798 if current_group_sounds and re.search( 

799 r"[,;]", text[: m.start()] 

800 ): 

801 current_group_sounds = [] 

802 pos_values = pron_pos_markers[int(m.group(1))] 

803 if current_group_sounds: 

804 for sound in current_group_sounds: 

805 set_sound_pos(sound, pos_values) 

806 line_sound_pos_candidates.add(pos_values) 

807 line_pos = pos_values 

808 text = text[: m.start()] + text[m.end() :] 

809 m = re.search(r"__PRON_POS_MARKER_(\d+)__", text) 

810 text = text.strip() 

811 if not text: 

812 continue 

813 # POS inheritance for normal pronunciation data: 

814 # 1. line_pos: explicit POS marker on this line, e.g. 

815 # "* {{q|noun}} {{IPA|...}}". 

816 # 2. parent_pronunciation_pos: structurally inherited from a 

817 # parent list item, e.g. "* {{q|noun}}" then "** {{IPA|...}}". 

818 # 3. active_pos: support for "* Noun:" followed by 

819 # "* {{IPA|...}}"; broad, so it stays after structural data. 

820 inherited_pos = ( 

821 line_pos or parent_pronunciation_pos(list_depth) or active_pos 

822 ) 

823 

824 if i % 2 == 1: 

825 # re.split (with capture groups) splits the lines so that 

826 # every even entry is a captured splitter; odd lines are either 

827 # empty strings or stuff around the splitters. 

828 base_pron_data, first_prons = pron_templates[int(text)] 

829 if base_pron_data: 

830 earlier_base_data = base_pron_data 

831 # print(f"Set {earlier_base_data=}") 

832 elif earlier_base_data is not None: 

833 # merge data from an earlier iteration of this loop 

834 for pr in first_prons: 

835 if "note" in pr and "note" in earlier_base_data: 835 ↛ 836line 835 didn't jump to line 836 because the condition on line 835 was never true

836 pr["note"] += ";" + earlier_base_data.get( 

837 "note", "" 

838 ) 

839 elif "note" in earlier_base_data: 839 ↛ 840line 839 didn't jump to line 840 because the condition on line 839 was never true

840 pr["note"] = earlier_base_data["note"] 

841 if "topics" in earlier_base_data: 841 ↛ 842line 841 didn't jump to line 842 because the condition on line 841 was never true

842 data_extend( 

843 pr, "topics", earlier_base_data["topics"] 

844 ) 

845 if "tags" in pr and "tags" in earlier_base_data: 845 ↛ 846line 845 didn't jump to line 846 because the condition on line 845 was never true

846 pr["tags"].extend(earlier_base_data["tags"]) 

847 elif "tags" in earlier_base_data: 847 ↛ 834line 847 didn't jump to line 834 because the condition on line 847 was always true

848 pr["tags"] = sorted(set(earlier_base_data["tags"])) 

849 for pr in first_prons: 

850 inherit_pronunciation_tag_data(pr, parent_tags) 

851 if sound_pos := set_sound_pos( 

852 pr, 

853 None if "pos" in pr else inherited_pos, 

854 ): 

855 line_sound_pos_candidates.add(sound_pos) 

856 if pr not in data.get("sounds", ()): 856 ↛ 858line 856 didn't jump to line 858 because the condition on line 856 was always true

857 data_append(data, "sounds", pr) 

858 current_group_sounds.append(pr) 

859 line_has_sound = True 

860 # This bit is handled 

861 continue 

862 

863 if "IPA" in text: 

864 field: Literal[ 

865 "audio", 

866 "audio-ipa", 

867 "enpr", 

868 "form", 

869 "hangeul", 

870 "homophone", 

871 "ipa", 

872 "mp3_url", 

873 "note", 

874 "ogg_url", 

875 "other", 

876 "rhymes", 

877 "tags", 

878 "text", 

879 "topics", 

880 "zh-pron", 

881 ] = "ipa" 

882 else: 

883 # This is used for Rhymes, Homophones, etc 

884 field = "other" 

885 

886 # Check if it contains Japanese "Tokyo" pronunciation with 

887 # special syntax 

888 m = re.search(r"(?m)\(Tokyo\) +([^ ]+) +\[", text) 

889 if m: 889 ↛ 890line 889 didn't jump to line 890 because the condition on line 889 was never true

890 pron: SoundData = {field: m.group(1)} # type: ignore[misc] 

891 if sound_pos := set_sound_pos(pron, inherited_pos): 

892 line_sound_pos_candidates.add(sound_pos) 

893 data_append(data, "sounds", pron) 

894 current_group_sounds.append(pron) 

895 line_has_sound = True 

896 # have_pronunciations = True 

897 continue 

898 

899 # Check if it contains Rhymes 

900 m = re.match(r"\s*Rhymes?: (.*)", text) 

901 if m: 

902 for ending in split_at_comma_semi(m.group(1)): 

903 ending = ending.strip() 

904 if ending: 904 ↛ 902line 904 didn't jump to line 902 because the condition on line 904 was always true

905 pron = {"rhymes": ending} 

906 if sound_pos := set_sound_pos(pron, inherited_pos): 

907 line_sound_pos_candidates.add(sound_pos) 

908 data_append(data, "sounds", pron) 

909 current_group_sounds.append(pron) 

910 line_has_sound = True 

911 # have_pronunciations = True 

912 continue 

913 

914 # Check if it contains homophones 

915 m = re.search(r"(?m)\bHomophones?: (.*)", text) 

916 if m: 

917 for w in split_at_comma_semi(m.group(1)): 

918 w = w.strip() 

919 if w: 919 ↛ 917line 919 didn't jump to line 917 because the condition on line 919 was always true

920 pron = {"homophone": w} 

921 if sound_pos := set_sound_pos(pron, inherited_pos): 

922 line_sound_pos_candidates.add(sound_pos) 

923 data_append(data, "sounds", pron) 

924 current_group_sounds.append(pron) 

925 line_has_sound = True 

926 # have_pronunciations = True 

927 continue 

928 

929 # Check if it contains Phonetic hangeul 

930 m = re.search(r"(?m)\bPhonetic hange?ul: \[([^]]+)\]", text) 

931 if m: 931 ↛ 932line 931 didn't jump to line 932 because the condition on line 931 was never true

932 seen = set() 

933 for w in m.group(1).split("/"): 

934 w = w.strip() 

935 if w and w not in seen: 

936 seen.add(w) 

937 pron = {"hangeul": w} 

938 if sound_pos := set_sound_pos(pron, inherited_pos): 

939 line_sound_pos_candidates.add(sound_pos) 

940 data_append(data, "sounds", pron) 

941 current_group_sounds.append(pron) 

942 line_has_sound = True 

943 # have_pronunciations = True 

944 

945 # This regex-based hyphenation detection left as backup 

946 m = re.search(r"\b(Syllabification|Hyphenation): *([^\n.]*)", text) 

947 if m: 

948 data_append(data, "hyphenation", m.group(2)) 

949 commaseparated = m.group(2).split(",") 

950 if len(commaseparated) > 1: 950 ↛ 961line 950 didn't jump to line 961 because the condition on line 950 was always true

951 for h in commaseparated: 

952 # That second characters looks like a dash but it's 

953 # actually unicode decimal code 8231, hyphenation dash 

954 # Add more delimiters here if needed. 

955 parts = re.split(r"-|‧", h.strip()) 

956 data_append( 

957 data, "hyphenations", Hyphenation(parts=parts) 

958 ) 

959 ... 

960 else: 

961 data_append( 

962 data, 

963 "hyphenations", 

964 Hyphenation(parts=m.group(2).split(sep="-")), 

965 ) 

966 # have_pronunciations = True 

967 

968 # See if it contains a word prefix restricting which forms the 

969 # pronunciation applies to (see amica/Latin) and/or parenthesized 

970 # tags. 

971 m = re.match( 

972 r"^[*#\s]*(([-\w]+):\s+)?\((([^()]|\([^()]*\))*?)\)", text 

973 ) 

974 if m: 

975 prefix = m.group(2) or "" 

976 tagstext = m.group(3) 

977 text = text[m.end() :] 

978 else: 

979 m = re.match(r"^[*#\s]*([-\w]+):\s+", text) 

980 if m: 

981 prefix = m.group(1) 

982 tagstext = "" 

983 text = text[m.end() :] 

984 else: 

985 # Spanish has tags before pronunciations, eg. aceite/Spanish 

986 m = re.match(r".*:\s+\(([^)]*)\)\s+(.*)", text) 

987 if m: 987 ↛ 988line 987 didn't jump to line 988 because the condition on line 987 was never true

988 tagstext = m.group(1) 

989 text = m.group(2) 

990 else: 

991 # No prefix. In this case, we inherit prefix 

992 # from previous entry. This particularly 

993 # applies for nested Audio files. 

994 tagstext = "" 

995 if tagstext: 

996 earlier_base_data = {} 

997 parse_pronunciation_tags_with_pos( 

998 wxr, tagstext, earlier_base_data 

999 ) 

1000 

1001 # Find romanizations from the pronunciation section (routinely 

1002 # produced for Korean by {{ko-IPA}}) 

1003 for m in re.finditer(pron_romanization_re, text): 1003 ↛ 1004line 1003 didn't jump to line 1004 because the loop on line 1003 never started

1004 prefix = m.group(1) 

1005 w = m.group(2).strip() 

1006 tag = pron_romanizations[prefix] 

1007 form = {"form": w, "tags": tag.split()} 

1008 data_append(data, "forms", form) 

1009 

1010 # Find IPA pronunciations 

1011 for m in re.finditer( 

1012 r"(?m)/[^][\n/,]+?/" r"|" r"\[[^]\n0-9,/][^],/]*?\]", text 

1013 ): 

1014 v = m.group(0) 

1015 # The regexp above can match file links. Skip them. 

1016 if v.startswith("[[File:"): 1016 ↛ 1017line 1016 didn't jump to line 1017 because the condition on line 1016 was never true

1017 continue 

1018 if v == "/wiki.local/": 1018 ↛ 1019line 1018 didn't jump to line 1019 because the condition on line 1018 was never true

1019 continue 

1020 if field == "ipa" and "__AUDIO_IGNORE_THIS__" in text: 1020 ↛ 1021line 1020 didn't jump to line 1021 because the condition on line 1020 was never true

1021 m = re.search(r"__AUDIO_IGNORE_THIS__(\d+)__", text) 

1022 assert m 

1023 idx = int(m.group(1)) 

1024 if idx >= len(audios): 

1025 continue 

1026 if not audios[idx].get("audio-ipa"): 

1027 audios[idx]["audio-ipa"] = v 

1028 if prefix: 

1029 audios[idx]["form"] = prefix 

1030 else: 

1031 if earlier_base_data: 

1032 pron = deepcopy(earlier_base_data) 

1033 pron[field] = v 

1034 else: 

1035 pron = {field: v} # type: ignore[misc] 

1036 if prefix: 

1037 pron["form"] = prefix 

1038 inherit_pronunciation_tag_data(pron, parent_tags) 

1039 if sound_pos := set_sound_pos( 

1040 pron, 

1041 None if "pos" in pron else inherited_pos, 

1042 ): 

1043 line_sound_pos_candidates.add(sound_pos) 

1044 if may_be_duplicates is True: 1044 ↛ 1045line 1044 didn't jump to line 1045 because the condition on line 1044 was never true

1045 ok = True 

1046 for comp_sound in data.get("sounds", []): 

1047 # Python has dict comparison since 3.8 

1048 if pron == comp_sound: 

1049 ok = False 

1050 break 

1051 if ok: 

1052 data_append(data, "sounds", pron) 

1053 else: 

1054 data_append(data, "sounds", pron) 

1055 current_group_sounds.append(pron) 

1056 line_has_sound = True 

1057 # have_pronunciations = True 

1058 if current_group_sounds and re.search(r"[,;]", text): 

1059 current_group_sounds = [] 

1060 # XXX what about {{hyphenation|...}}, {{hyph|...}} 

1061 # and those used to be stored under "hyphenation" 

1062 

1063 # Add data that was collected in template_fn 

1064 # POS inheritance for audio has one extra source: 

1065 # common_sound_pos(line_sound_pos_candidates), from pronunciations 

1066 # extracted earlier on the same physical line, e.g. 

1067 # "* {{IPA|en|/foo/|a=verb}} {{audio|en|foo.wav}}". 

1068 # Explicit line_pos still wins, then same-line sound agreement, then 

1069 # parent-list structure, then active_pos. 

1070 audio_inherited_pos = ( 

1071 line_pos 

1072 or common_sound_pos(line_sound_pos_candidates) 

1073 or parent_pronunciation_pos(list_depth) 

1074 or active_pos 

1075 ) 

1076 for audio in audios: 

1077 if "audio" in audio: 1077 ↛ 1134line 1077 didn't jump to line 1134 because the condition on line 1077 was always true

1078 # Compute audio file URLs 

1079 fn = audio["audio"] 

1080 # Strip certain characters, e.g., left-to-right mark 

1081 fn = re.sub(r"[\u200f\u200e]", "", fn) 

1082 fn = fn.strip() 

1083 fn = urllib.parse.unquote(fn) 

1084 # First character is usually uppercased 

1085 if re.match(r"^[a-z][a-z]+", fn): 

1086 fn = fn[0].upper() + fn[1:] 

1087 if fn in wxr.config.redirects: 1087 ↛ 1088line 1087 didn't jump to line 1088 because the condition on line 1087 was never true

1088 fn = wxr.config.redirects[fn] 

1089 # File extension is lowercased 

1090 # XXX some words seem to need this, some don't seem to 

1091 # have this??? what is the exact rule? 

1092 # fn = re.sub(r"\.[^.]*$", lambda m: m.group(0).lower(), fn) 

1093 # Spaces are converted to underscores 

1094 fn = re.sub(r"\s+", "_", fn) 

1095 # Compute hash digest part 

1096 h = hashlib.md5() 

1097 hname = fn.encode("utf-8") 

1098 h.update(hname) 

1099 digest = h.hexdigest() 

1100 # Quote filename for URL 

1101 qfn = urllib.parse.quote(fn) 

1102 # For safety when writing files 

1103 qfn = qfn.replace("/", "__slash__") 

1104 if re.search(r"(?i)\.(ogg|oga)$", fn): 

1105 ogg = ( 

1106 "https://upload.wikimedia.org/wikipedia/" 

1107 "commons/{}/{}/{}".format(digest[:1], digest[:2], qfn) 

1108 ) 

1109 else: 

1110 ogg = ( 

1111 "https://upload.wikimedia.org/wikipedia/" 

1112 "commons/transcoded/" 

1113 "{}/{}/{}/{}.ogg".format( 

1114 digest[:1], digest[:2], qfn, qfn 

1115 ) 

1116 ) 

1117 if re.search(r"(?i)\.(mp3)$", fn): 1117 ↛ 1118line 1117 didn't jump to line 1118 because the condition on line 1117 was never true

1118 mp3 = ( 

1119 "https://upload.wikimedia.org/wikipedia/" 

1120 "commons/{}/{}/{}".format(digest[:1], digest[:2], qfn) 

1121 ) 

1122 else: 

1123 mp3 = ( 

1124 "https://upload.wikimedia.org/wikipedia/" 

1125 "commons/transcoded/" 

1126 "{}/{}/{}/{}.mp3".format( 

1127 digest[:1], digest[:2], qfn, qfn 

1128 ) 

1129 ) 

1130 audio["ogg_url"] = ogg 

1131 audio["mp3_url"] = mp3 

1132 if "pos" not in audio: 1132 ↛ 1134line 1132 didn't jump to line 1134 because the condition on line 1132 was always true

1133 set_sound_pos(audio, audio_inherited_pos) 

1134 if audio not in data.get("sounds", ()): 

1135 data_append(data, "sounds", audio) 

1136 line_has_sound = True 

1137 

1138 # if audios: 

1139 # have_pronunciations = True 

1140 audios = [] 

1141 

1142 data_extend(data, "hyphenations", hyphenations) 

1143 hyphenations = [] 

1144 

1145 if line_pos and not line_has_sound: 

1146 active_pos = line_pos 

1147 pronunciation_pos_stack.append((list_depth, line_pos)) 

1148 elif line_pronunciation_pos := common_sound_pos( 

1149 line_sound_pos_candidates 

1150 ): 

1151 pronunciation_pos_stack.append((list_depth, line_pronunciation_pos)) 

1152 

1153 if not line_has_sound and earlier_base_data: 

1154 line_tags: SoundData = {} 

1155 merge_pronunciation_tag_data(line_tags, earlier_base_data) 

1156 if line_tags: 1156 ↛ 751line 1156 didn't jump to line 751 because the condition on line 1156 was always true

1157 inherit_pronunciation_tag_data(line_tags, parent_tags) 

1158 pronunciation_tags_stack.append((list_depth, line_tags)) 

1159 

1160 ## I have commented out the otherwise unused have_pronunciation 

1161 ## toggles; uncomment them to use this debug print 

1162 # if not have_pronunciations and not have_panel_templates: 

1163 # wxr.wtp.debug("no pronunciations found from pronunciation section", 

1164 # sortid="pronunciations/533") 

1165 

1166 

1167def extract_th_pron_template( 

1168 wxr: WiktextractContext, word_entry: WordData, t_node: TemplateNode 

1169): 

1170 # https://en.wiktionary.org/wiki/Template:th-pron 

1171 @dataclass 

1172 class TableHeader: 

1173 raw_tags: list[str] 

1174 rowspan: int 

1175 

1176 expanded_node = wxr.wtp.parse( 

1177 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

1178 ) 

1179 sounds = [] 

1180 for table_tag in expanded_node.find_html("table"): 

1181 row_headers = [] 

1182 for tr_tag in table_tag.find_html("tr"): 

1183 field = "other" 

1184 new_headers = [] 

1185 for header in row_headers: 

1186 if header.rowspan > 1: 

1187 header.rowspan -= 1 

1188 new_headers.append(header) 

1189 row_headers = new_headers 

1190 for th_tag in tr_tag.find_html("th"): 

1191 header_str = clean_node(wxr, None, th_tag) 

1192 if header_str.startswith("(standard) IPA"): 

1193 field = "ipa" 

1194 elif header_str.startswith("Homophones"): 1194 ↛ 1195line 1194 didn't jump to line 1195 because the condition on line 1194 was never true

1195 field = "homophone" 

1196 elif header_str == "Audio": 

1197 field = "audio" 

1198 elif header_str != "": 1198 ↛ 1190line 1198 didn't jump to line 1190 because the condition on line 1198 was always true

1199 rowspan = 1 

1200 rowspan_str = th_tag.attrs.get("rowspan", "1") 

1201 if re.fullmatch(r"\d+", rowspan_str): 1201 ↛ 1203line 1201 didn't jump to line 1203 because the condition on line 1201 was always true

1202 rowspan = int(rowspan_str) 

1203 header = TableHeader([], rowspan) 

1204 for line in header_str.splitlines(): 

1205 for raw_tag in line.strip("{}\n ").split(";"): 

1206 raw_tag = raw_tag.strip() 

1207 if raw_tag != "": 1207 ↛ 1205line 1207 didn't jump to line 1205 because the condition on line 1207 was always true

1208 header.raw_tags.append(raw_tag) 

1209 row_headers.append(header) 

1210 

1211 for td_tag in tr_tag.find_html("td"): 

1212 if field == "audio": 

1213 for link_node in td_tag.find_child(NodeKind.LINK): 

1214 filename = clean_node(wxr, None, link_node.largs[0]) 

1215 if filename != "": 1215 ↛ 1213line 1215 didn't jump to line 1213 because the condition on line 1215 was always true

1216 sound = create_audio_url_dict(filename) 

1217 sounds.append(sound) 

1218 elif field == "homophone": 1218 ↛ 1219line 1218 didn't jump to line 1219 because the condition on line 1218 was never true

1219 for span_tag in td_tag.find_html_recursively( 

1220 "span", attr_name="lang", attr_value="th" 

1221 ): 

1222 word = clean_node(wxr, None, span_tag) 

1223 if word != "": 

1224 sounds.append({"homophone": word}) 

1225 else: 

1226 raw_tags = [] 

1227 for html_node in td_tag.find_child_recursively( 

1228 NodeKind.HTML 

1229 ): 

1230 if html_node.tag == "small": 

1231 node_str = clean_node(wxr, None, html_node) 

1232 if node_str.startswith("[") and node_str.endswith( 

1233 "]" 

1234 ): 

1235 for raw_tag in node_str.strip("[]").split(","): 

1236 raw_tag = raw_tag.strip() 

1237 if raw_tag != "": 1237 ↛ 1235line 1237 didn't jump to line 1235 because the condition on line 1237 was always true

1238 raw_tags.append(raw_tag) 

1239 elif len(sounds) > 0: 1239 ↛ 1227line 1239 didn't jump to line 1227 because the condition on line 1239 was always true

1240 sounds[-1]["roman"] = node_str 

1241 elif html_node.tag == "span": 

1242 node_str = clean_node(wxr, None, html_node) 

1243 span_lang = html_node.attrs.get("lang", "") 

1244 span_class = html_node.attrs.get("class", "") 

1245 if node_str != "" and ( 

1246 span_lang == "th" or span_class in ["IPA", "tr"] 

1247 ): 

1248 sound = {} 

1249 for raw_tag in raw_tags: 

1250 if raw_tag in valid_tags: 1250 ↛ 1253line 1250 didn't jump to line 1253 because the condition on line 1250 was always true

1251 data_append(sound, "tags", raw_tag) 

1252 else: 

1253 data_append(sound, "raw_tags", raw_tag) 

1254 for header in row_headers: 

1255 for raw_tag in header.raw_tags: 

1256 if raw_tag.lower() in valid_tags: 

1257 data_append( 

1258 sound, "tags", raw_tag.lower() 

1259 ) 

1260 else: 

1261 data_append( 

1262 sound, "raw_tags", raw_tag 

1263 ) 

1264 if "romanization" in sound.get("tags", []): 

1265 field = "roman" 

1266 sound[field] = node_str 

1267 sounds.append(sound) 

1268 

1269 clean_node(wxr, word_entry, expanded_node) 

1270 data_extend(word_entry, "sounds", sounds) 

1271 

1272 

1273def extract_zh_pron_template( 

1274 wxr: WiktextractContext, word_entry: WordData, t_node: TemplateNode 

1275): 

1276 # https://en.wiktionary.org/wiki/Template:zh-pron 

1277 expanded_node = wxr.wtp.parse( 

1278 wxr.wtp.node_to_wikitext(t_node), expand_all=True 

1279 ) 

1280 seen_lists = set() 

1281 sounds = [] 

1282 for list_node in expanded_node.find_child_recursively(NodeKind.LIST): 

1283 if list_node not in seen_lists: 

1284 for list_item in list_node.find_child(NodeKind.LIST_ITEM): 

1285 sounds.extend( 

1286 extract_zh_pron_list_item(wxr, list_item, [], seen_lists) 

1287 ) 

1288 clean_node(wxr, word_entry, expanded_node) 

1289 data_extend(word_entry, "sounds", sounds) 

1290 

1291 

1292def extract_zh_pron_list_item( 

1293 wxr: WiktextractContext, 

1294 list_item: WikiNode, 

1295 raw_tags: list[str], 

1296 seen_lists: set[WikiNode], 

1297) -> list[SoundData]: 

1298 current_tags = raw_tags[:] 

1299 sounds = [] 

1300 is_first_small_tag = True 

1301 for node in list_item.children: 

1302 if isinstance(node, WikiNode) and node.kind == NodeKind.LINK: 

1303 link_str = clean_node(wxr, None, node.largs) 

1304 node_str = clean_node(wxr, None, node) 

1305 if link_str.startswith("File:"): 1305 ↛ 1306line 1305 didn't jump to line 1306 because the condition on line 1305 was never true

1306 sound = create_audio_url_dict(link_str.removeprefix("File:")) 

1307 sound["raw_tags"] = current_tags[:] 

1308 translate_zh_pron_raw_tags(sound) 

1309 sounds.append(sound) 

1310 elif node_str != "": 1310 ↛ 1301line 1310 didn't jump to line 1301 because the condition on line 1310 was always true

1311 current_tags.append(node_str) 

1312 elif isinstance(node, HTMLNode): 

1313 if node.tag == "small": 

1314 if is_first_small_tag: 1314 ↛ 1325line 1314 didn't jump to line 1325 because the condition on line 1314 was always true

1315 raw_tag_text = clean_node( 

1316 wxr, 

1317 None, 

1318 [ 

1319 n 

1320 for n in node.children 

1321 if not (isinstance(n, HTMLNode) and n.tag == "sup") 

1322 ], 

1323 ) 

1324 current_tags.extend(split_zh_pron_raw_tag(raw_tag_text)) 

1325 elif len(sounds) > 0: 

1326 data_extend( 

1327 sounds[-1], 

1328 "raw_tags", 

1329 split_zh_pron_raw_tag(clean_node(wxr, None, node)), 

1330 ) 

1331 translate_zh_pron_raw_tags(sounds[-1]) 

1332 is_first_small_tag = False 

1333 elif node.tag == "span": 

1334 sounds.extend(extract_zh_pron_span(wxr, node, current_tags)) 

1335 elif ( 1335 ↛ 1340line 1335 didn't jump to line 1340 because the condition on line 1335 was never true

1336 node.tag == "table" 

1337 and len(current_tags) > 0 

1338 and current_tags[-1] == "Homophones" 

1339 ): 

1340 sounds.extend( 

1341 extract_zh_pron_homophone_table(wxr, node, current_tags) 

1342 ) 

1343 elif isinstance(node, WikiNode) and node.kind == NodeKind.LIST: 

1344 seen_lists.add(node) 

1345 for child_list_item in node.find_child(NodeKind.LIST_ITEM): 

1346 sounds.extend( 

1347 extract_zh_pron_list_item( 

1348 wxr, child_list_item, current_tags, seen_lists 

1349 ) 

1350 ) 

1351 

1352 return sounds 

1353 

1354 

1355def extract_zh_pron_homophone_table( 

1356 wxr: WiktextractContext, table: HTMLNode, raw_tags: list[str] 

1357) -> list[SoundData]: 

1358 sounds = [] 

1359 for td_tag in table.find_html_recursively("td"): 

1360 for span_tag in td_tag.find_html("span"): 

1361 span_class = span_tag.attrs.get("class", "") 

1362 span_lang = span_tag.attrs.get("lang", "") 

1363 span_str = clean_node(wxr, None, span_tag) 

1364 if ( 

1365 span_str not in ["", "/"] 

1366 and span_lang != "" 

1367 and span_class in ["Hant", "Hans", "Hani"] 

1368 ): 

1369 sound = {"homophone": span_str, "raw_tags": raw_tags[:]} 

1370 if span_class == "Hant": 

1371 data_append(sound, "tags", "Traditional-Chinese") 

1372 elif span_class == "Hans": 

1373 data_append(sound, "tags", "Simplified-Chinese") 

1374 translate_zh_pron_raw_tags(sound) 

1375 sounds.append(sound) 

1376 

1377 return sounds 

1378 

1379 

1380def translate_zh_pron_raw_tags(sound: SoundData): 

1381 from .zh_pron_tags import ZH_PRON_TAGS 

1382 

1383 raw_tags = [] 

1384 for raw_tag in sound.get("raw_tags", []): 

1385 if raw_tag in ZH_PRON_TAGS: 

1386 tr_tag = ZH_PRON_TAGS[raw_tag] 

1387 if isinstance(tr_tag, str): 

1388 data_append(sound, "tags", tr_tag) 

1389 elif isinstance(tr_tag, list) and tr_tag not in sound.get( 1389 ↛ 1384line 1389 didn't jump to line 1384 because the condition on line 1389 was always true

1390 "tags", [] 

1391 ): 

1392 data_extend(sound, "tags", tr_tag) 

1393 elif raw_tag in valid_tags: 

1394 if raw_tag not in sound.get("tags", []): 1394 ↛ 1384line 1394 didn't jump to line 1384 because the condition on line 1394 was always true

1395 data_append(sound, "tags", raw_tag) 

1396 elif raw_tag not in raw_tags: 1396 ↛ 1384line 1396 didn't jump to line 1384 because the condition on line 1396 was always true

1397 raw_tags.append(raw_tag) 

1398 

1399 if len(raw_tags) > 0: 

1400 sound["raw_tags"] = raw_tags 

1401 elif "raw_tags" in sound: 1401 ↛ exitline 1401 didn't return from function 'translate_zh_pron_raw_tags' because the condition on line 1401 was always true

1402 del sound["raw_tags"] 

1403 

1404 

1405def split_zh_pron_raw_tag(raw_tag_text: str) -> list[str]: 

1406 raw_tags = [] 

1407 if "(" not in raw_tag_text: 

1408 for raw_tag in re.split(r",|:|;| and ", raw_tag_text): 

1409 raw_tag = raw_tag.strip().removeprefix("incl. ").strip() 

1410 if raw_tag != "": 

1411 raw_tags.append(raw_tag) 

1412 else: 

1413 processed_offsets = [] 

1414 for match in re.finditer(r"\([^()]+\)", raw_tag_text): 

1415 processed_offsets.append((match.start(), match.end())) 

1416 raw_tags.extend( 

1417 split_zh_pron_raw_tag( 

1418 raw_tag_text[match.start() + 1 : match.end() - 1] 

1419 ) 

1420 ) 

1421 not_processed = "" 

1422 last_end = 0 

1423 for start, end in processed_offsets: 

1424 not_processed += raw_tag_text[last_end:start] 

1425 last_end = end 

1426 not_processed += raw_tag_text[last_end:] 

1427 if not_processed != raw_tag_text: 1427 ↛ 1430line 1427 didn't jump to line 1430 because the condition on line 1427 was always true

1428 raw_tags = split_zh_pron_raw_tag(not_processed) + raw_tags 

1429 else: 

1430 raw_tags.append(not_processed) 

1431 

1432 return raw_tags 

1433 

1434 

1435def extract_zh_pron_span( 

1436 wxr: WiktextractContext, span_tag: HTMLNode, raw_tags: list[str] 

1437) -> list[SoundData]: 

1438 sounds = [] 

1439 small_tags = [] 

1440 pron_nodes = [] 

1441 roman = "" 

1442 phonetic_pron = "" 

1443 for index, node in enumerate(span_tag.children): 

1444 if isinstance(node, HTMLNode) and node.tag == "small": 1444 ↛ 1445line 1444 didn't jump to line 1445 because the condition on line 1444 was never true

1445 small_tags = split_zh_pron_raw_tag(clean_node(wxr, None, node)) 

1446 elif ( 1446 ↛ 1451line 1446 didn't jump to line 1451 because the condition on line 1446 was never true

1447 isinstance(node, HTMLNode) 

1448 and node.tag == "span" 

1449 and "-Latn" in node.attrs.get("lang", "") 

1450 ): 

1451 roman = clean_node(wxr, None, node).strip("() ") 

1452 elif isinstance(node, str) and node.strip() == "[Phonetic:": 1452 ↛ 1453line 1452 didn't jump to line 1453 because the condition on line 1452 was never true

1453 phonetic_pron = clean_node( 

1454 wxr, None, span_tag.children[index + 1 :] 

1455 ).strip("] ") 

1456 break 

1457 else: 

1458 pron_nodes.append(node) 

1459 for zh_pron in split_zh_pron(clean_node(wxr, None, pron_nodes)): 

1460 zh_pron = zh_pron.strip("[]: ") 

1461 if len(zh_pron) > 0: 1461 ↛ 1459line 1461 didn't jump to line 1459 because the condition on line 1461 was always true

1462 if "IPA" in span_tag.attrs.get("class", ""): 1462 ↛ 1463line 1462 didn't jump to line 1463 because the condition on line 1462 was never true

1463 sound = {"ipa": zh_pron, "raw_tags": raw_tags[:]} 

1464 else: 

1465 sound = {"zh_pron": zh_pron, "raw_tags": raw_tags[:]} 

1466 if roman != "": 1466 ↛ 1467line 1466 didn't jump to line 1467 because the condition on line 1466 was never true

1467 sound["roman"] = roman 

1468 sounds.append(sound) 

1469 if len(sounds) > 0: 1469 ↛ 1471line 1469 didn't jump to line 1471 because the condition on line 1469 was always true

1470 data_extend(sounds[-1], "raw_tags", small_tags) 

1471 if phonetic_pron != "": 1471 ↛ 1472line 1471 didn't jump to line 1472 because the condition on line 1471 was never true

1472 sound = { 

1473 "zh_pron": phonetic_pron, 

1474 "raw_tags": raw_tags[:] + ["Phonetic"], 

1475 } 

1476 if roman != "": 

1477 sound["roman"] = roman 

1478 sounds.append(sound) 

1479 for sound in sounds: 

1480 translate_zh_pron_raw_tags(sound) 

1481 return sounds 

1482 

1483 

1484def split_zh_pron(zh_pron: str) -> list[str]: 

1485 # split by comma and other symbols that outside parentheses 

1486 parentheses = 0 

1487 pron_list = [] 

1488 pron = "" 

1489 for c in zh_pron: 

1490 if ( 

1491 (c in [",", ";", "→"] or (c == "/" and not zh_pron.startswith("/"))) 

1492 and parentheses == 0 

1493 and len(pron.strip()) > 0 

1494 ): 

1495 pron_list.append(pron.strip()) 

1496 pron = "" 

1497 elif c == "(": 

1498 parentheses += 1 

1499 pron += c 

1500 elif c == ")": 

1501 parentheses -= 1 

1502 pron += c 

1503 else: 

1504 pron += c 

1505 

1506 if pron.strip() != "": 1506 ↛ 1508line 1506 didn't jump to line 1508 because the condition on line 1506 was always true

1507 pron_list.append(pron) 

1508 return pron_list