Coverage for src/wiktextract/page.py: 88%

307 statements  

« prev     ^ index     » next       coverage.py v7.16.2, created at 2026-10-07 00:55 +0000

1# Code for parsing information from a single Wiktionary page. 

2# 

3# Copyright (c) 2018-2022 Tatu Ylonen. See file LICENSE and https://ylonen.org 

4 

5import re 

6from collections import defaultdict 

7from copy import copy 

8from typing import Any, Callable, Optional, Union 

9 

10from mediawiki_langcodes import name_to_code 

11from wikitextprocessor.core import ( 

12 NamespaceDataEntry, 

13 PostTemplateFnCallable, 

14 TemplateArgs, 

15 TemplateFnCallable, 

16) 

17from wikitextprocessor.node_expand import NodeHandlerFnCallable 

18from wikitextprocessor.parser import GeneralNode, NodeKind, WikiNode 

19 

20from .clean import clean_value, remove_invisible_markup 

21from .datautils import data_append, data_extend 

22from .import_utils import import_extractor_module 

23from .wxr_context import WiktextractContext 

24 

25# NodeKind values for subtitles 

26LEVEL_KINDS = { 

27 NodeKind.LEVEL2, 

28 NodeKind.LEVEL3, 

29 NodeKind.LEVEL4, 

30 NodeKind.LEVEL5, 

31 NodeKind.LEVEL6, 

32} 

33 

34 

35def parse_page( 

36 wxr: WiktextractContext, page_title: str, page_text: str 

37) -> list[dict[str, Any]]: 

38 """Parses the text of a Wiktionary page and returns a list of 

39 dictionaries, one for each word/part-of-speech defined on the page 

40 for the languages specified by ``capture_language_codes`` (None means 

41 all available languages). ``word`` is page title, and ``text`` is 

42 page text in Wikimedia format. Other arguments indicate what is 

43 captured.""" 

44 page_extractor_mod = import_extractor_module(wxr.wtp.lang_code, "page") 

45 page_data = page_extractor_mod.parse_page(wxr, page_title, page_text) 

46 if wxr.config.extract_thesaurus_pages: 

47 inject_linkages(wxr, page_data) 

48 if wxr.config.dump_file_lang_code == "en": 

49 process_categories(wxr, page_data) 

50 remove_duplicate_data(page_data) 

51 return page_data 

52 

53 

54def is_panel_template(wxr: WiktextractContext, template_name: str) -> bool: 

55 """Checks if `Template_name` is a known panel template name (i.e., one that 

56 produces an infobox in Wiktionary, but this also recognizes certain other 

57 templates that we do not wish to expand).""" 

58 page_extractor_mod = import_extractor_module(wxr.wtp.lang_code, "page") 

59 if ( 

60 hasattr(page_extractor_mod, "PANEL_TEMPLATES") 

61 and template_name in page_extractor_mod.PANEL_TEMPLATES 

62 ): 

63 return True 

64 if hasattr( 

65 page_extractor_mod, "PANEL_PREFIXES" 

66 ) and template_name.startswith(tuple(page_extractor_mod.PANEL_PREFIXES)): 

67 return True 

68 return False 

69 

70 

71def recursively_extract( 

72 contents: Union[WikiNode, str, list[Union[str, WikiNode]]], 

73 fn: Callable[[Union[WikiNode, list[WikiNode]]], bool], 

74) -> tuple[list[Union[str, WikiNode]], list[Union[str, WikiNode]]]: 

75 """Recursively extracts elements from contents for which ``fn`` returns 

76 True. This returns two lists, the extracted elements and the remaining 

77 content (with the extracted elements removed at each level). Only 

78 WikiNode objects can be extracted.""" 

79 # If contents is a list, process each element separately 

80 extracted = [] 

81 new_contents = [] 

82 if isinstance(contents, (list, tuple)): 

83 for x in contents: 

84 e1, c1 = recursively_extract(x, fn) 

85 extracted.extend(e1) 

86 new_contents.extend(c1) 

87 return extracted, new_contents 

88 # If content is not WikiNode, just return it as new contents. 

89 if not isinstance(contents, WikiNode): 

90 return [], [contents] 

91 # Check if this content should be extracted 

92 if fn(contents): 

93 return [contents], [] 

94 # Otherwise content is WikiNode, and we must recurse into it. 

95 kind = contents.kind 

96 new_node = copy(contents) 

97 new_node.children = [] 

98 new_node.sarg = "" 

99 new_node.largs = [] 

100 new_node.attrs = {} 

101 new_contents.append(new_node) 

102 if kind in LEVEL_KINDS or kind == NodeKind.LINK: 

103 # Process args and children 

104 new_args = [] 

105 for arg in contents.largs: 

106 e1, c1 = recursively_extract(arg, fn) 

107 new_args.append(c1) 

108 extracted.extend(e1) 

109 new_node.largs = new_args 

110 e1, c1 = recursively_extract(contents.children, fn) 

111 extracted.extend(e1) 

112 new_node.children = c1 

113 elif kind in { 

114 NodeKind.ITALIC, 

115 NodeKind.BOLD, 

116 NodeKind.TABLE, 

117 NodeKind.TABLE_CAPTION, 

118 NodeKind.TABLE_ROW, 

119 NodeKind.TABLE_HEADER_CELL, 

120 NodeKind.TABLE_CELL, 

121 NodeKind.PRE, 

122 NodeKind.PREFORMATTED, 

123 }: 

124 # Process only children 

125 e1, c1 = recursively_extract(contents.children, fn) 

126 extracted.extend(e1) 

127 new_node.children = c1 

128 elif kind in (NodeKind.HLINE,): 128 ↛ 130line 128 didn't jump to line 130 because the condition on line 128 was never true

129 # No arguments or children 

130 pass 

131 elif kind in (NodeKind.LIST, NodeKind.LIST_ITEM): 

132 # Keep args as-is, process children 

133 new_node.sarg = contents.sarg 

134 e1, c1 = recursively_extract(contents.children, fn) 

135 extracted.extend(e1) 

136 new_node.children = c1 

137 elif kind in { 

138 NodeKind.TEMPLATE, 

139 NodeKind.TEMPLATE_ARG, 

140 NodeKind.PARSER_FN, 

141 NodeKind.URL, 

142 }: 

143 # Process only args 

144 new_args = [] 

145 for arg in contents.largs: 

146 e1, c1 = recursively_extract(arg, fn) 

147 new_args.append(c1) 

148 extracted.extend(e1) 

149 new_node.largs = new_args 

150 elif kind == NodeKind.HTML: 150 ↛ 158line 150 didn't jump to line 158 because the condition on line 150 was always true

151 # Keep attrs and args as-is, process children 

152 new_node.attrs = contents.attrs 

153 new_node.sarg = contents.sarg 

154 e1, c1 = recursively_extract(contents.children, fn) 

155 extracted.extend(e1) 

156 new_node.children = c1 

157 else: 

158 raise RuntimeError(f"recursively_extract: unhandled kind {kind}") 

159 return extracted, new_contents 

160 

161 

162def inject_linkages(wxr: WiktextractContext, page_data: list[dict]) -> None: 

163 # Inject linkages from thesaurus entries 

164 from .thesaurus import search_thesaurus 

165 

166 local_thesaurus_ns = wxr.wtp.NAMESPACE_DATA.get("Thesaurus", {}).get("name") # type: ignore[call-overload] 

167 for data in page_data: 

168 if "pos" not in data: 168 ↛ 169line 168 didn't jump to line 169 because the condition on line 168 was never true

169 continue 

170 word = data["word"] 

171 lang_code = data["lang_code"] 

172 pos = data["pos"] 

173 for term in search_thesaurus( 

174 wxr.thesaurus_db_conn, # type:ignore[arg-type] 

175 word, 

176 lang_code, 

177 pos, # type: ignore[arg-type] 

178 ): 

179 for dt in data.get(term.linkage, ()): 

180 if dt.get("word") == term.term and ( 180 ↛ 183line 180 didn't jump to line 183 because the condition on line 180 was never true

181 not term.sense or dt.get("sense") == term.sense 

182 ): 

183 break 

184 else: 

185 dt = { 

186 "word": term.term, 

187 "source": f"{local_thesaurus_ns}:{word}", 

188 } 

189 if len(term.sense) > 0: 189 ↛ 190line 189 didn't jump to line 190 because the condition on line 189 was never true

190 dt["sense"] = term.sense 

191 if len(term.tags) > 0: 191 ↛ 192line 191 didn't jump to line 192 because the condition on line 191 was never true

192 dt["tags"] = term.tags 

193 if len(term.raw_tags) > 0: 193 ↛ 194line 193 didn't jump to line 194 because the condition on line 193 was never true

194 dt["raw_tags"] = term.raw_tags 

195 if len(term.topics) > 0: 195 ↛ 196line 195 didn't jump to line 196 because the condition on line 195 was never true

196 dt["topics"] = term.topics 

197 if len(term.roman) > 0: 197 ↛ 198line 197 didn't jump to line 198 because the condition on line 197 was never true

198 dt["roman"] = term.roman 

199 data_append(data, term.linkage, dt) 

200 

201 

202def process_categories( 

203 wxr: WiktextractContext, page_data: list[dict[str, Any]] 

204) -> None: 

205 # Categories are not otherwise disambiguated, but if there is only 

206 # one sense and only one data in ret for the same language, move 

207 # categories to the only sense. Note that categories are commonly 

208 # specified for the page, and thus if we have multiple data in 

209 # ret, we don't know which one they belong to (not even which 

210 # language necessarily?). 

211 # XXX can Category links be specified globally (i.e., in a different 

212 # language?) 

213 by_lang = defaultdict(list) 

214 for data in page_data: 

215 by_lang[data["lang"]].append(data) 

216 for la, lst in by_lang.items(): 

217 if len(lst) > 1: 

218 # Propagate categories from the last entry for the language to 

219 # its other entries. It is common for them to only be specified 

220 # in the last part-of-speech. 

221 last = lst[-1] 

222 for field in ("categories",): 

223 if field not in last: 

224 continue 

225 vals = last[field] 

226 for data in lst[:-1]: 

227 assert data is not last 

228 assert data.get(field) is not vals 

229 if data.get("alt_of") or data.get("form_of"): 229 ↛ 230line 229 didn't jump to line 230 because the condition on line 229 was never true

230 continue # Don't add to alt-of/form-of entries 

231 data_extend(data, field, vals) 

232 continue 

233 if len(lst) != 1: 233 ↛ 234line 233 didn't jump to line 234 because the condition on line 233 was never true

234 continue 

235 data = lst[0] 

236 senses = data.get("senses") or [] 

237 if len(senses) != 1: 

238 continue 

239 # Only one sense for this language. Move categories and certain other 

240 # data to sense. 

241 for field in ("categories", "topics", "wikidata", "wikipedia"): 

242 if field in data: 

243 v = data[field] 

244 del data[field] 

245 data_extend(senses[0], field, v) 

246 

247 # If the last part-of-speech of the last language (i.e., last item in "ret") 

248 # has categories or topics not bound to a sense, propagate those 

249 # categories and topics to all datas on "ret". It is common for categories 

250 # to be specified at the end of an article. Apparently these can also 

251 # apply to different languages. 

252 if len(page_data) > 1: 

253 last = page_data[-1] 

254 for field in ("categories",): 

255 if field not in last: 

256 continue 

257 lst = last[field] 

258 for data in page_data[:-1]: 

259 if data.get("form_of") or data.get("alt_of"): 259 ↛ 260line 259 didn't jump to line 260 because the condition on line 259 was never true

260 continue # Don't add to form_of or alt_of entries 

261 data_extend(data, field, lst) 

262 

263 # Remove category links that start with a language name from entries for 

264 # different languages 

265 rhymes_ns_prefix = ( 

266 wxr.wtp.NAMESPACE_DATA.get("Rhymes", {}).get("name", "") + ":" # type: ignore[call-overload] 

267 ) 

268 for data in page_data: 

269 lang_code = data.get("lang_code") 

270 cats = data.get("categories", []) 

271 new_cats = [] 

272 for cat in cats: 

273 no_prefix_cat = cat.removeprefix(rhymes_ns_prefix) 

274 cat_lang = no_prefix_cat.split(maxsplit=1)[0].split( 

275 "/", maxsplit=1 

276 )[0] 

277 cat_lang_code = name_to_code(cat_lang, "en") 

278 if ( 

279 cat_lang_code != "" 

280 and cat_lang_code != lang_code 

281 and not (lang_code == "mul" and cat_lang_code == "en") 

282 ): 

283 continue 

284 new_cats.append(cat) 

285 if len(new_cats) == 0: 

286 if "categories" in data: 

287 del data["categories"] 

288 else: 

289 data["categories"] = new_cats 

290 

291 

292def remove_duplicate_data(page_data: dict) -> None: 

293 # Remove duplicates from tags, categories, etc. 

294 for data in page_data: 

295 for field in ("categories", "topics", "tags", "wikidata", "wikipedia"): 

296 if field in data: 

297 data[field] = sorted(set(data[field])) 

298 for sense in data.get("senses", ()): 

299 if field in sense: 

300 sense[field] = sorted(set(sense[field])) 

301 

302 # If raw_glosses is identical to glosses, remove it 

303 # If "empty-gloss" in tags and there are glosses, remove the tag 

304 for data in page_data: 

305 for s in data.get("senses", []): 

306 rglosses = s.get("raw_glosses", ()) 

307 if not rglosses: 

308 continue 

309 sglosses = s.get("glosses", ()) 

310 if sglosses: 310 ↛ 314line 310 didn't jump to line 314 because the condition on line 310 was always true

311 tags = s.get("tags", ()) 

312 while "empty-gloss" in s.get("tags", ()): 312 ↛ 313line 312 didn't jump to line 313 because the condition on line 312 was never true

313 tags.remove("empty-gloss") 

314 if len(rglosses) != len(sglosses): 

315 continue 

316 same = True 

317 for rg, sg in zip(rglosses, sglosses): 

318 if rg != sg: 

319 same = False 

320 break 

321 if same: 

322 del s["raw_glosses"] 

323 

324 

325def clean_node( 

326 wxr: WiktextractContext, 

327 sense_data: Optional[Any], 

328 wikinode: GeneralNode, 

329 template_fn: Optional[TemplateFnCallable] = None, 

330 post_template_fn: Optional[PostTemplateFnCallable] = None, 

331 node_handler_fn: Optional[NodeHandlerFnCallable] = None, 

332 collect_links: bool = False, 

333 remove_anchors_from_links: bool = False, 

334 no_strip=False, 

335 no_html_strip=False, 

336 link_collector: list[tuple[str, str]] | None = None, 

337) -> str: 

338 """ 

339 Expands node or nodes to text, cleaning up HTML tags and duplicate spaces. 

340 

341 If `sense_data` is a dictionary, expanded category links will be added to 

342 it under the `categories` key. And if `collect_link` is `True`, expanded 

343 links will be added to the `links` key. An optional `link_collector` also 

344 receives expanded links, independently of `sense_data`. This lets callers 

345 retain links alongside other text fields without changing category handling. 

346 """ 

347 

348 # print("CLEAN_NODE:", repr(value)) 

349 def clean_template_fn(name: str, ht: TemplateArgs) -> Optional[str]: 

350 if template_fn is not None: 

351 return template_fn(name, ht) 

352 if is_panel_template(wxr, name): 

353 return "" 

354 return None 

355 

356 def clean_node_handler_fn_default( 

357 node: WikiNode, 

358 ) -> Optional[list[Union[str, WikiNode]]]: 

359 assert isinstance(node, WikiNode) 

360 kind = node.kind 

361 if kind in { 

362 NodeKind.TABLE_CELL, 

363 NodeKind.TABLE_HEADER_CELL, 

364 }: 

365 return node.children 

366 return None 

367 

368 if node_handler_fn is not None: 

369 # override clean_node_handler_fn, the def above can't be accessed 

370 clean_node_handler_fn = node_handler_fn 

371 else: 

372 clean_node_handler_fn = clean_node_handler_fn_default 

373 

374 # print("clean_node: value={!r}".format(value)) 

375 v = wxr.wtp.node_to_html( 

376 wikinode, 

377 node_handler_fn=clean_node_handler_fn, 

378 template_fn=template_fn, 

379 post_template_fn=post_template_fn, 

380 ) 

381 # print("##########") 

382 # print(f"{wikinode=}") 

383 # print("clean_node: v={!r}".format(v)) 

384 

385 # Capture categories if sense_data has been given. We also track 

386 # Lua execution errors here. 

387 # If collect_links=True (for glosses), capture links 

388 category_ns_data: NamespaceDataEntry = wxr.wtp.NAMESPACE_DATA.get( 

389 "Category", 

390 {}, # type: ignore[typeddict-item] 

391 ) 

392 category_ns_names: set[str] = {category_ns_data.get("name")} | set( 

393 category_ns_data.get("aliases") # type:ignore[assignment,arg-type] 

394 ) 

395 category_ns_names |= {"Category", "category"} 

396 category_names_pattern = rf"(?:{'|'.join(category_ns_names)})" 

397 if sense_data is not None: 

398 # Check for Lua execution error 

399 if '<strong class="error">Lua execution error' in v: 399 ↛ 400line 399 didn't jump to line 400 because the condition on line 399 was never true

400 data_append(sense_data, "tags", "error-lua-exec") 

401 if '<strong class="error">Lua timeout error' in v: 401 ↛ 402line 401 didn't jump to line 402 because the condition on line 401 was never true

402 data_append(sense_data, "tags", "error-lua-timeout") 

403 # Capture Category tags 

404 if not collect_links: 

405 for m in re.finditer( 

406 rf"(?is)\[\[:?\s*{category_names_pattern}\s*:([^]|]+)", 

407 v, 

408 ): 

409 cat = clean_value(wxr, m.group(1)) 

410 cat = re.sub(r"\s+", " ", cat) 

411 cat = cat.strip() 

412 if not cat: 412 ↛ 413line 412 didn't jump to line 413 because the condition on line 412 was never true

413 continue 

414 if not sense_data_has_value(sense_data, "categories", cat): 

415 data_append(sense_data, "categories", cat) 

416 else: 

417 links, categories = extract_links_from_node( 

418 wxr, 

419 v, 

420 category_ns_names=category_ns_names, 

421 remove_anchor_tags=remove_anchors_from_links, 

422 ) 

423 for cat in categories: 

424 # do not keep duplicated category links 

425 if not sense_data_has_value(sense_data, "categories", cat): 425 ↛ 423line 425 didn't jump to line 423 because the condition on line 425 was always true

426 data_append(sense_data, "categories", cat) 

427 # print(f"{links=}") 

428 for ltuple in links: 

429 # We want to keep link data as is, even duplicated 

430 data_append(sense_data, "links", ltuple) 

431 

432 if link_collector is not None: 

433 # Use the same visibility rules as the returned prose, so citations, 

434 # tables and side panels cannot contribute unrelated destinations. 

435 link_text = remove_invisible_markup(v) 

436 captured_links, _ = extract_links_from_node( 

437 wxr, 

438 link_text, 

439 category_ns_names=category_ns_names, 

440 remove_anchor_tags=remove_anchors_from_links, 

441 ) 

442 # An empty namespace/interwiki link (e.g. [[w:|language]]) can 

443 # otherwise fall back to a bare "w:" target in the string parser. 

444 link_collector.extend( 

445 (label, target) 

446 for label, target in captured_links 

447 if not (target.endswith(":") and re.fullmatch(r"[^:]+:", target)) 

448 ) 

449 

450 v = clean_value(wxr, v, no_strip=no_strip, no_html_strip=no_html_strip) 

451 # print("After clean_value:", repr(v)) 

452 

453 # Strip any unhandled templates and other stuff. This is mostly intended 

454 # to clean up erroneous codings in the original text. 

455 # v = re.sub(r"(?s)\{\{.*", "", v) 

456 # Some templates create <sup>(Category: ...)</sup>; remove 

457 v = re.sub( 

458 rf"(?si)\s*(?:<sup>)?\({category_names_pattern}:[^)]+\)(?:</sup>)?", 

459 "", 

460 v, 

461 ) 

462 # Some templates create question mark in <sup>, e.g., 

463 # some Korean Hanja form 

464 v = re.sub(r"\^\?", "", v) 

465 return v 

466 

467 

468def sense_data_has_value( 

469 sense_data: dict[str, Any], name: str, value: Any 

470) -> bool: 

471 """ 

472 Return True if `sense_data` has value in the attribute `name`'s value or 

473 in the value of key `name` if `sense_date` is dictionary. 

474 """ 

475 if hasattr(sense_data, name): 

476 return value in getattr(sense_data, name) 

477 elif isinstance(sense_data, dict): 477 ↛ 479line 477 didn't jump to line 479 because the condition on line 477 was always true

478 return value in sense_data.get(name, ()) # type:ignore[operator] 

479 return False 

480 

481 

482def extract_links_from_node( 

483 wxr: WiktextractContext, 

484 nodes: WikiNode | list[WikiNode | str] | str, 

485 category_ns_names: set[str] | None = None, 

486 remove_anchor_tags=False, 

487 expand_nodes=False, 

488) -> tuple[list[tuple[str, str]], set[str]]: 

489 """Find link nodes and extract them as a list of tuples. If 

490 `category_ns_names` is passed, also extract category names separately and 

491 return them as a list of strings.""" 

492 ret: list[tuple[str, str]] = [] 

493 cat_ret: set[str] = set() 

494 

495 # Sometimes this function may receive nodes that have not been expanded, 

496 # and like a head template node. Expanding the involves turning the nodes 

497 # into wikitext and then parsing and expanding them, so it's expensive. 

498 if expand_nodes is True: 

499 nodes = wxr.wtp.parse(wxr.wtp.node_to_wikitext(nodes), expand_all=True) 

500 if not isinstance(nodes, list): 500 ↛ 502line 500 didn't jump to line 502 because the condition on line 500 was always true

501 nodes = [nodes] 

502 for node in nodes: 

503 # print(f"{node=}") 

504 if isinstance(node, str) and node.strip(): 

505 for m in re.finditer( 

506 r"(?is)\[\[:?(\s*([^][|:]+):)?\s*([^]|]+)(\|([^]|]+))?\]\]", 

507 # 1 2 3 4 5 

508 node, 

509 ): 

510 if ( 

511 m.group(2) 

512 and category_ns_names is not None 

513 and m.group(2).strip() in category_ns_names 

514 ): 

515 cat = clean_value(wxr, m.group(3)) 

516 cat = re.sub(r"\s+", " ", cat).strip() 

517 if not cat: 517 ↛ 518line 517 didn't jump to line 518 because the condition on line 517 was never true

518 continue 

519 cat_ret.add(cat) 

520 elif not m.group(1): 

521 if m.group(5): 

522 ltext = clean_value(wxr, m.group(5)) 

523 ltarget = clean_value(wxr, m.group(3)) 

524 elif not m.group(3): 524 ↛ 525line 524 didn't jump to line 525 because the condition on line 524 was never true

525 continue 

526 else: 

527 txt = clean_value(wxr, m.group(3)) 

528 ltext = txt 

529 ltarget = txt 

530 ltarget = re.sub(r"\s+", " ", ltarget).strip() 

531 ltext = re.sub(r"\s+", " ", ltext).strip() 

532 if not ltext and not ltarget: 532 ↛ 533line 532 didn't jump to line 533 because the condition on line 532 was never true

533 continue 

534 if not ltext and ltarget: 534 ↛ 535line 534 didn't jump to line 535 because the condition on line 534 was never true

535 ltext = ltarget 

536 ret.append((ltext, ltarget)) 

537 if not isinstance(node, WikiNode): 

538 continue 

539 for link_node in node.find_child_recursively(NodeKind.LINK): 

540 if len(link_node.largs) > 0: 540 ↛ 539line 540 didn't jump to line 539 because the condition on line 540 was always true

541 ltarget = clean_node(wxr, None, link_node.largs[0]).strip() 

542 if not ltarget: 542 ↛ 543line 542 didn't jump to line 543 because the condition on line 542 was never true

543 continue 

544 ltext = clean_node(wxr, None, link_node) 

545 ret.append((ltext, ltarget)) 

546 # XXX extract category links 

547 if category_ns_names is not None: 

548 new_ret: list[tuple[str, str]] = [] 

549 for ltext, ltarget in ret: 

550 if ltext.strip() or not ltarget.strip(): 

551 new_ret.append((ltext, ltarget)) 

552 continue 

553 m2 = re.match(r"([^:]+):.+", ltarget) 

554 if m2 is not None and m2.group(1).strip() in category_ns_names: 554 ↛ 557line 554 didn't jump to line 557 because the condition on line 554 was always true

555 cat_ret.add(ltarget[ltarget.index(":") + 1 :]) 

556 else: 

557 new_ret.append((ltext, ltarget)) 

558 ret = new_ret 

559 if remove_anchor_tags is True: 

560 new_ret = [] 

561 for ltext, ltarget in ret: 

562 if "#" in ltarget and not ltarget.startswith("#"): 

563 ltarget = ltarget[: ltarget.index("#")] 

564 new_ret.append((ltext, ltarget)) 

565 ret = new_ret 

566 return ret, cat_ret