Coverage for src/wiktextract/extractor/nl/models.py: 100%
103 statements
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
« prev ^ index » next coverage.py v7.16.2, created at 2026-10-07 00:55 +0000
1from pydantic import BaseModel, ConfigDict, Field
4class DutchBaseModel(BaseModel):
5 model_config = ConfigDict(
6 extra="forbid",
7 strict=True,
8 validate_assignment=True,
9 validate_default=True,
10 )
13class Example(DutchBaseModel):
14 text: str = ""
15 bold_text_offsets: list[tuple[int, int]] = []
16 translation: str = ""
17 bold_translation_offsets: list[tuple[int, int]] = []
18 ref: str = ""
21class AltForm(DutchBaseModel):
22 word: str
25class Sense(DutchBaseModel):
26 glosses: list[str] = []
27 tags: list[str] = []
28 raw_tags: list[str] = []
29 categories: list[str] = []
30 examples: list[Example] = []
31 form_of: list[AltForm] = []
32 topics: list[str] = []
35class Sound(DutchBaseModel):
36 ipa: str = Field(default="", description="International Phonetic Alphabet")
37 audio: str = Field(default="", description="Audio file name")
38 wav_url: str = ""
39 oga_url: str = ""
40 ogg_url: str = ""
41 mp3_url: str = ""
42 opus_url: str = ""
43 flac_url: str = ""
44 tags: list[str] = []
45 raw_tags: list[str] = []
48class Linkage(DutchBaseModel):
49 word: str
50 tags: list[str] = []
51 raw_tags: list[str] = []
52 roman: str = ""
53 sense: str = Field(default="", description="Definition of the word")
54 sense_index: int = Field(
55 default=0, ge=0, description="Number of the definition, start from 1"
56 )
59class Translation(DutchBaseModel):
60 lang_code: str = Field(
61 default="",
62 description="Wiktionary language code of the translation term",
63 )
64 lang: str = Field(default="", description="Translation language name")
65 word: str = Field(default="", description="Translation term")
66 sense: str = Field(default="", description="Translation gloss")
67 sense_index: int = Field(
68 default=0, ge=0, description="Number of the definition, start from 1"
69 )
70 tags: list[str] = []
71 raw_tags: list[str] = []
72 roman: str = ""
75class Etymology(DutchBaseModel):
76 text: str = ""
77 links: list[tuple[str, str]] = []
78 categories: list[str] = []
79 index: str = ""
82class Form(DutchBaseModel):
83 form: str = ""
84 note: str = ""
85 tags: list[str] = []
86 raw_tags: list[str] = []
87 ipa: str = ""
88 source: str = ""
89 sense: str = ""
92class Descendant(DutchBaseModel):
93 lang_code: str
94 lang: str
95 word: str
96 descendants: list["Descendant"] = []
99class Hyphenation(DutchBaseModel):
100 parts: list[str] = []
101 tags: list[str] = []
102 raw_tags: list[str] = []
105class WordEntry(DutchBaseModel):
106 model_config = ConfigDict(title="Dutch Wiktionary")
107 word: str = Field(description="Word string", min_length=1)
108 lang_code: str = Field(description="Wiktionary language code", min_length=1)
109 lang: str = Field(description="Localized language name", min_length=1)
110 pos: str = Field(description="Part of speech type", min_length=1)
111 pos_title: str = ""
112 senses: list[Sense] = []
113 categories: list[str] = []
114 tags: list[str] = []
115 raw_tags: list[str] = []
116 etymology_index: str = Field(default="", exclude=True)
117 etymology_texts: list[str] = []
118 etymology_links: list[tuple[str, str]] = []
119 sounds: list[Sound] = []
120 abbreviations: list[Linkage] = []
121 anagrams: list[Linkage] = []
122 antonyms: list[Linkage] = []
123 derived: list[Linkage] = []
124 proverbs: list[Linkage] = []
125 holonyms: list[Linkage] = []
126 homophones: list[Linkage] = []
127 hypernyms: list[Linkage] = []
128 hyponyms: list[Linkage] = []
129 metonyms: list[Linkage] = []
130 paronyms: list[Linkage] = []
131 related: list[Linkage] = []
132 rhymes: list[Linkage] = []
133 synonyms: list[Linkage] = []
134 translations: list[Translation] = []
135 hyphenations: list[Hyphenation] = []
136 forms: list[Form] = []
137 notes: list[str] = []
138 descendants: list[Descendant] = []
139 extracted_vervoeging_page: bool = Field(default=False, exclude=True)