Wiksiyonaryo tlwiktionary https://tl.wiktionary.org/wiki/Wiksiyonaryo:Unang_Pahina MediaWiki 1.47.0-wmf.20 case-sensitive Midya Natatangi Usapan Tagagamit Usapang tagagamit Wiksiyonaryo Usapang Wiksiyonaryo Talaksan Usapang talaksan MediaWiki Usapang MediaWiki Padron Usapang padron Tulong Usapang tulong Kategorya Usapang kategorya TimedText TimedText talk Module Module talk Event Event talk sa katunayan 0 3392 178037 9702 2026-09-22T15:55:19Z Yivan000 4078 Inilipat ni Yivan000 ang pahinang [[Sa katunayan]] sa [[sa katunayan]] nang walang iniwang redirect 9702 wikitext text/x-wiki "Sa Katunayan' isang bahagi ng pangungusap na katumbas ng "sa totoo lang". higit na malinaw at may paggalang ang pagbigkas nito dahil ito ang dating anyo nito. Halimbawa; 1.) Sa paghahambing ng "sa totoo lang" at ng "sa katunayan", may paggalang ang gamit ng pangalawa dahil hindi naman sa parlor naguusap kitang lahat. lqhtlxfvicqgj55holm3ffo3bxai1d1 siya nawa 0 5041 178038 15990 2026-09-23T05:02:57Z Yivan000 4078 Inilipat ni Yivan000 ang pahinang [[Siya Nawa]] sa [[siya nawa]] nang walang iniwang redirect 15990 wikitext text/x-wiki "Siya Nawa" ang huling bahagi ng panalangin na siyang "Amen" sa ingles at espaniol.Ang "Siya Nawa" ay nagpapahayag ng "Siya" at wala nang iba, kasunod ng "Nawa" na siyang gumawa sa lahat at maykapangyarihang papangyarihin ang hinihiling sa bisa ng banal na ngalan niya. g7b8ytiygitq91zjft0m7shy8iup43t Module:qualifier 828 31088 178044 166535 2026-09-23T05:16:12Z Yivan000 4078 enwikt parity 178044 Scribunto text/plain local export = {} local concat = table.concat --[==[ Wrap text in one or more CSS classes. `classes` should be a string; separate multiple classes with a space. ]==] function export.wrap_css(text, classes) return ("<span class=\"%s\">%s</span>"):format(classes, text) end --[==[ Wrap text in one or more qualifier CSS classes. `suffix` is the suffix describing the type of content, e.g. `brac` for parens, `content` for content, `comma` for commas. CSS classes <code>ib-<var>suffix</var></code> and i<code>qualifier-<var>suffix</var></code> are added. ]==] function export.wrap_qualifier_css(text, suffix) local css_classes = ("ib-%s qualifier-%s"):format(suffix, suffix) return export.wrap_css(text, css_classes) end --[==[ Format one or more qualifiers. `data` is an object with the following fields: * `qualifiers`: A single qualifier or a list or qualifiers. * `open`: Override the open paren displayed before the qualifiers. If `false` or an empty string, no paren is displayed. * `close`: Override the close paren displayed before the qualifiers. If `false` or an empty string, no paren is displayed. * `opencontent`: Content to display before the qualifiers, after the open paren. * `closecontent`: Content to display after the qualifiers, before the close paren. * `no_ib_content`: Suppress wrapping the content with classes `ib-content` and `qualifier-content`. Parens and commas will still be wrapped in CSS. * `raw`: Suppress all CSS wrapping. ]==] function export.format_qualifiers(data) local qualifiers, open, close = data.qualifiers, data.open, data.close if type(qualifiers) ~= "table" then qualifiers = {qualifiers} end if not qualifiers[1] then return "" end local parts = {} local function ins(text) table.insert(parts, text) end local function wrap_qualifier_css(text, suffix) if data.raw then return text end return export.wrap_qualifier_css(text, suffix) end if open ~= false and open ~= ""then ins(wrap_qualifier_css(open or "(", "brac")) end if data.opencontent then ins(data.opencontent) end local content = concat(qualifiers, wrap_qualifier_css(",", "comma") .. " ") if not data.no_ib_content then content = wrap_qualifier_css(content, "content") end ins(content) if data.closecontent then ins(data.closecontent) end if close ~= false and close ~= "" then ins(wrap_qualifier_css(close or ")", "brac")) end return concat(parts) end --[==[ An older interface onto `format_qualifiers`. Eventually code should be converted to use the new entry point. ]==] function export.format_qualifier(qualifiers, open, close, opencontent, closecontent, no_ib_content) return export.format_qualifiers { qualifiers = qualifiers, open = open, close = close, opencontent = opencontent, closecontent = closecontent, no_ib_content = no_ib_content, } end local function format_qualifiers_with_clarification(qualifiers, clarification, openquote, closequote) local opencontent = export.wrap_css(clarification, "qualifier-clarification") .. export.wrap_css(openquote or "“", "qualifier-clarification qualifier-quote") local closecontent = export.wrap_css(closequote or "”", "qualifier-clarification qualifier-quote") return export.format_qualifiers { qualifiers = qualifiers, open = "(", close = ")", opencontent = opencontent, closecontent = closecontent, } end --[==[ Internal implementation of {{tl|sense}}. ]==] function export.sense(qualifiers) return export.format_qualifiers { qualifiers = qualifiers }.. export.wrap_css(":", "ib-colon sense-qualifier-colon") end --[==[ Internal implementation of {{tl|antsense}}. ]==] function export.antsense(qualifiers) return format_qualifiers_with_clarification(qualifiers, "antonym(s) of ") .. export.wrap_css(":", "ib-colon sense-qualifier-colon") end return export tc3tigewud11x9n9900h1tvmhjpvcah Module:IPA 828 31186 178042 176332 2026-09-23T05:14:17Z Yivan000 4078 merge changes 178042 Scribunto text/plain local export = {} local force_cat = false -- for testing local decorations_module = "Module:decorations" local pages_module = "Module:pages" local qualifier_module = "Module:qualifier" local string_utilities_module = "Module:string utilities" local syllables_module = "Module:syllables" local utilities_module = "Module:utilities" local m_data = mw.loadData("Module:IPA/data") local m_str_utils = require(string_utilities_module) local m_syllables -- [[Module:syllables]]; loaded below if needed local m_symbols = mw.loadData("Module:IPA/data/symbols") local concat = table.concat local decode_entities = m_str_utils.decode_entities local find = string.find local gcodepoint = m_str_utils.gcodepoint local gmatch = m_str_utils.gmatch local gsub = string.gsub local insert = table.insert local is_preview = require(pages_module).is_preview local len = m_str_utils.len local listToText = mw.text.listToText local match = string.match local pattern_escape = m_str_utils.pattern_escape local sub = string.sub local u = m_str_utils.char local ugsub = m_str_utils.gsub local umatch = m_str_utils.match local usub = m_str_utils.sub local function with_codepoints(s) if find(s, "%S%s") then local parts = {} for ch in gmatch(s, "%S") do parts[#parts + 1] = with_codepoints(ch) end return concat(parts, ", ") end local cps = {} for cp in gcodepoint(s) do cps[#cps + 1] = ("U+%04X"):format(cp) end return s .. " [" .. concat(cps, " ") .. "]" end local namespace = mw.title.getCurrentTitle().nsText local function is_content_page(lang, namespace) return namespace == "" or namespace == "Reconstruction" or lang and lang:hasType("appendix-constructed") and namespace == "Appendix" end -- Etymology-only languages are not L2 entry languages; IPA should use the parent full language. local function assert_not_etymology_only_lang(lang) if lang and lang.hasType and lang:hasType("language", "etymology-only") then local parent_code = lang.getParentCode and lang:getParentCode() or nil error(("Cannot use IPA with the etymology-only language %q; use the parent full language %q instead."):format(lang:getCode(), parent_code)) end end local function track(page) require("Module:debug/track")("IPA/" .. page) return true end local function process_maybe_split_categories(split_output, categories, prontext, lang, errtext) if split_output ~= "raw" then if categories[1] then categories = require(utilities_module).format_categories(categories, lang, nil, nil, force_cat) else categories = "" end end if split_output then -- for use of IPA in links, etc. if errtext then return prontext, categories, errtext else return prontext, categories end else return prontext .. (errtext or "") .. categories end end --[==[ Format a line of one or more IPA pronunciations as {{tl|IPA}} would do it, i.e. with a preceding {"IPA:"} followed by the word {"key"} linking to an Appendix page describing the language's phonology, and with an added category ` ``lang`` terms with IPA pronunciation`. Other than the extra preceding text and category, this is identical to {format_IPA_multiple()}, and the considerations described there in the documentation apply here as well. There is a single parameter `data`, an object with the following fields: * `lang`: Object representing the language of the pronunciations, which is used when adding cleanup categories for pronunciations with invalid phonemes; for determining how many syllables the pronunciations have in them, in order to add a category such as [[:Category:Italian 2-syllable words]] (for certain languages only); for adding a category ` ``lang`` terms with IPA pronunciation`; and for determining the proper sort keys for categories. Unlike for {format_IPA_multiple()}, `lang` may not be {nil}. * `items`: List of pronunciations, in exactly the same format as for {format_IPA_multiple()}. * `err`: If not {nil}, a string containing an error message to use in place of the link to the language's phonology. * `separator`: The default separator to use when separating formatted items. Defaults to {", "}. Does not apply to the first item, where the default separator is always the empty string. Overridden by the per-item `separator` field in `items`. * `sort_key`: Explicit sort key used for categories. * `no_count`: Suppress adding a {#-syllable words} category such as [[:Category:Italian 2-syllable words]]. Note that only certain languages add such categories to begin with, because it depends on knowing how to count syllables in a given language, which depends on the phonology of the language. Also, this does not suppress the addition of cleanup or other categories. If you need them suppressed, use `split_output` to return the categories separately and ignore them. * `split_output`: If not given, the return value is a concatenation of the formatted pronunciation and formatted categories. Otherwise, two values are returned: the formatted pronunciation and the categories. If `split_output` is the value {"raw"}, the categories are returned in list form, where the list elements are a combination of category strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]]. If `split_output` is any other value besides {nil}, the categories are returned as a pre-formatted concatenated string. * `include_langname`: If specified, prefix the result with the language name, followed by a colon. * `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display at the beginning, before the formatted pronunciations and preceding {"IPA:"}. * `qq`: {nil} or a list of right qualifiers to display after all formatted pronunciations. * `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display at the beginning, before the formatted pronunciations and preceding {"IPA:"}. * `aa`: {nil} or a list of right accent qualifiers to display after all formatted pronunciations. ]==] function export.format_IPA_full(data) if type(data) ~= "table" or data.getCode then error("Must now supply a table of arguments to format_IPA_full(); first argument should be that table, not a language object") end local lang = data.lang local items = data.items local err = data.err local separator = data.separator local sort_key = data.sort_key local no_count = data.no_count local split_output = data.split_output local q = data.q local qq = data.qq local a = data.a local aa = data.aa local include_langname = data.include_langname if data.qualifiers then -- FIXME: added 2026-09-18; consider removing eventually. error("overall `.qualifiers` is no longer supported; change the code to use `.q` or `.qq`") end local hasKey = m_data.langs_with_infopages if not lang or not lang.getCode then error("Must specify language to format_IPA_full()") end assert_not_etymology_only_lang(lang) local langname = lang:getCanonicalName() local prefix_text if err then prefix_text = '<span class="error">' .. err .. '</span>' else if hasKey[lang:getCode()] then prefix_text = "Apendise:" .. langname .. " na pagbigkas" --TLCHANGE else prefix_text = "wikipedia:" .. langname .. " na ponolohiya" --TLCHANGE end prefix_text = "[[" .. prefix_text .. "|gabay]]" --TLCHANGE "[[" .. prefix_text .. "|key]]" end local prefix = "[[Wiktionary:International Phonetic Alphabet|IPA]]<sup>(" .. prefix_text .. ")</sup>:&#32;" local IPAs, categories = export.format_IPA_multiple(lang, items, separator, no_count, "raw") if is_content_page(lang, namespace) then insert(categories, { cat = langname .. " na salitang may pagbigkas na IPA", --TLCHANGE sort_key = sort_key }) end local prontext = prefix .. IPAs if q and q[1] or qq and qq[1] or a and a[1] or aa and aa[1] then prontext = require(pron_qualifier_module).format_qualifiers { lang = lang, text = prontext, q = q, qq = qq, a = a, aa = aa, } end if include_langname then prontext = langname .. ": " .. prontext end return process_maybe_split_categories(split_output, categories, prontext, lang) end local function split_phonemic_phonetic(pron) local reconstructed, phonemic, phonetic = match(pron, "^(%*?)(/.-/)%s+(%[.-%])$") if reconstructed then return reconstructed .. phonemic, reconstructed .. phonetic else return pron, nil end end local function determine_repr(pron) local reconstructed -- Temporarily remove any initial asterisk before representation marks, -- which avoids having to account for it in the data, but set the -- `reconstructed` flag. if sub(pron, 1, 1) == "*" then reconstructed = true pron = sub(pron, 2) end -- Some representation types have aliases for convenience (e.g. "// //" is -- an alias for "⫽ ⫽"). and these need to be substituted in before checking -- for other data. local opening, n = match(pron, "^.[\128-\191]*") local subs_data = m_data.representation_subs[opening] if subs_data then pron, n = ugsub(pron, subs_data[1], subs_data[2]) -- If the substitution was made, `opening` needs to be changed to the -- new opening character. if n ~= 0 then opening = subs_data[3] end end -- Get the type data based on the opening character (if any), and set the -- representation type if the closing character matches. local type_data, repr, closing = m_data.representation_types[opening] if type_data then closing = type_data[2] if type_data and match(pron, pattern_escape(closing) .. "$", #opening + 1) then repr = type_data[1] end end -- Default to the empty string. if not repr then opening, closing = "", "" end -- Reattach the asterisk if reconstructed. if reconstructed then pron = "*" .. pron end return pron, repr, opening, closing, reconstructed end local function hasInvalidSeparators(transcription) -- Escape certain characters as well as pauses, which have the format "(...)" (with any number of dots), to avoid false-positives. transcription = transcription:gsub(".[\128-\191]*", m_symbols.separator_escapes) :gsub("%(%.+%)", "\3") :gsub("[()]+", "") return ( transcription:find("..", nil, true) or transcription:match("%.%f[%z \1\2\3,:;]") or transcription:match("\1%f[%z \2\3,:;]") or transcription:match("\2%f[%z \1\3,:;]") or transcription:match("\3[:;]") or transcription:match("%f[^%z \1\2\3,]%.") ) and true or false end --[==[ Format a line of one or more bare IPA pronunciations (i.e. without any preceding {"IPA:"} and without adding to a category ` ``lang`` terms with IPA pronunciation`). Individual pronunciations are formatted using {format_IPA()} and are combined with separators, decorations, pre-text, post-text, etc. to form a line of pronunciations. Parameters accepted are: * `lang` is an object representing the language of the pronunciations, which is used when adding cleanup categories for pronunciations with invalid phonemes; for determining how many syllables the pronunciations have in them, in order to add a category such as [[:Category:Italian 2-syllable words]] (for certain languages only); and for computing the proper sort keys for categories. `lang` may be {nil}. * `items` is a list of pronunciations, each of which is an object with the following properties: ** `pron`: the pronunciation, in the same format as is accepted by {format_IPA()}, i.e. it should be either phonemic (surrounded by {/.../}), phonetic (surrounded by {[...]}), orthographic (surrounded by {⟨...⟩}) or a rhyme (beginning with a hyphen); ** `pretext`: text to display directly before the formatted pronunciation, inside of any qualifiers or accent qualifiers; ** `posttext`: text to display directly after the formatted pronunciation, inside of any qualifiers or accent qualifiers; ** `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display before the formatted pronunciation; ** `qq`: {nil} or a list of right qualifiers to display after the formatted pronunciation; ** `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display before the formatted pronunciation; ** `aa`: {nil} or a list of right accent qualifiers to after before the formatted pronunciation; ** `refs`: {nil} or a list of references or reference specs to add after the pronunciation and any posttext and qualifiers; the value of a list item is either a string containing the reference text (typically a call to a citation template such as {{tl|cite-book}}, or a template wrapping such a call), or an object with fields `text` (the reference text), `name` (the name of the reference, as in {{cd|<nowiki><ref name="foo">...</ref></nowiki>}} or {{cd|<nowiki><ref name="foo" /></nowiki>}}) and/or `group` (the group of the reference, as in {{cd|<nowiki><ref name="foo" group="bar">...</ref></nowiki>}} or {{cd|<nowiki><ref name="foo" group="bar"/></nowiki>}}); this uses a parser function to format the reference appropriately and insert a footnote number that hyperlinks to the actual reference, located in the {{cd|<nowiki><references /></nowiki>}} section; ** `gloss`: {nil} or a gloss (definition) for this item, if different definitions have different pronunciations; ** `pos`: {nil} or a part of speech for this item, if different parts of speech have different pronunciations; ** `separator`: the separator text to insert directly before the formatted pronunciation and all decorations and pre-text; defaults to the outer `separator` parameter. * `separator`: The default separator to use when separating formatted items. Defaults to {", "}. Does not apply to the first item, where the default separator is always the empty string. Overridden by the per-item `separator` field in `items`. * `no_count`: Suppress adding a {#-syllable words} category such as [[:Category:Italian 2-syllable words]]. Note that only certain languages add such categories to begin with, because it depends on knowing how to count syllables in a given language, which depends on the phonology of the language. Also, this does not suppress the addition of cleanup categories. If you need them suppressed, use `split_output` to return the categories separately and ignore them. * `split_output`: If not given, the return value is a concatenation of the formatted pronunciation and formatted categories. Otherwise, two values are returned: the formatted pronunciation and the categories. If `split_output` is the value {"raw"}, the categories are returned in list form, where the list elements are a combination of category strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]]. If `split_output` is any other value besides {nil}, the categories are returned as a pre-formatted concatenated string. ]==] function export.format_IPA_multiple(lang, items, separator, no_count, split_output) local categories = {} separator = separator or ", " if not lang then track("format-multiple-nolang") else assert_not_etymology_only_lang(lang) end -- Format if not items[1] then if namespace == "Padron" then --TLCHANGE "Template" insert(items, {pron = "/aɪ piː ˈeɪ/"}) else insert(categories, "Pronunciation templates without a pronunciation") end end local bits = {} for i, item in ipairs(items) do local bit -- If the pronunciation is entirely empty, allow this and don't do anything, so that e.g. the pretext and/or -- posttext can be specified to force something like ''unknown'' to appear in place of the pronunciation -- (as happens e.g. when ? is used as a respelling in [[Module:ca-IPA]]; see [[guèiser]] for an example). if item.pron == "" then bit = "" else local item_categories, errtext bit, item_categories, errtext = export.format_IPA(lang, item.pron, "raw") bit = bit .. errtext for _, cat in ipairs(item_categories) do insert(categories, cat) end end if item.pretext then bit = item.pretext .. bit end if item.posttext then bit = bit .. item.posttext end if item.qualifiers then -- FIXME: added 2026-09-18; consider removing eventually. error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`") end local has_decorations = item.q and item.q[1] or item.qq and item.qq[1] or item.a and item.a[1] or item.aa and item.aa[1] or item.refs and item.refs[1] local has_gloss_or_pos = item.gloss or item.pos if has_decorations or has_gloss_or_pos then -- FIXME: Currently we tack the gloss and POS (in that order) onto the end of the regular left qualifiers. -- Should we do something different? local q = item.q if has_gloss_or_pos then q = mw.clone(item.q) or {} if item.gloss then local m_qualifier = require(qualifier_module) insert(q, m_qualifier.wrap_qualifier_css("“", "quote") .. item.gloss .. m_qualifier.wrap_qualifier_css("”", "quote")) end if item.pos then -- FIXME: Consider expanding aliases as found in [[Module:headword/data]] or similar. insert(q, item.pos) end end bit = require(decorations_module).format_decorations { lang = lang, text = bit, q = q, qq = item.qq, a = item.a, aa = item.aa, refs = item.refs, } end bit = (item.separator or (i == 1 and "" or separator)) .. bit insert(bits, bit) --[=[ [[Special:WhatLinksHere/Wiktionary:Tracking/IPA/syntax-error]] The length or gemination symbol should not appear after a syllable break or stress symbol. ]=] -- The nature of the following pattern match is such that we don't have to split a combined '/.../ [...]' spec -- into its parts in order to process. if match(item.pron, "[.\203][\136\140]?\203[\144\145]") then -- [.ˈˌ][ːˑ] track("syntax-error") end if lang then -- Add syllable count if the language's diphthongs are listed in [[Module:syllables]]. -- Don't do this if the term has spaces, a liaison mark (‿) or isn't in mainspace. if not no_count and namespace == "" then m_syllables = m_syllables or require(syllables_module) local langcode = lang:getCode() if m_data.langs_to_generate_syllable_count_categories[langcode] then local raw_phonemic, phonetic, use_it = split_phonemic_phonetic(item.pron) local phonemic, repr = determine_repr(raw_phonemic) if not phonetic then -- not a '/.../ [...]' combined pronunciation if m_data.langs_to_use_phonetic_or_phonemic_notation[langcode] then use_it = phonemic elseif m_data.langs_to_use_phonetic_notation[langcode] then use_it = repr == "phonetic" and phonemic or nil else use_it = repr == "phonemic" and phonemic or nil end elseif repr == "phonetic" then use_it = phonetic elseif repr == "phonemic" then use_it = phonemic end -- Note: two uses of find with plain patterns is much faster than umatch with [ ‿]. if use_it and not (find(use_it, " ") or find(use_it, "‿")) then local syllable_count = m_syllables.getVowels(use_it, lang) if syllable_count then insert(categories, lang:getCanonicalName() .. " na salitang may " .. syllable_count .. " pantig") -- TLCHANGE end end end end end end return process_maybe_split_categories(split_output, categories, concat(bits), lang) end --[=[ Format a single IPA pronunciation, which cannot be a combined spec (such as {/.../ [...]}). This has been extracted from {format_IPA()} to allow the latter to handle such combined specs. This works like {format_IPA()} but requires that pre-created {err} (for error messages) and {categories} lists be passed in, and adds any generated error messages and categories to those lists. A single value is returned, the pronunciation, which is usually the same as passed in, but may have HTML added surrounding invalid characters so they appear in red. ]=] local function format_one_IPA(lang, raw_pron, err, categories) -- Disallow wikilinks. if match(raw_pron, "%[%[.-%]%]") then error("IPA input must not contain wikilinks.") end raw_pron = decode_entities(raw_pron) -- Detect the type of transcription. local pron, repr, opening, closing, reconstructed = determine_repr(raw_pron) -- Strip any reconstruction asterisk and representation marks. pron = sub(pron, #opening + 1 + (reconstructed and 1 or 0), -#closing - 1) if not repr then insert(categories, "IPA pronunciations with invalid representation marks") -- insert(err, "invalid representation marks") -- Removed because it's annoying when previewing pronunciation pages. end if repr ~= "orthographic" and lang and lang:getCode() == "en" and hasInvalidSeparators(pron) then insert(categories, "English IPA pronunciations with invalid separators") end if pron == "" then insert(categories, "IPA pronunciations with no pronunciation present") end -- Check for obsolete and nonstandard symbols for _, symbol in ipairs(m_data.nonstandard) do local result for nonstandard in gmatch(pron, symbol) do if not result then result = {} end insert(result, nonstandard) insert(categories, {cat = "IPA pronunciations with obsolete or nonstandard characters", sort_key = nonstandard} ) end if result then insert(err, "obsolete or nonstandard characters (" .. concat(result) .. ")") break end end --[[ Check for invalid symbols after removing the following: 1. wikilinks (handled above) 2. paired HTML tags 3. bolding 4. italics 5. asterisk at beginning of transcription 6. comma followed by spacing characters 7. superscripts enclosed in superscript parentheses ]] local found_HTML local result = gsub(pron, "<(%a+)[^>]*>([^<]+)</%1>", function(tagName, content) found_HTML = true return content end) result = gsub(result, "'''([^']*)'''", "%1") result = gsub(result, "''([^']*)''", "%1") result = gsub(result, "^%*", "") result = ugsub(result, ",%s+", "") -- VS15 local vs15_class = "[" .. m_symbols.add_vs15 .. "]" if umatch(pron, vs15_class) then local vs15 = u(0xFE0E) if find(result, vs15) then result = gsub(result, vs15, "") pron = gsub(pron, vs15, "") end pron = ugsub(pron, vs15_class, "%0" .. vs15) end if result ~= "" then local content_page = is_content_page(lang, namespace) if lang then -- Get the per_lang_valid data, and convert any per-language valid sequences to spaces. local per_lang_valid = m_symbols.per_lang_valid[lang:getCode()] if per_lang_valid then if type(per_lang_valid) == "table" then for _, pattern in pairs(per_lang_valid) do result = ugsub(result, pattern, " ") end else -- Should be a string. result = ugsub(result, per_lang_valid, " ") end end end local suggestions = {} -- Check for any invalid sequences, excluding anything in the per-language lookup table. for k, v in pairs(m_symbols.invalid) do if find(result, k, nil, true) then insert(suggestions, with_codepoints(k) .. " with " .. with_codepoints(v)) end end if suggestions[1] then local replacements = "replace " .. listToText(suggestions) if content_page then error("Invalid IPA: " .. replacements) end insert(err, replacements) end -- Convert any valid character sequences to spaces for _, pattern in pairs(m_symbols.valid) do result = ugsub(result, pattern, " ") end if not match(result, "^ *$") then local category = "IPA pronunciations with invalid IPA characters" if not content_page then category = category .. "/non_mainspace" end insert(categories, category) insert(err, "invalid IPA characters: " .. with_codepoints(result)) end end if found_HTML then insert(categories, "IPA pronunciations with paired HTML tags") end if (repr == "phonemic" or repr == "rhyme") and lang and m_data.phonemes[lang:getCode()] then local valid_phonemes = m_data.phonemes[lang:getCode()] local rest = pron local phonemes = {} while #rest > 0 do local longestmatch, longestmatch_len = "", 0 local rest_init = sub(rest, 1, 1) if rest_init == "(" or rest_init == ")" then longestmatch = rest_init longestmatch_len = 1 else for _, phoneme in ipairs(valid_phonemes) do local phoneme_len = len(phoneme) if phoneme_len > longestmatch_len and usub(rest, 1, phoneme_len) == phoneme then longestmatch = phoneme longestmatch_len = len(longestmatch) end end end if longestmatch_len > 0 then insert(phonemes, longestmatch) rest = usub(rest, longestmatch_len + 1) else local phoneme = usub(rest, 1, 1) insert(phonemes, "<span style=\"color: var(--wikt-palette-red,red)\">" .. phoneme .. "</span>") rest = usub(rest, 2) insert(categories, "IPA pronunciations with invalid phonemes/" .. lang:getCode()) track("invalid phonemes/" .. phoneme) end end pron = concat(phonemes) end return (reconstructed and "*" or "") .. opening .. pron .. closing end --[==[ Format an IPA pronunciation. This wraps the pronunciation in appropriate CSS classes and adds cleanup categories and error messages as needed. The pronunciation `pron` should be either phonemic (surrounded by {/.../}), phonetic (surrounded by {[...]}), orthographic (surrounded by {⟨...⟩}), a rhyme (beginning with a hyphen) or a combined phonemic/phonetic spec (of the form {/.../ [...]}). `lang` indicates the language of the pronunciation and can be {nil}. If not {nil}, and the specified language has data in [[Module:IPA/data]] indicating the allowed phonemes, then the page will be added to a cleanup category and an error message displayed next to the outputted pronunciation. Note that {lang} also determines sort key processing in the added cleanup categories. If `split_output` is not given, the return value is a concatenation of the formatted pronunciation, error messages and formatted cleanup categories. Otherwise, three values are returned: the formatted pronunciation, the cleanup categories and the concatenated error messages. If `split_output` is the value {"raw"}, the cleanup categories are returned in list form, where the list elements are a combination of category strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]]. If `split_output` is any other value besides {nil}, the cleanup categories are returned as a pre-formatted concatenated string. ]==] function export.format_IPA(lang, pron, split_output) local err = {} local categories = {} -- `pron` shouldn't contain ref tags. if match(pron, "\127'\"`UNIQ%-%-ref%-[%dA-F]+%-QINU`\"'\127") then error("<ref> tags found inside pronunciation parameter.") end if not lang then track("format-nolang") else assert_not_etymology_only_lang(lang) end local phonemic, phonetic = split_phonemic_phonetic(pron) pron = format_one_IPA(lang, phonemic, err, categories) if phonetic then track("phonemic-phonetic") -- There's no benefit to supporting the "/.../ [...]" format within one parameter. phonetic = format_one_IPA(lang, phonetic, err, categories) pron = pron .. " " .. phonetic end if err[1] and is_preview() then err = '<span class="error" style="font-size: small;>&#32;' .. concat(err, ", ") .. "</span>" else err = "" end return process_maybe_split_categories(split_output, categories, '<span class="IPA nowrap">' .. pron .. "</span>", lang, err) end --[==[ Format a line of one or more enPR pronunciations as {{tl|enPR}} would do it, i.e. with a preceding {"enPR:"} (linked to [[Appendix:English pronunciation]]) followed by one or more formatted, comma-separated enPR pronunciations. The pronunciations are formatted by wrapping them in the `AHD` and `enPR` CSS classes and adding any decorations (qualifiers, accent qualifiers and references). In addition, the overall result is wrapped in any overall decorations. There is a single parameter `data`, an object with the following fields: * `items` is a list of enPR pronunciations, each of which is an object with the following properties: ** `pron`: the enPR pronunciation; ** `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display before the formatted pronunciation; ** `qq`: {nil} or a list of right qualifiers to display after the formatted pronunciation; ** `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display before the formatted pronunciation; ** `aa`: {nil} or a list of right accent qualifiers to after before the formatted pronunciation. * `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display at the beginning, before the formatted pronunciations and preceding {"enPR:"}. * `qq`: {nil} or a list of right qualifiers to display after all formatted pronunciations. * `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display at the beginning, before the formatted pronunciations and preceding {"enPR:"}. * `aa`: {nil} or a list of right accent qualifiers to display after all formatted pronunciations. ]==] function export.format_enPR_full(data) local prefix = "[[Appendix:English pronunciation|enPR]]: " local lang = require("Module:languages").getByCode("en") local parts = {} for _, item in ipairs(data.items) do local part = '<span class="AHD enPR">' .. item.pron .. "</span>" if item.qualifiers then -- FIXME: added 2026-09-18; consider removing eventually. error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`") end if item.q and item.q[1] or item.qq and item.qq[1] or item.a and item.a[1] or item.aa and item.aa[1] then part = require(decorations_module).format_decorations { lang = lang, text = part, q = item.q, qq = item.qq, a = item.a, aa = item.aa, } end insert(parts, part) end local prontext = prefix .. concat(parts, ", ") if data.qualifiers then -- FIXME: added 2026-09-18; consider removing eventually. error("overall `.qualifiers` is no longer supported; change the code to use `.q` or `.qq`") end if data.q and data.q[1] or data.qq and data.qq[1] or data.a and data.a[1] or data.aa and data.aa[1] then prontext = require(decorations_module).format_decorations { lang = lang, text = prontext, q = data.q, qq = data.qq, a = data.a, aa = data.aa, } end return prontext end return export g9jloaqepe6sw5c8g6tddxskfppnui7 buenos días 0 31649 178039 157330 2026-09-23T05:05:46Z Yivan000 4078 178039 wikitext text/x-wiki =={{=es=}}== ===Pagbigkas=== {{es-pr|+<audio:Es-buenos días.ogg><audio:Es-buenos días.oga>}} ===Pandamdam=== {{head|es|interjection|head=[[bueno]]s [[día]]s}} # [[magandang araw]], [[magandang umaga]] #: {{syn|es|buen día}} #: {{cot|es|buenas tardes|buenas noches}} ongixx3z19mdj37kido1uw4upf895u4 Module:decorations 828 33383 178043 177965 2026-09-23T05:15:19Z Yivan000 4078 enwikt parity 178043 Scribunto text/plain local export = {} local gender_and_number_module = "Module:gender and number" local labels_module = "Module:labels" local qualifier_module = "Module:qualifier" local references_module = "Module:references" local function track(page) require("Module:debug/track")("decorations/" .. page) return true end --[==[ This function is used by any module that wants to add support for adding decorations (i.e. left and right regular and accent qualifiers, labels and references, and potentially other types of decorations in the future) to an item, where an "item" is any string of text that might want to be decorated. It is currently used for: # pronunciations and other pronunciation-related items, such as rhymes, hyphenations and homophones, as implemented in [[Module:IPA]], [[Module:rhymes]], [[Module:hyphenation]], [[Module:homophones]] and various language-specific modules such as [[Module:es-pronunc]]; # arbitrary links, as implemented in [[Module:links]]; # headwords, as implemented in [[Module:headword]]; # gender/number specs, as implemented in [[Module:gender and number]]; # *nyms of all sorts (e.g. synonyms, antonyms, hypernyms, hyponyms, coordinate terms, etc.) as implemented in [[Module:nyms]]; # affixes of all sorts, as implemented in [[Module:affix]]; # affix usage examples, as implemented in [[Module:affixusex]]; # column templates as such as {{tl|col}} and {{tl|col3}}, as implemented in [[Module:columns]]; # names (surnames, given names, patronymics, etc.) as implemented in [[Module:names]]; # object-usage qualifiers such as {{tl|+obj}}, as implemented in [[Module:object usage]]; # various inflection templates such as {{tl|ar-conj}} in [[Module:ar-verb]]; # demonyms, as implemented in [[Module:demonym]]; and several other places. Modules that implement templates that occur frequently on a page, such as [[Module:headword]] and [[Module:links]], should consider checking that any decorations exist before loading the module. This is less necessary for other, less-used templates, as this module is not very heavy and does appropriate checks itself to make sure that any decorations are present before taking action. `data` is a structure containing the following fields: * `q`: Optional list of left regular qualifiers, each a string. * `qq`: Optional list of right regular qualifiers, each a string. * `a`: Optional list of left accent qualifiers, each a string. * `aa`: Optional list of right accent qualifiers, each a string. * `l`: Optional list of left labels, each a string. * `ll`: Optional list of right labels, each a string. * `refs`: Optional list of references or reference specs to add directly after the text; the value of a list item is either a string containing the reference text (typically a call to a citation template such as {{tl|cite-book}}, or a template wrapping such a call), or an object with fields `text` (the reference text), `name` (the name of the reference, as in `<nowiki><ref name="foo">...</ref></nowiki>` or `<nowiki><ref name="foo" /></nowiki>`) and/or `group` (the group of the reference, as in `<nowiki><ref name="foo" group="bar">...</ref></nowiki>` or `<nowiki><ref name="foo" group="bar"/></nowiki>`); this uses a parser function to format the reference appropriately and insert a footnote number that hyperlinks to the actual reference, located in the `<nowiki><references /></nowiki>` section. * `genders`: Optional list of gender/number specs as accepted by [[Module:gender and number]]. * `lang`: Language object for accent qualifiers. * `text`: The text to wrap with qualifiers. *` raw`: Don't do any CSS wrapping of the formatted text. The order of qualifiers and labels, on both the left and right, is (1) labels, (2) accent qualifiers, (3) regular qualifiers. This goes in order of relative importance. References and genders go on the right, inside of labels and qualifiers, with references before genders. ]==] function export.format_decorations(data) if not data.text then error("Missing `data.text`; did you try to pass `text` as a separate param?") end if not data.lang then track("nolang") end local text = data.text -- Format the qualifiers and labels that go either before or after the main text. They are ordered as follows, on -- both the left and the right: (1) labels, (2) accent qualifiers, (3) regular qualifiers. This puts the different -- types of qualifiers/labels in order of relative importance. Return nil if no qualifiers or labels, otherwise -- a string containing all formatted qualifiers and labels surrounded by parens. local function format_qualifier_like(labels, accent_qualifiers, qualifiers) local has_qualifiers = qualifiers and qualifiers[1] local has_accent_qualifiers = accent_qualifiers and accent_qualifiers[1] local has_labels = labels and labels[1] if not has_qualifiers and not has_accent_qualifiers and not has_labels then return nil end local qualifier_like_parts = {} local function ins(part) table.insert(qualifier_like_parts, part) end local function format_label_like(labels, mode) return require(labels_module).show_labels { lang = data.lang, labels = labels, nocat = true, mode = mode, open = false, close = false, no_ib_content = true, no_track_already_seen = true, ok_to_destructively_modify = true, -- doesn't apply to `labels` raw = data.raw, } end local m_qualifier = require(qualifier_module) if has_labels then ins(format_label_like(labels)) end if has_accent_qualifiers then ins(format_label_like(accent_qualifiers, "accent")) end if has_qualifiers then ins(m_qualifier.format_qualifiers { qualifiers = qualifiers, open = false, close = false, no_ib_content = true, raw = data.raw, }) end local qualifier_inside local function wrap_qualifier_css(txt, suffix) if data.raw then return txt else return m_qualifier.wrap_qualifier_css(txt, suffix) end end if qualifier_like_parts[2] then qualifier_inside = table.concat(qualifier_like_parts, wrap_qualifier_css(",", "comma") .. " ") else qualifier_inside = qualifier_like_parts[1] end qualifier_like_parts = {} ins(wrap_qualifier_css("(", "brac")) ins(wrap_qualifier_css(qualifier_inside, "content")) ins(wrap_qualifier_css(")", "brac")) return table.concat(qualifier_like_parts) end if data.refs then text = text .. require(references_module).format_references(data.refs) end if data.genders and data.genders[1] then -- NOTE, format_genders() returns a second value (categories) but we ignore it. text = text .. "&nbsp;" .. require(gender_and_number_module).format_genders(data.genders, data.lang) end if data.qualifiers then -- FIXME: added 2026-09-18; consider removing eventually. error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`") end local leftq = format_qualifier_like(data.l, data.a, data.q) local rightq = format_qualifier_like(data.ll, data.aa, data.qq) if leftq then text = leftq .. " " .. text end if rightq then text = text .. " " .. rightq end return text end function export.format_qualifiers(...) -- FIXME: Added 2026-09-17. Remove after a month or less. error("use format_decorations instead") end return export gvyopr8wgnkfw3arhfcjf23ibitjoha Module:IPA/data 828 33750 178046 169474 2026-09-23T05:19:56Z Yivan000 4078 enwikt parity 178046 Scribunto text/plain local list_to_set = require("Module:table").listToSet local data = {} --[=[ A list of representation types (e.g. /foo/ for phonemic and [bar] for phonetic), given as a table. The key is the opening character, the first value the representation type, and the second value the closing symbol.]=] data.representation_types = { ["/"] = {"phonemic", "/"}, ["["] = {"phonetic", "]"}, ["⫽"] = {"morphophonemic", "⫽"}, ["⟨"] = {"orthographic", "⟩"}, ["-"] = {"rhyme", ""}, } --[=[ A list of convenience inputs for certain representation types. The key is the opening character, and the table is a three-item array consisting of (1) an mw.ustring.gsub pattern which is anchored to the start and end of the string, with a single capture group that excludes the characters to be substituted, (2) a corresponding replacement pattern to be used with the pattern, and (3) the replacement opening character.]=] data.representation_subs = { ["<"] = {"^<(.*)>$", "⟨%1⟩", "⟨"}, ["/"] = {"^//(.*)//$", "⫽%1⫽", "⫽"}, } --[=[ This should list the language codes of all languages that have a pronunciation page in the appendix of the form ''Appendix:LANG pronunciation'', e.g. [[Appendix:Russian pronunciation]]. For these languages, the text "key" next to the generated pronunciation links to such pages; for other languages, it links to the "LANG phonology" page in Wikipedia (which may or may not exist). [[Module:IPA]] is responsible for this linking; see format_IPA_full().]=] data.langs_with_infopages = list_to_set{ "acw", "ady", "ang", "arc", "ba", "bg", "bo", "ca", "cho", "cmn", "cs", "cv", "cy", "da", "de", "dsb", "dz", "egl", "egy", "el", "en", "enm", "eo", "es", "fa", "fi", "fo", "fr", "fy", "ga", "gd", "ghc", "gmh", "gmw-msc", "got", "he", "hi", "hrx", "hu", "hy", "id", "ii", "is", "it", "iu", "ja", "jbo", "ka", "kls", "ko", "kw", "la", "lb", "liv", "lt", "lv", "mdf", "mfe", "mic", "mk", "mns-nor", "ms", "mt", "mul", "my", "nan", "nci", "nl", "nn", "no", "nov", "nv", "pjt", "pl", "ps", "pt", "ro", "ru", "scn", "sco", "sga", "sh", "sl", "sq", "sv", "sw", "syc", "szl", "tg", "th", "tl", "tpw", "tr", "tyv", "ug", "uk", "vi", "vo", "wlm", "yi", "yrl", "yue", "zlw-mas" } --[=[ This should list the diphthongs of a language (in the form of Lua patterns), provided they do *NOT* contain semivowel symbols such as /j w ɰ ɥ/ or vowels with nonsyllabic diacritics such as /i̯ u̯/. For example, list /au/ or /aʊ/, but do not list /aw/ or /au̯/. The data in this table is used to count the number of syllables in a word. [[Module:syllables]] automatically knows how to correctly handle semivowel symbols and nonsyllabic diacritics. Any language listed here will automatically have categories of the form "LANG #-syllable words" generated. In addition, any language listed below under `langs_to_generate_syllable_count_categories` will also have such categories generated. NOTE: There are some additional languages that have these categories. For example: * Thai words have these categories added by [[Module:th-pron]].]=] data.diphthongs = { ["cs"] = { -- [[w:Czech phonology#Diphthongs]] "[aeo]u", }, ["de"] = { "a[ɪʊ]", "ɔ[ʏɪ]", }, ["en"] = { -- from [[Appendix:English pronunciation]] mostly, but /ʌɪ/ is from the OED "[aɑæeɛoɔʌ][ɪi]", "[ɑɒæo]e", "[əɐ]ʉ", "[aɒəoɔæ]ʊ", "æo", "[ɛeɪiɔʊʉ]ə", -- /iə/ is a diphthong in NZE, but a disyllabic sequence in GA. -- /ɪə/ is both a disyllabic sequence and a diphthong in old-fashioned RP. "[aʌ][ʊɪ]ə", -- May be a disyllabic sequence in some or all dialects? }, ["grc"] = { "[aeyo]i", "[ae]u", "[ɛɔa]ː[iu]", }, ["hrx"] = { "aɪ̯", "aʊ̯", "oɪ̯", "eʊ̯", }, ["is"] = { -- [[w:Icelandic phonology#Vowels]] "[aeɔœʏ]i", -- diphthongs as the module generates them "[ao]u", -- diphthongs as the module generates them "ø[iɪy]", -- additional forms that may occur; Wikipedia is oddly specific about the second element: ei and ai, but øɪ. }, ["it"] = { "[aeɛoɔu]i", "[aeɛioɔ]u", }, ["lb"] = { "[iu]ə", "[ɜoæɑ]ɪ", "[əæɑ]ʊ", }, ["lt"] = { "ɐɪ", "ɒʊ", "ɛɪ", "ɛʊ", "ʊɪ", "ɔɪ", "ɔʊ", -- Simple diphthongs (unstressed forms) "iɛ", "uɔ", -- Complex diphthongs "ɑˑɪ", "ɑˑʊ", "æˑɪ", "æˑʊ", "oˑɪ", -- Falling tone (acute) "ɐɪˑ", "ɒʊˑ", "ɛɪˑ", "ɛʊˑ", "ʊɪˑ", -- Rising tone (tilde) - lengthened second element -- Note: Mixed diphthongs (e.g., ɐlˑ, æˑn, ʊl, etc.) are omitted since they are inherently monosyllabic }, } --[=[ This should list any languages for which categories of the form "LANG #-syllable words", e.g. [[:Category:Russian 3-syllable words]], should be generated. Do not list languages here if they have an entry above under `data.diphthongs`; such languages are automatically added to this list.]=] local langs_to_generate_syllable_count_categories = list_to_set{ "ar", -- Arabic has diphthongs, but they are transcribed -- with semivowel symbols. "ary", -- Moroccan Arabic has diphthongs, but they are transcribed -- with semivowel symbols. "bg", -- Bulgarian has diphthongs with /j/ and marginally with /w/, -- but these are semivowels. "ca", -- Catalan has diphthongs, but they are generally transcribed using -- /w/ and /j/, so do not need to be listed (see [[w:Catalan language#Diphthongs and triphthongs]]. "eo", "es", -- Spanish has diphthongs, but they are transcribed with i̯ etc. "eu", -- Basque has dipthongs, but they are transcribed with i̯ and u̯. "fi", -- Finnish has diphthongs, but they are now automatically transcribed with -- the nonsyllabic diacritic "fr", -- French has diphthongs, but they are transcribed -- with semivowel symbols: [[w:French phonology#Glides and diphthongs]]. "hnn", "id", -- Indonesian has diphthongs, but they are transcribed with i̯ or /j/ etc. "ka", "kne", "kmr", "ku", "la", -- All diphthongs transcribed with e̯ or /j/ etc. "mk", "ms", -- Malay has diphthongs, but they are transcribed with i̯ or /j/ etc. "mt", -- Maltese has diphthongs, but they are transcribed -- with semivowel symbols. "pl", -- No diphthongs, properly speaking; sequences of a vowel and /w/ or /j/ though. "pt", -- Portuguese has diphthongs, but they are transcribed with i̯ or /j/ etc. "rsk", -- No diphthongs but there are sequences of vowel and /j/ or /w/. "ru", -- No diphthongs, properly speaking; sequences of a vowel and /j/ though. "sk", -- Slovak has rising diphthongs, /i̯e, i̯a, i̯u, u̯o/, which are probably always spelled with the nonsyllabic diacritic, so do not need to be listed. "sl", -- No diphthongs, properly speaking; sequences of a vowel, /j/ and /w/ though "sq", -- [[w:Albanian language#Vowels]] doesn't mention anything about diphthongs. "szy", -- All diphthongs are transcribed with /j/ or /w/ "tl", -- Tagalog has diphthongs, but they are transcribed with i̯ or /j/ etc "tsg", "ug", -- No diphthongs. } -- Also add languages listed under `data.diphthongs`. for langcode, _ in pairs(data.diphthongs) do langs_to_generate_syllable_count_categories[langcode] = true end data.langs_to_generate_syllable_count_categories = langs_to_generate_syllable_count_categories -- Languages to use the phonetic not phonemic notation to compute syllable counts. data.langs_to_use_phonetic_notation = list_to_set{ "bg", "es", "id", "la", "lt", "mk", "ms", "rsk", "ru", } -- Languages to use the phonetic or phonemic notation to compute syllable counts, whichever is available. data.langs_to_use_phonetic_or_phonemic_notation = list_to_set{ -- [[Module:is-IPA]] generates [...] but many manual pronuns use /.../. "is", } -- Non-standard or obsolete IPA symbols. data.nonstandard = { --[[ The following symbols consist of more than one character, so we can't put them in the line below. ]] "ɑ̢", "ɔ̗", "ɔ̖", "[?ƍσƺƪƞƛłščžǰǧǯẋⱻʚω∅ØȣᴀᴇⱻQKPT]" } -- See valid IPA characters at [[Module:IPA/data/symbols]]. data.phonemes = {} data.phonemes["dz"] = { "m", "n", "ŋ", "p", "t", "ʈ", "k", "pʰ", "tʰ", "ʈʰ", "kʰ", "t͡s", "t͡ɕ", "t͡sʰ", "t͡ɕʰ", "w", "s", "z", "ɬ", "l", "r", "ɕ", "ʑ", "j", "h", "ɑ", "e", "i", "o", "u", "ɑː", "eː", "ɛː", "iː", "oː", "øː", "uː", "yː", "ɑ˥", "e˥", "i˥", "o˥", "u˥", "ɑː˥", "eː˥", "ɛː˥", "iː˥", "oː˥", "øː˥", "uː˥", "yː˥", "m˥", "n˥", "ŋ˥", "p˥", "k˥", "k̚˥", "w˥", "l˥", "r˥", "ɕ˥", "j˥", ")˥", "ɑ˩", "e˩", "i˩", "o˩", "u˩", "ɑː˩", "eː˩", "ɛː˩", "iː˩", "oː˩", "øː˩", "uː˩", "yː˩", "m˩", "n˩", "ŋ˩", "p˩", "k˩", "k̚˩", "w˩", "l˩", "r˩", "ɕ˩", "j˩", ")˩", ".", ",", "-", } data.phonemes["eo"] = { "a", "b", "d", "d͡ʒ", "d͡z", "e", "f", "h", "i", "j", "k", "l", "m", "n", "o", "p", "r", "s", "t", "t͡s", "t͡ʃ", "u", "u̯", "v", "w", "x", "z", "ɡ", "ʃ", "ʒ", "ˈ", ".", " ", "-", "u̯", "i̯" } data.phonemes["hy"] = { "ɑ", "b", "ɡ", "d", "e", "z", "ə", "tʰ", "ʒ", "i", "l", "χ", "t͡s", "k", "h", "d͡z", "ʁ", "t͡ʃ", "m", "j", "n", "ʃ", "ɔ", "t͡ʃʰ", "p", "d͡ʒ", "r", "s", "v", "t", "ɾ", "t͡sʰ", "v", "pʰ", "kʰ", "o", "f", "ŋɡ", "ŋk", "ŋχ", "u", "œ", "ʏ", "ˈ", "ˌ", ".", " ", "ː", } data.phonemes["nl"] = { "m", "n", "ŋ", "p", "b", "t", "d", "k", "ɡ", "f", "v", "s", "z", "ʃ", "ʒ", "x", "ɣ", "ɦ", "ʋ", "l", "j", "r", "ɪ", "ʏ", "ɛ", "ə", "ɔ", "ɑ", "i", "iː", "y", "yː", "u", "uː", "eː", "øː", "oː", "ɛː", "œː", "ɔː", "aː", "ɛi̯", "œy̯", "ɔi̯", "ɑu̯", "ɑi̯", "iu̯", "yu̯", "ui̯", "eːu̯", "oːi̯", "aːi̯", "ˈ", "ˌ", ".", " ", "-", } data.phonemes["mt"] = { "m", "n", "p", "t", "k", "ʔ", "b", "d", "ɡ", "t͡s", "t͡ʃ", "d͡z", "d͡ʒ", "f", "s", "ʃ", "ħ", "v", "z", "ʒ", "ɣ", "l", "j", "w", "r", "ɪ", "ɛ", "ɔ", "a", "u", "ɛˤ", "ɔˤ", "aˤ", "əˤ", "ɛˤː", "ɔˤː", "aˤː", "əˤː", "ɪˤː", "iː", "ɪː", "ɛː", "ɔː", "aː", "uː", "ˈ", "ˌ", ".", " ", "‿", "-" } return data 9j3yfr5dzmr70htnh6pfy97jzb11aph Module:es-common 828 34964 178045 168430 2026-09-23T05:18:08Z Yivan000 4078 enwikt parity 178045 Scribunto text/plain local export = {} local romut_module = "Module:romance utilities" local u = require("Module:string/char") local rsplit = mw.text.split local rfind = mw.ustring.find local rmatch = mw.ustring.match local rsubn = mw.ustring.gsub local toNFD = mw.ustring.toNFD local TILDE = u(0x0303) -- tilde = ̃ local DIA = u(0x0308) -- diaeresis = ̈ local CEDILLA = u(0x0327) -- cedilla = ̧ local TEMPC1 = u(0xFFF1) local TEMPC2 = u(0xFFF2) local TEMPV1 = u(0xFFF3) local DIV = u(0xFFF4) local vowel = "aeiouáéíóúý" .. TEMPV1 local V = "[" .. vowel .. "]" local AV = "[áéíóúý]" -- accented vowel local W = "[iyuw]" -- glide local C = "[^" .. vowel .. ".]" export.vowel = vowel export.V = V export.AV = AV export.W = W export.C = C local remove_accent = { ["á"] = "a", ["é"] = "e", ["í"] = "i", ["ó"] = "o", ["ú"] = "u", ["ý"] = "y" } local add_accent = { ["a"] = "á", ["e"] = "é", ["i"] = "í", ["o"] = "ó", ["u"] = "ú", ["y"] = "ý" } export.remove_accent = remove_accent export.add_accent = add_accent local prepositions = { "al? ", "del? ", "como ", "con ", "en ", "para ", "por ", } -- version of rsubn() that discards all but the first return value local function rsub(term, foo, bar) local retval = rsubn(term, foo, bar) return retval end export.rsub = rsub -- apply rsub() repeatedly until no change local function rsub_repeatedly(term, foo, bar) while true do local new_term = rsub(term, foo, bar) if new_term == term then return term end term = new_term end end export.rsub_repeatedly = rsub_repeatedly function export.decompose(text) -- decompose everything but ç, ñ and ü text = toNFD(text) text = rsub(text, ".[" .. TILDE .. DIA .. CEDILLA .. "]", { ["c" .. CEDILLA] = "ç", ["C" .. CEDILLA] = "Ç", ["n" .. TILDE] = "ñ", ["N" .. TILDE] = "Ñ", ["u" .. DIA] = "ü", ["U" .. DIA] = "Ü", }) return text end -- Apply vowel alternation to stem. function export.apply_vowel_alternation(stem, alternation) local ret, err -- Treat final -gu, -qu as a consonant, so the previous vowel can alternate (e.g. conseguir -> consigo). -- This means a verb in -guar can't have a u-ú alternation but I don't think there are any verbs like that. stem = rsub(stem, "([gq])u$", "%1" .. TEMPC1) local before_last_vowel, last_vowel, after_last_vowel = rmatch(stem, "^(.*)(" .. V .. ")(.-)$") if alternation == "ie" then if last_vowel == "e" or last_vowel == "i" then -- allow i for adquirir -> adquiero, inquirir -> inquiero, etc. ret = before_last_vowel .. "ie" .. after_last_vowel else err = "should have -e- or -i- as the last vowel" end elseif alternation == "ye" then if last_vowel == "e" then ret = before_last_vowel .. "ye" .. after_last_vowel else err = "should have -e- as the last vowel" end elseif alternation == "ue" then if last_vowel == "o" or last_vowel == "u" then -- allow u for jugar -> juego; correctly handle avergonzar -> avergüenzo ret = ( last_vowel == "o" and before_last_vowel:find("g$") and before_last_vowel .. "üe" .. after_last_vowel or before_last_vowel .. "ue" .. after_last_vowel ) else err = "should have -o- or -u- as the last vowel" end elseif alternation == "hue" then if last_vowel == "o" then ret = before_last_vowel .. "hue" .. after_last_vowel else err = "should have -o- as the last vowel" end elseif alternation == "i" then if last_vowel == "e" then ret = before_last_vowel .. "i" .. after_last_vowel else err = "should have -i- as the last vowel" end elseif alternation == "í" then if last_vowel == "e" or last_vowel == "i" then -- allow e for reír -> río, sonreír -> sonrío ret = before_last_vowel .. "í" .. after_last_vowel else err = "should have -e- or -i- as the last vowel" end elseif alternation == "ú" then if last_vowel == "u" then ret = before_last_vowel .. "ú" .. after_last_vowel else err = "should have -u- as the last vowel" end else error("Unrecognized vowel alternation '" .. alternation .. "'") end ret = ret and ret:gsub(TEMPC1, "u") or nil return {ret = ret, err = err} end -- Syllabify a word. This implements the full syllabification algorithm, based on the corresponding code -- in [[Module:es-pronunc]]. This is more than is needed for the purpose of this module, which doesn't -- care so much about syllable boundaries, but won't hurt. function export.syllabify(word) word = DIV .. word .. DIV -- gu/qu + front vowel; make sure we treat the u as a consonant; a following -- i should not be treated as a consonant ([[alguien]] would become ''álguienes'' -- if pluralized) word = rsub(word, "([gq])u([eiéí])", "%1" .. TEMPC2 .. "%2") local vowel_to_glide = { ["i"] = TEMPC1, ["u"] = TEMPC2 } -- i and u between vowels should behave like consonants ([[paranoia]], [[baiano]], [[abreuense]], -- [[alauita]], [[Malaui]], etc.) word = rsub_repeatedly(word, "(" .. V .. ")([iu])(" .. V .. ")", function(v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end ) -- y between consonants or after a consonant at the end of the word should behave like a vowel -- ([[ankylosaurio]], [[cryptomeria]], [[brandy]], [[cherry]], etc.) word = rsub_repeatedly(word, "(" .. C .. ")y(" .. C .. ")", function(c1, c2) return c1 .. TEMPV1 .. c2 end ) word = rsub_repeatedly(word, "(" .. V .. ")(" .. C .. W .. "?" .. V .. ")", "%1.%2") word = rsub_repeatedly(word, "(" .. V .. C .. ")(" .. C .. V .. ")", "%1.%2") word = rsub_repeatedly(word, "(" .. V .. C .. "+)(" .. C .. C .. V .. ")", "%1.%2") word = rsub(word, "([pbcktdg])%.([lr])", ".%1%2") word = rsub_repeatedly(word, "(" .. C .. ")%.s(" .. C .. ")", "%1s.%2") -- Any aeo, or stressed iu, should be syllabically divided from a following aeo or stressed iu. word = rsub_repeatedly(word, "([aeoáéíóúý])([aeoáéíóúý])", "%1.%2") word = rsub_repeatedly(word, "([ií])([ií])", "%1.%2") word = rsub_repeatedly(word, "([uú])([uú])", "%1.%2") word = rsub(word, "([" .. DIV .. TEMPC1 .. TEMPC2 .. TEMPV1 .. "])", { [DIV] = "", [TEMPC1] = "i", [TEMPC2] = "u", [TEMPV1] = "y", }) return rsplit(word, "%.") end -- Return the index of the (last) stressed syllable. function export.stressed_syllable(syllables) -- If a syllable is stressed, return it. for i = #syllables, 1, -1 do if rfind(syllables[i], AV) then return i end end -- Monosyllabic words are stressed on that syllable. if #syllables == 1 then return 1 end local i = #syllables -- Unaccented words ending in a vowel or a vowel + s/n are stressed on the preceding syllable. if rfind(syllables[i], V .. "[sn]?$") then return i - 1 end -- Remaining words are stressed on the last syllable. return i end -- Add an accent to the appropriate vowel in a syllable, if not already accented. function export.add_accent_to_syllable(syllable) -- Don't do anything if syllable already stressed. if rfind(syllable, AV) then return syllable end -- Prefer to accent an a/e/o in case of a diphthong or triphthong (the first one if for some reason -- there are multiple, which should not occur with the standard syllabification algorithm); -- otherwise, do the last i or u in case of a diphthong ui or iu. if rfind(syllable, "[aeo]") then return rsub(syllable, "^(.-)([aeo])", function(prev, v) return prev .. add_accent[v] end) end return rsub(syllable, "^(.*)([iu])", function(prev, v) return prev .. add_accent[v] end) end -- Remove any accent from a syllable. function export.remove_accent_from_syllable(syllable) return rsub(syllable, AV, remove_accent) end -- Return true if an accent is needed on syllable number `sylno` if that syllable were to receive the stress, -- given the syllables of a word. The current accent may be on any syllable. function export.accent_needed(syllables, sylno) -- Diphthongs iu and ui are normally stressed on the second vowel, so if the accent is on the first vowel, -- it's needed. if rfind(syllables[sylno], "íu") or rfind(syllables[sylno], "úi") then return true end -- If the default-stressed syllable is different from `sylno`, accent is needed. local unaccented_syllables = {} for _, syl in ipairs(syllables) do table.insert(unaccented_syllables, export.remove_accent_from_syllable(syl)) end local would_be_stressed_syl = export.stressed_syllable(unaccented_syllables) if would_be_stressed_syl ~= sylno then return true end -- At this point, we know that the stress would by default go on `sylno`, given the syllabification in -- `syllables`. Now we have to check for situations where removing the accent mark would result in a -- different syllabification. For example, países -> `pa.i.ses` but removing the accent mark would lead -- to `pai.ses`. Similarly, río -> `ri.o` but removing the accent mark would lead to single-syllable `rio`. -- We need to check whether (a) the stress falls on an i or u; (b) in the absence of an accent mark, the -- i or u would form a diphthong with a preceding or following vowel and the stress would be on that vowel. -- The conditions are slightly different when dealing with preceding or following vowels because ui and ui -- diphthongs are by default stressed on the second vowel. We also have to ignore h between the vowels. local accented_syllable = export.add_accent_to_syllable(unaccented_syllables[sylno]) if sylno > 1 and rfind(unaccented_syllables[sylno - 1], "[aeo]$") and rfind(accented_syllable, "^h?[íú]") then return true end if sylno < #syllables then if rfind(accented_syllable, "í$") and rfind(unaccented_syllables[sylno + 1], "^h?[aeou]") or rfind(accented_syllable, "ú$") and rfind(unaccented_syllables[sylno + 1], "^h?[aeio]") then return true end end return false end function export.make_plural(form, gender, special) local retval = require(romut_module).handle_multiword(form, special, function(term) return export.make_plural(term, gender) end, prepositions) if retval then return retval end if gender == "gneut" and rfind(form, "[x@]$") then return {form .. "s"} end -- ends in unstressed vowel or á, é, ó if rfind(form, "[aeiouáéó]$") then return {form .. "s"} end -- ends in í or ú if rfind(form, "[íú]$") then return {form .. "es", form .. "s"} end -- ends in a vowel + z if rfind(form, V .. "z$") then return {rsub(form, "z$", "ces")} end -- ends in cons + s/z if rfind(form, C.."[sz]$") then return {form} end -- ends in s/z + cons if rfind(form, "[sz]"..C.."$") then return {form} end local syllables = export.syllabify(form) -- ends in s or x with more than 1 syllable, last syllable unstressed if syllables[2] and rfind(form, "[sx]$") and not rfind(syllables[#syllables], AV) then return {form} end -- ends in l, r, n, d, z, or j with 3 or more syllables, stressed on third to last syllable if syllables[3] and rfind(form, "[lrndzj]$") and rfind(syllables[#syllables - 2], AV) then return {form} end -- ends in an accented vowel + consonant if rfind(form, AV .. C .. "$") then return {rsub(form, "(.)(.)$", function(vowel, consonant) return export.remove_accent[vowel] .. consonant .. "es" end)} end -- ends in a vowel + y, l, r, n, d, j, s, x if rfind(form, "[aeiou][ylrndjsx]$") then -- two or more syllables: add stress mark to plural; e.g. joven -> jóvenes if syllables[2] and rfind(form, "n$") then syllables[#syllables - 1] = export.add_accent_to_syllable(syllables[#syllables - 1]) return {table.concat(syllables, "") .. "es"} end return {form .. "es"} end -- ends in a vowel + ch if rfind(form, "[aeiou]ch$") then return {form .. "es"} end -- ends in two consonants if rfind(form, C .. C .. "$") then return {form .. "s"} end -- ends in a vowel + consonant other than l, r, n, d, z, j, s, or x if rfind(form, "[aeiou][^aeioulrndzjsx]$") then return {form .. "s"} end return nil end function export.make_feminine(form, special) local retval = require(romut_module).handle_multiword(form, special, export.make_feminine, prepositions) if retval then if #retval ~= 1 then error("Internal error: Should have one return value for make_feminine: " .. table.concat(retval, ",")) end return retval[1] end if form:find("o$") then local retval = form:gsub("o$", "a") -- discard second retval return retval end local function make_stem(form) return rsub( form, "^(.+)(.)(.)$", function (before_stress, stressed_vowel, after_stress) return before_stress .. (export.remove_accent[stressed_vowel] or stressed_vowel) .. after_stress end) end if rfind(form, "[áíó]n$") or rfind(form, "[éí]s$") or rfind(form, "[dtszxñ]or$") or rfind(form, "ol$") then -- holgazán, comodín, bretón (not común); francés, kirguís (not mandamás); -- volador, agricultor, defensor, avizor, flexor, señor (not posterior, bicolor, mayor, mejor, menor, peor); -- español, mongol return make_stem(form) .. "a" end return form end function export.make_masculine(form, special) local retval = require(romut_module).handle_multiword(form, special, export.make_masculine, prepositions) if retval then if #retval ~= 1 then error("Internal error: Should have one return value for make_masculine: " .. table.concat(retval, ",")) end return retval[1] end if form:find("dora$") then local retval = form:gsub("a$", "") -- discard second retval return retval end if form:find("a$") then local retval = form:gsub("a$", "o") -- discard second retval return retval end return form end return export qq46tgo6gtotmf220p5pp0dpz3e6kus Module:es-pronunc 828 35650 178040 169830 2026-09-23T05:08:17Z Yivan000 4078 enwikt parity 178040 Scribunto text/plain --[=[ This module implements the templates {{es-pr}} and {{es-IPA}}. Author: Benwing2 ]=] local export = {} local m_IPA = require("Module:IPA") local m_str_utils = require("Module:string utilities") local m_table = require("Module:table") local audio_module = "Module:audio" local decorations_module = "Module:decorations" local headword_data_module = "Module:headword/data" local homophones_module = "Module:homophones" local hyphenation_module = "Module:hyphenation" local labels_module = "Module:labels" local links_module = "Module:links" local parameters_module = "Module:parameters" local parse_utilities_module = "Module:parse utilities" local references_module = "Module:references" local rhymes_module = "Module:rhymes" local force_cat = false -- for testing --[=[ FIXME: 1. Port latest changes to production module. [DONE] 2. Finish work on rhymes and hyphenation. [DONE] 3. Handle <hmp:...> for homophones. [DONE] 4. Don't add comma before phonetic IPA. [DONE] 5. Handle secondary stress, suffixes, etc. in syllabification. [DONE] 6. Need some changes to syllable splitting in consonant clusters. (e.g. 'cum‧min‧gto‧ni‧ta') [DONE] 7. Fix handling of references to correspond to Portuguese module. [DONE] 8. Propagate qualifiers on individual pronun terms to rhymes and hyph. 9. Support raw phonemic/phonetic pronunciations. [DONE] 10. Support overall audio. [DONE] 11. Keep th/ph/kh/gh/tz ([[Ertzaintza]]) together when syllabifying (but not bh due to [[subhumano]], [[subhistoria]], etc.). [DONE] 12. Support <q:...> and <qq:...> on audio. [DONE] 13. Support <a:...> and <aa:...> (using {{a|...}}, left and right) on terms, rhymes, hyphenation, homophones and audio. [DONE] 14. Support # instead of ; as separator between audio file and gloss and make sure it works if gloss has embedded # or ;. [DONE] 15. Use parse_inline_modifiers() in [[Module:parse utilities]]. [DONE] ]=] --[=[ About styles, dialects and isoglosses: From the standpoint of pronunciation, a given dialect is defined by isoglosses, which specify differences in the way of pronouncing certain phonemes. You can think of a dialect as a collection of isoglosses. For example, one isogloss is "distinción" (pronouncing written ''s'' and ''c/z'' differently) vs. "seseo" (pronouncing them the same). Another is "lleísmo" (pronouncing written ''ll'' and ''y'' differently) vs. "yeísmo" (pronouncing them the same). The dominant pronunciation in Spain can be described as distinción + yeísmo, while the pronunciation in rural northern Spain can be described as distinción + lleísmo and the pronunciation across much of the Andes mountains, Paraguay, and the Philippines can be described as seseo + lleísmo. Specifically, the following isoglosses are recognized (note, the isogloss specs as used in this module dispense with written accents): -- "distincion" = pronouncing ''s'' and ''c/z'' differently -- "seseo" = pronouncing ''s'' and ''c/z'' the same -- "lleismo" = pronouncing ''ll'' and ''y'' differently -- "yeismo" = pronouncing ''ll'' and ''y'' the same -- "rioplatense" = Rioplatense speech, i.e. seseo+yeismo with ''ll'' and ''y'' pronounced specially, and a clear distinction between initial ''hi-'' vs. initial ''ll-/y-'' -- "sheismo" = a type of Rioplatense speech, characteristic of Buenos Aires, where ''ll'' and ''y'' are pronounced as /ʃ/ -- "zheismo" = a type of Rioplatense speech, found outside of Buenos Aires, where ''ll'' and ''y'' are pronounced as /ʒ/ -- "quito" = seseo + lleismo, but pronouncing ''ll'' as /ʒ/ -- "yucatan" = seseo + yeismo, intervocalic ''y'' is pronounced ''i'' and lost in contact with ''i'' or ''e'' These isoglosses can be combined to yield one of the following eight dialects: -- "distincion-lleismo": distinción + lleísmo -- "distincion-yeismo": distinción + yeísmo -- "seseo-lleismo": seseo + lleísmo -- "seseo-yeismo": seseo + yeísmo -- "rioplatense-sheismo": Rioplatense with /ʃ/ (Buenos Aires) -- "rioplatense-zheismo": Rioplatense with /ʒ/ (non-Buenos Aires) -- "quito" -- "yucatan" A "style" here is a set of dialects that pronounce a given word in a given fashion. For example, if we are only considering the distinción/seseo and lleísmo/yeísmo isoglosses, there are four conceivable dialects (all of which in fact exist). However, for a given word, more than one dialect may pronounce it the same. For example, a word like [[paz]] has a ''z'' but no ''ll'', and so there are only two possible pronunciations for the four dialects. Here, the two styles are "Spain" and "Latin America". Correspondingly, a word like [[pollo]] with an ''ll'' but no ''z'' has two styles, which can approximately be described as "most of Spain and Latin America" vs. "rural northern Spain, Andes Mountains, Paraguay, Philippines". A "style spec" (indicated by the style= parameter to {{es-IPA}}) restricts the output to certain styles. A style spec can be one of the following: 1. An isogloss, e.g. "distincion", "rioplatense"; if specified, only styles containing this isogloss are output. 2. A negated isogloss, e.g. "-rioplatense". 3. An intersection of isoglosses ("A and B"), e.g. "distincion+lleismo". This can be used to restrict to specific dialects. 4. A union of isoglosses ("A or B"), e.g. "distincion,zheismo". If both plus and comma are used, plus takes precedence, e.g. "seseo+lleismo,zheismo" means either the "seseo+lleismo" dialect or the "rioplatense-zheismo" dialect. An example where the style= parameter might be used is with the word [[bluetooth]], which has one pronunciation in Spain/distinción (respelled "blutuz") but another in Latin America/seseo (respelled "blutud"). This might be represented using {{es-pr}} as {{es-pr|blutuz<style:distincion>|blutud<style:seseo>}}. ]=] local lang = require("Module:languages").getByCode("es") local decompose = require("Module:es-common").decompose local u = m_str_utils.char local rfind = m_str_utils.find local rsubn = m_str_utils.gsub local rsplit = m_str_utils.split local ulower = m_str_utils.lower local ulen = m_str_utils.len local unfd = mw.ustring.toNFD local unfc = mw.ustring.toNFC local AC = u(0x0301) -- acute = ́ local GR = u(0x0300) -- grave = ̀ local CFLEX = u(0x0302) -- circumflex = ̂ local TILDE = u(0x0303) -- tilde = ̃ local SYLDIV = u(0xFFF0) -- used to represent a user-specific syllable divider (.) so we won't change it local vowel = "aeiouüyAEIOUÜY" -- vowel; include y so we get single-word y correct and for syllabifying from spelling local V = "[" .. vowel .. "]" -- vowel class local accent = AC .. GR .. CFLEX local accent_c = "[" .. accent .. "]" local stress = AC .. GR local stress_c = "[" .. AC .. GR .. "]" local ipa_stress = "ˈˌ" local ipa_stress_c = "[" .. ipa_stress .. "]" local sylsep = "%-." .. SYLDIV -- hyphen included for syllabifying from spelling local sylsep_c = "[" .. sylsep .. "]" local wordsep = "# " local separator_not_wordsep = accent .. ipa_stress .. sylsep local separator = separator_not_wordsep .. wordsep local separator_c = "[" .. separator .. "]" local C = "[^" .. vowel .. separator .. "]" -- consonant class including h local C_NOT_H = "[^" .. vowel .. separator .. "h]" -- consonant class not including h local C_OR_WORDSEP = "[^" .. vowel .. separator_not_wordsep .. "]" -- consonant class including h, or word separator local T = "[^" .. vowel .. "lrɾjw" .. separator .. "]" -- obstruent or nasal local unstressed_words = m_table.listToSet({ "el", "la", "los", "las", -- definite articles "un", -- single-syllable indefinite articles "me", "te", "se", "lo", "le", "nos", "os", "les", -- unstressed object pronouns "mi", "mis", "tu", "tus", "su", "sus", -- unstressed possessive pronouns "que", "si", -- subordinating conjunctions "y", "e", "o", "u", "mas", -- coordinating conjunctions "de", "del", "a", "al", -- basic prepositions + combinations with articles "por", "en", "con", -- other prepositions }) -- version of rsubn() that discards all but the first return value local function rsub(term, foo, bar) local retval = rsubn(term, foo, bar) return retval end -- version of rsubn() that returns a 2nd argument boolean indicating whether -- a substitution was made. local function rsubb(term, foo, bar) local retval, nsubs = rsubn(term, foo, bar) return retval, nsubs > 0 end -- apply rsub() repeatedly until no change local function rsub_repeatedly(term, foo, bar) while true do local new_term = rsub(term, foo, bar) if new_term == term then return term end term = new_term end end local function split_on_comma(term) if not term then return nil end if term:find(",%s") then return require(parse_utilities_module).split_on_comma(term) elseif term:find(",") then return rsplit(term, ",") else return {term} end end -- Remove any HTML from the formatted text and resolve links, since the extra characters don't contribute to the -- displayed length. local function convert_to_raw_text(text) text = rsub(text, "<.->", "") if text:find("%[%[") then text = require(links_module).remove_links(text) end return text end -- Return the approximate displayed length in characters. local function textual_len(text) return ulen(convert_to_raw_text(text)) end local function construct_default_differences(dialect) if dialect == "distincion-lleismo" then return { distincion_different = false, lleismo_different = false, sheismo_different = false, need_rioplat = false, need_quito = false, need_yucatan = false, } end return nil end -- Main syllable-division algorithm. Can be called either directly on spelling (when hyphenating) or after -- non-trivial processing of respelling in the direction of pronunciation (when generating pronunciation). local function syllabify_from_spelling_or_pronun(text, is_spelling) -- Part 1: Divide before the last consonant in a cluster of consonants between vowels (but don't divide a VhV -- sequence; [[prohibir]] should be prohi.bir). Then move the syllable division marker leftwards over clusters that -- can form onsets. text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*)(" .. C_NOT_H .. V .. ")", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. C .. "+)(" .. C .. V .. ")", "%1.%2") -- Puerto Rico + most of Spain divide tl as t.l. Mexico and the Canary Islands have .tl. Unclear what other regions -- do. Here we choose to go with .tl. See https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf -- and https://www.spanishdict.com/guide/spanish-syllables-and-syllabification-rules. -- NOTE: When run on pronun, we have already eliminated c and v, but not when run on spelling. -- When run on pronun, don't include r, which at this point represents the trill. local cluster_r = is_spelling and "rɾ" or "ɾ" -- Don't divide Cl or Cr where C is a stop or fricative, except for dl. text = rsub(text, "([pbfvkctg])%.([l" .. cluster_r .. "])", ".%1%2") text = text:gsub("d%.([" .. cluster_r .. "])", ".d%1") -- Don't divide ch, sh, ph, th, dh, fh, kh or gh. Do allow bh to be divided ([[subhumano]], [[subhúmedo]], etc.). text = rsub(text, "([csptdfkg])%.h", ".%1h") -- Don't divide ll or rr. text = rsub(text, "([lr])%.%1", ".%1%1") -- Don't divide tz ([[Ertzaintza]], [[quetzal]], [[hertziano]] and other words of Basque, Nahuatl and German -- origin). text = rsub(text, "t%.z", ".tz") -- Per https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf, tl at the end of a word -- (as in nahuatl, Popocatepetl etc.) is divided .tl from the previous vowel. if is_spelling then text = text:gsub("([^. %-])tl$", "%1.tl") text = text:gsub("([^. %-])(tl[ %-])", "%1.%2") else text = text:gsub("([^.#])tl#", "%1.tl") end -- Part 2: Divide hiatuses. Any aeo, or stressed iuüy, should be syllabically divided from a following aeo or -- stressed iuüy. Also divide ii and uu sequences ([[antiincendios]], [[shiita]], [[vacuum]]). Note that words with -- ii or uu next to a vowel (e.g. [[hawaiiano]]) will not make it to this point unchanged; the i or u adjacent to -- a vowel (or the second one if both are adjacent to vowels) will get converted to a consonant symbol (temporarily -- when syllabifying spelling). text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?[aeo])", "%1.%2") text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?" .. V .. stress_c .. ")", "%1.%2") text = rsub(text, "([iuüyIUÜY]" .. stress_c .. ")(h?[aeo])", "%1.%2") text = rsub_repeatedly(text, "([iuüyIUÜY]" .. stress_c .. ")(h?" .. V .. stress_c .. ")", "%1.%2") text = rsub_repeatedly(text, "([iI]" .. accent_c .. "*)(h?i)", "%1.%2") text = rsub_repeatedly(text, "([uU]" .. accent_c .. "*)(h?u)", "%1.%2") return text end local function syllabify_from_spelling(text) text = decompose(text) -- start at FFF1 because FFF0 is used for SYLDIV -- Temporary replacements for characters we want treated as default consonants. The C and related consonant regexes -- treat all unknown characters as consonants. local TEMP_I = u(0xFFF1) local TEMP_U = u(0xFFF2) local TEMP_Y_CONS = u(0xFFF3) local TEMP_QU = u(0xFFF4) local TEMP_QU_CAPS = u(0xFFF5) local TEMP_GU = u(0xFFF6) local TEMP_GU_CAPS = u(0xFFF7) local TEMP_H = u(0xFFF8) -- Change user-specified . into SYLDIV so we don't shuffle it around when dividing into syllables. text = text:gsub("%.", SYLDIV) text = rsub(text, "y(" .. V .. ")", TEMP_Y_CONS .. "%1") -- We don't want to break -sh- except in desh-, e.g. [[deshuesar]], [[deshonra]], [[deshecho]]. Normally, -sh- is -- automatically preserved, so we replace the h with a temporary symbol to avoid this. text = text:gsub("^([Dd]es)h", "%1" .. TEMP_H) text = text:gsub("([ %-][Dd]es)h", "%1" .. TEMP_H) -- qu mostly handled correctly automatically, but not in quietud text = rsub(text, "qu(" .. V .. ")", TEMP_QU .. "%1") text = rsub(text, "Qu(" .. V .. ")", TEMP_QU_CAPS .. "%1") text = rsub(text, "gu(" .. V .. ")", TEMP_GU .. "%1") text = rsub(text, "Gu(" .. V .. ")", TEMP_GU_CAPS .. "%1") local vowel_to_glide = { ["i"] = TEMP_I, ["u"] = TEMP_U } -- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]], -- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]], -- etc. When we do this we need to help the syllabification particularly of words with -hiV- and -huV- in them, -- otherwise we get e.g. 'an.tih.ie.lo' because we converted the i following the h to a consonant. Add .* at the -- beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano. text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*)(h?)([iu])(" .. V .. ")", function (v1, h, iu, v2) return v1 .. "." .. h .. vowel_to_glide[iu] .. v2 end ) text = syllabify_from_spelling_or_pronun(text, "is spelling") text = text:gsub(SYLDIV, ".") text = text:gsub(TEMP_I, "i") text = text:gsub(TEMP_U, "u") text = text:gsub(TEMP_Y_CONS, "y") text = text:gsub(TEMP_QU, "qu") text = text:gsub(TEMP_QU_CAPS, "Qu") text = text:gsub(TEMP_GU, "gu") text = text:gsub(TEMP_GU_CAPS, "Gu") text = text:gsub(TEMP_H, "h") text = unfc(text) -- No qualifiers from dialect tags because we assume all dialects hyphenate the same way. -- FIXME: There are region-specific ways of hyphenating -tl-. See above. We don't currently handle this properly. return text end -- Generate the IPA of a given respelling, where a respelling is the representation of the pronunciation of a given -- Spanish term using Spanish spelling conventions (augmented in a few cases with extra conventions such as 'sh' for -- /ʃ/). -- ɟ and ĉ are used internally to represent [ʝ⁓ɟ͡ʝ] and [t͡ʃ] -- function export.IPA(text, dialect, phonetic) local distincion = dialect == "distincion-lleismo" or dialect == "distincion-yeismo" local lleismo = dialect == "distincion-lleismo" or dialect == "seseo-lleismo" or dialect == "quito" local rioplat = dialect == "rioplatense-sheismo" or dialect == "rioplatense-zheismo" local sheismo = dialect == "rioplatense-sheismo" local quito = dialect == "quito" local yucatan = dialect == "yucatan" local distincion_different = false local lleismo_different = false local need_rioplat = false local need_quito = false local need_yucatan = false local initial_hi = false local sheismo_different = false -- start at FFF1 because FFF0 is used for SYLDIV local TEMP_Y = u(0xFFF1) local TEMP_W = u(0xFFF2) text = ulower(text or mw.loadData("Module:headword/data").pagename) -- decompose everything but ç, ñ and ü text = decompose(text) -- convert commas and en/en dashes to IPA foot boundaries text = rsub(text, "%s*[,–—]%s*", " | ") -- question mark or exclamation point in the middle of a sentence -> IPA foot boundary text = rsub(text, "([^%s])%s*[¡!¿?]%s*([^%s])", "%1 | %2") -- canonicalize multiple spaces and remove leading and trailing spaces local function canon_spaces(text) text = rsub(text, "%s+", " ") text = rsub(text, "^ ", "") text = rsub(text, " $", "") return text end text = canon_spaces(text) -- Make prefixes unstressed unless they have an explicit stress marker; also make certain -- monosyllabic words (e.g. [[el]], [[la]], [[de]], [[en]], etc.) without stress marks be -- unstressed. local words = rsplit(text, " ") for i, word in ipairs(words) do if rfind(word, "%-$") and not rfind(word, accent_c) or unstressed_words[word] then -- add CFLEX to the last vowel not the first one, or we will mess up 'que' by -- adding the CFLEX after the 'u' words[i] = rsub(word, "^(.*" .. V .. ")", "%1" .. CFLEX) end end text = table.concat(words, " ") -- Convert hyphens to spaces, to handle [[Austria-Hungría]], [[franco-italiano]], etc. text = rsub(text, "%-", " ") -- canonicalize multiple spaces again, which may have been introduced by hyphens text = canon_spaces(text) -- now eliminate punctuation text = rsub(text, "[¡!¿?']", "") -- put # at word beginning and end and double ## at text/foot boundary beginning/end text = rsub(text, " | ", "# | #") text = "##" .. rsub(text, " ", "# #") .. "##" --determining whether "y" is a consonant or a vowel text = rsub(text, "y(" .. V .. ")", "ɟ%1") -- not the real sound -- word-final -ay/-ey/-oy/-uy is stressed whereas word-final -ai/-ei/-oi/-ui is not; in addition, -- word-final -uy is /uj/ whereas word-final -ui is /wi/ (e.g. [[muy]] vs. [[fui]]) text = rsub(text, "([aeou])y#", "%1" .. TEMP_Y .. "#") -- a temporary symbol; replaced with i below text = rsub(text, "y", "i") -- handle certain combinations; sh handling needs to go before x handling to avoid issues with [[exhausto]] text = rsub(text, "ch", "ĉ") --not the real sound -- We want to keep desh- ([[deshuesar]]) as-is. Converting to des- won't work because we want it syllabified as -- 'des.we.saɾ' not #'de.swe.saɾ' (cf. [[desuelo]] /de.swe.lo/ from [[desolar]]). text = rsub(text, "#desh", "!") --temporary symbol text = rsub(text, "sh", "ʃ") text = rsub(text, "!", "#desh") --restore text = rsub(text, "#[ckp]([st])", "#%1") -- [[ctónico]], [[psicología]], [[pterodáctilo]] --x text = rsub(text, "#x", "#s") -- xenofobia, xilófono, etc. text = rsub(text, "x", "ks") --c, g, q text = rsub(text, "c([ie])", (distincion and "θ" or "z") .. "%1") -- not the real LatAm sound text = rsub(text, "g([ie])", "x%1") -- must happen after handling of x above text = rsub(text, "gu([ie])", "g%1") text = rsub(text, "gü([ie])", "gu%1") -- following must happen before stress assignment; [[branding]] has initial stress like 'brandin' text = rsub(text, "ng([^aeiouüwhlr])", "n%1") -- [[Bangkok]], [[ángstrom]], [[branding]] text = rsub(text, "qu([ie])", "k%1") text = rsub(text, "ü", "u") -- [[Düsseldorf]], [[hübnerita]], obsolete [[freqüentemente]], etc. text = rsub(text, "q", "k") -- [[quark]], [[Qatar]], [[burqa]], [[Iraq]], etc. text = rsub(text, "[zç]", distincion and "θ" or "z") -- not the real LatAm sound; "ç" became "z" in 1726 if rfind(text, "[θz]") then distincion_different = true end -- map various consonants to their phoneme equivalent text = rsub(text, "[cjñrv]", {["c"]="k", ["j"]="x", ["ñ"]="ɲ", ["r"]="ɾ", ["v"]="b" }) -- handle word- and syllable-initial hiV ([[hielo]], [[enhiesto]], [[deshielo]], ...) local word_initial_hi, syl_initial_hi text, word_initial_hi = rsubb(text, "#h?i(" .. V .. ")", rioplat and "#j%1" or "#ɟ%1") text, syl_initial_hi = rsubb(text, "(" .. C .. sylsep_c .. "*)hi(" .. V .. ")", rioplat and "%1j%2" or "%1ɟ%2") initial_hi = word_initial_hi or syl_initial_hi -- handle word- and syllable-initial huV ([[huevo]], [[deshuesar]]) text = rsubb(text, "(" .. C_OR_WORDSEP .. sylsep_c .. "*)hu(" .. V .. ")", "%1" .. TEMP_W .. "%2") -- handle double consonants that have a pronunciation different from their single equivalents -- double l lleismo_different = rfind(text, "ll") need_quito = lleismo_different and rfind(text, "ɟ") text = rsub(text, "ll", lleismo and "ʎ" or "ɟ") -- handle intervocalic -y- need_yucatan = rfind(text, V .. accent_c .. "*" .. sylsep_c .. "*[ʎɟ]" .. V) if yucatan then text = rsub_repeatedly(text, "([ei]" .. accent_c .. "*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. sylsep_c .. "*)ɟ" .. "([ei])", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .."*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1i%2") end -- trill in #r, lr ([[alrededor]], [[malrotar]]), nr ([[enriquecer]], [[sonrisa]], etc.), sr ([[Israel]], -- [[desregular]], etc.), zr ([[Azrael]], [[cruzrojista]]), rr text = rsub(text, "ɾɾ", "r") text = rsub(text, "([#lnszθ])ɾ", "%1r") -- double n (e.g. [[[ennoblecer]]) text = rsub(text, "nn", "N") -- double b (e.g. [[subbase]]) text = rsub(text, "bb", "B") -- reduce any remaining double consonants ([[Addis Abeba]], [[cappa]], [[descender]] in Latin America ...); -- do this before handling of -nm- e.g. in [[inmigración]], which generates a double consonant, and do this -- before voicing stops before obstruents, to avoid problems with [[cappa]] and [[crackear]] text = rsub(text, "(" .. C .. ")%1", "%1") -- also reduce sz (Latin American in [[fascinante]], etc.) text = rsub(text, "sz", "s") -- restore double n, b text = rsub(text, "N", "nn") text = rsub(text, "B", "bb") -- voiceless stop to voiced before obstruent or nasal; but intercept -ts-, -tz- local voice_stop = { ["p"] = "b", ["t"] = "d", ["k"] = "g" } text = rsub(text, "t(" .. separator_c .. "*[szθ])", "!%1") -- temporary symbol text = rsub(text, "([ptk])(" .. separator_c .. "*" .. T .. ")", function(stop, after) return voice_stop[stop] .. after end) text = rsub(text, "!", "t") text = rsub(text, "n([# .]*[bpm])", "m%1") -- remove silent h before syllable division text = rsub(text, "h", "") -- convert i/u between vowels to glide local vowel_to_glide = { ["i"] = "j", ["u"] = "w" } -- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]], -- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]], -- etc. Add .* at the beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano. text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*h?)([iu])(" .. V .. ")", function (v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end ) --syllable division text = syllabify_from_spelling_or_pronun(text, false) --diphthongs; do not include TEMP_Y here text = rsub(text, "i([aeou])", "j%1") text = rsub(text, "u([aeio])", "w%1") local accent_to_stress_mark = { [AC] = "ˈ", [GR] = "ˌ", [CFLEX] = "" } local function accent_word(word, syllables) -- Now stress the word. If any accent exists in the word (including ^ indicating an unaccented word), -- put the stress mark(s) at the beginning of the indicated syllable(s). Otherwise, apply the default -- stress rule. if rfind(word, accent_c) then for i = 1, #syllables do syllables[i] = rsub(syllables[i], "^(.*)(" .. accent_c .. ")(.*)$", function(pre, accent, post) return accent_to_stress_mark[accent] .. pre .. post end ) end else -- Default stress rule. Words without vowels (e.g. IPA foot boundaries) don't get stress. if #syllables > 1 and (rfind(word, "[^" .. vowel .. "ns#]#") or rfind(word, C .. "[ns]#")) or #syllables == 1 and rfind(word, V) then syllables[#syllables] = "ˈ" .. syllables[#syllables] elseif #syllables > 1 then syllables[#syllables - 1] = "ˈ" .. syllables[#syllables - 1] end end end local words = rsplit(text, " ") for j, word in ipairs(words) do -- accentuation local syllables = rsplit(word, "%.") if rfind(word, "men%.te#") then local mente_syllables -- Words ends in -mente (converted above to ménte); add a stress to the preceding portion -- (e.g. [[agriamente]] -> 'ágriaménte') unless already stressed (e.g. [[rápidamente]]). -- It will be converted to secondary stress further below. Essentially, we rip the word apart -- into two words ('mente' and the preceding portion) and stress each one independently. mente_syllables = {} mente_syllables[2] = table.remove(syllables) mente_syllables[1] = table.remove(syllables) accent_word(table.concat(syllables, "."), syllables) accent_word(table.concat(mente_syllables, "."), mente_syllables) table.insert(syllables, mente_syllables[1]) table.insert(syllables, mente_syllables[2]) else accent_word(word, syllables) end -- Vowels are nasalized if followed by nasal in same syllable. if phonetic then for i = 1, #syllables do -- first check for two vowels (veinte) syllables[i] = rsub(syllables[i], "(" .. V .. ")(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2" .. TILDE .. "%3") -- then for one vowel syllables[i] = rsub(syllables[i], "(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2") end end -- Reconstruct the word. words[j] = table.concat(syllables, ".") end text = table.concat(words, " ") text = rsub(text, TEMP_Y, "i") --final -ay/-ey/-oy/-uy text = rsub(text, "z", "s") --real sound of LatAm Z -- suppress syllable mark before IPA stress indicator text = rsub(text, "%.(" .. ipa_stress_c .. ")", "%1") --make all primary stresses but the last one be secondary text = rsub_repeatedly(text, "ˈ(.+)ˈ", "ˌ%1ˈ") if (not initial_hi and rfind(text, "[ʎɟ]")) or (rfind(text, sylsep_c .. "[ʎɟ]")) then sheismo_different = true end if rioplat then if not initial_hi then if sheismo then text = rsub(text, "ɟ", "ʃ") else text = rsub(text, "ɟ", "ʒ") end else if sheismo then text = rsub(text, sylsep_c .. "(ɟ)", "ʃ") else text = rsub(text, sylsep_c .. "(ɟ)", "ʒ") end end end if quito then text = rsub(text, "ʎ", "ʒ") end --phonetic transcription if phonetic then -- θ, s, f before voiced consonants local voiced = "mnɲbdɟgʎ" .. TEMP_W local r = "ɾr" local tovoiced = { ["θ"] = "θ̬", ["s"] = "z", ["f"] = "v", } local function voice(sound, following) return tovoiced[sound] .. following end text = rsub(text, "([θs])(" .. separator_c .. "*[" .. voiced .. r .. "])", voice) text = rsub(text, "(f)(" .. separator_c .. "*[" .. voiced .. "])", voice) -- fricative vs. stop allophones; first convert stops to fricatives, then back to stops -- after nasals and sometimes after l local stop_to_fricative = {["b"] = "β", ["d"] = "ð", ["ɟ"] = "ʝ", ["g"] = "ɣ"} local fricative_to_stop = {["β"] = "b", ["ð"] = "d", ["ʝ"] = "ɟ", ["ɣ"] = "g"} text = rsub(text, "[bdɟg]", stop_to_fricative) text = rsub(text, "([mnɲ]" .. separator_c .. "*)([βɣ])", function(nasal, fricative) return nasal .. fricative_to_stop[fricative] end ) text = rsub(text, "([lʎmnɲ]" .. separator_c .. "*)([ðʝ])", function(nasal_l, fricative) return nasal_l .. fricative_to_stop[fricative] end ) text = rsub(text, "(##" .. ipa_stress_c .. "*)([βɣðʝ])", function(stress, fricative) return stress .. fricative_to_stop[fricative] end ) text = rsub(text, "[td]", {["t"] = "t̪", ["d"] = "d̪"}) -- nasal assimilation before consonants local labiodental, dentialveolar, dental, alveolopalatal, palatal, velar = "ɱ", "n̪", "n̟", "nʲ", "ɲ", "ŋ" local nasal_assimilation = { ["f"] = labiodental, ["t"] = dentialveolar, ["d"] = dentialveolar, ["θ"] = dental, ["ĉ"] = alveolopalatal, ["ʃ"] = alveolopalatal, ["ʒ"] = alveolopalatal, ["ɟ"] = palatal, ["ʎ"] = palatal, ["k"] = velar, ["x"] = velar, ["g"] = velar, } text = rsub(text, "n(" .. separator_c .. "*)(.)", function(stress, following) return (nasal_assimilation[following] or "n") .. stress .. following end ) -- lateral assimilation before consonants text = rsub(text, "l(" .. separator_c .. "*)(.)", function(stress, following) local l = "l" if following == "t" or following == "d" then -- dentialveolar l = "l̪" elseif following == "θ" then -- dental l = "l̟" elseif following == "ĉ" or following == "ʃ" then -- alveolopalatal l = "lʲ" end return l .. stress .. following end) --semivowels text = rsub(text, "([aeouãẽõũ][iĩ])", "%1̯") text = rsub(text, "([aeioãẽĩõ][uũ])", "%1̯") -- voiced fricatives are actually approximants text = rsub(text, "([βðɣ])", "%1̞") end -- convert fake symbols to real ones local final_conversions = { ["ħ"] = "h", -- fake aspirated "h" to real "h" ["ĉ"] = "t͡ʃ", -- fake "ch" to real "ch" ["ɟ"] = phonetic and "ɟ͡ʝ" or "ʝ", -- fake "y" to real "y" -- do the following at the very end so we can use regular g throughout ["g"] = "ɡ", -- U+0067 LATIN SMALL LETTER G → U+0261 LATIN SMALL LETTER SCRIPT G [TEMP_W] = "w̝", -- see https://en.wikipedia.org/wiki/Spanish_orthography for this } text = rsub(text, "[ħĉɟg" .. TEMP_W .. "]", final_conversions) -- remove # symbols at word and text boundaries text = rsub(text, "#", "") text = unfc(text) -- The values in `differences` are only accurate when the dialect is 'distincion-lleismo' -- because we look for sounds like /θ/ and /ʎ/ that are only present in that dialect. -- The calling code knows to only use this structure in conjunction with this dialect. -- but to make sure of this we set the structure to nil for other dialects. local differences = nil if dialect == "distincion-lleismo" then differences = { distincion_different = distincion_different, lleismo_different = lleismo_different, need_rioplat = initial_hi or sheismo_different, sheismo_different = sheismo_different, need_quito = need_quito, need_yucatan = need_yucatan, } end local ret = { text = text, differences = differences, } return ret end -- For bot usage; {{#invoke:es-pronunc|IPA_string|SPELLING|style=STYLE|phonetic=PHONETIC}} -- where -- -- 1. SPELLING is the word or respelling to generate pronunciation for; -- 2. required parameter style= indicates the pronunciation style to generate -- (e.g. "distincion-yeismo" for distinción+yeísmo, as is common in Spain; -- see the comment above export.IPA() above for the full list); -- 3. phonetic=1 specifies to generate the phonetic rather than phonemic pronunciation; function export.IPA_string(frame) local iparams = { [1] = {}, ["style"] = {required = true}, ["phonetic"] = {type = "boolean"}, } local iargs = require(parameters_module).process(frame.args, iparams) local retval = export.IPA(iargs[1], iargs.style, iargs.phonetic) return retval.text end -- Generate all relevant dialect pronunciations and group into styles. See the comment above about dialects and styles. -- A "pronunciation" here could be for example the IPA phonemic/phonetic representation of the term or the IPA form of -- the rhyme that the term belongs to. If `style_spec` is nil, this generates all styles for all dialects, but -- `style_spec` can also be a style spec such as "seseo" or "distincion+yeismo" (see comment above) to restrict the -- output. `dodialect` is a function of two arguments, `ret` and `dialect`, where `ret` is the return-value table (see -- below), and `dialect` is a string naming a particular dialect, such as "distincion-lleismo" or "rioplatense-sheismo". -- `dodialect` should side-effect the `ret` table by adding an entry to `ret.pronun` for the dialect in question. -- -- The return value is a table of the form -- -- { -- pronun = {DIALECT = {PRONUN, PRONUN, ...}, DIALECT = {PRONUN, PRONUN, ...}, ...}, -- expressed_styles = {STYLE_GROUP, STYLE_GROUP, ...}, -- } -- -- where: -- 1. DIALECT is a string such as "distincion-lleismo" naming a specific dialect. -- 2. PRONUN is a table describing a particular pronunciation. If the dialect is "distincion-lleismo", there should be -- a field in this table named `differences`, but where other fields may vary depending on the type of pronunciation -- (e.g. phonemic/phonetic or rhyme). See below for the form of the PRONUN table for phonemic/phonetic pronunciation -- vs. rhyme and the form of the `differences` field. -- 3. STYLE_GROUP is a table of the form {tag = "HIDDEN_TAG", styles = {INNER_STYLE, INNER_STYLE, ...}}. This describes -- a group of related styles (such as those for Latin America) that by default (the "hidden" form) are displayed as -- a single line, with an icon on the right to "open" the style group into the "shown" form, with multiple lines -- for each style in the group. The tag of the style group is the text displayed before the pronunciation in the -- default "hidden" form, such as "Spain" or "Latin America". It can have the special value of `false` to indicate -- that no tag text is to be displayed. Note that the pronunciation shown in the default "hidden" form is taken -- from the first style in the style group. -- 4. INNER_STYLE is a table of the form {tag = "SHOWN_TAG", pronun = {PRONUN, PRONUN, ...}}. This describes a single -- style (such as for the Andes Mountains and Paraguay in the case where the seseo+lleismo accent differs from all others), to -- be shown on a single line. `tag` is the text preceding the displayed pronunciation, or `false` if no tag text -- is to be displayed. PRONUN is a table as described above and describes a particular pronunciation. -- -- The PRONUN table has the following form for the full phonemic/phonetic pronunciation: -- -- { -- phonemic = "PHONEMIC", -- phonetic = "PHONETIC", -- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...}, -- } -- -- Here, `phonemic` is the phonemic pronunciation (displayed as /.../) and `phonetic` is the phonetic pronunciation -- (displayed as [...]). -- -- The PRONUN table has the following form for the rhyme pronunciation: -- -- { -- rhyme = "RHYME_PRONUN", -- num_syl = {NUM, NUM, ...}, -- q = nil or {QUALIFIER, QUALIFIER, ...}, -- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...}, -- } -- -- Here, `rhyme` is a phonemic pronunciation such as "ado" for [[abogado]] or "iʝa"/"iʎa" for [[tortilla]] (depending -- on the dialect), and `num_syl` is a list of the possible numbers of syllables for the term(s) that have this rhyme -- (e.g. {4} for [[abogado]], {3} for [[tortilla]] and {4, 5} for [[biología]], which may be syllabified as -- bio.lo.gí.a or bi.o.lo.gí.a). `num_syl` is used to generate syllable-count categories such as -- [[Category:Rhymes:Spanish/ia/4 syllables]] in addition to [[Category:Rhymes:Spanish/ia]]. `num_syl` may be nil to -- suppress the generation of syllable-count categories; this is typically the case with multiword terms. -- `q`, if non-nil, comes from the user using the syntax e.g. <rhyme:iʃa<q:Buenos Aires>>. -- -- The value of the `differences` field in the PRONUN table (which, as noted above, only needs to be present for the -- "distincion-lleismo" dialect, and otherwise should be nil) is a table containing flags indicating whether and how -- the per-dialect pronunciations differ. This is an optimization to avoid having to generate all six dialectal -- pronunciations and compare them. It has the following form: -- -- { -- distincion_different = BOOLEAN, -- lleismo_different = BOOLEAN, -- need_rioplat = BOOLEAN, -- sheismo_different = BOOLEAN, -- need_quito = BOOLEAN, -- need_yucatan = BOOLEAN, -- } -- -- where: -- 1. `distincion_different` should be `true` if the "distincion" and "seseo" pronunciations differ; -- 2. `lleismo_different` should be `true` if the "lleismo" and "yeismo" pronunciations differ; -- 3. `need_rioplat` should be `true` if the Rioplatense pronunciations differ from the seseo+yeismo pronunciation; -- 4. `sheismo_different` should be `true` if the "sheismo" and "zheismo" pronunciations differ. -- 5. `need_quito` should be `true` if the "quito" and "zheismo" pronunciations differ. -- 6. `need_yucatan` should be `true` if the "yucatan" and "yeismo" pronunciations differ; local function express_all_styles(style_spec, dodialect) local ret = { pronun = {}, expressed_styles = {}, } local need_rioplat local need_quito local need_yucatan -- Add a style object (see INNER_STYLE above) that represents a particular style to `ret.expressed_styles`. -- `hidden_tag` is the tag text to be used when the style group containing the style is in the default "hidden" -- state (e.g. "Spain", "Latin America" or false if there is only one style group and no tag text should be -- shown), while `tag` is the tag text to be used when the individual style is shown (e.g. a description such as -- "most of Spain and Latin America", "Andes Mountains and Paraguay" or "everywhere but Argentina and Uruguay"). -- `representative_dialect` is one of the dialects that this style represents, and whose pronunciation is stored in -- the style object. `matching_styles` is a hyphen separated string listing the isoglosses described by this style. -- For example, if the term has an ''ll'' but no ''c/z'', the `tag` text for the yeismo pronunciation will be -- "most of Spain and Latin America" and `matching_styles` will be "distincion-seseo-yeismo", indicating that -- it corresponds to both the "distincion" and "seseo" isoglosses as well as the "yeismo" isogloss. This is used -- when a particular style spec is given. If `matching_styles` is omitted, it takes its value from -- `representative_dialect`; this is used when the style contains only a single dialect. local function express_style(hidden_tag, tag, representative_dialect, matching_styles) matching_styles = matching_styles or representative_dialect -- If the Rioplatense pronunciation isn't distinctive, add all Rioplatense isoglosses. if not need_rioplat then matching_styles = matching_styles .. "-rioplatense-sheismo-zheismo" end -- also Quito if not need_quito then matching_styles = matching_styles .. "-quito" end -- Yucatan if not need_yucatan then matching_styles = matching_styles .. "-yucatan" end -- If style specified, make sure it matches the requested style. local style_matches if not style_spec then style_matches = true else local style_parts = rsplit(matching_styles, "%-") local or_styles = rsplit(style_spec, "%s*,%s*") for _, or_style in ipairs(or_styles) do local and_styles = rsplit(or_style, "%s*%+%s*") local and_matches = true for _, and_style in ipairs(and_styles) do local negate if and_style:find("^%-") then and_style = and_style:gsub("^%-", "") negate = true end local this_style_matches = false for _, part in ipairs(style_parts) do if part == and_style then this_style_matches = true break end end if negate then this_style_matches = not this_style_matches end if not this_style_matches then and_matches = false end end if and_matches then style_matches = true break end end end if not style_matches then return end -- Fetch the representative dialect's pronunciation if not already present. if not ret.pronun[representative_dialect] then dodialect(ret, representative_dialect) end -- Insert the new style into the style group, creating the group if necessary. local new_style = { tag = tag, pronun = ret.pronun[representative_dialect], } for _, hidden_tag_style in ipairs(ret.expressed_styles) do if hidden_tag_style.tag == hidden_tag then table.insert(hidden_tag_style.styles, new_style) return end end table.insert(ret.expressed_styles, { tag = hidden_tag, styles = {new_style}, }) end -- For each type of difference, figure out if the difference exists in any of the given respellings. We do this by -- generating the pronunciation for the dialect "distincion-lleismo", for each respelling. In the process of -- generating the pronunciation for a given respelling, it computes how the other dialects for that respelling -- differ. Then we take the union of these differences across the respellings. dodialect(ret, "distincion-lleismo") local differences = {} for _, difftype in ipairs { "distincion_different", "lleismo_different", "need_rioplat", "sheismo_different", "need_quito", "need_yucatan" } do for _, pronun in ipairs(ret.pronun["distincion-lleismo"]) do if pronun.differences[difftype] then differences[difftype] = true end end end local distincion_different = differences.distincion_different local lleismo_different = differences.lleismo_different need_rioplat = differences.need_rioplat local sheismo_different = differences.sheismo_different need_quito = differences.need_quito need_yucatan = differences.need_yucatan -- Now, based on the observed differences, figure out how to combine the individual dialects into styles and -- style groups. if not distincion_different and not lleismo_different then if not need_rioplat then if not need_yucatan then express_style(false, false, "distincion-lleismo", "distincion-seseo-lleismo-yeismo") else express_style(false, "everywhere but northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") end else if not need_yucatan then express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>> and <<Uruguay>>", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") else express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>, northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") end end elseif distincion_different and not lleismo_different then express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, <<Spain>>", "distincion-lleismo", "distincion-lleismo-yeismo") if not need_rioplat and not need_yucatan then express_style("<<Latin America>>, <<Philippines>>", "<<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo") else express_style("<<Latin America>>, <<Philippines>>", "most of <<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo") end elseif not distincion_different and lleismo_different then express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo") express_style(false, "<<rural>> <<northern Spain>>, northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "distincion-lleismo", "distincion-seseo-lleismo") else express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, most of <<Spain>>", "distincion-yeismo") express_style("<<Latin America>>", "most of <<Latin America>>", "seseo-yeismo") express_style("<<Equatorial Guinea>>, <<Spain>>", "<<rural>> <<northern Spain>>", "distincion-lleismo") express_style("<<Latin America>>", "northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "seseo-lleismo") end if need_rioplat then if lleismo_different then local hidden_tag = distincion_different and "<<Latin America>>" or false if sheismo_different then if not need_quito then express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs, elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo-quito") else express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs", "quito", "seseo-quito") express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo") end else express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo") end else local hidden_tag = distincion_different and "<<Latin America>>, <<Philippines>>" or false if sheismo_different then express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo") else express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo") end end end if need_yucatan then local hidden_tag = distincion_different and (lleismo_different and "<<Latin America>>" or "<<Latin America>>, <<Philippines>>") or false express_style(hidden_tag, "northern <<Mexico>>, <<Yucatán>>, <<Central America>> (except <<Panama>>)", "yucatan") end -- If only one style group, don't indicate the style. -- Not clear we want this in reality. --if #ret.expressed_styles == 1 then -- ret.expressed_styles[1].tag = false -- if #ret.expressed_styles[1].styles == 1 then -- ret.expressed_styles[1].styles[1].tag = false -- end --end return ret end local function format_all_styles(expressed_styles, format_style) for i, style_group in ipairs(expressed_styles) do if #style_group.styles == 1 then style_group.formatted, style_group.formatted_len = format_style(style_group.styles[1].tag, style_group.styles[1], i == 1) else style_group.formatted, style_group.formatted_len = format_style(style_group.tag, style_group.styles[1], i == 1) for j, style in ipairs(style_group.styles) do style.formatted, style.formatted_len = format_style(style.tag, style, i == 1 and j == 1) end end end local maxlen = 0 for i, style_group in ipairs(expressed_styles) do local this_len = style_group.formatted_len if #style_group.styles > 1 then for _, style in ipairs(style_group.styles) do this_len = math.max(this_len, style.formatted_len) end end maxlen = math.max(maxlen, this_len) end local lines = {} local need_major_hack = false for i, style_group in ipairs(expressed_styles) do if #style_group.styles == 1 then table.insert(lines, style_group.formatted) need_major_hack = false else local inline = '\n<div class="vsShow" style="display:none">\n' .. style_group.formatted .. "</div>" local full_prons = {} for _, style in ipairs(style_group.styles) do table.insert(full_prons, style.formatted) end local full = '\n<div class="vsHide">\n' .. table.concat(full_prons, "\n") .. "</div>" local em_length = math.floor(maxlen * 0.68) -- from [[Module:grc-pronunciation]] table.insert(lines, '<div class="vsSwitcher" data-toggle-category="pronunciations" style="width: ' .. em_length .. 'em; max-width:100%;"><span class="vsToggleElement" style="float: right;">&nbsp;</span>' .. inline .. full .. "</div>") need_major_hack = true end end -- major hack to get bullets working on the next line after a div box return table.concat(lines, "\n") .. (need_major_hack and "\n<span></span>" or "") end local function dodialect_pronun(args, ret, dialect) ret.pronun[dialect] = {} for i, term in ipairs(args.terms) do local phonemic, phonetic, differences if term.raw then phonemic = term.raw_phonemic phonetic = term.raw_phonetic differences = construct_default_differences(dialect) else phonemic = export.IPA(term.term, dialect, false) phonetic = export.IPA(term.term, dialect, true) differences = phonemic.differences phonemic = phonemic.text phonetic = phonetic.text end ret.pronun[dialect][i] = { raw = term.raw, phonemic = phonemic, phonetic = phonetic, refs = term.refs, q = term.q, qq = term.qq, a = term.a, aa = term.aa, differences = differences, } end end local function generate_pronun(args) local function this_dodialect_pronun(ret, dialect) dodialect_pronun(args, ret, dialect) end local ret = express_all_styles(args.style, this_dodialect_pronun) local function format_style(tag, expressed_style, is_first) local pronunciations = {} local formatted_pronuns = {} local function ins(formatted_part) table.insert(formatted_pronuns, formatted_part) end -- Loop through each pronunciation. For each one, add the phonemic and phonetic versions to `pronunciations`, -- for formatting by [[Module:IPA]], and also create an approximation of the formatted version so that we can -- compute the appropriate width of the HTML switcher div box that holds the different per-dialect variants. -- NOTE: The code below constructs the formatted approximation out-of-order in some cases but that doesn't -- currently matter because we assume all characters have the same width. If we change the width computation -- in a way that requires the correct order, we need changes to the code below. for j, pronun in ipairs(expressed_style.pronun) do -- Add tag to right accent qualifiers if last one local aas = pronun.aa if j == #expressed_style.pronun and tag then if aas then aas = m_table.deepCopy(aas) table.insert(aas, tag) else aas = {tag} end end local first_pronun = #pronunciations + 1 if not pronun.phonemic and not pronun.phonetic then error("Internal error: Saw neither phonemic nor phonetic pronunciation") end if pronun.phonemic then -- missing if 'raw:[...]' given -- don't display syllable division markers in phonemic local slash_pron = "/" .. pronun.phonemic:gsub("%.", "") .. "/" table.insert(pronunciations, { pron = slash_pron, }) ins(slash_pron) end if pronun.phonetic then -- missing if 'raw:/.../' given local bracket_pron = "[" .. pronun.phonetic .. "]" table.insert(pronunciations, { pron = bracket_pron, }) ins(bracket_pron) end local last_pronun = #pronunciations if pronun.q then pronunciations[first_pronun].q = pronun.q end if pronun.a then pronunciations[first_pronun].a = pronun.a end if j > 1 then pronunciations[first_pronun].separator = ", " ins(", ") end if pronun.qq then pronunciations[last_pronun].qq = pronun.qq end if aas then pronunciations[last_pronun].aa = aas end if pronun.q or pronun.qq or pronun.a or aas then -- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call -- to textual_len() removes all HTML and reduces links. ins(require(decorations_module).format_decorations { lang = lang, text = "", q = pronun.q, qq = pronun.qq, a = pronun.a, aa = aas, }) end if pronun.refs then pronunciations[last_pronun].refs = pronun.refs -- Approximate the reference using a footnote notation. This will be slightly inaccurate if there are -- more than nine references but that is rare. ins(string.rep("[1]", #pronun.refs)) end if first_pronun ~= last_pronun then pronunciations[last_pronun].separator = " " ins(" ") end end local bullet = string.rep("*", args.bullets) .. " " -- Here we construct the formatted line in `formatted`, and also try to construct the equivalent without HTML -- and wiki markup in `formatted_for_len`, so we can compute the approximate textual length for use in sizing -- the toggle box with the "more" button on the right. local pre = is_first and args.pre and args.pre .. " " or "" local post = is_first and args.post and " " .. args.post or "" local formatted = bullet .. pre .. m_IPA.format_IPA_full { lang = lang, items = pronunciations, separator = "" } .. post local formatted_for_len = bullet .. pre .. "IPA(key): " .. table.concat(formatted_pronuns) .. post return formatted, textual_len(formatted_for_len) end ret.text = format_all_styles(ret.expressed_styles, format_style) return ret end local function parse_respelling(respelling, pagename, parse_err) local raw_respelling = respelling:match("^raw:(.*)$") if raw_respelling then local raw_phonemic, raw_phonetic = raw_respelling:match("^/(.*)/ %[(.*)%]$") if not raw_phonemic then raw_phonemic = raw_respelling:match("^/(.*)/$") end if not raw_phonemic then raw_phonetic = raw_respelling:match("^%[(.*)%]$") end if not raw_phonemic and not raw_phonetic then parse_err(("Unable to parse raw respelling '%s', should be one of /.../, [...] or /.../ [...]") :format(raw_respelling)) end return { raw = true, raw_phonemic = raw_phonemic, raw_phonetic = raw_phonetic, } end if respelling == "+" then respelling = pagename end return {term = respelling} end -- External entry point for {{es-IPA}}. function export.show(frame) local params = { [1] = {}, ["pre"] = {}, ["post"] = {}, ["ref"] = {}, ["style"] = {}, ["bullets"] = {type = "number", default = 1}, } local parargs = frame:getParent().args local args = require(parameters_module).process(parargs, params) local text = args[1] or mw.loadData("Module:headword/data").pagename args.terms = {{term = text}} local ret = generate_pronun(args) return ret.text end -- Return the number of syllables of a phonemic representation, which should have syllable dividers in it but no -- hyphens. local function get_num_syl_from_phonemic(phonemic) -- Maybe we should just count vowels instead of the below code. phonemic = rsub(phonemic, "|", " ") -- remove IPA foot boundaries local words = rsplit(phonemic, " +") for i, word in ipairs(words) do -- IPA stress marks are syllable divisions if between characters; otherwise just remove. word = rsub(word, "(.)[ˌˈ](.)", "%1.%2") word = rsub(word, "[ˌˈ]", "") words[i] = word end -- There should be a syllable boundary between words. phonemic = table.concat(words, ".") return ulen(rsub(phonemic, "[^.]", "")) + 1 end -- Get the rhyme by truncating everything up through the last stress mark + any following consonants, and remove -- syllable boundary markers. local function convert_phonemic_to_rhyme(phonemic) -- NOTE: This works because the phonemic vowels are just [aeiou] possibly with diacritics that are separate -- Unicode chars. If we want to handle things like ɛ or ɔ we need to add them to `vowel`. return rsub(rsub(phonemic, ".*[ˌˈ]", ""), "^[^" .. vowel .. "]*", ""):gsub("%.", ""):gsub("t͡ʃ", "tʃ") end local function split_syllabified_spelling(spelling) return rsplit(spelling, "%.") end -- "Align" syllabification to original spelling by matching character-by-character, allowing for extra syllable and -- accent markers in the syllabification. If we encounter an extra syllable marker (.), we allow and keep it. If we -- encounter an extra accent marker in the syllabification, we drop it. In any other case, we return nil indicating -- the alignment failed. local function align_syllabification_to_spelling(syllab, spelling) local result = {} local syll_chars = rsplit(decompose(syllab), "") local spelling_chars = rsplit(decompose(spelling), "") local i = 1 local j = 1 while i <= #syll_chars or j <= #spelling_chars do local ci = syll_chars[i] local cj = spelling_chars[j] if ci == cj then table.insert(result, ci) i = i + 1 j = j + 1 elseif ci == "." then table.insert(result, ci) i = i + 1 elseif ci == AC or ci == GR or ci == CFLEX then -- skip character i = i + 1 else -- non-matching character return nil end end if i <= #syll_chars or j <= #spelling_chars then -- left-over characters on one side or the other return nil end return unfc(table.concat(result)) end local function generate_hyph_obj(term) return {syllabification = term, hyph = split_syllabified_spelling(term)} end -- Word should already be decomposed. local function word_has_vowels(word) return rfind(word, V) end local function all_words_have_vowels(term) local words = rsplit(decompose(term), "[ %-]") for i, word in ipairs(words) do -- Allow empty word; this occurs with prefixes and suffixes. if word ~= "" and not word_has_vowels(word) then return false end end return true end local function should_generate_rhyme_from_respelling(term) local words = rsplit(decompose(term), " +") return #words == 1 and -- no if multiple words not words[1]:find(".%-.") and -- no if word is composed of hyphenated parts (e.g. [[Austria-Hungría]]) not words[1]:find("%-$") and -- no if word is a prefix not (words[1]:find("^%-") and words[1]:find(CFLEX)) and -- no if word is an unstressed suffix word_has_vowels(words[1]) -- no if word has no vowels (e.g. a single letter) end local function should_generate_rhyme_from_ipa(ipa) return not ipa:find("%s") and word_has_vowels(decompose(ipa)) end local function dodialect_specified_rhymes(rhymes, hyphs, parsed_respellings, rhyme_ret, dialect) rhyme_ret.pronun[dialect] = {} for _, rhyme in ipairs(rhymes) do local num_syl = rhyme.num_syl local no_num_syl = false -- If user explicitly gave the rhyme but didn't explicitly specify the number of syllables, try to take it from -- the hyphenation. if not num_syl then num_syl = {} for _, hyph in ipairs(hyphs) do if should_generate_rhyme_from_respelling(hyph.syllabification) then local this_num_syl = 1 + ulen(rsub(hyph.syllabification, "[^.]", "")) m_table.insertIfNot(num_syl, this_num_syl) else no_num_syl = true break end end if no_num_syl or #num_syl == 0 then num_syl = nil end end -- If that fails and term is single-word, try to take it from the phonemic. if not no_num_syl and not num_syl then for _, parsed in ipairs(parsed_respellings) do for dialect, pronun in pairs(parsed.pronun.pronun[dialect]) do -- Check that pronun.phonemic exists (it may not if raw phonetic-only pronun is given). if pronun.phonemic then if not should_generate_rhyme_from_ipa(pronun.phonemic) then no_num_syl = true break end -- Count number of syllables by looking at syllable boundaries (including stress marks). local this_num_syl = get_num_syl_from_phonemic(pronun.phonemic) m_table.insertIfNot(num_syl, this_num_syl) end end if no_num_syl then break end end if no_num_syl or #num_syl == 0 then num_syl = nil end end table.insert(rhyme_ret.pronun[dialect], { rhyme = rhyme.rhyme, num_syl = num_syl, q = rhyme.q, qq = rhyme.qq, a = rhyme.a, aa = rhyme.aa, differences = construct_default_differences(dialect), }) end end local q_qq_inline_modifier_spec = { store = "insert-flattened", type = "qualifier", } local a_aa_inline_modifier_spec = { store = "insert-flattened", type = "labels", } local ref_inline_modifier_spec = { store = "insert-flattened", item_dest = "refs", type = "references", } -- Parse a pronunciation modifier in `arg`, the argument portion in an inline modifier (after the prefix), which -- specifies a pronunciation property such as rhyme, hyphenation/syllabification, homophones or audio. The argument -- can itself have inline modifiers, e.g. <audio:Foo.ogg<a:Colombia>>. The allowed inline modifiers are specified -- by `param_mods` (of the format expected by `parse_inline_modifiers()`); in addition to any modifiers specified -- there, the modifiers <q:...>, <qq:...>, <a:...>, <aa:...> and <ref:...> are always accepted (and can be repeated). -- `generate_obj` and `parse_err` are like in `parse_inline_modifiers()` and specify respectively a function to -- generate the object into which modifier properties are stored given the non-modifier part of the argument, and -- a function to generate an error message (given the message). Normally, a comma-separated list of pronunciation -- properties is accepted and parsed, where each element in the list can have its own inline modifiers and where -- no spaces are allowed next to the commas in order for them to be recognized as separators. If `no_split_on_comma` -- is given, only a single pronunciation property is accepted. In all cases, however, the return value is a list -- of property objects (when `no_split_on_comma` is given, the return value is a one-element list). local function parse_pron_modifier(arg, parse_err, generate_obj, param_mods, no_split_on_comma) if arg:find("<") then param_mods.q = q_qq_inline_modifier_spec param_mods.qq = q_qq_inline_modifier_spec param_mods.a = a_aa_inline_modifier_spec param_mods.aa = a_aa_inline_modifier_spec param_mods.ref = ref_inline_modifier_spec local retval = require(parse_utilities_module).parse_inline_modifiers(arg, { param_mods = param_mods, generate_obj = generate_obj, parse_err = parse_err, splitchar = not no_split_on_comma and "," or nil, }) if no_split_on_comma then retval = {retval} end return retval elseif no_split_on_comma then return {generate_obj(arg)} else local retval = {} for _, term in ipairs(split_on_comma(arg)) do table.insert(retval, generate_obj(term)) end return retval end end local function parse_rhyme(arg, parse_err) local function generate_obj(term) return {rhyme = term} end local param_mods = { s = { item_dest = "num_syl", type = "number", sublist = true, }, } return parse_pron_modifier(arg, parse_err, generate_obj, param_mods) end local function parse_hyph(arg, parse_err) -- None other than decorations local param_mods = {} return parse_pron_modifier(arg, parse_err, generate_hyph_obj, param_mods) end local function parse_homophone(arg, parse_err) local function generate_obj(term) return {term = term} end local param_mods = { t = { -- [[Module:links]] expects the gloss in "gloss". item_dest = "gloss", }, gloss = {}, -- No tr=, ts=, or sc=; doesn't make sense for Spanish. pos = {}, alt = {}, lit = {}, id = {}, g = { -- [[Module:links]] expects the genders in "genders". item_dest = "genders", sublist = true, }, } return parse_pron_modifier(arg, parse_err, generate_obj, param_mods) end local function generate_audio_obj(arg) local file, caption = arg:match("^(.-)%s*#%s*(.*)$") file = file or arg return {file = file, caption = caption} end local function parse_audio(arg, parse_err) local param_mods = { IPA = { sublist = true, }, text = {}, t = { item_dest = "gloss", }, -- No tr=, ts=, or sc=; doesn't make sense for Spanish. gloss = {}, pos = {}, -- No alt=; text= already goes in alt=. lit = {}, -- No id=; text= already goes in alt= and isn't normally linked. g = { item_dest = "genders", sublist = true, }, bad = {}, } -- Don't split on comma because some filenames have embedded commas not followed by a space -- (typically followed by an underscore). local retvals = parse_pron_modifier(arg, parse_err, generate_audio_obj, param_mods, "no split on comma") local retval = retvals[1] retval.lang = lang local textobj = require(audio_module).construct_audio_textobj(retval) retval.text = textobj retval.gloss = nil retval.pos = nil retval.lit = nil retval.genders = nil return retval end -- External entry point for {{es-pr}}. function export.show_pr(frame) local params = { [1] = {list = true}, ["rhyme"] = {convert = parse_rhyme}, ["hyph"] = {convert = parse_hyph}, ["hmp"] = {convert = parse_homophone}, ["audio"] = {list = true}, ["pagename"] = {}, } local parargs = frame:getParent().args local args = require(parameters_module).process(parargs, params) local pagename = args.pagename or mw.loadData(headword_data_module).pagename -- Parse the arguments. local respellings = #args[1] > 0 and args[1] or {"+"} local parsed_respellings = {} local overall_rhyme = args.rhyme local overall_hyph = args.hyph local overall_hmp = args.hmp local overall_audio if args.audio then -- We can't specify parse_audio() as a `convert` function because it needs access to `pagename` (i.e. another -- parameter). overall_audio = {} for i, audio in ipairs(args.audio) do local function parse_err(msg) error(("%s: parameter audio%s=%s"):format(msg, i == 1 and "" or i, audio)) end local parsed_audio = parse_audio(audio, parse_err, pagename) table.insert(overall_audio, parsed_audio) end end for i, respelling in ipairs(respellings) do if respelling:find("<") then local param_mods = { pre = { overall = true }, post = { overall = true }, style = { overall = true }, bullets = { overall = true, type = "number", }, rhyme = { overall = true, store = "insert-flattened", convert = parse_rhyme, }, hyph = { overall = true, store = "insert-flattened", convert = parse_hyph, }, hmp = { overall = true, store = "insert-flattened", convert = parse_homophone, }, audio = { overall = true, store = "insert", convert = function(arg, parse_err) return parse_audio(arg, parse_err, pagename) end, }, ref = ref_inline_modifier_spec, q = q_qq_inline_modifier_spec, qq = q_qq_inline_modifier_spec, a = a_aa_inline_modifier_spec, aa = a_aa_inline_modifier_spec, } local parsed = require(parse_utilities_module).parse_inline_modifiers(respelling, { paramname = i, param_mods = param_mods, generate_obj = function(term, parse_err) return parse_respelling(term, pagename, parse_err) end, splitchar = ",", outer_container = { audio = {}, rhyme = {}, hyph = {}, hmp = {} } }) if not parsed.bullets then parsed.bullets = 1 end table.insert(parsed_respellings, parsed) else local termobjs = {} local function parse_err(msg) error(msg .. ": " .. i .. "=" .. respelling) end for _, term in ipairs(split_on_comma(respelling)) do table.insert(termobjs, parse_respelling(term, pagename, parse_err)) end table.insert(parsed_respellings, { terms = termobjs, audio = {}, rhyme = {}, hyph = {}, hmp = {}, bullets = 1, }) end end if overall_hyph then local hyphs = {} for _, hyph in ipairs(overall_hyph) do if hyph.syllabification == "+" then hyph.syllabification = syllabify_from_spelling(pagename) hyph.hyph = split_syllabified_spelling(hyph.syllabification) elseif hyph.syllabification == "-" then overall_hyph = {} break end end end -- Loop over individual respellings, processing each. for _, parsed in ipairs(parsed_respellings) do parsed.pronun = generate_pronun(parsed) local no_auto_rhyme = false for _, term in ipairs(parsed.terms) do if term.raw then if not should_generate_rhyme_from_ipa(term.raw_phonemic or term.raw_phonetic) then no_auto_rhyme = true break end elseif not should_generate_rhyme_from_respelling(term.term) then no_auto_rhyme = true break end end if #parsed.hyph == 0 then if not overall_hyph and all_words_have_vowels(pagename) then for _, term in ipairs(parsed.terms) do if not term.raw then local syllabification = syllabify_from_spelling(term.term) local aligned_syll = align_syllabification_to_spelling(syllabification, pagename) if aligned_syll then m_table.insertIfNot(parsed.hyph, generate_hyph_obj(aligned_syll)) end end end end else for _, hyph in ipairs(parsed.hyph) do if hyph.syllabification == "+" then hyph.syllabification = syllabify_from_spelling(pagename) hyph.hyph = split_syllabified_spelling(hyph.syllabification) elseif hyph.syllabification == "-" then parsed.hyph = {} break end end end -- Generate the rhymes. local function dodialect_rhymes_from_pronun(rhyme_ret, dialect) rhyme_ret.pronun[dialect] = {} -- It's possible the pronunciation for a passed-in dialect was never generated. This happens e.g. with -- {{es-pr|cebolla<style:seseo>}}. The initial call to generate_pronun() fails to generate a pronunciation -- for the dialect 'distinction-yeismo' because the pronunciation of 'cebolla' differs between distincion -- and seseo and so the seseo style restriction rules out generation of pronunciation for distincion -- dialects (other than 'distincion-lleismo', which always gets generated so as to determine on which axes -- the dialects differ). However, when generating the rhyme, it is based only on -olla, whose pronunciation -- does not differ between distincion and seseo, but does differ between lleismo and yeismo, so it needs to -- generate a yeismo-specific rhyme, and 'distincion-yeismo' is the representative dialect for yeismo in the -- situation where distincion and seseo do not have distinct results (based on the following line in -- express_all_styles()): -- express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo") -- In this case we need to generate the missing overall pronunciation ourselves since we need it to generate -- the dialect-specific rhyme pronunciation. if not parsed.pronun.pronun[dialect] then dodialect_pronun(parsed, parsed.pronun, dialect) end for _, pronun in ipairs(parsed.pronun.pronun[dialect]) do -- We should have already excluded multiword terms and terms without vowels from rhyme generation (see -- `no_auto_rhyme` below). But make sure to check that pronun.phonemic exists (it may not if raw -- phonetic-only pronun is given). if pronun.phonemic then -- Count number of syllables by looking at syllable boundaries (including stress marks). local num_syl = get_num_syl_from_phonemic(pronun.phonemic) -- Get the rhyme by truncating everything up through the last stress mark + any following -- consonants, and remove syllable boundary markers. local rhyme = convert_phonemic_to_rhyme(pronun.phonemic) local saw_already = false for _, existing in ipairs(rhyme_ret.pronun[dialect]) do if existing.rhyme == rhyme then saw_already = true -- We already saw this rhyme but possibly with a different number of syllables, -- e.g. if the user specified two pronunciations 'biología' (4 syllables) and -- 'bi.ología' (5 syllables), both of which have the same rhyme /ia/. m_table.insertIfNot(existing.num_syl, num_syl) break end end if not saw_already then local rhyme_diffs = nil if dialect == "distincion-lleismo" then rhyme_diffs = {} if rhyme:find("θ") then rhyme_diffs.distincion_different = true end if rhyme:find("ʎ") then rhyme_diffs.lleismo_different = true if rhyme:find("ɟ") then rhyme_diffs.need_quito = true end end if rfind(rhyme, "[ʎɟ]") then rhyme_diffs.sheismo_different = true rhyme_diffs.need_rioplat = true if rfind(rhyme, V .. "[ʎɟ]" .. V) then rhyme_diffs.need_yucatan = true end end end table.insert(rhyme_ret.pronun[dialect], { rhyme = rhyme, num_syl = {num_syl}, differences = rhyme_diffs, }) end end end end if #parsed.rhyme == 0 then if overall_rhyme or no_auto_rhyme then parsed.rhyme = nil else parsed.rhyme = express_all_styles(parsed.style, dodialect_rhymes_from_pronun) end else local no_rhyme = false for _, rhyme in ipairs(parsed.rhyme) do if rhyme.rhyme == "-" then no_rhyme = true break end end if no_rhyme then parsed.rhyme = nil else local function this_dodialect(rhyme_ret, dialect) return dodialect_specified_rhymes(parsed.rhyme, parsed.hyph, {parsed}, rhyme_ret, dialect) end parsed.rhyme = express_all_styles(parsed.style, this_dodialect) end end end if overall_rhyme then local no_overall_rhyme = false for _, orhyme in ipairs(overall_rhyme) do if orhyme.rhyme == "-" then no_overall_rhyme = true break end end if no_overall_rhyme then overall_rhyme = nil else local all_hyphs if overall_hyph then all_hyphs = overall_hyph else all_hyphs = {} for _, parsed in ipairs(parsed_respellings) do for _, hyph in ipairs(parsed.hyph) do m_table.insertIfNot(all_hyphs, hyph) end end end local function dodialect_overall_rhyme(rhyme_ret, dialect) return dodialect_specified_rhymes(overall_rhyme, all_hyphs, parsed_respellings, rhyme_ret, dialect) end overall_rhyme = express_all_styles(parsed.style, dodialect_overall_rhyme) end end -- If all sets of pronunciations have the same rhymes, display them only once at the bottom. -- Otherwise, display rhymes beneath each set, indented. local first_rhyme_ret local all_rhyme_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_rhyme_ret = parsed.rhyme elseif not m_table.deepEquals(first_rhyme_ret, parsed.rhyme) then all_rhyme_sets_eq = false break end end local function format_rhyme(rhyme_ret, num_bullets) local function format_rhyme_style(tag, expressed_style, is_first) local pronunciations = {} local rhymes = {} for _, pronun in ipairs(expressed_style.pronun) do table.insert(rhymes, pronun) end local data = { lang = lang, rhymes = rhymes, aa = tag and {tag} or nil, force_cat = force_cat, } local bullet = string.rep("*", num_bullets) .. " " local formatted = bullet .. require(rhymes_module).format_rhymes(data) local formatted_for_len_parts = {} table.insert(formatted_for_len_parts, bullet .. "Rhymes: " .. (tag and "(" .. tag .. ") " or "")) for j, pronun in ipairs(expressed_style.pronun) do if j > 1 then table.insert(formatted_for_len_parts, ", ") end if pronun.q or pronun.qq or pronun.a or pronun.aa then -- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call -- to textual_len() removes all HTML and reduces links. table.insert(formatted_for_len_parts, require(decorations_module).format_decorations { lang = lang, text = "", q = pronun.q, qq = pronun.qq, a = pronun.a, aa = pronun.aa, }) end table.insert(formatted_for_len_parts, "-" .. pronun.rhyme) end return formatted, textual_len(table.concat(formatted_for_len_parts)) end return format_all_styles(rhyme_ret.expressed_styles, format_rhyme_style) end -- If all sets of pronunciations have the same hyphenations, display them only once at the bottom. -- Otherwise, display hyphenations beneath each set, indented. local first_hyphs local all_hyph_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_hyphs = parsed.hyph elseif not m_table.deepEquals(first_hyphs, parsed.hyph) then all_hyph_sets_eq = false break end end local function format_hyphenations(hyphs, num_bullets) local hyphtext = require(hyphenation_module).format_hyphenations { lang = lang, hyphs = hyphs, caption = "Syllabification" } return string.rep("*", num_bullets) .. " " .. hyphtext end -- If all sets of pronunciations have the same homophones, display them only once at the bottom. -- Otherwise, display homophones beneath each set, indented. local first_hmps local all_hmp_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_hmps = parsed.hmp elseif not m_table.deepEquals(first_hmps, parsed.hmp) then all_hmp_sets_eq = false break end end local function format_homophones(hmps, num_bullets) local hmptext = require(homophones_module).format_homophones { lang = lang, homophones = hmps } return string.rep("*", num_bullets) .. " " .. hmptext end local function format_audio(audios, num_bullets) local ret = {} for i, audio in ipairs(audios) do local text = require(audio_module).format_audio(audio) table.insert(ret, string.rep("*", num_bullets) .. " " .. text) end return table.concat(ret, "\n") end local textparts = {} local min_num_bullets = math.huge for j, parsed in ipairs(parsed_respellings) do if parsed.bullets < min_num_bullets then min_num_bullets = parsed.bullets end if j > 1 then table.insert(textparts, "\n") end table.insert(textparts, parsed.pronun.text) if #parsed.audio > 0 then table.insert(textparts, "\n") -- If only one pronunciation set, add the audio with the same number of bullets, otherwise -- indent audio by one more bullet. table.insert(textparts, format_audio(parsed.audio, #parsed_respellings == 1 and parsed.bullets or parsed.bullets + 1)) end if not all_rhyme_sets_eq and parsed.rhyme then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(parsed.rhyme, parsed.bullets + 1)) end if not all_hyph_sets_eq and #parsed.hyph > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(parsed.hyph, parsed.bullets + 1)) end if not all_hmp_sets_eq and #parsed.hmp > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(parsed.hmp, parsed.bullets + 1)) end end if overall_audio and #overall_audio > 0 then table.insert(textparts, "\n") table.insert(textparts, format_audio(overall_audio, min_num_bullets)) end if all_rhyme_sets_eq and first_rhyme_ret then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(first_rhyme_ret, min_num_bullets)) end if overall_rhyme then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(overall_rhyme, min_num_bullets)) end if all_hyph_sets_eq and #first_hyphs > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(first_hyphs, min_num_bullets)) end if overall_hyph and #overall_hyph > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(overall_hyph, min_num_bullets)) end if all_hmp_sets_eq and #first_hmps > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(first_hmps, min_num_bullets)) end if overall_hmp and #overall_hmp > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(overall_hmp, min_num_bullets)) end return table.concat(textparts) end return export 7qwnb8u8ekg7lzajzlk6jnkfpja905m 178041 178040 2026-09-23T05:08:56Z Yivan000 4078 178041 Scribunto text/plain --[=[ This module implements the templates {{es-pr}} and {{es-IPA}}. Author: Benwing2 ]=] local export = {} local m_IPA = require("Module:IPA") local m_str_utils = require("Module:string utilities") local m_table = require("Module:table") local audio_module = "Module:audio" local decorations_module = "Module:decorations" local headword_data_module = "Module:headword/data" local homophones_module = "Module:homophones" local hyphenation_module = "Module:hyphenation" local labels_module = "Module:labels" local links_module = "Module:links" local parameters_module = "Module:parameters" local parse_utilities_module = "Module:parse utilities" local references_module = "Module:references" local rhymes_module = "Module:rhymes" local force_cat = false -- for testing --[=[ FIXME: 1. Port latest changes to production module. [DONE] 2. Finish work on rhymes and hyphenation. [DONE] 3. Handle <hmp:...> for homophones. [DONE] 4. Don't add comma before phonetic IPA. [DONE] 5. Handle secondary stress, suffixes, etc. in syllabification. [DONE] 6. Need some changes to syllable splitting in consonant clusters. (e.g. 'cum‧min‧gto‧ni‧ta') [DONE] 7. Fix handling of references to correspond to Portuguese module. [DONE] 8. Propagate qualifiers on individual pronun terms to rhymes and hyph. 9. Support raw phonemic/phonetic pronunciations. [DONE] 10. Support overall audio. [DONE] 11. Keep th/ph/kh/gh/tz ([[Ertzaintza]]) together when syllabifying (but not bh due to [[subhumano]], [[subhistoria]], etc.). [DONE] 12. Support <q:...> and <qq:...> on audio. [DONE] 13. Support <a:...> and <aa:...> (using {{a|...}}, left and right) on terms, rhymes, hyphenation, homophones and audio. [DONE] 14. Support # instead of ; as separator between audio file and gloss and make sure it works if gloss has embedded # or ;. [DONE] 15. Use parse_inline_modifiers() in [[Module:parse utilities]]. [DONE] ]=] --[=[ About styles, dialects and isoglosses: From the standpoint of pronunciation, a given dialect is defined by isoglosses, which specify differences in the way of pronouncing certain phonemes. You can think of a dialect as a collection of isoglosses. For example, one isogloss is "distinción" (pronouncing written ''s'' and ''c/z'' differently) vs. "seseo" (pronouncing them the same). Another is "lleísmo" (pronouncing written ''ll'' and ''y'' differently) vs. "yeísmo" (pronouncing them the same). The dominant pronunciation in Spain can be described as distinción + yeísmo, while the pronunciation in rural northern Spain can be described as distinción + lleísmo and the pronunciation across much of the Andes mountains, Paraguay, and the Philippines can be described as seseo + lleísmo. Specifically, the following isoglosses are recognized (note, the isogloss specs as used in this module dispense with written accents): -- "distincion" = pronouncing ''s'' and ''c/z'' differently -- "seseo" = pronouncing ''s'' and ''c/z'' the same -- "lleismo" = pronouncing ''ll'' and ''y'' differently -- "yeismo" = pronouncing ''ll'' and ''y'' the same -- "rioplatense" = Rioplatense speech, i.e. seseo+yeismo with ''ll'' and ''y'' pronounced specially, and a clear distinction between initial ''hi-'' vs. initial ''ll-/y-'' -- "sheismo" = a type of Rioplatense speech, characteristic of Buenos Aires, where ''ll'' and ''y'' are pronounced as /ʃ/ -- "zheismo" = a type of Rioplatense speech, found outside of Buenos Aires, where ''ll'' and ''y'' are pronounced as /ʒ/ -- "quito" = seseo + lleismo, but pronouncing ''ll'' as /ʒ/ -- "yucatan" = seseo + yeismo, intervocalic ''y'' is pronounced ''i'' and lost in contact with ''i'' or ''e'' These isoglosses can be combined to yield one of the following eight dialects: -- "distincion-lleismo": distinción + lleísmo -- "distincion-yeismo": distinción + yeísmo -- "seseo-lleismo": seseo + lleísmo -- "seseo-yeismo": seseo + yeísmo -- "rioplatense-sheismo": Rioplatense with /ʃ/ (Buenos Aires) -- "rioplatense-zheismo": Rioplatense with /ʒ/ (non-Buenos Aires) -- "quito" -- "yucatan" A "style" here is a set of dialects that pronounce a given word in a given fashion. For example, if we are only considering the distinción/seseo and lleísmo/yeísmo isoglosses, there are four conceivable dialects (all of which in fact exist). However, for a given word, more than one dialect may pronounce it the same. For example, a word like [[paz]] has a ''z'' but no ''ll'', and so there are only two possible pronunciations for the four dialects. Here, the two styles are "Spain" and "Latin America". Correspondingly, a word like [[pollo]] with an ''ll'' but no ''z'' has two styles, which can approximately be described as "most of Spain and Latin America" vs. "rural northern Spain, Andes Mountains, Paraguay, Philippines". A "style spec" (indicated by the style= parameter to {{es-IPA}}) restricts the output to certain styles. A style spec can be one of the following: 1. An isogloss, e.g. "distincion", "rioplatense"; if specified, only styles containing this isogloss are output. 2. A negated isogloss, e.g. "-rioplatense". 3. An intersection of isoglosses ("A and B"), e.g. "distincion+lleismo". This can be used to restrict to specific dialects. 4. A union of isoglosses ("A or B"), e.g. "distincion,zheismo". If both plus and comma are used, plus takes precedence, e.g. "seseo+lleismo,zheismo" means either the "seseo+lleismo" dialect or the "rioplatense-zheismo" dialect. An example where the style= parameter might be used is with the word [[bluetooth]], which has one pronunciation in Spain/distinción (respelled "blutuz") but another in Latin America/seseo (respelled "blutud"). This might be represented using {{es-pr}} as {{es-pr|blutuz<style:distincion>|blutud<style:seseo>}}. ]=] local lang = require("Module:languages").getByCode("es") local decompose = require("Module:es-common").decompose local u = m_str_utils.char local rfind = m_str_utils.find local rsubn = m_str_utils.gsub local rsplit = m_str_utils.split local ulower = m_str_utils.lower local ulen = m_str_utils.len local unfd = mw.ustring.toNFD local unfc = mw.ustring.toNFC local AC = u(0x0301) -- acute = ́ local GR = u(0x0300) -- grave = ̀ local CFLEX = u(0x0302) -- circumflex = ̂ local TILDE = u(0x0303) -- tilde = ̃ local SYLDIV = u(0xFFF0) -- used to represent a user-specific syllable divider (.) so we won't change it local vowel = "aeiouüyAEIOUÜY" -- vowel; include y so we get single-word y correct and for syllabifying from spelling local V = "[" .. vowel .. "]" -- vowel class local accent = AC .. GR .. CFLEX local accent_c = "[" .. accent .. "]" local stress = AC .. GR local stress_c = "[" .. AC .. GR .. "]" local ipa_stress = "ˈˌ" local ipa_stress_c = "[" .. ipa_stress .. "]" local sylsep = "%-." .. SYLDIV -- hyphen included for syllabifying from spelling local sylsep_c = "[" .. sylsep .. "]" local wordsep = "# " local separator_not_wordsep = accent .. ipa_stress .. sylsep local separator = separator_not_wordsep .. wordsep local separator_c = "[" .. separator .. "]" local C = "[^" .. vowel .. separator .. "]" -- consonant class including h local C_NOT_H = "[^" .. vowel .. separator .. "h]" -- consonant class not including h local C_OR_WORDSEP = "[^" .. vowel .. separator_not_wordsep .. "]" -- consonant class including h, or word separator local T = "[^" .. vowel .. "lrɾjw" .. separator .. "]" -- obstruent or nasal local unstressed_words = m_table.listToSet({ "el", "la", "los", "las", -- definite articles "un", -- single-syllable indefinite articles "me", "te", "se", "lo", "le", "nos", "os", "les", -- unstressed object pronouns "mi", "mis", "tu", "tus", "su", "sus", -- unstressed possessive pronouns "que", "si", -- subordinating conjunctions "y", "e", "o", "u", "mas", -- coordinating conjunctions "de", "del", "a", "al", -- basic prepositions + combinations with articles "por", "en", "con", -- other prepositions }) -- version of rsubn() that discards all but the first return value local function rsub(term, foo, bar) local retval = rsubn(term, foo, bar) return retval end -- version of rsubn() that returns a 2nd argument boolean indicating whether -- a substitution was made. local function rsubb(term, foo, bar) local retval, nsubs = rsubn(term, foo, bar) return retval, nsubs > 0 end -- apply rsub() repeatedly until no change local function rsub_repeatedly(term, foo, bar) while true do local new_term = rsub(term, foo, bar) if new_term == term then return term end term = new_term end end local function split_on_comma(term) if not term then return nil end if term:find(",%s") then return require(parse_utilities_module).split_on_comma(term) elseif term:find(",") then return rsplit(term, ",") else return {term} end end -- Remove any HTML from the formatted text and resolve links, since the extra characters don't contribute to the -- displayed length. local function convert_to_raw_text(text) text = rsub(text, "<.->", "") if text:find("%[%[") then text = require(links_module).remove_links(text) end return text end -- Return the approximate displayed length in characters. local function textual_len(text) return ulen(convert_to_raw_text(text)) end local function construct_default_differences(dialect) if dialect == "distincion-lleismo" then return { distincion_different = false, lleismo_different = false, sheismo_different = false, need_rioplat = false, need_quito = false, need_yucatan = false, } end return nil end -- Main syllable-division algorithm. Can be called either directly on spelling (when hyphenating) or after -- non-trivial processing of respelling in the direction of pronunciation (when generating pronunciation). local function syllabify_from_spelling_or_pronun(text, is_spelling) -- Part 1: Divide before the last consonant in a cluster of consonants between vowels (but don't divide a VhV -- sequence; [[prohibir]] should be prohi.bir). Then move the syllable division marker leftwards over clusters that -- can form onsets. text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*)(" .. C_NOT_H .. V .. ")", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. C .. "+)(" .. C .. V .. ")", "%1.%2") -- Puerto Rico + most of Spain divide tl as t.l. Mexico and the Canary Islands have .tl. Unclear what other regions -- do. Here we choose to go with .tl. See https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf -- and https://www.spanishdict.com/guide/spanish-syllables-and-syllabification-rules. -- NOTE: When run on pronun, we have already eliminated c and v, but not when run on spelling. -- When run on pronun, don't include r, which at this point represents the trill. local cluster_r = is_spelling and "rɾ" or "ɾ" -- Don't divide Cl or Cr where C is a stop or fricative, except for dl. text = rsub(text, "([pbfvkctg])%.([l" .. cluster_r .. "])", ".%1%2") text = text:gsub("d%.([" .. cluster_r .. "])", ".d%1") -- Don't divide ch, sh, ph, th, dh, fh, kh or gh. Do allow bh to be divided ([[subhumano]], [[subhúmedo]], etc.). text = rsub(text, "([csptdfkg])%.h", ".%1h") -- Don't divide ll or rr. text = rsub(text, "([lr])%.%1", ".%1%1") -- Don't divide tz ([[Ertzaintza]], [[quetzal]], [[hertziano]] and other words of Basque, Nahuatl and German -- origin). text = rsub(text, "t%.z", ".tz") -- Per https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf, tl at the end of a word -- (as in nahuatl, Popocatepetl etc.) is divided .tl from the previous vowel. if is_spelling then text = text:gsub("([^. %-])tl$", "%1.tl") text = text:gsub("([^. %-])(tl[ %-])", "%1.%2") else text = text:gsub("([^.#])tl#", "%1.tl") end -- Part 2: Divide hiatuses. Any aeo, or stressed iuüy, should be syllabically divided from a following aeo or -- stressed iuüy. Also divide ii and uu sequences ([[antiincendios]], [[shiita]], [[vacuum]]). Note that words with -- ii or uu next to a vowel (e.g. [[hawaiiano]]) will not make it to this point unchanged; the i or u adjacent to -- a vowel (or the second one if both are adjacent to vowels) will get converted to a consonant symbol (temporarily -- when syllabifying spelling). text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?[aeo])", "%1.%2") text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?" .. V .. stress_c .. ")", "%1.%2") text = rsub(text, "([iuüyIUÜY]" .. stress_c .. ")(h?[aeo])", "%1.%2") text = rsub_repeatedly(text, "([iuüyIUÜY]" .. stress_c .. ")(h?" .. V .. stress_c .. ")", "%1.%2") text = rsub_repeatedly(text, "([iI]" .. accent_c .. "*)(h?i)", "%1.%2") text = rsub_repeatedly(text, "([uU]" .. accent_c .. "*)(h?u)", "%1.%2") return text end local function syllabify_from_spelling(text) text = decompose(text) -- start at FFF1 because FFF0 is used for SYLDIV -- Temporary replacements for characters we want treated as default consonants. The C and related consonant regexes -- treat all unknown characters as consonants. local TEMP_I = u(0xFFF1) local TEMP_U = u(0xFFF2) local TEMP_Y_CONS = u(0xFFF3) local TEMP_QU = u(0xFFF4) local TEMP_QU_CAPS = u(0xFFF5) local TEMP_GU = u(0xFFF6) local TEMP_GU_CAPS = u(0xFFF7) local TEMP_H = u(0xFFF8) -- Change user-specified . into SYLDIV so we don't shuffle it around when dividing into syllables. text = text:gsub("%.", SYLDIV) text = rsub(text, "y(" .. V .. ")", TEMP_Y_CONS .. "%1") -- We don't want to break -sh- except in desh-, e.g. [[deshuesar]], [[deshonra]], [[deshecho]]. Normally, -sh- is -- automatically preserved, so we replace the h with a temporary symbol to avoid this. text = text:gsub("^([Dd]es)h", "%1" .. TEMP_H) text = text:gsub("([ %-][Dd]es)h", "%1" .. TEMP_H) -- qu mostly handled correctly automatically, but not in quietud text = rsub(text, "qu(" .. V .. ")", TEMP_QU .. "%1") text = rsub(text, "Qu(" .. V .. ")", TEMP_QU_CAPS .. "%1") text = rsub(text, "gu(" .. V .. ")", TEMP_GU .. "%1") text = rsub(text, "Gu(" .. V .. ")", TEMP_GU_CAPS .. "%1") local vowel_to_glide = { ["i"] = TEMP_I, ["u"] = TEMP_U } -- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]], -- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]], -- etc. When we do this we need to help the syllabification particularly of words with -hiV- and -huV- in them, -- otherwise we get e.g. 'an.tih.ie.lo' because we converted the i following the h to a consonant. Add .* at the -- beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano. text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*)(h?)([iu])(" .. V .. ")", function (v1, h, iu, v2) return v1 .. "." .. h .. vowel_to_glide[iu] .. v2 end ) text = syllabify_from_spelling_or_pronun(text, "is spelling") text = text:gsub(SYLDIV, ".") text = text:gsub(TEMP_I, "i") text = text:gsub(TEMP_U, "u") text = text:gsub(TEMP_Y_CONS, "y") text = text:gsub(TEMP_QU, "qu") text = text:gsub(TEMP_QU_CAPS, "Qu") text = text:gsub(TEMP_GU, "gu") text = text:gsub(TEMP_GU_CAPS, "Gu") text = text:gsub(TEMP_H, "h") text = unfc(text) -- No qualifiers from dialect tags because we assume all dialects hyphenate the same way. -- FIXME: There are region-specific ways of hyphenating -tl-. See above. We don't currently handle this properly. return text end -- Generate the IPA of a given respelling, where a respelling is the representation of the pronunciation of a given -- Spanish term using Spanish spelling conventions (augmented in a few cases with extra conventions such as 'sh' for -- /ʃ/). -- ɟ and ĉ are used internally to represent [ʝ⁓ɟ͡ʝ] and [t͡ʃ] -- function export.IPA(text, dialect, phonetic) local distincion = dialect == "distincion-lleismo" or dialect == "distincion-yeismo" local lleismo = dialect == "distincion-lleismo" or dialect == "seseo-lleismo" or dialect == "quito" local rioplat = dialect == "rioplatense-sheismo" or dialect == "rioplatense-zheismo" local sheismo = dialect == "rioplatense-sheismo" local quito = dialect == "quito" local yucatan = dialect == "yucatan" local distincion_different = false local lleismo_different = false local need_rioplat = false local need_quito = false local need_yucatan = false local initial_hi = false local sheismo_different = false -- start at FFF1 because FFF0 is used for SYLDIV local TEMP_Y = u(0xFFF1) local TEMP_W = u(0xFFF2) text = ulower(text or mw.loadData("Module:headword/data").pagename) -- decompose everything but ç, ñ and ü text = decompose(text) -- convert commas and en/en dashes to IPA foot boundaries text = rsub(text, "%s*[,–—]%s*", " | ") -- question mark or exclamation point in the middle of a sentence -> IPA foot boundary text = rsub(text, "([^%s])%s*[¡!¿?]%s*([^%s])", "%1 | %2") -- canonicalize multiple spaces and remove leading and trailing spaces local function canon_spaces(text) text = rsub(text, "%s+", " ") text = rsub(text, "^ ", "") text = rsub(text, " $", "") return text end text = canon_spaces(text) -- Make prefixes unstressed unless they have an explicit stress marker; also make certain -- monosyllabic words (e.g. [[el]], [[la]], [[de]], [[en]], etc.) without stress marks be -- unstressed. local words = rsplit(text, " ") for i, word in ipairs(words) do if rfind(word, "%-$") and not rfind(word, accent_c) or unstressed_words[word] then -- add CFLEX to the last vowel not the first one, or we will mess up 'que' by -- adding the CFLEX after the 'u' words[i] = rsub(word, "^(.*" .. V .. ")", "%1" .. CFLEX) end end text = table.concat(words, " ") -- Convert hyphens to spaces, to handle [[Austria-Hungría]], [[franco-italiano]], etc. text = rsub(text, "%-", " ") -- canonicalize multiple spaces again, which may have been introduced by hyphens text = canon_spaces(text) -- now eliminate punctuation text = rsub(text, "[¡!¿?']", "") -- put # at word beginning and end and double ## at text/foot boundary beginning/end text = rsub(text, " | ", "# | #") text = "##" .. rsub(text, " ", "# #") .. "##" --determining whether "y" is a consonant or a vowel text = rsub(text, "y(" .. V .. ")", "ɟ%1") -- not the real sound -- word-final -ay/-ey/-oy/-uy is stressed whereas word-final -ai/-ei/-oi/-ui is not; in addition, -- word-final -uy is /uj/ whereas word-final -ui is /wi/ (e.g. [[muy]] vs. [[fui]]) text = rsub(text, "([aeou])y#", "%1" .. TEMP_Y .. "#") -- a temporary symbol; replaced with i below text = rsub(text, "y", "i") -- handle certain combinations; sh handling needs to go before x handling to avoid issues with [[exhausto]] text = rsub(text, "ch", "ĉ") --not the real sound -- We want to keep desh- ([[deshuesar]]) as-is. Converting to des- won't work because we want it syllabified as -- 'des.we.saɾ' not #'de.swe.saɾ' (cf. [[desuelo]] /de.swe.lo/ from [[desolar]]). text = rsub(text, "#desh", "!") --temporary symbol text = rsub(text, "sh", "ʃ") text = rsub(text, "!", "#desh") --restore text = rsub(text, "#[ckp]([st])", "#%1") -- [[ctónico]], [[psicología]], [[pterodáctilo]] --x text = rsub(text, "#x", "#s") -- xenofobia, xilófono, etc. text = rsub(text, "x", "ks") --c, g, q text = rsub(text, "c([ie])", (distincion and "θ" or "z") .. "%1") -- not the real LatAm sound text = rsub(text, "g([ie])", "x%1") -- must happen after handling of x above text = rsub(text, "gu([ie])", "g%1") text = rsub(text, "gü([ie])", "gu%1") -- following must happen before stress assignment; [[branding]] has initial stress like 'brandin' text = rsub(text, "ng([^aeiouüwhlr])", "n%1") -- [[Bangkok]], [[ángstrom]], [[branding]] text = rsub(text, "qu([ie])", "k%1") text = rsub(text, "ü", "u") -- [[Düsseldorf]], [[hübnerita]], obsolete [[freqüentemente]], etc. text = rsub(text, "q", "k") -- [[quark]], [[Qatar]], [[burqa]], [[Iraq]], etc. text = rsub(text, "[zç]", distincion and "θ" or "z") -- not the real LatAm sound; "ç" became "z" in 1726 if rfind(text, "[θz]") then distincion_different = true end -- map various consonants to their phoneme equivalent text = rsub(text, "[cjñrv]", {["c"]="k", ["j"]="x", ["ñ"]="ɲ", ["r"]="ɾ", ["v"]="b" }) -- handle word- and syllable-initial hiV ([[hielo]], [[enhiesto]], [[deshielo]], ...) local word_initial_hi, syl_initial_hi text, word_initial_hi = rsubb(text, "#h?i(" .. V .. ")", rioplat and "#j%1" or "#ɟ%1") text, syl_initial_hi = rsubb(text, "(" .. C .. sylsep_c .. "*)hi(" .. V .. ")", rioplat and "%1j%2" or "%1ɟ%2") initial_hi = word_initial_hi or syl_initial_hi -- handle word- and syllable-initial huV ([[huevo]], [[deshuesar]]) text = rsubb(text, "(" .. C_OR_WORDSEP .. sylsep_c .. "*)hu(" .. V .. ")", "%1" .. TEMP_W .. "%2") -- handle double consonants that have a pronunciation different from their single equivalents -- double l lleismo_different = rfind(text, "ll") need_quito = lleismo_different and rfind(text, "ɟ") text = rsub(text, "ll", lleismo and "ʎ" or "ɟ") -- handle intervocalic -y- need_yucatan = rfind(text, V .. accent_c .. "*" .. sylsep_c .. "*[ʎɟ]" .. V) if yucatan then text = rsub_repeatedly(text, "([ei]" .. accent_c .. "*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. sylsep_c .. "*)ɟ" .. "([ei])", "%1.%2") text = rsub_repeatedly(text, "(" .. V .. accent_c .."*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1i%2") end -- trill in #r, lr ([[alrededor]], [[malrotar]]), nr ([[enriquecer]], [[sonrisa]], etc.), sr ([[Israel]], -- [[desregular]], etc.), zr ([[Azrael]], [[cruzrojista]]), rr text = rsub(text, "ɾɾ", "r") text = rsub(text, "([#lnszθ])ɾ", "%1r") -- double n (e.g. [[[ennoblecer]]) text = rsub(text, "nn", "N") -- double b (e.g. [[subbase]]) text = rsub(text, "bb", "B") -- reduce any remaining double consonants ([[Addis Abeba]], [[cappa]], [[descender]] in Latin America ...); -- do this before handling of -nm- e.g. in [[inmigración]], which generates a double consonant, and do this -- before voicing stops before obstruents, to avoid problems with [[cappa]] and [[crackear]] text = rsub(text, "(" .. C .. ")%1", "%1") -- also reduce sz (Latin American in [[fascinante]], etc.) text = rsub(text, "sz", "s") -- restore double n, b text = rsub(text, "N", "nn") text = rsub(text, "B", "bb") -- voiceless stop to voiced before obstruent or nasal; but intercept -ts-, -tz- local voice_stop = { ["p"] = "b", ["t"] = "d", ["k"] = "g" } text = rsub(text, "t(" .. separator_c .. "*[szθ])", "!%1") -- temporary symbol text = rsub(text, "([ptk])(" .. separator_c .. "*" .. T .. ")", function(stop, after) return voice_stop[stop] .. after end) text = rsub(text, "!", "t") text = rsub(text, "n([# .]*[bpm])", "m%1") -- remove silent h before syllable division text = rsub(text, "h", "") -- convert i/u between vowels to glide local vowel_to_glide = { ["i"] = "j", ["u"] = "w" } -- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]], -- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]], -- etc. Add .* at the beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano. text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*h?)([iu])(" .. V .. ")", function (v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end ) --syllable division text = syllabify_from_spelling_or_pronun(text, false) --diphthongs; do not include TEMP_Y here text = rsub(text, "i([aeou])", "j%1") text = rsub(text, "u([aeio])", "w%1") local accent_to_stress_mark = { [AC] = "ˈ", [GR] = "ˌ", [CFLEX] = "" } local function accent_word(word, syllables) -- Now stress the word. If any accent exists in the word (including ^ indicating an unaccented word), -- put the stress mark(s) at the beginning of the indicated syllable(s). Otherwise, apply the default -- stress rule. if rfind(word, accent_c) then for i = 1, #syllables do syllables[i] = rsub(syllables[i], "^(.*)(" .. accent_c .. ")(.*)$", function(pre, accent, post) return accent_to_stress_mark[accent] .. pre .. post end ) end else -- Default stress rule. Words without vowels (e.g. IPA foot boundaries) don't get stress. if #syllables > 1 and (rfind(word, "[^" .. vowel .. "ns#]#") or rfind(word, C .. "[ns]#")) or #syllables == 1 and rfind(word, V) then syllables[#syllables] = "ˈ" .. syllables[#syllables] elseif #syllables > 1 then syllables[#syllables - 1] = "ˈ" .. syllables[#syllables - 1] end end end local words = rsplit(text, " ") for j, word in ipairs(words) do -- accentuation local syllables = rsplit(word, "%.") if rfind(word, "men%.te#") then local mente_syllables -- Words ends in -mente (converted above to ménte); add a stress to the preceding portion -- (e.g. [[agriamente]] -> 'ágriaménte') unless already stressed (e.g. [[rápidamente]]). -- It will be converted to secondary stress further below. Essentially, we rip the word apart -- into two words ('mente' and the preceding portion) and stress each one independently. mente_syllables = {} mente_syllables[2] = table.remove(syllables) mente_syllables[1] = table.remove(syllables) accent_word(table.concat(syllables, "."), syllables) accent_word(table.concat(mente_syllables, "."), mente_syllables) table.insert(syllables, mente_syllables[1]) table.insert(syllables, mente_syllables[2]) else accent_word(word, syllables) end -- Vowels are nasalized if followed by nasal in same syllable. if phonetic then for i = 1, #syllables do -- first check for two vowels (veinte) syllables[i] = rsub(syllables[i], "(" .. V .. ")(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2" .. TILDE .. "%3") -- then for one vowel syllables[i] = rsub(syllables[i], "(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2") end end -- Reconstruct the word. words[j] = table.concat(syllables, ".") end text = table.concat(words, " ") text = rsub(text, TEMP_Y, "i") --final -ay/-ey/-oy/-uy text = rsub(text, "z", "s") --real sound of LatAm Z -- suppress syllable mark before IPA stress indicator text = rsub(text, "%.(" .. ipa_stress_c .. ")", "%1") --make all primary stresses but the last one be secondary text = rsub_repeatedly(text, "ˈ(.+)ˈ", "ˌ%1ˈ") if (not initial_hi and rfind(text, "[ʎɟ]")) or (rfind(text, sylsep_c .. "[ʎɟ]")) then sheismo_different = true end if rioplat then if not initial_hi then if sheismo then text = rsub(text, "ɟ", "ʃ") else text = rsub(text, "ɟ", "ʒ") end else if sheismo then text = rsub(text, sylsep_c .. "(ɟ)", "ʃ") else text = rsub(text, sylsep_c .. "(ɟ)", "ʒ") end end end if quito then text = rsub(text, "ʎ", "ʒ") end --phonetic transcription if phonetic then -- θ, s, f before voiced consonants local voiced = "mnɲbdɟgʎ" .. TEMP_W local r = "ɾr" local tovoiced = { ["θ"] = "θ̬", ["s"] = "z", ["f"] = "v", } local function voice(sound, following) return tovoiced[sound] .. following end text = rsub(text, "([θs])(" .. separator_c .. "*[" .. voiced .. r .. "])", voice) text = rsub(text, "(f)(" .. separator_c .. "*[" .. voiced .. "])", voice) -- fricative vs. stop allophones; first convert stops to fricatives, then back to stops -- after nasals and sometimes after l local stop_to_fricative = {["b"] = "β", ["d"] = "ð", ["ɟ"] = "ʝ", ["g"] = "ɣ"} local fricative_to_stop = {["β"] = "b", ["ð"] = "d", ["ʝ"] = "ɟ", ["ɣ"] = "g"} text = rsub(text, "[bdɟg]", stop_to_fricative) text = rsub(text, "([mnɲ]" .. separator_c .. "*)([βɣ])", function(nasal, fricative) return nasal .. fricative_to_stop[fricative] end ) text = rsub(text, "([lʎmnɲ]" .. separator_c .. "*)([ðʝ])", function(nasal_l, fricative) return nasal_l .. fricative_to_stop[fricative] end ) text = rsub(text, "(##" .. ipa_stress_c .. "*)([βɣðʝ])", function(stress, fricative) return stress .. fricative_to_stop[fricative] end ) text = rsub(text, "[td]", {["t"] = "t̪", ["d"] = "d̪"}) -- nasal assimilation before consonants local labiodental, dentialveolar, dental, alveolopalatal, palatal, velar = "ɱ", "n̪", "n̟", "nʲ", "ɲ", "ŋ" local nasal_assimilation = { ["f"] = labiodental, ["t"] = dentialveolar, ["d"] = dentialveolar, ["θ"] = dental, ["ĉ"] = alveolopalatal, ["ʃ"] = alveolopalatal, ["ʒ"] = alveolopalatal, ["ɟ"] = palatal, ["ʎ"] = palatal, ["k"] = velar, ["x"] = velar, ["g"] = velar, } text = rsub(text, "n(" .. separator_c .. "*)(.)", function(stress, following) return (nasal_assimilation[following] or "n") .. stress .. following end ) -- lateral assimilation before consonants text = rsub(text, "l(" .. separator_c .. "*)(.)", function(stress, following) local l = "l" if following == "t" or following == "d" then -- dentialveolar l = "l̪" elseif following == "θ" then -- dental l = "l̟" elseif following == "ĉ" or following == "ʃ" then -- alveolopalatal l = "lʲ" end return l .. stress .. following end) --semivowels text = rsub(text, "([aeouãẽõũ][iĩ])", "%1̯") text = rsub(text, "([aeioãẽĩõ][uũ])", "%1̯") -- voiced fricatives are actually approximants text = rsub(text, "([βðɣ])", "%1̞") end -- convert fake symbols to real ones local final_conversions = { ["ħ"] = "h", -- fake aspirated "h" to real "h" ["ĉ"] = "t͡ʃ", -- fake "ch" to real "ch" ["ɟ"] = phonetic and "ɟ͡ʝ" or "ʝ", -- fake "y" to real "y" -- do the following at the very end so we can use regular g throughout ["g"] = "ɡ", -- U+0067 LATIN SMALL LETTER G → U+0261 LATIN SMALL LETTER SCRIPT G [TEMP_W] = "w̝", -- see https://en.wikipedia.org/wiki/Spanish_orthography for this } text = rsub(text, "[ħĉɟg" .. TEMP_W .. "]", final_conversions) -- remove # symbols at word and text boundaries text = rsub(text, "#", "") text = unfc(text) -- The values in `differences` are only accurate when the dialect is 'distincion-lleismo' -- because we look for sounds like /θ/ and /ʎ/ that are only present in that dialect. -- The calling code knows to only use this structure in conjunction with this dialect. -- but to make sure of this we set the structure to nil for other dialects. local differences = nil if dialect == "distincion-lleismo" then differences = { distincion_different = distincion_different, lleismo_different = lleismo_different, need_rioplat = initial_hi or sheismo_different, sheismo_different = sheismo_different, need_quito = need_quito, need_yucatan = need_yucatan, } end local ret = { text = text, differences = differences, } return ret end -- For bot usage; {{#invoke:es-pronunc|IPA_string|SPELLING|style=STYLE|phonetic=PHONETIC}} -- where -- -- 1. SPELLING is the word or respelling to generate pronunciation for; -- 2. required parameter style= indicates the pronunciation style to generate -- (e.g. "distincion-yeismo" for distinción+yeísmo, as is common in Spain; -- see the comment above export.IPA() above for the full list); -- 3. phonetic=1 specifies to generate the phonetic rather than phonemic pronunciation; function export.IPA_string(frame) local iparams = { [1] = {}, ["style"] = {required = true}, ["phonetic"] = {type = "boolean"}, } local iargs = require(parameters_module).process(frame.args, iparams) local retval = export.IPA(iargs[1], iargs.style, iargs.phonetic) return retval.text end -- Generate all relevant dialect pronunciations and group into styles. See the comment above about dialects and styles. -- A "pronunciation" here could be for example the IPA phonemic/phonetic representation of the term or the IPA form of -- the rhyme that the term belongs to. If `style_spec` is nil, this generates all styles for all dialects, but -- `style_spec` can also be a style spec such as "seseo" or "distincion+yeismo" (see comment above) to restrict the -- output. `dodialect` is a function of two arguments, `ret` and `dialect`, where `ret` is the return-value table (see -- below), and `dialect` is a string naming a particular dialect, such as "distincion-lleismo" or "rioplatense-sheismo". -- `dodialect` should side-effect the `ret` table by adding an entry to `ret.pronun` for the dialect in question. -- -- The return value is a table of the form -- -- { -- pronun = {DIALECT = {PRONUN, PRONUN, ...}, DIALECT = {PRONUN, PRONUN, ...}, ...}, -- expressed_styles = {STYLE_GROUP, STYLE_GROUP, ...}, -- } -- -- where: -- 1. DIALECT is a string such as "distincion-lleismo" naming a specific dialect. -- 2. PRONUN is a table describing a particular pronunciation. If the dialect is "distincion-lleismo", there should be -- a field in this table named `differences`, but where other fields may vary depending on the type of pronunciation -- (e.g. phonemic/phonetic or rhyme). See below for the form of the PRONUN table for phonemic/phonetic pronunciation -- vs. rhyme and the form of the `differences` field. -- 3. STYLE_GROUP is a table of the form {tag = "HIDDEN_TAG", styles = {INNER_STYLE, INNER_STYLE, ...}}. This describes -- a group of related styles (such as those for Latin America) that by default (the "hidden" form) are displayed as -- a single line, with an icon on the right to "open" the style group into the "shown" form, with multiple lines -- for each style in the group. The tag of the style group is the text displayed before the pronunciation in the -- default "hidden" form, such as "Spain" or "Latin America". It can have the special value of `false` to indicate -- that no tag text is to be displayed. Note that the pronunciation shown in the default "hidden" form is taken -- from the first style in the style group. -- 4. INNER_STYLE is a table of the form {tag = "SHOWN_TAG", pronun = {PRONUN, PRONUN, ...}}. This describes a single -- style (such as for the Andes Mountains and Paraguay in the case where the seseo+lleismo accent differs from all others), to -- be shown on a single line. `tag` is the text preceding the displayed pronunciation, or `false` if no tag text -- is to be displayed. PRONUN is a table as described above and describes a particular pronunciation. -- -- The PRONUN table has the following form for the full phonemic/phonetic pronunciation: -- -- { -- phonemic = "PHONEMIC", -- phonetic = "PHONETIC", -- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...}, -- } -- -- Here, `phonemic` is the phonemic pronunciation (displayed as /.../) and `phonetic` is the phonetic pronunciation -- (displayed as [...]). -- -- The PRONUN table has the following form for the rhyme pronunciation: -- -- { -- rhyme = "RHYME_PRONUN", -- num_syl = {NUM, NUM, ...}, -- q = nil or {QUALIFIER, QUALIFIER, ...}, -- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...}, -- } -- -- Here, `rhyme` is a phonemic pronunciation such as "ado" for [[abogado]] or "iʝa"/"iʎa" for [[tortilla]] (depending -- on the dialect), and `num_syl` is a list of the possible numbers of syllables for the term(s) that have this rhyme -- (e.g. {4} for [[abogado]], {3} for [[tortilla]] and {4, 5} for [[biología]], which may be syllabified as -- bio.lo.gí.a or bi.o.lo.gí.a). `num_syl` is used to generate syllable-count categories such as -- [[Category:Rhymes:Spanish/ia/4 syllables]] in addition to [[Category:Rhymes:Spanish/ia]]. `num_syl` may be nil to -- suppress the generation of syllable-count categories; this is typically the case with multiword terms. -- `q`, if non-nil, comes from the user using the syntax e.g. <rhyme:iʃa<q:Buenos Aires>>. -- -- The value of the `differences` field in the PRONUN table (which, as noted above, only needs to be present for the -- "distincion-lleismo" dialect, and otherwise should be nil) is a table containing flags indicating whether and how -- the per-dialect pronunciations differ. This is an optimization to avoid having to generate all six dialectal -- pronunciations and compare them. It has the following form: -- -- { -- distincion_different = BOOLEAN, -- lleismo_different = BOOLEAN, -- need_rioplat = BOOLEAN, -- sheismo_different = BOOLEAN, -- need_quito = BOOLEAN, -- need_yucatan = BOOLEAN, -- } -- -- where: -- 1. `distincion_different` should be `true` if the "distincion" and "seseo" pronunciations differ; -- 2. `lleismo_different` should be `true` if the "lleismo" and "yeismo" pronunciations differ; -- 3. `need_rioplat` should be `true` if the Rioplatense pronunciations differ from the seseo+yeismo pronunciation; -- 4. `sheismo_different` should be `true` if the "sheismo" and "zheismo" pronunciations differ. -- 5. `need_quito` should be `true` if the "quito" and "zheismo" pronunciations differ. -- 6. `need_yucatan` should be `true` if the "yucatan" and "yeismo" pronunciations differ; local function express_all_styles(style_spec, dodialect) local ret = { pronun = {}, expressed_styles = {}, } local need_rioplat local need_quito local need_yucatan -- Add a style object (see INNER_STYLE above) that represents a particular style to `ret.expressed_styles`. -- `hidden_tag` is the tag text to be used when the style group containing the style is in the default "hidden" -- state (e.g. "Spain", "Latin America" or false if there is only one style group and no tag text should be -- shown), while `tag` is the tag text to be used when the individual style is shown (e.g. a description such as -- "most of Spain and Latin America", "Andes Mountains and Paraguay" or "everywhere but Argentina and Uruguay"). -- `representative_dialect` is one of the dialects that this style represents, and whose pronunciation is stored in -- the style object. `matching_styles` is a hyphen separated string listing the isoglosses described by this style. -- For example, if the term has an ''ll'' but no ''c/z'', the `tag` text for the yeismo pronunciation will be -- "most of Spain and Latin America" and `matching_styles` will be "distincion-seseo-yeismo", indicating that -- it corresponds to both the "distincion" and "seseo" isoglosses as well as the "yeismo" isogloss. This is used -- when a particular style spec is given. If `matching_styles` is omitted, it takes its value from -- `representative_dialect`; this is used when the style contains only a single dialect. local function express_style(hidden_tag, tag, representative_dialect, matching_styles) matching_styles = matching_styles or representative_dialect -- If the Rioplatense pronunciation isn't distinctive, add all Rioplatense isoglosses. if not need_rioplat then matching_styles = matching_styles .. "-rioplatense-sheismo-zheismo" end -- also Quito if not need_quito then matching_styles = matching_styles .. "-quito" end -- Yucatan if not need_yucatan then matching_styles = matching_styles .. "-yucatan" end -- If style specified, make sure it matches the requested style. local style_matches if not style_spec then style_matches = true else local style_parts = rsplit(matching_styles, "%-") local or_styles = rsplit(style_spec, "%s*,%s*") for _, or_style in ipairs(or_styles) do local and_styles = rsplit(or_style, "%s*%+%s*") local and_matches = true for _, and_style in ipairs(and_styles) do local negate if and_style:find("^%-") then and_style = and_style:gsub("^%-", "") negate = true end local this_style_matches = false for _, part in ipairs(style_parts) do if part == and_style then this_style_matches = true break end end if negate then this_style_matches = not this_style_matches end if not this_style_matches then and_matches = false end end if and_matches then style_matches = true break end end end if not style_matches then return end -- Fetch the representative dialect's pronunciation if not already present. if not ret.pronun[representative_dialect] then dodialect(ret, representative_dialect) end -- Insert the new style into the style group, creating the group if necessary. local new_style = { tag = tag, pronun = ret.pronun[representative_dialect], } for _, hidden_tag_style in ipairs(ret.expressed_styles) do if hidden_tag_style.tag == hidden_tag then table.insert(hidden_tag_style.styles, new_style) return end end table.insert(ret.expressed_styles, { tag = hidden_tag, styles = {new_style}, }) end -- For each type of difference, figure out if the difference exists in any of the given respellings. We do this by -- generating the pronunciation for the dialect "distincion-lleismo", for each respelling. In the process of -- generating the pronunciation for a given respelling, it computes how the other dialects for that respelling -- differ. Then we take the union of these differences across the respellings. dodialect(ret, "distincion-lleismo") local differences = {} for _, difftype in ipairs { "distincion_different", "lleismo_different", "need_rioplat", "sheismo_different", "need_quito", "need_yucatan" } do for _, pronun in ipairs(ret.pronun["distincion-lleismo"]) do if pronun.differences[difftype] then differences[difftype] = true end end end local distincion_different = differences.distincion_different local lleismo_different = differences.lleismo_different need_rioplat = differences.need_rioplat local sheismo_different = differences.sheismo_different need_quito = differences.need_quito need_yucatan = differences.need_yucatan -- Now, based on the observed differences, figure out how to combine the individual dialects into styles and -- style groups. if not distincion_different and not lleismo_different then if not need_rioplat then if not need_yucatan then express_style(false, false, "distincion-lleismo", "distincion-seseo-lleismo-yeismo") else express_style(false, "everywhere but northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") end else if not need_yucatan then express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>> and <<Uruguay>>", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") else express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>, northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo") end end elseif distincion_different and not lleismo_different then express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, <<Spain>>", "distincion-lleismo", "distincion-lleismo-yeismo") if not need_rioplat and not need_yucatan then express_style("<<Latin America>>, <<Philippines>>", "<<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo") else express_style("<<Latin America>>, <<Philippines>>", "most of <<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo") end elseif not distincion_different and lleismo_different then express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo") express_style(false, "<<rural>> <<northern Spain>>, northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "distincion-lleismo", "distincion-seseo-lleismo") else express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, most of <<Spain>>", "distincion-yeismo") express_style("<<Latin America>>", "most of <<Latin America>>", "seseo-yeismo") express_style("<<Equatorial Guinea>>, <<Spain>>", "<<rural>> <<northern Spain>>", "distincion-lleismo") express_style("<<Latin America>>", "northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "seseo-lleismo") end if need_rioplat then if lleismo_different then local hidden_tag = distincion_different and "<<Latin America>>" or false if sheismo_different then if not need_quito then express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs, elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo-quito") else express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs", "quito", "seseo-quito") express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo") end else express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo") end else local hidden_tag = distincion_different and "<<Latin America>>, <<Philippines>>" or false if sheismo_different then express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo") express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo") else express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo") end end end if need_yucatan then local hidden_tag = distincion_different and (lleismo_different and "<<Latin America>>" or "<<Latin America>>, <<Philippines>>") or false express_style(hidden_tag, "northern <<Mexico>>, <<Yucatán>>, <<Central America>> (except <<Panama>>)", "yucatan") end -- If only one style group, don't indicate the style. -- Not clear we want this in reality. --if #ret.expressed_styles == 1 then -- ret.expressed_styles[1].tag = false -- if #ret.expressed_styles[1].styles == 1 then -- ret.expressed_styles[1].styles[1].tag = false -- end --end return ret end local function format_all_styles(expressed_styles, format_style) for i, style_group in ipairs(expressed_styles) do if #style_group.styles == 1 then style_group.formatted, style_group.formatted_len = format_style(style_group.styles[1].tag, style_group.styles[1], i == 1) else style_group.formatted, style_group.formatted_len = format_style(style_group.tag, style_group.styles[1], i == 1) for j, style in ipairs(style_group.styles) do style.formatted, style.formatted_len = format_style(style.tag, style, i == 1 and j == 1) end end end local maxlen = 0 for i, style_group in ipairs(expressed_styles) do local this_len = style_group.formatted_len if #style_group.styles > 1 then for _, style in ipairs(style_group.styles) do this_len = math.max(this_len, style.formatted_len) end end maxlen = math.max(maxlen, this_len) end local lines = {} local need_major_hack = false for i, style_group in ipairs(expressed_styles) do if #style_group.styles == 1 then table.insert(lines, style_group.formatted) need_major_hack = false else local inline = '\n<div class="vsShow" style="display:none">\n' .. style_group.formatted .. "</div>" local full_prons = {} for _, style in ipairs(style_group.styles) do table.insert(full_prons, style.formatted) end local full = '\n<div class="vsHide">\n' .. table.concat(full_prons, "\n") .. "</div>" local em_length = math.floor(maxlen * 0.68) -- from [[Module:grc-pronunciation]] table.insert(lines, '<div class="vsSwitcher" data-toggle-category="pronunciations" style="width: ' .. em_length .. 'em; max-width:100%;"><span class="vsToggleElement" style="float: right;">&nbsp;</span>' .. inline .. full .. "</div>") need_major_hack = true end end -- major hack to get bullets working on the next line after a div box return table.concat(lines, "\n") .. (need_major_hack and "\n<span></span>" or "") end local function dodialect_pronun(args, ret, dialect) ret.pronun[dialect] = {} for i, term in ipairs(args.terms) do local phonemic, phonetic, differences if term.raw then phonemic = term.raw_phonemic phonetic = term.raw_phonetic differences = construct_default_differences(dialect) else phonemic = export.IPA(term.term, dialect, false) phonetic = export.IPA(term.term, dialect, true) differences = phonemic.differences phonemic = phonemic.text phonetic = phonetic.text end ret.pronun[dialect][i] = { raw = term.raw, phonemic = phonemic, phonetic = phonetic, refs = term.refs, q = term.q, qq = term.qq, a = term.a, aa = term.aa, differences = differences, } end end local function generate_pronun(args) local function this_dodialect_pronun(ret, dialect) dodialect_pronun(args, ret, dialect) end local ret = express_all_styles(args.style, this_dodialect_pronun) local function format_style(tag, expressed_style, is_first) local pronunciations = {} local formatted_pronuns = {} local function ins(formatted_part) table.insert(formatted_pronuns, formatted_part) end -- Loop through each pronunciation. For each one, add the phonemic and phonetic versions to `pronunciations`, -- for formatting by [[Module:IPA]], and also create an approximation of the formatted version so that we can -- compute the appropriate width of the HTML switcher div box that holds the different per-dialect variants. -- NOTE: The code below constructs the formatted approximation out-of-order in some cases but that doesn't -- currently matter because we assume all characters have the same width. If we change the width computation -- in a way that requires the correct order, we need changes to the code below. for j, pronun in ipairs(expressed_style.pronun) do -- Add tag to right accent qualifiers if last one local aas = pronun.aa if j == #expressed_style.pronun and tag then if aas then aas = m_table.deepCopy(aas) table.insert(aas, tag) else aas = {tag} end end local first_pronun = #pronunciations + 1 if not pronun.phonemic and not pronun.phonetic then error("Internal error: Saw neither phonemic nor phonetic pronunciation") end if pronun.phonemic then -- missing if 'raw:[...]' given -- don't display syllable division markers in phonemic local slash_pron = "/" .. pronun.phonemic:gsub("%.", "") .. "/" table.insert(pronunciations, { pron = slash_pron, }) ins(slash_pron) end if pronun.phonetic then -- missing if 'raw:/.../' given local bracket_pron = "[" .. pronun.phonetic .. "]" table.insert(pronunciations, { pron = bracket_pron, }) ins(bracket_pron) end local last_pronun = #pronunciations if pronun.q then pronunciations[first_pronun].q = pronun.q end if pronun.a then pronunciations[first_pronun].a = pronun.a end if j > 1 then pronunciations[first_pronun].separator = ", " ins(", ") end if pronun.qq then pronunciations[last_pronun].qq = pronun.qq end if aas then pronunciations[last_pronun].aa = aas end if pronun.q or pronun.qq or pronun.a or aas then -- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call -- to textual_len() removes all HTML and reduces links. ins(require(decorations_module).format_decorations { lang = lang, text = "", q = pronun.q, qq = pronun.qq, a = pronun.a, aa = aas, }) end if pronun.refs then pronunciations[last_pronun].refs = pronun.refs -- Approximate the reference using a footnote notation. This will be slightly inaccurate if there are -- more than nine references but that is rare. ins(string.rep("[1]", #pronun.refs)) end if first_pronun ~= last_pronun then pronunciations[last_pronun].separator = " " ins(" ") end end local bullet = string.rep("*", args.bullets) .. " " -- Here we construct the formatted line in `formatted`, and also try to construct the equivalent without HTML -- and wiki markup in `formatted_for_len`, so we can compute the approximate textual length for use in sizing -- the toggle box with the "more" button on the right. local pre = is_first and args.pre and args.pre .. " " or "" local post = is_first and args.post and " " .. args.post or "" local formatted = bullet .. pre .. m_IPA.format_IPA_full { lang = lang, items = pronunciations, separator = "" } .. post local formatted_for_len = bullet .. pre .. "IPA(key): " .. table.concat(formatted_pronuns) .. post return formatted, textual_len(formatted_for_len) end ret.text = format_all_styles(ret.expressed_styles, format_style) return ret end local function parse_respelling(respelling, pagename, parse_err) local raw_respelling = respelling:match("^raw:(.*)$") if raw_respelling then local raw_phonemic, raw_phonetic = raw_respelling:match("^/(.*)/ %[(.*)%]$") if not raw_phonemic then raw_phonemic = raw_respelling:match("^/(.*)/$") end if not raw_phonemic then raw_phonetic = raw_respelling:match("^%[(.*)%]$") end if not raw_phonemic and not raw_phonetic then parse_err(("Unable to parse raw respelling '%s', should be one of /.../, [...] or /.../ [...]") :format(raw_respelling)) end return { raw = true, raw_phonemic = raw_phonemic, raw_phonetic = raw_phonetic, } end if respelling == "+" then respelling = pagename end return {term = respelling} end -- External entry point for {{es-IPA}}. function export.show(frame) local params = { [1] = {}, ["pre"] = {}, ["post"] = {}, ["ref"] = {}, ["style"] = {}, ["bullets"] = {type = "number", default = 1}, } local parargs = frame:getParent().args local args = require(parameters_module).process(parargs, params) local text = args[1] or mw.loadData("Module:headword/data").pagename args.terms = {{term = text}} local ret = generate_pronun(args) return ret.text end -- Return the number of syllables of a phonemic representation, which should have syllable dividers in it but no -- hyphens. local function get_num_syl_from_phonemic(phonemic) -- Maybe we should just count vowels instead of the below code. phonemic = rsub(phonemic, "|", " ") -- remove IPA foot boundaries local words = rsplit(phonemic, " +") for i, word in ipairs(words) do -- IPA stress marks are syllable divisions if between characters; otherwise just remove. word = rsub(word, "(.)[ˌˈ](.)", "%1.%2") word = rsub(word, "[ˌˈ]", "") words[i] = word end -- There should be a syllable boundary between words. phonemic = table.concat(words, ".") return ulen(rsub(phonemic, "[^.]", "")) + 1 end -- Get the rhyme by truncating everything up through the last stress mark + any following consonants, and remove -- syllable boundary markers. local function convert_phonemic_to_rhyme(phonemic) -- NOTE: This works because the phonemic vowels are just [aeiou] possibly with diacritics that are separate -- Unicode chars. If we want to handle things like ɛ or ɔ we need to add them to `vowel`. return rsub(rsub(phonemic, ".*[ˌˈ]", ""), "^[^" .. vowel .. "]*", ""):gsub("%.", ""):gsub("t͡ʃ", "tʃ") end local function split_syllabified_spelling(spelling) return rsplit(spelling, "%.") end -- "Align" syllabification to original spelling by matching character-by-character, allowing for extra syllable and -- accent markers in the syllabification. If we encounter an extra syllable marker (.), we allow and keep it. If we -- encounter an extra accent marker in the syllabification, we drop it. In any other case, we return nil indicating -- the alignment failed. local function align_syllabification_to_spelling(syllab, spelling) local result = {} local syll_chars = rsplit(decompose(syllab), "") local spelling_chars = rsplit(decompose(spelling), "") local i = 1 local j = 1 while i <= #syll_chars or j <= #spelling_chars do local ci = syll_chars[i] local cj = spelling_chars[j] if ci == cj then table.insert(result, ci) i = i + 1 j = j + 1 elseif ci == "." then table.insert(result, ci) i = i + 1 elseif ci == AC or ci == GR or ci == CFLEX then -- skip character i = i + 1 else -- non-matching character return nil end end if i <= #syll_chars or j <= #spelling_chars then -- left-over characters on one side or the other return nil end return unfc(table.concat(result)) end local function generate_hyph_obj(term) return {syllabification = term, hyph = split_syllabified_spelling(term)} end -- Word should already be decomposed. local function word_has_vowels(word) return rfind(word, V) end local function all_words_have_vowels(term) local words = rsplit(decompose(term), "[ %-]") for i, word in ipairs(words) do -- Allow empty word; this occurs with prefixes and suffixes. if word ~= "" and not word_has_vowels(word) then return false end end return true end local function should_generate_rhyme_from_respelling(term) local words = rsplit(decompose(term), " +") return #words == 1 and -- no if multiple words not words[1]:find(".%-.") and -- no if word is composed of hyphenated parts (e.g. [[Austria-Hungría]]) not words[1]:find("%-$") and -- no if word is a prefix not (words[1]:find("^%-") and words[1]:find(CFLEX)) and -- no if word is an unstressed suffix word_has_vowels(words[1]) -- no if word has no vowels (e.g. a single letter) end local function should_generate_rhyme_from_ipa(ipa) return not ipa:find("%s") and word_has_vowels(decompose(ipa)) end local function dodialect_specified_rhymes(rhymes, hyphs, parsed_respellings, rhyme_ret, dialect) rhyme_ret.pronun[dialect] = {} for _, rhyme in ipairs(rhymes) do local num_syl = rhyme.num_syl local no_num_syl = false -- If user explicitly gave the rhyme but didn't explicitly specify the number of syllables, try to take it from -- the hyphenation. if not num_syl then num_syl = {} for _, hyph in ipairs(hyphs) do if should_generate_rhyme_from_respelling(hyph.syllabification) then local this_num_syl = 1 + ulen(rsub(hyph.syllabification, "[^.]", "")) m_table.insertIfNot(num_syl, this_num_syl) else no_num_syl = true break end end if no_num_syl or #num_syl == 0 then num_syl = nil end end -- If that fails and term is single-word, try to take it from the phonemic. if not no_num_syl and not num_syl then for _, parsed in ipairs(parsed_respellings) do for dialect, pronun in pairs(parsed.pronun.pronun[dialect]) do -- Check that pronun.phonemic exists (it may not if raw phonetic-only pronun is given). if pronun.phonemic then if not should_generate_rhyme_from_ipa(pronun.phonemic) then no_num_syl = true break end -- Count number of syllables by looking at syllable boundaries (including stress marks). local this_num_syl = get_num_syl_from_phonemic(pronun.phonemic) m_table.insertIfNot(num_syl, this_num_syl) end end if no_num_syl then break end end if no_num_syl or #num_syl == 0 then num_syl = nil end end table.insert(rhyme_ret.pronun[dialect], { rhyme = rhyme.rhyme, num_syl = num_syl, q = rhyme.q, qq = rhyme.qq, a = rhyme.a, aa = rhyme.aa, differences = construct_default_differences(dialect), }) end end local q_qq_inline_modifier_spec = { store = "insert-flattened", type = "qualifier", } local a_aa_inline_modifier_spec = { store = "insert-flattened", type = "labels", } local ref_inline_modifier_spec = { store = "insert-flattened", item_dest = "refs", type = "references", } -- Parse a pronunciation modifier in `arg`, the argument portion in an inline modifier (after the prefix), which -- specifies a pronunciation property such as rhyme, hyphenation/syllabification, homophones or audio. The argument -- can itself have inline modifiers, e.g. <audio:Foo.ogg<a:Colombia>>. The allowed inline modifiers are specified -- by `param_mods` (of the format expected by `parse_inline_modifiers()`); in addition to any modifiers specified -- there, the modifiers <q:...>, <qq:...>, <a:...>, <aa:...> and <ref:...> are always accepted (and can be repeated). -- `generate_obj` and `parse_err` are like in `parse_inline_modifiers()` and specify respectively a function to -- generate the object into which modifier properties are stored given the non-modifier part of the argument, and -- a function to generate an error message (given the message). Normally, a comma-separated list of pronunciation -- properties is accepted and parsed, where each element in the list can have its own inline modifiers and where -- no spaces are allowed next to the commas in order for them to be recognized as separators. If `no_split_on_comma` -- is given, only a single pronunciation property is accepted. In all cases, however, the return value is a list -- of property objects (when `no_split_on_comma` is given, the return value is a one-element list). local function parse_pron_modifier(arg, parse_err, generate_obj, param_mods, no_split_on_comma) if arg:find("<") then param_mods.q = q_qq_inline_modifier_spec param_mods.qq = q_qq_inline_modifier_spec param_mods.a = a_aa_inline_modifier_spec param_mods.aa = a_aa_inline_modifier_spec param_mods.ref = ref_inline_modifier_spec local retval = require(parse_utilities_module).parse_inline_modifiers(arg, { param_mods = param_mods, generate_obj = generate_obj, parse_err = parse_err, splitchar = not no_split_on_comma and "," or nil, }) if no_split_on_comma then retval = {retval} end return retval elseif no_split_on_comma then return {generate_obj(arg)} else local retval = {} for _, term in ipairs(split_on_comma(arg)) do table.insert(retval, generate_obj(term)) end return retval end end local function parse_rhyme(arg, parse_err) local function generate_obj(term) return {rhyme = term} end local param_mods = { s = { item_dest = "num_syl", type = "number", sublist = true, }, } return parse_pron_modifier(arg, parse_err, generate_obj, param_mods) end local function parse_hyph(arg, parse_err) -- None other than decorations local param_mods = {} return parse_pron_modifier(arg, parse_err, generate_hyph_obj, param_mods) end local function parse_homophone(arg, parse_err) local function generate_obj(term) return {term = term} end local param_mods = { t = { -- [[Module:links]] expects the gloss in "gloss". item_dest = "gloss", }, gloss = {}, -- No tr=, ts=, or sc=; doesn't make sense for Spanish. pos = {}, alt = {}, lit = {}, id = {}, g = { -- [[Module:links]] expects the genders in "genders". item_dest = "genders", sublist = true, }, } return parse_pron_modifier(arg, parse_err, generate_obj, param_mods) end local function generate_audio_obj(arg) local file, caption = arg:match("^(.-)%s*#%s*(.*)$") file = file or arg return {file = file, caption = caption} end local function parse_audio(arg, parse_err) local param_mods = { IPA = { sublist = true, }, text = {}, t = { item_dest = "gloss", }, -- No tr=, ts=, or sc=; doesn't make sense for Spanish. gloss = {}, pos = {}, -- No alt=; text= already goes in alt=. lit = {}, -- No id=; text= already goes in alt= and isn't normally linked. g = { item_dest = "genders", sublist = true, }, bad = {}, } -- Don't split on comma because some filenames have embedded commas not followed by a space -- (typically followed by an underscore). local retvals = parse_pron_modifier(arg, parse_err, generate_audio_obj, param_mods, "no split on comma") local retval = retvals[1] retval.lang = lang local textobj = require(audio_module).construct_audio_textobj(retval) retval.text = textobj retval.gloss = nil retval.pos = nil retval.lit = nil retval.genders = nil return retval end -- External entry point for {{es-pr}}. function export.show_pr(frame) local params = { [1] = {list = true}, ["rhyme"] = {convert = parse_rhyme}, ["hyph"] = {convert = parse_hyph}, ["hmp"] = {convert = parse_homophone}, ["audio"] = {list = true}, ["pagename"] = {}, } local parargs = frame:getParent().args local args = require(parameters_module).process(parargs, params) local pagename = args.pagename or mw.loadData(headword_data_module).pagename -- Parse the arguments. local respellings = #args[1] > 0 and args[1] or {"+"} local parsed_respellings = {} local overall_rhyme = args.rhyme local overall_hyph = args.hyph local overall_hmp = args.hmp local overall_audio if args.audio then -- We can't specify parse_audio() as a `convert` function because it needs access to `pagename` (i.e. another -- parameter). overall_audio = {} for i, audio in ipairs(args.audio) do local function parse_err(msg) error(("%s: parameter audio%s=%s"):format(msg, i == 1 and "" or i, audio)) end local parsed_audio = parse_audio(audio, parse_err, pagename) table.insert(overall_audio, parsed_audio) end end for i, respelling in ipairs(respellings) do if respelling:find("<") then local param_mods = { pre = { overall = true }, post = { overall = true }, style = { overall = true }, bullets = { overall = true, type = "number", }, rhyme = { overall = true, store = "insert-flattened", convert = parse_rhyme, }, hyph = { overall = true, store = "insert-flattened", convert = parse_hyph, }, hmp = { overall = true, store = "insert-flattened", convert = parse_homophone, }, audio = { overall = true, store = "insert", convert = function(arg, parse_err) return parse_audio(arg, parse_err, pagename) end, }, ref = ref_inline_modifier_spec, q = q_qq_inline_modifier_spec, qq = q_qq_inline_modifier_spec, a = a_aa_inline_modifier_spec, aa = a_aa_inline_modifier_spec, } local parsed = require(parse_utilities_module).parse_inline_modifiers(respelling, { paramname = i, param_mods = param_mods, generate_obj = function(term, parse_err) return parse_respelling(term, pagename, parse_err) end, splitchar = ",", outer_container = { audio = {}, rhyme = {}, hyph = {}, hmp = {} } }) if not parsed.bullets then parsed.bullets = 1 end table.insert(parsed_respellings, parsed) else local termobjs = {} local function parse_err(msg) error(msg .. ": " .. i .. "=" .. respelling) end for _, term in ipairs(split_on_comma(respelling)) do table.insert(termobjs, parse_respelling(term, pagename, parse_err)) end table.insert(parsed_respellings, { terms = termobjs, audio = {}, rhyme = {}, hyph = {}, hmp = {}, bullets = 1, }) end end if overall_hyph then local hyphs = {} for _, hyph in ipairs(overall_hyph) do if hyph.syllabification == "+" then hyph.syllabification = syllabify_from_spelling(pagename) hyph.hyph = split_syllabified_spelling(hyph.syllabification) elseif hyph.syllabification == "-" then overall_hyph = {} break end end end -- Loop over individual respellings, processing each. for _, parsed in ipairs(parsed_respellings) do parsed.pronun = generate_pronun(parsed) local no_auto_rhyme = false for _, term in ipairs(parsed.terms) do if term.raw then if not should_generate_rhyme_from_ipa(term.raw_phonemic or term.raw_phonetic) then no_auto_rhyme = true break end elseif not should_generate_rhyme_from_respelling(term.term) then no_auto_rhyme = true break end end if #parsed.hyph == 0 then if not overall_hyph and all_words_have_vowels(pagename) then for _, term in ipairs(parsed.terms) do if not term.raw then local syllabification = syllabify_from_spelling(term.term) local aligned_syll = align_syllabification_to_spelling(syllabification, pagename) if aligned_syll then m_table.insertIfNot(parsed.hyph, generate_hyph_obj(aligned_syll)) end end end end else for _, hyph in ipairs(parsed.hyph) do if hyph.syllabification == "+" then hyph.syllabification = syllabify_from_spelling(pagename) hyph.hyph = split_syllabified_spelling(hyph.syllabification) elseif hyph.syllabification == "-" then parsed.hyph = {} break end end end -- Generate the rhymes. local function dodialect_rhymes_from_pronun(rhyme_ret, dialect) rhyme_ret.pronun[dialect] = {} -- It's possible the pronunciation for a passed-in dialect was never generated. This happens e.g. with -- {{es-pr|cebolla<style:seseo>}}. The initial call to generate_pronun() fails to generate a pronunciation -- for the dialect 'distinction-yeismo' because the pronunciation of 'cebolla' differs between distincion -- and seseo and so the seseo style restriction rules out generation of pronunciation for distincion -- dialects (other than 'distincion-lleismo', which always gets generated so as to determine on which axes -- the dialects differ). However, when generating the rhyme, it is based only on -olla, whose pronunciation -- does not differ between distincion and seseo, but does differ between lleismo and yeismo, so it needs to -- generate a yeismo-specific rhyme, and 'distincion-yeismo' is the representative dialect for yeismo in the -- situation where distincion and seseo do not have distinct results (based on the following line in -- express_all_styles()): -- express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo") -- In this case we need to generate the missing overall pronunciation ourselves since we need it to generate -- the dialect-specific rhyme pronunciation. if not parsed.pronun.pronun[dialect] then dodialect_pronun(parsed, parsed.pronun, dialect) end for _, pronun in ipairs(parsed.pronun.pronun[dialect]) do -- We should have already excluded multiword terms and terms without vowels from rhyme generation (see -- `no_auto_rhyme` below). But make sure to check that pronun.phonemic exists (it may not if raw -- phonetic-only pronun is given). if pronun.phonemic then -- Count number of syllables by looking at syllable boundaries (including stress marks). local num_syl = get_num_syl_from_phonemic(pronun.phonemic) -- Get the rhyme by truncating everything up through the last stress mark + any following -- consonants, and remove syllable boundary markers. local rhyme = convert_phonemic_to_rhyme(pronun.phonemic) local saw_already = false for _, existing in ipairs(rhyme_ret.pronun[dialect]) do if existing.rhyme == rhyme then saw_already = true -- We already saw this rhyme but possibly with a different number of syllables, -- e.g. if the user specified two pronunciations 'biología' (4 syllables) and -- 'bi.ología' (5 syllables), both of which have the same rhyme /ia/. m_table.insertIfNot(existing.num_syl, num_syl) break end end if not saw_already then local rhyme_diffs = nil if dialect == "distincion-lleismo" then rhyme_diffs = {} if rhyme:find("θ") then rhyme_diffs.distincion_different = true end if rhyme:find("ʎ") then rhyme_diffs.lleismo_different = true if rhyme:find("ɟ") then rhyme_diffs.need_quito = true end end if rfind(rhyme, "[ʎɟ]") then rhyme_diffs.sheismo_different = true rhyme_diffs.need_rioplat = true if rfind(rhyme, V .. "[ʎɟ]" .. V) then rhyme_diffs.need_yucatan = true end end end table.insert(rhyme_ret.pronun[dialect], { rhyme = rhyme, num_syl = {num_syl}, differences = rhyme_diffs, }) end end end end if #parsed.rhyme == 0 then if overall_rhyme or no_auto_rhyme then parsed.rhyme = nil else parsed.rhyme = express_all_styles(parsed.style, dodialect_rhymes_from_pronun) end else local no_rhyme = false for _, rhyme in ipairs(parsed.rhyme) do if rhyme.rhyme == "-" then no_rhyme = true break end end if no_rhyme then parsed.rhyme = nil else local function this_dodialect(rhyme_ret, dialect) return dodialect_specified_rhymes(parsed.rhyme, parsed.hyph, {parsed}, rhyme_ret, dialect) end parsed.rhyme = express_all_styles(parsed.style, this_dodialect) end end end if overall_rhyme then local no_overall_rhyme = false for _, orhyme in ipairs(overall_rhyme) do if orhyme.rhyme == "-" then no_overall_rhyme = true break end end if no_overall_rhyme then overall_rhyme = nil else local all_hyphs if overall_hyph then all_hyphs = overall_hyph else all_hyphs = {} for _, parsed in ipairs(parsed_respellings) do for _, hyph in ipairs(parsed.hyph) do m_table.insertIfNot(all_hyphs, hyph) end end end local function dodialect_overall_rhyme(rhyme_ret, dialect) return dodialect_specified_rhymes(overall_rhyme, all_hyphs, parsed_respellings, rhyme_ret, dialect) end overall_rhyme = express_all_styles(parsed.style, dodialect_overall_rhyme) end end -- If all sets of pronunciations have the same rhymes, display them only once at the bottom. -- Otherwise, display rhymes beneath each set, indented. local first_rhyme_ret local all_rhyme_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_rhyme_ret = parsed.rhyme elseif not m_table.deepEquals(first_rhyme_ret, parsed.rhyme) then all_rhyme_sets_eq = false break end end local function format_rhyme(rhyme_ret, num_bullets) local function format_rhyme_style(tag, expressed_style, is_first) local pronunciations = {} local rhymes = {} for _, pronun in ipairs(expressed_style.pronun) do table.insert(rhymes, pronun) end local data = { lang = lang, rhymes = rhymes, aa = tag and {tag} or nil, force_cat = force_cat, } local bullet = string.rep("*", num_bullets) .. " " local formatted = bullet .. require(rhymes_module).format_rhymes(data) local formatted_for_len_parts = {} table.insert(formatted_for_len_parts, bullet .. "Rhymes: " .. (tag and "(" .. tag .. ") " or "")) for j, pronun in ipairs(expressed_style.pronun) do if j > 1 then table.insert(formatted_for_len_parts, ", ") end if pronun.q or pronun.qq or pronun.a or pronun.aa then -- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call -- to textual_len() removes all HTML and reduces links. table.insert(formatted_for_len_parts, require(decorations_module).format_decorations { lang = lang, text = "", q = pronun.q, qq = pronun.qq, a = pronun.a, aa = pronun.aa, }) end table.insert(formatted_for_len_parts, "-" .. pronun.rhyme) end return formatted, textual_len(table.concat(formatted_for_len_parts)) end return format_all_styles(rhyme_ret.expressed_styles, format_rhyme_style) end -- If all sets of pronunciations have the same hyphenations, display them only once at the bottom. -- Otherwise, display hyphenations beneath each set, indented. local first_hyphs local all_hyph_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_hyphs = parsed.hyph elseif not m_table.deepEquals(first_hyphs, parsed.hyph) then all_hyph_sets_eq = false break end end local function format_hyphenations(hyphs, num_bullets) local hyphtext = require(hyphenation_module).format_hyphenations { lang = lang, hyphs = hyphs, caption = "Pagpapantig" } --TLCHANGE "Syllabification" return string.rep("*", num_bullets) .. " " .. hyphtext end -- If all sets of pronunciations have the same homophones, display them only once at the bottom. -- Otherwise, display homophones beneath each set, indented. local first_hmps local all_hmp_sets_eq = true for j, parsed in ipairs(parsed_respellings) do if j == 1 then first_hmps = parsed.hmp elseif not m_table.deepEquals(first_hmps, parsed.hmp) then all_hmp_sets_eq = false break end end local function format_homophones(hmps, num_bullets) local hmptext = require(homophones_module).format_homophones { lang = lang, homophones = hmps } return string.rep("*", num_bullets) .. " " .. hmptext end local function format_audio(audios, num_bullets) local ret = {} for i, audio in ipairs(audios) do local text = require(audio_module).format_audio(audio) table.insert(ret, string.rep("*", num_bullets) .. " " .. text) end return table.concat(ret, "\n") end local textparts = {} local min_num_bullets = math.huge for j, parsed in ipairs(parsed_respellings) do if parsed.bullets < min_num_bullets then min_num_bullets = parsed.bullets end if j > 1 then table.insert(textparts, "\n") end table.insert(textparts, parsed.pronun.text) if #parsed.audio > 0 then table.insert(textparts, "\n") -- If only one pronunciation set, add the audio with the same number of bullets, otherwise -- indent audio by one more bullet. table.insert(textparts, format_audio(parsed.audio, #parsed_respellings == 1 and parsed.bullets or parsed.bullets + 1)) end if not all_rhyme_sets_eq and parsed.rhyme then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(parsed.rhyme, parsed.bullets + 1)) end if not all_hyph_sets_eq and #parsed.hyph > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(parsed.hyph, parsed.bullets + 1)) end if not all_hmp_sets_eq and #parsed.hmp > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(parsed.hmp, parsed.bullets + 1)) end end if overall_audio and #overall_audio > 0 then table.insert(textparts, "\n") table.insert(textparts, format_audio(overall_audio, min_num_bullets)) end if all_rhyme_sets_eq and first_rhyme_ret then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(first_rhyme_ret, min_num_bullets)) end if overall_rhyme then table.insert(textparts, "\n") table.insert(textparts, format_rhyme(overall_rhyme, min_num_bullets)) end if all_hyph_sets_eq and #first_hyphs > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(first_hyphs, min_num_bullets)) end if overall_hyph and #overall_hyph > 0 then table.insert(textparts, "\n") table.insert(textparts, format_hyphenations(overall_hyph, min_num_bullets)) end if all_hmp_sets_eq and #first_hmps > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(first_hmps, min_num_bullets)) end if overall_hmp and #overall_hmp > 0 then table.insert(textparts, "\n") table.insert(textparts, format_homophones(overall_hmp, min_num_bullets)) end return table.concat(textparts) end return export ad08xkdkfd3zcgx3eicevfi4jt8ln8v