Wiksiyonaryo
tlwiktionary
https://tl.wiktionary.org/wiki/Wiksiyonaryo:Unang_Pahina
MediaWiki 1.47.0-wmf.20
case-sensitive
Midya
Natatangi
Usapan
Tagagamit
Usapang tagagamit
Wiksiyonaryo
Usapang Wiksiyonaryo
Talaksan
Usapang talaksan
MediaWiki
Usapang MediaWiki
Padron
Usapang padron
Tulong
Usapang tulong
Kategorya
Usapang kategorya
TimedText
TimedText talk
Module
Module talk
Event
Event talk
sa katunayan
0
3392
178037
9702
2026-09-22T15:55:19Z
Yivan000
4078
Inilipat ni Yivan000 ang pahinang [[Sa katunayan]] sa [[sa katunayan]] nang walang iniwang redirect
9702
wikitext
text/x-wiki
"Sa Katunayan' isang bahagi ng pangungusap na katumbas ng "sa totoo lang". higit na malinaw at may paggalang ang pagbigkas nito dahil ito ang dating anyo nito. Halimbawa; 1.) Sa paghahambing ng "sa totoo lang" at ng "sa katunayan", may paggalang ang gamit ng pangalawa dahil hindi naman sa parlor naguusap kitang lahat.
lqhtlxfvicqgj55holm3ffo3bxai1d1
siya nawa
0
5041
178038
15990
2026-09-23T05:02:57Z
Yivan000
4078
Inilipat ni Yivan000 ang pahinang [[Siya Nawa]] sa [[siya nawa]] nang walang iniwang redirect
15990
wikitext
text/x-wiki
"Siya Nawa" ang huling bahagi ng panalangin na siyang "Amen" sa ingles at espaniol.Ang "Siya Nawa" ay nagpapahayag ng "Siya" at wala nang iba, kasunod ng "Nawa" na siyang gumawa sa lahat at maykapangyarihang papangyarihin ang hinihiling sa bisa ng banal na ngalan niya.
g7b8ytiygitq91zjft0m7shy8iup43t
Module:qualifier
828
31088
178044
166535
2026-09-23T05:16:12Z
Yivan000
4078
enwikt parity
178044
Scribunto
text/plain
local export = {}
local concat = table.concat
--[==[
Wrap text in one or more CSS classes. `classes` should be a string; separate multiple classes with a space.
]==]
function export.wrap_css(text, classes)
return ("<span class=\"%s\">%s</span>"):format(classes, text)
end
--[==[
Wrap text in one or more qualifier CSS classes. `suffix` is the suffix describing the type of content, e.g. `brac`
for parens, `content` for content, `comma` for commas. CSS classes <code>ib-<var>suffix</var></code> and
i<code>qualifier-<var>suffix</var></code> are added.
]==]
function export.wrap_qualifier_css(text, suffix)
local css_classes = ("ib-%s qualifier-%s"):format(suffix, suffix)
return export.wrap_css(text, css_classes)
end
--[==[
Format one or more qualifiers. `data` is an object with the following fields:
* `qualifiers`: A single qualifier or a list or qualifiers.
* `open`: Override the open paren displayed before the qualifiers. If `false` or an empty string, no paren is displayed.
* `close`: Override the close paren displayed before the qualifiers. If `false` or an empty string, no paren is
displayed.
* `opencontent`: Content to display before the qualifiers, after the open paren.
* `closecontent`: Content to display after the qualifiers, before the close paren.
* `no_ib_content`: Suppress wrapping the content with classes `ib-content` and `qualifier-content`. Parens and commas
will still be wrapped in CSS.
* `raw`: Suppress all CSS wrapping.
]==]
function export.format_qualifiers(data)
local qualifiers, open, close = data.qualifiers, data.open, data.close
if type(qualifiers) ~= "table" then
qualifiers = {qualifiers}
end
if not qualifiers[1] then
return ""
end
local parts = {}
local function ins(text)
table.insert(parts, text)
end
local function wrap_qualifier_css(text, suffix)
if data.raw then
return text
end
return export.wrap_qualifier_css(text, suffix)
end
if open ~= false and open ~= ""then
ins(wrap_qualifier_css(open or "(", "brac"))
end
if data.opencontent then
ins(data.opencontent)
end
local content = concat(qualifiers, wrap_qualifier_css(",", "comma") .. " ")
if not data.no_ib_content then
content = wrap_qualifier_css(content, "content")
end
ins(content)
if data.closecontent then
ins(data.closecontent)
end
if close ~= false and close ~= "" then
ins(wrap_qualifier_css(close or ")", "brac"))
end
return concat(parts)
end
--[==[
An older interface onto `format_qualifiers`. Eventually code should be converted to use the new entry point.
]==]
function export.format_qualifier(qualifiers, open, close, opencontent, closecontent, no_ib_content)
return export.format_qualifiers {
qualifiers = qualifiers,
open = open,
close = close,
opencontent = opencontent,
closecontent = closecontent,
no_ib_content = no_ib_content,
}
end
local function format_qualifiers_with_clarification(qualifiers, clarification, openquote, closequote)
local opencontent = export.wrap_css(clarification, "qualifier-clarification") ..
export.wrap_css(openquote or "“", "qualifier-clarification qualifier-quote")
local closecontent = export.wrap_css(closequote or "”", "qualifier-clarification qualifier-quote")
return export.format_qualifiers {
qualifiers = qualifiers,
open = "(",
close = ")",
opencontent = opencontent,
closecontent = closecontent,
}
end
--[==[
Internal implementation of {{tl|sense}}.
]==]
function export.sense(qualifiers)
return export.format_qualifiers {
qualifiers = qualifiers
}.. export.wrap_css(":", "ib-colon sense-qualifier-colon")
end
--[==[
Internal implementation of {{tl|antsense}}.
]==]
function export.antsense(qualifiers)
return format_qualifiers_with_clarification(qualifiers, "antonym(s) of ") ..
export.wrap_css(":", "ib-colon sense-qualifier-colon")
end
return export
tc3tigewud11x9n9900h1tvmhjpvcah
Module:IPA
828
31186
178042
176332
2026-09-23T05:14:17Z
Yivan000
4078
merge changes
178042
Scribunto
text/plain
local export = {}
local force_cat = false -- for testing
local decorations_module = "Module:decorations"
local pages_module = "Module:pages"
local qualifier_module = "Module:qualifier"
local string_utilities_module = "Module:string utilities"
local syllables_module = "Module:syllables"
local utilities_module = "Module:utilities"
local m_data = mw.loadData("Module:IPA/data")
local m_str_utils = require(string_utilities_module)
local m_syllables -- [[Module:syllables]]; loaded below if needed
local m_symbols = mw.loadData("Module:IPA/data/symbols")
local concat = table.concat
local decode_entities = m_str_utils.decode_entities
local find = string.find
local gcodepoint = m_str_utils.gcodepoint
local gmatch = m_str_utils.gmatch
local gsub = string.gsub
local insert = table.insert
local is_preview = require(pages_module).is_preview
local len = m_str_utils.len
local listToText = mw.text.listToText
local match = string.match
local pattern_escape = m_str_utils.pattern_escape
local sub = string.sub
local u = m_str_utils.char
local ugsub = m_str_utils.gsub
local umatch = m_str_utils.match
local usub = m_str_utils.sub
local function with_codepoints(s)
if find(s, "%S%s") then
local parts = {}
for ch in gmatch(s, "%S") do
parts[#parts + 1] = with_codepoints(ch)
end
return concat(parts, ", ")
end
local cps = {}
for cp in gcodepoint(s) do
cps[#cps + 1] = ("U+%04X"):format(cp)
end
return s .. " [" .. concat(cps, " ") .. "]"
end
local namespace = mw.title.getCurrentTitle().nsText
local function is_content_page(lang, namespace)
return namespace == "" or namespace == "Reconstruction" or
lang and lang:hasType("appendix-constructed") and namespace == "Appendix"
end
-- Etymology-only languages are not L2 entry languages; IPA should use the parent full language.
local function assert_not_etymology_only_lang(lang)
if lang and lang.hasType and lang:hasType("language", "etymology-only") then
local parent_code = lang.getParentCode and lang:getParentCode() or nil
error(("Cannot use IPA with the etymology-only language %q; use the parent full language %q instead."):format(lang:getCode(), parent_code))
end
end
local function track(page)
require("Module:debug/track")("IPA/" .. page)
return true
end
local function process_maybe_split_categories(split_output, categories, prontext, lang, errtext)
if split_output ~= "raw" then
if categories[1] then
categories = require(utilities_module).format_categories(categories, lang, nil, nil, force_cat)
else
categories = ""
end
end
if split_output then -- for use of IPA in links, etc.
if errtext then
return prontext, categories, errtext
else
return prontext, categories
end
else
return prontext .. (errtext or "") .. categories
end
end
--[==[
Format a line of one or more IPA pronunciations as {{tl|IPA}} would do it, i.e. with a preceding {"IPA:"} followed by
the word {"key"} linking to an Appendix page describing the language's phonology, and with an added category
` ``lang`` terms with IPA pronunciation`. Other than the extra preceding text and category, this is identical
to {format_IPA_multiple()}, and the considerations described there in the documentation apply here as well. There is a
single parameter `data`, an object with the following fields:
* `lang`: Object representing the language of the pronunciations, which is used when adding cleanup categories for
pronunciations with invalid phonemes; for determining how many syllables the pronunciations have in them, in order to
add a category such as [[:Category:Italian 2-syllable words]] (for certain languages only); for adding a category
` ``lang`` terms with IPA pronunciation`; and for determining the proper sort keys for categories. Unlike
for {format_IPA_multiple()}, `lang` may not be {nil}.
* `items`: List of pronunciations, in exactly the same format as for {format_IPA_multiple()}.
* `err`: If not {nil}, a string containing an error message to use in place of the link to the language's phonology.
* `separator`: The default separator to use when separating formatted items. Defaults to {", "}. Does not apply to the
first item, where the default separator is always the empty string. Overridden by the per-item `separator` field in
`items`.
* `sort_key`: Explicit sort key used for categories.
* `no_count`: Suppress adding a {#-syllable words} category such as [[:Category:Italian 2-syllable words]]. Note that
only certain languages add such categories to begin with, because it depends on knowing how to count syllables in a
given language, which depends on the phonology of the language. Also, this does not suppress the addition of cleanup
or other categories. If you need them suppressed, use `split_output` to return the categories separately and ignore
them.
* `split_output`: If not given, the return value is a concatenation of the formatted pronunciation and formatted
categories. Otherwise, two values are returned: the formatted pronunciation and the categories. If `split_output` is
the value {"raw"}, the categories are returned in list form, where the list elements are a combination of category
strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]]. If
`split_output` is any other value besides {nil}, the categories are returned as a pre-formatted concatenated string.
* `include_langname`: If specified, prefix the result with the language name, followed by a colon.
* `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display at the beginning, before the formatted
pronunciations and preceding {"IPA:"}.
* `qq`: {nil} or a list of right qualifiers to display after all formatted pronunciations.
* `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display at the beginning, before the formatted
pronunciations and preceding {"IPA:"}.
* `aa`: {nil} or a list of right accent qualifiers to display after all formatted pronunciations.
]==]
function export.format_IPA_full(data)
if type(data) ~= "table" or data.getCode then
error("Must now supply a table of arguments to format_IPA_full(); first argument should be that table, not a language object")
end
local lang = data.lang
local items = data.items
local err = data.err
local separator = data.separator
local sort_key = data.sort_key
local no_count = data.no_count
local split_output = data.split_output
local q = data.q
local qq = data.qq
local a = data.a
local aa = data.aa
local include_langname = data.include_langname
if data.qualifiers then
-- FIXME: added 2026-09-18; consider removing eventually.
error("overall `.qualifiers` is no longer supported; change the code to use `.q` or `.qq`")
end
local hasKey = m_data.langs_with_infopages
if not lang or not lang.getCode then
error("Must specify language to format_IPA_full()")
end
assert_not_etymology_only_lang(lang)
local langname = lang:getCanonicalName()
local prefix_text
if err then
prefix_text = '<span class="error">' .. err .. '</span>'
else
if hasKey[lang:getCode()] then
prefix_text = "Apendise:" .. langname .. " na pagbigkas" --TLCHANGE
else
prefix_text = "wikipedia:" .. langname .. " na ponolohiya" --TLCHANGE
end
prefix_text = "[[" .. prefix_text .. "|gabay]]" --TLCHANGE "[[" .. prefix_text .. "|key]]"
end
local prefix = "[[Wiktionary:International Phonetic Alphabet|IPA]]<sup>(" .. prefix_text .. ")</sup>: "
local IPAs, categories = export.format_IPA_multiple(lang, items, separator, no_count, "raw")
if is_content_page(lang, namespace) then
insert(categories, {
cat = langname .. " na salitang may pagbigkas na IPA", --TLCHANGE
sort_key = sort_key
})
end
local prontext = prefix .. IPAs
if q and q[1] or qq and qq[1] or a and a[1] or aa and aa[1] then
prontext = require(pron_qualifier_module).format_qualifiers {
lang = lang,
text = prontext,
q = q,
qq = qq,
a = a,
aa = aa,
}
end
if include_langname then
prontext = langname .. ": " .. prontext
end
return process_maybe_split_categories(split_output, categories, prontext, lang)
end
local function split_phonemic_phonetic(pron)
local reconstructed, phonemic, phonetic = match(pron, "^(%*?)(/.-/)%s+(%[.-%])$")
if reconstructed then
return reconstructed .. phonemic, reconstructed .. phonetic
else
return pron, nil
end
end
local function determine_repr(pron)
local reconstructed
-- Temporarily remove any initial asterisk before representation marks,
-- which avoids having to account for it in the data, but set the
-- `reconstructed` flag.
if sub(pron, 1, 1) == "*" then
reconstructed = true
pron = sub(pron, 2)
end
-- Some representation types have aliases for convenience (e.g. "// //" is
-- an alias for "⫽ ⫽"). and these need to be substituted in before checking
-- for other data.
local opening, n = match(pron, "^.[\128-\191]*")
local subs_data = m_data.representation_subs[opening]
if subs_data then
pron, n = ugsub(pron, subs_data[1], subs_data[2])
-- If the substitution was made, `opening` needs to be changed to the
-- new opening character.
if n ~= 0 then
opening = subs_data[3]
end
end
-- Get the type data based on the opening character (if any), and set the
-- representation type if the closing character matches.
local type_data, repr, closing = m_data.representation_types[opening]
if type_data then
closing = type_data[2]
if type_data and match(pron, pattern_escape(closing) .. "$", #opening + 1) then
repr = type_data[1]
end
end
-- Default to the empty string.
if not repr then
opening, closing = "", ""
end
-- Reattach the asterisk if reconstructed.
if reconstructed then
pron = "*" .. pron
end
return pron, repr, opening, closing, reconstructed
end
local function hasInvalidSeparators(transcription)
-- Escape certain characters as well as pauses, which have the format "(...)" (with any number of dots), to avoid false-positives.
transcription = transcription:gsub(".[\128-\191]*", m_symbols.separator_escapes)
:gsub("%(%.+%)", "\3")
:gsub("[()]+", "")
return (
transcription:find("..", nil, true) or
transcription:match("%.%f[%z \1\2\3,:;]") or
transcription:match("\1%f[%z \2\3,:;]") or
transcription:match("\2%f[%z \1\3,:;]") or
transcription:match("\3[:;]") or
transcription:match("%f[^%z \1\2\3,]%.")
) and true or false
end
--[==[
Format a line of one or more bare IPA pronunciations (i.e. without any preceding {"IPA:"} and without adding to a
category ` ``lang`` terms with IPA pronunciation`). Individual pronunciations are formatted using
{format_IPA()} and are combined with separators, decorations, pre-text, post-text, etc. to form a line of pronunciations.
Parameters accepted are:
* `lang` is an object representing the language of the pronunciations, which is used when adding cleanup categories for
pronunciations with invalid phonemes; for determining how many syllables the pronunciations have in them, in order to
add a category such as [[:Category:Italian 2-syllable words]] (for certain languages only); and for computing the
proper sort keys for categories. `lang` may be {nil}.
* `items` is a list of pronunciations, each of which is an object with the following properties:
** `pron`: the pronunciation, in the same format as is accepted by {format_IPA()}, i.e. it should be either phonemic
(surrounded by {/.../}), phonetic (surrounded by {[...]}), orthographic (surrounded by {⟨...⟩}) or a rhyme
(beginning with a hyphen);
** `pretext`: text to display directly before the formatted pronunciation, inside of any qualifiers or accent
qualifiers;
** `posttext`: text to display directly after the formatted pronunciation, inside of any qualifiers or accent
qualifiers;
** `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display before the formatted pronunciation;
** `qq`: {nil} or a list of right qualifiers to display after the formatted pronunciation;
** `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display before the formatted pronunciation;
** `aa`: {nil} or a list of right accent qualifiers to after before the formatted pronunciation;
** `refs`: {nil} or a list of references or reference specs to add after the pronunciation and any posttext and
qualifiers; the value of a list item is either a string containing the reference text (typically a call to a
citation template such as {{tl|cite-book}}, or a template wrapping such a call), or an object with fields `text`
(the reference text), `name` (the name of the reference, as in {{cd|<nowiki><ref name="foo">...</ref></nowiki>}}
or {{cd|<nowiki><ref name="foo" /></nowiki>}}) and/or `group` (the group of the reference, as in
{{cd|<nowiki><ref name="foo" group="bar">...</ref></nowiki>}} or
{{cd|<nowiki><ref name="foo" group="bar"/></nowiki>}}); this uses a parser function to format the reference
appropriately and insert a footnote number that hyperlinks to the actual reference, located in the
{{cd|<nowiki><references /></nowiki>}} section;
** `gloss`: {nil} or a gloss (definition) for this item, if different definitions have different pronunciations;
** `pos`: {nil} or a part of speech for this item, if different parts of speech have different pronunciations;
** `separator`: the separator text to insert directly before the formatted pronunciation and all decorations
and pre-text; defaults to the outer `separator` parameter.
* `separator`: The default separator to use when separating formatted items. Defaults to {", "}. Does not apply to the
first item, where the default separator is always the empty string. Overridden by the per-item `separator` field in
`items`.
* `no_count`: Suppress adding a {#-syllable words} category such as [[:Category:Italian 2-syllable words]]. Note that
only certain languages add such categories to begin with, because it depends on knowing how to count syllables in a
given language, which depends on the phonology of the language. Also, this does not suppress the addition of cleanup
categories. If you need them suppressed, use `split_output` to return the categories separately and ignore them.
* `split_output`: If not given, the return value is a concatenation of the formatted pronunciation and formatted
categories. Otherwise, two values are returned: the formatted pronunciation and the categories. If `split_output` is
the value {"raw"}, the categories are returned in list form, where the list elements are a combination of category
strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]]. If
`split_output` is any other value besides {nil}, the categories are returned as a pre-formatted concatenated string.
]==]
function export.format_IPA_multiple(lang, items, separator, no_count, split_output)
local categories = {}
separator = separator or ", "
if not lang then
track("format-multiple-nolang")
else
assert_not_etymology_only_lang(lang)
end
-- Format
if not items[1] then
if namespace == "Padron" then --TLCHANGE "Template"
insert(items, {pron = "/aɪ piː ˈeɪ/"})
else
insert(categories, "Pronunciation templates without a pronunciation")
end
end
local bits = {}
for i, item in ipairs(items) do
local bit
-- If the pronunciation is entirely empty, allow this and don't do anything, so that e.g. the pretext and/or
-- posttext can be specified to force something like ''unknown'' to appear in place of the pronunciation
-- (as happens e.g. when ? is used as a respelling in [[Module:ca-IPA]]; see [[guèiser]] for an example).
if item.pron == "" then
bit = ""
else
local item_categories, errtext
bit, item_categories, errtext = export.format_IPA(lang, item.pron, "raw")
bit = bit .. errtext
for _, cat in ipairs(item_categories) do
insert(categories, cat)
end
end
if item.pretext then
bit = item.pretext .. bit
end
if item.posttext then
bit = bit .. item.posttext
end
if item.qualifiers then
-- FIXME: added 2026-09-18; consider removing eventually.
error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`")
end
local has_decorations = item.q and item.q[1] or item.qq and item.qq[1] or item.a and item.a[1] or
item.aa and item.aa[1] or item.refs and item.refs[1]
local has_gloss_or_pos = item.gloss or item.pos
if has_decorations or has_gloss_or_pos then
-- FIXME: Currently we tack the gloss and POS (in that order) onto the end of the regular left qualifiers.
-- Should we do something different?
local q = item.q
if has_gloss_or_pos then
q = mw.clone(item.q) or {}
if item.gloss then
local m_qualifier = require(qualifier_module)
insert(q, m_qualifier.wrap_qualifier_css("“", "quote") .. item.gloss ..
m_qualifier.wrap_qualifier_css("”", "quote"))
end
if item.pos then
-- FIXME: Consider expanding aliases as found in [[Module:headword/data]] or similar.
insert(q, item.pos)
end
end
bit = require(decorations_module).format_decorations {
lang = lang,
text = bit,
q = q,
qq = item.qq,
a = item.a,
aa = item.aa,
refs = item.refs,
}
end
bit = (item.separator or (i == 1 and "" or separator)) .. bit
insert(bits, bit)
--[=[ [[Special:WhatLinksHere/Wiktionary:Tracking/IPA/syntax-error]]
The length or gemination symbol should not appear after a syllable break or stress symbol. ]=]
-- The nature of the following pattern match is such that we don't have to split a combined '/.../ [...]' spec
-- into its parts in order to process.
if match(item.pron, "[.\203][\136\140]?\203[\144\145]") then -- [.ˈˌ][ːˑ]
track("syntax-error")
end
if lang then
-- Add syllable count if the language's diphthongs are listed in [[Module:syllables]].
-- Don't do this if the term has spaces, a liaison mark (‿) or isn't in mainspace.
if not no_count and namespace == "" then
m_syllables = m_syllables or require(syllables_module)
local langcode = lang:getCode()
if m_data.langs_to_generate_syllable_count_categories[langcode] then
local raw_phonemic, phonetic, use_it = split_phonemic_phonetic(item.pron)
local phonemic, repr = determine_repr(raw_phonemic)
if not phonetic then -- not a '/.../ [...]' combined pronunciation
if m_data.langs_to_use_phonetic_or_phonemic_notation[langcode] then
use_it = phonemic
elseif m_data.langs_to_use_phonetic_notation[langcode] then
use_it = repr == "phonetic" and phonemic or nil
else
use_it = repr == "phonemic" and phonemic or nil
end
elseif repr == "phonetic" then
use_it = phonetic
elseif repr == "phonemic" then
use_it = phonemic
end
-- Note: two uses of find with plain patterns is much faster than umatch with [ ‿].
if use_it and not (find(use_it, " ") or find(use_it, "‿")) then
local syllable_count = m_syllables.getVowels(use_it, lang)
if syllable_count then
insert(categories, lang:getCanonicalName() .. " na salitang may " .. syllable_count ..
" pantig") -- TLCHANGE
end
end
end
end
end
end
return process_maybe_split_categories(split_output, categories, concat(bits), lang)
end
--[=[
Format a single IPA pronunciation, which cannot be a combined spec (such as {/.../ [...]}). This has been extracted from
{format_IPA()} to allow the latter to handle such combined specs. This works like {format_IPA()} but requires that
pre-created {err} (for error messages) and {categories} lists be passed in, and adds any generated error messages and
categories to those lists. A single value is returned, the pronunciation, which is usually the same as passed in, but
may have HTML added surrounding invalid characters so they appear in red.
]=]
local function format_one_IPA(lang, raw_pron, err, categories)
-- Disallow wikilinks.
if match(raw_pron, "%[%[.-%]%]") then
error("IPA input must not contain wikilinks.")
end
raw_pron = decode_entities(raw_pron)
-- Detect the type of transcription.
local pron, repr, opening, closing, reconstructed = determine_repr(raw_pron)
-- Strip any reconstruction asterisk and representation marks.
pron = sub(pron, #opening + 1 + (reconstructed and 1 or 0), -#closing - 1)
if not repr then
insert(categories, "IPA pronunciations with invalid representation marks")
-- insert(err, "invalid representation marks")
-- Removed because it's annoying when previewing pronunciation pages.
end
if repr ~= "orthographic" and lang and lang:getCode() == "en" and hasInvalidSeparators(pron) then
insert(categories, "English IPA pronunciations with invalid separators")
end
if pron == "" then
insert(categories, "IPA pronunciations with no pronunciation present")
end
-- Check for obsolete and nonstandard symbols
for _, symbol in ipairs(m_data.nonstandard) do
local result
for nonstandard in gmatch(pron, symbol) do
if not result then
result = {}
end
insert(result, nonstandard)
insert(categories,
{cat = "IPA pronunciations with obsolete or nonstandard characters", sort_key = nonstandard}
)
end
if result then
insert(err, "obsolete or nonstandard characters (" .. concat(result) .. ")")
break
end
end
--[[ Check for invalid symbols after removing the following:
1. wikilinks (handled above)
2. paired HTML tags
3. bolding
4. italics
5. asterisk at beginning of transcription
6. comma followed by spacing characters
7. superscripts enclosed in superscript parentheses ]]
local found_HTML
local result = gsub(pron, "<(%a+)[^>]*>([^<]+)</%1>",
function(tagName, content)
found_HTML = true
return content
end)
result = gsub(result, "'''([^']*)'''", "%1")
result = gsub(result, "''([^']*)''", "%1")
result = gsub(result, "^%*", "")
result = ugsub(result, ",%s+", "")
-- VS15
local vs15_class = "[" .. m_symbols.add_vs15 .. "]"
if umatch(pron, vs15_class) then
local vs15 = u(0xFE0E)
if find(result, vs15) then
result = gsub(result, vs15, "")
pron = gsub(pron, vs15, "")
end
pron = ugsub(pron, vs15_class, "%0" .. vs15)
end
if result ~= "" then
local content_page = is_content_page(lang, namespace)
if lang then
-- Get the per_lang_valid data, and convert any per-language valid sequences to spaces.
local per_lang_valid = m_symbols.per_lang_valid[lang:getCode()]
if per_lang_valid then
if type(per_lang_valid) == "table" then
for _, pattern in pairs(per_lang_valid) do
result = ugsub(result, pattern, " ")
end
else -- Should be a string.
result = ugsub(result, per_lang_valid, " ")
end
end
end
local suggestions = {}
-- Check for any invalid sequences, excluding anything in the per-language lookup table.
for k, v in pairs(m_symbols.invalid) do
if find(result, k, nil, true) then
insert(suggestions, with_codepoints(k) .. " with " .. with_codepoints(v))
end
end
if suggestions[1] then
local replacements = "replace " .. listToText(suggestions)
if content_page then
error("Invalid IPA: " .. replacements)
end
insert(err, replacements)
end
-- Convert any valid character sequences to spaces
for _, pattern in pairs(m_symbols.valid) do
result = ugsub(result, pattern, " ")
end
if not match(result, "^ *$") then
local category = "IPA pronunciations with invalid IPA characters"
if not content_page then
category = category .. "/non_mainspace"
end
insert(categories, category)
insert(err, "invalid IPA characters: " .. with_codepoints(result))
end
end
if found_HTML then
insert(categories, "IPA pronunciations with paired HTML tags")
end
if (repr == "phonemic" or repr == "rhyme") and lang and m_data.phonemes[lang:getCode()] then
local valid_phonemes = m_data.phonemes[lang:getCode()]
local rest = pron
local phonemes = {}
while #rest > 0 do
local longestmatch, longestmatch_len = "", 0
local rest_init = sub(rest, 1, 1)
if rest_init == "(" or rest_init == ")" then
longestmatch = rest_init
longestmatch_len = 1
else
for _, phoneme in ipairs(valid_phonemes) do
local phoneme_len = len(phoneme)
if phoneme_len > longestmatch_len and usub(rest, 1, phoneme_len) == phoneme then
longestmatch = phoneme
longestmatch_len = len(longestmatch)
end
end
end
if longestmatch_len > 0 then
insert(phonemes, longestmatch)
rest = usub(rest, longestmatch_len + 1)
else
local phoneme = usub(rest, 1, 1)
insert(phonemes, "<span style=\"color: var(--wikt-palette-red,red)\">" .. phoneme .. "</span>")
rest = usub(rest, 2)
insert(categories, "IPA pronunciations with invalid phonemes/" .. lang:getCode())
track("invalid phonemes/" .. phoneme)
end
end
pron = concat(phonemes)
end
return (reconstructed and "*" or "") .. opening .. pron .. closing
end
--[==[
Format an IPA pronunciation. This wraps the pronunciation in appropriate CSS classes and adds cleanup categories and
error messages as needed. The pronunciation `pron` should be either phonemic (surrounded by {/.../}), phonetic
(surrounded by {[...]}), orthographic (surrounded by {⟨...⟩}), a rhyme (beginning with a hyphen) or a combined
phonemic/phonetic spec (of the form {/.../ [...]}). `lang` indicates the language of the pronunciation and can be {nil}.
If not {nil}, and the specified language has data in [[Module:IPA/data]] indicating the allowed phonemes, then the page
will be added to a cleanup category and an error message displayed next to the outputted pronunciation. Note that {lang}
also determines sort key processing in the added cleanup categories. If `split_output` is not given, the return value is
a concatenation of the formatted pronunciation, error messages and formatted cleanup categories. Otherwise, three values
are returned: the formatted pronunciation, the cleanup categories and the concatenated error messages. If `split_output`
is the value {"raw"}, the cleanup categories are returned in list form, where the list elements are a combination of
category strings and category objects of the form suitable for passing to {format_categories()} in [[Module:utilities]].
If `split_output` is any other value besides {nil}, the cleanup categories are returned as a pre-formatted concatenated
string.
]==]
function export.format_IPA(lang, pron, split_output)
local err = {}
local categories = {}
-- `pron` shouldn't contain ref tags.
if match(pron, "\127'\"`UNIQ%-%-ref%-[%dA-F]+%-QINU`\"'\127") then
error("<ref> tags found inside pronunciation parameter.")
end
if not lang then
track("format-nolang")
else
assert_not_etymology_only_lang(lang)
end
local phonemic, phonetic = split_phonemic_phonetic(pron)
pron = format_one_IPA(lang, phonemic, err, categories)
if phonetic then
track("phonemic-phonetic") -- There's no benefit to supporting the "/.../ [...]" format within one parameter.
phonetic = format_one_IPA(lang, phonetic, err, categories)
pron = pron .. " " .. phonetic
end
if err[1] and is_preview() then
err = '<span class="error" style="font-size: small;> ' .. concat(err, ", ") .. "</span>"
else
err = ""
end
return process_maybe_split_categories(split_output, categories, '<span class="IPA nowrap">' .. pron .. "</span>", lang,
err)
end
--[==[
Format a line of one or more enPR pronunciations as {{tl|enPR}} would do it, i.e. with a preceding {"enPR:"} (linked to
[[Appendix:English pronunciation]]) followed by one or more formatted, comma-separated enPR pronunciations. The
pronunciations are formatted by wrapping them in the `AHD` and `enPR` CSS classes and adding any decorations
(qualifiers, accent qualifiers and references). In addition, the overall result is wrapped in any overall decorations.
There is a single parameter `data`, an object with the following fields:
* `items` is a list of enPR pronunciations, each of which is an object with the following properties:
** `pron`: the enPR pronunciation;
** `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display before the formatted pronunciation;
** `qq`: {nil} or a list of right qualifiers to display after the formatted pronunciation;
** `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display before the formatted pronunciation;
** `aa`: {nil} or a list of right accent qualifiers to after before the formatted pronunciation.
* `q`: {nil} or a list of left qualifiers (as in {{tl|q}}) to display at the beginning, before the formatted
pronunciations and preceding {"enPR:"}.
* `qq`: {nil} or a list of right qualifiers to display after all formatted pronunciations.
* `a`: {nil} or a list of left accent qualifiers (as in {{tl|a}}) to display at the beginning, before the formatted
pronunciations and preceding {"enPR:"}.
* `aa`: {nil} or a list of right accent qualifiers to display after all formatted pronunciations.
]==]
function export.format_enPR_full(data)
local prefix = "[[Appendix:English pronunciation|enPR]]: "
local lang = require("Module:languages").getByCode("en")
local parts = {}
for _, item in ipairs(data.items) do
local part = '<span class="AHD enPR">' .. item.pron .. "</span>"
if item.qualifiers then
-- FIXME: added 2026-09-18; consider removing eventually.
error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`")
end
if item.q and item.q[1] or item.qq and item.qq[1] or item.a and item.a[1] or item.aa and item.aa[1] then
part = require(decorations_module).format_decorations {
lang = lang,
text = part,
q = item.q,
qq = item.qq,
a = item.a,
aa = item.aa,
}
end
insert(parts, part)
end
local prontext = prefix .. concat(parts, ", ")
if data.qualifiers then
-- FIXME: added 2026-09-18; consider removing eventually.
error("overall `.qualifiers` is no longer supported; change the code to use `.q` or `.qq`")
end
if data.q and data.q[1] or data.qq and data.qq[1] or data.a and data.a[1] or data.aa and data.aa[1] then
prontext = require(decorations_module).format_decorations {
lang = lang,
text = prontext,
q = data.q,
qq = data.qq,
a = data.a,
aa = data.aa,
}
end
return prontext
end
return export
g9jloaqepe6sw5c8g6tddxskfppnui7
buenos días
0
31649
178039
157330
2026-09-23T05:05:46Z
Yivan000
4078
178039
wikitext
text/x-wiki
=={{=es=}}==
===Pagbigkas===
{{es-pr|+<audio:Es-buenos días.ogg><audio:Es-buenos días.oga>}}
===Pandamdam===
{{head|es|interjection|head=[[bueno]]s [[día]]s}}
# [[magandang araw]], [[magandang umaga]]
#: {{syn|es|buen día}}
#: {{cot|es|buenas tardes|buenas noches}}
ongixx3z19mdj37kido1uw4upf895u4
Module:decorations
828
33383
178043
177965
2026-09-23T05:15:19Z
Yivan000
4078
enwikt parity
178043
Scribunto
text/plain
local export = {}
local gender_and_number_module = "Module:gender and number"
local labels_module = "Module:labels"
local qualifier_module = "Module:qualifier"
local references_module = "Module:references"
local function track(page)
require("Module:debug/track")("decorations/" .. page)
return true
end
--[==[
This function is used by any module that wants to add support for adding decorations (i.e. left and right regular and
accent qualifiers, labels and references, and potentially other types of decorations in the future) to an item, where
an "item" is any string of text that might want to be decorated. It is currently used for:
# pronunciations and other pronunciation-related items, such as rhymes, hyphenations and homophones, as implemented in
[[Module:IPA]], [[Module:rhymes]], [[Module:hyphenation]], [[Module:homophones]] and various language-specific modules
such as [[Module:es-pronunc]];
# arbitrary links, as implemented in [[Module:links]];
# headwords, as implemented in [[Module:headword]];
# gender/number specs, as implemented in [[Module:gender and number]];
# *nyms of all sorts (e.g. synonyms, antonyms, hypernyms, hyponyms, coordinate terms, etc.) as implemented in
[[Module:nyms]];
# affixes of all sorts, as implemented in [[Module:affix]];
# affix usage examples, as implemented in [[Module:affixusex]];
# column templates as such as {{tl|col}} and {{tl|col3}}, as implemented in [[Module:columns]];
# names (surnames, given names, patronymics, etc.) as implemented in [[Module:names]];
# object-usage qualifiers such as {{tl|+obj}}, as implemented in [[Module:object usage]];
# various inflection templates such as {{tl|ar-conj}} in [[Module:ar-verb]];
# demonyms, as implemented in [[Module:demonym]];
and several other places.
Modules that implement templates that occur frequently on a page, such as [[Module:headword]] and [[Module:links]],
should consider checking that any decorations exist before loading the module. This is less necessary for other,
less-used templates, as this module is not very heavy and does appropriate checks itself to make sure that any
decorations are present before taking action.
`data` is a structure containing the following fields:
* `q`: Optional list of left regular qualifiers, each a string.
* `qq`: Optional list of right regular qualifiers, each a string.
* `a`: Optional list of left accent qualifiers, each a string.
* `aa`: Optional list of right accent qualifiers, each a string.
* `l`: Optional list of left labels, each a string.
* `ll`: Optional list of right labels, each a string.
* `refs`: Optional list of references or reference specs to add directly after the text; the value of a list item is
either a string containing the reference text (typically a call to a citation template such as {{tl|cite-book}}, or a
template wrapping such a call), or an object with fields `text` (the reference text), `name` (the name of the
reference, as in `<nowiki><ref name="foo">...</ref></nowiki>` or `<nowiki><ref name="foo" /></nowiki>`) and/or `group`
(the group of the reference, as in `<nowiki><ref name="foo" group="bar">...</ref></nowiki>` or
`<nowiki><ref name="foo" group="bar"/></nowiki>`); this uses a parser function to format the reference appropriately
and insert a footnote number that hyperlinks to the actual reference, located in the `<nowiki><references /></nowiki>`
section.
* `genders`: Optional list of gender/number specs as accepted by [[Module:gender and number]].
* `lang`: Language object for accent qualifiers.
* `text`: The text to wrap with qualifiers.
*` raw`: Don't do any CSS wrapping of the formatted text.
The order of qualifiers and labels, on both the left and right, is (1) labels, (2) accent qualifiers, (3) regular
qualifiers. This goes in order of relative importance. References and genders go on the right, inside of labels and
qualifiers, with references before genders.
]==]
function export.format_decorations(data)
if not data.text then
error("Missing `data.text`; did you try to pass `text` as a separate param?")
end
if not data.lang then
track("nolang")
end
local text = data.text
-- Format the qualifiers and labels that go either before or after the main text. They are ordered as follows, on
-- both the left and the right: (1) labels, (2) accent qualifiers, (3) regular qualifiers. This puts the different
-- types of qualifiers/labels in order of relative importance. Return nil if no qualifiers or labels, otherwise
-- a string containing all formatted qualifiers and labels surrounded by parens.
local function format_qualifier_like(labels, accent_qualifiers, qualifiers)
local has_qualifiers = qualifiers and qualifiers[1]
local has_accent_qualifiers = accent_qualifiers and accent_qualifiers[1]
local has_labels = labels and labels[1]
if not has_qualifiers and not has_accent_qualifiers and not has_labels then
return nil
end
local qualifier_like_parts = {}
local function ins(part)
table.insert(qualifier_like_parts, part)
end
local function format_label_like(labels, mode)
return require(labels_module).show_labels {
lang = data.lang,
labels = labels,
nocat = true,
mode = mode,
open = false,
close = false,
no_ib_content = true,
no_track_already_seen = true,
ok_to_destructively_modify = true, -- doesn't apply to `labels`
raw = data.raw,
}
end
local m_qualifier = require(qualifier_module)
if has_labels then
ins(format_label_like(labels))
end
if has_accent_qualifiers then
ins(format_label_like(accent_qualifiers, "accent"))
end
if has_qualifiers then
ins(m_qualifier.format_qualifiers {
qualifiers = qualifiers,
open = false,
close = false,
no_ib_content = true,
raw = data.raw,
})
end
local qualifier_inside
local function wrap_qualifier_css(txt, suffix)
if data.raw then
return txt
else
return m_qualifier.wrap_qualifier_css(txt, suffix)
end
end
if qualifier_like_parts[2] then
qualifier_inside = table.concat(qualifier_like_parts, wrap_qualifier_css(",", "comma") .. " ")
else
qualifier_inside = qualifier_like_parts[1]
end
qualifier_like_parts = {}
ins(wrap_qualifier_css("(", "brac"))
ins(wrap_qualifier_css(qualifier_inside, "content"))
ins(wrap_qualifier_css(")", "brac"))
return table.concat(qualifier_like_parts)
end
if data.refs then
text = text .. require(references_module).format_references(data.refs)
end
if data.genders and data.genders[1] then
-- NOTE, format_genders() returns a second value (categories) but we ignore it.
text = text .. " " .. require(gender_and_number_module).format_genders(data.genders, data.lang)
end
if data.qualifiers then
-- FIXME: added 2026-09-18; consider removing eventually.
error("`.qualifiers` is no longer supported; change the code to use `.q` or `.qq`")
end
local leftq = format_qualifier_like(data.l, data.a, data.q)
local rightq = format_qualifier_like(data.ll, data.aa, data.qq)
if leftq then
text = leftq .. " " .. text
end
if rightq then
text = text .. " " .. rightq
end
return text
end
function export.format_qualifiers(...)
-- FIXME: Added 2026-09-17. Remove after a month or less.
error("use format_decorations instead")
end
return export
gvyopr8wgnkfw3arhfcjf23ibitjoha
Module:IPA/data
828
33750
178046
169474
2026-09-23T05:19:56Z
Yivan000
4078
enwikt parity
178046
Scribunto
text/plain
local list_to_set = require("Module:table").listToSet
local data = {}
--[=[
A list of representation types (e.g. /foo/ for phonemic and [bar] for phonetic),
given as a table. The key is the opening character, the first value the
representation type, and the second value the closing symbol.]=]
data.representation_types = {
["/"] = {"phonemic", "/"},
["["] = {"phonetic", "]"},
["⫽"] = {"morphophonemic", "⫽"},
["⟨"] = {"orthographic", "⟩"},
["-"] = {"rhyme", ""},
}
--[=[
A list of convenience inputs for certain representation types. The key is the
opening character, and the table is a three-item array consisting of (1) an
mw.ustring.gsub pattern which is anchored to the start and end of the string,
with a single capture group that excludes the characters to be substituted,
(2) a corresponding replacement pattern to be used with the pattern, and (3) the
replacement opening character.]=]
data.representation_subs = {
["<"] = {"^<(.*)>$", "⟨%1⟩", "⟨"},
["/"] = {"^//(.*)//$", "⫽%1⫽", "⫽"},
}
--[=[
This should list the language codes of all languages that have a pronunciation
page in the appendix of the form ''Appendix:LANG pronunciation'', e.g.
[[Appendix:Russian pronunciation]]. For these languages, the text "key" next to
the generated pronunciation links to such pages; for other languages, it links
to the "LANG phonology" page in Wikipedia (which may or may not exist).
[[Module:IPA]] is responsible for this linking; see format_IPA_full().]=]
data.langs_with_infopages = list_to_set{
"acw",
"ady",
"ang",
"arc",
"ba",
"bg",
"bo",
"ca",
"cho",
"cmn",
"cs",
"cv",
"cy",
"da",
"de",
"dsb",
"dz",
"egl",
"egy",
"el",
"en",
"enm",
"eo",
"es",
"fa",
"fi",
"fo",
"fr",
"fy",
"ga",
"gd",
"ghc",
"gmh",
"gmw-msc",
"got",
"he",
"hi",
"hrx",
"hu",
"hy",
"id",
"ii",
"is",
"it",
"iu",
"ja",
"jbo",
"ka",
"kls",
"ko",
"kw",
"la",
"lb",
"liv",
"lt",
"lv",
"mdf",
"mfe",
"mic",
"mk",
"mns-nor",
"ms",
"mt",
"mul",
"my",
"nan",
"nci",
"nl",
"nn",
"no",
"nov",
"nv",
"pjt",
"pl",
"ps",
"pt",
"ro",
"ru",
"scn",
"sco",
"sga",
"sh",
"sl",
"sq",
"sv",
"sw",
"syc",
"szl",
"tg",
"th",
"tl",
"tpw",
"tr",
"tyv",
"ug",
"uk",
"vi",
"vo",
"wlm",
"yi",
"yrl",
"yue",
"zlw-mas"
}
--[=[
This should list the diphthongs of a language (in the form of Lua patterns),
provided they do *NOT* contain semivowel symbols such as /j w ɰ ɥ/ or vowels
with nonsyllabic diacritics such as /i̯ u̯/. For example, list /au/ or /aʊ/,
but do not list /aw/ or /au̯/. The data in this table is used to count the
number of syllables in a word. [[Module:syllables]] automatically knows how
to correctly handle semivowel symbols and nonsyllabic diacritics.
Any language listed here will automatically have categories of the form
"LANG #-syllable words" generated. In addition, any language listed below under
`langs_to_generate_syllable_count_categories` will also have such categories
generated.
NOTE: There are some additional languages that have these categories.
For example:
* Thai words have these categories added by [[Module:th-pron]].]=]
data.diphthongs = {
["cs"] = { -- [[w:Czech phonology#Diphthongs]]
"[aeo]u",
},
["de"] = {
"a[ɪʊ]",
"ɔ[ʏɪ]",
},
["en"] = { -- from [[Appendix:English pronunciation]] mostly, but /ʌɪ/ is from the OED
"[aɑæeɛoɔʌ][ɪi]",
"[ɑɒæo]e",
"[əɐ]ʉ",
"[aɒəoɔæ]ʊ",
"æo",
"[ɛeɪiɔʊʉ]ə", -- /iə/ is a diphthong in NZE, but a disyllabic sequence in GA.
-- /ɪə/ is both a disyllabic sequence and a diphthong in old-fashioned RP.
"[aʌ][ʊɪ]ə", -- May be a disyllabic sequence in some or all dialects?
},
["grc"] = {
"[aeyo]i",
"[ae]u",
"[ɛɔa]ː[iu]",
},
["hrx"] = {
"aɪ̯",
"aʊ̯",
"oɪ̯",
"eʊ̯",
},
["is"] = { -- [[w:Icelandic phonology#Vowels]]
"[aeɔœʏ]i", -- diphthongs as the module generates them
"[ao]u", -- diphthongs as the module generates them
"ø[iɪy]", -- additional forms that may occur; Wikipedia is oddly specific about the second element: ei and ai, but øɪ.
},
["it"] = {
"[aeɛoɔu]i",
"[aeɛioɔ]u",
},
["lb"] = {
"[iu]ə",
"[ɜoæɑ]ɪ",
"[əæɑ]ʊ",
},
["lt"] = {
"ɐɪ", "ɒʊ", "ɛɪ", "ɛʊ", "ʊɪ", "ɔɪ", "ɔʊ", -- Simple diphthongs (unstressed forms)
"iɛ", "uɔ", -- Complex diphthongs
"ɑˑɪ", "ɑˑʊ", "æˑɪ", "æˑʊ", "oˑɪ", -- Falling tone (acute)
"ɐɪˑ", "ɒʊˑ", "ɛɪˑ", "ɛʊˑ", "ʊɪˑ", -- Rising tone (tilde) - lengthened second element
-- Note: Mixed diphthongs (e.g., ɐlˑ, æˑn, ʊl, etc.) are omitted since they are inherently monosyllabic
},
}
--[=[
This should list any languages for which categories of the form
"LANG #-syllable words", e.g. [[:Category:Russian 3-syllable words]], should be
generated. Do not list languages here if they have an entry above under
`data.diphthongs`; such languages are automatically added to this list.]=]
local langs_to_generate_syllable_count_categories = list_to_set{
"ar", -- Arabic has diphthongs, but they are transcribed
-- with semivowel symbols.
"ary", -- Moroccan Arabic has diphthongs, but they are transcribed
-- with semivowel symbols.
"bg", -- Bulgarian has diphthongs with /j/ and marginally with /w/,
-- but these are semivowels.
"ca", -- Catalan has diphthongs, but they are generally transcribed using
-- /w/ and /j/, so do not need to be listed (see [[w:Catalan language#Diphthongs and triphthongs]].
"eo",
"es", -- Spanish has diphthongs, but they are transcribed with i̯ etc.
"eu", -- Basque has dipthongs, but they are transcribed with i̯ and u̯.
"fi", -- Finnish has diphthongs, but they are now automatically transcribed with
-- the nonsyllabic diacritic
"fr", -- French has diphthongs, but they are transcribed
-- with semivowel symbols: [[w:French phonology#Glides and diphthongs]].
"hnn",
"id", -- Indonesian has diphthongs, but they are transcribed with i̯ or /j/ etc.
"ka",
"kne",
"kmr",
"ku",
"la", -- All diphthongs transcribed with e̯ or /j/ etc.
"mk",
"ms", -- Malay has diphthongs, but they are transcribed with i̯ or /j/ etc.
"mt", -- Maltese has diphthongs, but they are transcribed
-- with semivowel symbols.
"pl", -- No diphthongs, properly speaking; sequences of a vowel and /w/ or /j/ though.
"pt", -- Portuguese has diphthongs, but they are transcribed with i̯ or /j/ etc.
"rsk", -- No diphthongs but there are sequences of vowel and /j/ or /w/.
"ru", -- No diphthongs, properly speaking; sequences of a vowel and /j/ though.
"sk", -- Slovak has rising diphthongs, /i̯e, i̯a, i̯u, u̯o/, which are probably always spelled with the nonsyllabic diacritic, so do not need to be listed.
"sl", -- No diphthongs, properly speaking; sequences of a vowel, /j/ and /w/ though
"sq", -- [[w:Albanian language#Vowels]] doesn't mention anything about diphthongs.
"szy", -- All diphthongs are transcribed with /j/ or /w/
"tl", -- Tagalog has diphthongs, but they are transcribed with i̯ or /j/ etc
"tsg",
"ug", -- No diphthongs.
}
-- Also add languages listed under `data.diphthongs`.
for langcode, _ in pairs(data.diphthongs) do
langs_to_generate_syllable_count_categories[langcode] = true
end
data.langs_to_generate_syllable_count_categories = langs_to_generate_syllable_count_categories
-- Languages to use the phonetic not phonemic notation to compute syllable counts.
data.langs_to_use_phonetic_notation = list_to_set{
"bg",
"es",
"id",
"la",
"lt",
"mk",
"ms",
"rsk",
"ru",
}
-- Languages to use the phonetic or phonemic notation to compute syllable counts, whichever is available.
data.langs_to_use_phonetic_or_phonemic_notation = list_to_set{
-- [[Module:is-IPA]] generates [...] but many manual pronuns use /.../.
"is",
}
-- Non-standard or obsolete IPA symbols.
data.nonstandard = {
--[[ The following symbols consist of more than one character,
so we can't put them in the line below. ]]
"ɑ̢", "ɔ̗", "ɔ̖",
"[?ƍσƺƪƞƛłščžǰǧǯẋⱻʚω∅ØȣᴀᴇⱻQKPT]"
}
-- See valid IPA characters at [[Module:IPA/data/symbols]].
data.phonemes = {}
data.phonemes["dz"] = {
"m", "n", "ŋ",
"p", "t", "ʈ", "k",
"pʰ", "tʰ", "ʈʰ", "kʰ",
"t͡s", "t͡ɕ",
"t͡sʰ", "t͡ɕʰ",
"w", "s", "z", "ɬ", "l", "r", "ɕ", "ʑ", "j", "h",
"ɑ", "e", "i", "o", "u",
"ɑː", "eː", "ɛː", "iː", "oː", "øː", "uː", "yː",
"ɑ˥", "e˥", "i˥", "o˥", "u˥",
"ɑː˥", "eː˥", "ɛː˥", "iː˥", "oː˥", "øː˥", "uː˥", "yː˥",
"m˥", "n˥", "ŋ˥", "p˥", "k˥", "k̚˥", "w˥", "l˥", "r˥", "ɕ˥", "j˥", ")˥",
"ɑ˩", "e˩", "i˩", "o˩", "u˩",
"ɑː˩", "eː˩", "ɛː˩", "iː˩", "oː˩", "øː˩", "uː˩", "yː˩",
"m˩", "n˩", "ŋ˩", "p˩", "k˩", "k̚˩", "w˩", "l˩", "r˩", "ɕ˩", "j˩", ")˩",
".", ",", "-",
}
data.phonemes["eo"] = {
"a", "b", "d", "d͡ʒ", "d͡z", "e", "f", "h", "i", "j", "k",
"l", "m", "n", "o", "p", "r", "s", "t", "t͡s", "t͡ʃ",
"u", "u̯", "v", "w", "x", "z", "ɡ", "ʃ", "ʒ",
"ˈ", ".", " ", "-", "u̯", "i̯"
}
data.phonemes["hy"] = {
"ɑ", "b", "ɡ", "d", "e", "z", "ə", "tʰ", "ʒ", "i", "l", "χ", "t͡s",
"k", "h", "d͡z", "ʁ", "t͡ʃ", "m", "j", "n", "ʃ", "ɔ", "t͡ʃʰ", "p", "d͡ʒ",
"r", "s", "v", "t", "ɾ", "t͡sʰ", "v", "pʰ", "kʰ", "o", "f", "ŋɡ", "ŋk",
"ŋχ", "u", "œ", "ʏ", "ˈ", "ˌ", ".", " ", "ː",
}
data.phonemes["nl"] = {
"m", "n", "ŋ",
"p", "b", "t", "d", "k", "ɡ",
"f", "v", "s", "z", "ʃ", "ʒ", "x", "ɣ", "ɦ",
"ʋ", "l", "j", "r",
"ɪ", "ʏ", "ɛ", "ə", "ɔ", "ɑ",
"i", "iː", "y", "yː", "u", "uː", "eː", "øː", "oː", "ɛː", "œː", "ɔː", "aː",
"ɛi̯", "œy̯", "ɔi̯", "ɑu̯", "ɑi̯",
"iu̯", "yu̯", "ui̯", "eːu̯", "oːi̯", "aːi̯",
"ˈ", "ˌ", ".", " ", "-",
}
data.phonemes["mt"] = {
"m", "n",
"p", "t", "k", "ʔ",
"b", "d", "ɡ",
"t͡s", "t͡ʃ",
"d͡z", "d͡ʒ",
"f", "s", "ʃ", "ħ",
"v", "z", "ʒ", "ɣ",
"l", "j", "w",
"r",
"ɪ", "ɛ", "ɔ", "a", "u",
"ɛˤ", "ɔˤ", "aˤ", "əˤ",
"ɛˤː", "ɔˤː", "aˤː", "əˤː", "ɪˤː",
"iː", "ɪː", "ɛː", "ɔː", "aː", "uː",
"ˈ", "ˌ", ".", " ", "‿", "-"
}
return data
9j3yfr5dzmr70htnh6pfy97jzb11aph
Module:es-common
828
34964
178045
168430
2026-09-23T05:18:08Z
Yivan000
4078
enwikt parity
178045
Scribunto
text/plain
local export = {}
local romut_module = "Module:romance utilities"
local u = require("Module:string/char")
local rsplit = mw.text.split
local rfind = mw.ustring.find
local rmatch = mw.ustring.match
local rsubn = mw.ustring.gsub
local toNFD = mw.ustring.toNFD
local TILDE = u(0x0303) -- tilde = ̃
local DIA = u(0x0308) -- diaeresis = ̈
local CEDILLA = u(0x0327) -- cedilla = ̧
local TEMPC1 = u(0xFFF1)
local TEMPC2 = u(0xFFF2)
local TEMPV1 = u(0xFFF3)
local DIV = u(0xFFF4)
local vowel = "aeiouáéíóúý" .. TEMPV1
local V = "[" .. vowel .. "]"
local AV = "[áéíóúý]" -- accented vowel
local W = "[iyuw]" -- glide
local C = "[^" .. vowel .. ".]"
export.vowel = vowel
export.V = V
export.AV = AV
export.W = W
export.C = C
local remove_accent = {
["á"] = "a", ["é"] = "e", ["í"] = "i", ["ó"] = "o", ["ú"] = "u", ["ý"] = "y"
}
local add_accent = {
["a"] = "á", ["e"] = "é", ["i"] = "í", ["o"] = "ó", ["u"] = "ú", ["y"] = "ý"
}
export.remove_accent = remove_accent
export.add_accent = add_accent
local prepositions = {
"al? ",
"del? ",
"como ",
"con ",
"en ",
"para ",
"por ",
}
-- version of rsubn() that discards all but the first return value
local function rsub(term, foo, bar)
local retval = rsubn(term, foo, bar)
return retval
end
export.rsub = rsub
-- apply rsub() repeatedly until no change
local function rsub_repeatedly(term, foo, bar)
while true do
local new_term = rsub(term, foo, bar)
if new_term == term then
return term
end
term = new_term
end
end
export.rsub_repeatedly = rsub_repeatedly
function export.decompose(text)
-- decompose everything but ç, ñ and ü
text = toNFD(text)
text = rsub(text, ".[" .. TILDE .. DIA .. CEDILLA .. "]", {
["c" .. CEDILLA] = "ç",
["C" .. CEDILLA] = "Ç",
["n" .. TILDE] = "ñ",
["N" .. TILDE] = "Ñ",
["u" .. DIA] = "ü",
["U" .. DIA] = "Ü",
})
return text
end
-- Apply vowel alternation to stem.
function export.apply_vowel_alternation(stem, alternation)
local ret, err
-- Treat final -gu, -qu as a consonant, so the previous vowel can alternate (e.g. conseguir -> consigo).
-- This means a verb in -guar can't have a u-ú alternation but I don't think there are any verbs like that.
stem = rsub(stem, "([gq])u$", "%1" .. TEMPC1)
local before_last_vowel, last_vowel, after_last_vowel = rmatch(stem, "^(.*)(" .. V .. ")(.-)$")
if alternation == "ie" then
if last_vowel == "e" or last_vowel == "i" then
-- allow i for adquirir -> adquiero, inquirir -> inquiero, etc.
ret = before_last_vowel .. "ie" .. after_last_vowel
else
err = "should have -e- or -i- as the last vowel"
end
elseif alternation == "ye" then
if last_vowel == "e" then
ret = before_last_vowel .. "ye" .. after_last_vowel
else
err = "should have -e- as the last vowel"
end
elseif alternation == "ue" then
if last_vowel == "o" or last_vowel == "u" then
-- allow u for jugar -> juego; correctly handle avergonzar -> avergüenzo
ret = (
last_vowel == "o" and before_last_vowel:find("g$") and before_last_vowel .. "üe" .. after_last_vowel or
before_last_vowel .. "ue" .. after_last_vowel
)
else
err = "should have -o- or -u- as the last vowel"
end
elseif alternation == "hue" then
if last_vowel == "o" then
ret = before_last_vowel .. "hue" .. after_last_vowel
else
err = "should have -o- as the last vowel"
end
elseif alternation == "i" then
if last_vowel == "e" then
ret = before_last_vowel .. "i" .. after_last_vowel
else
err = "should have -i- as the last vowel"
end
elseif alternation == "í" then
if last_vowel == "e" or last_vowel == "i" then
-- allow e for reír -> río, sonreír -> sonrío
ret = before_last_vowel .. "í" .. after_last_vowel
else
err = "should have -e- or -i- as the last vowel"
end
elseif alternation == "ú" then
if last_vowel == "u" then
ret = before_last_vowel .. "ú" .. after_last_vowel
else
err = "should have -u- as the last vowel"
end
else
error("Unrecognized vowel alternation '" .. alternation .. "'")
end
ret = ret and ret:gsub(TEMPC1, "u") or nil
return {ret = ret, err = err}
end
-- Syllabify a word. This implements the full syllabification algorithm, based on the corresponding code
-- in [[Module:es-pronunc]]. This is more than is needed for the purpose of this module, which doesn't
-- care so much about syllable boundaries, but won't hurt.
function export.syllabify(word)
word = DIV .. word .. DIV
-- gu/qu + front vowel; make sure we treat the u as a consonant; a following
-- i should not be treated as a consonant ([[alguien]] would become ''álguienes''
-- if pluralized)
word = rsub(word, "([gq])u([eiéí])", "%1" .. TEMPC2 .. "%2")
local vowel_to_glide = { ["i"] = TEMPC1, ["u"] = TEMPC2 }
-- i and u between vowels should behave like consonants ([[paranoia]], [[baiano]], [[abreuense]],
-- [[alauita]], [[Malaui]], etc.)
word = rsub_repeatedly(word, "(" .. V .. ")([iu])(" .. V .. ")",
function(v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end
)
-- y between consonants or after a consonant at the end of the word should behave like a vowel
-- ([[ankylosaurio]], [[cryptomeria]], [[brandy]], [[cherry]], etc.)
word = rsub_repeatedly(word, "(" .. C .. ")y(" .. C .. ")",
function(c1, c2) return c1 .. TEMPV1 .. c2 end
)
word = rsub_repeatedly(word, "(" .. V .. ")(" .. C .. W .. "?" .. V .. ")", "%1.%2")
word = rsub_repeatedly(word, "(" .. V .. C .. ")(" .. C .. V .. ")", "%1.%2")
word = rsub_repeatedly(word, "(" .. V .. C .. "+)(" .. C .. C .. V .. ")", "%1.%2")
word = rsub(word, "([pbcktdg])%.([lr])", ".%1%2")
word = rsub_repeatedly(word, "(" .. C .. ")%.s(" .. C .. ")", "%1s.%2")
-- Any aeo, or stressed iu, should be syllabically divided from a following aeo or stressed iu.
word = rsub_repeatedly(word, "([aeoáéíóúý])([aeoáéíóúý])", "%1.%2")
word = rsub_repeatedly(word, "([ií])([ií])", "%1.%2")
word = rsub_repeatedly(word, "([uú])([uú])", "%1.%2")
word = rsub(word, "([" .. DIV .. TEMPC1 .. TEMPC2 .. TEMPV1 .. "])", {
[DIV] = "",
[TEMPC1] = "i",
[TEMPC2] = "u",
[TEMPV1] = "y",
})
return rsplit(word, "%.")
end
-- Return the index of the (last) stressed syllable.
function export.stressed_syllable(syllables)
-- If a syllable is stressed, return it.
for i = #syllables, 1, -1 do
if rfind(syllables[i], AV) then
return i
end
end
-- Monosyllabic words are stressed on that syllable.
if #syllables == 1 then
return 1
end
local i = #syllables
-- Unaccented words ending in a vowel or a vowel + s/n are stressed on the preceding syllable.
if rfind(syllables[i], V .. "[sn]?$") then
return i - 1
end
-- Remaining words are stressed on the last syllable.
return i
end
-- Add an accent to the appropriate vowel in a syllable, if not already accented.
function export.add_accent_to_syllable(syllable)
-- Don't do anything if syllable already stressed.
if rfind(syllable, AV) then
return syllable
end
-- Prefer to accent an a/e/o in case of a diphthong or triphthong (the first one if for some reason
-- there are multiple, which should not occur with the standard syllabification algorithm);
-- otherwise, do the last i or u in case of a diphthong ui or iu.
if rfind(syllable, "[aeo]") then
return rsub(syllable, "^(.-)([aeo])", function(prev, v) return prev .. add_accent[v] end)
end
return rsub(syllable, "^(.*)([iu])", function(prev, v) return prev .. add_accent[v] end)
end
-- Remove any accent from a syllable.
function export.remove_accent_from_syllable(syllable)
return rsub(syllable, AV, remove_accent)
end
-- Return true if an accent is needed on syllable number `sylno` if that syllable were to receive the stress,
-- given the syllables of a word. The current accent may be on any syllable.
function export.accent_needed(syllables, sylno)
-- Diphthongs iu and ui are normally stressed on the second vowel, so if the accent is on the first vowel,
-- it's needed.
if rfind(syllables[sylno], "íu") or rfind(syllables[sylno], "úi") then
return true
end
-- If the default-stressed syllable is different from `sylno`, accent is needed.
local unaccented_syllables = {}
for _, syl in ipairs(syllables) do
table.insert(unaccented_syllables, export.remove_accent_from_syllable(syl))
end
local would_be_stressed_syl = export.stressed_syllable(unaccented_syllables)
if would_be_stressed_syl ~= sylno then
return true
end
-- At this point, we know that the stress would by default go on `sylno`, given the syllabification in
-- `syllables`. Now we have to check for situations where removing the accent mark would result in a
-- different syllabification. For example, países -> `pa.i.ses` but removing the accent mark would lead
-- to `pai.ses`. Similarly, río -> `ri.o` but removing the accent mark would lead to single-syllable `rio`.
-- We need to check whether (a) the stress falls on an i or u; (b) in the absence of an accent mark, the
-- i or u would form a diphthong with a preceding or following vowel and the stress would be on that vowel.
-- The conditions are slightly different when dealing with preceding or following vowels because ui and ui
-- diphthongs are by default stressed on the second vowel. We also have to ignore h between the vowels.
local accented_syllable = export.add_accent_to_syllable(unaccented_syllables[sylno])
if sylno > 1 and rfind(unaccented_syllables[sylno - 1], "[aeo]$") and rfind(accented_syllable, "^h?[íú]") then
return true
end
if sylno < #syllables then
if rfind(accented_syllable, "í$") and rfind(unaccented_syllables[sylno + 1], "^h?[aeou]") or
rfind(accented_syllable, "ú$") and rfind(unaccented_syllables[sylno + 1], "^h?[aeio]") then
return true
end
end
return false
end
function export.make_plural(form, gender, special)
local retval = require(romut_module).handle_multiword(form, special,
function(term) return export.make_plural(term, gender) end, prepositions)
if retval then
return retval
end
if gender == "gneut" and rfind(form, "[x@]$") then
return {form .. "s"}
end
-- ends in unstressed vowel or á, é, ó
if rfind(form, "[aeiouáéó]$") then return {form .. "s"} end
-- ends in í or ú
if rfind(form, "[íú]$") then
return {form .. "es", form .. "s"}
end
-- ends in a vowel + z
if rfind(form, V .. "z$") then
return {rsub(form, "z$", "ces")}
end
-- ends in cons + s/z
if rfind(form, C.."[sz]$") then return {form} end
-- ends in s/z + cons
if rfind(form, "[sz]"..C.."$") then return {form} end
local syllables = export.syllabify(form)
-- ends in s or x with more than 1 syllable, last syllable unstressed
if syllables[2] and rfind(form, "[sx]$") and not rfind(syllables[#syllables], AV) then
return {form}
end
-- ends in l, r, n, d, z, or j with 3 or more syllables, stressed on third to last syllable
if syllables[3] and rfind(form, "[lrndzj]$") and rfind(syllables[#syllables - 2], AV) then
return {form}
end
-- ends in an accented vowel + consonant
if rfind(form, AV .. C .. "$") then
return {rsub(form, "(.)(.)$", function(vowel, consonant)
return export.remove_accent[vowel] .. consonant .. "es"
end)}
end
-- ends in a vowel + y, l, r, n, d, j, s, x
if rfind(form, "[aeiou][ylrndjsx]$") then
-- two or more syllables: add stress mark to plural; e.g. joven -> jóvenes
if syllables[2] and rfind(form, "n$") then
syllables[#syllables - 1] = export.add_accent_to_syllable(syllables[#syllables - 1])
return {table.concat(syllables, "") .. "es"}
end
return {form .. "es"}
end
-- ends in a vowel + ch
if rfind(form, "[aeiou]ch$") then return {form .. "es"} end
-- ends in two consonants
if rfind(form, C .. C .. "$") then return {form .. "s"} end
-- ends in a vowel + consonant other than l, r, n, d, z, j, s, or x
if rfind(form, "[aeiou][^aeioulrndzjsx]$") then return {form .. "s"} end
return nil
end
function export.make_feminine(form, special)
local retval = require(romut_module).handle_multiword(form, special, export.make_feminine, prepositions)
if retval then
if #retval ~= 1 then
error("Internal error: Should have one return value for make_feminine: " .. table.concat(retval, ","))
end
return retval[1]
end
if form:find("o$") then
local retval = form:gsub("o$", "a") -- discard second retval
return retval
end
local function make_stem(form)
return rsub(
form,
"^(.+)(.)(.)$",
function (before_stress, stressed_vowel, after_stress)
return before_stress .. (export.remove_accent[stressed_vowel] or stressed_vowel) .. after_stress
end)
end
if rfind(form, "[áíó]n$") or rfind(form, "[éí]s$") or rfind(form, "[dtszxñ]or$") or rfind(form, "ol$") then
-- holgazán, comodín, bretón (not común); francés, kirguís (not mandamás);
-- volador, agricultor, defensor, avizor, flexor, señor (not posterior, bicolor, mayor, mejor, menor, peor);
-- español, mongol
return make_stem(form) .. "a"
end
return form
end
function export.make_masculine(form, special)
local retval = require(romut_module).handle_multiword(form, special, export.make_masculine, prepositions)
if retval then
if #retval ~= 1 then
error("Internal error: Should have one return value for make_masculine: " .. table.concat(retval, ","))
end
return retval[1]
end
if form:find("dora$") then
local retval = form:gsub("a$", "") -- discard second retval
return retval
end
if form:find("a$") then
local retval = form:gsub("a$", "o") -- discard second retval
return retval
end
return form
end
return export
qq46tgo6gtotmf220p5pp0dpz3e6kus
Module:es-pronunc
828
35650
178040
169830
2026-09-23T05:08:17Z
Yivan000
4078
enwikt parity
178040
Scribunto
text/plain
--[=[
This module implements the templates {{es-pr}} and {{es-IPA}}.
Author: Benwing2
]=]
local export = {}
local m_IPA = require("Module:IPA")
local m_str_utils = require("Module:string utilities")
local m_table = require("Module:table")
local audio_module = "Module:audio"
local decorations_module = "Module:decorations"
local headword_data_module = "Module:headword/data"
local homophones_module = "Module:homophones"
local hyphenation_module = "Module:hyphenation"
local labels_module = "Module:labels"
local links_module = "Module:links"
local parameters_module = "Module:parameters"
local parse_utilities_module = "Module:parse utilities"
local references_module = "Module:references"
local rhymes_module = "Module:rhymes"
local force_cat = false -- for testing
--[=[
FIXME:
1. Port latest changes to production module. [DONE]
2. Finish work on rhymes and hyphenation. [DONE]
3. Handle <hmp:...> for homophones. [DONE]
4. Don't add comma before phonetic IPA. [DONE]
5. Handle secondary stress, suffixes, etc. in syllabification. [DONE]
6. Need some changes to syllable splitting in consonant clusters. (e.g. 'cum‧min‧gto‧ni‧ta') [DONE]
7. Fix handling of references to correspond to Portuguese module. [DONE]
8. Propagate qualifiers on individual pronun terms to rhymes and hyph.
9. Support raw phonemic/phonetic pronunciations. [DONE]
10. Support overall audio. [DONE]
11. Keep th/ph/kh/gh/tz ([[Ertzaintza]]) together when syllabifying (but not bh due to [[subhumano]], [[subhistoria]], etc.). [DONE]
12. Support <q:...> and <qq:...> on audio. [DONE]
13. Support <a:...> and <aa:...> (using {{a|...}}, left and right) on terms, rhymes, hyphenation, homophones and
audio. [DONE]
14. Support # instead of ; as separator between audio file and gloss and make sure it works if gloss has embedded # or
;. [DONE]
15. Use parse_inline_modifiers() in [[Module:parse utilities]]. [DONE]
]=]
--[=[
About styles, dialects and isoglosses:
From the standpoint of pronunciation, a given dialect is defined by isoglosses, which specify differences
in the way of pronouncing certain phonemes. You can think of a dialect as a collection of isoglosses.
For example, one isogloss is "distinción" (pronouncing written ''s'' and ''c/z'' differently) vs. "seseo"
(pronouncing them the same). Another is "lleísmo" (pronouncing written ''ll'' and ''y'' differently) vs.
"yeísmo" (pronouncing them the same). The dominant pronunciation in Spain can be described as
distinción + yeísmo, while the pronunciation in rural northern Spain can be described as distinción + lleísmo
and the pronunciation across much of the Andes mountains, Paraguay, and the Philippines can be described as seseo + lleísmo.
Specifically, the following isoglosses are recognized (note, the isogloss specs as used in this module
dispense with written accents):
-- "distincion" = pronouncing ''s'' and ''c/z'' differently
-- "seseo" = pronouncing ''s'' and ''c/z'' the same
-- "lleismo" = pronouncing ''ll'' and ''y'' differently
-- "yeismo" = pronouncing ''ll'' and ''y'' the same
-- "rioplatense" = Rioplatense speech, i.e. seseo+yeismo with ''ll'' and ''y'' pronounced specially, and a
clear distinction between initial ''hi-'' vs. initial ''ll-/y-''
-- "sheismo" = a type of Rioplatense speech, characteristic of Buenos Aires, where ''ll'' and ''y'' are
pronounced as /ʃ/
-- "zheismo" = a type of Rioplatense speech, found outside of Buenos Aires, where ''ll'' and ''y'' are
pronounced as /ʒ/
-- "quito" = seseo + lleismo, but pronouncing ''ll'' as /ʒ/
-- "yucatan" = seseo + yeismo, intervocalic ''y'' is pronounced ''i'' and lost in contact with ''i'' or ''e''
These isoglosses can be combined to yield one of the following eight dialects:
-- "distincion-lleismo": distinción + lleísmo
-- "distincion-yeismo": distinción + yeísmo
-- "seseo-lleismo": seseo + lleísmo
-- "seseo-yeismo": seseo + yeísmo
-- "rioplatense-sheismo": Rioplatense with /ʃ/ (Buenos Aires)
-- "rioplatense-zheismo": Rioplatense with /ʒ/ (non-Buenos Aires)
-- "quito"
-- "yucatan"
A "style" here is a set of dialects that pronounce a given word in a given fashion. For example, if we are only
considering the distinción/seseo and lleísmo/yeísmo isoglosses, there are four conceivable dialects (all of
which in fact exist). However, for a given word, more than one dialect may pronounce it the same. For
example, a word like [[paz]] has a ''z'' but no ''ll'', and so there are only two possible pronunciations for
the four dialects. Here, the two styles are "Spain" and "Latin America". Correspondingly, a word like [[pollo]]
with an ''ll'' but no ''z'' has two styles, which can approximately be described as "most of Spain and Latin
America" vs. "rural northern Spain, Andes Mountains, Paraguay, Philippines".
A "style spec" (indicated by the style= parameter to {{es-IPA}}) restricts the output to certain styles.
A style spec can be one of the following:
1. An isogloss, e.g. "distincion", "rioplatense"; if specified, only styles containing this isogloss are output.
2. A negated isogloss, e.g. "-rioplatense".
3. An intersection of isoglosses ("A and B"), e.g. "distincion+lleismo". This can be used to restrict to specific
dialects.
4. A union of isoglosses ("A or B"), e.g. "distincion,zheismo". If both plus and comma are used, plus takes
precedence, e.g. "seseo+lleismo,zheismo" means either the "seseo+lleismo" dialect or the "rioplatense-zheismo"
dialect.
An example where the style= parameter might be used is with the word [[bluetooth]], which has one pronunciation
in Spain/distinción (respelled "blutuz") but another in Latin America/seseo (respelled "blutud"). This might be
represented using {{es-pr}} as {{es-pr|blutuz<style:distincion>|blutud<style:seseo>}}.
]=]
local lang = require("Module:languages").getByCode("es")
local decompose = require("Module:es-common").decompose
local u = m_str_utils.char
local rfind = m_str_utils.find
local rsubn = m_str_utils.gsub
local rsplit = m_str_utils.split
local ulower = m_str_utils.lower
local ulen = m_str_utils.len
local unfd = mw.ustring.toNFD
local unfc = mw.ustring.toNFC
local AC = u(0x0301) -- acute = ́
local GR = u(0x0300) -- grave = ̀
local CFLEX = u(0x0302) -- circumflex = ̂
local TILDE = u(0x0303) -- tilde = ̃
local SYLDIV = u(0xFFF0) -- used to represent a user-specific syllable divider (.) so we won't change it
local vowel = "aeiouüyAEIOUÜY" -- vowel; include y so we get single-word y correct and for syllabifying from spelling
local V = "[" .. vowel .. "]" -- vowel class
local accent = AC .. GR .. CFLEX
local accent_c = "[" .. accent .. "]"
local stress = AC .. GR
local stress_c = "[" .. AC .. GR .. "]"
local ipa_stress = "ˈˌ"
local ipa_stress_c = "[" .. ipa_stress .. "]"
local sylsep = "%-." .. SYLDIV -- hyphen included for syllabifying from spelling
local sylsep_c = "[" .. sylsep .. "]"
local wordsep = "# "
local separator_not_wordsep = accent .. ipa_stress .. sylsep
local separator = separator_not_wordsep .. wordsep
local separator_c = "[" .. separator .. "]"
local C = "[^" .. vowel .. separator .. "]" -- consonant class including h
local C_NOT_H = "[^" .. vowel .. separator .. "h]" -- consonant class not including h
local C_OR_WORDSEP = "[^" .. vowel .. separator_not_wordsep .. "]" -- consonant class including h, or word separator
local T = "[^" .. vowel .. "lrɾjw" .. separator .. "]" -- obstruent or nasal
local unstressed_words = m_table.listToSet({
"el", "la", "los", "las", -- definite articles
"un", -- single-syllable indefinite articles
"me", "te", "se", "lo", "le", "nos", "os", "les", -- unstressed object pronouns
"mi", "mis", "tu", "tus", "su", "sus", -- unstressed possessive pronouns
"que", "si", -- subordinating conjunctions
"y", "e", "o", "u", "mas", -- coordinating conjunctions
"de", "del", "a", "al", -- basic prepositions + combinations with articles
"por", "en", "con", -- other prepositions
})
-- version of rsubn() that discards all but the first return value
local function rsub(term, foo, bar)
local retval = rsubn(term, foo, bar)
return retval
end
-- version of rsubn() that returns a 2nd argument boolean indicating whether
-- a substitution was made.
local function rsubb(term, foo, bar)
local retval, nsubs = rsubn(term, foo, bar)
return retval, nsubs > 0
end
-- apply rsub() repeatedly until no change
local function rsub_repeatedly(term, foo, bar)
while true do
local new_term = rsub(term, foo, bar)
if new_term == term then
return term
end
term = new_term
end
end
local function split_on_comma(term)
if not term then
return nil
end
if term:find(",%s") then
return require(parse_utilities_module).split_on_comma(term)
elseif term:find(",") then
return rsplit(term, ",")
else
return {term}
end
end
-- Remove any HTML from the formatted text and resolve links, since the extra characters don't contribute to the
-- displayed length.
local function convert_to_raw_text(text)
text = rsub(text, "<.->", "")
if text:find("%[%[") then
text = require(links_module).remove_links(text)
end
return text
end
-- Return the approximate displayed length in characters.
local function textual_len(text)
return ulen(convert_to_raw_text(text))
end
local function construct_default_differences(dialect)
if dialect == "distincion-lleismo" then
return {
distincion_different = false,
lleismo_different = false,
sheismo_different = false,
need_rioplat = false,
need_quito = false,
need_yucatan = false,
}
end
return nil
end
-- Main syllable-division algorithm. Can be called either directly on spelling (when hyphenating) or after
-- non-trivial processing of respelling in the direction of pronunciation (when generating pronunciation).
local function syllabify_from_spelling_or_pronun(text, is_spelling)
-- Part 1: Divide before the last consonant in a cluster of consonants between vowels (but don't divide a VhV
-- sequence; [[prohibir]] should be prohi.bir). Then move the syllable division marker leftwards over clusters that
-- can form onsets.
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*)(" .. C_NOT_H .. V .. ")", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. C .. "+)(" .. C .. V .. ")", "%1.%2")
-- Puerto Rico + most of Spain divide tl as t.l. Mexico and the Canary Islands have .tl. Unclear what other regions
-- do. Here we choose to go with .tl. See https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf
-- and https://www.spanishdict.com/guide/spanish-syllables-and-syllabification-rules.
-- NOTE: When run on pronun, we have already eliminated c and v, but not when run on spelling.
-- When run on pronun, don't include r, which at this point represents the trill.
local cluster_r = is_spelling and "rɾ" or "ɾ"
-- Don't divide Cl or Cr where C is a stop or fricative, except for dl.
text = rsub(text, "([pbfvkctg])%.([l" .. cluster_r .. "])", ".%1%2")
text = text:gsub("d%.([" .. cluster_r .. "])", ".d%1")
-- Don't divide ch, sh, ph, th, dh, fh, kh or gh. Do allow bh to be divided ([[subhumano]], [[subhúmedo]], etc.).
text = rsub(text, "([csptdfkg])%.h", ".%1h")
-- Don't divide ll or rr.
text = rsub(text, "([lr])%.%1", ".%1%1")
-- Don't divide tz ([[Ertzaintza]], [[quetzal]], [[hertziano]] and other words of Basque, Nahuatl and German
-- origin).
text = rsub(text, "t%.z", ".tz")
-- Per https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf, tl at the end of a word
-- (as in nahuatl, Popocatepetl etc.) is divided .tl from the previous vowel.
if is_spelling then
text = text:gsub("([^. %-])tl$", "%1.tl")
text = text:gsub("([^. %-])(tl[ %-])", "%1.%2")
else
text = text:gsub("([^.#])tl#", "%1.tl")
end
-- Part 2: Divide hiatuses. Any aeo, or stressed iuüy, should be syllabically divided from a following aeo or
-- stressed iuüy. Also divide ii and uu sequences ([[antiincendios]], [[shiita]], [[vacuum]]). Note that words with
-- ii or uu next to a vowel (e.g. [[hawaiiano]]) will not make it to this point unchanged; the i or u adjacent to
-- a vowel (or the second one if both are adjacent to vowels) will get converted to a consonant symbol (temporarily
-- when syllabifying spelling).
text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?[aeo])", "%1.%2")
text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?" .. V .. stress_c .. ")", "%1.%2")
text = rsub(text, "([iuüyIUÜY]" .. stress_c .. ")(h?[aeo])", "%1.%2")
text = rsub_repeatedly(text, "([iuüyIUÜY]" .. stress_c .. ")(h?" .. V .. stress_c .. ")", "%1.%2")
text = rsub_repeatedly(text, "([iI]" .. accent_c .. "*)(h?i)", "%1.%2")
text = rsub_repeatedly(text, "([uU]" .. accent_c .. "*)(h?u)", "%1.%2")
return text
end
local function syllabify_from_spelling(text)
text = decompose(text)
-- start at FFF1 because FFF0 is used for SYLDIV
-- Temporary replacements for characters we want treated as default consonants. The C and related consonant regexes
-- treat all unknown characters as consonants.
local TEMP_I = u(0xFFF1)
local TEMP_U = u(0xFFF2)
local TEMP_Y_CONS = u(0xFFF3)
local TEMP_QU = u(0xFFF4)
local TEMP_QU_CAPS = u(0xFFF5)
local TEMP_GU = u(0xFFF6)
local TEMP_GU_CAPS = u(0xFFF7)
local TEMP_H = u(0xFFF8)
-- Change user-specified . into SYLDIV so we don't shuffle it around when dividing into syllables.
text = text:gsub("%.", SYLDIV)
text = rsub(text, "y(" .. V .. ")", TEMP_Y_CONS .. "%1")
-- We don't want to break -sh- except in desh-, e.g. [[deshuesar]], [[deshonra]], [[deshecho]]. Normally, -sh- is
-- automatically preserved, so we replace the h with a temporary symbol to avoid this.
text = text:gsub("^([Dd]es)h", "%1" .. TEMP_H)
text = text:gsub("([ %-][Dd]es)h", "%1" .. TEMP_H)
-- qu mostly handled correctly automatically, but not in quietud
text = rsub(text, "qu(" .. V .. ")", TEMP_QU .. "%1")
text = rsub(text, "Qu(" .. V .. ")", TEMP_QU_CAPS .. "%1")
text = rsub(text, "gu(" .. V .. ")", TEMP_GU .. "%1")
text = rsub(text, "Gu(" .. V .. ")", TEMP_GU_CAPS .. "%1")
local vowel_to_glide = { ["i"] = TEMP_I, ["u"] = TEMP_U }
-- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]],
-- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]],
-- etc. When we do this we need to help the syllabification particularly of words with -hiV- and -huV- in them,
-- otherwise we get e.g. 'an.tih.ie.lo' because we converted the i following the h to a consonant. Add .* at the
-- beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano.
text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*)(h?)([iu])(" .. V .. ")",
function (v1, h, iu, v2) return v1 .. "." .. h .. vowel_to_glide[iu] .. v2 end
)
text = syllabify_from_spelling_or_pronun(text, "is spelling")
text = text:gsub(SYLDIV, ".")
text = text:gsub(TEMP_I, "i")
text = text:gsub(TEMP_U, "u")
text = text:gsub(TEMP_Y_CONS, "y")
text = text:gsub(TEMP_QU, "qu")
text = text:gsub(TEMP_QU_CAPS, "Qu")
text = text:gsub(TEMP_GU, "gu")
text = text:gsub(TEMP_GU_CAPS, "Gu")
text = text:gsub(TEMP_H, "h")
text = unfc(text)
-- No qualifiers from dialect tags because we assume all dialects hyphenate the same way.
-- FIXME: There are region-specific ways of hyphenating -tl-. See above. We don't currently handle this properly.
return text
end
-- Generate the IPA of a given respelling, where a respelling is the representation of the pronunciation of a given
-- Spanish term using Spanish spelling conventions (augmented in a few cases with extra conventions such as 'sh' for
-- /ʃ/).
-- ɟ and ĉ are used internally to represent [ʝ⁓ɟ͡ʝ] and [t͡ʃ]
--
function export.IPA(text, dialect, phonetic)
local distincion = dialect == "distincion-lleismo" or dialect == "distincion-yeismo"
local lleismo = dialect == "distincion-lleismo" or dialect == "seseo-lleismo" or dialect == "quito"
local rioplat = dialect == "rioplatense-sheismo" or dialect == "rioplatense-zheismo"
local sheismo = dialect == "rioplatense-sheismo"
local quito = dialect == "quito"
local yucatan = dialect == "yucatan"
local distincion_different = false
local lleismo_different = false
local need_rioplat = false
local need_quito = false
local need_yucatan = false
local initial_hi = false
local sheismo_different = false
-- start at FFF1 because FFF0 is used for SYLDIV
local TEMP_Y = u(0xFFF1)
local TEMP_W = u(0xFFF2)
text = ulower(text or mw.loadData("Module:headword/data").pagename)
-- decompose everything but ç, ñ and ü
text = decompose(text)
-- convert commas and en/en dashes to IPA foot boundaries
text = rsub(text, "%s*[,–—]%s*", " | ")
-- question mark or exclamation point in the middle of a sentence -> IPA foot boundary
text = rsub(text, "([^%s])%s*[¡!¿?]%s*([^%s])", "%1 | %2")
-- canonicalize multiple spaces and remove leading and trailing spaces
local function canon_spaces(text)
text = rsub(text, "%s+", " ")
text = rsub(text, "^ ", "")
text = rsub(text, " $", "")
return text
end
text = canon_spaces(text)
-- Make prefixes unstressed unless they have an explicit stress marker; also make certain
-- monosyllabic words (e.g. [[el]], [[la]], [[de]], [[en]], etc.) without stress marks be
-- unstressed.
local words = rsplit(text, " ")
for i, word in ipairs(words) do
if rfind(word, "%-$") and not rfind(word, accent_c) or unstressed_words[word] then
-- add CFLEX to the last vowel not the first one, or we will mess up 'que' by
-- adding the CFLEX after the 'u'
words[i] = rsub(word, "^(.*" .. V .. ")", "%1" .. CFLEX)
end
end
text = table.concat(words, " ")
-- Convert hyphens to spaces, to handle [[Austria-Hungría]], [[franco-italiano]], etc.
text = rsub(text, "%-", " ")
-- canonicalize multiple spaces again, which may have been introduced by hyphens
text = canon_spaces(text)
-- now eliminate punctuation
text = rsub(text, "[¡!¿?']", "")
-- put # at word beginning and end and double ## at text/foot boundary beginning/end
text = rsub(text, " | ", "# | #")
text = "##" .. rsub(text, " ", "# #") .. "##"
--determining whether "y" is a consonant or a vowel
text = rsub(text, "y(" .. V .. ")", "ɟ%1") -- not the real sound
-- word-final -ay/-ey/-oy/-uy is stressed whereas word-final -ai/-ei/-oi/-ui is not; in addition,
-- word-final -uy is /uj/ whereas word-final -ui is /wi/ (e.g. [[muy]] vs. [[fui]])
text = rsub(text, "([aeou])y#", "%1" .. TEMP_Y .. "#") -- a temporary symbol; replaced with i below
text = rsub(text, "y", "i")
-- handle certain combinations; sh handling needs to go before x handling to avoid issues with [[exhausto]]
text = rsub(text, "ch", "ĉ") --not the real sound
-- We want to keep desh- ([[deshuesar]]) as-is. Converting to des- won't work because we want it syllabified as
-- 'des.we.saɾ' not #'de.swe.saɾ' (cf. [[desuelo]] /de.swe.lo/ from [[desolar]]).
text = rsub(text, "#desh", "!") --temporary symbol
text = rsub(text, "sh", "ʃ")
text = rsub(text, "!", "#desh") --restore
text = rsub(text, "#[ckp]([st])", "#%1") -- [[ctónico]], [[psicología]], [[pterodáctilo]]
--x
text = rsub(text, "#x", "#s") -- xenofobia, xilófono, etc.
text = rsub(text, "x", "ks")
--c, g, q
text = rsub(text, "c([ie])", (distincion and "θ" or "z") .. "%1") -- not the real LatAm sound
text = rsub(text, "g([ie])", "x%1") -- must happen after handling of x above
text = rsub(text, "gu([ie])", "g%1")
text = rsub(text, "gü([ie])", "gu%1")
-- following must happen before stress assignment; [[branding]] has initial stress like 'brandin'
text = rsub(text, "ng([^aeiouüwhlr])", "n%1") -- [[Bangkok]], [[ángstrom]], [[branding]]
text = rsub(text, "qu([ie])", "k%1")
text = rsub(text, "ü", "u") -- [[Düsseldorf]], [[hübnerita]], obsolete [[freqüentemente]], etc.
text = rsub(text, "q", "k") -- [[quark]], [[Qatar]], [[burqa]], [[Iraq]], etc.
text = rsub(text, "[zç]", distincion and "θ" or "z") -- not the real LatAm sound; "ç" became "z" in 1726
if rfind(text, "[θz]") then
distincion_different = true
end
-- map various consonants to their phoneme equivalent
text = rsub(text, "[cjñrv]", {["c"]="k", ["j"]="x", ["ñ"]="ɲ", ["r"]="ɾ", ["v"]="b" })
-- handle word- and syllable-initial hiV ([[hielo]], [[enhiesto]], [[deshielo]], ...)
local word_initial_hi, syl_initial_hi
text, word_initial_hi = rsubb(text, "#h?i(" .. V .. ")", rioplat and "#j%1" or "#ɟ%1")
text, syl_initial_hi = rsubb(text, "(" .. C .. sylsep_c .. "*)hi(" .. V .. ")", rioplat and "%1j%2" or "%1ɟ%2")
initial_hi = word_initial_hi or syl_initial_hi
-- handle word- and syllable-initial huV ([[huevo]], [[deshuesar]])
text = rsubb(text, "(" .. C_OR_WORDSEP .. sylsep_c .. "*)hu(" .. V .. ")", "%1" .. TEMP_W .. "%2")
-- handle double consonants that have a pronunciation different from their single equivalents
-- double l
lleismo_different = rfind(text, "ll")
need_quito = lleismo_different and rfind(text, "ɟ")
text = rsub(text, "ll", lleismo and "ʎ" or "ɟ")
-- handle intervocalic -y-
need_yucatan = rfind(text, V .. accent_c .. "*" .. sylsep_c .. "*[ʎɟ]" .. V)
if yucatan then
text = rsub_repeatedly(text, "([ei]" .. accent_c .. "*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. sylsep_c .. "*)ɟ" .. "([ei])", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .."*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1i%2")
end
-- trill in #r, lr ([[alrededor]], [[malrotar]]), nr ([[enriquecer]], [[sonrisa]], etc.), sr ([[Israel]],
-- [[desregular]], etc.), zr ([[Azrael]], [[cruzrojista]]), rr
text = rsub(text, "ɾɾ", "r")
text = rsub(text, "([#lnszθ])ɾ", "%1r")
-- double n (e.g. [[[ennoblecer]])
text = rsub(text, "nn", "N")
-- double b (e.g. [[subbase]])
text = rsub(text, "bb", "B")
-- reduce any remaining double consonants ([[Addis Abeba]], [[cappa]], [[descender]] in Latin America ...);
-- do this before handling of -nm- e.g. in [[inmigración]], which generates a double consonant, and do this
-- before voicing stops before obstruents, to avoid problems with [[cappa]] and [[crackear]]
text = rsub(text, "(" .. C .. ")%1", "%1")
-- also reduce sz (Latin American in [[fascinante]], etc.)
text = rsub(text, "sz", "s")
-- restore double n, b
text = rsub(text, "N", "nn")
text = rsub(text, "B", "bb")
-- voiceless stop to voiced before obstruent or nasal; but intercept -ts-, -tz-
local voice_stop = { ["p"] = "b", ["t"] = "d", ["k"] = "g" }
text = rsub(text, "t(" .. separator_c .. "*[szθ])", "!%1") -- temporary symbol
text = rsub(text, "([ptk])(" .. separator_c .. "*" .. T .. ")",
function(stop, after) return voice_stop[stop] .. after end)
text = rsub(text, "!", "t")
text = rsub(text, "n([# .]*[bpm])", "m%1")
-- remove silent h before syllable division
text = rsub(text, "h", "")
-- convert i/u between vowels to glide
local vowel_to_glide = { ["i"] = "j", ["u"] = "w" }
-- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]],
-- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]],
-- etc. Add .* at the beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano.
text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*h?)([iu])(" .. V .. ")",
function (v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end
)
--syllable division
text = syllabify_from_spelling_or_pronun(text, false)
--diphthongs; do not include TEMP_Y here
text = rsub(text, "i([aeou])", "j%1")
text = rsub(text, "u([aeio])", "w%1")
local accent_to_stress_mark = { [AC] = "ˈ", [GR] = "ˌ", [CFLEX] = "" }
local function accent_word(word, syllables)
-- Now stress the word. If any accent exists in the word (including ^ indicating an unaccented word),
-- put the stress mark(s) at the beginning of the indicated syllable(s). Otherwise, apply the default
-- stress rule.
if rfind(word, accent_c) then
for i = 1, #syllables do
syllables[i] = rsub(syllables[i], "^(.*)(" .. accent_c .. ")(.*)$",
function(pre, accent, post) return accent_to_stress_mark[accent] .. pre .. post end
)
end
else
-- Default stress rule. Words without vowels (e.g. IPA foot boundaries) don't get stress.
if #syllables > 1 and (rfind(word, "[^" .. vowel .. "ns#]#") or rfind(word, C .. "[ns]#")) or #syllables == 1 and rfind(word, V) then
syllables[#syllables] = "ˈ" .. syllables[#syllables]
elseif #syllables > 1 then
syllables[#syllables - 1] = "ˈ" .. syllables[#syllables - 1]
end
end
end
local words = rsplit(text, " ")
for j, word in ipairs(words) do
-- accentuation
local syllables = rsplit(word, "%.")
if rfind(word, "men%.te#") then
local mente_syllables
-- Words ends in -mente (converted above to ménte); add a stress to the preceding portion
-- (e.g. [[agriamente]] -> 'ágriaménte') unless already stressed (e.g. [[rápidamente]]).
-- It will be converted to secondary stress further below. Essentially, we rip the word apart
-- into two words ('mente' and the preceding portion) and stress each one independently.
mente_syllables = {}
mente_syllables[2] = table.remove(syllables)
mente_syllables[1] = table.remove(syllables)
accent_word(table.concat(syllables, "."), syllables)
accent_word(table.concat(mente_syllables, "."), mente_syllables)
table.insert(syllables, mente_syllables[1])
table.insert(syllables, mente_syllables[2])
else
accent_word(word, syllables)
end
-- Vowels are nasalized if followed by nasal in same syllable.
if phonetic then
for i = 1, #syllables do
-- first check for two vowels (veinte)
syllables[i] = rsub(syllables[i], "(" .. V .. ")(" .. V .. ")([mnɲ])",
"%1" .. TILDE .. "%2" .. TILDE .. "%3")
-- then for one vowel
syllables[i] = rsub(syllables[i], "(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2")
end
end
-- Reconstruct the word.
words[j] = table.concat(syllables, ".")
end
text = table.concat(words, " ")
text = rsub(text, TEMP_Y, "i") --final -ay/-ey/-oy/-uy
text = rsub(text, "z", "s") --real sound of LatAm Z
-- suppress syllable mark before IPA stress indicator
text = rsub(text, "%.(" .. ipa_stress_c .. ")", "%1")
--make all primary stresses but the last one be secondary
text = rsub_repeatedly(text, "ˈ(.+)ˈ", "ˌ%1ˈ")
if (not initial_hi and rfind(text, "[ʎɟ]")) or (rfind(text, sylsep_c .. "[ʎɟ]")) then
sheismo_different = true
end
if rioplat then
if not initial_hi then
if sheismo then
text = rsub(text, "ɟ", "ʃ")
else
text = rsub(text, "ɟ", "ʒ")
end
else
if sheismo then
text = rsub(text, sylsep_c .. "(ɟ)", "ʃ")
else
text = rsub(text, sylsep_c .. "(ɟ)", "ʒ")
end
end
end
if quito then text = rsub(text, "ʎ", "ʒ") end
--phonetic transcription
if phonetic then
-- θ, s, f before voiced consonants
local voiced = "mnɲbdɟgʎ" .. TEMP_W
local r = "ɾr"
local tovoiced = {
["θ"] = "θ̬",
["s"] = "z",
["f"] = "v",
}
local function voice(sound, following)
return tovoiced[sound] .. following
end
text = rsub(text, "([θs])(" .. separator_c .. "*[" .. voiced .. r .. "])", voice)
text = rsub(text, "(f)(" .. separator_c .. "*[" .. voiced .. "])", voice)
-- fricative vs. stop allophones; first convert stops to fricatives, then back to stops
-- after nasals and sometimes after l
local stop_to_fricative = {["b"] = "β", ["d"] = "ð", ["ɟ"] = "ʝ", ["g"] = "ɣ"}
local fricative_to_stop = {["β"] = "b", ["ð"] = "d", ["ʝ"] = "ɟ", ["ɣ"] = "g"}
text = rsub(text, "[bdɟg]", stop_to_fricative)
text = rsub(text, "([mnɲ]" .. separator_c .. "*)([βɣ])",
function(nasal, fricative) return nasal .. fricative_to_stop[fricative] end
)
text = rsub(text, "([lʎmnɲ]" .. separator_c .. "*)([ðʝ])",
function(nasal_l, fricative) return nasal_l .. fricative_to_stop[fricative] end
)
text = rsub(text, "(##" .. ipa_stress_c .. "*)([βɣðʝ])",
function(stress, fricative) return stress .. fricative_to_stop[fricative] end
)
text = rsub(text, "[td]", {["t"] = "t̪", ["d"] = "d̪"})
-- nasal assimilation before consonants
local labiodental, dentialveolar, dental, alveolopalatal, palatal, velar =
"ɱ", "n̪", "n̟", "nʲ", "ɲ", "ŋ"
local nasal_assimilation = {
["f"] = labiodental,
["t"] = dentialveolar, ["d"] = dentialveolar,
["θ"] = dental,
["ĉ"] = alveolopalatal,
["ʃ"] = alveolopalatal,
["ʒ"] = alveolopalatal,
["ɟ"] = palatal, ["ʎ"] = palatal,
["k"] = velar, ["x"] = velar, ["g"] = velar,
}
text = rsub(text, "n(" .. separator_c .. "*)(.)",
function(stress, following) return (nasal_assimilation[following] or "n") .. stress .. following end
)
-- lateral assimilation before consonants
text = rsub(text, "l(" .. separator_c .. "*)(.)",
function(stress, following)
local l = "l"
if following == "t" or following == "d" then -- dentialveolar
l = "l̪"
elseif following == "θ" then -- dental
l = "l̟"
elseif following == "ĉ" or following == "ʃ" then -- alveolopalatal
l = "lʲ"
end
return l .. stress .. following
end)
--semivowels
text = rsub(text, "([aeouãẽõũ][iĩ])", "%1̯")
text = rsub(text, "([aeioãẽĩõ][uũ])", "%1̯")
-- voiced fricatives are actually approximants
text = rsub(text, "([βðɣ])", "%1̞")
end
-- convert fake symbols to real ones
local final_conversions = {
["ħ"] = "h", -- fake aspirated "h" to real "h"
["ĉ"] = "t͡ʃ", -- fake "ch" to real "ch"
["ɟ"] = phonetic and "ɟ͡ʝ" or "ʝ", -- fake "y" to real "y"
-- do the following at the very end so we can use regular g throughout
["g"] = "ɡ", -- U+0067 LATIN SMALL LETTER G → U+0261 LATIN SMALL LETTER SCRIPT G
[TEMP_W] = "w̝", -- see https://en.wikipedia.org/wiki/Spanish_orthography for this
}
text = rsub(text, "[ħĉɟg" .. TEMP_W .. "]", final_conversions)
-- remove # symbols at word and text boundaries
text = rsub(text, "#", "")
text = unfc(text)
-- The values in `differences` are only accurate when the dialect is 'distincion-lleismo'
-- because we look for sounds like /θ/ and /ʎ/ that are only present in that dialect.
-- The calling code knows to only use this structure in conjunction with this dialect.
-- but to make sure of this we set the structure to nil for other dialects.
local differences = nil
if dialect == "distincion-lleismo" then
differences = {
distincion_different = distincion_different,
lleismo_different = lleismo_different,
need_rioplat = initial_hi or sheismo_different,
sheismo_different = sheismo_different,
need_quito = need_quito,
need_yucatan = need_yucatan,
}
end
local ret = {
text = text,
differences = differences,
}
return ret
end
-- For bot usage; {{#invoke:es-pronunc|IPA_string|SPELLING|style=STYLE|phonetic=PHONETIC}}
-- where
--
-- 1. SPELLING is the word or respelling to generate pronunciation for;
-- 2. required parameter style= indicates the pronunciation style to generate
-- (e.g. "distincion-yeismo" for distinción+yeísmo, as is common in Spain;
-- see the comment above export.IPA() above for the full list);
-- 3. phonetic=1 specifies to generate the phonetic rather than phonemic pronunciation;
function export.IPA_string(frame)
local iparams = {
[1] = {},
["style"] = {required = true},
["phonetic"] = {type = "boolean"},
}
local iargs = require(parameters_module).process(frame.args, iparams)
local retval = export.IPA(iargs[1], iargs.style, iargs.phonetic)
return retval.text
end
-- Generate all relevant dialect pronunciations and group into styles. See the comment above about dialects and styles.
-- A "pronunciation" here could be for example the IPA phonemic/phonetic representation of the term or the IPA form of
-- the rhyme that the term belongs to. If `style_spec` is nil, this generates all styles for all dialects, but
-- `style_spec` can also be a style spec such as "seseo" or "distincion+yeismo" (see comment above) to restrict the
-- output. `dodialect` is a function of two arguments, `ret` and `dialect`, where `ret` is the return-value table (see
-- below), and `dialect` is a string naming a particular dialect, such as "distincion-lleismo" or "rioplatense-sheismo".
-- `dodialect` should side-effect the `ret` table by adding an entry to `ret.pronun` for the dialect in question.
--
-- The return value is a table of the form
--
-- {
-- pronun = {DIALECT = {PRONUN, PRONUN, ...}, DIALECT = {PRONUN, PRONUN, ...}, ...},
-- expressed_styles = {STYLE_GROUP, STYLE_GROUP, ...},
-- }
--
-- where:
-- 1. DIALECT is a string such as "distincion-lleismo" naming a specific dialect.
-- 2. PRONUN is a table describing a particular pronunciation. If the dialect is "distincion-lleismo", there should be
-- a field in this table named `differences`, but where other fields may vary depending on the type of pronunciation
-- (e.g. phonemic/phonetic or rhyme). See below for the form of the PRONUN table for phonemic/phonetic pronunciation
-- vs. rhyme and the form of the `differences` field.
-- 3. STYLE_GROUP is a table of the form {tag = "HIDDEN_TAG", styles = {INNER_STYLE, INNER_STYLE, ...}}. This describes
-- a group of related styles (such as those for Latin America) that by default (the "hidden" form) are displayed as
-- a single line, with an icon on the right to "open" the style group into the "shown" form, with multiple lines
-- for each style in the group. The tag of the style group is the text displayed before the pronunciation in the
-- default "hidden" form, such as "Spain" or "Latin America". It can have the special value of `false` to indicate
-- that no tag text is to be displayed. Note that the pronunciation shown in the default "hidden" form is taken
-- from the first style in the style group.
-- 4. INNER_STYLE is a table of the form {tag = "SHOWN_TAG", pronun = {PRONUN, PRONUN, ...}}. This describes a single
-- style (such as for the Andes Mountains and Paraguay in the case where the seseo+lleismo accent differs from all others), to
-- be shown on a single line. `tag` is the text preceding the displayed pronunciation, or `false` if no tag text
-- is to be displayed. PRONUN is a table as described above and describes a particular pronunciation.
--
-- The PRONUN table has the following form for the full phonemic/phonetic pronunciation:
--
-- {
-- phonemic = "PHONEMIC",
-- phonetic = "PHONETIC",
-- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...},
-- }
--
-- Here, `phonemic` is the phonemic pronunciation (displayed as /.../) and `phonetic` is the phonetic pronunciation
-- (displayed as [...]).
--
-- The PRONUN table has the following form for the rhyme pronunciation:
--
-- {
-- rhyme = "RHYME_PRONUN",
-- num_syl = {NUM, NUM, ...},
-- q = nil or {QUALIFIER, QUALIFIER, ...},
-- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...},
-- }
--
-- Here, `rhyme` is a phonemic pronunciation such as "ado" for [[abogado]] or "iʝa"/"iʎa" for [[tortilla]] (depending
-- on the dialect), and `num_syl` is a list of the possible numbers of syllables for the term(s) that have this rhyme
-- (e.g. {4} for [[abogado]], {3} for [[tortilla]] and {4, 5} for [[biología]], which may be syllabified as
-- bio.lo.gí.a or bi.o.lo.gí.a). `num_syl` is used to generate syllable-count categories such as
-- [[Category:Rhymes:Spanish/ia/4 syllables]] in addition to [[Category:Rhymes:Spanish/ia]]. `num_syl` may be nil to
-- suppress the generation of syllable-count categories; this is typically the case with multiword terms.
-- `q`, if non-nil, comes from the user using the syntax e.g. <rhyme:iʃa<q:Buenos Aires>>.
--
-- The value of the `differences` field in the PRONUN table (which, as noted above, only needs to be present for the
-- "distincion-lleismo" dialect, and otherwise should be nil) is a table containing flags indicating whether and how
-- the per-dialect pronunciations differ. This is an optimization to avoid having to generate all six dialectal
-- pronunciations and compare them. It has the following form:
--
-- {
-- distincion_different = BOOLEAN,
-- lleismo_different = BOOLEAN,
-- need_rioplat = BOOLEAN,
-- sheismo_different = BOOLEAN,
-- need_quito = BOOLEAN,
-- need_yucatan = BOOLEAN,
-- }
--
-- where:
-- 1. `distincion_different` should be `true` if the "distincion" and "seseo" pronunciations differ;
-- 2. `lleismo_different` should be `true` if the "lleismo" and "yeismo" pronunciations differ;
-- 3. `need_rioplat` should be `true` if the Rioplatense pronunciations differ from the seseo+yeismo pronunciation;
-- 4. `sheismo_different` should be `true` if the "sheismo" and "zheismo" pronunciations differ.
-- 5. `need_quito` should be `true` if the "quito" and "zheismo" pronunciations differ.
-- 6. `need_yucatan` should be `true` if the "yucatan" and "yeismo" pronunciations differ;
local function express_all_styles(style_spec, dodialect)
local ret = {
pronun = {},
expressed_styles = {},
}
local need_rioplat
local need_quito
local need_yucatan
-- Add a style object (see INNER_STYLE above) that represents a particular style to `ret.expressed_styles`.
-- `hidden_tag` is the tag text to be used when the style group containing the style is in the default "hidden"
-- state (e.g. "Spain", "Latin America" or false if there is only one style group and no tag text should be
-- shown), while `tag` is the tag text to be used when the individual style is shown (e.g. a description such as
-- "most of Spain and Latin America", "Andes Mountains and Paraguay" or "everywhere but Argentina and Uruguay").
-- `representative_dialect` is one of the dialects that this style represents, and whose pronunciation is stored in
-- the style object. `matching_styles` is a hyphen separated string listing the isoglosses described by this style.
-- For example, if the term has an ''ll'' but no ''c/z'', the `tag` text for the yeismo pronunciation will be
-- "most of Spain and Latin America" and `matching_styles` will be "distincion-seseo-yeismo", indicating that
-- it corresponds to both the "distincion" and "seseo" isoglosses as well as the "yeismo" isogloss. This is used
-- when a particular style spec is given. If `matching_styles` is omitted, it takes its value from
-- `representative_dialect`; this is used when the style contains only a single dialect.
local function express_style(hidden_tag, tag, representative_dialect, matching_styles)
matching_styles = matching_styles or representative_dialect
-- If the Rioplatense pronunciation isn't distinctive, add all Rioplatense isoglosses.
if not need_rioplat then
matching_styles = matching_styles .. "-rioplatense-sheismo-zheismo"
end
-- also Quito
if not need_quito then
matching_styles = matching_styles .. "-quito"
end
-- Yucatan
if not need_yucatan then
matching_styles = matching_styles .. "-yucatan"
end
-- If style specified, make sure it matches the requested style.
local style_matches
if not style_spec then
style_matches = true
else
local style_parts = rsplit(matching_styles, "%-")
local or_styles = rsplit(style_spec, "%s*,%s*")
for _, or_style in ipairs(or_styles) do
local and_styles = rsplit(or_style, "%s*%+%s*")
local and_matches = true
for _, and_style in ipairs(and_styles) do
local negate
if and_style:find("^%-") then
and_style = and_style:gsub("^%-", "")
negate = true
end
local this_style_matches = false
for _, part in ipairs(style_parts) do
if part == and_style then
this_style_matches = true
break
end
end
if negate then
this_style_matches = not this_style_matches
end
if not this_style_matches then
and_matches = false
end
end
if and_matches then
style_matches = true
break
end
end
end
if not style_matches then
return
end
-- Fetch the representative dialect's pronunciation if not already present.
if not ret.pronun[representative_dialect] then
dodialect(ret, representative_dialect)
end
-- Insert the new style into the style group, creating the group if necessary.
local new_style = {
tag = tag,
pronun = ret.pronun[representative_dialect],
}
for _, hidden_tag_style in ipairs(ret.expressed_styles) do
if hidden_tag_style.tag == hidden_tag then
table.insert(hidden_tag_style.styles, new_style)
return
end
end
table.insert(ret.expressed_styles, {
tag = hidden_tag,
styles = {new_style},
})
end
-- For each type of difference, figure out if the difference exists in any of the given respellings. We do this by
-- generating the pronunciation for the dialect "distincion-lleismo", for each respelling. In the process of
-- generating the pronunciation for a given respelling, it computes how the other dialects for that respelling
-- differ. Then we take the union of these differences across the respellings.
dodialect(ret, "distincion-lleismo")
local differences = {}
for _, difftype in ipairs { "distincion_different", "lleismo_different", "need_rioplat", "sheismo_different", "need_quito", "need_yucatan" } do
for _, pronun in ipairs(ret.pronun["distincion-lleismo"]) do
if pronun.differences[difftype] then
differences[difftype] = true
end
end
end
local distincion_different = differences.distincion_different
local lleismo_different = differences.lleismo_different
need_rioplat = differences.need_rioplat
local sheismo_different = differences.sheismo_different
need_quito = differences.need_quito
need_yucatan = differences.need_yucatan
-- Now, based on the observed differences, figure out how to combine the individual dialects into styles and
-- style groups.
if not distincion_different and not lleismo_different then
if not need_rioplat then
if not need_yucatan then
express_style(false, false, "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
else
express_style(false, "everywhere but northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
end
else
if not need_yucatan then
express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>> and <<Uruguay>>", "distincion-lleismo",
"distincion-seseo-lleismo-yeismo")
else
express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>, northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
end
end
elseif distincion_different and not lleismo_different then
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, <<Spain>>", "distincion-lleismo", "distincion-lleismo-yeismo")
if not need_rioplat and not need_yucatan then
express_style("<<Latin America>>, <<Philippines>>", "<<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo")
else
express_style("<<Latin America>>, <<Philippines>>", "most of <<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo")
end
elseif not distincion_different and lleismo_different then
express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo")
express_style(false, "<<rural>> <<northern Spain>>, northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "distincion-lleismo", "distincion-seseo-lleismo")
else
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, most of <<Spain>>", "distincion-yeismo")
express_style("<<Latin America>>", "most of <<Latin America>>", "seseo-yeismo")
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<rural>> <<northern Spain>>", "distincion-lleismo")
express_style("<<Latin America>>", "northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "seseo-lleismo")
end
if need_rioplat then
if lleismo_different then
local hidden_tag = distincion_different and "<<Latin America>>" or false
if sheismo_different then
if not need_quito then
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs, elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo-quito")
else
express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs", "quito", "seseo-quito")
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo")
end
else
express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo")
end
else
local hidden_tag = distincion_different and "<<Latin America>>, <<Philippines>>" or false
if sheismo_different then
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo")
else
express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo")
end
end
end
if need_yucatan then
local hidden_tag = distincion_different and (lleismo_different and "<<Latin America>>" or "<<Latin America>>, <<Philippines>>") or false
express_style(hidden_tag, "northern <<Mexico>>, <<Yucatán>>, <<Central America>> (except <<Panama>>)", "yucatan")
end
-- If only one style group, don't indicate the style.
-- Not clear we want this in reality.
--if #ret.expressed_styles == 1 then
-- ret.expressed_styles[1].tag = false
-- if #ret.expressed_styles[1].styles == 1 then
-- ret.expressed_styles[1].styles[1].tag = false
-- end
--end
return ret
end
local function format_all_styles(expressed_styles, format_style)
for i, style_group in ipairs(expressed_styles) do
if #style_group.styles == 1 then
style_group.formatted, style_group.formatted_len =
format_style(style_group.styles[1].tag, style_group.styles[1], i == 1)
else
style_group.formatted, style_group.formatted_len =
format_style(style_group.tag, style_group.styles[1], i == 1)
for j, style in ipairs(style_group.styles) do
style.formatted, style.formatted_len =
format_style(style.tag, style, i == 1 and j == 1)
end
end
end
local maxlen = 0
for i, style_group in ipairs(expressed_styles) do
local this_len = style_group.formatted_len
if #style_group.styles > 1 then
for _, style in ipairs(style_group.styles) do
this_len = math.max(this_len, style.formatted_len)
end
end
maxlen = math.max(maxlen, this_len)
end
local lines = {}
local need_major_hack = false
for i, style_group in ipairs(expressed_styles) do
if #style_group.styles == 1 then
table.insert(lines, style_group.formatted)
need_major_hack = false
else
local inline = '\n<div class="vsShow" style="display:none">\n' .. style_group.formatted .. "</div>"
local full_prons = {}
for _, style in ipairs(style_group.styles) do
table.insert(full_prons, style.formatted)
end
local full = '\n<div class="vsHide">\n' .. table.concat(full_prons, "\n") .. "</div>"
local em_length = math.floor(maxlen * 0.68) -- from [[Module:grc-pronunciation]]
table.insert(lines, '<div class="vsSwitcher" data-toggle-category="pronunciations" style="width: ' .. em_length .. 'em; max-width:100%;"><span class="vsToggleElement" style="float: right;"> </span>' .. inline .. full .. "</div>")
need_major_hack = true
end
end
-- major hack to get bullets working on the next line after a div box
return table.concat(lines, "\n") .. (need_major_hack and "\n<span></span>" or "")
end
local function dodialect_pronun(args, ret, dialect)
ret.pronun[dialect] = {}
for i, term in ipairs(args.terms) do
local phonemic, phonetic, differences
if term.raw then
phonemic = term.raw_phonemic
phonetic = term.raw_phonetic
differences = construct_default_differences(dialect)
else
phonemic = export.IPA(term.term, dialect, false)
phonetic = export.IPA(term.term, dialect, true)
differences = phonemic.differences
phonemic = phonemic.text
phonetic = phonetic.text
end
ret.pronun[dialect][i] = {
raw = term.raw,
phonemic = phonemic,
phonetic = phonetic,
refs = term.refs,
q = term.q,
qq = term.qq,
a = term.a,
aa = term.aa,
differences = differences,
}
end
end
local function generate_pronun(args)
local function this_dodialect_pronun(ret, dialect)
dodialect_pronun(args, ret, dialect)
end
local ret = express_all_styles(args.style, this_dodialect_pronun)
local function format_style(tag, expressed_style, is_first)
local pronunciations = {}
local formatted_pronuns = {}
local function ins(formatted_part)
table.insert(formatted_pronuns, formatted_part)
end
-- Loop through each pronunciation. For each one, add the phonemic and phonetic versions to `pronunciations`,
-- for formatting by [[Module:IPA]], and also create an approximation of the formatted version so that we can
-- compute the appropriate width of the HTML switcher div box that holds the different per-dialect variants.
-- NOTE: The code below constructs the formatted approximation out-of-order in some cases but that doesn't
-- currently matter because we assume all characters have the same width. If we change the width computation
-- in a way that requires the correct order, we need changes to the code below.
for j, pronun in ipairs(expressed_style.pronun) do
-- Add tag to right accent qualifiers if last one
local aas = pronun.aa
if j == #expressed_style.pronun and tag then
if aas then
aas = m_table.deepCopy(aas)
table.insert(aas, tag)
else
aas = {tag}
end
end
local first_pronun = #pronunciations + 1
if not pronun.phonemic and not pronun.phonetic then
error("Internal error: Saw neither phonemic nor phonetic pronunciation")
end
if pronun.phonemic then -- missing if 'raw:[...]' given
-- don't display syllable division markers in phonemic
local slash_pron = "/" .. pronun.phonemic:gsub("%.", "") .. "/"
table.insert(pronunciations, {
pron = slash_pron,
})
ins(slash_pron)
end
if pronun.phonetic then -- missing if 'raw:/.../' given
local bracket_pron = "[" .. pronun.phonetic .. "]"
table.insert(pronunciations, {
pron = bracket_pron,
})
ins(bracket_pron)
end
local last_pronun = #pronunciations
if pronun.q then
pronunciations[first_pronun].q = pronun.q
end
if pronun.a then
pronunciations[first_pronun].a = pronun.a
end
if j > 1 then
pronunciations[first_pronun].separator = ", "
ins(", ")
end
if pronun.qq then
pronunciations[last_pronun].qq = pronun.qq
end
if aas then
pronunciations[last_pronun].aa = aas
end
if pronun.q or pronun.qq or pronun.a or aas then
-- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call
-- to textual_len() removes all HTML and reduces links.
ins(require(decorations_module).format_decorations {
lang = lang,
text = "",
q = pronun.q,
qq = pronun.qq,
a = pronun.a,
aa = aas,
})
end
if pronun.refs then
pronunciations[last_pronun].refs = pronun.refs
-- Approximate the reference using a footnote notation. This will be slightly inaccurate if there are
-- more than nine references but that is rare.
ins(string.rep("[1]", #pronun.refs))
end
if first_pronun ~= last_pronun then
pronunciations[last_pronun].separator = " "
ins(" ")
end
end
local bullet = string.rep("*", args.bullets) .. " "
-- Here we construct the formatted line in `formatted`, and also try to construct the equivalent without HTML
-- and wiki markup in `formatted_for_len`, so we can compute the approximate textual length for use in sizing
-- the toggle box with the "more" button on the right.
local pre = is_first and args.pre and args.pre .. " " or ""
local post = is_first and args.post and " " .. args.post or ""
local formatted = bullet .. pre ..
m_IPA.format_IPA_full { lang = lang, items = pronunciations, separator = "" } .. post
local formatted_for_len = bullet .. pre .. "IPA(key): " .. table.concat(formatted_pronuns) .. post
return formatted, textual_len(formatted_for_len)
end
ret.text = format_all_styles(ret.expressed_styles, format_style)
return ret
end
local function parse_respelling(respelling, pagename, parse_err)
local raw_respelling = respelling:match("^raw:(.*)$")
if raw_respelling then
local raw_phonemic, raw_phonetic = raw_respelling:match("^/(.*)/ %[(.*)%]$")
if not raw_phonemic then
raw_phonemic = raw_respelling:match("^/(.*)/$")
end
if not raw_phonemic then
raw_phonetic = raw_respelling:match("^%[(.*)%]$")
end
if not raw_phonemic and not raw_phonetic then
parse_err(("Unable to parse raw respelling '%s', should be one of /.../, [...] or /.../ [...]")
:format(raw_respelling))
end
return {
raw = true,
raw_phonemic = raw_phonemic,
raw_phonetic = raw_phonetic,
}
end
if respelling == "+" then
respelling = pagename
end
return {term = respelling}
end
-- External entry point for {{es-IPA}}.
function export.show(frame)
local params = {
[1] = {},
["pre"] = {},
["post"] = {},
["ref"] = {},
["style"] = {},
["bullets"] = {type = "number", default = 1},
}
local parargs = frame:getParent().args
local args = require(parameters_module).process(parargs, params)
local text = args[1] or mw.loadData("Module:headword/data").pagename
args.terms = {{term = text}}
local ret = generate_pronun(args)
return ret.text
end
-- Return the number of syllables of a phonemic representation, which should have syllable dividers in it but no
-- hyphens.
local function get_num_syl_from_phonemic(phonemic)
-- Maybe we should just count vowels instead of the below code.
phonemic = rsub(phonemic, "|", " ") -- remove IPA foot boundaries
local words = rsplit(phonemic, " +")
for i, word in ipairs(words) do
-- IPA stress marks are syllable divisions if between characters; otherwise just remove.
word = rsub(word, "(.)[ˌˈ](.)", "%1.%2")
word = rsub(word, "[ˌˈ]", "")
words[i] = word
end
-- There should be a syllable boundary between words.
phonemic = table.concat(words, ".")
return ulen(rsub(phonemic, "[^.]", "")) + 1
end
-- Get the rhyme by truncating everything up through the last stress mark + any following consonants, and remove
-- syllable boundary markers.
local function convert_phonemic_to_rhyme(phonemic)
-- NOTE: This works because the phonemic vowels are just [aeiou] possibly with diacritics that are separate
-- Unicode chars. If we want to handle things like ɛ or ɔ we need to add them to `vowel`.
return rsub(rsub(phonemic, ".*[ˌˈ]", ""), "^[^" .. vowel .. "]*", ""):gsub("%.", ""):gsub("t͡ʃ", "tʃ")
end
local function split_syllabified_spelling(spelling)
return rsplit(spelling, "%.")
end
-- "Align" syllabification to original spelling by matching character-by-character, allowing for extra syllable and
-- accent markers in the syllabification. If we encounter an extra syllable marker (.), we allow and keep it. If we
-- encounter an extra accent marker in the syllabification, we drop it. In any other case, we return nil indicating
-- the alignment failed.
local function align_syllabification_to_spelling(syllab, spelling)
local result = {}
local syll_chars = rsplit(decompose(syllab), "")
local spelling_chars = rsplit(decompose(spelling), "")
local i = 1
local j = 1
while i <= #syll_chars or j <= #spelling_chars do
local ci = syll_chars[i]
local cj = spelling_chars[j]
if ci == cj then
table.insert(result, ci)
i = i + 1
j = j + 1
elseif ci == "." then
table.insert(result, ci)
i = i + 1
elseif ci == AC or ci == GR or ci == CFLEX then
-- skip character
i = i + 1
else
-- non-matching character
return nil
end
end
if i <= #syll_chars or j <= #spelling_chars then
-- left-over characters on one side or the other
return nil
end
return unfc(table.concat(result))
end
local function generate_hyph_obj(term)
return {syllabification = term, hyph = split_syllabified_spelling(term)}
end
-- Word should already be decomposed.
local function word_has_vowels(word)
return rfind(word, V)
end
local function all_words_have_vowels(term)
local words = rsplit(decompose(term), "[ %-]")
for i, word in ipairs(words) do
-- Allow empty word; this occurs with prefixes and suffixes.
if word ~= "" and not word_has_vowels(word) then
return false
end
end
return true
end
local function should_generate_rhyme_from_respelling(term)
local words = rsplit(decompose(term), " +")
return #words == 1 and -- no if multiple words
not words[1]:find(".%-.") and -- no if word is composed of hyphenated parts (e.g. [[Austria-Hungría]])
not words[1]:find("%-$") and -- no if word is a prefix
not (words[1]:find("^%-") and words[1]:find(CFLEX)) and -- no if word is an unstressed suffix
word_has_vowels(words[1]) -- no if word has no vowels (e.g. a single letter)
end
local function should_generate_rhyme_from_ipa(ipa)
return not ipa:find("%s") and word_has_vowels(decompose(ipa))
end
local function dodialect_specified_rhymes(rhymes, hyphs, parsed_respellings, rhyme_ret, dialect)
rhyme_ret.pronun[dialect] = {}
for _, rhyme in ipairs(rhymes) do
local num_syl = rhyme.num_syl
local no_num_syl = false
-- If user explicitly gave the rhyme but didn't explicitly specify the number of syllables, try to take it from
-- the hyphenation.
if not num_syl then
num_syl = {}
for _, hyph in ipairs(hyphs) do
if should_generate_rhyme_from_respelling(hyph.syllabification) then
local this_num_syl = 1 + ulen(rsub(hyph.syllabification, "[^.]", ""))
m_table.insertIfNot(num_syl, this_num_syl)
else
no_num_syl = true
break
end
end
if no_num_syl or #num_syl == 0 then
num_syl = nil
end
end
-- If that fails and term is single-word, try to take it from the phonemic.
if not no_num_syl and not num_syl then
for _, parsed in ipairs(parsed_respellings) do
for dialect, pronun in pairs(parsed.pronun.pronun[dialect]) do
-- Check that pronun.phonemic exists (it may not if raw phonetic-only pronun is given).
if pronun.phonemic then
if not should_generate_rhyme_from_ipa(pronun.phonemic) then
no_num_syl = true
break
end
-- Count number of syllables by looking at syllable boundaries (including stress marks).
local this_num_syl = get_num_syl_from_phonemic(pronun.phonemic)
m_table.insertIfNot(num_syl, this_num_syl)
end
end
if no_num_syl then
break
end
end
if no_num_syl or #num_syl == 0 then
num_syl = nil
end
end
table.insert(rhyme_ret.pronun[dialect], {
rhyme = rhyme.rhyme,
num_syl = num_syl,
q = rhyme.q,
qq = rhyme.qq,
a = rhyme.a,
aa = rhyme.aa,
differences = construct_default_differences(dialect),
})
end
end
local q_qq_inline_modifier_spec = {
store = "insert-flattened",
type = "qualifier",
}
local a_aa_inline_modifier_spec = {
store = "insert-flattened",
type = "labels",
}
local ref_inline_modifier_spec = {
store = "insert-flattened",
item_dest = "refs",
type = "references",
}
-- Parse a pronunciation modifier in `arg`, the argument portion in an inline modifier (after the prefix), which
-- specifies a pronunciation property such as rhyme, hyphenation/syllabification, homophones or audio. The argument
-- can itself have inline modifiers, e.g. <audio:Foo.ogg<a:Colombia>>. The allowed inline modifiers are specified
-- by `param_mods` (of the format expected by `parse_inline_modifiers()`); in addition to any modifiers specified
-- there, the modifiers <q:...>, <qq:...>, <a:...>, <aa:...> and <ref:...> are always accepted (and can be repeated).
-- `generate_obj` and `parse_err` are like in `parse_inline_modifiers()` and specify respectively a function to
-- generate the object into which modifier properties are stored given the non-modifier part of the argument, and
-- a function to generate an error message (given the message). Normally, a comma-separated list of pronunciation
-- properties is accepted and parsed, where each element in the list can have its own inline modifiers and where
-- no spaces are allowed next to the commas in order for them to be recognized as separators. If `no_split_on_comma`
-- is given, only a single pronunciation property is accepted. In all cases, however, the return value is a list
-- of property objects (when `no_split_on_comma` is given, the return value is a one-element list).
local function parse_pron_modifier(arg, parse_err, generate_obj, param_mods, no_split_on_comma)
if arg:find("<") then
param_mods.q = q_qq_inline_modifier_spec
param_mods.qq = q_qq_inline_modifier_spec
param_mods.a = a_aa_inline_modifier_spec
param_mods.aa = a_aa_inline_modifier_spec
param_mods.ref = ref_inline_modifier_spec
local retval = require(parse_utilities_module).parse_inline_modifiers(arg, {
param_mods = param_mods,
generate_obj = generate_obj,
parse_err = parse_err,
splitchar = not no_split_on_comma and "," or nil,
})
if no_split_on_comma then
retval = {retval}
end
return retval
elseif no_split_on_comma then
return {generate_obj(arg)}
else
local retval = {}
for _, term in ipairs(split_on_comma(arg)) do
table.insert(retval, generate_obj(term))
end
return retval
end
end
local function parse_rhyme(arg, parse_err)
local function generate_obj(term)
return {rhyme = term}
end
local param_mods = {
s = {
item_dest = "num_syl",
type = "number",
sublist = true,
},
}
return parse_pron_modifier(arg, parse_err, generate_obj, param_mods)
end
local function parse_hyph(arg, parse_err)
-- None other than decorations
local param_mods = {}
return parse_pron_modifier(arg, parse_err, generate_hyph_obj, param_mods)
end
local function parse_homophone(arg, parse_err)
local function generate_obj(term)
return {term = term}
end
local param_mods = {
t = {
-- [[Module:links]] expects the gloss in "gloss".
item_dest = "gloss",
},
gloss = {},
-- No tr=, ts=, or sc=; doesn't make sense for Spanish.
pos = {},
alt = {},
lit = {},
id = {},
g = {
-- [[Module:links]] expects the genders in "genders".
item_dest = "genders",
sublist = true,
},
}
return parse_pron_modifier(arg, parse_err, generate_obj, param_mods)
end
local function generate_audio_obj(arg)
local file, caption = arg:match("^(.-)%s*#%s*(.*)$")
file = file or arg
return {file = file, caption = caption}
end
local function parse_audio(arg, parse_err)
local param_mods = {
IPA = {
sublist = true,
},
text = {},
t = {
item_dest = "gloss",
},
-- No tr=, ts=, or sc=; doesn't make sense for Spanish.
gloss = {},
pos = {},
-- No alt=; text= already goes in alt=.
lit = {},
-- No id=; text= already goes in alt= and isn't normally linked.
g = {
item_dest = "genders",
sublist = true,
},
bad = {},
}
-- Don't split on comma because some filenames have embedded commas not followed by a space
-- (typically followed by an underscore).
local retvals = parse_pron_modifier(arg, parse_err, generate_audio_obj, param_mods, "no split on comma")
local retval = retvals[1]
retval.lang = lang
local textobj = require(audio_module).construct_audio_textobj(retval)
retval.text = textobj
retval.gloss = nil
retval.pos = nil
retval.lit = nil
retval.genders = nil
return retval
end
-- External entry point for {{es-pr}}.
function export.show_pr(frame)
local params = {
[1] = {list = true},
["rhyme"] = {convert = parse_rhyme},
["hyph"] = {convert = parse_hyph},
["hmp"] = {convert = parse_homophone},
["audio"] = {list = true},
["pagename"] = {},
}
local parargs = frame:getParent().args
local args = require(parameters_module).process(parargs, params)
local pagename = args.pagename or mw.loadData(headword_data_module).pagename
-- Parse the arguments.
local respellings = #args[1] > 0 and args[1] or {"+"}
local parsed_respellings = {}
local overall_rhyme = args.rhyme
local overall_hyph = args.hyph
local overall_hmp = args.hmp
local overall_audio
if args.audio then
-- We can't specify parse_audio() as a `convert` function because it needs access to `pagename` (i.e. another
-- parameter).
overall_audio = {}
for i, audio in ipairs(args.audio) do
local function parse_err(msg)
error(("%s: parameter audio%s=%s"):format(msg, i == 1 and "" or i, audio))
end
local parsed_audio = parse_audio(audio, parse_err, pagename)
table.insert(overall_audio, parsed_audio)
end
end
for i, respelling in ipairs(respellings) do
if respelling:find("<") then
local param_mods = {
pre = { overall = true },
post = { overall = true },
style = { overall = true },
bullets = {
overall = true,
type = "number",
},
rhyme = {
overall = true,
store = "insert-flattened",
convert = parse_rhyme,
},
hyph = {
overall = true,
store = "insert-flattened",
convert = parse_hyph,
},
hmp = {
overall = true,
store = "insert-flattened",
convert = parse_homophone,
},
audio = {
overall = true,
store = "insert",
convert = function(arg, parse_err)
return parse_audio(arg, parse_err, pagename)
end,
},
ref = ref_inline_modifier_spec,
q = q_qq_inline_modifier_spec,
qq = q_qq_inline_modifier_spec,
a = a_aa_inline_modifier_spec,
aa = a_aa_inline_modifier_spec,
}
local parsed = require(parse_utilities_module).parse_inline_modifiers(respelling, {
paramname = i,
param_mods = param_mods,
generate_obj = function(term, parse_err)
return parse_respelling(term, pagename, parse_err)
end,
splitchar = ",",
outer_container = {
audio = {}, rhyme = {}, hyph = {}, hmp = {}
}
})
if not parsed.bullets then
parsed.bullets = 1
end
table.insert(parsed_respellings, parsed)
else
local termobjs = {}
local function parse_err(msg)
error(msg .. ": " .. i .. "=" .. respelling)
end
for _, term in ipairs(split_on_comma(respelling)) do
table.insert(termobjs, parse_respelling(term, pagename, parse_err))
end
table.insert(parsed_respellings, {
terms = termobjs,
audio = {},
rhyme = {},
hyph = {},
hmp = {},
bullets = 1,
})
end
end
if overall_hyph then
local hyphs = {}
for _, hyph in ipairs(overall_hyph) do
if hyph.syllabification == "+" then
hyph.syllabification = syllabify_from_spelling(pagename)
hyph.hyph = split_syllabified_spelling(hyph.syllabification)
elseif hyph.syllabification == "-" then
overall_hyph = {}
break
end
end
end
-- Loop over individual respellings, processing each.
for _, parsed in ipairs(parsed_respellings) do
parsed.pronun = generate_pronun(parsed)
local no_auto_rhyme = false
for _, term in ipairs(parsed.terms) do
if term.raw then
if not should_generate_rhyme_from_ipa(term.raw_phonemic or term.raw_phonetic) then
no_auto_rhyme = true
break
end
elseif not should_generate_rhyme_from_respelling(term.term) then
no_auto_rhyme = true
break
end
end
if #parsed.hyph == 0 then
if not overall_hyph and all_words_have_vowels(pagename) then
for _, term in ipairs(parsed.terms) do
if not term.raw then
local syllabification = syllabify_from_spelling(term.term)
local aligned_syll = align_syllabification_to_spelling(syllabification, pagename)
if aligned_syll then
m_table.insertIfNot(parsed.hyph, generate_hyph_obj(aligned_syll))
end
end
end
end
else
for _, hyph in ipairs(parsed.hyph) do
if hyph.syllabification == "+" then
hyph.syllabification = syllabify_from_spelling(pagename)
hyph.hyph = split_syllabified_spelling(hyph.syllabification)
elseif hyph.syllabification == "-" then
parsed.hyph = {}
break
end
end
end
-- Generate the rhymes.
local function dodialect_rhymes_from_pronun(rhyme_ret, dialect)
rhyme_ret.pronun[dialect] = {}
-- It's possible the pronunciation for a passed-in dialect was never generated. This happens e.g. with
-- {{es-pr|cebolla<style:seseo>}}. The initial call to generate_pronun() fails to generate a pronunciation
-- for the dialect 'distinction-yeismo' because the pronunciation of 'cebolla' differs between distincion
-- and seseo and so the seseo style restriction rules out generation of pronunciation for distincion
-- dialects (other than 'distincion-lleismo', which always gets generated so as to determine on which axes
-- the dialects differ). However, when generating the rhyme, it is based only on -olla, whose pronunciation
-- does not differ between distincion and seseo, but does differ between lleismo and yeismo, so it needs to
-- generate a yeismo-specific rhyme, and 'distincion-yeismo' is the representative dialect for yeismo in the
-- situation where distincion and seseo do not have distinct results (based on the following line in
-- express_all_styles()):
-- express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo")
-- In this case we need to generate the missing overall pronunciation ourselves since we need it to generate
-- the dialect-specific rhyme pronunciation.
if not parsed.pronun.pronun[dialect] then
dodialect_pronun(parsed, parsed.pronun, dialect)
end
for _, pronun in ipairs(parsed.pronun.pronun[dialect]) do
-- We should have already excluded multiword terms and terms without vowels from rhyme generation (see
-- `no_auto_rhyme` below). But make sure to check that pronun.phonemic exists (it may not if raw
-- phonetic-only pronun is given).
if pronun.phonemic then
-- Count number of syllables by looking at syllable boundaries (including stress marks).
local num_syl = get_num_syl_from_phonemic(pronun.phonemic)
-- Get the rhyme by truncating everything up through the last stress mark + any following
-- consonants, and remove syllable boundary markers.
local rhyme = convert_phonemic_to_rhyme(pronun.phonemic)
local saw_already = false
for _, existing in ipairs(rhyme_ret.pronun[dialect]) do
if existing.rhyme == rhyme then
saw_already = true
-- We already saw this rhyme but possibly with a different number of syllables,
-- e.g. if the user specified two pronunciations 'biología' (4 syllables) and
-- 'bi.ología' (5 syllables), both of which have the same rhyme /ia/.
m_table.insertIfNot(existing.num_syl, num_syl)
break
end
end
if not saw_already then
local rhyme_diffs = nil
if dialect == "distincion-lleismo" then
rhyme_diffs = {}
if rhyme:find("θ") then
rhyme_diffs.distincion_different = true
end
if rhyme:find("ʎ") then
rhyme_diffs.lleismo_different = true
if rhyme:find("ɟ") then
rhyme_diffs.need_quito = true
end
end
if rfind(rhyme, "[ʎɟ]") then
rhyme_diffs.sheismo_different = true
rhyme_diffs.need_rioplat = true
if rfind(rhyme, V .. "[ʎɟ]" .. V) then
rhyme_diffs.need_yucatan = true
end
end
end
table.insert(rhyme_ret.pronun[dialect], {
rhyme = rhyme,
num_syl = {num_syl},
differences = rhyme_diffs,
})
end
end
end
end
if #parsed.rhyme == 0 then
if overall_rhyme or no_auto_rhyme then
parsed.rhyme = nil
else
parsed.rhyme = express_all_styles(parsed.style, dodialect_rhymes_from_pronun)
end
else
local no_rhyme = false
for _, rhyme in ipairs(parsed.rhyme) do
if rhyme.rhyme == "-" then
no_rhyme = true
break
end
end
if no_rhyme then
parsed.rhyme = nil
else
local function this_dodialect(rhyme_ret, dialect)
return dodialect_specified_rhymes(parsed.rhyme, parsed.hyph, {parsed}, rhyme_ret, dialect)
end
parsed.rhyme = express_all_styles(parsed.style, this_dodialect)
end
end
end
if overall_rhyme then
local no_overall_rhyme = false
for _, orhyme in ipairs(overall_rhyme) do
if orhyme.rhyme == "-" then
no_overall_rhyme = true
break
end
end
if no_overall_rhyme then
overall_rhyme = nil
else
local all_hyphs
if overall_hyph then
all_hyphs = overall_hyph
else
all_hyphs = {}
for _, parsed in ipairs(parsed_respellings) do
for _, hyph in ipairs(parsed.hyph) do
m_table.insertIfNot(all_hyphs, hyph)
end
end
end
local function dodialect_overall_rhyme(rhyme_ret, dialect)
return dodialect_specified_rhymes(overall_rhyme, all_hyphs, parsed_respellings, rhyme_ret, dialect)
end
overall_rhyme = express_all_styles(parsed.style, dodialect_overall_rhyme)
end
end
-- If all sets of pronunciations have the same rhymes, display them only once at the bottom.
-- Otherwise, display rhymes beneath each set, indented.
local first_rhyme_ret
local all_rhyme_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_rhyme_ret = parsed.rhyme
elseif not m_table.deepEquals(first_rhyme_ret, parsed.rhyme) then
all_rhyme_sets_eq = false
break
end
end
local function format_rhyme(rhyme_ret, num_bullets)
local function format_rhyme_style(tag, expressed_style, is_first)
local pronunciations = {}
local rhymes = {}
for _, pronun in ipairs(expressed_style.pronun) do
table.insert(rhymes, pronun)
end
local data = {
lang = lang,
rhymes = rhymes,
aa = tag and {tag} or nil,
force_cat = force_cat,
}
local bullet = string.rep("*", num_bullets) .. " "
local formatted = bullet .. require(rhymes_module).format_rhymes(data)
local formatted_for_len_parts = {}
table.insert(formatted_for_len_parts, bullet .. "Rhymes: " .. (tag and "(" .. tag .. ") " or ""))
for j, pronun in ipairs(expressed_style.pronun) do
if j > 1 then
table.insert(formatted_for_len_parts, ", ")
end
if pronun.q or pronun.qq or pronun.a or pronun.aa then
-- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call
-- to textual_len() removes all HTML and reduces links.
table.insert(formatted_for_len_parts, require(decorations_module).format_decorations {
lang = lang,
text = "",
q = pronun.q,
qq = pronun.qq,
a = pronun.a,
aa = pronun.aa,
})
end
table.insert(formatted_for_len_parts, "-" .. pronun.rhyme)
end
return formatted, textual_len(table.concat(formatted_for_len_parts))
end
return format_all_styles(rhyme_ret.expressed_styles, format_rhyme_style)
end
-- If all sets of pronunciations have the same hyphenations, display them only once at the bottom.
-- Otherwise, display hyphenations beneath each set, indented.
local first_hyphs
local all_hyph_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_hyphs = parsed.hyph
elseif not m_table.deepEquals(first_hyphs, parsed.hyph) then
all_hyph_sets_eq = false
break
end
end
local function format_hyphenations(hyphs, num_bullets)
local hyphtext = require(hyphenation_module).format_hyphenations { lang = lang, hyphs = hyphs, caption = "Syllabification" }
return string.rep("*", num_bullets) .. " " .. hyphtext
end
-- If all sets of pronunciations have the same homophones, display them only once at the bottom.
-- Otherwise, display homophones beneath each set, indented.
local first_hmps
local all_hmp_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_hmps = parsed.hmp
elseif not m_table.deepEquals(first_hmps, parsed.hmp) then
all_hmp_sets_eq = false
break
end
end
local function format_homophones(hmps, num_bullets)
local hmptext = require(homophones_module).format_homophones { lang = lang, homophones = hmps }
return string.rep("*", num_bullets) .. " " .. hmptext
end
local function format_audio(audios, num_bullets)
local ret = {}
for i, audio in ipairs(audios) do
local text = require(audio_module).format_audio(audio)
table.insert(ret, string.rep("*", num_bullets) .. " " .. text)
end
return table.concat(ret, "\n")
end
local textparts = {}
local min_num_bullets = math.huge
for j, parsed in ipairs(parsed_respellings) do
if parsed.bullets < min_num_bullets then
min_num_bullets = parsed.bullets
end
if j > 1 then
table.insert(textparts, "\n")
end
table.insert(textparts, parsed.pronun.text)
if #parsed.audio > 0 then
table.insert(textparts, "\n")
-- If only one pronunciation set, add the audio with the same number of bullets, otherwise
-- indent audio by one more bullet.
table.insert(textparts, format_audio(parsed.audio,
#parsed_respellings == 1 and parsed.bullets or parsed.bullets + 1))
end
if not all_rhyme_sets_eq and parsed.rhyme then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(parsed.rhyme, parsed.bullets + 1))
end
if not all_hyph_sets_eq and #parsed.hyph > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(parsed.hyph, parsed.bullets + 1))
end
if not all_hmp_sets_eq and #parsed.hmp > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(parsed.hmp, parsed.bullets + 1))
end
end
if overall_audio and #overall_audio > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_audio(overall_audio, min_num_bullets))
end
if all_rhyme_sets_eq and first_rhyme_ret then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(first_rhyme_ret, min_num_bullets))
end
if overall_rhyme then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(overall_rhyme, min_num_bullets))
end
if all_hyph_sets_eq and #first_hyphs > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(first_hyphs, min_num_bullets))
end
if overall_hyph and #overall_hyph > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(overall_hyph, min_num_bullets))
end
if all_hmp_sets_eq and #first_hmps > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(first_hmps, min_num_bullets))
end
if overall_hmp and #overall_hmp > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(overall_hmp, min_num_bullets))
end
return table.concat(textparts)
end
return export
7qwnb8u8ekg7lzajzlk6jnkfpja905m
178041
178040
2026-09-23T05:08:56Z
Yivan000
4078
178041
Scribunto
text/plain
--[=[
This module implements the templates {{es-pr}} and {{es-IPA}}.
Author: Benwing2
]=]
local export = {}
local m_IPA = require("Module:IPA")
local m_str_utils = require("Module:string utilities")
local m_table = require("Module:table")
local audio_module = "Module:audio"
local decorations_module = "Module:decorations"
local headword_data_module = "Module:headword/data"
local homophones_module = "Module:homophones"
local hyphenation_module = "Module:hyphenation"
local labels_module = "Module:labels"
local links_module = "Module:links"
local parameters_module = "Module:parameters"
local parse_utilities_module = "Module:parse utilities"
local references_module = "Module:references"
local rhymes_module = "Module:rhymes"
local force_cat = false -- for testing
--[=[
FIXME:
1. Port latest changes to production module. [DONE]
2. Finish work on rhymes and hyphenation. [DONE]
3. Handle <hmp:...> for homophones. [DONE]
4. Don't add comma before phonetic IPA. [DONE]
5. Handle secondary stress, suffixes, etc. in syllabification. [DONE]
6. Need some changes to syllable splitting in consonant clusters. (e.g. 'cum‧min‧gto‧ni‧ta') [DONE]
7. Fix handling of references to correspond to Portuguese module. [DONE]
8. Propagate qualifiers on individual pronun terms to rhymes and hyph.
9. Support raw phonemic/phonetic pronunciations. [DONE]
10. Support overall audio. [DONE]
11. Keep th/ph/kh/gh/tz ([[Ertzaintza]]) together when syllabifying (but not bh due to [[subhumano]], [[subhistoria]], etc.). [DONE]
12. Support <q:...> and <qq:...> on audio. [DONE]
13. Support <a:...> and <aa:...> (using {{a|...}}, left and right) on terms, rhymes, hyphenation, homophones and
audio. [DONE]
14. Support # instead of ; as separator between audio file and gloss and make sure it works if gloss has embedded # or
;. [DONE]
15. Use parse_inline_modifiers() in [[Module:parse utilities]]. [DONE]
]=]
--[=[
About styles, dialects and isoglosses:
From the standpoint of pronunciation, a given dialect is defined by isoglosses, which specify differences
in the way of pronouncing certain phonemes. You can think of a dialect as a collection of isoglosses.
For example, one isogloss is "distinción" (pronouncing written ''s'' and ''c/z'' differently) vs. "seseo"
(pronouncing them the same). Another is "lleísmo" (pronouncing written ''ll'' and ''y'' differently) vs.
"yeísmo" (pronouncing them the same). The dominant pronunciation in Spain can be described as
distinción + yeísmo, while the pronunciation in rural northern Spain can be described as distinción + lleísmo
and the pronunciation across much of the Andes mountains, Paraguay, and the Philippines can be described as seseo + lleísmo.
Specifically, the following isoglosses are recognized (note, the isogloss specs as used in this module
dispense with written accents):
-- "distincion" = pronouncing ''s'' and ''c/z'' differently
-- "seseo" = pronouncing ''s'' and ''c/z'' the same
-- "lleismo" = pronouncing ''ll'' and ''y'' differently
-- "yeismo" = pronouncing ''ll'' and ''y'' the same
-- "rioplatense" = Rioplatense speech, i.e. seseo+yeismo with ''ll'' and ''y'' pronounced specially, and a
clear distinction between initial ''hi-'' vs. initial ''ll-/y-''
-- "sheismo" = a type of Rioplatense speech, characteristic of Buenos Aires, where ''ll'' and ''y'' are
pronounced as /ʃ/
-- "zheismo" = a type of Rioplatense speech, found outside of Buenos Aires, where ''ll'' and ''y'' are
pronounced as /ʒ/
-- "quito" = seseo + lleismo, but pronouncing ''ll'' as /ʒ/
-- "yucatan" = seseo + yeismo, intervocalic ''y'' is pronounced ''i'' and lost in contact with ''i'' or ''e''
These isoglosses can be combined to yield one of the following eight dialects:
-- "distincion-lleismo": distinción + lleísmo
-- "distincion-yeismo": distinción + yeísmo
-- "seseo-lleismo": seseo + lleísmo
-- "seseo-yeismo": seseo + yeísmo
-- "rioplatense-sheismo": Rioplatense with /ʃ/ (Buenos Aires)
-- "rioplatense-zheismo": Rioplatense with /ʒ/ (non-Buenos Aires)
-- "quito"
-- "yucatan"
A "style" here is a set of dialects that pronounce a given word in a given fashion. For example, if we are only
considering the distinción/seseo and lleísmo/yeísmo isoglosses, there are four conceivable dialects (all of
which in fact exist). However, for a given word, more than one dialect may pronounce it the same. For
example, a word like [[paz]] has a ''z'' but no ''ll'', and so there are only two possible pronunciations for
the four dialects. Here, the two styles are "Spain" and "Latin America". Correspondingly, a word like [[pollo]]
with an ''ll'' but no ''z'' has two styles, which can approximately be described as "most of Spain and Latin
America" vs. "rural northern Spain, Andes Mountains, Paraguay, Philippines".
A "style spec" (indicated by the style= parameter to {{es-IPA}}) restricts the output to certain styles.
A style spec can be one of the following:
1. An isogloss, e.g. "distincion", "rioplatense"; if specified, only styles containing this isogloss are output.
2. A negated isogloss, e.g. "-rioplatense".
3. An intersection of isoglosses ("A and B"), e.g. "distincion+lleismo". This can be used to restrict to specific
dialects.
4. A union of isoglosses ("A or B"), e.g. "distincion,zheismo". If both plus and comma are used, plus takes
precedence, e.g. "seseo+lleismo,zheismo" means either the "seseo+lleismo" dialect or the "rioplatense-zheismo"
dialect.
An example where the style= parameter might be used is with the word [[bluetooth]], which has one pronunciation
in Spain/distinción (respelled "blutuz") but another in Latin America/seseo (respelled "blutud"). This might be
represented using {{es-pr}} as {{es-pr|blutuz<style:distincion>|blutud<style:seseo>}}.
]=]
local lang = require("Module:languages").getByCode("es")
local decompose = require("Module:es-common").decompose
local u = m_str_utils.char
local rfind = m_str_utils.find
local rsubn = m_str_utils.gsub
local rsplit = m_str_utils.split
local ulower = m_str_utils.lower
local ulen = m_str_utils.len
local unfd = mw.ustring.toNFD
local unfc = mw.ustring.toNFC
local AC = u(0x0301) -- acute = ́
local GR = u(0x0300) -- grave = ̀
local CFLEX = u(0x0302) -- circumflex = ̂
local TILDE = u(0x0303) -- tilde = ̃
local SYLDIV = u(0xFFF0) -- used to represent a user-specific syllable divider (.) so we won't change it
local vowel = "aeiouüyAEIOUÜY" -- vowel; include y so we get single-word y correct and for syllabifying from spelling
local V = "[" .. vowel .. "]" -- vowel class
local accent = AC .. GR .. CFLEX
local accent_c = "[" .. accent .. "]"
local stress = AC .. GR
local stress_c = "[" .. AC .. GR .. "]"
local ipa_stress = "ˈˌ"
local ipa_stress_c = "[" .. ipa_stress .. "]"
local sylsep = "%-." .. SYLDIV -- hyphen included for syllabifying from spelling
local sylsep_c = "[" .. sylsep .. "]"
local wordsep = "# "
local separator_not_wordsep = accent .. ipa_stress .. sylsep
local separator = separator_not_wordsep .. wordsep
local separator_c = "[" .. separator .. "]"
local C = "[^" .. vowel .. separator .. "]" -- consonant class including h
local C_NOT_H = "[^" .. vowel .. separator .. "h]" -- consonant class not including h
local C_OR_WORDSEP = "[^" .. vowel .. separator_not_wordsep .. "]" -- consonant class including h, or word separator
local T = "[^" .. vowel .. "lrɾjw" .. separator .. "]" -- obstruent or nasal
local unstressed_words = m_table.listToSet({
"el", "la", "los", "las", -- definite articles
"un", -- single-syllable indefinite articles
"me", "te", "se", "lo", "le", "nos", "os", "les", -- unstressed object pronouns
"mi", "mis", "tu", "tus", "su", "sus", -- unstressed possessive pronouns
"que", "si", -- subordinating conjunctions
"y", "e", "o", "u", "mas", -- coordinating conjunctions
"de", "del", "a", "al", -- basic prepositions + combinations with articles
"por", "en", "con", -- other prepositions
})
-- version of rsubn() that discards all but the first return value
local function rsub(term, foo, bar)
local retval = rsubn(term, foo, bar)
return retval
end
-- version of rsubn() that returns a 2nd argument boolean indicating whether
-- a substitution was made.
local function rsubb(term, foo, bar)
local retval, nsubs = rsubn(term, foo, bar)
return retval, nsubs > 0
end
-- apply rsub() repeatedly until no change
local function rsub_repeatedly(term, foo, bar)
while true do
local new_term = rsub(term, foo, bar)
if new_term == term then
return term
end
term = new_term
end
end
local function split_on_comma(term)
if not term then
return nil
end
if term:find(",%s") then
return require(parse_utilities_module).split_on_comma(term)
elseif term:find(",") then
return rsplit(term, ",")
else
return {term}
end
end
-- Remove any HTML from the formatted text and resolve links, since the extra characters don't contribute to the
-- displayed length.
local function convert_to_raw_text(text)
text = rsub(text, "<.->", "")
if text:find("%[%[") then
text = require(links_module).remove_links(text)
end
return text
end
-- Return the approximate displayed length in characters.
local function textual_len(text)
return ulen(convert_to_raw_text(text))
end
local function construct_default_differences(dialect)
if dialect == "distincion-lleismo" then
return {
distincion_different = false,
lleismo_different = false,
sheismo_different = false,
need_rioplat = false,
need_quito = false,
need_yucatan = false,
}
end
return nil
end
-- Main syllable-division algorithm. Can be called either directly on spelling (when hyphenating) or after
-- non-trivial processing of respelling in the direction of pronunciation (when generating pronunciation).
local function syllabify_from_spelling_or_pronun(text, is_spelling)
-- Part 1: Divide before the last consonant in a cluster of consonants between vowels (but don't divide a VhV
-- sequence; [[prohibir]] should be prohi.bir). Then move the syllable division marker leftwards over clusters that
-- can form onsets.
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*)(" .. C_NOT_H .. V .. ")", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. C .. "+)(" .. C .. V .. ")", "%1.%2")
-- Puerto Rico + most of Spain divide tl as t.l. Mexico and the Canary Islands have .tl. Unclear what other regions
-- do. Here we choose to go with .tl. See https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf
-- and https://www.spanishdict.com/guide/spanish-syllables-and-syllabification-rules.
-- NOTE: When run on pronun, we have already eliminated c and v, but not when run on spelling.
-- When run on pronun, don't include r, which at this point represents the trill.
local cluster_r = is_spelling and "rɾ" or "ɾ"
-- Don't divide Cl or Cr where C is a stop or fricative, except for dl.
text = rsub(text, "([pbfvkctg])%.([l" .. cluster_r .. "])", ".%1%2")
text = text:gsub("d%.([" .. cluster_r .. "])", ".d%1")
-- Don't divide ch, sh, ph, th, dh, fh, kh or gh. Do allow bh to be divided ([[subhumano]], [[subhúmedo]], etc.).
text = rsub(text, "([csptdfkg])%.h", ".%1h")
-- Don't divide ll or rr.
text = rsub(text, "([lr])%.%1", ".%1%1")
-- Don't divide tz ([[Ertzaintza]], [[quetzal]], [[hertziano]] and other words of Basque, Nahuatl and German
-- origin).
text = rsub(text, "t%.z", ".tz")
-- Per https://catalog.ldc.upenn.edu/docs/LDC2019S07/Syllabification_Rules_in_Spanish.pdf, tl at the end of a word
-- (as in nahuatl, Popocatepetl etc.) is divided .tl from the previous vowel.
if is_spelling then
text = text:gsub("([^. %-])tl$", "%1.tl")
text = text:gsub("([^. %-])(tl[ %-])", "%1.%2")
else
text = text:gsub("([^.#])tl#", "%1.tl")
end
-- Part 2: Divide hiatuses. Any aeo, or stressed iuüy, should be syllabically divided from a following aeo or
-- stressed iuüy. Also divide ii and uu sequences ([[antiincendios]], [[shiita]], [[vacuum]]). Note that words with
-- ii or uu next to a vowel (e.g. [[hawaiiano]]) will not make it to this point unchanged; the i or u adjacent to
-- a vowel (or the second one if both are adjacent to vowels) will get converted to a consonant symbol (temporarily
-- when syllabifying spelling).
text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?[aeo])", "%1.%2")
text = rsub_repeatedly(text, "([aeoAEO]" .. accent_c .. "*)(h?" .. V .. stress_c .. ")", "%1.%2")
text = rsub(text, "([iuüyIUÜY]" .. stress_c .. ")(h?[aeo])", "%1.%2")
text = rsub_repeatedly(text, "([iuüyIUÜY]" .. stress_c .. ")(h?" .. V .. stress_c .. ")", "%1.%2")
text = rsub_repeatedly(text, "([iI]" .. accent_c .. "*)(h?i)", "%1.%2")
text = rsub_repeatedly(text, "([uU]" .. accent_c .. "*)(h?u)", "%1.%2")
return text
end
local function syllabify_from_spelling(text)
text = decompose(text)
-- start at FFF1 because FFF0 is used for SYLDIV
-- Temporary replacements for characters we want treated as default consonants. The C and related consonant regexes
-- treat all unknown characters as consonants.
local TEMP_I = u(0xFFF1)
local TEMP_U = u(0xFFF2)
local TEMP_Y_CONS = u(0xFFF3)
local TEMP_QU = u(0xFFF4)
local TEMP_QU_CAPS = u(0xFFF5)
local TEMP_GU = u(0xFFF6)
local TEMP_GU_CAPS = u(0xFFF7)
local TEMP_H = u(0xFFF8)
-- Change user-specified . into SYLDIV so we don't shuffle it around when dividing into syllables.
text = text:gsub("%.", SYLDIV)
text = rsub(text, "y(" .. V .. ")", TEMP_Y_CONS .. "%1")
-- We don't want to break -sh- except in desh-, e.g. [[deshuesar]], [[deshonra]], [[deshecho]]. Normally, -sh- is
-- automatically preserved, so we replace the h with a temporary symbol to avoid this.
text = text:gsub("^([Dd]es)h", "%1" .. TEMP_H)
text = text:gsub("([ %-][Dd]es)h", "%1" .. TEMP_H)
-- qu mostly handled correctly automatically, but not in quietud
text = rsub(text, "qu(" .. V .. ")", TEMP_QU .. "%1")
text = rsub(text, "Qu(" .. V .. ")", TEMP_QU_CAPS .. "%1")
text = rsub(text, "gu(" .. V .. ")", TEMP_GU .. "%1")
text = rsub(text, "Gu(" .. V .. ")", TEMP_GU_CAPS .. "%1")
local vowel_to_glide = { ["i"] = TEMP_I, ["u"] = TEMP_U }
-- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]],
-- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]],
-- etc. When we do this we need to help the syllabification particularly of words with -hiV- and -huV- in them,
-- otherwise we get e.g. 'an.tih.ie.lo' because we converted the i following the h to a consonant. Add .* at the
-- beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano.
text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*)(h?)([iu])(" .. V .. ")",
function (v1, h, iu, v2) return v1 .. "." .. h .. vowel_to_glide[iu] .. v2 end
)
text = syllabify_from_spelling_or_pronun(text, "is spelling")
text = text:gsub(SYLDIV, ".")
text = text:gsub(TEMP_I, "i")
text = text:gsub(TEMP_U, "u")
text = text:gsub(TEMP_Y_CONS, "y")
text = text:gsub(TEMP_QU, "qu")
text = text:gsub(TEMP_QU_CAPS, "Qu")
text = text:gsub(TEMP_GU, "gu")
text = text:gsub(TEMP_GU_CAPS, "Gu")
text = text:gsub(TEMP_H, "h")
text = unfc(text)
-- No qualifiers from dialect tags because we assume all dialects hyphenate the same way.
-- FIXME: There are region-specific ways of hyphenating -tl-. See above. We don't currently handle this properly.
return text
end
-- Generate the IPA of a given respelling, where a respelling is the representation of the pronunciation of a given
-- Spanish term using Spanish spelling conventions (augmented in a few cases with extra conventions such as 'sh' for
-- /ʃ/).
-- ɟ and ĉ are used internally to represent [ʝ⁓ɟ͡ʝ] and [t͡ʃ]
--
function export.IPA(text, dialect, phonetic)
local distincion = dialect == "distincion-lleismo" or dialect == "distincion-yeismo"
local lleismo = dialect == "distincion-lleismo" or dialect == "seseo-lleismo" or dialect == "quito"
local rioplat = dialect == "rioplatense-sheismo" or dialect == "rioplatense-zheismo"
local sheismo = dialect == "rioplatense-sheismo"
local quito = dialect == "quito"
local yucatan = dialect == "yucatan"
local distincion_different = false
local lleismo_different = false
local need_rioplat = false
local need_quito = false
local need_yucatan = false
local initial_hi = false
local sheismo_different = false
-- start at FFF1 because FFF0 is used for SYLDIV
local TEMP_Y = u(0xFFF1)
local TEMP_W = u(0xFFF2)
text = ulower(text or mw.loadData("Module:headword/data").pagename)
-- decompose everything but ç, ñ and ü
text = decompose(text)
-- convert commas and en/en dashes to IPA foot boundaries
text = rsub(text, "%s*[,–—]%s*", " | ")
-- question mark or exclamation point in the middle of a sentence -> IPA foot boundary
text = rsub(text, "([^%s])%s*[¡!¿?]%s*([^%s])", "%1 | %2")
-- canonicalize multiple spaces and remove leading and trailing spaces
local function canon_spaces(text)
text = rsub(text, "%s+", " ")
text = rsub(text, "^ ", "")
text = rsub(text, " $", "")
return text
end
text = canon_spaces(text)
-- Make prefixes unstressed unless they have an explicit stress marker; also make certain
-- monosyllabic words (e.g. [[el]], [[la]], [[de]], [[en]], etc.) without stress marks be
-- unstressed.
local words = rsplit(text, " ")
for i, word in ipairs(words) do
if rfind(word, "%-$") and not rfind(word, accent_c) or unstressed_words[word] then
-- add CFLEX to the last vowel not the first one, or we will mess up 'que' by
-- adding the CFLEX after the 'u'
words[i] = rsub(word, "^(.*" .. V .. ")", "%1" .. CFLEX)
end
end
text = table.concat(words, " ")
-- Convert hyphens to spaces, to handle [[Austria-Hungría]], [[franco-italiano]], etc.
text = rsub(text, "%-", " ")
-- canonicalize multiple spaces again, which may have been introduced by hyphens
text = canon_spaces(text)
-- now eliminate punctuation
text = rsub(text, "[¡!¿?']", "")
-- put # at word beginning and end and double ## at text/foot boundary beginning/end
text = rsub(text, " | ", "# | #")
text = "##" .. rsub(text, " ", "# #") .. "##"
--determining whether "y" is a consonant or a vowel
text = rsub(text, "y(" .. V .. ")", "ɟ%1") -- not the real sound
-- word-final -ay/-ey/-oy/-uy is stressed whereas word-final -ai/-ei/-oi/-ui is not; in addition,
-- word-final -uy is /uj/ whereas word-final -ui is /wi/ (e.g. [[muy]] vs. [[fui]])
text = rsub(text, "([aeou])y#", "%1" .. TEMP_Y .. "#") -- a temporary symbol; replaced with i below
text = rsub(text, "y", "i")
-- handle certain combinations; sh handling needs to go before x handling to avoid issues with [[exhausto]]
text = rsub(text, "ch", "ĉ") --not the real sound
-- We want to keep desh- ([[deshuesar]]) as-is. Converting to des- won't work because we want it syllabified as
-- 'des.we.saɾ' not #'de.swe.saɾ' (cf. [[desuelo]] /de.swe.lo/ from [[desolar]]).
text = rsub(text, "#desh", "!") --temporary symbol
text = rsub(text, "sh", "ʃ")
text = rsub(text, "!", "#desh") --restore
text = rsub(text, "#[ckp]([st])", "#%1") -- [[ctónico]], [[psicología]], [[pterodáctilo]]
--x
text = rsub(text, "#x", "#s") -- xenofobia, xilófono, etc.
text = rsub(text, "x", "ks")
--c, g, q
text = rsub(text, "c([ie])", (distincion and "θ" or "z") .. "%1") -- not the real LatAm sound
text = rsub(text, "g([ie])", "x%1") -- must happen after handling of x above
text = rsub(text, "gu([ie])", "g%1")
text = rsub(text, "gü([ie])", "gu%1")
-- following must happen before stress assignment; [[branding]] has initial stress like 'brandin'
text = rsub(text, "ng([^aeiouüwhlr])", "n%1") -- [[Bangkok]], [[ángstrom]], [[branding]]
text = rsub(text, "qu([ie])", "k%1")
text = rsub(text, "ü", "u") -- [[Düsseldorf]], [[hübnerita]], obsolete [[freqüentemente]], etc.
text = rsub(text, "q", "k") -- [[quark]], [[Qatar]], [[burqa]], [[Iraq]], etc.
text = rsub(text, "[zç]", distincion and "θ" or "z") -- not the real LatAm sound; "ç" became "z" in 1726
if rfind(text, "[θz]") then
distincion_different = true
end
-- map various consonants to their phoneme equivalent
text = rsub(text, "[cjñrv]", {["c"]="k", ["j"]="x", ["ñ"]="ɲ", ["r"]="ɾ", ["v"]="b" })
-- handle word- and syllable-initial hiV ([[hielo]], [[enhiesto]], [[deshielo]], ...)
local word_initial_hi, syl_initial_hi
text, word_initial_hi = rsubb(text, "#h?i(" .. V .. ")", rioplat and "#j%1" or "#ɟ%1")
text, syl_initial_hi = rsubb(text, "(" .. C .. sylsep_c .. "*)hi(" .. V .. ")", rioplat and "%1j%2" or "%1ɟ%2")
initial_hi = word_initial_hi or syl_initial_hi
-- handle word- and syllable-initial huV ([[huevo]], [[deshuesar]])
text = rsubb(text, "(" .. C_OR_WORDSEP .. sylsep_c .. "*)hu(" .. V .. ")", "%1" .. TEMP_W .. "%2")
-- handle double consonants that have a pronunciation different from their single equivalents
-- double l
lleismo_different = rfind(text, "ll")
need_quito = lleismo_different and rfind(text, "ɟ")
text = rsub(text, "ll", lleismo and "ʎ" or "ɟ")
-- handle intervocalic -y-
need_yucatan = rfind(text, V .. accent_c .. "*" .. sylsep_c .. "*[ʎɟ]" .. V)
if yucatan then
text = rsub_repeatedly(text, "([ei]" .. accent_c .. "*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .. "*" .. sylsep_c .. "*)ɟ" .. "([ei])", "%1.%2")
text = rsub_repeatedly(text, "(" .. V .. accent_c .."*" .. sylsep_c .. "*)ɟ(" .. V .. ")", "%1i%2")
end
-- trill in #r, lr ([[alrededor]], [[malrotar]]), nr ([[enriquecer]], [[sonrisa]], etc.), sr ([[Israel]],
-- [[desregular]], etc.), zr ([[Azrael]], [[cruzrojista]]), rr
text = rsub(text, "ɾɾ", "r")
text = rsub(text, "([#lnszθ])ɾ", "%1r")
-- double n (e.g. [[[ennoblecer]])
text = rsub(text, "nn", "N")
-- double b (e.g. [[subbase]])
text = rsub(text, "bb", "B")
-- reduce any remaining double consonants ([[Addis Abeba]], [[cappa]], [[descender]] in Latin America ...);
-- do this before handling of -nm- e.g. in [[inmigración]], which generates a double consonant, and do this
-- before voicing stops before obstruents, to avoid problems with [[cappa]] and [[crackear]]
text = rsub(text, "(" .. C .. ")%1", "%1")
-- also reduce sz (Latin American in [[fascinante]], etc.)
text = rsub(text, "sz", "s")
-- restore double n, b
text = rsub(text, "N", "nn")
text = rsub(text, "B", "bb")
-- voiceless stop to voiced before obstruent or nasal; but intercept -ts-, -tz-
local voice_stop = { ["p"] = "b", ["t"] = "d", ["k"] = "g" }
text = rsub(text, "t(" .. separator_c .. "*[szθ])", "!%1") -- temporary symbol
text = rsub(text, "([ptk])(" .. separator_c .. "*" .. T .. ")",
function(stop, after) return voice_stop[stop] .. after end)
text = rsub(text, "!", "t")
text = rsub(text, "n([# .]*[bpm])", "m%1")
-- remove silent h before syllable division
text = rsub(text, "h", "")
-- convert i/u between vowels to glide
local vowel_to_glide = { ["i"] = "j", ["u"] = "w" }
-- i and u between vowels -> consonant-like substitutions: [[paranoia]], [[baiano]], [[abreuense]], [[alauita]],
-- [[Malaui]], etc.; also with h, as in [[marihuana]], [[parihuela]], [[antihielo]], [[pelluhuano]], [[náhuatl]],
-- etc. Add .* at the beginning so we go right-to-left, in the case of [[hawaiiano]] -> ha.wai.iano.
text = rsub_repeatedly(text, "(.*" .. V .. accent_c .. "*h?)([iu])(" .. V .. ")",
function (v1, iu, v2) return v1 .. vowel_to_glide[iu] .. v2 end
)
--syllable division
text = syllabify_from_spelling_or_pronun(text, false)
--diphthongs; do not include TEMP_Y here
text = rsub(text, "i([aeou])", "j%1")
text = rsub(text, "u([aeio])", "w%1")
local accent_to_stress_mark = { [AC] = "ˈ", [GR] = "ˌ", [CFLEX] = "" }
local function accent_word(word, syllables)
-- Now stress the word. If any accent exists in the word (including ^ indicating an unaccented word),
-- put the stress mark(s) at the beginning of the indicated syllable(s). Otherwise, apply the default
-- stress rule.
if rfind(word, accent_c) then
for i = 1, #syllables do
syllables[i] = rsub(syllables[i], "^(.*)(" .. accent_c .. ")(.*)$",
function(pre, accent, post) return accent_to_stress_mark[accent] .. pre .. post end
)
end
else
-- Default stress rule. Words without vowels (e.g. IPA foot boundaries) don't get stress.
if #syllables > 1 and (rfind(word, "[^" .. vowel .. "ns#]#") or rfind(word, C .. "[ns]#")) or #syllables == 1 and rfind(word, V) then
syllables[#syllables] = "ˈ" .. syllables[#syllables]
elseif #syllables > 1 then
syllables[#syllables - 1] = "ˈ" .. syllables[#syllables - 1]
end
end
end
local words = rsplit(text, " ")
for j, word in ipairs(words) do
-- accentuation
local syllables = rsplit(word, "%.")
if rfind(word, "men%.te#") then
local mente_syllables
-- Words ends in -mente (converted above to ménte); add a stress to the preceding portion
-- (e.g. [[agriamente]] -> 'ágriaménte') unless already stressed (e.g. [[rápidamente]]).
-- It will be converted to secondary stress further below. Essentially, we rip the word apart
-- into two words ('mente' and the preceding portion) and stress each one independently.
mente_syllables = {}
mente_syllables[2] = table.remove(syllables)
mente_syllables[1] = table.remove(syllables)
accent_word(table.concat(syllables, "."), syllables)
accent_word(table.concat(mente_syllables, "."), mente_syllables)
table.insert(syllables, mente_syllables[1])
table.insert(syllables, mente_syllables[2])
else
accent_word(word, syllables)
end
-- Vowels are nasalized if followed by nasal in same syllable.
if phonetic then
for i = 1, #syllables do
-- first check for two vowels (veinte)
syllables[i] = rsub(syllables[i], "(" .. V .. ")(" .. V .. ")([mnɲ])",
"%1" .. TILDE .. "%2" .. TILDE .. "%3")
-- then for one vowel
syllables[i] = rsub(syllables[i], "(" .. V .. ")([mnɲ])", "%1" .. TILDE .. "%2")
end
end
-- Reconstruct the word.
words[j] = table.concat(syllables, ".")
end
text = table.concat(words, " ")
text = rsub(text, TEMP_Y, "i") --final -ay/-ey/-oy/-uy
text = rsub(text, "z", "s") --real sound of LatAm Z
-- suppress syllable mark before IPA stress indicator
text = rsub(text, "%.(" .. ipa_stress_c .. ")", "%1")
--make all primary stresses but the last one be secondary
text = rsub_repeatedly(text, "ˈ(.+)ˈ", "ˌ%1ˈ")
if (not initial_hi and rfind(text, "[ʎɟ]")) or (rfind(text, sylsep_c .. "[ʎɟ]")) then
sheismo_different = true
end
if rioplat then
if not initial_hi then
if sheismo then
text = rsub(text, "ɟ", "ʃ")
else
text = rsub(text, "ɟ", "ʒ")
end
else
if sheismo then
text = rsub(text, sylsep_c .. "(ɟ)", "ʃ")
else
text = rsub(text, sylsep_c .. "(ɟ)", "ʒ")
end
end
end
if quito then text = rsub(text, "ʎ", "ʒ") end
--phonetic transcription
if phonetic then
-- θ, s, f before voiced consonants
local voiced = "mnɲbdɟgʎ" .. TEMP_W
local r = "ɾr"
local tovoiced = {
["θ"] = "θ̬",
["s"] = "z",
["f"] = "v",
}
local function voice(sound, following)
return tovoiced[sound] .. following
end
text = rsub(text, "([θs])(" .. separator_c .. "*[" .. voiced .. r .. "])", voice)
text = rsub(text, "(f)(" .. separator_c .. "*[" .. voiced .. "])", voice)
-- fricative vs. stop allophones; first convert stops to fricatives, then back to stops
-- after nasals and sometimes after l
local stop_to_fricative = {["b"] = "β", ["d"] = "ð", ["ɟ"] = "ʝ", ["g"] = "ɣ"}
local fricative_to_stop = {["β"] = "b", ["ð"] = "d", ["ʝ"] = "ɟ", ["ɣ"] = "g"}
text = rsub(text, "[bdɟg]", stop_to_fricative)
text = rsub(text, "([mnɲ]" .. separator_c .. "*)([βɣ])",
function(nasal, fricative) return nasal .. fricative_to_stop[fricative] end
)
text = rsub(text, "([lʎmnɲ]" .. separator_c .. "*)([ðʝ])",
function(nasal_l, fricative) return nasal_l .. fricative_to_stop[fricative] end
)
text = rsub(text, "(##" .. ipa_stress_c .. "*)([βɣðʝ])",
function(stress, fricative) return stress .. fricative_to_stop[fricative] end
)
text = rsub(text, "[td]", {["t"] = "t̪", ["d"] = "d̪"})
-- nasal assimilation before consonants
local labiodental, dentialveolar, dental, alveolopalatal, palatal, velar =
"ɱ", "n̪", "n̟", "nʲ", "ɲ", "ŋ"
local nasal_assimilation = {
["f"] = labiodental,
["t"] = dentialveolar, ["d"] = dentialveolar,
["θ"] = dental,
["ĉ"] = alveolopalatal,
["ʃ"] = alveolopalatal,
["ʒ"] = alveolopalatal,
["ɟ"] = palatal, ["ʎ"] = palatal,
["k"] = velar, ["x"] = velar, ["g"] = velar,
}
text = rsub(text, "n(" .. separator_c .. "*)(.)",
function(stress, following) return (nasal_assimilation[following] or "n") .. stress .. following end
)
-- lateral assimilation before consonants
text = rsub(text, "l(" .. separator_c .. "*)(.)",
function(stress, following)
local l = "l"
if following == "t" or following == "d" then -- dentialveolar
l = "l̪"
elseif following == "θ" then -- dental
l = "l̟"
elseif following == "ĉ" or following == "ʃ" then -- alveolopalatal
l = "lʲ"
end
return l .. stress .. following
end)
--semivowels
text = rsub(text, "([aeouãẽõũ][iĩ])", "%1̯")
text = rsub(text, "([aeioãẽĩõ][uũ])", "%1̯")
-- voiced fricatives are actually approximants
text = rsub(text, "([βðɣ])", "%1̞")
end
-- convert fake symbols to real ones
local final_conversions = {
["ħ"] = "h", -- fake aspirated "h" to real "h"
["ĉ"] = "t͡ʃ", -- fake "ch" to real "ch"
["ɟ"] = phonetic and "ɟ͡ʝ" or "ʝ", -- fake "y" to real "y"
-- do the following at the very end so we can use regular g throughout
["g"] = "ɡ", -- U+0067 LATIN SMALL LETTER G → U+0261 LATIN SMALL LETTER SCRIPT G
[TEMP_W] = "w̝", -- see https://en.wikipedia.org/wiki/Spanish_orthography for this
}
text = rsub(text, "[ħĉɟg" .. TEMP_W .. "]", final_conversions)
-- remove # symbols at word and text boundaries
text = rsub(text, "#", "")
text = unfc(text)
-- The values in `differences` are only accurate when the dialect is 'distincion-lleismo'
-- because we look for sounds like /θ/ and /ʎ/ that are only present in that dialect.
-- The calling code knows to only use this structure in conjunction with this dialect.
-- but to make sure of this we set the structure to nil for other dialects.
local differences = nil
if dialect == "distincion-lleismo" then
differences = {
distincion_different = distincion_different,
lleismo_different = lleismo_different,
need_rioplat = initial_hi or sheismo_different,
sheismo_different = sheismo_different,
need_quito = need_quito,
need_yucatan = need_yucatan,
}
end
local ret = {
text = text,
differences = differences,
}
return ret
end
-- For bot usage; {{#invoke:es-pronunc|IPA_string|SPELLING|style=STYLE|phonetic=PHONETIC}}
-- where
--
-- 1. SPELLING is the word or respelling to generate pronunciation for;
-- 2. required parameter style= indicates the pronunciation style to generate
-- (e.g. "distincion-yeismo" for distinción+yeísmo, as is common in Spain;
-- see the comment above export.IPA() above for the full list);
-- 3. phonetic=1 specifies to generate the phonetic rather than phonemic pronunciation;
function export.IPA_string(frame)
local iparams = {
[1] = {},
["style"] = {required = true},
["phonetic"] = {type = "boolean"},
}
local iargs = require(parameters_module).process(frame.args, iparams)
local retval = export.IPA(iargs[1], iargs.style, iargs.phonetic)
return retval.text
end
-- Generate all relevant dialect pronunciations and group into styles. See the comment above about dialects and styles.
-- A "pronunciation" here could be for example the IPA phonemic/phonetic representation of the term or the IPA form of
-- the rhyme that the term belongs to. If `style_spec` is nil, this generates all styles for all dialects, but
-- `style_spec` can also be a style spec such as "seseo" or "distincion+yeismo" (see comment above) to restrict the
-- output. `dodialect` is a function of two arguments, `ret` and `dialect`, where `ret` is the return-value table (see
-- below), and `dialect` is a string naming a particular dialect, such as "distincion-lleismo" or "rioplatense-sheismo".
-- `dodialect` should side-effect the `ret` table by adding an entry to `ret.pronun` for the dialect in question.
--
-- The return value is a table of the form
--
-- {
-- pronun = {DIALECT = {PRONUN, PRONUN, ...}, DIALECT = {PRONUN, PRONUN, ...}, ...},
-- expressed_styles = {STYLE_GROUP, STYLE_GROUP, ...},
-- }
--
-- where:
-- 1. DIALECT is a string such as "distincion-lleismo" naming a specific dialect.
-- 2. PRONUN is a table describing a particular pronunciation. If the dialect is "distincion-lleismo", there should be
-- a field in this table named `differences`, but where other fields may vary depending on the type of pronunciation
-- (e.g. phonemic/phonetic or rhyme). See below for the form of the PRONUN table for phonemic/phonetic pronunciation
-- vs. rhyme and the form of the `differences` field.
-- 3. STYLE_GROUP is a table of the form {tag = "HIDDEN_TAG", styles = {INNER_STYLE, INNER_STYLE, ...}}. This describes
-- a group of related styles (such as those for Latin America) that by default (the "hidden" form) are displayed as
-- a single line, with an icon on the right to "open" the style group into the "shown" form, with multiple lines
-- for each style in the group. The tag of the style group is the text displayed before the pronunciation in the
-- default "hidden" form, such as "Spain" or "Latin America". It can have the special value of `false` to indicate
-- that no tag text is to be displayed. Note that the pronunciation shown in the default "hidden" form is taken
-- from the first style in the style group.
-- 4. INNER_STYLE is a table of the form {tag = "SHOWN_TAG", pronun = {PRONUN, PRONUN, ...}}. This describes a single
-- style (such as for the Andes Mountains and Paraguay in the case where the seseo+lleismo accent differs from all others), to
-- be shown on a single line. `tag` is the text preceding the displayed pronunciation, or `false` if no tag text
-- is to be displayed. PRONUN is a table as described above and describes a particular pronunciation.
--
-- The PRONUN table has the following form for the full phonemic/phonetic pronunciation:
--
-- {
-- phonemic = "PHONEMIC",
-- phonetic = "PHONETIC",
-- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...},
-- }
--
-- Here, `phonemic` is the phonemic pronunciation (displayed as /.../) and `phonetic` is the phonetic pronunciation
-- (displayed as [...]).
--
-- The PRONUN table has the following form for the rhyme pronunciation:
--
-- {
-- rhyme = "RHYME_PRONUN",
-- num_syl = {NUM, NUM, ...},
-- q = nil or {QUALIFIER, QUALIFIER, ...},
-- differences = {FLAG = BOOLEAN, FLAG = BOOLEAN, ...},
-- }
--
-- Here, `rhyme` is a phonemic pronunciation such as "ado" for [[abogado]] or "iʝa"/"iʎa" for [[tortilla]] (depending
-- on the dialect), and `num_syl` is a list of the possible numbers of syllables for the term(s) that have this rhyme
-- (e.g. {4} for [[abogado]], {3} for [[tortilla]] and {4, 5} for [[biología]], which may be syllabified as
-- bio.lo.gí.a or bi.o.lo.gí.a). `num_syl` is used to generate syllable-count categories such as
-- [[Category:Rhymes:Spanish/ia/4 syllables]] in addition to [[Category:Rhymes:Spanish/ia]]. `num_syl` may be nil to
-- suppress the generation of syllable-count categories; this is typically the case with multiword terms.
-- `q`, if non-nil, comes from the user using the syntax e.g. <rhyme:iʃa<q:Buenos Aires>>.
--
-- The value of the `differences` field in the PRONUN table (which, as noted above, only needs to be present for the
-- "distincion-lleismo" dialect, and otherwise should be nil) is a table containing flags indicating whether and how
-- the per-dialect pronunciations differ. This is an optimization to avoid having to generate all six dialectal
-- pronunciations and compare them. It has the following form:
--
-- {
-- distincion_different = BOOLEAN,
-- lleismo_different = BOOLEAN,
-- need_rioplat = BOOLEAN,
-- sheismo_different = BOOLEAN,
-- need_quito = BOOLEAN,
-- need_yucatan = BOOLEAN,
-- }
--
-- where:
-- 1. `distincion_different` should be `true` if the "distincion" and "seseo" pronunciations differ;
-- 2. `lleismo_different` should be `true` if the "lleismo" and "yeismo" pronunciations differ;
-- 3. `need_rioplat` should be `true` if the Rioplatense pronunciations differ from the seseo+yeismo pronunciation;
-- 4. `sheismo_different` should be `true` if the "sheismo" and "zheismo" pronunciations differ.
-- 5. `need_quito` should be `true` if the "quito" and "zheismo" pronunciations differ.
-- 6. `need_yucatan` should be `true` if the "yucatan" and "yeismo" pronunciations differ;
local function express_all_styles(style_spec, dodialect)
local ret = {
pronun = {},
expressed_styles = {},
}
local need_rioplat
local need_quito
local need_yucatan
-- Add a style object (see INNER_STYLE above) that represents a particular style to `ret.expressed_styles`.
-- `hidden_tag` is the tag text to be used when the style group containing the style is in the default "hidden"
-- state (e.g. "Spain", "Latin America" or false if there is only one style group and no tag text should be
-- shown), while `tag` is the tag text to be used when the individual style is shown (e.g. a description such as
-- "most of Spain and Latin America", "Andes Mountains and Paraguay" or "everywhere but Argentina and Uruguay").
-- `representative_dialect` is one of the dialects that this style represents, and whose pronunciation is stored in
-- the style object. `matching_styles` is a hyphen separated string listing the isoglosses described by this style.
-- For example, if the term has an ''ll'' but no ''c/z'', the `tag` text for the yeismo pronunciation will be
-- "most of Spain and Latin America" and `matching_styles` will be "distincion-seseo-yeismo", indicating that
-- it corresponds to both the "distincion" and "seseo" isoglosses as well as the "yeismo" isogloss. This is used
-- when a particular style spec is given. If `matching_styles` is omitted, it takes its value from
-- `representative_dialect`; this is used when the style contains only a single dialect.
local function express_style(hidden_tag, tag, representative_dialect, matching_styles)
matching_styles = matching_styles or representative_dialect
-- If the Rioplatense pronunciation isn't distinctive, add all Rioplatense isoglosses.
if not need_rioplat then
matching_styles = matching_styles .. "-rioplatense-sheismo-zheismo"
end
-- also Quito
if not need_quito then
matching_styles = matching_styles .. "-quito"
end
-- Yucatan
if not need_yucatan then
matching_styles = matching_styles .. "-yucatan"
end
-- If style specified, make sure it matches the requested style.
local style_matches
if not style_spec then
style_matches = true
else
local style_parts = rsplit(matching_styles, "%-")
local or_styles = rsplit(style_spec, "%s*,%s*")
for _, or_style in ipairs(or_styles) do
local and_styles = rsplit(or_style, "%s*%+%s*")
local and_matches = true
for _, and_style in ipairs(and_styles) do
local negate
if and_style:find("^%-") then
and_style = and_style:gsub("^%-", "")
negate = true
end
local this_style_matches = false
for _, part in ipairs(style_parts) do
if part == and_style then
this_style_matches = true
break
end
end
if negate then
this_style_matches = not this_style_matches
end
if not this_style_matches then
and_matches = false
end
end
if and_matches then
style_matches = true
break
end
end
end
if not style_matches then
return
end
-- Fetch the representative dialect's pronunciation if not already present.
if not ret.pronun[representative_dialect] then
dodialect(ret, representative_dialect)
end
-- Insert the new style into the style group, creating the group if necessary.
local new_style = {
tag = tag,
pronun = ret.pronun[representative_dialect],
}
for _, hidden_tag_style in ipairs(ret.expressed_styles) do
if hidden_tag_style.tag == hidden_tag then
table.insert(hidden_tag_style.styles, new_style)
return
end
end
table.insert(ret.expressed_styles, {
tag = hidden_tag,
styles = {new_style},
})
end
-- For each type of difference, figure out if the difference exists in any of the given respellings. We do this by
-- generating the pronunciation for the dialect "distincion-lleismo", for each respelling. In the process of
-- generating the pronunciation for a given respelling, it computes how the other dialects for that respelling
-- differ. Then we take the union of these differences across the respellings.
dodialect(ret, "distincion-lleismo")
local differences = {}
for _, difftype in ipairs { "distincion_different", "lleismo_different", "need_rioplat", "sheismo_different", "need_quito", "need_yucatan" } do
for _, pronun in ipairs(ret.pronun["distincion-lleismo"]) do
if pronun.differences[difftype] then
differences[difftype] = true
end
end
end
local distincion_different = differences.distincion_different
local lleismo_different = differences.lleismo_different
need_rioplat = differences.need_rioplat
local sheismo_different = differences.sheismo_different
need_quito = differences.need_quito
need_yucatan = differences.need_yucatan
-- Now, based on the observed differences, figure out how to combine the individual dialects into styles and
-- style groups.
if not distincion_different and not lleismo_different then
if not need_rioplat then
if not need_yucatan then
express_style(false, false, "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
else
express_style(false, "everywhere but northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
end
else
if not need_yucatan then
express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>> and <<Uruguay>>", "distincion-lleismo",
"distincion-seseo-lleismo-yeismo")
else
express_style(false, "everywhere but <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>, northern <<Mexico>>, <<Yucatán>> and <<Central America>> (except <<Panama>>)", "distincion-lleismo", "distincion-seseo-lleismo-yeismo")
end
end
elseif distincion_different and not lleismo_different then
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, <<Spain>>", "distincion-lleismo", "distincion-lleismo-yeismo")
if not need_rioplat and not need_yucatan then
express_style("<<Latin America>>, <<Philippines>>", "<<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo")
else
express_style("<<Latin America>>, <<Philippines>>", "most of <<Latin America>>, <<Philippines>>", "seseo-lleismo", "seseo-lleismo-yeismo")
end
elseif not distincion_different and lleismo_different then
express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo")
express_style(false, "<<rural>> <<northern Spain>>, northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "distincion-lleismo", "distincion-seseo-lleismo")
else
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<Equatorial Guinea>>, most of <<Spain>>", "distincion-yeismo")
express_style("<<Latin America>>", "most of <<Latin America>>", "seseo-yeismo")
express_style("<<Equatorial Guinea>>, <<Spain>>", "<<rural>> <<northern Spain>>", "distincion-lleismo")
express_style("<<Latin America>>", "northern and central <<Andes>> (except central <<Ecuador>>), <<Bolivia>>, <<Paraguay>>, northeastern <<Argentina>>, <<Philippines>>", "seseo-lleismo")
end
if need_rioplat then
if lleismo_different then
local hidden_tag = distincion_different and "<<Latin America>>" or false
if sheismo_different then
if not need_quito then
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs, elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo-quito")
else
express_style(hidden_tag, "central <<Ecuador>>, <<Santiago del Estero>> and environs", "quito", "seseo-quito")
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo")
end
else
express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo")
end
else
local hidden_tag = distincion_different and "<<Latin America>>, <<Philippines>>" or false
if sheismo_different then
express_style(hidden_tag, "<<Buenos Aires>> and environs", "rioplatense-sheismo", "seseo-rioplatense-sheismo")
express_style(hidden_tag, "elsewhere in <<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-zheismo", "seseo-rioplatense-zheismo")
else
express_style(hidden_tag, "<<{{w|Pampas}}>> and southern <<Argentina>>, <<Uruguay>>", "rioplatense-sheismo", "seseo-rioplatense-sheismo-zheismo")
end
end
end
if need_yucatan then
local hidden_tag = distincion_different and (lleismo_different and "<<Latin America>>" or "<<Latin America>>, <<Philippines>>") or false
express_style(hidden_tag, "northern <<Mexico>>, <<Yucatán>>, <<Central America>> (except <<Panama>>)", "yucatan")
end
-- If only one style group, don't indicate the style.
-- Not clear we want this in reality.
--if #ret.expressed_styles == 1 then
-- ret.expressed_styles[1].tag = false
-- if #ret.expressed_styles[1].styles == 1 then
-- ret.expressed_styles[1].styles[1].tag = false
-- end
--end
return ret
end
local function format_all_styles(expressed_styles, format_style)
for i, style_group in ipairs(expressed_styles) do
if #style_group.styles == 1 then
style_group.formatted, style_group.formatted_len =
format_style(style_group.styles[1].tag, style_group.styles[1], i == 1)
else
style_group.formatted, style_group.formatted_len =
format_style(style_group.tag, style_group.styles[1], i == 1)
for j, style in ipairs(style_group.styles) do
style.formatted, style.formatted_len =
format_style(style.tag, style, i == 1 and j == 1)
end
end
end
local maxlen = 0
for i, style_group in ipairs(expressed_styles) do
local this_len = style_group.formatted_len
if #style_group.styles > 1 then
for _, style in ipairs(style_group.styles) do
this_len = math.max(this_len, style.formatted_len)
end
end
maxlen = math.max(maxlen, this_len)
end
local lines = {}
local need_major_hack = false
for i, style_group in ipairs(expressed_styles) do
if #style_group.styles == 1 then
table.insert(lines, style_group.formatted)
need_major_hack = false
else
local inline = '\n<div class="vsShow" style="display:none">\n' .. style_group.formatted .. "</div>"
local full_prons = {}
for _, style in ipairs(style_group.styles) do
table.insert(full_prons, style.formatted)
end
local full = '\n<div class="vsHide">\n' .. table.concat(full_prons, "\n") .. "</div>"
local em_length = math.floor(maxlen * 0.68) -- from [[Module:grc-pronunciation]]
table.insert(lines, '<div class="vsSwitcher" data-toggle-category="pronunciations" style="width: ' .. em_length .. 'em; max-width:100%;"><span class="vsToggleElement" style="float: right;"> </span>' .. inline .. full .. "</div>")
need_major_hack = true
end
end
-- major hack to get bullets working on the next line after a div box
return table.concat(lines, "\n") .. (need_major_hack and "\n<span></span>" or "")
end
local function dodialect_pronun(args, ret, dialect)
ret.pronun[dialect] = {}
for i, term in ipairs(args.terms) do
local phonemic, phonetic, differences
if term.raw then
phonemic = term.raw_phonemic
phonetic = term.raw_phonetic
differences = construct_default_differences(dialect)
else
phonemic = export.IPA(term.term, dialect, false)
phonetic = export.IPA(term.term, dialect, true)
differences = phonemic.differences
phonemic = phonemic.text
phonetic = phonetic.text
end
ret.pronun[dialect][i] = {
raw = term.raw,
phonemic = phonemic,
phonetic = phonetic,
refs = term.refs,
q = term.q,
qq = term.qq,
a = term.a,
aa = term.aa,
differences = differences,
}
end
end
local function generate_pronun(args)
local function this_dodialect_pronun(ret, dialect)
dodialect_pronun(args, ret, dialect)
end
local ret = express_all_styles(args.style, this_dodialect_pronun)
local function format_style(tag, expressed_style, is_first)
local pronunciations = {}
local formatted_pronuns = {}
local function ins(formatted_part)
table.insert(formatted_pronuns, formatted_part)
end
-- Loop through each pronunciation. For each one, add the phonemic and phonetic versions to `pronunciations`,
-- for formatting by [[Module:IPA]], and also create an approximation of the formatted version so that we can
-- compute the appropriate width of the HTML switcher div box that holds the different per-dialect variants.
-- NOTE: The code below constructs the formatted approximation out-of-order in some cases but that doesn't
-- currently matter because we assume all characters have the same width. If we change the width computation
-- in a way that requires the correct order, we need changes to the code below.
for j, pronun in ipairs(expressed_style.pronun) do
-- Add tag to right accent qualifiers if last one
local aas = pronun.aa
if j == #expressed_style.pronun and tag then
if aas then
aas = m_table.deepCopy(aas)
table.insert(aas, tag)
else
aas = {tag}
end
end
local first_pronun = #pronunciations + 1
if not pronun.phonemic and not pronun.phonetic then
error("Internal error: Saw neither phonemic nor phonetic pronunciation")
end
if pronun.phonemic then -- missing if 'raw:[...]' given
-- don't display syllable division markers in phonemic
local slash_pron = "/" .. pronun.phonemic:gsub("%.", "") .. "/"
table.insert(pronunciations, {
pron = slash_pron,
})
ins(slash_pron)
end
if pronun.phonetic then -- missing if 'raw:/.../' given
local bracket_pron = "[" .. pronun.phonetic .. "]"
table.insert(pronunciations, {
pron = bracket_pron,
})
ins(bracket_pron)
end
local last_pronun = #pronunciations
if pronun.q then
pronunciations[first_pronun].q = pronun.q
end
if pronun.a then
pronunciations[first_pronun].a = pronun.a
end
if j > 1 then
pronunciations[first_pronun].separator = ", "
ins(", ")
end
if pronun.qq then
pronunciations[last_pronun].qq = pronun.qq
end
if aas then
pronunciations[last_pronun].aa = aas
end
if pronun.q or pronun.qq or pronun.a or aas then
-- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call
-- to textual_len() removes all HTML and reduces links.
ins(require(decorations_module).format_decorations {
lang = lang,
text = "",
q = pronun.q,
qq = pronun.qq,
a = pronun.a,
aa = aas,
})
end
if pronun.refs then
pronunciations[last_pronun].refs = pronun.refs
-- Approximate the reference using a footnote notation. This will be slightly inaccurate if there are
-- more than nine references but that is rare.
ins(string.rep("[1]", #pronun.refs))
end
if first_pronun ~= last_pronun then
pronunciations[last_pronun].separator = " "
ins(" ")
end
end
local bullet = string.rep("*", args.bullets) .. " "
-- Here we construct the formatted line in `formatted`, and also try to construct the equivalent without HTML
-- and wiki markup in `formatted_for_len`, so we can compute the approximate textual length for use in sizing
-- the toggle box with the "more" button on the right.
local pre = is_first and args.pre and args.pre .. " " or ""
local post = is_first and args.post and " " .. args.post or ""
local formatted = bullet .. pre ..
m_IPA.format_IPA_full { lang = lang, items = pronunciations, separator = "" } .. post
local formatted_for_len = bullet .. pre .. "IPA(key): " .. table.concat(formatted_pronuns) .. post
return formatted, textual_len(formatted_for_len)
end
ret.text = format_all_styles(ret.expressed_styles, format_style)
return ret
end
local function parse_respelling(respelling, pagename, parse_err)
local raw_respelling = respelling:match("^raw:(.*)$")
if raw_respelling then
local raw_phonemic, raw_phonetic = raw_respelling:match("^/(.*)/ %[(.*)%]$")
if not raw_phonemic then
raw_phonemic = raw_respelling:match("^/(.*)/$")
end
if not raw_phonemic then
raw_phonetic = raw_respelling:match("^%[(.*)%]$")
end
if not raw_phonemic and not raw_phonetic then
parse_err(("Unable to parse raw respelling '%s', should be one of /.../, [...] or /.../ [...]")
:format(raw_respelling))
end
return {
raw = true,
raw_phonemic = raw_phonemic,
raw_phonetic = raw_phonetic,
}
end
if respelling == "+" then
respelling = pagename
end
return {term = respelling}
end
-- External entry point for {{es-IPA}}.
function export.show(frame)
local params = {
[1] = {},
["pre"] = {},
["post"] = {},
["ref"] = {},
["style"] = {},
["bullets"] = {type = "number", default = 1},
}
local parargs = frame:getParent().args
local args = require(parameters_module).process(parargs, params)
local text = args[1] or mw.loadData("Module:headword/data").pagename
args.terms = {{term = text}}
local ret = generate_pronun(args)
return ret.text
end
-- Return the number of syllables of a phonemic representation, which should have syllable dividers in it but no
-- hyphens.
local function get_num_syl_from_phonemic(phonemic)
-- Maybe we should just count vowels instead of the below code.
phonemic = rsub(phonemic, "|", " ") -- remove IPA foot boundaries
local words = rsplit(phonemic, " +")
for i, word in ipairs(words) do
-- IPA stress marks are syllable divisions if between characters; otherwise just remove.
word = rsub(word, "(.)[ˌˈ](.)", "%1.%2")
word = rsub(word, "[ˌˈ]", "")
words[i] = word
end
-- There should be a syllable boundary between words.
phonemic = table.concat(words, ".")
return ulen(rsub(phonemic, "[^.]", "")) + 1
end
-- Get the rhyme by truncating everything up through the last stress mark + any following consonants, and remove
-- syllable boundary markers.
local function convert_phonemic_to_rhyme(phonemic)
-- NOTE: This works because the phonemic vowels are just [aeiou] possibly with diacritics that are separate
-- Unicode chars. If we want to handle things like ɛ or ɔ we need to add them to `vowel`.
return rsub(rsub(phonemic, ".*[ˌˈ]", ""), "^[^" .. vowel .. "]*", ""):gsub("%.", ""):gsub("t͡ʃ", "tʃ")
end
local function split_syllabified_spelling(spelling)
return rsplit(spelling, "%.")
end
-- "Align" syllabification to original spelling by matching character-by-character, allowing for extra syllable and
-- accent markers in the syllabification. If we encounter an extra syllable marker (.), we allow and keep it. If we
-- encounter an extra accent marker in the syllabification, we drop it. In any other case, we return nil indicating
-- the alignment failed.
local function align_syllabification_to_spelling(syllab, spelling)
local result = {}
local syll_chars = rsplit(decompose(syllab), "")
local spelling_chars = rsplit(decompose(spelling), "")
local i = 1
local j = 1
while i <= #syll_chars or j <= #spelling_chars do
local ci = syll_chars[i]
local cj = spelling_chars[j]
if ci == cj then
table.insert(result, ci)
i = i + 1
j = j + 1
elseif ci == "." then
table.insert(result, ci)
i = i + 1
elseif ci == AC or ci == GR or ci == CFLEX then
-- skip character
i = i + 1
else
-- non-matching character
return nil
end
end
if i <= #syll_chars or j <= #spelling_chars then
-- left-over characters on one side or the other
return nil
end
return unfc(table.concat(result))
end
local function generate_hyph_obj(term)
return {syllabification = term, hyph = split_syllabified_spelling(term)}
end
-- Word should already be decomposed.
local function word_has_vowels(word)
return rfind(word, V)
end
local function all_words_have_vowels(term)
local words = rsplit(decompose(term), "[ %-]")
for i, word in ipairs(words) do
-- Allow empty word; this occurs with prefixes and suffixes.
if word ~= "" and not word_has_vowels(word) then
return false
end
end
return true
end
local function should_generate_rhyme_from_respelling(term)
local words = rsplit(decompose(term), " +")
return #words == 1 and -- no if multiple words
not words[1]:find(".%-.") and -- no if word is composed of hyphenated parts (e.g. [[Austria-Hungría]])
not words[1]:find("%-$") and -- no if word is a prefix
not (words[1]:find("^%-") and words[1]:find(CFLEX)) and -- no if word is an unstressed suffix
word_has_vowels(words[1]) -- no if word has no vowels (e.g. a single letter)
end
local function should_generate_rhyme_from_ipa(ipa)
return not ipa:find("%s") and word_has_vowels(decompose(ipa))
end
local function dodialect_specified_rhymes(rhymes, hyphs, parsed_respellings, rhyme_ret, dialect)
rhyme_ret.pronun[dialect] = {}
for _, rhyme in ipairs(rhymes) do
local num_syl = rhyme.num_syl
local no_num_syl = false
-- If user explicitly gave the rhyme but didn't explicitly specify the number of syllables, try to take it from
-- the hyphenation.
if not num_syl then
num_syl = {}
for _, hyph in ipairs(hyphs) do
if should_generate_rhyme_from_respelling(hyph.syllabification) then
local this_num_syl = 1 + ulen(rsub(hyph.syllabification, "[^.]", ""))
m_table.insertIfNot(num_syl, this_num_syl)
else
no_num_syl = true
break
end
end
if no_num_syl or #num_syl == 0 then
num_syl = nil
end
end
-- If that fails and term is single-word, try to take it from the phonemic.
if not no_num_syl and not num_syl then
for _, parsed in ipairs(parsed_respellings) do
for dialect, pronun in pairs(parsed.pronun.pronun[dialect]) do
-- Check that pronun.phonemic exists (it may not if raw phonetic-only pronun is given).
if pronun.phonemic then
if not should_generate_rhyme_from_ipa(pronun.phonemic) then
no_num_syl = true
break
end
-- Count number of syllables by looking at syllable boundaries (including stress marks).
local this_num_syl = get_num_syl_from_phonemic(pronun.phonemic)
m_table.insertIfNot(num_syl, this_num_syl)
end
end
if no_num_syl then
break
end
end
if no_num_syl or #num_syl == 0 then
num_syl = nil
end
end
table.insert(rhyme_ret.pronun[dialect], {
rhyme = rhyme.rhyme,
num_syl = num_syl,
q = rhyme.q,
qq = rhyme.qq,
a = rhyme.a,
aa = rhyme.aa,
differences = construct_default_differences(dialect),
})
end
end
local q_qq_inline_modifier_spec = {
store = "insert-flattened",
type = "qualifier",
}
local a_aa_inline_modifier_spec = {
store = "insert-flattened",
type = "labels",
}
local ref_inline_modifier_spec = {
store = "insert-flattened",
item_dest = "refs",
type = "references",
}
-- Parse a pronunciation modifier in `arg`, the argument portion in an inline modifier (after the prefix), which
-- specifies a pronunciation property such as rhyme, hyphenation/syllabification, homophones or audio. The argument
-- can itself have inline modifiers, e.g. <audio:Foo.ogg<a:Colombia>>. The allowed inline modifiers are specified
-- by `param_mods` (of the format expected by `parse_inline_modifiers()`); in addition to any modifiers specified
-- there, the modifiers <q:...>, <qq:...>, <a:...>, <aa:...> and <ref:...> are always accepted (and can be repeated).
-- `generate_obj` and `parse_err` are like in `parse_inline_modifiers()` and specify respectively a function to
-- generate the object into which modifier properties are stored given the non-modifier part of the argument, and
-- a function to generate an error message (given the message). Normally, a comma-separated list of pronunciation
-- properties is accepted and parsed, where each element in the list can have its own inline modifiers and where
-- no spaces are allowed next to the commas in order for them to be recognized as separators. If `no_split_on_comma`
-- is given, only a single pronunciation property is accepted. In all cases, however, the return value is a list
-- of property objects (when `no_split_on_comma` is given, the return value is a one-element list).
local function parse_pron_modifier(arg, parse_err, generate_obj, param_mods, no_split_on_comma)
if arg:find("<") then
param_mods.q = q_qq_inline_modifier_spec
param_mods.qq = q_qq_inline_modifier_spec
param_mods.a = a_aa_inline_modifier_spec
param_mods.aa = a_aa_inline_modifier_spec
param_mods.ref = ref_inline_modifier_spec
local retval = require(parse_utilities_module).parse_inline_modifiers(arg, {
param_mods = param_mods,
generate_obj = generate_obj,
parse_err = parse_err,
splitchar = not no_split_on_comma and "," or nil,
})
if no_split_on_comma then
retval = {retval}
end
return retval
elseif no_split_on_comma then
return {generate_obj(arg)}
else
local retval = {}
for _, term in ipairs(split_on_comma(arg)) do
table.insert(retval, generate_obj(term))
end
return retval
end
end
local function parse_rhyme(arg, parse_err)
local function generate_obj(term)
return {rhyme = term}
end
local param_mods = {
s = {
item_dest = "num_syl",
type = "number",
sublist = true,
},
}
return parse_pron_modifier(arg, parse_err, generate_obj, param_mods)
end
local function parse_hyph(arg, parse_err)
-- None other than decorations
local param_mods = {}
return parse_pron_modifier(arg, parse_err, generate_hyph_obj, param_mods)
end
local function parse_homophone(arg, parse_err)
local function generate_obj(term)
return {term = term}
end
local param_mods = {
t = {
-- [[Module:links]] expects the gloss in "gloss".
item_dest = "gloss",
},
gloss = {},
-- No tr=, ts=, or sc=; doesn't make sense for Spanish.
pos = {},
alt = {},
lit = {},
id = {},
g = {
-- [[Module:links]] expects the genders in "genders".
item_dest = "genders",
sublist = true,
},
}
return parse_pron_modifier(arg, parse_err, generate_obj, param_mods)
end
local function generate_audio_obj(arg)
local file, caption = arg:match("^(.-)%s*#%s*(.*)$")
file = file or arg
return {file = file, caption = caption}
end
local function parse_audio(arg, parse_err)
local param_mods = {
IPA = {
sublist = true,
},
text = {},
t = {
item_dest = "gloss",
},
-- No tr=, ts=, or sc=; doesn't make sense for Spanish.
gloss = {},
pos = {},
-- No alt=; text= already goes in alt=.
lit = {},
-- No id=; text= already goes in alt= and isn't normally linked.
g = {
item_dest = "genders",
sublist = true,
},
bad = {},
}
-- Don't split on comma because some filenames have embedded commas not followed by a space
-- (typically followed by an underscore).
local retvals = parse_pron_modifier(arg, parse_err, generate_audio_obj, param_mods, "no split on comma")
local retval = retvals[1]
retval.lang = lang
local textobj = require(audio_module).construct_audio_textobj(retval)
retval.text = textobj
retval.gloss = nil
retval.pos = nil
retval.lit = nil
retval.genders = nil
return retval
end
-- External entry point for {{es-pr}}.
function export.show_pr(frame)
local params = {
[1] = {list = true},
["rhyme"] = {convert = parse_rhyme},
["hyph"] = {convert = parse_hyph},
["hmp"] = {convert = parse_homophone},
["audio"] = {list = true},
["pagename"] = {},
}
local parargs = frame:getParent().args
local args = require(parameters_module).process(parargs, params)
local pagename = args.pagename or mw.loadData(headword_data_module).pagename
-- Parse the arguments.
local respellings = #args[1] > 0 and args[1] or {"+"}
local parsed_respellings = {}
local overall_rhyme = args.rhyme
local overall_hyph = args.hyph
local overall_hmp = args.hmp
local overall_audio
if args.audio then
-- We can't specify parse_audio() as a `convert` function because it needs access to `pagename` (i.e. another
-- parameter).
overall_audio = {}
for i, audio in ipairs(args.audio) do
local function parse_err(msg)
error(("%s: parameter audio%s=%s"):format(msg, i == 1 and "" or i, audio))
end
local parsed_audio = parse_audio(audio, parse_err, pagename)
table.insert(overall_audio, parsed_audio)
end
end
for i, respelling in ipairs(respellings) do
if respelling:find("<") then
local param_mods = {
pre = { overall = true },
post = { overall = true },
style = { overall = true },
bullets = {
overall = true,
type = "number",
},
rhyme = {
overall = true,
store = "insert-flattened",
convert = parse_rhyme,
},
hyph = {
overall = true,
store = "insert-flattened",
convert = parse_hyph,
},
hmp = {
overall = true,
store = "insert-flattened",
convert = parse_homophone,
},
audio = {
overall = true,
store = "insert",
convert = function(arg, parse_err)
return parse_audio(arg, parse_err, pagename)
end,
},
ref = ref_inline_modifier_spec,
q = q_qq_inline_modifier_spec,
qq = q_qq_inline_modifier_spec,
a = a_aa_inline_modifier_spec,
aa = a_aa_inline_modifier_spec,
}
local parsed = require(parse_utilities_module).parse_inline_modifiers(respelling, {
paramname = i,
param_mods = param_mods,
generate_obj = function(term, parse_err)
return parse_respelling(term, pagename, parse_err)
end,
splitchar = ",",
outer_container = {
audio = {}, rhyme = {}, hyph = {}, hmp = {}
}
})
if not parsed.bullets then
parsed.bullets = 1
end
table.insert(parsed_respellings, parsed)
else
local termobjs = {}
local function parse_err(msg)
error(msg .. ": " .. i .. "=" .. respelling)
end
for _, term in ipairs(split_on_comma(respelling)) do
table.insert(termobjs, parse_respelling(term, pagename, parse_err))
end
table.insert(parsed_respellings, {
terms = termobjs,
audio = {},
rhyme = {},
hyph = {},
hmp = {},
bullets = 1,
})
end
end
if overall_hyph then
local hyphs = {}
for _, hyph in ipairs(overall_hyph) do
if hyph.syllabification == "+" then
hyph.syllabification = syllabify_from_spelling(pagename)
hyph.hyph = split_syllabified_spelling(hyph.syllabification)
elseif hyph.syllabification == "-" then
overall_hyph = {}
break
end
end
end
-- Loop over individual respellings, processing each.
for _, parsed in ipairs(parsed_respellings) do
parsed.pronun = generate_pronun(parsed)
local no_auto_rhyme = false
for _, term in ipairs(parsed.terms) do
if term.raw then
if not should_generate_rhyme_from_ipa(term.raw_phonemic or term.raw_phonetic) then
no_auto_rhyme = true
break
end
elseif not should_generate_rhyme_from_respelling(term.term) then
no_auto_rhyme = true
break
end
end
if #parsed.hyph == 0 then
if not overall_hyph and all_words_have_vowels(pagename) then
for _, term in ipairs(parsed.terms) do
if not term.raw then
local syllabification = syllabify_from_spelling(term.term)
local aligned_syll = align_syllabification_to_spelling(syllabification, pagename)
if aligned_syll then
m_table.insertIfNot(parsed.hyph, generate_hyph_obj(aligned_syll))
end
end
end
end
else
for _, hyph in ipairs(parsed.hyph) do
if hyph.syllabification == "+" then
hyph.syllabification = syllabify_from_spelling(pagename)
hyph.hyph = split_syllabified_spelling(hyph.syllabification)
elseif hyph.syllabification == "-" then
parsed.hyph = {}
break
end
end
end
-- Generate the rhymes.
local function dodialect_rhymes_from_pronun(rhyme_ret, dialect)
rhyme_ret.pronun[dialect] = {}
-- It's possible the pronunciation for a passed-in dialect was never generated. This happens e.g. with
-- {{es-pr|cebolla<style:seseo>}}. The initial call to generate_pronun() fails to generate a pronunciation
-- for the dialect 'distinction-yeismo' because the pronunciation of 'cebolla' differs between distincion
-- and seseo and so the seseo style restriction rules out generation of pronunciation for distincion
-- dialects (other than 'distincion-lleismo', which always gets generated so as to determine on which axes
-- the dialects differ). However, when generating the rhyme, it is based only on -olla, whose pronunciation
-- does not differ between distincion and seseo, but does differ between lleismo and yeismo, so it needs to
-- generate a yeismo-specific rhyme, and 'distincion-yeismo' is the representative dialect for yeismo in the
-- situation where distincion and seseo do not have distinct results (based on the following line in
-- express_all_styles()):
-- express_style(false, "<<Equatorial Guinea>>, most of <<Latin America>> and <<Spain>>", "distincion-yeismo", "distincion-seseo-yeismo")
-- In this case we need to generate the missing overall pronunciation ourselves since we need it to generate
-- the dialect-specific rhyme pronunciation.
if not parsed.pronun.pronun[dialect] then
dodialect_pronun(parsed, parsed.pronun, dialect)
end
for _, pronun in ipairs(parsed.pronun.pronun[dialect]) do
-- We should have already excluded multiword terms and terms without vowels from rhyme generation (see
-- `no_auto_rhyme` below). But make sure to check that pronun.phonemic exists (it may not if raw
-- phonetic-only pronun is given).
if pronun.phonemic then
-- Count number of syllables by looking at syllable boundaries (including stress marks).
local num_syl = get_num_syl_from_phonemic(pronun.phonemic)
-- Get the rhyme by truncating everything up through the last stress mark + any following
-- consonants, and remove syllable boundary markers.
local rhyme = convert_phonemic_to_rhyme(pronun.phonemic)
local saw_already = false
for _, existing in ipairs(rhyme_ret.pronun[dialect]) do
if existing.rhyme == rhyme then
saw_already = true
-- We already saw this rhyme but possibly with a different number of syllables,
-- e.g. if the user specified two pronunciations 'biología' (4 syllables) and
-- 'bi.ología' (5 syllables), both of which have the same rhyme /ia/.
m_table.insertIfNot(existing.num_syl, num_syl)
break
end
end
if not saw_already then
local rhyme_diffs = nil
if dialect == "distincion-lleismo" then
rhyme_diffs = {}
if rhyme:find("θ") then
rhyme_diffs.distincion_different = true
end
if rhyme:find("ʎ") then
rhyme_diffs.lleismo_different = true
if rhyme:find("ɟ") then
rhyme_diffs.need_quito = true
end
end
if rfind(rhyme, "[ʎɟ]") then
rhyme_diffs.sheismo_different = true
rhyme_diffs.need_rioplat = true
if rfind(rhyme, V .. "[ʎɟ]" .. V) then
rhyme_diffs.need_yucatan = true
end
end
end
table.insert(rhyme_ret.pronun[dialect], {
rhyme = rhyme,
num_syl = {num_syl},
differences = rhyme_diffs,
})
end
end
end
end
if #parsed.rhyme == 0 then
if overall_rhyme or no_auto_rhyme then
parsed.rhyme = nil
else
parsed.rhyme = express_all_styles(parsed.style, dodialect_rhymes_from_pronun)
end
else
local no_rhyme = false
for _, rhyme in ipairs(parsed.rhyme) do
if rhyme.rhyme == "-" then
no_rhyme = true
break
end
end
if no_rhyme then
parsed.rhyme = nil
else
local function this_dodialect(rhyme_ret, dialect)
return dodialect_specified_rhymes(parsed.rhyme, parsed.hyph, {parsed}, rhyme_ret, dialect)
end
parsed.rhyme = express_all_styles(parsed.style, this_dodialect)
end
end
end
if overall_rhyme then
local no_overall_rhyme = false
for _, orhyme in ipairs(overall_rhyme) do
if orhyme.rhyme == "-" then
no_overall_rhyme = true
break
end
end
if no_overall_rhyme then
overall_rhyme = nil
else
local all_hyphs
if overall_hyph then
all_hyphs = overall_hyph
else
all_hyphs = {}
for _, parsed in ipairs(parsed_respellings) do
for _, hyph in ipairs(parsed.hyph) do
m_table.insertIfNot(all_hyphs, hyph)
end
end
end
local function dodialect_overall_rhyme(rhyme_ret, dialect)
return dodialect_specified_rhymes(overall_rhyme, all_hyphs, parsed_respellings, rhyme_ret, dialect)
end
overall_rhyme = express_all_styles(parsed.style, dodialect_overall_rhyme)
end
end
-- If all sets of pronunciations have the same rhymes, display them only once at the bottom.
-- Otherwise, display rhymes beneath each set, indented.
local first_rhyme_ret
local all_rhyme_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_rhyme_ret = parsed.rhyme
elseif not m_table.deepEquals(first_rhyme_ret, parsed.rhyme) then
all_rhyme_sets_eq = false
break
end
end
local function format_rhyme(rhyme_ret, num_bullets)
local function format_rhyme_style(tag, expressed_style, is_first)
local pronunciations = {}
local rhymes = {}
for _, pronun in ipairs(expressed_style.pronun) do
table.insert(rhymes, pronun)
end
local data = {
lang = lang,
rhymes = rhymes,
aa = tag and {tag} or nil,
force_cat = force_cat,
}
local bullet = string.rep("*", num_bullets) .. " "
local formatted = bullet .. require(rhymes_module).format_rhymes(data)
local formatted_for_len_parts = {}
table.insert(formatted_for_len_parts, bullet .. "Rhymes: " .. (tag and "(" .. tag .. ") " or ""))
for j, pronun in ipairs(expressed_style.pronun) do
if j > 1 then
table.insert(formatted_for_len_parts, ", ")
end
if pronun.q or pronun.qq or pronun.a or pronun.aa then
-- Note: This inserts the actual formatted decoration text, including HTML and such, but the later call
-- to textual_len() removes all HTML and reduces links.
table.insert(formatted_for_len_parts, require(decorations_module).format_decorations {
lang = lang,
text = "",
q = pronun.q,
qq = pronun.qq,
a = pronun.a,
aa = pronun.aa,
})
end
table.insert(formatted_for_len_parts, "-" .. pronun.rhyme)
end
return formatted, textual_len(table.concat(formatted_for_len_parts))
end
return format_all_styles(rhyme_ret.expressed_styles, format_rhyme_style)
end
-- If all sets of pronunciations have the same hyphenations, display them only once at the bottom.
-- Otherwise, display hyphenations beneath each set, indented.
local first_hyphs
local all_hyph_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_hyphs = parsed.hyph
elseif not m_table.deepEquals(first_hyphs, parsed.hyph) then
all_hyph_sets_eq = false
break
end
end
local function format_hyphenations(hyphs, num_bullets)
local hyphtext = require(hyphenation_module).format_hyphenations { lang = lang, hyphs = hyphs, caption = "Pagpapantig" } --TLCHANGE "Syllabification"
return string.rep("*", num_bullets) .. " " .. hyphtext
end
-- If all sets of pronunciations have the same homophones, display them only once at the bottom.
-- Otherwise, display homophones beneath each set, indented.
local first_hmps
local all_hmp_sets_eq = true
for j, parsed in ipairs(parsed_respellings) do
if j == 1 then
first_hmps = parsed.hmp
elseif not m_table.deepEquals(first_hmps, parsed.hmp) then
all_hmp_sets_eq = false
break
end
end
local function format_homophones(hmps, num_bullets)
local hmptext = require(homophones_module).format_homophones { lang = lang, homophones = hmps }
return string.rep("*", num_bullets) .. " " .. hmptext
end
local function format_audio(audios, num_bullets)
local ret = {}
for i, audio in ipairs(audios) do
local text = require(audio_module).format_audio(audio)
table.insert(ret, string.rep("*", num_bullets) .. " " .. text)
end
return table.concat(ret, "\n")
end
local textparts = {}
local min_num_bullets = math.huge
for j, parsed in ipairs(parsed_respellings) do
if parsed.bullets < min_num_bullets then
min_num_bullets = parsed.bullets
end
if j > 1 then
table.insert(textparts, "\n")
end
table.insert(textparts, parsed.pronun.text)
if #parsed.audio > 0 then
table.insert(textparts, "\n")
-- If only one pronunciation set, add the audio with the same number of bullets, otherwise
-- indent audio by one more bullet.
table.insert(textparts, format_audio(parsed.audio,
#parsed_respellings == 1 and parsed.bullets or parsed.bullets + 1))
end
if not all_rhyme_sets_eq and parsed.rhyme then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(parsed.rhyme, parsed.bullets + 1))
end
if not all_hyph_sets_eq and #parsed.hyph > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(parsed.hyph, parsed.bullets + 1))
end
if not all_hmp_sets_eq and #parsed.hmp > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(parsed.hmp, parsed.bullets + 1))
end
end
if overall_audio and #overall_audio > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_audio(overall_audio, min_num_bullets))
end
if all_rhyme_sets_eq and first_rhyme_ret then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(first_rhyme_ret, min_num_bullets))
end
if overall_rhyme then
table.insert(textparts, "\n")
table.insert(textparts, format_rhyme(overall_rhyme, min_num_bullets))
end
if all_hyph_sets_eq and #first_hyphs > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(first_hyphs, min_num_bullets))
end
if overall_hyph and #overall_hyph > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_hyphenations(overall_hyph, min_num_bullets))
end
if all_hmp_sets_eq and #first_hmps > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(first_hmps, min_num_bullets))
end
if overall_hmp and #overall_hmp > 0 then
table.insert(textparts, "\n")
table.insert(textparts, format_homophones(overall_hmp, min_num_bullets))
end
return table.concat(textparts)
end
return export
ad08xkdkfd3zcgx3eicevfi4jt8ln8v