Jump to content

Մոդուլ:Text


Text՝ մոդուլ մը, որ կը պարունակէ գրութիւն, ուիքի նշագրում եւ HTML մշակելու ֆունկցիաներ։

Կաղապարներու համար նախատեսուած ֆունկցիաներ

[Խմբագրել աղբիւրը]

Բոլոր ֆունկցիաները ունին անանուն պարամետր մը, որ կը պարունակէ գրութիւնը։

Եթէ պարամետրը պայմաններուն չի համապատասխաներ, վերադարձուած արժէքը պարապ տող է։ Երբ պայմանը կը բաւարարուի կամ արդիւնք մը կը գտնուի, կը վերադարձուի առնուազն մէկ նիշ պարունակող տող։

char
Կը կազմէ տող մը՝ նիշերու կոդերու ցանկէ մը։
1
Բացատով բաժնուած նիշերու կոդերու ցանկ
*
1 պարամետրի ցանկին կրկնութիւններուն թիւը (լռելեայն՝ 1)
errors
0՝ սխալի հաղորդագրութիւնները չցուցադրել։ Այս պարամետրը կ՚անցնի Module:Yesno-էն, ուստի կ՚ընդունի նաեւ վերջինիս ճանչցած ուրիշ արժէքները։
concatParams
Կը միացնէ ցանկացած թիւով տարրեր՝ մէկ տողի մէջ, Lua-ի table.concat()-ին նման։
Կաղապարէ՝
1
Առաջին տարրը. բացակայ եւ պարապ տարրերը կ՚անտեսուին։
2 3 4 5 6 …
Յաջորդ տարրերը
separator
Տարրերու միջեւ բաժանիչը. լռելեայն՝ |
format
Ընտրովի ձեւաչափ, որ կը կիրառուի իւրաքանչիւր տարրի վրայ. պէտք է պարունակէ %s։
template=1
Տարրերը կանչող կաղապարէն առնել։
Lua-էն՝
args
տարրերու աղիւսակ (յաջորդականութիւն)
apply
Բաժանիչ. լռելեայն՝ |
adapt
Ընտրովի ձեւաչափ. պէտք է պարունակէ %s։
containsCJK
Կը վերադարձնէ արժէք մը, եթէ մուտքային տողը կը պարունակէ CJK (չինական, ճաբոնական, քորէական) նիշեր։
  • Ոչինչ կը վերադարձնէ, եթէ CJK նիշ չկայ։
  • Կը պահանջէ Module:Text/data։
getPlain
Կը հեռացնէ ուիքի նշագրումը (բացի կաղապարներէն)՝ մեկնաբանութիւններ, HTML պիտակներ, թաւ, շեղ, անխզելի բացատ։
isLatinRange
Կը վերադարձնէ արժէք մը, բացի այն պարագայէն, երբ տողը կը պարունակէ նիշ մը, որ սովորաբար լատինատառ գրութեան մէջ չի հանդիպիր։
  • Ոչինչ կը վերադարձնէ ոչ լատինատառ տողի պարագային, ուստի հայերէն գրութեան միշտ ոչինչ կը վերադարձնէ։
  • Կը պահանջէ Module:Text/data։
isQuote
Կը վերադարձնէ արժէք մը, եթէ փոխանցուած պարամետրը մէկ նիշ է եւ այդ նիշը չակերտ է, օրինակ '։
  • Ոչինչ կը վերադարձնէ բազմաթիւ նիշերու պարագային, կամ եթէ նիշը չակերտ չէ։
  • Կը պահանջէ Module:Text/data։
listToFormat
Կը բաժնէ մէկ կամ քանի մը ցանկ եւ իւրաքանչիւր տողի համար կը կրկնէ տրուած ձեւաչափը՝ ամէն %s-ի տեղ դնելով համապատասխան ցանկին տարրը։
  • 1, 2, 3, …՝ բաժնուելիք ցանկերը
  • sep՝ ցանկերը բաժնելու բաժանիչը. լռելեայն՝ ;
  • format՝ ձեւաչափը. պէտք է պարունակէ այնքան %s, որքան ցանկ կայ։
listToText
Կը ձեւաւորէ ցանկի տարրերը՝ mw.text.listToText()-ի նման։
Տարրերը կը բաժնուին ստորակէտով, իսկ վերջին երկուքին միջեւ կը դրուի շաղկապ։ Բաժանիչն ու շաղկապը կու գան MediaWiki-ի համակարգային հաղորդագրութիւններէն՝ ուիքիին բովանդակութեան լեզուին համաձայն, ուստի հոս թարգմանելիք բան չկայ։
Անանուն պարամետրերը կը դառնան ցանկին տարրերը։
#invoke-ի ընտրովի պարամետրեր՝
  • format՝ իւրաքանչիւր տարր նախ պիտի ձեւաւորուի այս ձեւաչափով։ Տե՛ս string.format։ Տողը պէտք է պարունակէ առնուազն մէկ %s։
  • template=1՝ ցանկին տարրերը կանչող կաղապարէն առնել։
quote
Տողը կը փակցնէ չակերտներու մէջ։ Չակերտները կրնան ընտրուիլ ըստ լեզուի։
1
Մուտքային գրութիւն (եզրային բացատները ինքնաբերաբար կը հեռացուին). կրնայ պարապ ըլլալ։
2
(ընտրովի) չակերտներուն համար ISO 639 լեզուի ծածկագիրը։ hyw եւ hy ծածկագիրները կը մշակուին այս մոդուլին մէջ եւ կու տան «…»։ Ուրիշ լեզուներ կը կարդացուին Module:Text/data-էն։
3
(ընտրովի) 2՝ երկրորդ մակարդակի չակերտներու համար, այսինքն չակերտի մէջ չակերտ։
quoteUnquoted
Տողը կը փակցնէ չակերտներու մէջ, եթէ արդէն չակերտուած չէ եւ պարապ չէ։ Պարապ տողը չի չակերտուիր, ոչ ալ այն տողը, որուն սկիզբը կամ վերջը արդէն չակերտ կայ։
1
Մուտքային գրութիւն (եզրային բացատները ինքնաբերաբար կը հեռացուին). կրնայ պարապ ըլլալ։
2
(ընտրովի) ISO 639 լեզուի ծածկագիր
3
(ընտրովի) 2՝ երկրորդ մակարդակի չակերտներու համար
  • Վերջին նիշին ստուգումը կը պահանջէ Module:Text/data։
removeDiacritics
Կը հեռացնէ բոլոր տարբերանշանները։
1
Մուտքային գրութիւն
  • Կը պահանջէ Module:Text/data։
sentenceTerminated
Կը ստուգէ թէ նախադասութիւնը աւարտած է։ Յաջորդող չակերտները արգելք չեն։
  • Ոչինչ կը վերադարձնէ, եթէ նախադասութիւնը աւարտած չէ։
  • Զգուշացում. օրինաչափութիւնը կու գայ Module:Text/data-էն եւ կը ճանչնայ միայն լատինատառ կէտադրութիւն։ Հայերէն վերջակէտը, հարցականն ու բացականչականը չեն ճանչցուիր։
split
Կը բաժնէ տողը նշուած բաժանիչով եւ կը վերադարձնէ առաջին (կամ նշուած) մասը։ Ասիկա mw.text.split-ի ոչ Unicode տարբերակն է, որ ASCII գրութեան համար մինչեւ 60 անգամ աւելի արագ է։
  • 1 (կամ text)՝ բաժնուելիք գրութիւնը
  • 2 (կամ pattern)՝ բաժնելու համար գործածուելիք օրինաչափութիւնը
  • 3 (կամ plain)՝ եթէ «true» է, pattern-ը կ՚ընկալուի իբրեւ պարզ գրութիւն, ոչ թէ օրինաչափութիւն
  • 4 (կամ index)՝ վերադարձուելիք մասը։ Բացակայութեան պարագային՝ առաջինը։ Բացասական թիւը կը հաշուէ վերջէն (օրինակ -1՝ վերջին մասը)։
  • Զգուշացում. այս ֆունկցիան բայթի հիմքով կ՚աշխատի։ Պարզ բաժանիչով կամ ASCII օրինաչափութեամբ (,, %s+) ապահով է, բայց %a, %w, %l, %u դասերով հայերէն գրութեան վրայ կիրառուելով՝ կը կտրէ նիշին մէջտեղէն եւ կ՚արտադրէ աղաւաղուած գրութիւն՝ առանց սխալ ցոյց տալու։ Այդ պարագային գործածէ՛ mw.text.split։
ucfirstAll
Իւրաքանչիւր ճանչցուած բառին առաջին տառը կը դարձնէ մեծատառ։ Ասիկա կը տարբերի ucfirst: վերլուծիչի ֆունկցիայէն, որ միայն ամբողջ տողին առաջին նիշը կը փոխէ։
Քանի մը յաճախ գործածուող HTML էնթիթիներ պաշտպանուած են։ Ասոր հետեւանքով թուային էնթիթիները (օրինակ &) կրնան վերածուիլ & ձեւին։
  • Հայերէնը վերնագիրի մէջ ամէն բառ մեծատառով չի գրեր։ Այս ֆունկցիան պահուած է en-էն ներմուծուած կաղապարներուն համար. ընթերցողին ցուցադրուող գրութեան վրայ չկիրարկել։
uprightNonlatin
Կ՚ապահովէ որ ոչ լատինատառ մասերը շեղ չըլլան։ Մէկ յունարէն տառ կրնայ բացառութիւն ըլլալ։
  • Զգուշացում. ֆունկցիան կ՚ենթադրէ լատինատառ գրութիւն՝ ոչ լատինատառ ներդիրներով։ Այս ուիքիին վրայ սովորական պարագան հակառակն է, ուստի ան կը փաթթէ հայերէնը, ոչ թէ լատինատառը։
  • Կը պահանջէ Module:Text/data։
zip
Կը միացնէ քանի մը ցանկ՝ խաչաձեւելով զանոնք։ Օրինակով աւելի դիւրին կը բացատրուի. եթէ list1 = "a,b,c" եւ list2 = "1,2,3", ապա
zip(list1, list2, sep = ",", isep = "-", osep = "/")
կու տայ
a-1/b-2/c-3
  • 1, 2, 3, …՝ միացուելիք ցանկերը
  • sep՝ ցանկերը բաժնելու բաժանիչը (Lua-ի օրինաչափութեան ձեւով)։ Եթէ պարապ է, ցանկերը կը բաժնուին առանձին նիշերու։
  • sep1, sep2, sep3, …՝ իւրաքանչիւր ցանկի համար տարբեր բաժանիչ գործածելու կարելիութիւն
  • isep՝ ելքի բաժանիչ. կը դրուի իրենց ցանկերուն մէջ նոյն դիրքը գրաւող տարրերուն միջեւ
  • osep՝ ելքի բաժանիչ. կը դրուի տարբեր դիրք ունեցող տարրերուն, այսինքն isep-ով միացուած խումբերուն միջեւ
failsafe
Կը վերադարձնէ մոդուլին տարբերակին թուականը։ Կը ծառայէ ստուգելու թէ մոդուլը ճիշդ բեռնուած է։

Գործածութիւն ուրիշ Lua մոդուլի մէջ

[Խմբագրել աղբիւրը]

Վերոյիշեալ բոլոր ֆունկցիաները կրնան կանչուիլ ուրիշ Lua մոդուլներէ։ Գործածէ՛ require()։ Ստորեւ բերուած ծածկագիրը կը ստուգէ բեռնումի սխալները՝

local lucky, Text = pcall( require, "Module:Text" )
if type( Text ) == "table" then
    Text = Text.Text()
else
    -- In the event of errors, Text is an error message.
    return "<span class=\"error\">" .. Text .. "</span>"
end

Ապա կրնաս կանչել՝

  • Text.char( apply, again, accept )
  • Text.concatParams( args, separator, format )
  • Text.containsCJK( s )
  • Text.removeDelimited( s, prefix, suffix )
  • Text.getPlain( s )
  • Text.isLatinRange( s )
  • Text.isQuote( c )
  • Text.listToText( table, format )
  • Text.quote( s, lang, mode )
  • Text.quoteUnquoted( s, lang, mode )
  • Text.removeDiacritics( s )
  • Text.sentenceTerminated( s )
  • Text.split( text, pattern, plain )՝ mw.text.split-ի ոչ Unicode տարբերակը
  • Text.gsplit( text, pattern, plain )՝ mw.text.gsplit-ի ոչ Unicode տարբերակը
  • Text.ucfirstAll( s )
  • Text.uprightNonlatin( s )

Text.removeDelimited միայն Lua-էն կը կանչուի. կաղապարներու համար արտածուած չէ։ Կը հեռացնէ prefix-ի եւ


local yesNo = require("Module:Yesno")
local unpack = table.unpack or unpack
local Text = { serial = "2026-08-10",
               suite  = "Text" }
--[=[
Text utilities.
Localised for hyw from the en.wikipedia version, upstream serial 2024-09-21.
Almost all of this module is language neutral. The reader facing surface is
the localisation block immediately below; everything else is Unicode plumbing
and should be kept in step with upstream.
]=]

--------------------------------------------------------------------------------
-- hyw localisation block.
--------------------------------------------------------------------------------

-- Shown by Text.char() when the codepoint list cannot be used.
local MSG_BAD_CODEPOINTS = "անվաւեր կոդային կէտեր՝ "

-- Quotation marks, consulted before Module:Text/data so that quoting works on
-- this wiki even if that data page has not been imported yet.
-- Structure matches QuoteType in the data page:
--   [1] = { open, close } codepoints for level 1
--   [2] = { open, close } codepoints for level 2
--   [3] = true if a non breaking space goes inside the marks (French style)
-- Armenian uses guillemets with no inner spacing.
local quote_override = {
	hyw = { { 0x00AB, 0x00BB }, { 0x201C, 0x201D }, false },
	hy  = { { 0x00AB, 0x00BB }, { 0x201C, 0x201D }, false },
}

--------------------------------------------------------------------------------
-- End of localisation block.
--------------------------------------------------------------------------------

local function textData()
	-- Module:Text/data holds language neutral Unicode tables (quote suites,
	-- script ranges, combining mark patterns). It must be imported from
	-- en.wikipedia unchanged; nothing in it gets translated.
	-- Loaded lazily so the rest of this module works without it.
	local ok, data = pcall( mw.loadData, 'Module:Text/data' )
	if not ok then
		error( 'Module:Text/data is not present on this wiki; import it from en.wikipedia unchanged', 2 )
	end
	return data
end

local function fiatQuote( apply, alien, advance )
    -- Quote text
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code
    --     advance  -- number, with level 1 or 2
    local r = apply and tostring(apply) or ""
    alien = alien or "en"
    advance = tonumber(advance) or 0
    -- Language codes are ASCII, so a byte based match is safe here.
    local slang = alien:match( "^(%l+)-" )
    local quotes = quote_override[ alien ] or ( slang and quote_override[ slang ] )
    if not quotes then
        local data = textData()
        local QuoteLang = data.QuoteLang
        local QuoteType = data.QuoteType
        local suite = QuoteLang[alien] or slang and QuoteLang[slang] or QuoteLang["en"]
        if suite then
            quotes = QuoteType[ suite ]
            if not quotes then
                mw.log( "fiatQuote() " .. suite )
            end
        end
    end
    if quotes then
        local space = quotes[ 3 ] and "&#160;" or ""
        local pair = quotes[ advance ]
        if pair then
            r = mw.ustring.format( "%s%s%s%s%s",
                                   mw.ustring.char( pair[ 1 ] ),
                                   space,
                                   apply,
                                   space,
                                   mw.ustring.char( pair[ 2 ] ) )
        end
    end
    return r
end -- fiatQuote()



Text.char = function ( apply, again, accept )
    -- Create string from codepoints
    -- Parameter:
    --     apply   -- table (sequence) with numerical codepoints, or nil
    --     again   -- number of repetitions, or nil
    --     accept  -- true, if no error messages to be appended
    -- Returns: string
    local r = ""
    apply = type(apply) == "table" and apply or {}
    again = math.floor(tonumber(again) or 1)
    if again < 1 then
    	return ""
    end
    local bad   = { }
    local codes = { }
    for _, v in ipairs( apply ) do
    	local n = tonumber(v)
    	if not n or (n < 32 and n ~= 9 and n ~= 10) then
    		table.insert(bad, tostring(v))
    	else
    		table.insert(codes, math.floor(n))
		end
    end 
    if #bad > 0 then
    	if not accept then
    		r = tostring(  mw.html.create( "span" )
                    		:addClass( "error" )
                    		:wikitext( MSG_BAD_CODEPOINTS .. table.concat( bad, " " )) )
    	end
    	return r
    end
    if #codes > 0 then
    	r = mw.ustring.char( unpack( codes ) )
    	if again > 1 then
    		r = r:rep(again)
    	end
	end
    return r
end -- Text.char()

local function trimAndFormat(args, fmt)
	local result = {}
	if type(args) ~= 'table' then
		args = {args}
	end
	for _, v in ipairs(args) do
		v = mw.text.trim(tostring(v))
		if v ~= "" then
			table.insert(result,fmt and mw.ustring.format(fmt, v) or v)
		end
	end
	return result
end

Text.concatParams = function ( args, apply, adapt )
    -- Concat list items into one string
    -- Parameter:
    --     args   -- table (sequence) with numKey=string
    --     apply  -- string (optional); separator (default: "|")
    --     adapt  -- string (optional); format including "%s"
    -- Returns: string
    return table.concat(trimAndFormat(args,adapt), apply or "|")
end -- Text.concatParams()



Text.containsCJK = function ( s )
    -- Is any CJK code within?
    -- Parameter:
    --     s  -- string
    -- Returns: true, if CJK detected
    s = s and tostring(s) or ""
    local patternCJK = textData().PatternCJK
    return mw.ustring.find( s, patternCJK ) ~= nil
end -- Text.containsCJK()

Text.removeDelimited = function (s, prefix, suffix)
	-- Remove all text in s delimited by prefix and suffix (inclusive)
	-- Arguments:
	--    s = string to process
	--    prefix = initial delimiter
	--    suffix = ending delimiter
	-- Returns: stripped string
	s = s and tostring(s) or ""
	prefix = prefix and tostring(prefix) or ""
	suffix = suffix and tostring(suffix) or ""
	-- Byte lengths, because find() and sub() below are byte based. Upstream used
	-- mw.ustring.len here, which mismatched as soon as a delimiter contained a
	-- multi byte character. Plain find on UTF-8 always lands on a character
	-- boundary, so byte slicing at those offsets is safe.
	local prefixLen = #prefix
	local suffixLen = #suffix
	if prefixLen == 0 or suffixLen == 0 then
		return s
	end
	local i = s:find(prefix, 1, true)
	local r = s
	local j
	while i do
		j = r:find(suffix, i + prefixLen, true)
		if j then
			r = r:sub(1, i - 1)..r:sub(j+suffixLen)
		else
			r = r:sub(1, i - 1)
		end
		i = r:find(prefix, 1, true)
	end
	return r
end

Text.getPlain = function ( adjust )
    -- Remove wikisyntax from string, except templates
    -- Parameter:
    --     adjust  -- string
    -- Returns: string
    local r = Text.removeDelimited(adjust,"<!--","-->")
    r = r:gsub( "(</?%l[^>]*>)", "" )
         :gsub( "'''", "" )
         :gsub( "''", "" )
         :gsub( "&nbsp;", " " )
    return r
end -- Text.getPlain()

Text.isLatinRange = function (s)
    -- Are characters expected to be latin or symbols within latin texts?
    -- Arguments:
    --  s = string to analyze
    -- Returns: true, if valid for latin only
    s = s and tostring(s) or ""  --- ensure input is always string
    local PatternLatin = textData().PatternLatin
    return mw.ustring.match(s, PatternLatin) ~= nil
end -- Text.isLatinRange()



Text.isQuote = function ( s )
    -- Is this character any quotation mark?
    -- Parameter:
    --     s = single character to analyze
    -- Returns: true, if s is quotation mark
    s = s and tostring(s) or ""
    if s == "" then
    	return false
    end
    local SeekQuote = textData().SeekQuote
    return mw.ustring.find( SeekQuote, s, 1, true ) ~= nil
end -- Text.isQuote()



Text.listToText = function ( args, adapt )
    -- Format list items similar to mw.text.listToText()
    -- The separator and the final conjunction come from MediaWiki core messages
    -- for the wiki content language, so no localisation is needed here.
    -- Parameter:
    --     args   -- table (sequence) with numKey=string
    --     adapt  -- string (optional); format including "%s"
    -- Returns: string
    return mw.text.listToText(trimAndFormat(args, adapt))
end -- Text.listToText()



Text.quote = function ( apply, alien, advance )
    -- Quote text
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code, or nil
    --     advance  -- number, with level 1 or 2, or nil
    -- Returns: quoted string
    apply = apply and tostring(apply) or ""
    local mode, slang
    if type( alien ) == "string" then
        slang = mw.ustring.lower( mw.text.trim( alien ) )
    else
        slang = mw.title.getCurrentTitle().pageLanguage
        if not slang then
            slang = mw.language.getContentLanguage():getCode()
        end
    end
    if advance == 2 then
        mode = 2
    else
        mode = 1
    end
    return fiatQuote( mw.text.trim( apply ), slang, mode )
end -- Text.quote()



Text.quoteUnquoted = function ( apply, alien, advance )
    -- Quote text, if not yet quoted and not empty
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code, or nil
    --     advance  -- number, with level 1 or 2, or nil
    -- Returns: string; possibly quoted
    local r = mw.text.trim( apply and tostring(apply) or "" )
    local s = mw.ustring.sub( r, 1, 1 )
    if s ~= ""  and  not Text.isQuote( s ) then
        -- Upstream had mw.ustring.sub( r, -1, 1 ), which always returns an empty
        -- string, so the closing mark was never actually tested.
        s = mw.ustring.sub( r, -1 )
        if not Text.isQuote( s ) then
            r = Text.quote( r, alien, advance )
        end
    end
    return r
end -- Text.quoteUnquoted()



Text.removeDiacritics = function ( adjust )
    -- Remove all diacritics
    -- Parameter:
    --     adjust  -- string
    -- Returns: string; all latin letters should be ASCII
    --                  or basic greek or cyrillic or symbols etc.
    local cleanup, decomposed
    local PatternCombined = textData().PatternCombined
    decomposed = mw.ustring.toNFD( adjust and tostring(adjust) or "" )
    cleanup    = mw.ustring.gsub( decomposed, PatternCombined, "" )
    return mw.ustring.toNFC( cleanup )
end -- Text.removeDiacritics()



Text.sentenceTerminated = function ( analyse )
    -- Is string terminated by dot, question or exclamation mark?
    --     Quotation, link termination and so on granted
    -- NOTE: the pattern comes from Module:Text/data and knows only Latin
    -- punctuation. Armenian verjaket, hartsakan nshan and yerkarnshan are not
    -- recognised. See the notes on the talk page before relying on this.
    -- Parameter:
    --     analyse  -- string
    -- Returns: true, if sentence terminated
    local r
    local PatternTerminated = textData().PatternTerminated
    if mw.ustring.find( analyse, PatternTerminated ) then
        r = true
    else
        r = false
    end
    return r
end -- Text.sentenceTerminated()



Text.ucfirstAll = function ( adjust)
    -- Capitalize all words
    -- NOTE: Armenian does not use title case. This is kept for imported
    -- templates that call it; do not wire it into reader facing output.
    -- Arguments:
    --     adjust = string to adjust
    -- Returns: string with all first letters in upper case
    adjust = adjust and tostring(adjust) or ""
    local r = mw.text.decode(adjust,true)
    local i = 1
    local c, j, m
    m = (r ~= adjust)
    r = " "..r
    while i do
        i = mw.ustring.find( r, "%W%l", i )
        if i then
            j = i + 1
            c = mw.ustring.upper( mw.ustring.sub( r, j, j ) )
            r = string.format( "%s%s%s",
                               mw.ustring.sub( r, 1, i ),
                               c,
                               mw.ustring.sub( r, i + 2 ) )
            i = j
        end
    end -- while i
    r = r:sub( 2 )
    if m then
    	r = mw.text.encode(r)
    end
    return r
end -- Text.ucfirstAll()


Text.uprightNonlatin = function ( adjust )
    -- Ensure non-italics for non-latin text parts
    --     One single greek letter might be granted
    -- NOTE: on this wiki the ordinary case is Armenian text with Latin inserts,
    -- which is the reverse of what this function assumes. It will wrap the
    -- Armenian, not the Latin.
    -- Precondition:
    --     adjust  -- string
    -- Returns: string with non-latin parts enclosed in <span>
    local r
    local data = textData()
    local PatternLatin = data.PatternLatin
    local RangesLatin = data.RangesLatin
    local NumLatinRanges = data.NumLatinRanges
    if mw.ustring.match( adjust, PatternLatin ) then
        -- latin only, horizontal dashes, quotes
        r = adjust
    else
        local c
        local j    = false
        local k    = 1
        local m    = false
        local n    = mw.ustring.len( adjust )
        local span = "%s%s<span dir='auto' style='font-style:normal'>%s</span>"
        local flat = function ( a )
                  -- isLatin
                  local range
                  -- NumLatinRanges has to be precomputed because # does not work from loadData
                  for i = 1, NumLatinRanges do
                      range = RangesLatin[ i ]
                      if a >= range[ 1 ]  and  a <= range[ 2 ] then
                          return true
                      end
                  end    -- for i
              end -- flat()
        local focus = function ( a )
                  -- char is not ambivalent
                  local r = ( a > 64 )
                  if r then
                      r = ( a < 8192  or  a > 8212 )
                  else
                      r = ( a == 38  or  a == 60 )    -- '&' '<'
                  end
                  return r
              end -- focus()
        local form = function ( a )
                return string.format( span,
                                      r,
                                      mw.ustring.sub( adjust, k, j - 1 ),
                                      mw.ustring.sub( adjust, j, a ) )
              end -- form()
        r = ""
        for i = 1, n do
            c = mw.ustring.codepoint( adjust, i, i )
            if focus( c ) then
                if flat( c ) then
                    if j then
                        if m then
                            if i == m then
                                -- single greek letter.
                                j = false
                            end
                            m = false
                        end
                        if j then
                            local nx = i - 1
                            local s  = ""
                            for ix = nx, 1, -1 do
                                c = mw.ustring.sub( adjust, ix, ix )
                                if c == " "  or  c == "(" then
                                    nx = nx - 1
                                    s  = c .. s
                                else
                                    break -- for ix
                                end
                            end -- for ix
                            r = form( nx ) .. s
                            j = false
                            k = i
                        end
                    end
                elseif not j then
                    j = i
                    if c >= 880  and  c <= 1023 then
                        -- single greek letter?
                        m = i + 1
                    else
                        m = false
                    end
                end
            elseif m then
                m = m + 1
            end
        end    -- for i
        if j  and  ( not m  or  m < n ) then
            r = form( n )
        else
            r = r .. mw.ustring.sub( adjust, k )
        end
    end
    return r
end -- Text.uprightNonlatin()


Text.test = function ( about )
    local r
    if about == "quote" then
        local data = textData()   -- upstream leaked this as a global
        r = { }
        r.QuoteLang = data.QuoteLang
        r.QuoteType = data.QuoteType
    end
    return r
end -- Text.test()

-- Non Unicode-aware version of mw.text.split and mw.text.gsplit
-- based on [[phab:diffusion/ELUA/browse/master/includes/Engines/LuaCommon/lualib/mw.text.lua]]
-- These run up to 60 times faster than the Unicode-aware versions
--
-- WARNING for this wiki: these are byte based. They are safe with a plain
-- separator or an ASCII pattern such as "," or "%s+", because those always land
-- on a character boundary. They are NOT safe with the letter classes %a %w %l %u
-- against Armenian text, which will split mid character and produce mojibake
-- rather than an error. Use mw.text.split for anything matching Armenian letters.
Text.split = function ( text, pattern, plain )
	local ret = {}
	for m in Text.gsplit( text, pattern, plain ) do
		ret[#ret+1] = m
	end
	return ret
end

Text.gsplit = function ( text, pattern, plain )
	local s, l = 1, string.len( text )
	return function ()
		if s then
			local e, n = string.find( text, pattern, s, plain )
			local ret
			if not e then
				ret = string.sub( text, s )
				s = nil
			elseif n < e then
				-- Empty separator!
				ret = string.sub( text, s, e )
				if e < l then
					s = e + 1
				else
					s = nil
				end
			else
				ret = e > s and string.sub( text, s, e - 1 ) or ''
				s = n + 1
			end
			return ret
		end
	end, nil, nil
end

-- Export
local p = { }

for _, func in ipairs({'containsCJK','isLatinRange','isQuote','sentenceTerminated'}) do
	p[func] = function (frame) 
		return Text[func]( frame.args[ 1 ] or "" ) and "1" or ""
	end
end

for _, func in ipairs({'getPlain','removeDiacritics','ucfirstAll','uprightNonlatin'}) do
	p[func] = function (frame) 
		return Text[func]( frame.args[ 1 ] or "" )
	end
end

function p.char( frame )
    local params = frame:getParent().args
    local story = params[ 1 ]
    local codes, lenient, multiple
    if not story then
        params = frame.args
        story  = params[ 1 ]
    end
    if story then
        local items = mw.text.split( mw.text.trim(story), "%s+" )
        if #items > 0 then
            local j
            lenient  = (yesNo(params.errors) == false)
            codes    = { }
            multiple = tonumber( params[ "*" ] )
            for _, v in ipairs( items ) do
            	j = tonumber((mw.ustring.sub( v, 1, 1 ) == "x" and "0" or "") .. v)
                table.insert( codes,  j or v )
            end 
        end
    end
    return Text.char( codes, multiple, lenient )
end

function p.concatParams( frame )
    local args
    local template = frame.args.template
    if type( template ) == "string" then
        template = mw.text.trim( template )
        template = ( template == "1" )
    end
    if template then
        args = frame:getParent().args
    else
        args = frame.args
    end
    return Text.concatParams( args,
                              frame.args.separator,
                              frame.args.format )
end


function p.listToFormat(frame)
    local lists = {}
    local pformat = frame.args["format"] or ""
    local sep = frame.args["sep"] or ";"

    -- Read the numbered list parameters.
    for k, v in pairs(frame.args) do
        local knum = tonumber(k)
        if knum then lists[knum] = v end
    end

    -- Split each list.
    local maxListLen = 0
    for i = 1, #lists do
        lists[i] = mw.text.split(lists[i], sep)
        if #lists[i] > maxListLen then maxListLen = #lists[i] end
    end

    -- Build the result.
    local result = ""
    local result_line = ""
    for i = 1, maxListLen do
        result_line = pformat
        for j = 1, #lists do
            -- Guard against a short list, and escape % so a percent sign in the
            -- data is not read as a capture reference by gsub.
            local value = (lists[j][i] or ""):gsub("%%", "%%%%")
            result_line = mw.ustring.gsub(result_line, "%%s", value, 1)
        end
        result = result .. result_line
    end

    return result
end



function p.listToText( frame )
    local args
    local template = frame.args.template
    if type( template ) == "string" then
        template = mw.text.trim( template )
        template = ( template == "1" )
    end
    if template then
        args = frame:getParent().args
    else
        args = frame.args
    end
    return Text.listToText( args, frame.args.format )
end



function p.quote( frame )
    local slang = frame.args[2]
    if type( slang ) == "string" then
        slang = mw.text.trim( slang )
        if slang == "" then
            slang = false
        end
    end
    return Text.quote( frame.args[ 1 ] or "",
                       slang,
                       tonumber( frame.args[3] ) )
end



function p.quoteUnquoted( frame )
    local slang = frame.args[2]
    if type( slang ) == "string" then
        slang = mw.text.trim( slang )
        if slang == "" then
            slang = false
        end
    end
    return Text.quoteUnquoted( frame.args[ 1 ] or "",
                               slang,
                               tonumber( frame.args[3] ) )
end


function p.zip(frame)
    local lists = {}
    local seps = {}
    local defaultsep = frame.args["sep"] or ""
    local innersep = frame.args["isep"] or ""
    local outersep = frame.args["osep"] or ""

    -- Read the numbered list parameters and any explicit separators.
    for k, v in pairs(frame.args) do
        local knum = tonumber(k)
        if knum then lists[knum] = v else
            if string.sub(k, 1, 3) == "sep" then
                local sepnum = tonumber(string.sub(k, 4))
                if sepnum then seps[sepnum] = v end
            end
        end
    end
    -- Where no explicit separator was given, use the default.
    for i = 1, math.max(#seps, #lists) do
        if not seps[i] then seps[i] = defaultsep end
    end

    -- Split each list.
    local maxListLen = 0
    for i = 1, #lists do
        lists[i] = mw.text.split(lists[i], seps[i])
        if #lists[i] > maxListLen then maxListLen = #lists[i] end
    end

    local result = ""
    for i = 1, maxListLen do
        if i ~= 1 then result = result .. outersep end
        for j = 1, #lists do
            if j ~= 1 then result = result .. innersep end
            result = result .. (lists[j][i] or "")
        end
    end
    return result
end


function p.split(frame)
	local text = frame.args.text or frame.args[1] or ''
	local pattern = frame.args.pattern or frame.args[2] or ''
	local plain = yesNo(frame.args.plain or frame.args[3])
	local index = tonumber(frame.args.index) or tonumber(frame.args[4]) or 1
	local a = Text.split(text, pattern, plain)
	if index < 0 then index = #a + index + 1 end
	return a[index]
end


function p.failsafe()
    return Text.serial
end


p.Text = function ()
    return Text
end -- p.Text

return p