Modul:Text: Unterschied zwischen den Versionen

Version vom 21. Juli 2022, 18:43 Uhr

Die Dokumentation für dieses Modul kann unter Modul:Text/Doku erstellt werden

local yesNo = require("Module:Yesno")
local Text = { serial = "2022-07-21",
               suite  = "Text" }
--[=[
Text utilities
]=]



-- local globals
local PatternCJK        = false
local PatternCombined   = false
local PatternLatin      = false
local PatternTerminated = false
local QuoteLang         = false
local QuoteType         = false
local RangesLatin       = false
local SeekQuote         = false

local function initLatinData()
    if not RangesLatin then
        RangesLatin = { {    7,  687 },
                        { 7531, 7578 },
                        { 7680, 7935 },
                        { 8194, 8250 } }
    end
    if not PatternLatin then
        local range
        PatternLatin = "^["
        for i = 1, #RangesLatin do
            range = RangesLatin[ i ]
            PatternLatin = PatternLatin ..
                           mw.ustring.char( range[ 1 ], 45, range[ 2 ] )
        end    -- for i
        PatternLatin = PatternLatin .. "]*$"
    end
end

local function initQuoteData()
    -- Create quote definitions
    if not QuoteLang then
    	QuoteLang = 
    	        { af        = "bd",
                  ar        = "la",
                  be        = "labd",
                  bg        = "bd",
                  ca        = "la",
                  cs        = "bd",
                  da        = "bd",
                  de        = "bd",
                  dsb       = "bd",
                  et        = "bd",
                  el        = "lald",
                  en        = "ld",
                  es        = "la",
                  eu        = "la",
            --    fa        = "la",
                  fi        = "rd",
                  fr        = "laSPC",
                  ga        = "ld",
                  he        = "ldla",
                  hr        = "bd",
                  hsb       = "bd",
                  hu        = "bd",
                  hy        = "labd",
                  id        = "rd",
                  is        = "bd",
                  it        = "ld",
                  ja        = "x300C",
                  ka        = "bd",
                  ko        = "ld",
                  lt        = "bd",
                  lv        = "bd",
                  nl        = "ld",
                  nn        = "la",
                  no        = "la",
                  pl        = "bdla",
                  pt        = "lald",
                  ro        = "bdla",
                  ru        = "labd",
                  sk        = "bd",
                  sl        = "bd",
                  sq        = "la",
                  sr        = "bx",
                  sv        = "rd",
                  th        = "ld",
                  tr        = "ld",
                  uk        = "la",
                  zh        = "ld",
                  ["de-ch"] = "la",
                  ["en-gb"] = "lsld",
                  ["en-us"] = "ld",
                  ["fr-ch"] = "la",
                  ["it-ch"] = "la",
                  ["pt-br"] = "ldla",
                  ["zh-tw"] = "x300C",
                  ["zh-cn"] = "ld" }
    end
    if not QuoteType then
    	QuoteType = 
    	        { bd    = { { 8222, 8220 },  { 8218, 8217 } },
                  bdla  = { { 8222, 8220 },  {  171,  187 } },
                  bx    = { { 8222, 8221 },  { 8218, 8217 } },
                  la    = { {  171,  187 },  { 8249, 8250 } },
                  laSPC = { {  171,  187 },  { 8249, 8250 },  true },
                  labd  = { {  171,  187 },  { 8222, 8220 } },
                  lald  = { {  171,  187 },  { 8220, 8221 } },
                  ld    = { { 8220, 8221 },  { 8216, 8217 } },
                  ldla  = { { 8220, 8221 },  {  171,  187 } },
                  lsld  = { { 8216, 8217 },  { 8220, 8221 } },
                  rd    = { { 8221, 8221 },  { 8217, 8217 } },
                  x300C = { { 0x300C, 0x300D },
                            { 0x300E, 0x300F } } }
    end
end -- initQuoteData()



local function fiatQuote( apply, alien, advance )
    -- Quote text
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code
    --     advance  -- number, with level 1 or 2
    local r = apply and tostring(apply) or ""
    alien = alien or "en"
    advance = tonumber(advance) or 0
    local suite
    initQuoteData()
    local slang = alien:match( "^(%l+)-" )
    suite = QuoteLang[alien] or slang and QuoteLang[slang] or QuoteLang["en"]
    if suite then
        local quotes = QuoteType[ suite ]
        if quotes then
            local space
            if quotes[ 3 ] then
                space = "&#160;"
            else
                space = ""
            end
            quotes = quotes[ advance ]
            if quotes then
                r = mw.ustring.format( "%s%s%s%s%s",
                                       mw.ustring.char( quotes[ 1 ] ),
                                       space,
                                       apply,
                                       space,
                                       mw.ustring.char( quotes[ 2 ] ) )
            end
        else
            mw.log( "fiatQuote() " .. suite )
        end
    end
    return r
end -- fiatQuote()



Text.char = function ( apply, again, accept )
    -- Create string from codepoints
    -- Parameter:
    --     apply   -- table (sequence) with numerical codepoints, or nil
    --     again   -- number of repetitions, or nil
    --     accept  -- true, if no error messages to be appended
    -- Returns: string
    local r = ""
    apply = type(apply) == "table" and apply or {}
    again = math.floor(tonumber(again) or 1)
    if again < 1 then
    	return ""
    end
    local bad   = { }
    local codes = { }
    for _, v in ipairs( apply ) do
    	local n = tonumber(v)
    	if not n or (n < 32 and n ~= 9 and n ~= 10) then
    		table.insert(bad, tostring(v))
    	else
    		table.insert(codes, math.floor(n))
		end
    end 
    if #bad > 0 then
    	if not accept then
    		r = tostring(  mw.html.create( "span" )
                    		:addClass( "error" )
                    		:wikitext( "bad codepoints: " .. table.concat( bad, " " )) )
    	end
    	return r
    end
    if #codes > 0 then
    	r = mw.ustring.char( unpack( codes ) )
    	if again > 1 then
    		r = r:rep(again)
    	end
	end
    return r
end -- Text.char()

local function trimAndFormat(args, fmt)
	local result = {}
	if type(args) ~= 'table' then
		args = {args}
	end
	for _, v in ipairs(args) do
		v = mw.text.trim(tostring(v))
		if v ~= "" then
			table.insert(result,fmt and mw.ustring.format(fmt, v) or v)
		end
	end
	return result
end

Text.concatParams = function ( args, apply, adapt )
    -- Concat list items into one string
    -- Parameter:
    --     args   -- table (sequence) with numKey=string
    --     apply  -- string (optional); separator (default: "|")
    --     adapt  -- string (optional); format including "%s"
    -- Returns: string
    local collect = { }
    return table.concat(trimAndFormat(args,adapt), apply or "|")
end -- Text.concatParams()



Text.containsCJK = function ( s )
    -- Is any CJK code within?
    -- Parameter:
    --     s  -- string
    -- Returns: true, if CJK detected
    s = s and tostring(s) or ""
    if not patternCJK then
        patternCJK = mw.ustring.char( 91,
        	                            4352, 45,   4607,
        	                           11904, 45,  42191,
        	                           43072, 45,  43135,
        	                           44032, 45,  55215,
        	                           63744, 45,  64255,
        	                           65072, 45,  65103,
        	                           65381, 45,  65500,
                                      131072, 45, 196607,
                                      93 )
    end
    return mw.ustring.find( s, patternCJK ) ~= nil
end -- Text.containsCJK()

Text.removeDelimited = function (s, prefix, suffix)
	-- Remove all text in s delimited by prefix and suffix (inclusive)
	-- Arguments:
	--    s = string to process
	--    prefix = initial delimiter
	--    suffix = ending delimiter
	-- Returns: stripped string
	s = s and tostring(s) or ""
	prefix = prefix and tostring(prefix) or ""
	suffix = suffix and tostring(suffix) or ""
	local prefixLen = mw.ustring.len(prefix)
	local suffixLen = mw.ustring.len(suffix)
	if prefixLen == 0 or suffixLen == 0 then
		return s
	end
	local i = s:find(prefix, 1, true)
	local r = s
	local j
	while i do
		j = r:find(suffix, i + prefixLen)
		if j then
			r = r:sub(1, i - 1)..r:sub(j+suffixLen)
		else
			r = r:sub(1, i - 1)
		end
		i = r:find(prefix, 1, true)
	end
	return r
end

Text.getPlain = function ( adjust )
    -- Remove wikisyntax from string, except templates
    -- Parameter:
    --     adjust  -- string
    -- Returns: string
    local r = Text.removeDelimited(adjust,"<!--","-->")
    r = r:gsub( "(</?%l[^>]*>)", "" )
         :gsub( "'''", "" )
         :gsub( "''", "" )
         :gsub( "&nbsp;", " " )
    return r
end -- Text.getPlain()

Text.isLatinRange = function (s)
    -- Are characters expected to be latin or symbols within latin texts?
    -- Arguments:
    --  s = string to analyze
    -- Returns: true, if valid for latin only
    s = s and tostring(s) or ""  --- ensure input is always string
    initLatinData()
    return mw.ustring.match(s, PatternLatin) ~= nil
end -- Text.isLatinRange()



Text.isQuote = function ( s )
    -- Is this character any quotation mark?
    -- Parameter:
    --     s = single character to analyze
    -- Returns: true, if s is quotation mark
    s = s and tostring(s) or ""
    if s == "" then
    	return false
    end
    if not SeekQuote then
        SeekQuote = mw.ustring.char(   34,       -- "
                                       39,       -- '
                                      171,       -- laquo
                                      187,       -- raquo
                                     8216,       -- lsquo
                                     8217,       -- rsquo
                                     8218,       -- sbquo
                                     8220,       -- ldquo
                                     8221,       -- rdquo
                                     8222,       -- bdquo
                                     8249,       -- lsaquo
                                     8250,       -- rsaquo
                                     0x300C,     -- CJK
                                     0x300D,     -- CJK
                                     0x300E,     -- CJK
                                     0x300F )    -- CJK
    end
    return mw.ustring.find( SeekQuote, s, 1, true ) ~= nil
end -- Text.isQuote()



Text.listToText = function ( args, adapt )
    -- Format list items similar to mw.text.listToText()
    -- Parameter:
    --     args   -- table (sequence) with numKey=string
    --     adapt  -- string (optional); format including "%s"
    -- Returns: string
    return mw.text.listToText(trimAndFormat(args, adapt))
end -- Text.listToText()



Text.quote = function ( apply, alien, advance )
    -- Quote text
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code, or nil
    --     advance  -- number, with level 1 or 2, or nil
    -- Returns: quoted string
    apply = apply and tostring(apply) or ""
    local mode, slang
    if type( alien ) == "string" then
        slang = mw.text.trim( alien ):lower()
    else
        slang = mw.title.getCurrentTitle().pageLanguage
        if not slang then
            -- TODO FIXME: Introduction expected 2017-04
            slang = mw.language.getContentLanguage():getCode()
        end
    end
    if advance == 2 then
        mode = 2
    else
        mode = 1
    end
    return fiatQuote( mw.text.trim( apply ), slang, mode )
end -- Text.quote()



Text.quoteUnquoted = function ( apply, alien, advance )
    -- Quote text, if not yet quoted and not empty
    -- Parameter:
    --     apply    -- string, with text
    --     alien    -- string, with language code, or nil
    --     advance  -- number, with level 1 or 2, or nil
    -- Returns: string; possibly quoted
    local r = mw.text.trim( apply and tostring(apply) or "" )
    local s = mw.ustring.sub( r, 1, 1 )
    if s ~= ""  and  not Text.isQuote( s, advance ) then
        s = mw.ustring.sub( r, -1, 1 )
        if not Text.isQuote( s ) then
            r = Text.quote( r, alien, advance )
        end
    end
    return r
end -- Text.quoteUnquoted()



Text.removeDiacritics = function ( adjust )
    -- Remove all diacritics
    -- Parameter:
    --     adjust  -- string
    -- Returns: string; all latin letters should be ASCII
    --                  or basic greek or cyrillic or symbols etc.
    local cleanup, decomposed
    if not PatternCombined then
        PatternCombined = mw.ustring.char( 91,
                                            0x0300, 45, 0x036F,
                                            0x1AB0, 45, 0x1AFF,
                                            0x1DC0, 45, 0x1DFF,
                                            0xFE20, 45, 0xFE2F,
                                           93 )
    end
    decomposed = mw.ustring.toNFD( adjust and tostring(adjust) or "" )
    cleanup    = mw.ustring.gsub( decomposed, PatternCombined, "" )
    return mw.ustring.toNFC( cleanup )
end -- Text.removeDiacritics()



Text.sentenceTerminated = function ( analyse )
    -- Is string terminated by dot, question or exclamation mark?
    --     Quotation, link termination and so on granted
    -- Parameter:
    --     analyse  -- string
    -- Returns: true, if sentence terminated
    local r
    if not PatternTerminated then
        PatternTerminated = mw.ustring.char( 91,
                                             12290,
                                             65281,
                                             65294,
                                             65311 )
                            .. "!%.%?…][\"'%]‹›«»‘’“”]*$"
    end
    if mw.ustring.find( analyse, PatternTerminated ) then
        r = true
    else
        r = false
    end
    return r
end -- Text.sentenceTerminated()



Text.ucfirstAll = function ( adjust)
    -- Capitalize all words
    -- Arguments:
    --     adjust = string to adjust
    -- Returns: string with all first letters in upper case
    adjust = adjust and tostring(adjust) or ""
    local r = mw.text.decode(adjust,true)
    local i = 1
    local c, j, m
    m = (r ~= adjust)
    r = " "..r
    while i do
        i = mw.ustring.find( r, "%W%l", i )
        if i then
            j = i + 1
            c = mw.ustring.upper( mw.ustring.sub( r, j, j ) )
            r = string.format( "%s%s%s",
                               mw.ustring.sub( r, 1, i ),
                               c,
                               mw.ustring.sub( r, i + 2 ) )
            i = j
        end
    end -- while i
    r = r:sub( 2 )
    if m then
    	r = mw.text.encode(r)
    end
    return r
end -- Text.ucfirstAll()


Text.uprightNonlatin = function ( adjust )
    -- Ensure non-italics for non-latin text parts
    --     One single greek letter might be granted
    -- Precondition:
    --     adjust  -- string
    -- Returns: string with non-latin parts enclosed in <span>
    local r
    initLatinData()
    if mw.ustring.match( adjust, PatternLatin ) then
        -- latin only, horizontal dashes, quotes
        r = adjust
    else
        local c
        local j    = false
        local k    = 1
        local m    = false
        local n    = mw.ustring.len( adjust )
        local span = "%s%s<span dir='auto' style='font-style:normal'>%s</span>"
        local flat = function ( a )
                  -- isLatin
                  local range
                  for i = 1, #RangesLatin do
                      range = RangesLatin[ i ]
                      if a >= range[ 1 ]  and  a <= range[ 2 ] then
                          return true
                      end
                  end    -- for i
              end -- flat()
        local focus = function ( a )
                  -- char is not ambivalent
                  local r = ( a > 64 )
                  if r then
                      r = ( a < 8192  or  a > 8212 )
                  else
                      r = ( a == 38  or  a == 60 )    -- '&' '<'
                  end
                  return r
              end -- focus()
        local form = function ( a )
                return string.format( span,
                                      r,
                                      mw.ustring.sub( adjust, k, j - 1 ),
                                      mw.ustring.sub( adjust, j, a ) )
              end -- form()
        r = ""
        for i = 1, n do
            c = mw.ustring.codepoint( adjust, i, i )
            if focus( c ) then
                if flat( c ) then
                    if j then
                        if m then
                            if i == m then
                                -- single greek letter.
                                j = false
                            end
                            m = false
                        end
                        if j then
                            local nx = i - 1
                            local s  = ""
                            for ix = nx, 1, -1 do
                                c = mw.ustring.sub( adjust, ix, ix )
                                if c == " "  or  c == "(" then
                                    nx = nx - 1
                                    s  = c .. s
                                else
                                    break -- for ix
                                end
                            end -- for ix
                            r = form( nx ) .. s
                            j = false
                            k = i
                        end
                    end
                elseif not j then
                    j = i
                    if c >= 880  and  c <= 1023 then
                        -- single greek letter?
                        m = i + 1
                    else
                        m = false
                    end
                end
            elseif m then
                m = m + 1
            end
        end    -- for i
        if j  and  ( not m  or  m < n ) then
            r = form( n )
        else
            r = r .. mw.ustring.sub( adjust, k )
        end
    end
    return r
end -- Text.uprightNonlatin()


Text.test = function ( about )
    local r
    if about == "quote" then
        initQuoteData()
        r = { }
        r.QuoteLang = QuoteLang
        r.QuoteType = QuoteType
    end
    return r
end -- Text.test()



-- Export
local p = { }

for _, func in ipairs({'containsCJK','isLatinRange','isQuote','sentenceTerminated'}) do
	p[func] = function (frame) 
		return Text[func]( frame.args[ 1 ] or "" ) and "1" or ""
	end
end

for _, func in ipairs({'getPlain','removeDiacritics','ucfirstAll','uprightNonlatin'}) do
	p[func] = function (frame) 
		return Text[func]( frame.args[ 1 ] or "" )
	end
end

function p.char( frame )
    local params = frame:getParent().args
    local story = params[ 1 ]
    local codes, lenient, multiple
    if not story then
        params = frame.args
        story  = params[ 1 ]
    end
    if story then
        local items = mw.text.split( mw.text.trim(story), "%s+" )
        if #items > 0 then
            local j
            lenient  = (yesNo(params.errors) == false)
            codes    = { }
            multiple = tonumber( params[ "*" ] )
            for _, v in ipairs( items ) do
            	j = tonumber((v:sub( 1, 1 ) == "x" and "0" or "") .. v)
                table.insert( codes,  j or v )
            end 
        end
    end
    return Text.char( codes, multiple, lenient )
end

function p.concatParams( frame )
    local args
    local template = frame.args.template
    if type( template ) == "string" then
        template = mw.text.trim( template )
        template = ( template == "1" )
    end
    if template then
        args = frame:getParent().args
    else
        args = frame.args
    end
    return Text.concatParams( args,
                              frame.args.separator,
                              frame.args.format )
end


function p.listToFormat(frame)
    local lists = {}
    local pformat = frame.args["format"]
    local sep = frame.args["sep"] or ";"

    -- Parameter parsen: Listen
    for k, v in pairs(frame.args) do
        local knum = tonumber(k)
        if knum then lists[knum] = v end
    end

    -- Listen splitten
    local maxListLen = 0
    for i = 1, #lists do
        lists[i] = mw.text.split(lists[i], sep)
        if #lists[i] > maxListLen then maxListLen = #lists[i] end
    end

    -- Ergebnisstring generieren
    local result = ""
    local result_line = ""
    for i = 1, maxListLen do
        result_line = pformat
        for j = 1, #lists do
            result_line = mw.ustring.gsub(result_line, "%%s", lists[j][i], 1)
        end
        result = result .. result_line
    end

    return result
end



function p.listToText( frame )
    local args
    local template = frame.args.template
    if type( template ) == "string" then
        template = mw.text.trim( template )
        template = ( template == "1" )
    end
    if template then
        args = frame:getParent().args
    else
        args = frame.args
    end
    return Text.listToText( args, frame.args.format )
end



function p.quote( frame )
    local slang = frame.args[2]
    if type( slang ) == "string" then
        slang = mw.text.trim( slang )
        if slang == "" then
            slang = false
        end
    end
    return Text.quote( frame.args[ 1 ] or "",
                       slang,
                       tonumber( frame.args[3] ) )
end



function p.quoteUnquoted( frame )
    local slang = frame.args[2]
    if type( slang ) == "string" then
        slang = mw.text.trim( slang )
        if slang == "" then
            slang = false
        end
    end
    return Text.quoteUnquoted( frame.args[ 1 ] or "",
                               slang,
                               tonumber( frame.args[3] ) )
end


function p.zip(frame)
    local lists = {}
    local seps = {}
    local defaultsep = frame.args["sep"] or ""
    local innersep = frame.args["isep"] or ""
    local outersep = frame.args["osep"] or ""

    -- Parameter parsen
    for k, v in pairs(frame.args) do
        local knum = tonumber(k)
        if knum then lists[knum] = v else
            if string.sub(k, 1, 3) == "sep" then
                local sepnum = tonumber(string.sub(k, 4))
                if sepnum then seps[sepnum] = v end
            end
        end
    end
    -- sofern keine expliziten Separatoren angegeben sind, den Standardseparator verwenden
    for i = 1, math.max(#seps, #lists) do
        if not seps[i] then seps[i] = defaultsep end
    end

    -- Listen splitten
    local maxListLen = 0
    for i = 1, #lists do
        lists[i] = mw.text.split(lists[i], seps[i])
        if #lists[i] > maxListLen then maxListLen = #lists[i] end
    end

    local result = ""
    for i = 1, maxListLen do
        if i ~= 1 then result = result .. outersep end
        for j = 1, #lists do
            if j ~= 1 then result = result .. innersep end
            result = result .. (lists[j][i] or "")
        end
    end
    return result
end



function p.failsafe()
    return Text.serial
end



p.Text = function ()
    return Text
end -- p.Text

return p

@@ Zeile 1: / Zeile 1: @@
-local Text = { serial = "2019-11-12",
+local yesNo = require("Module:Yesno")
-                suite  = "Text",
+local Text = { serial = "2022-07-21",
-               item   = 29387871 }
+                suite  = "Text" }
 --[=[
 Text utilities
 ]=]
-local Failsafe  = Text
-local GlobalMod = Text
 -- local globals
@@ Zeile 13: / Zeile 13: @@
 local PatternLatin      = false
 local PatternTerminated = false
+local QuoteLang         = false
+local QuoteType         = false
 local RangesLatin       = false
 local SeekQuote         = false
+local function initLatinData()
+     if not RangesLatin then
-local foreignModule = function ( access, advanced, append, alt, alert )
+        RangesLatin = { {    7,  687 },
-     -- Fetch global module
+                        { 7531, 7578 },
-    -- Precondition:
+                        { 7680, 7935 },
-    --     access    -- string, with name of base module
+                        { 8194, 8250 } }
-    --     advanced  -- true, for require(); else mw.loadData()
-    --     append    -- string, with subpage part, if any; or false
-    --     alt       -- number, of wikidata item of root; or false
-    --     alert     -- true, for throwing error on data problem
-    -- Postcondition:
-    --     Returns whatever, probably table
-    -- 2019-10-29
-    local storage = access
-    local finer = function ()
-                      if append then
-                          storage = string.format( "%s/%s",
-                                                   storage,
-                                                   append )
-                      end
-                  end
-    local fun, lucky, r, suited
-    if advanced then
-        fun = require
-    else
-        fun = mw.loadData
      end
-     GlobalMod.globalModules = GlobalMod.globalModules or { }
+     if not PatternLatin then
-    suited = GlobalMod.globalModules[ access ]
+        local range
-    if not suited then
+        PatternLatin = "^["
-        finer()
+        for i = 1, #RangesLatin do
-         lucky, r = pcall( fun,  "Module:" .. storage )
+            range = RangesLatin[ i ]
+            PatternLatin = PatternLatin ..
+                           mw.ustring.char( range[ 1 ], 45, range[ 2 ] )
+         end    -- for i
+        PatternLatin = PatternLatin .. "]*$"
      end
-    if not lucky then
+end
-        if not suited  and
-           type( alt ) == "number"  and
-           alt > 0 then
-            suited = string.format( "Q%d", alt )
-            suited = mw.wikibase.getSitelink( suited )
-            GlobalMod.globalModules[ access ] = suited or true
-        end
-        if type( suited ) == "string" then
-            storage = suited
-            finer()
-            lucky, r = pcall( fun, storage )
-        end
-        if not lucky and alert then
-            error( "Missing or invalid page: " .. storage, 0 )
-        end
-    end
-    return r
-end -- foreignModule()
+local function initQuoteData()
-local function factoryQuote()
      -- Create quote definitions
-     if not Text.quoteLang then
+     if not QuoteLang then
-        local quoting = foreignModule( "Text",
+    	QuoteLang =
-                                       false,
+    	        { af        = "bd",
-                                       "quoting",
+                  ar        = "la",
-                                       Text.item )
+                  be        = "labd",
-        if type( quoting ) == "table" then
+                  bg        = "bd",
-             Text.quoteLang = quoting.langs
+                  ca        = "la",
-            Text.quoteType = quoting.types
+                  cs        = "bd",
-        end
+                  da        = "bd",
-        if type( Text.quoteLang ) ~= "table" then
+                  de        = "bd",
-            Text.quoteLang = { }
+                  dsb       = "bd",
-        end
+                  et        = "bd",
-        if type( Text.quoteType ) ~= "table" then
+                  el        = "lald",
-            Text.quoteType = { }
+                  en        = "ld",
-        end
+                  es        = "la",
-        if type( Text.quoteLang.en ) ~= "string" then
+                  eu        = "la",
-            Text.quoteLang.en = "ld"
+             --    fa        = "la",
-        end
+                  fi        = "rd",
-        if type( Text.quoteType[ Text.quoteLang.en ] ) ~= "table" then
+                  fr        = "laSPC",
-            Text.quoteType[ Text.quoteLang.en ] = { { 8220, 8221 },
+                  ga        = "ld",
-                                                    { 8216, 8217 } }
+                  he        = "ldla",
-        end
+                  hr        = "bd",
+                  hsb       = "bd",
+                  hu        = "bd",
+                  hy        = "labd",
+                  id        = "rd",
+                  is        = "bd",
+                  it        = "ld",
+                  ja        = "x300C",
+                  ka        = "bd",
+                  ko        = "ld",
+                  lt        = "bd",
+                  lv        = "bd",
+                  nl        = "ld",
+                  nn        = "la",
+                  no        = "la",
+                  pl        = "bdla",
+                  pt        = "lald",
+                  ro        = "bdla",
+                  ru        = "labd",
+                  sk        = "bd",
+                  sl        = "bd",
+                  sq        = "la",
+                  sr        = "bx",
+                  sv        = "rd",
+                  th        = "ld",
+                  tr        = "ld",
+                  uk        = "la",
+                  zh        = "ld",
+                  ["de-ch"] = "la",
+                  ["en-gb"] = "lsld",
+                  ["en-us"] = "ld",
+                  ["fr-ch"] = "la",
+                  ["it-ch"] = "la",
+                  ["pt-br"] = "ldla",
+                  ["zh-tw"] = "x300C",
+                  ["zh-cn"] = "ld" }
+    end
+    if not QuoteType then
+    	QuoteType =
+    	        { bd    = { { 8222, 8220 },  { 8218, 8217 } },
+                  bdla  = { { 8222, 8220 },  {  171,  187 } },
+                  bx    = { { 8222, 8221 },  { 8218, 8217 } },
+                  la    = { {  171,  187 },  { 8249, 8250 } },
+                  laSPC = { {  171,  187 },  { 8249, 8250 },  true },
+                  labd  = { {  171,  187 },  { 8222, 8220 } },
+                  lald  = { {  171,  187 },  { 8220, 8221 } },
+                  ld    = { { 8220, 8221 },  { 8216, 8217 } },
+                  ldla  = { { 8220, 8221 },  {  171,  187 } },
+                  lsld  = { { 8216, 8217 },  { 8220, 8221 } },
+                  rd    = { { 8221, 8221 },  { 8217, 8217 } },
+                  x300C = { { 0x300C, 0x300D },
+                            { 0x300E, 0x300F } } }
      end
-end -- factoryQuote()
+end -- initQuoteData()
@@ Zeile 106: / Zeile 123: @@
      --     alien    -- string, with language code
      --     advance  -- number, with level 1 or 2
-     local r = apply
+     local r = apply and tostring(apply) or ""
+    alien = alien or "en"
+    advance = tonumber(advance) or 0
      local suite
-     factoryQuote()
+     initQuoteData()
-     suite = Text.quoteLang[ alien ]
+     local slang = alien:match( "^(%l+)-" )
-    if not suite then
+    suite = QuoteLang[alien] or slang and QuoteLang[slang] or QuoteLang["en"]
-        local slang = alien:match( "^(%l+)-" )
-        if slang then
-            suite = Text.quoteLang[ slang ]
-        end
-        if not suite then
-            suite = Text.quoteLang.en
-        end
-    end
      if suite then
-         local quotes = Text.quoteType[ suite ]
+         local quotes = QuoteType[ suite ]
          if quotes then
              local space
@@ Zeile 153: / Zeile 164: @@
      --     accept  -- true, if no error messages to be appended
      -- Returns: string
-     local r
+     local r = ""
-     if type( apply ) == "table" then
+     apply = type(apply) == "table" and apply or {}
-        local bad   = { }
+    again = math.floor(tonumber(again) or 1)
-        local codes = { }
+    if again < 1 then
-        local s
+    	return ""
-        for k, v in pairs( apply ) do
+    end
-            s = type( v )
+    local bad   = { }
-            if s == "number" then
+    local codes = { }
-                if v < 32  and  v ~= 9  and  v ~= 10 then
+    for _, v in ipairs( apply ) do
-                    v = tostring( v )
+    	local n = tonumber(v)
-                else
+    	if not n or (n < 32 and n ~= 9 and n ~= 10) then
-                    v = math.floor( v )
+    		table.insert(bad, tostring(v))
-                    s = false
+    	else
-                end
+    		table.insert(codes, math.floor(n))
-            elseif s ~= "string" then
+		end
-                v = tostring( v )
+    end
-            end
+    if #bad > 0 then
-            if s then
+    	if not accept then
-                table.insert( bad, v )
+    		r = tostring(  mw.html.create( "span" )
-            else
+                    		:addClass( "error" )
-                table.insert( codes, v )
+                    		:wikitext( "bad codepoints: " .. table.concat( bad, " " )) )
-            end
+    	end
-        end -- for k, v
+    	return r
-        if #bad == 0 then
-            if #codes > 0 then
-                r = mw.ustring.char( unpack( codes ) )
-                if again then
-                    if type( again ) == "number" then
-                        local n = math.floor( again )
-                        if n > 1 then
-                            r = r:rep( n )
-                        elseif n < 1 then
-                            r = ""
-                        end
-                    else
-                        s = "bad repetitions: " .. tostring( again )
-                    end
-                end
-            end
-        else
-            s = "bad codepoints: " .. table.concat( bad, " " )
-        end
-        if s  and  not accept then
-            r = tostring(  mw.html.create( "span" )
-                                  :addClass( "error" )
-                                  :wikitext( s ) )
-        end
      end
-     return r or ""
+    if #codes > 0 then
+    	r = mw.ustring.char( unpack( codes ) )
+    	if again > 1 then
+    		r = r:rep(again)
+    	end
+	end
+     return r
 end -- Text.char()
+local function trimAndFormat(args, fmt)
+	local result = {}
+	if type(args) ~= 'table' then
+		args = {args}
+	end
+	for _, v in ipairs(args) do
+		v = mw.text.trim(tostring(v))
+		if v ~= "" then
+			table.insert(result,fmt and mw.ustring.format(fmt, v) or v)
+		end
+	end
+	return result
+end
 Text.concatParams = function ( args, apply, adapt )
@@ Zeile 214: / Zeile 219: @@
      -- Returns: string
      local collect = { }
-     for k, v in pairs( args ) do
+     return table.concat(trimAndFormat(args,adapt), apply or "|")
-        if type( k ) == "number" then
-            v = mw.text.trim( v )
-            if v ~= "" then
-                if adapt then
-                    v = mw.ustring.format( adapt, v )
-                end
-                table.insert( collect, v )
-            end
-        end
-    end -- for k, v
-    return table.concat( collect,  apply or "|" )
 end -- Text.concatParams()
-Text.containsCJK = function ( analyse )
+Text.containsCJK = function ( s )
      -- Is any CJK code within?
      -- Parameter:
-     --     analyse  -- string
+     --     s  -- string
      -- Returns: true, if CJK detected
-     local r
+     s = s and tostring(s) or ""
-     if not PatternCJK then
+     if not patternCJK then
-         PatternCJK = mw.ustring.char( 91,
+         patternCJK = mw.ustring.char( 91,
-, 45,  40959,
+, 45,   4607,
-, 45, 178207,
+, 45,  42191,
+, 45,  43135,
+, 45,  55215,
+, 45,  64255,
+, 45,  65103,
+, 45,  65500,
+, 45, 196607,
 )
      end
-     if mw.ustring.find( analyse, PatternCJK ) then
+     return mw.ustring.find( s, patternCJK ) ~= nil
-        r = true
-    else
-        r = false
-    end
-    return r
 end -- Text.containsCJK()
+Text.removeDelimited = function (s, prefix, suffix)
+	-- Remove all text in s delimited by prefix and suffix (inclusive)
+	-- Arguments:
+	--    s = string to process
+	--    prefix = initial delimiter
+	--    suffix = ending delimiter
+	-- Returns: stripped string
+	s = s and tostring(s) or ""
+	prefix = prefix and tostring(prefix) or ""
+	suffix = suffix and tostring(suffix) or ""
+	local prefixLen = mw.ustring.len(prefix)
+	local suffixLen = mw.ustring.len(suffix)
+	if prefixLen == 0 or suffixLen == 0 then
+		return s
+	end
+	local i = s:find(prefix, 1, true)
+	local r = s
+	local j
+	while i do
+		j = r:find(suffix, i + prefixLen)
+		if j then
+			r = r:sub(1, i - 1)..r:sub(j+suffixLen)
+		else
+			r = r:sub(1, i - 1)
+		end
+		i = r:find(prefix, 1, true)
+	end
+	return r
+end
 Text.getPlain = function ( adjust )
@@ Zeile 257: / Zeile 280: @@
      --     adjust  -- string
      -- Returns: string
-     local i = adjust:find( "<!--", 1, true )
+     local r = Text.removeDelimited(adjust,"<!--","-->")
-    local r = adjust
-    local j
-    while i do
-        j = r:find( "-->",  i + 3,  true )
-        if j then
-            r = r:sub( 1, i ) .. r:sub( j + 3 )
-        else
-            r = r:sub( 1, i )
-        end
-        i = r:find( "<!--", i, true )
-    end    -- "<!--"
      r = r:gsub( "(</?%l[^>]*>)", "" )
-          :gsub( "'''(.+)'''", "%1" )
+          :gsub( "'''", "" )
-          :gsub( "''(.+)''", "%1" )
+          :gsub( "''", "" )
           :gsub( "&nbsp;", " " )
-     return mw.text.unstrip( r )
+     return r
 end -- Text.getPlain()
+Text.isLatinRange = function (s)
-Text.isLatinRange = function ( adjust )
      -- Are characters expected to be latin or symbols within latin texts?
-     -- Precondition:
+     -- Arguments:
-     --     adjust  -- string, or nil for initialization
+     --  s = string to analyze
      -- Returns: true, if valid for latin only
-     local r
+     s = s and tostring(s) or ""  --- ensure input is always string
-    if not RangesLatin then
+     initLatinData()
-        RangesLatin = { {    7,  687 },
+     return mw.ustring.match(s, PatternLatin) ~= nil
-                        { 7531, 7578 },
-                        { 7680, 7935 },
-                        { 8194, 8250 } }
-    end
-    if not PatternLatin then
-        local range
-        PatternLatin = "^["
-        for i = 1, #RangesLatin do
-            range = RangesLatin[ i ]
-            PatternLatin = PatternLatin ..
-                           mw.ustring.char( range[ 1 ], 45, range[ 2 ] )
-        end    -- for i
-        PatternLatin = PatternLatin .. "]*$"
-     end
-     if adjust then
-        if mw.ustring.match( adjust, PatternLatin ) then
-            r = true
-        else
-            r = false
-        end
-    end
-    return r
 end -- Text.isLatinRange()
-Text.isQuote = function ( ask )
+Text.isQuote = function ( s )
      -- Is this character any quotation mark?
      -- Parameter:
-     --     ask  -- string, with single character
+     --     s = single character to analyze
-     -- Returns: true, if ask is quotation mark
+     -- Returns: true, if s is quotation mark
-     local r
+     s = s and tostring(s) or ""
+    if s == "" then
+    	return false
+    end
      if not SeekQuote then
          SeekQuote = mw.ustring.char(   34,       -- "
@@ Zeile 336: / Zeile 327: @@
 x300F )    -- CJK
      end
-     if ask == "" then
+     return mw.ustring.find( SeekQuote, s, 1, true ) ~= nil
-        r = false
-    elseif mw.ustring.find( SeekQuote, ask, 1, true ) then
-        r = true
-    else
-        r = false
-    end
-    return r
 end -- Text.isQuote()
@@ Zeile 354: / Zeile 338: @@
      --     adapt  -- string (optional); format including "%s"
      -- Returns: string
-     local collect = { }
+     return mw.text.listToText(trimAndFormat(args, adapt))
-    for k, v in pairs( args ) do
-        if type( k ) == "number" then
-            v = mw.text.trim( v )
-            if v ~= "" then
-                if adapt then
-                    v = mw.ustring.format( adapt, v )
-                end
-                table.insert( collect, v )
-            end
-        end
-    end -- for k, v
-    return mw.text.listToText( collect )
 end -- Text.listToText()
@@ Zeile 378: / Zeile 350: @@
      --     advance  -- number, with level 1 or 2, or nil
      -- Returns: quoted string
+    apply = apply and tostring(apply) or ""
      local mode, slang
      if type( alien ) == "string" then
@@ Zeile 405: / Zeile 378: @@
      --     advance  -- number, with level 1 or 2, or nil
      -- Returns: string; possibly quoted
-     local r = mw.text.trim( apply )
+     local r = mw.text.trim( apply and tostring(apply) or "" )
      local s = mw.ustring.sub( r, 1, 1 )
      if s ~= ""  and  not Text.isQuote( s, advance ) then
@@ Zeile 433: / Zeile 406: @@
 )
      end
-     decomposed = mw.ustring.toNFD( adjust )
+     decomposed = mw.ustring.toNFD( adjust and tostring(adjust) or "" )
      cleanup    = mw.ustring.gsub( decomposed, PatternCombined, "" )
      return mw.ustring.toNFC( cleanup )
@@ Zeile 446: / Zeile 419: @@
      --     analyse  -- string
      -- Returns: true, if sentence terminated
-     local r = mw.text.trim( analyse )
+     local r
      if not PatternTerminated then
          PatternTerminated = mw.ustring.char( 91,
@@ Zeile 455: / Zeile 428: @@
                              .. "!%.%?…][\"'%]‹›«»‘’“”]*$"
      end
-     if mw.ustring.find( r, PatternTerminated ) then
+     if mw.ustring.find( analyse, PatternTerminated ) then
          r = true
      else
@@ Zeile 465: / Zeile 438: @@
-Text.ucfirstAll = function ( adjust )
+Text.ucfirstAll = function ( adjust)
      -- Capitalize all words
-     -- Precondition:
+     -- Arguments:
-     --     adjust  -- string
+     --     adjust = string to adjust
      -- Returns: string with all first letters in upper case
-     local r = " " .. adjust
+    adjust = adjust and tostring(adjust) or ""
+     local r = mw.text.decode(adjust,true)
      local i = 1
      local c, j, m
-     if adjust:find( "&" ) then
+     m = (r ~= adjust)
-        r = r:gsub( "&amp;",      "&#38;" )
+    r = " "..r
-             :gsub( "&lt;",       "&#60;" )
-             :gsub( "&gt;",       "&#62;" )
-             :gsub( "&nbsp;",    "&#160;" )
-             :gsub( "&thinsp;", "&#8201;" )
-             :gsub( "&zwnj;",   "&#8204;" )
-             :gsub( "&zwj;",    "&#8205;" )
-             :gsub( "&lrm;",    "&#8206;" )
-             :gsub( "&rlm;",    "&#8207;" )
-        m = true
-    end
      while i do
          i = mw.ustring.find( r, "%W%l", i )
@@ Zeile 499: / Zeile 463: @@
      r = r:sub( 2 )
      if m then
-        r = r:gsub(     "&#38;", "&amp;" )
+    	r = mw.text.encode(r)
-             :gsub(     "&#60;", "&lt;" )
-             :gsub(     "&#62;", "&gt;" )
-             :gsub(    "&#160;", "&nbsp;" )
-             :gsub(   "&#8201;", "&thinsp;" )
-             :gsub(   "&#8204;", "&zwnj;" )
-             :gsub(   "&#8205;", "&zwj;" )
-             :gsub(   "&#8206;", "&lrm;" )
-             :gsub(   "&#8207;", "&rlm;" )
-             :gsub( "&#X(%x+);", "&#x%1;" )
      end
      return r
 end -- Text.ucfirstAll()
@@ Zeile 522: / Zeile 476: @@
      -- Returns: string with non-latin parts enclosed in <span>
      local r
-     Text.isLatinRange()
+     initLatinData()
      if mw.ustring.match( adjust, PatternLatin ) then
          -- latin only, horizontal dashes, quotes
@@ Zeile 610: / Zeile 564: @@
      return r
 end -- Text.uprightNonlatin()
-Failsafe.failsafe = function ( atleast )
-    -- Retrieve versioning and check for compliance
-    -- Precondition:
-    --     atleast  -- string, with required version or "wikidata" or "~"
-    --                 or false
-    -- Postcondition:
-    --     Returns  string  -- with queried version, also if problem
-    --              false   -- if appropriate
-    -- 2019-10-15
-    local last  = ( atleast == "~" )
-    local since = atleast
-    local r
-    if last  or  since == "wikidata" then
-        local item = Failsafe.item
-        since = false
-        if type( item ) == "number"  and  item > 0 then
-            local entity = mw.wikibase.getEntity( string.format( "Q%d",
-                                                                 item ) )
-            if type( entity ) == "table" then
-                local seek = Failsafe.serialProperty or "P348"
-                local vsn  = entity:formatPropertyValues( seek )
-                if type( vsn ) == "table"  and
-                   type( vsn.value ) == "string"  and
-                   vsn.value ~= "" then
-                    if last  and  vsn.value == Failsafe.serial then
-                        r = false
-                    else
-                        r = vsn.value
-                    end
-                end
-            end
-        end
-    end
-    if type( r ) == "nil" then
-        if not since  or  since <= Failsafe.serial then
-            r = Failsafe.serial
-        else
-            r = false
-        end
-    end
-    return r
-end -- Failsafe.failsafe()
@@ Zeile 661: / Zeile 569: @@
      local r
      if about == "quote" then
-         factoryQuote()
+         initQuoteData()
-         r = { QuoteLang = Text.quoteLang,
+         r = { }
-              QuoteType = Text.quoteType }
+        r.QuoteLang = QuoteLang
+        r.QuoteType = QuoteType
      end
      return r
@@ Zeile 672: / Zeile 581: @@
 -- Export
 local p = { }
+for _, func in ipairs({'containsCJK','isLatinRange','isQuote','sentenceTerminated'}) do
+	p[func] = function (frame)
+		return Text[func]( frame.args[ 1 ] or "" ) and "1" or ""
+	end
+end
+for _, func in ipairs({'getPlain','removeDiacritics','ucfirstAll','uprightNonlatin'}) do
+	p[func] = function (frame)
+		return Text[func]( frame.args[ 1 ] or "" )
+	end
+end
 function p.char( frame )
@@ Zeile 682: / Zeile 603: @@
      end
      if story then
-         local items = mw.text.split( story, "%s+" )
+         local items = mw.text.split( mw.text.trim(story), "%s+" )
          if #items > 0 then
              local j
-             lenient  = ( params.errors == "0" )
+             lenient  = (yesNo(params.errors) == false)
              codes    = { }
              multiple = tonumber( params[ "*" ] )
-             for k, v in pairs( items ) do
+             for _, v in ipairs( items ) do
-                if v:sub( 1, 1 ) == "x" then
+            	j = tonumber((v:sub( 1, 1 ) == "x" and "0" or "") .. v)
-                    j = tonumber( "0" .. v )
+                 table.insert( codes,  j or v )
-                 elseif v == "" then
+             end
-                    v = false
-                else
-                    j = tonumber( v )
-                end
-                if v then
-                    table.insert( codes,  j or v )
-                end
-             end -- for k, v
          end
      end
@@ Zeile 720: / Zeile 633: @@
                                frame.args.separator,
                                frame.args.format )
-end
-function p.containsCJK( frame )
-    return Text.containsCJK( frame.args[ 1 ] or "" ) and "1" or ""
-end
-function p.getPlain( frame )
-    return Text.getPlain( frame.args[ 1 ] or "" )
-end
-function p.isLatinRange( frame )
-    return Text.isLatinRange( frame.args[ 1 ] or "" ) and "1" or ""
-end
-function p.isQuote( frame )
-    return Text.isQuote( frame.args[ 1 ] or "" ) and "1" or ""
 end
@@ Zeile 818: / Zeile 714: @@
                                 tonumber( frame.args[3] ) )
 end
-function p.removeDiacritics( frame )
-    return Text.removeDiacritics( frame.args[ 1 ] or "" )
-end
-function p.sentenceTerminated( frame )
-    return Text.sentenceTerminated( frame.args[ 1 ] or "" ) and "1" or ""
-end
-function p.ucfirstAll( frame )
-    return Text.ucfirstAll( frame.args[ 1 ] or "" )
-end
-function p.unstrip( frame )
-    return mw.text.trim( mw.text.unstrip( frame.args[ 1 ] or "" ) )
-end
-function p.uprightNonlatin( frame )
-    return Text.uprightNonlatin( frame.args[ 1 ] or "" )
-end
@@ Zeile 885: / Zeile 758: @@
-p.failsafe = function ( frame )
+function p.failsafe()
-    -- Versioning interface
+     return Text.serial
-    local s = type( frame )
+end
-    local since
-    if s == "table" then
-        since = frame.args[ 1 ]
-    elseif s == "string" then
-        since = frame
-    end
-    if since then
-        since = mw.text.trim( since )
-        if since == "" then
-            since = false
-        end
-    end
-     return Failsafe.failsafe( since )  or  ""
-end -- p.failsafe()

Modul:Text: Unterschied zwischen den Versionen

Version vom 21. Juli 2022, 18:43 Uhr

Navigationsmenü

Suche