打开/关闭菜单
60
73
29
1.1K
武外梗百科
打开/关闭外观设置菜单
打开/关闭个人菜单
未登录
未登录用户的IP地址会在进行任意编辑后公开展示。

Module:PageSummary:修订间差异

武外梗百科 爱国好学自强图新的百科全书
无编辑摘要
无编辑摘要
 
(未显示17个用户的27个中间版本)
第1行: 第1行:
local p = {}
local p = {}


-- ────────────────────────────────────────────────
-- 強化版清理:徹底根除內文圖片,保留加粗和連結
-- 剥离图片、分类等 [[File:…]] [[Category:…]] 结构,避免出现在摘要中
local function cleanContent(text)
-- ────────────────────────────────────────────────
    if not text then return "" end
local function stripImages(text)
 
     if not text or text == "" then return "" end
    -- 1. 基礎清理(注釋、腳注、表格、模板)
    text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
      
      
     -- 统一常见命名空间写法,减少匹配分支
     -- 移除表格 {| ... |}
     text = text:gsub("%[%[(文件|File|Image|图像):", "[[File:")
     local prev
          :gsub("%[%[(分类|Category):", "[[Category:")
    repeat
   
        prev = text
     local out = {}
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
    local i = 1
    until text == prev
     local len = #text
 
      
    -- 移除模板 {{ ... }}
    while i <= len do
    local count = 0
         if text:byte(i) == 91 and text:byte(i+1) == 91 then  -- [[
    repeat
             local chunk = text:sub(i, i+15)
        prev = text
             if chunk:match("^%[%[File:") or chunk:match("^%[%[Category:") then
        text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
                -- 跳过整段嵌套链接 [[ ... ]]
        count = count + 1
                local depth = 1
     until text == prev or count > 10
                local j = i + 2
 
                while j <= len and depth > 0 do
     -- 2. 使用安全占位符保護連結,剔除圖片
                    if text:byte(j) == 91 and text:byte(j+1) == 91 then
     repeat
                        depth = depth + 1
        prev = text
                        j = j + 2
         text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
                    elseif text:byte(j) == 93 and text:byte(j+1) == 93 then
             local low = mw.ustring.lower(inner)
                        depth = depth - 1
             if low:match("^file:") or low:match("^image:") or
                        j = j + 2
              low:match("^文件:") or low:match("^圖像:") or
                    else
              low:match("^category:") or low:match("^分類:") then
                        j = j + 1
                 return ""  
                    end
                 end
                i = j  -- 跳到 ]] 之后或文本末尾
            else
                -- 普通内部链接,保留
                table.insert(out, "[[")
                i = i + 2
             end
             end
         else
            return "LINKSTART" .. inner .. "LINKEND"
            table.insert(out, text:sub(i,i))
         end)
            i = i + 1
    until text == prev
        end
 
       
    -- 3. 清理剩餘雜質
        if i > 1200 then break end  -- 极端安全阀
    text = mw.ustring.gsub(text, "\n==+.-==+", " ")
     end
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
      
    text = mw.ustring.gsub(text, "\n+", " ")
    return table.concat(out)
    text = mw.ustring.gsub(text, "%s+", " ")
end
 
    -- 4. 初步恢復連結標籤
     text = mw.ustring.gsub(text, "LINKSTART", "[[")
     text = mw.ustring.gsub(text, "LINKEND", "]]")


-- ────────────────────────────────────────────────
-- 清理模板、参考文献、标题、列表等,得到纯文本摘要
-- ────────────────────────────────────────────────
local function cleanFinal(text)
    if not text or text == "" then return "" end
   
    text = text
        :gsub("{{[^{}]*}}", "")                  -- 模板(简单版,够用)
        :gsub("<ref.-</?ref[^>]*>", "")          -- 参考文献
        :gsub("<!%-%-.-%-%->", "")              -- HTML 注释
        :gsub("\n==+[^=]+==+", " ")              -- 各级标题
        :gsub("\n%s*[*#:;]+", " ")              -- 列表、定义列表
        :gsub("%s*\n%s*", " ")                  -- 所有换行 → 单个空格
        :gsub("%s+", " ")                        -- 压缩连续空格
   
     return mw.text.trim(text)
     return mw.text.trim(text)
end
end


-- ────────────────────────────────────────────────
-- 主函数:生成带首图的简短摘要 + 标题
-- ────────────────────────────────────────────────
function p.getSummaryAndImage(frame)
function p.getSummaryAndImage(frame)
     local pageName = mw.text.trim(frame.args[1] or "")
     local pageName = frame.args[1] or ""
    if pageName == "" then return "" end
   
     local title = mw.title.new(pageName)
     local title = mw.title.new(pageName)
     if not title or not title.exists then return "" end
     if not title or not title.exists then return "" end
    -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
    local rawContent = title:getContent() or ""
    local limitedContent = mw.ustring.sub(rawContent, 1, 1000)
    -- 1. 提取第一張縮圖名
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or
                      mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                      mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")
    -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
    local cleanText = cleanContent(limitedContent)
      
      
     local content = title:getContent()
    -- 3. 【調整】最終截取約 80 字
     if not content or content == "" then return "" end
     local targetLen = 80
     local summary = ""
      
      
     -- 安全取前 300 个 unicode 字符(避免字节截断中文)
     if mw.ustring.len(cleanText) <= targetLen then
    local rawExcerpt = mw.ustring.sub(content, 1, 300)
        summary = cleanText
      
     else
    -- 提取第一张图片(优先标准写法)
        -- 截取前 80 字並加上省略號
    local firstImage = mw.ustring.match(rawExcerpt, "%[%[(File|文件|Image|图像):([^|%]]+)")
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
   
       
    -- 备选:更宽松匹配,但只接受文件类
        -- 閉合加粗標籤 '''
    if not firstImage then
         local _, opens = mw.ustring.gsub(summary, "'''", "")
         local candidate = mw.ustring.match(rawExcerpt, "%[%[([^|%]]+)%]")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
        if candidate then
            local ns = candidate:match("^([^:]+):")
            if ns and (ns:lower() == "file" or ns == "文件" or ns:lower() == "image" or ns == "图像") then
                firstImage = candidate
            end
        end
    end
   
    local step1    = stripImages(rawExcerpt) or ""
    local cleanText = cleanFinal(step1) or ""
   
    local targetLen = 120
    local summary  = cleanText:sub(1, targetLen * 2) or ""
   
    local ulen = mw.ustring.len(summary) or 0
    if ulen > targetLen then
        summary = mw.ustring.sub(summary, 1, targetLen) or ""
          
          
         -- 尽量避免截在中文单词/句子中间
         -- 閉合或清理截斷的連結 [[...
         local rev = summary:reverse()
         if mw.ustring.match(summary, "%[%[[^%]]*$") then
        local lastBreak = rev:find("[ %s%p,。!?;:,…]") or 1
             summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        if lastBreak > 1 and lastBreak < 20 then
             summary = mw.ustring.sub(summary, 1, #summary - lastBreak + 1) or ""
         end
         end
       
        summary = mw.text.trim(summary) .. "..."
     end
     end
   
 
    summary = mw.text.trim(summary)
     -- 4. 渲染 HTML
   
     local res = mw.html.create('div'):css({
     -- ── HTML 输出 ───────────────────────────────────
        ['display'] = 'flow-root',  
     local container = mw.html.create('div')
        ['line-height'] = '1.5',
        :css('margin-bottom', '25px')
         ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
         :css('display', 'flow-root')
    })
      
      
     -- 标题
     -- 标题
     container:tag('div')
     res:tag('div')
        :css('font-size', '1.3em')
      :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
        :css('font-weight', 'bold')
      :wikitext('[[' .. pageName .. ']]')
        :css('margin-bottom', '6px')
 
        :css('border-bottom', '1px solid #eee')
     -- 右侧缩略图
        :wikitext('[[' .. pageName .. ']]')
     if firstImage then
   
         res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
    -- 内容区(右浮图 + 文字)
     local textDiv = container:tag('div')
        :css('line-height', '1.6')
        :css('color', '#222')
        :css('font-size', '14px')
   
     if firstImage and firstImage ~= "" then
         firstImage = mw.text.trim(firstImage)
        local img = '[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]'
        textDiv:wikitext(img)
     end
     end
      
 
     textDiv:wikitext(summary)
     -- 摘要正文:在末尾增加空行
   
     -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
     return tostring(container)
    res:tag('div')
      :wikitext(summary .. '<br /><br />')  
 
     return tostring(res)
end
end


return p
return p

2026年2月18日 (三) 13:35的最新版本

此模块的文档可以在Module:PageSummary/doc创建

local p = {}

-- 強化版清理:徹底根除內文圖片,保留加粗和連結
local function cleanContent(text)
    if not text then return "" end

    -- 1. 基礎清理(注釋、腳注、表格、模板)
    text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
    text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
    
    -- 移除表格 {| ... |}
    local prev
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
    until text == prev

    -- 移除模板 {{ ... }}
    local count = 0
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
        count = count + 1
    until text == prev or count > 10

    -- 2. 使用安全占位符保護連結,剔除圖片
    repeat
        prev = text
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
            local low = mw.ustring.lower(inner)
            if low:match("^file:") or low:match("^image:") or 
               low:match("^文件:") or low:match("^圖像:") or
               low:match("^category:") or low:match("^分類:") then
                return "" 
            end
            return "LINKSTART" .. inner .. "LINKEND"
        end)
    until text == prev

    -- 3. 清理剩餘雜質
    text = mw.ustring.gsub(text, "\n==+.-==+", " ") 
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
    text = mw.ustring.gsub(text, "%s+", " ")

    -- 4. 初步恢復連結標籤
    text = mw.ustring.gsub(text, "LINKSTART", "[[")
    text = mw.ustring.gsub(text, "LINKEND", "]]")

    return mw.text.trim(text)
end

function p.getSummaryAndImage(frame)
    local pageName = frame.args[1] or ""
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end

    -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
    local rawContent = title:getContent() or ""
    local limitedContent = mw.ustring.sub(rawContent, 1, 1000)

    -- 1. 提取第一張縮圖名
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or 
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")

    -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
    local cleanText = cleanContent(limitedContent)
    
    -- 3. 【調整】最終截取約 80 字
    local targetLen = 80
    local summary = ""
    
    if mw.ustring.len(cleanText) <= targetLen then
        summary = cleanText
    else
        -- 截取前 80 字並加上省略號
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
        
        -- 閉合加粗標籤 '''
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
        
        -- 閉合或清理截斷的連結 [[...
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
    end

    -- 4. 渲染 HTML
    local res = mw.html.create('div'):css({
        ['display'] = 'flow-root', 
        ['line-height'] = '1.5',
        ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
    })
    
    -- 标题
    res:tag('div')
       :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
       :wikitext('[[' .. pageName .. ']]')

    -- 右侧缩略图
    if firstImage then
        res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
    end

    -- 摘要正文:在末尾增加空行
    -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
    res:tag('div')
       :wikitext(summary .. '<br /><br />') 

    return tostring(res)
end

return p