打开/关闭菜单
60
73
29
1.1K
武外梗百科
打开/关闭外观设置菜单
打开/关闭个人菜单
未登录
未登录用户的IP地址会在进行任意编辑后公开展示。

Module:PageSummary:修订间差异

武外梗百科 爱国好学自强图新的百科全书
无编辑摘要
无编辑摘要
 
(未显示1个用户的4个中间版本)
第1行: 第1行:
local p = {}
local p = {}


-- 高性能清理函数
-- 強化版清理:徹底根除內文圖片,保留加粗和連結
local function smartClean(text)
local function cleanContent(text)
     if not text then return "" end
     if not text then return "" end


     -- 1. 预处理:去掉 HTML 注释、脚注、数学公式等标签块
     -- 1. 基礎清理(注釋、腳注、表格、模板)
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
     text = mw.ustring.gsub(text, "<math[^>]*>.-</math>", "")
      
 
     -- 移除表格 {| ... |}
     -- 2. 去除表格 (MediaWiki 表格以 {| 开始,|} 结束)
     local prev
     -- 使用非贪婪匹配,重复执行以处理多个表格
     repeat
     text = mw.ustring.gsub(text, "{|[^{}]*|}", "") -- 先处理最内层不含嵌套的
        prev = text
     text = mw.ustring.gsub(text, "{|.-|}", "")    -- 扩展清理
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
     until text == prev


     -- 3. 递归剥离模板 {{...}}
     -- 移除模板 {{ ... }}
    -- 性能优化:限制循环次数,防止死循环
     local count = 0
     local prev, count = "", 0
     repeat
     repeat
         prev = text
         prev = text
第25行: 第25行:
     until text == prev or count > 10
     until text == prev or count > 10


     -- 4. 剥离图片、分类等 [[File:xxx]]
     -- 2. 使用安全占位符保護連結,剔除圖片
     -- 逻辑:匹配所有中括号,如果是媒体文件则删掉,如果是普通链接则保留文字
     repeat
    text = mw.ustring.gsub(text, "%[%[([^%[%]]-)%]%]", function(inner)
        prev = text
        local l_inner = mw.ustring.lower(inner)
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
        if l_inner:match("^file:") or l_inner:match("^image:") or  
            local low = mw.ustring.lower(inner)
          l_inner:match("^category:") or l_inner:match("^文件:") or  
            if low:match("^file:") or low:match("^image:") or  
          l_inner:match("^图像:") or l_inner:match("^分类:") then
              low:match("^文件:") or low:match("^圖像:") or
             return "" -- 丢弃附件
              low:match("^category:") or low:match("^分類:") then
         end
                return ""
        -- 保留链接文字:[[A|B]] -> B, [[A]] -> A
            end
        local parts = mw.text.split(inner, "|")
             return "LINKSTART" .. inner .. "LINKEND"
        return parts[#parts]
         end)
     end)
    until text == prev
 
    -- 3. 清理剩餘雜質
    text = mw.ustring.gsub(text, "\n==+.-==+", " ")
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
     text = mw.ustring.gsub(text, "%s+", " ")


     -- 5. 清理剩余的格式字符
     -- 4. 初步恢復連結標籤
     text = mw.ustring.gsub(text, "''+", "")         -- 去掉粗斜体
     text = mw.ustring.gsub(text, "LINKSTART", "[[")
     text = mw.ustring.gsub(text, "==+.-==+", "")    -- 去掉标题
     text = mw.ustring.gsub(text, "LINKEND", "]]")
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ") -- 列表转空格
    text = mw.ustring.gsub(text, "\n+", " ")        -- 换行转空格
    text = mw.ustring.gsub(text, "%s+", " ")       -- 压缩多余空格


     return mw.text.trim(text)
     return mw.text.trim(text)
第54行: 第57行:
     if not title or not title.exists then return "" end
     if not title or not title.exists then return "" end


     -- 性能优化:只读取前 2000 个字符进行解析,而不是全篇,防止超大页面卡死
     -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
     local content = mw.ustring.sub(title:getContent() or "", 1, 2000)
     local rawContent = title:getContent() or ""
    local limitedContent = mw.ustring.sub(rawContent, 1, 1000)


     -- 1. 提取第一张图(直接在原文找,不干扰摘要)
     -- 1. 提取第一張縮圖名
     local firstImage = mw.ustring.match(content, "%[%[%s*[Ff]ile%s*:([^|%]%s]+)") or  
     local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or  
                       mw.ustring.match(content, "%[%[%s*文件%s*:([^|%]%s]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(content, "%[%[%s*[Ii]mage%s*:([^|%]%s]+)")
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")


     -- 2. 提取摘要
     -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
     local cleanText = smartClean(content)
     local cleanText = cleanContent(limitedContent)
     local targetLen = 140
   
     local summary = mw.ustring.sub(cleanText, 1, targetLen)
    -- 3. 【調整】最終截取約 80 字
 
     local targetLen = 80
    if mw.ustring.len(cleanText) > targetLen then
     local summary = ""
        summary = summary .. "..."
   
    if mw.ustring.len(cleanText) <= targetLen then
        summary = cleanText
    else
        -- 截取前 80 字並加上省略號
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
       
        -- 閉合加粗標籤 '''
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
       
        -- 閉合或清理截斷的連結 [[...
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
     end
     end


     -- 3. 渲染 UI
     -- 4. 渲染 HTML
     local container = mw.html.create('div'):css({
     local res = mw.html.create('div'):css({
         ['margin-bottom'] = '20px',
         ['display'] = 'flow-root',  
        ['padding'] = '15px',
         ['line-height'] = '1.5',
         ['border'] = '1px solid #e2e2e2',
         ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
        ['border-radius'] = '8px',
         ['background-color'] = '#fff',
        ['box-shadow'] = '0 2px 5px rgba(0,0,0,0.05)',
        ['display'] = 'flow-root'
     })
     })
 
   
     -- 标题
     -- 标题
     container:tag('div')
     res:tag('div')
        :css({['font-size'] = '1.2em', ['font-weight'] = 'bold', ['margin-bottom'] = '8px', ['color'] = '#000'})
      :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
        :wikitext('[[' .. pageName .. ']]')
      :wikitext('[[' .. pageName .. ']]')
 
    -- 正文容器
    local body = container:tag('div'):css({['font-size'] = '14px', ['line-height'] = '1.6', ['color'] = '#444'})


     -- 图片右浮动
     -- 右侧缩略图
     if firstImage then
     if firstImage then
         body:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
         res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
     end
     end


     body:wikitext(summary)
     -- 摘要正文:在末尾增加空行
    -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
    res:tag('div')
      :wikitext(summary .. '<br /><br />')  


     return tostring(container)
     return tostring(res)
end
end


return p
return p

2026年2月18日 (三) 13:35的最新版本

此模块的文档可以在Module:PageSummary/doc创建

local p = {}

-- 強化版清理:徹底根除內文圖片,保留加粗和連結
local function cleanContent(text)
    if not text then return "" end

    -- 1. 基礎清理(注釋、腳注、表格、模板)
    text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
    text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
    
    -- 移除表格 {| ... |}
    local prev
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
    until text == prev

    -- 移除模板 {{ ... }}
    local count = 0
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
        count = count + 1
    until text == prev or count > 10

    -- 2. 使用安全占位符保護連結,剔除圖片
    repeat
        prev = text
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
            local low = mw.ustring.lower(inner)
            if low:match("^file:") or low:match("^image:") or 
               low:match("^文件:") or low:match("^圖像:") or
               low:match("^category:") or low:match("^分類:") then
                return "" 
            end
            return "LINKSTART" .. inner .. "LINKEND"
        end)
    until text == prev

    -- 3. 清理剩餘雜質
    text = mw.ustring.gsub(text, "\n==+.-==+", " ") 
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
    text = mw.ustring.gsub(text, "%s+", " ")

    -- 4. 初步恢復連結標籤
    text = mw.ustring.gsub(text, "LINKSTART", "[[")
    text = mw.ustring.gsub(text, "LINKEND", "]]")

    return mw.text.trim(text)
end

function p.getSummaryAndImage(frame)
    local pageName = frame.args[1] or ""
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end

    -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
    local rawContent = title:getContent() or ""
    local limitedContent = mw.ustring.sub(rawContent, 1, 1000)

    -- 1. 提取第一張縮圖名
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or 
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")

    -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
    local cleanText = cleanContent(limitedContent)
    
    -- 3. 【調整】最終截取約 80 字
    local targetLen = 80
    local summary = ""
    
    if mw.ustring.len(cleanText) <= targetLen then
        summary = cleanText
    else
        -- 截取前 80 字並加上省略號
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
        
        -- 閉合加粗標籤 '''
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
        
        -- 閉合或清理截斷的連結 [[...
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
    end

    -- 4. 渲染 HTML
    local res = mw.html.create('div'):css({
        ['display'] = 'flow-root', 
        ['line-height'] = '1.5',
        ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
    })
    
    -- 标题
    res:tag('div')
       :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
       :wikitext('[[' .. pageName .. ']]')

    -- 右侧缩略图
    if firstImage then
        res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
    end

    -- 摘要正文:在末尾增加空行
    -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
    res:tag('div')
       :wikitext(summary .. '<br /><br />') 

    return tostring(res)
end

return p