打开/关闭菜单
60
73
29
1.1K
武外梗百科
打开/关闭外观设置菜单
打开/关闭个人菜单
未登录
未登录用户的IP地址会在进行任意编辑后公开展示。

Module:PageSummary:修订间差异

武外梗百科 爱国好学自强图新的百科全书
无编辑摘要
无编辑摘要
 
(未显示6个用户的9个中间版本)
第1行: 第1行:
local p = {}
local p = {}


-- 供其他 Lua 模块调用的内部函数
-- 強化版清理:徹底根除內文圖片,保留加粗和連結
function p._getSummary(pageName)
local function cleanContent(text)
     if not pageName or pageName == "" then
     if not text then return "" end
        return "<span class='error'>请输入有效的条目名称。</span>"
    end


    local title = mw.title.new(pageName)
     -- 1. 基礎清理(注釋、腳注、表格、模板)
    if not title or not title.exists then
        return "<span class='error'>条目“" .. pageName .. "”不存在。</span>"
    end
 
    local content = title:getContent()
    if not content then
        return "<span class='error'>无法获取条目内容。</span>"
    end
 
     -- 1. 提取首张图片 (匹配常用图片命名空间)
    -- 注意:不区分大小写,使用简单的正则匹配查找最早出现的文件
    local firstImage = mw.ustring.match(content, "%[%[%s*[Ff][Ii][Ll][Ee]%s*:([^|%]]+)")
        or mw.ustring.match(content, "%[%[%s*[Ii][Mm][Aa][Gg][Ee]%s*:([^|%]]+)")
        or mw.ustring.match(content, "%[%[%s*文件%s*:([^|%]]+)")
        or mw.ustring.match(content, "%[%[%s*檔案%s*:([^|%]]+)")
        or mw.ustring.match(content, "%[%[%s*图片%s*:([^|%]]+)")
        or mw.ustring.match(content, "%[%[%s*图像%s*:([^|%]]+)")
 
    -- 2. 纯文本清理流程
    local text = content
 
    -- 屏蔽重定向页面
    if mw.ustring.match(text, "^%s*#REDIRECT") or mw.ustring.match(text, "^%s*#重定向") then
        return "该页面是一个重定向页面。"
    end
 
    -- 去除 HTML 注释
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    -- 去除部分可能干扰正文的 HTML 标签及其内容
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
     text = mw.ustring.gsub(text, "<math[^>]*>.-</math>", "")
      
     text = mw.ustring.gsub(text, "<gallery[^>]*>.-</gallery>", "")
     -- 移除表格 {| ... |}
    text = mw.ustring.gsub(text, "<div[^>]*>.-</div>", "")
     local prev
    text = mw.ustring.gsub(text, "<table[^>]*>.-</table>", "")
 
    -- 循环去除模板 {{ ... }} (添加计数器防死循环)
     local prev, count
    count = 0
     repeat
     repeat
         prev = text
         prev = text
         text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
         text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
        count = count + 1
     until text == prev
     until text == prev or count > 50


     -- 循环去除表格 {| ... |}  
     -- 移除模板 {{ ... }}
     count = 0
     local count = 0
     repeat
     repeat
         prev = text
         prev = text
         text = mw.ustring.gsub(text, "{|.-|}", "")
         text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
         count = count + 1
         count = count + 1
     until text == prev or count > 30
     until text == prev or count > 10


     -- 去除章节标题 == 标题 ==
     -- 2. 使用安全占位符保護連結,剔除圖片
    text = mw.ustring.gsub(text, "==+.-==+", "")
 
    -- 核心逻辑:由内向外剥离中括号 [[ ... ]]
    -- 这样可以安全处理诸如 [[File:xx.jpg|thumb|这是一个[[内链]]]] 的嵌套情况
    count = 0
     repeat
     repeat
         prev = text
         prev = text
         text = mw.ustring.gsub(text, "%[%[([^%[%]]-)%]%]", function(inner)
         text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
             local upperInner = mw.ustring.upper(inner)
             local low = mw.ustring.lower(inner)
            -- 如果是文件、图片或分类,直接整块抹除
             if low:match("^file:") or low:match("^image:") or  
             if mw.ustring.match(upperInner, "^%s*FILE%s*:") or
               low:match("^文件:") or low:match("^圖像:") or
              mw.ustring.match(upperInner, "^%s*IMAGE%s*:") or
               low:match("^category:") or low:match("^分類:") then
               mw.ustring.match(inner, "^%s*文件%s*:") or
                 return ""  
              mw.ustring.match(inner, "^%s*檔案%s*:") or
              mw.ustring.match(inner, "^%s*图片%s*:") or
               mw.ustring.match(inner, "^%s*图像%s*:") or
              mw.ustring.match(upperInner, "^%s*CATEGORY%s*:") or
              mw.ustring.match(inner, "^%s*分类%s*:") or
              mw.ustring.match(inner, "^%s*分類%s*:") then
                 return ""
             end
             end
             -- 如果是普通内链 [[链接|显示文本]],保留最后的显示文本
             return "LINKSTART" .. inner .. "LINKEND"
            local parts = mw.text.split(inner, "|")
            return parts[#parts]
         end)
         end)
        count = count + 1
    until text == prev
     until text == prev or count > 50
 
    -- 3. 清理剩餘雜質
    text = mw.ustring.gsub(text, "\n==+.-==+", " ")
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
    text = mw.ustring.gsub(text, "%s+", " ")
 
    -- 4. 初步恢復連結標籤
    text = mw.ustring.gsub(text, "LINKSTART", "[[")
    text = mw.ustring.gsub(text, "LINKEND", "]]")
 
     return mw.text.trim(text)
end
 
function p.getSummaryAndImage(frame)
    local pageName = frame.args[1] or ""
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end


     -- 去除粗体和斜体 (''' 和 '')
     -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
     text = mw.ustring.gsub(text, "''+", "")
    local rawContent = title:getContent() or ""
     local limitedContent = mw.ustring.sub(rawContent, 1, 1000)


     -- 去除残留的单层 HTML 标签 (如 <br>, <span>)
     -- 1. 提取第一張縮圖名
    text = mw.ustring.gsub(text, "<[^>]+>", "")
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or
                      mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                      mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")


     -- 合并多余的换行和空格
     -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
     text = mw.ustring.gsub(text, "%s+", " ")
     local cleanText = cleanContent(limitedContent)
   
    -- 3. 【調整】最終截取約 80 字
    local targetLen = 80
    local summary = ""
      
      
     -- 去除开头可能残留的乱码标点(因为移除了前面的信息框,这里容易留下逗号或右括号)
     if mw.ustring.len(cleanText) <= targetLen then
    text = mw.ustring.gsub(text, "^[%s,%.:;%?!)]】」》”’>|]+", "")
        summary = cleanText
    text = mw.ustring.gsub(text, "^[,。、;:?!]+", "")
    else
    text = mw.text.trim(text)
        -- 截取前 80 字並加上省略號
 
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
    -- 3. 截取前约 100 个字符 (由于使用的是 mw.ustring,完美支持中文字符计算)
       
    local summary = mw.ustring.sub(text, 1, 100)
        -- 閉合加粗標籤 '''
    if mw.ustring.len(text) > 100 then
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        summary = summary .. "..."
        if opens % 2 ~= 0 then summary = summary .. "'''" end
       
        -- 閉合或清理截斷的連結 [[...
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
     end
     end


     -- 4. 使用 mw.html 组装最终输出的 UI
     -- 4. 渲染 HTML
     local container = mw.html.create('div')
     local res = mw.html.create('div'):css({
         :cssText('display: flex; gap: 15px; align-items: flex-start; padding: 15px; border: 1px solid #eaecf0; border-radius: 6px; background: #f8f9fa; margin-bottom: 1em;')
         ['display'] = 'flow-root',
        ['line-height'] = '1.5',
        ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
    })
   
    -- 标题
    res:tag('div')
      :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
      :wikitext('[[' .. pageName .. ']]')


    -- 右侧缩略图
     if firstImage then
     if firstImage then
         container:tag('div')
         res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
            :cssText('flex: 0 0 120px; border-radius: 4px; overflow: hidden;')
            :wikitext("[[File:" .. firstImage .. "|120px|frameless]]")
     end
     end


     container:tag('div')
     -- 摘要正文:在末尾增加空行
        :cssText('flex-grow: 1; font-size: 0.95em; line-height: 1.6; color: #202122;')
    -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
        :wikitext(summary)
    res:tag('div')
 
      :wikitext(summary .. '<br /><br />')  
    return tostring(container)
end


-- 供 Wikitext 中 {{#invoke:}} 调用的主函数
     return tostring(res)
function p.main(frame)
    -- 优先获取模板传递的参数,如果没有,再尝试获取直接 Invoke 传递的参数
    local args = frame:getParent().args
    if not args[1] or args[1] == "" then
        args = frame.args
    end
   
     return p._getSummary(args[1])
end
end


return p
return p

2026年2月18日 (三) 13:35的最新版本

此模块的文档可以在Module:PageSummary/doc创建

local p = {}

-- 強化版清理:徹底根除內文圖片,保留加粗和連結
local function cleanContent(text)
    if not text then return "" end

    -- 1. 基礎清理(注釋、腳注、表格、模板)
    text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
    text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
    
    -- 移除表格 {| ... |}
    local prev
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
    until text == prev

    -- 移除模板 {{ ... }}
    local count = 0
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
        count = count + 1
    until text == prev or count > 10

    -- 2. 使用安全占位符保護連結,剔除圖片
    repeat
        prev = text
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
            local low = mw.ustring.lower(inner)
            if low:match("^file:") or low:match("^image:") or 
               low:match("^文件:") or low:match("^圖像:") or
               low:match("^category:") or low:match("^分類:") then
                return "" 
            end
            return "LINKSTART" .. inner .. "LINKEND"
        end)
    until text == prev

    -- 3. 清理剩餘雜質
    text = mw.ustring.gsub(text, "\n==+.-==+", " ") 
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
    text = mw.ustring.gsub(text, "%s+", " ")

    -- 4. 初步恢復連結標籤
    text = mw.ustring.gsub(text, "LINKSTART", "[[")
    text = mw.ustring.gsub(text, "LINKEND", "]]")

    return mw.text.trim(text)
end

function p.getSummaryAndImage(frame)
    local pageName = frame.args[1] or ""
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end

    -- 【調整】讀取前 1000 字以確保跨過開頭的表格/模板區
    local rawContent = title:getContent() or ""
    local limitedContent = mw.ustring.sub(rawContent, 1, 1000)

    -- 1. 提取第一張縮圖名
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or 
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")

    -- 2. 執行清理(此時會得到過濾掉干擾後的純文字)
    local cleanText = cleanContent(limitedContent)
    
    -- 3. 【調整】最終截取約 80 字
    local targetLen = 80
    local summary = ""
    
    if mw.ustring.len(cleanText) <= targetLen then
        summary = cleanText
    else
        -- 截取前 80 字並加上省略號
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
        
        -- 閉合加粗標籤 '''
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
        
        -- 閉合或清理截斷的連結 [[...
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
    end

    -- 4. 渲染 HTML
    local res = mw.html.create('div'):css({
        ['display'] = 'flow-root', 
        ['line-height'] = '1.5',
        ['margin-bottom'] = '1em' -- 也可以通过这里控制整个组件下方的间距
    })
    
    -- 标题
    res:tag('div')
       :css({['font-size'] = '1.15em', ['font-weight'] = 'bold', ['margin-bottom'] = '4px'})
       :wikitext('[[' .. pageName .. ']]')

    -- 右侧缩略图
    if firstImage then
        res:wikitext('[[File:' .. firstImage .. '|100px|right|link=' .. pageName .. ']]')
    end

    -- 摘要正文:在末尾增加空行
    -- 方法:在 summary 字符串后直接加上 <br /> 或 \n\n
    res:tag('div')
       :wikitext(summary .. '<br /><br />') 

    return tostring(res)
end

return p