打开/关闭菜单
60
73
29
1.1K
武外梗百科
打开/关闭外观设置菜单
打开/关闭个人菜单
未登录
未登录用户的IP地址会在进行任意编辑后公开展示。

Module:PageSummary:修订间差异

武外梗百科 爱国好学自强图新的百科全书
无编辑摘要
无编辑摘要
第1行: 第1行:
local p = {}
local p = {}


-- 精准清理函数:保留加粗和链接,去除表格、模板和正文图片
-- 强化版清理:彻底根除内文图片,保留加粗和链接
local function cleanContent(text)
local function cleanContent(text)
     if not text then return "" end
     if not text then return "" end


     -- 1. 移除 HTML 注释、脚注、数学公式
     -- 1. 基础清理(注释、脚注、表格、模板)
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
     text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
     text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
 
   
     -- 2. 移除表格 {| ... |}
     -- 移除表格
    -- 处理嵌套表格:从内向外剥离
     local prev
     local prev
     repeat
     repeat
第18行: 第17行:
     until text == prev
     until text == prev


     -- 3. 递归移除模板 {{ ... }} (保留正文文字,移除干扰块)
     -- 移除模板
     local count = 0
     local count = 0
     repeat
     repeat
第26行: 第25行:
     until text == prev or count > 10
     until text == prev or count > 10


     -- 4. 【核心】去除正文中的图片/文件,但保留普通链接
     -- 2. 【强化】彻底剔除正文中的图片引用
     -- 逻辑:匹配 [[...]],如果是文件则删掉,如果是链接则原样保留
     -- 这里的正则覆盖了包含换行、管道符、嵌套括号的复杂情况
     text = mw.ustring.gsub(text, "%[%[([^%[%]]-)%]%]", function(inner)
    -- 我们使用递归匹配,确保嵌套的 [[文件:[[链接]]]] 也能被干掉
        local low = mw.ustring.lower(inner)
     repeat
        -- 识别文件、分类前缀
        prev = text
        if low:match("^file:") or low:match("^image:") or  
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
          low:match("^文件:") or low:match("^图像:") or
            local low = mw.ustring.lower(inner)
          low:match("^category:") or low:match("^分类:") then
            -- 如果是文件、分类、图像等,返回空字符串(彻底剔除)
            return "" -- 移除这些内容
            if low:match("^file:") or low:match("^image:") or  
        end
              low:match("^文件:") or low:match("^图像:") or
        return "[[" .. inner .. "]]" -- 保留普通链接
              low:match("^category:") or low:match("^分类:") then
     end)
                return ""  
            end
            -- 如果是普通链接,暂时保留
            return "%%LINKSTART%%" .. inner .. "%%LINKEND%%"
        end)
    until text == prev
 
    -- 3. 清理剩余的 Wikitext 杂质
    text = mw.ustring.gsub(text, "\n==+.-==+", " ")
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
     text = mw.ustring.gsub(text, "%s+", " ")


     -- 5. 清理多余的标题符号、换行和空格
     -- 4. 恢复链接(将占位符转回标准链接)
     text = mw.ustring.gsub(text, "\n==+.-==+", " ") -- 移除标题行
     text = mw.ustring.gsub(text, "%%LINKSTART%%(.-)%%LINKEND%%", "[[%1]]")
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ") -- 列表转为空格
    text = mw.ustring.gsub(text, "\n+", " ")        -- 换行转空格
    text = mw.ustring.gsub(text, "%s+", " ")       -- 压缩空格


     return mw.text.trim(text)
     return mw.text.trim(text)
第53行: 第60行:
     if not title or not title.exists then return "" end
     if not title or not title.exists then return "" end


    -- 读取前 1500 字,保证性能的同时覆盖首段
     local rawContent = title:getContent() or ""
     local rawContent = title:getContent() or ""
     local limitedContent = mw.ustring.sub(rawContent, 1, 1500)
    -- 只读取前 2000 字符,确保护盖首段图片
     local limitedContent = mw.ustring.sub(rawContent, 1, 2000)


     -- 1. 提取缩略图(在清理前先抓取第一张图名)
     -- 1. 提取缩略图(在彻底清理前提取出文件名)
     local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s]+)") or  
     local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or  
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s]+)")
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")


     -- 2. 执行清理(保留加粗和链接)
     -- 2. 执行强力清理
     local cleanText = cleanContent(limitedContent)
     local cleanText = cleanContent(limitedContent)
      
      
     -- 3. 截取摘要长度
     -- 3. 安全截断(处理中文和标签闭合)
     local targetLen = 180
     local targetLen = 150
     local summary = mw.ustring.sub(cleanText, 1, targetLen)
     local summary = ""
      
      
    -- 4. 补全因截断可能破坏的加粗或链接标签
     if mw.ustring.len(cleanText) <= targetLen then
     if mw.ustring.len(cleanText) > targetLen then
         summary = cleanText
         summary = summary .. "..."
    else
         -- 简单的标签闭合检查(防止截断导致的页面错乱)
        -- 使用 ustring 截断防止乱码,并加上省略号
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
       
         -- 闭合加粗标签
         local _, opens = mw.ustring.gsub(summary, "'''", "")
         local _, opens = mw.ustring.gsub(summary, "'''", "")
         if opens % 2 ~= 0 then summary = summary .. "'''" end
         if opens % 2 ~= 0 then summary = summary .. "'''" end
       
        -- 闭合链接标签(如果截断在链接中间,直接舍弃最后一部分)
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
     end
     end


     -- 5. 渲染(不加外框,仅保持文字环绕)
     -- 4. 最终渲染
     local res = mw.html.create('div'):css({['display'] = 'flow-root'})
     local res = mw.html.create('div'):css({['display'] = 'flow-root', ['line-height'] = '1.6'})
      
      
     -- 渲染标题
     -- 标题
     res:tag('div')
     res:tag('div')
       :css({['font-size'] = '1.2em', ['font-weight'] = 'bold', ['margin-bottom'] = '5px'})
       :css({['font-size'] = '1.2em', ['font-weight'] = 'bold', ['margin-bottom'] = '5px'})
       :wikitext('[[' .. pageName .. ']]')
       :wikitext('[[' .. pageName .. ']]')


     -- 渲染图片(缩略图)
     -- 图片
     if firstImage then
     if firstImage then
         res:wikitext('[[File:' .. firstImage .. '|120px|right|link=' .. pageName .. ']]')
         res:wikitext('[[File:' .. firstImage .. '|120px|right|link=' .. pageName .. ']]')
     end
     end


     -- 渲染带加粗和链接的正文
     -- 正文(带加粗和链接)
     res:wikitext(summary)
     res:wikitext(summary)



2026年2月18日 (三) 13:31的版本

此模块的文档可以在Module:PageSummary/doc创建

local p = {}

-- 强化版清理:彻底根除内文图片,保留加粗和链接
local function cleanContent(text)
    if not text then return "" end

    -- 1. 基础清理(注释、脚注、表格、模板)
    text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
    text = mw.ustring.gsub(text, "<ref[^>]*>.-</ref>", "")
    text = mw.ustring.gsub(text, "<ref[^>]-/>", "")
    
    -- 移除表格
    local prev
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{|[^{}]*|}", "")
    until text == prev

    -- 移除模板
    local count = 0
    repeat
        prev = text
        text = mw.ustring.gsub(text, "{{[^{}]-}}", "")
        count = count + 1
    until text == prev or count > 10

    -- 2. 【强化】彻底剔除正文中的图片引用
    -- 这里的正则覆盖了包含换行、管道符、嵌套括号的复杂情况
    -- 我们使用递归匹配,确保嵌套的 [[文件:[[链接]]]] 也能被干掉
    repeat
        prev = text
        text = mw.ustring.gsub(text, "%[%[%s*([^%[%]]-)%s*%]%]", function(inner)
            local low = mw.ustring.lower(inner)
            -- 如果是文件、分类、图像等,返回空字符串(彻底剔除)
            if low:match("^file:") or low:match("^image:") or 
               low:match("^文件:") or low:match("^图像:") or
               low:match("^category:") or low:match("^分类:") then
                return "" 
            end
            -- 如果是普通链接,暂时保留
            return "%%LINKSTART%%" .. inner .. "%%LINKEND%%"
        end)
    until text == prev

    -- 3. 清理剩余的 Wikitext 杂质
    text = mw.ustring.gsub(text, "\n==+.-==+", " ") 
    text = mw.ustring.gsub(text, "\n%s*[*#:]+", " ")
    text = mw.ustring.gsub(text, "\n+", " ")
    text = mw.ustring.gsub(text, "%s+", " ")

    -- 4. 恢复链接(将占位符转回标准链接)
    text = mw.ustring.gsub(text, "%%LINKSTART%%(.-)%%LINKEND%%", "[[%1]]")

    return mw.text.trim(text)
end

function p.getSummaryAndImage(frame)
    local pageName = frame.args[1] or ""
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end

    local rawContent = title:getContent() or ""
    -- 只读取前 2000 字符,确保护盖首段图片
    local limitedContent = mw.ustring.sub(rawContent, 1, 2000)

    -- 1. 提取缩略图(在彻底清理前提取出文件名)
    local firstImage = mw.ustring.match(limitedContent, "%[%[%s*[Ff]ile%s*:([^|%]%s\n]+)") or 
                       mw.ustring.match(limitedContent, "%[%[%s*文件%s*:([^|%]%s\n]+)") or
                       mw.ustring.match(limitedContent, "%[%[%s*[Ii]mage%s*:([^|%]%s\n]+)")

    -- 2. 执行强力清理
    local cleanText = cleanContent(limitedContent)
    
    -- 3. 安全截断(处理中文和标签闭合)
    local targetLen = 150
    local summary = ""
    
    if mw.ustring.len(cleanText) <= targetLen then
        summary = cleanText
    else
        -- 使用 ustring 截断防止乱码,并加上省略号
        summary = mw.ustring.sub(cleanText, 1, targetLen) .. "..."
        
        -- 闭合加粗标签
        local _, opens = mw.ustring.gsub(summary, "'''", "")
        if opens % 2 ~= 0 then summary = summary .. "'''" end
        
        -- 闭合链接标签(如果截断在链接中间,直接舍弃最后一部分)
        if mw.ustring.match(summary, "%[%[[^%]]*$") then
            summary = mw.ustring.gsub(summary, "%[%[[^%]]*$", "") .. "..."
        end
    end

    -- 4. 最终渲染
    local res = mw.html.create('div'):css({['display'] = 'flow-root', ['line-height'] = '1.6'})
    
    -- 标题
    res:tag('div')
       :css({['font-size'] = '1.2em', ['font-weight'] = 'bold', ['margin-bottom'] = '5px'})
       :wikitext('[[' .. pageName .. ']]')

    -- 图片
    if firstImage then
        res:wikitext('[[File:' .. firstImage .. '|120px|right|link=' .. pageName .. ']]')
    end

    -- 正文(带加粗和链接)
    res:wikitext(summary)

    return tostring(res)
end

return p