打开/关闭菜单
60
73
29
1.1K
武外梗百科
打开/关闭外观设置菜单
打开/关闭个人菜单
未登录
未登录用户的IP地址会在进行任意编辑后公开展示。

Module:PageSummary:修订间差异

武外梗百科 爱国好学自强图新的百科全书
无编辑摘要
无编辑摘要
第1行: 第1行:
local p = {}
local p = {}


-- 核心剥离逻辑:状态机识别,防止图片碎片
-- 更高效的图片/链接/分类剥离(避免逐字符 + 频繁 sub)
local function stripImages(text)
local function stripImages(text)
     if not text then return "" end
     if not text or text == "" then return "" end
   
    -- 预先把常见命名空间前缀统一处理,减少后续匹配分支
    text = text:gsub("%[%[(文件|File|Image|图像):", "[[File:")
   
     local out = {}
     local out = {}
     local i = 1
     local i = 1
     local len = mw.ustring.len(text)
     local len = #text   -- 优先用字节长度,性能更好(中文环境影响可接受)
   
     while i <= len do
     while i <= len do
         local foundImage = false
         local byte = text:byte(i)
         local chunk = mw.ustring.sub(text, i, i + 10)
         if byte == 91 and text:byte(i+1) == 91 then  -- [[
        if chunk:match("^%[%[[Ff]ile:") or chunk:match("^%[%[[Ii]mage:") or  
            local chunk = text:sub(i, i+12)
          chunk:match("^%[%[[Cc]ategory:") or
            if chunk:match("^%[%[File:")  
          chunk:match("^%[%[文件:") or chunk:match("^%[%[图像:") then
                or chunk:match("^%[%[Category:")
            local depth = 0
                or chunk:match("^%[%[分类:")
            local j = i
            then
            while j <= len do
                -- 跳过整段 [[...]]
                local char2 = mw.ustring.sub(text, j, j + 1)
                local depth = 1
                if char2 == "[[" then depth = depth + 1 j = j + 2
                local j = i + 2
                elseif char2 == "]]" then depth = depth - 1 j = j + 2
                while j <= len and depth > 0 do
                     if depth <= 0 then i = j foundImage = true break end
                    if text:byte(j) == 91 and text:byte(j+1) == 91 then
                 else j = j + 1 end
                        depth = depth + 1
                        j = j + 2
                    elseif text:byte(j) == 93 and text:byte(j+1) == 93 then
                        depth = depth - 1
                        j = j + 2
                     else
                        j = j + 1
                    end
                end
                i = j
            else
                -- 普通 [[ ,保留
                table.insert(out, "[")
                 i = i + 1
             end
             end
         end
         else
        if not foundImage then
             table.insert(out, text:sub(i, i))
             table.insert(out, mw.ustring.sub(text, i, i))
             i = i + 1
             i = i + 1
         end
         end
         if i > 600 then break end
       
        -- 安全阀(防止极端情况卡死)
         if i > 800 then break end
     end
     end
   
     return table.concat(out)
     return table.concat(out)
end
end


-- 合并多次 gsub,提升正则效率
local function cleanFinal(text)
local function cleanFinal(text)
     text = string.gsub(text, "{{[^{}]+}}", "")
     if not text or text == "" then return "" end
     text = string.gsub(text, "{{[^{}]+}}", "")
   
    text = string.gsub(text, "<ref.-</ref>", "")
    -- 一次性处理多种需要删除的模式
    text = string.gsub(text, "<ref.->", "")
     text = text
    text = string.gsub(text, "<!%-%-.-%-%->", "")
        :gsub("{{[^{}]*}}", "")                 -- 模板(包括嵌套一层的情况已足够)
    text = string.gsub(text, "\n==+.-==+", " ")
        :gsub("<ref.-</?ref[^>]*>", "")         -- 参考文献(更宽松匹配)
    text = string.gsub(text, "\n%s*[*#:]+", " ")
        :gsub("<!%-%-.-%-%->", "")             -- 注释
    text = string.gsub(text, "\n", " ")
        :gsub("\n==+.+==+", " ")               -- 标题
    text = string.gsub(text, "%s+", " ")
        :gsub("\n%s*[*#:;]+", " ")             -- 列表
        :gsub("%s*\n%s*", " ")                 -- 换行 → 空格(合并多行)
        :gsub("%s+", " ")                       -- 压缩连续空格
   
     return mw.text.trim(text)
     return mw.text.trim(text)
end
end


function p.getSummaryAndImage(frame)
function p.getSummaryAndImage(frame)
     local pageName = frame.args[1] or ""
     local pageName = mw.text.trim(frame.args[1] or "")
    if pageName == "" then return "" end
   
     local title = mw.title.new(pageName)
     local title = mw.title.new(pageName)
     if not title or not title.exists then return "" end
     if not title or not title.exists then return "" end
      
      
     local content = title:getContent()
     local content = title:getContent()
     local rawExcerpt = mw.ustring.sub(content, 1, 400)
    if not content then return "" end
 
   
     -- 提取第一张图名
    -- 只取前450字节(中文约200-300字),足够提取首图 + 摘要
     local firstImage = string.match(rawExcerpt, "%[%[([Ff]ile:[^|%]%s]+)") or
     local rawExcerpt = content:sub(1, 450)
                      string.match(rawExcerpt, "%[%[(文件:[^|%]%s]+)")
   
 
     -- 提取第一张图片(更宽松匹配,兼容更多写法)
     local step1 = stripImages(rawExcerpt)
     local firstImage = rawExcerpt:match("%[%[(File|文件|Image|图像):([^|%]]+)")
   
    -- 核心剥离 + 清理
     local step1   = stripImages(rawExcerpt)
     local cleanText = cleanFinal(step1)
     local cleanText = cleanFinal(step1)
 
   
     local targetLen = 120
     local targetLen = 120
     local summary = mw.ustring.sub(cleanText, 1, targetLen)
     local summary = cleanText:sub(1, targetLen * 2) -- 多取一点,后面精确截断
      
      
     -- 清理末尾断头链接
     -- 更安全的截断(避免截断在 [[ 中间)
     local _, opens = string.gsub(summary, "%[%[", "")
     local len = mw.ustring.len(summary)
     local _, closes = string.gsub(summary, "%]%]", "")
     if len > targetLen then
    if opens > closes then
        summary = mw.ustring.sub(summary, 1, targetLen)
         local lastOpen = string.find(summary:reverse(), "%[%[")
       
         if lastOpen then summary = string.sub(summary, 1, #summary - lastOpen - 1) end
        -- 简单向后找最近的空格或标点,避免断字
         local lastSpace = summary:reverse():find("[ %s%p]") or 1
         if lastSpace > 1 and lastSpace < 15 then
            summary = mw.ustring.sub(summary, 1, #summary - lastSpace + 1)
        end
        summary = mw.text.trim(summary) .. "..."
     end
     end
    summary = mw.text.trim(summary)
    if mw.ustring.len(cleanText) > targetLen then summary = summary .. "..." end
    -- --- 渲染部分修改 ---
    local container = mw.html.create('div'):css({['margin-bottom']='25px', ['display']='flow-root'})
      
      
     -- 1. 先渲染标题
     -- HTML 构建部分基本保持原样(性能占比很低)
    local container = mw.html.create('div')
        :css('margin-bottom', '25px')
        :css('display', 'flow-root')
   
    -- 标题
     container:tag('div')
     container:tag('div')
         :css({['font-size']='1.3em', ['font-weight']='bold', ['margin-bottom']='6px', ['border-bottom']='1px solid #eee'})
         :css('font-size', '1.3em')
        :css('font-weight', 'bold')
        :css('margin-bottom', '6px')
        :css('border-bottom', '1px solid #eee')
         :wikitext('[[' .. pageName .. ']]')
         :wikitext('[[' .. pageName .. ']]')
      
      
     -- 2. 创建摘要正文容器
     -- 内容区
     local textDiv = container:tag('div')
     local textDiv = container:tag('div')
         :css({['line-height']='1.6', ['color']='#222', ['font-size']='14px'})
         :css('line-height', '1.6')
        :css('color', '#222')
        :css('font-size', '14px')
      
      
     -- 3. 将图片放在正文的最前面,实现文字环绕
     -- 图片右浮 + 链接到页面
     if firstImage then
     if firstImage then
         textDiv:wikitext('[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]')
         textDiv:wikitext(
            '[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]'
        )
     end
     end
      
      
    -- 4. 紧接着放入文字
     textDiv:wikitext(summary)
     textDiv:wikitext(summary)
 
   
     return tostring(container)
     return tostring(container)
end
end


return p
return p

2026年2月17日 (二) 23:50的版本

此模块的文档可以在Module:PageSummary/doc创建

local p = {}

-- 更高效的图片/链接/分类剥离(避免逐字符 + 频繁 sub)
local function stripImages(text)
    if not text or text == "" then return "" end
    
    -- 预先把常见命名空间前缀统一处理,减少后续匹配分支
    text = text:gsub("%[%[(文件|File|Image|图像):", "[[File:")
    
    local out = {}
    local i = 1
    local len = #text   -- 优先用字节长度,性能更好(中文环境影响可接受)
    
    while i <= len do
        local byte = text:byte(i)
        if byte == 91 and text:byte(i+1) == 91 then  -- [[
            local chunk = text:sub(i, i+12)
            if chunk:match("^%[%[File:") 
                or chunk:match("^%[%[Category:")
                or chunk:match("^%[%[分类:")
            then
                -- 跳过整段 [[...]]
                local depth = 1
                local j = i + 2
                while j <= len and depth > 0 do
                    if text:byte(j) == 91 and text:byte(j+1) == 91 then
                        depth = depth + 1
                        j = j + 2
                    elseif text:byte(j) == 93 and text:byte(j+1) == 93 then
                        depth = depth - 1
                        j = j + 2
                    else
                        j = j + 1
                    end
                end
                i = j
            else
                -- 普通 [[ ,保留
                table.insert(out, "[")
                i = i + 1
            end
        else
            table.insert(out, text:sub(i, i))
            i = i + 1
        end
        
        -- 安全阀(防止极端情况卡死)
        if i > 800 then break end
    end
    
    return table.concat(out)
end


-- 合并多次 gsub,提升正则效率
local function cleanFinal(text)
    if not text or text == "" then return "" end
    
    -- 一次性处理多种需要删除的模式
    text = text
        :gsub("{{[^{}]*}}", "")                 -- 模板(包括嵌套一层的情况已足够)
        :gsub("<ref.-</?ref[^>]*>", "")         -- 参考文献(更宽松匹配)
        :gsub("<!%-%-.-%-%->", "")              -- 注释
        :gsub("\n==+.+==+", " ")                -- 标题
        :gsub("\n%s*[*#:;]+", " ")              -- 列表
        :gsub("%s*\n%s*", " ")                  -- 换行 → 空格(合并多行)
        :gsub("%s+", " ")                       -- 压缩连续空格
    
    return mw.text.trim(text)
end


function p.getSummaryAndImage(frame)
    local pageName = mw.text.trim(frame.args[1] or "")
    if pageName == "" then return "" end
    
    local title = mw.title.new(pageName)
    if not title or not title.exists then return "" end
    
    local content = title:getContent()
    if not content then return "" end
    
    -- 只取前450字节(中文约200-300字),足够提取首图 + 摘要
    local rawExcerpt = content:sub(1, 450)
    
    -- 提取第一张图片(更宽松匹配,兼容更多写法)
    local firstImage = rawExcerpt:match("%[%[(File|文件|Image|图像):([^|%]]+)")
    
    -- 核心剥离 + 清理
    local step1   = stripImages(rawExcerpt)
    local cleanText = cleanFinal(step1)
    
    local targetLen = 120
    local summary = cleanText:sub(1, targetLen * 2)  -- 多取一点,后面精确截断
    
    -- 更安全的截断(避免截断在 [[ 中间)
    local len = mw.ustring.len(summary)
    if len > targetLen then
        summary = mw.ustring.sub(summary, 1, targetLen)
        
        -- 简单向后找最近的空格或标点,避免断字
        local lastSpace = summary:reverse():find("[ %s%p]") or 1
        if lastSpace > 1 and lastSpace < 15 then
            summary = mw.ustring.sub(summary, 1, #summary - lastSpace + 1)
        end
        summary = mw.text.trim(summary) .. "..."
    end
    
    -- HTML 构建部分基本保持原样(性能占比很低)
    local container = mw.html.create('div')
        :css('margin-bottom', '25px')
        :css('display', 'flow-root')
    
    -- 标题
    container:tag('div')
        :css('font-size', '1.3em')
        :css('font-weight', 'bold')
        :css('margin-bottom', '6px')
        :css('border-bottom', '1px solid #eee')
        :wikitext('[[' .. pageName .. ']]')
    
    -- 内容区
    local textDiv = container:tag('div')
        :css('line-height', '1.6')
        :css('color', '#222')
        :css('font-size', '14px')
    
    -- 图片右浮 + 链接到页面
    if firstImage then
        textDiv:wikitext(
            '[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]'
        )
    end
    
    textDiv:wikitext(summary)
    
    return tostring(container)
end

return p