Module:PageSummary:修订间差异
武外梗百科 爱国好学自强图新的百科全书
更多操作
无编辑摘要 |
无编辑摘要 |
||
| 第1行: | 第1行: | ||
local p = {} | local p = {} | ||
-- 剥离 [[File:]] [[文件:]] 等,避免污染摘要 | |||
local function stripImages(text) | local function stripImages(text) | ||
if not text or text == "" then return "" end | if not text or text == "" then return "" end | ||
text = | |||
text = mw.ustring.gsub(text, "%[%[(文件|Image|图像):", "[[File:") | |||
text = mw.ustring.gsub(text, "%[%[(分类|Category):", "[[Category:") | |||
local out = {} | local out = {} | ||
local i = 1 | local i = 1 | ||
local len = #text | local len = #text | ||
while i <= len do | while i <= len do | ||
if text:byte(i) == 91 and text:byte(i+1) == 91 then | if text:byte(i) == 91 and text:byte(i+1) == 91 then | ||
| 第36行: | 第40行: | ||
if i > 1200 then break end | if i > 1200 then break end | ||
end | end | ||
return table.concat(out) | return table.concat(out) | ||
end | end | ||
-- 清理函数,全用 mw.ustring.gsub | |||
local function cleanFinal(text) | local function cleanFinal(text) | ||
if not text or text == "" then return "" end | if not text or text == "" then return "" end | ||
text = | |||
text = mw.ustring.gsub(text, "{{[^{}]*}}", "") | |||
text = mw.ustring.gsub(text, "<ref.-</?ref[^>]*>", "") | |||
text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "") | |||
text = mw.ustring.gsub(text, "\n==+[^=]+==+", " ") | |||
text = mw.ustring.gsub(text, "\n%s*[*#:;]+", " ") | |||
text = mw.ustring.gsub(text, "%s*\n%s*", " ") | |||
text = mw.ustring.gsub(text, "%s+", " ") | |||
return mw.text.trim(text) | return mw.text.trim(text) | ||
end | end | ||
| 第55行: | 第62行: | ||
local pageName = mw.text.trim(frame.args[1] or "") | local pageName = mw.text.trim(frame.args[1] or "") | ||
if pageName == "" then return "" end | if pageName == "" then return "" end | ||
local title = mw.title.new(pageName) | local title = mw.title.new(pageName) | ||
if not title or not title.exists then return "" end | if not title or not title.exists then return "" end | ||
local content = title:getContent() | local content = title:getContent() | ||
if not content or content == "" then return "" end | if not content or content == "" then return "" end | ||
-- 关键:清洗无效 UTF-8,避免 gsub 报错 | |||
content = mw.ustring.gsub(content, "[\128-\191][\128-\191]*[^\128-\191]", "�") -- 替换常见无效序列 | |||
content = mw.text.decode(content) or content -- 尝试解码实体 | |||
local rawExcerpt = mw.ustring.sub(content, 1, 800) or "" | local rawExcerpt = mw.ustring.sub(content, 1, 800) or "" | ||
-- | -- 提取第一张图片(find + 手动截取,更可靠) | ||
local firstImage = nil | local firstImage = nil | ||
local prefixes = {"File:", "文件:"} | local prefixes = {"File:", "文件:"} | ||
for _, prefix in ipairs(prefixes) do | for _, prefix in ipairs(prefixes) do | ||
local | local pattern = "%[%[" .. prefix | ||
if | local start = mw.ustring.find(rawExcerpt, pattern) | ||
local | if start then | ||
if | local pipePos = mw.ustring.find(rawExcerpt, "|", start) | ||
local candidate = mw.ustring.sub(rawExcerpt, | local endPos = mw.ustring.find(rawExcerpt, "%]%]", start) | ||
local boundary = pipePos or endPos | |||
if boundary then | |||
local candidate = mw.ustring.sub(rawExcerpt, start + #pattern + 1, boundary - 1) | |||
candidate = mw.text.trim(candidate) | candidate = mw.text.trim(candidate) | ||
if candidate ~= "" and mw.ustring.find(candidate, "%.") then -- | if candidate ~= "" and mw.ustring.find(candidate, "%.") then -- 有扩展名才算图片 | ||
firstImage = prefix .. candidate | firstImage = prefix .. candidate | ||
break | break | ||
| 第81行: | 第95行: | ||
end | end | ||
end | end | ||
local step1 = stripImages(rawExcerpt) or "" | |||
local cleanText = cleanFinal(step1) or "" | local cleanText = cleanFinal(step1) or "" | ||
-- 摘要截断:全程 ustring + 安全断点 | |||
local summary = cleanText | local targetLen = 120 | ||
local summary = cleanText | |||
local ulen = mw.ustring.len(summary) or 0 | |||
local ulen = mw.ustring.len(summary) | |||
if ulen > targetLen then | if ulen > targetLen then | ||
local temp = mw.ustring.sub(summary, 1, targetLen + 20) | local temp = mw.ustring.sub(summary, 1, targetLen + 20) | ||
local breakPos = nil | local breakPos = nil | ||
for i = | for i = mw.ustring.len(temp), 1, -1 do | ||
local char = mw.ustring.sub(temp, i, i) | local char = mw.ustring.sub(temp, i, i) | ||
if char:match("[ %s,。!?;:,…、]") then | if char:match("[ %s,。!?;:,…、]") then | ||
| 第116行: | 第126行: | ||
summary = mw.text.trim(summary) | summary = mw.text.trim(summary) | ||
end | end | ||
-- HTML 输出 | |||
local container = mw.html.create('div') | local container = mw.html.create('div') | ||
:css('margin-bottom', '25px') | :css('margin-bottom', '25px') | ||
:css('display', 'flow-root') | :css('display', 'flow-root') | ||
container:tag('div') | container:tag('div') | ||
:css('font-size', '1.3em') | :css('font-size', '1.3em') | ||
| 第127行: | 第138行: | ||
:css('border-bottom', '1px solid #eee') | :css('border-bottom', '1px solid #eee') | ||
:wikitext('[[' .. pageName .. ']]') | :wikitext('[[' .. pageName .. ']]') | ||
local textDiv = container:tag('div') | local textDiv = container:tag('div') | ||
:css('line-height', '1.6') | :css('line-height', '1.6') | ||
:css('color', '#222') | :css('color', '#222') | ||
:css('font-size', '14px') | :css('font-size', '14px') | ||
if firstImage and firstImage ~= "" then | if firstImage and firstImage ~= "" then | ||
textDiv:wikitext('[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]') | textDiv:wikitext('[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]') | ||
end | end | ||
textDiv:wikitext(summary) | textDiv:wikitext(summary) | ||
return tostring(container) | return tostring(container) | ||
end | end | ||
return p | return p | ||
2026年2月18日 (三) 09:03的版本
此模块的文档可以在Module:PageSummary/doc创建
local p = {}
-- 剥离 [[File:]] [[文件:]] 等,避免污染摘要
local function stripImages(text)
if not text or text == "" then return "" end
text = mw.ustring.gsub(text, "%[%[(文件|Image|图像):", "[[File:")
text = mw.ustring.gsub(text, "%[%[(分类|Category):", "[[Category:")
local out = {}
local i = 1
local len = #text
while i <= len do
if text:byte(i) == 91 and text:byte(i+1) == 91 then
local chunk = text:sub(i, i+15)
if chunk:match("^%[%[File:") or chunk:match("^%[%[Category:") then
local depth = 1
local j = i + 2
while j <= len and depth > 0 do
if text:byte(j) == 91 and text:byte(j+1) == 91 then
depth = depth + 1
j = j + 2
elseif text:byte(j) == 93 and text:byte(j+1) == 93 then
depth = depth - 1
j = j + 2
else
j = j + 1
end
end
i = j
else
table.insert(out, text:sub(i, i+1))
i = i + 2
end
else
table.insert(out, text:sub(i, i))
i = i + 1
end
if i > 1200 then break end
end
return table.concat(out)
end
-- 清理函数,全用 mw.ustring.gsub
local function cleanFinal(text)
if not text or text == "" then return "" end
text = mw.ustring.gsub(text, "{{[^{}]*}}", "")
text = mw.ustring.gsub(text, "<ref.-</?ref[^>]*>", "")
text = mw.ustring.gsub(text, "<!%-%-.-%-%->", "")
text = mw.ustring.gsub(text, "\n==+[^=]+==+", " ")
text = mw.ustring.gsub(text, "\n%s*[*#:;]+", " ")
text = mw.ustring.gsub(text, "%s*\n%s*", " ")
text = mw.ustring.gsub(text, "%s+", " ")
return mw.text.trim(text)
end
function p.getSummaryAndImage(frame)
local pageName = mw.text.trim(frame.args[1] or "")
if pageName == "" then return "" end
local title = mw.title.new(pageName)
if not title or not title.exists then return "" end
local content = title:getContent()
if not content or content == "" then return "" end
-- 关键:清洗无效 UTF-8,避免 gsub 报错
content = mw.ustring.gsub(content, "[\128-\191][\128-\191]*[^\128-\191]", "�") -- 替换常见无效序列
content = mw.text.decode(content) or content -- 尝试解码实体
local rawExcerpt = mw.ustring.sub(content, 1, 800) or ""
-- 提取第一张图片(find + 手动截取,更可靠)
local firstImage = nil
local prefixes = {"File:", "文件:"}
for _, prefix in ipairs(prefixes) do
local pattern = "%[%[" .. prefix
local start = mw.ustring.find(rawExcerpt, pattern)
if start then
local pipePos = mw.ustring.find(rawExcerpt, "|", start)
local endPos = mw.ustring.find(rawExcerpt, "%]%]", start)
local boundary = pipePos or endPos
if boundary then
local candidate = mw.ustring.sub(rawExcerpt, start + #pattern + 1, boundary - 1)
candidate = mw.text.trim(candidate)
if candidate ~= "" and mw.ustring.find(candidate, "%.") then -- 有扩展名才算图片
firstImage = prefix .. candidate
break
end
end
end
end
local step1 = stripImages(rawExcerpt) or ""
local cleanText = cleanFinal(step1) or ""
-- 摘要截断:全程 ustring + 安全断点
local targetLen = 120
local summary = cleanText
local ulen = mw.ustring.len(summary) or 0
if ulen > targetLen then
local temp = mw.ustring.sub(summary, 1, targetLen + 20)
local breakPos = nil
for i = mw.ustring.len(temp), 1, -1 do
local char = mw.ustring.sub(temp, i, i)
if char:match("[ %s,。!?;:,…、]") then
breakPos = i
break
end
end
if breakPos and breakPos > targetLen - 30 then
summary = mw.ustring.sub(temp, 1, breakPos)
else
summary = mw.ustring.sub(temp, 1, targetLen)
end
summary = mw.text.trim(summary) .. "…"
else
summary = mw.text.trim(summary)
end
-- HTML 输出
local container = mw.html.create('div')
:css('margin-bottom', '25px')
:css('display', 'flow-root')
container:tag('div')
:css('font-size', '1.3em')
:css('font-weight', 'bold')
:css('margin-bottom', '6px')
:css('border-bottom', '1px solid #eee')
:wikitext('[[' .. pageName .. ']]')
local textDiv = container:tag('div')
:css('line-height', '1.6')
:css('color', '#222')
:css('font-size', '14px')
if firstImage and firstImage ~= "" then
textDiv:wikitext('[[' .. firstImage .. '|120px|right|link=' .. pageName .. ']]')
end
textDiv:wikitext(summary)
return tostring(container)
end
return p