跳转到内容

Module:署名解析

来自Bleap Wiki

此模块的文档可以在Module:署名解析/doc创建

-- 名称署名解析: 归一化 + 拆分合作署名
-- 由 模块:曲师列表 / 模块:谱师列表 / 模块:映射检查 共用, 避免拆词规则各写一份而漂移
local p = {}

-- 含分隔符但其实是"单人名义"的写法, 不拆分 (键为 p.norm() 后的值)
p.NO_SPLIT = {
	['w/csθ'] = true,              -- 单人名, 含斜杠
	['↑=192.333kb/s'] = true,      -- 单人名, 含斜杠
	['t+pazolite'] = true,         -- 单人名, 含加号
}

function p.norm(s)
	return (mw.ustring.lower(tostring(s or '')):gsub('%s+', ''))
end

-- 拆分合作者: 先归一化 vs./feat./x 为 '/', 再按分隔符切, 并抽出括号注释
function p.splitRaw(raw)
	local out = {}
	raw = tostring(raw or '')
	raw = raw:gsub('%s+[Vv][Ss]%.?%s+', '/'):gsub('%s+[Ff][Ee][Aa][Tt]%.?%s*', '/')
	raw = raw:gsub('%s+[Xx]%s+', '/')
	for part in mw.text.gsplit(raw, '[/、,,&&%+]') do
		part = mw.text.trim(part)
		if part ~= '' then
			local base, ann = mw.ustring.match(part, '^(.-)%s*[((]([^))]*)[))]')
			if base then
				table.insert(out, { base = mw.text.trim(base), ann = ann })
			else
				table.insert(out, { base = part, ann = '' })
			end
		end
	end
	return out
end

-- 带 NO_SPLIT 白名单的拆分(列表与检查统一走这里)
function p.parts(raw)
	if p.NO_SPLIT[p.norm(raw)] then
		return { { base = raw, ann = '' } }
	end
	return p.splitRaw(raw)
end

-- ===== 两段式映射页 → 查找表(曲师列表/谱师列表/曲目列表/曲目页面 共用)=====
function p.buildLookup(data)
	if type(data) ~= 'table' or type(data['主名义']) ~= 'table' then return nil end
	local alias, collab, extra = {}, {}, {}
	local majia, writing = {}, {}
	for _, m in pairs(data['主名义']) do                     -- 第一趟: 显示名/写法
		if type(m) == 'table' and m['显示名'] then
			local dn = m['显示名']
			extra[p.norm(dn)] = m
			alias[p.norm(dn)] = dn
			writing[p.norm(dn)] = true
			for _, w in ipairs(m['_写法'] or {}) do alias[p.norm(w)] = dn; writing[p.norm(w)] = true end
		end
	end
	for _, m in pairs(data['主名义']) do                     -- 第二趟: 马甲优先
		if type(m) == 'table' and m['显示名'] then
			for _, a in ipairs(m['马甲'] or {}) do
				alias[p.norm(a)] = m['显示名']
				majia[p.norm(a)] = true
			end
		end
	end
	for _, u in pairs(data['未确认名义'] or {}) do            -- 第三段: 主名义未确认
		if type(u) == 'table' and u['名义'] then
			extra[p.norm(u['名义'])] = { ['显示名'] = u['名义'], ['_未确认'] = true }
			alias[p.norm(u['名义'])] = u['名义']
			majia[p.norm(u['名义'])] = true
		end
	end
	for _, c in pairs(data['合作名义']) do
		if type(c) == 'table' and c['名义'] then
			collab[p.norm(c['名义'])] = c['组合'] or {}
		end
	end
	return { alias = alias, collab = collab, extra = extra, majia = majia, writing = writing }
end

-- 是否需要提示「本曲使用的名义」: 只有真·马甲 / 组合标签 / 带括号注释才提示,
-- 纯写法差异(如 TSAR → TSAR崔瀚普、Shiroi-Ice → Shiroi-Ice白井冰)不提示
function p.creditNote(raw, lk)
	if not lk or not raw or raw == '' then return false end
	if lk.writing[p.norm(raw)] then return false end   -- 原名本身就是主名义/写法 → 不提示
	local parts = p.splitRaw(raw)
	local bases = {}
	for _, pt in ipairs(parts) do
		bases[#bases + 1] = pt.base
		if lk.majia[p.norm(pt.base)] then return true end     -- 马甲/未确认
	end
	local key = p.norm(raw)
	local c = lk.collab[key]
	if c and #c > #parts then return true end                 -- 组合标签(名称≠成员名)
	local stripped = p.norm((raw:gsub('[/、,,&&+%s]', '')))
	if stripped ~= p.norm(table.concat(bases, '')) then return true end   -- 含括号注释等额外信息
	return false
end

-- 署名原串 → 归属列表: 合作名义→每位成员; 马甲→主名义(备注标出名义); 其余→原样
function p.attribute(lk, raw)
	if not lk then return nil end
	local key = p.norm(raw)
	local c = lk.collab[key]
	if c then
		-- 组合名义: 若原串段数与成员数一致, 逐段对应(rawPart = 该成员在本曲使用的署名)
		local segs = p.splitRaw(raw)
		local out = {}
		for i, m in ipairs(c) do
			-- 逐段对应必须"该段真能解析到这一位成员",否则视为标签式组合(如 ↑=192.333KB/s)
			local rp = nil
			if #segs == #c then
				local seg = segs[i]
				local resolved = lk.alias[p.norm(seg and seg.base or '')]
				if resolved and p.norm(resolved) == p.norm(m) then rp = seg.base end
			end
			out[#out + 1] = { base = m, ann = '', other = c, rawPart = rp }
		end
		return out
	end
	local m = lk.alias[key]
	if m then
		local ann = (p.norm(raw) == p.norm(m)) and '' or ('名义:' .. raw)
		return { { base = m, ann = ann, rawPart = raw } }   -- 本曲使用的署名 = 原串
	end
	return { { base = raw, ann = '', rawPart = raw } }
end

-- ===== 行内链接化: 保留数据页原串, 把其中认得出的名字各自变成链接 =====
-- 例: ariiol (.feat符白牙) → [[曲师列表#ariiol黄河源|ariiol]] (.feat[[曲师列表#符白牙|符白牙]])
--    B(链接目标) 来自映射页; B'(链接文本) 直接用数据页原串 → 无需重建字符串
-- 返回: 链接化后的字符串; 若原串里没有任何可识别名字则返回 nil(调用方回退到旧逻辑)
local function asciiWord(ch)
	return ch ~= '' and ch:match('^[%w_]$') ~= nil
end

function p.linkify(raw, lk, opts)
	if not lk or not raw or raw == '' then return nil end
	opts = opts or {}
	local page, known = opts.page, opts.known

	-- 归一化串 + 码点位置映射(忽略空白; 小写)
	local chars, map, ci = {}, {}, 0
	for cp in mw.ustring.gcodepoint(raw) do
		ci = ci + 1
		local ch = mw.ustring.char(cp)
		if not mw.ustring.match(ch, '%s') then
			local lch = mw.ustring.lower(ch)
			chars[#chars + 1] = lch
			for _ = 1, #lch do map[#map + 1] = ci end   -- 按字节填充: find 返回的是字节位置
		end
	end
	local nstr = table.concat(chars)
	if nstr == '' then return nil end

	-- 收集所有"马甲/写法"名在归一化串里的出现位置(合作名义不参与行内匹配)
	local cands = {}
	for key, main in pairs(lk.alias) do
		if not lk.collab[key] and #key >= 2 then
			local from = 1
			while true do
				local s, e = nstr:find(key, from, true)
				if not s then break end
				cands[#cands + 1] = { s = map[s], e = map[e], main = main }
				from = e + 1
			end
		end
	end
	if #cands == 0 then return nil end

	-- 先长后短, 去重叠; 并做"词边界"检查(避免 ani 命中 Animosity 这类)
	table.sort(cands, function(a, b)
		if a.s ~= b.s then return a.s < b.s end
		return a.e > b.e
	end)
	local kept, lastEnd = {}, 0
	for _, c in ipairs(cands) do
		if c.s > lastEnd then
			local rlen = mw.ustring.len(raw)
			local prevCh = (c.s > 1) and mw.ustring.sub(raw, c.s - 1, c.s - 1) or ''
			local firstCh = mw.ustring.sub(raw, c.s, c.s)
			local lastCh = mw.ustring.sub(raw, c.e, c.e)
			local nextCh = (c.e < rlen) and mw.ustring.sub(raw, c.e + 1, c.e + 1) or ''
			local badLeft = asciiWord(prevCh) and asciiWord(firstCh)
			local badRight = asciiWord(nextCh) and asciiWord(lastCh)
			if not badLeft and not badRight then
				kept[#kept + 1] = c
				lastEnd = c.e
			end
		end
	end
	if #kept == 0 then return nil end

	-- 拼装: 原串切片 + 链接
	local out, cursor = {}, 1
	for _, c in ipairs(kept) do
		if c.s > cursor then out[#out + 1] = mw.ustring.sub(raw, cursor, c.s - 1) end
		local text = mw.ustring.sub(raw, c.s, c.e):gsub('|', '&#124;')
		if known and not known[c.main] then
			out[#out + 1] = text
		else
			out[#out + 1] = '[[' .. page .. '#' .. c.main:gsub('|', '&#124;') .. '|' .. text .. ']]'
		end
		cursor = c.e + 1
	end
	if cursor <= #raw then
		out[#out + 1] = mw.ustring.sub(raw, cursor)
	end
	return table.concat(out)
end

return p