Module:SscCitations: Difference between revisions

From Bodhicitta
(Narrow the query with a wildcard, verify exactly in Lua: 11s -> under 1s)
(Narrow the query with a wildcard, verify exactly in Lua: 11s -> under 1s)
Line 90: Line 90:
-- Every quote segment naming this id, grouped into quotations, per alignment.
-- Every quote segment naming this id, grouped into quotations, per alignment.
local function fetchByAlignment(id, selfTitle)
local function fetchByAlignment(id, selfTitle)
-- Narrowed by a wildcard, then verified exactly in Lua below.
-- Narrowed by wildcard, then verified exactly in Lua below.
--
--
-- The wildcard alone is not safe: "~*RKTSK 12*" also matches RKTSK 127 and
-- Fetching all 3,031 quote segments and filtering in Lua is correct but
-- 129. But it cuts 3,031 quote segments down to a few dozen, which takes the
-- costs ~2.5s, which made every cited text page 11 seconds slower to parse.
-- query from ~2.5s to ~0.2s — fetching them all made every cited text page
-- A wildcard cuts that to a few dozen rows in ~0.2s. It is not safe on its
-- 11 seconds slower to parse. The exact test still happens in Lua, so the
-- own — "~*RKTSK 12*" also matches RKTSK 127 and 129 — but it only has to
-- wildcard only has to be a superset.
-- return a SUPERSET, because the exact test still happens below.
local query = mw.smw.ask{
--
'[[QuoteSource::~*' .. id .. '*]]',
-- Two queries, because SMW's wildcard does not match across a newline and 86
-- QuoteSource values carry one (a parenthetical "also known as" on its own
-- line). Those are fetched separately and merged; without this the
-- Ratnameghasūtra silently lost a quotation.
local printouts = {
'?#-=Page',
'?#-=Page',
'?TransMemID#=TransMemID',
'?TransMemID#=TransMemID',
Line 108: Line 112:
'?SegmentTranslation#=SegmentTranslation',
'?SegmentTranslation#=SegmentTranslation',
'?TranslationWikiPage#=TranslationWikiPage',
'?TranslationWikiPage#=TranslationWikiPage',
limit = SEGMENT_LIMIT
}
}
if not query then return {} end
 
local function ask(condition)
local args = { condition }
for _, po in ipairs(printouts) do table.insert(args, po) end
args.limit = SEGMENT_LIMIT
return mw.smw.ask(args) or {}
end
 
local rows = ask('[[QuoteSource::~*' .. id .. '*]]')
local seen = {}
for _, row in ipairs(rows) do
local pg = firstValue(row.Page)
if pg then seen[pg] = true end
end
for _, row in ipairs(ask('[[QuoteSource::~*\n*]]')) do
local pg = firstValue(row.Page)
if not pg or not seen[pg] then table.insert(rows, row) end
end
 
local query = rows


-- Matched in Lua, not by the query. A wildcard on QuoteSource cannot do this
-- Matched in Lua, not by the query. A wildcard on QuoteSource cannot do this

Revision as of 14:34, 28 August 2026

Documentation for this module may be created at Module:SscCitations/doc

local p = {}

-- Where this text is quoted across the wiki's aligned translations.
--
-- Several works have been segmented and aligned in the Translation Memory
-- namespace. A segment that quotes another work carries a QuoteSource naming it,
-- usually with its RKTS catalogue number:
--
--   QuoteSource=Ratnameghasūtra (RKTSK 231)
--
-- Text pages hold the same identifier in DrlPageName ("RKTSK 231"), so that is
-- the join: no title matching, unaffected by spelling, and it keeps Kangyur
-- (RKTSK) and Tengyur (RKTST) texts distinct.
--
-- Sixteen alignments carry quote data, 3,031 segments between them. 112 text
-- pages are cited by at least one, and 53 by more than one — the
-- Laṅkāvatārasūtra is quoted in six different works. Showing only one alignment,
-- as this module first did for the Śikṣāsamuccaya, hid most of that.
--
-- COUNTS ARE QUOTATIONS, NOT SEGMENTS. One quotation is often split across
-- consecutive segments, so counting segments overstates badly: the
-- Pitāputrasamāgamanasūtra has 62 quote segments in the SSC but is quoted 14
-- times, and one of its runs is 36 segments long. SMW cannot express
-- "consecutive", so the grouping is done here.

local SEGMENT_LIMIT = 5000

-- A quotation past this many words of translated text is flagged as long.
-- Word count, not segment count: segments vary wildly in size, and the single
-- longest quotation in the corpus (8,330 words) is ONE segment.
local LONG_QUOTE_WORDS = 400

-- Alignments whose bilingual layout is a single page with no chapter subpages.
-- The segments carry no TranslationChapter, so the link is the base page plus
-- the fragment.
local SINGLE_PAGE_BILINGUAL = {
	['013-Tsadra-BCA-Com-Minyak-Ch9'] = true,
	['011-Tsadra-BCA-Com-Pawo'] = true,
}

-- Alignments whose bilingual subpage is named for the work rather than a
-- chapter. Three Bhāvanākramas share one translation, one page each.
local NAMED_BILINGUAL_SUBPAGE = {
	['020-Tsadra-BC-Root-Bhavanakrama-1'] = 'Bhavanakrama-1',
	['020-Tsadra-BC-Root-Bhavanakrama-2'] = 'Bhavanakrama-2',
	['020-Tsadra-BC-Root-Bhavanakrama-3'] = 'Bhavanakrama-3',
}

local function firstValue(v)
	if type(v) == 'table' then return v[1] end
	return v
end

local function trim(s)
	if type(s) ~= 'string' then return nil end
	s = mw.text.trim(s)
	if s == '' then return nil end
	return s
end

-- The RKTS id for this page, e.g. "RKTSK 231". Absent on anything that is not a
-- canonical text, which is what keeps the feature off every other page without
-- needing a flag.
local function rktsId()
	local title = mw.title.getCurrentTitle().prefixedText
	local query = mw.smw.ask{
		'[[' .. title .. ']]',
		'?DrlPageName#=DrlPageName',
		limit = 1
	}
	if not query or not query[1] then return nil, title end
	local id = trim(firstValue(query[1].DrlPageName))
	if id and id:match('^RKTS[KT]%s') then return id, title end
	return nil, title
end

-- Where a quotation links to in its alignment's bilingual layout.
local function bilingualLink(tm, base, chapter, order)
	if not base then return nil end
	local named = NAMED_BILINGUAL_SUBPAGE[tm]
	if named then
		return base .. '/Bilingual/' .. named .. '#seg-' .. order
	end
	if SINGLE_PAGE_BILINGUAL[tm] or not chapter or chapter == '' then
		return base .. '/Bilingual#seg-' .. order
	end
	return base .. '/Bilingual/' .. chapter .. '#seg-' .. order
end

-- Every quote segment naming this id, grouped into quotations, per alignment.
local function fetchByAlignment(id, selfTitle)
	-- Narrowed by wildcard, then verified exactly in Lua below.
	--
	-- Fetching all 3,031 quote segments and filtering in Lua is correct but
	-- costs ~2.5s, which made every cited text page 11 seconds slower to parse.
	-- A wildcard cuts that to a few dozen rows in ~0.2s. It is not safe on its
	-- own — "~*RKTSK 12*" also matches RKTSK 127 and 129 — but it only has to
	-- return a SUPERSET, because the exact test still happens below.
	--
	-- Two queries, because SMW's wildcard does not match across a newline and 86
	-- QuoteSource values carry one (a parenthetical "also known as" on its own
	-- line). Those are fetched separately and merged; without this the
	-- Ratnameghasūtra silently lost a quotation.
	local printouts = {
		'?#-=Page',
		'?TransMemID#=TransMemID',
		'?QuoteSource#=QuoteSource',
		'?SegmentOrder#=SegmentOrder',
		'?TranslationChapter#=TranslationChapter',
		'?TranslationPageNumber#=TranslationPageNumber',
		'?SourcePageNumber#=SourcePageNumber',
		'?SegmentTranslation#=SegmentTranslation',
		'?TranslationWikiPage#=TranslationWikiPage',
	}

	local function ask(condition)
		local args = { condition }
		for _, po in ipairs(printouts) do table.insert(args, po) end
		args.limit = SEGMENT_LIMIT
		return mw.smw.ask(args) or {}
	end

	local rows = ask('[[QuoteSource::~*' .. id .. '*]]')
	local seen = {}
	for _, row in ipairs(rows) do
		local pg = firstValue(row.Page)
		if pg then seen[pg] = true end
	end
	for _, row in ipairs(ask('[[QuoteSource::~*\n*]]')) do
		local pg = firstValue(row.Page)
		if not pg or not seen[pg] then table.insert(rows, row) end
	end

	local query = rows

	-- Matched in Lua, not by the query. A wildcard on QuoteSource cannot do this
	-- correctly: "~*RKTSK 12*" also matches RKTSK 127 and 129, and anchoring on
	-- the closing bracket to fix that then misses segments where the id is not
	-- last — a segment may name several sources at once:
	--
	--   Bodhisattvapratimokṣa… (RKTSK 248);Vinayaviniścayopāli… (RKTSK 68)
	local needle = '(' .. id .. ')'

	local byTm = {}
	for _, row in ipairs(query) do
		local source = firstValue(row.QuoteSource)
		local order = tonumber(firstValue(row.SegmentOrder))
		if order and type(source) == 'string' and source:find(needle, 1, true) then
			local tm = trim(firstValue(row.TransMemID)) or '(unknown)'
			local text = firstValue(row.SegmentTranslation)
			local words = 0
			if type(text) == 'string' then
				for _ in text:gmatch('%S+') do words = words + 1 end
			end
			byTm[tm] = byTm[tm] or { segments = {} }
			byTm[tm].base = byTm[tm].base or trim(firstValue(row.TranslationWikiPage))
			table.insert(byTm[tm].segments, {
				order   = order,
				chapter = trim(firstValue(row.TranslationChapter)),
				folio   = trim(firstValue(row.SourcePageNumber)),
				pageNum = trim(firstValue(row.TranslationPageNumber)),
				words   = words,
			})
		end
	end

	-- Collapse runs of consecutive SegmentOrder into one quotation, anchored to
	-- the lowest order so a link lands at the start of the passage. A chapter
	-- change breaks a run: two adjacent quotes in different chapters are two
	-- quotations.
	local alignments = {}
	for tm, data in pairs(byTm) do
		table.sort(data.segments, function(a, b) return a.order < b.order end)
		local quotations = {}
		for _, seg in ipairs(data.segments) do
			local last = quotations[#quotations]
			if last and seg.order == last.lastOrder + 1 and seg.chapter == last.chapter then
				last.lastOrder = seg.order
				last.words = last.words + seg.words
			else
				table.insert(quotations, {
					order     = seg.order,
					lastOrder = seg.order,
					chapter   = seg.chapter,
					folio     = seg.folio,
					pageNum   = seg.pageNum,
					words     = seg.words,
				})
			end
		end

		-- A work does not cite itself: an alignment OF this text quoting this
		-- text is the alignment cross-referencing itself, not a citation.
		local isSelf = data.base and (data.base == selfTitle)
		if #quotations > 0 and not isSelf then
			table.insert(alignments, {
				tm         = tm,
				base       = data.base,
				quotations = quotations,
				count      = #quotations,
				isRoot     = tm:find('%-Root%-') ~= nil or tm:match('%-Root$') ~= nil,
			})
		end
	end

	-- Root texts before commentaries, then most-quoted first. A root text
	-- quoting this work is a stronger fact than a commentary doing so.
	table.sort(alignments, function(a, b)
		if a.isRoot ~= b.isRoot then return a.isRoot end
		if a.count ~= b.count then return a.count > b.count end
		return a.tm < b.tm
	end)

	return alignments
end

-- Cached: the template asks for the count to decide whether to render, then for
-- the body. One query per page rather than two.
local cache = nil
local function alignments()
	if cache == nil then
		local id, title = rktsId()
		cache = id and fetchByAlignment(id, title) or {}
	end
	return cache
end

-- Total quotations across every alignment, for the heading.
function p.count()
	local list = alignments()
	local total = 0
	for _, a in ipairs(list) do total = total + a.count end
	if total == 0 then return '' end
	return tostring(total)
end

-- How many works cite this one, for the heading.
function p.works()
	local list = alignments()
	if #list == 0 then return '' end
	return tostring(#list)
end

local function renderQuotation(out, q, a, index)
	table.insert(out, '<div class="ssc-citation">')
	table.insert(out, '<span class="ssc-citation-index">' .. index .. '</span>')
	table.insert(out, '<span class="ssc-citation-body">')
	if q.chapter and q.chapter ~= '' then
		table.insert(out, '<span class="ssc-citation-chapter">Chapter ' .. q.chapter .. '</span>')
	end
	-- Folio first: SourcePageNumber is the block-print folio ("114a"), which is
	-- what a scholar cites. It can hold several values when a passage spans
	-- folios, so only the first is shown.
	if q.folio then
		table.insert(out, '<span class="ssc-citation-page">folio&nbsp;' ..
			(q.folio:gsub(';.*$', '')) .. '</span>')
	elseif q.pageNum then
		table.insert(out, '<span class="ssc-citation-page">p.&nbsp;' .. q.pageNum .. '</span>')
	end
	if q.words > LONG_QUOTE_WORDS then
		table.insert(out, '<span class="ssc-citation-long">long quotation</span>')
	end
	table.insert(out, '</span>')

	-- External-link syntax on a fullurl, not a [[wikilink]]: wiki links cannot
	-- carry target, and these pages take seconds to load — some over twenty — so
	-- following one and coming back is expensive. $wgExternalLinkTarget supplies
	-- target=_blank and the rel attributes.
	local link = bilingualLink(a.tm, a.base, q.chapter, q.order)
	if link then
		local page, fragment = link:match('^(.-)(#.*)$')
		table.insert(out, '<span class="ssc-citation-link ssc-citation-external">[' ..
			'{{fullurl:' .. page .. '}}' .. fragment .. ' Read the passage]</span>')
	end
	table.insert(out, '</div>')
end

-- One panel, with a section per citing work.
function p.list(frame)
	local list = alignments()
	if #list == 0 then return '' end

	local out = {'<div class="ssc-citations">'}
	for _, a in ipairs(list) do
		local label = a.base and a.base:gsub('^[^/]+/', '') or a.tm
		-- Three Bhāvanākramas share one translation, so the page title alone
		-- gives three identical headings. The subpage name distinguishes them.
		local named = NAMED_BILINGUAL_SUBPAGE[a.tm]
		if named then
			label = label .. ' — ' .. named:gsub('Bhavanakrama%-', 'Bhāvanākrama ')
		end
		table.insert(out, '<div class="ssc-work">')
		table.insert(out, '<div class="ssc-work-head">')
		table.insert(out, '<span class="ssc-work-title">' ..
			(a.base and ('[[' .. a.base .. '|' .. label .. ']]') or label) .. '</span>')
		if a.isRoot then
			table.insert(out, '<span class="ssc-work-kind">root text</span>')
		end
		table.insert(out, '<span class="ssc-work-count">' .. a.count ..
			(a.count == 1 and ' quotation' or ' quotations') .. '</span>')
		table.insert(out, '</div>')
		for i, q in ipairs(a.quotations) do
			renderQuotation(out, q, a, i)
		end
		table.insert(out, '</div>')
	end
	table.insert(out, '</div>')

	return frame:preprocess(table.concat(out))
end

return p