Module:SscCitations: Difference between revisions

From Bodhicitta
(VSCode edit)
(VSCode edit)
Line 257: Line 257:
local function excerpt(text)
local function excerpt(text)
if type(text) ~= 'string' then return nil end
if type(text) ~= 'string' then return nil end
-- Stripped rather than escaped: this sits inside a row that already carries
-- Translation notes come through as SMW highlighter tooltips carrying the
-- a link, where stray markup can swallow the rest of the line.
-- ENTIRE note in a nested span. Stripping only the tags glues the note into
-- the prose ("give upxxvPLV emends parivarja to…"), so the marker and the
-- note body are removed as elements before any tag stripping happens.
text = text:gsub('<span[^>]-class="[^"]-smw%-highlighter.-</span></span>', ' ')
text = text:gsub('<span[^>]-class="[^"]-smwttcontent.-</span>', ' ')
text = text:gsub('<span[^>]-class="[^"]-smwtticon.-</span>', ' ')
text = text:gsub('<sup.-</sup>', ' ')
text = text:gsub('<ref.-</ref>', ' '):gsub('<ref[^>]-/>', ' ')
-- Remaining inline markup is stripped rather than escaped: this sits in a
-- row that already carries a link, where stray markup can swallow the line.
text = text:gsub('<[^>]->', '')
text = text:gsub('<[^>]->', '')
text = text:gsub("'''", ''):gsub("''", '')
text = text:gsub("'''", ''):gsub("''", '')

Revision as of 15:59, 9 September 2026

Documentation for this module may be created at Module:SscCitations/doc

local p = {}

-- Where this text is quoted across the wiki's aligned translations.
--
-- Several works have been segmented and aligned in the Translation Memory
-- namespace. A segment that quotes another work carries a QuoteSource naming it,
-- usually with its RKTS catalogue number:
--
--   QuoteSource=Ratnameghasūtra (RKTSK 231)
--
-- Text pages hold the same identifier in DrlPageName ("RKTSK 231"), so that is
-- the join: no title matching, unaffected by spelling, and it keeps Kangyur
-- (RKTSK) and Tengyur (RKTST) texts distinct.
--
-- Sixteen alignments carry quote data, 3,031 segments between them. 112 text
-- pages are cited by at least one, and 53 by more than one — the
-- Laṅkāvatārasūtra is quoted in six different works. Showing only one alignment,
-- as this module first did for the Śikṣāsamuccaya, hid most of that.
--
-- COUNTS ARE QUOTATIONS, NOT SEGMENTS. One quotation is often split across
-- consecutive segments, so counting segments overstates badly: the
-- Pitāputrasamāgamanasūtra has 62 quote segments in the SSC but is quoted 14
-- times, and one of its runs is 36 segments long. SMW cannot express
-- "consecutive", so the grouping is done here.

local SEGMENT_LIMIT = 5000

-- A quotation past this many words of translated text is flagged as long.
-- Word count, not segment count: segments vary wildly in size, and the single
-- longest quotation in the corpus (8,330 words) is ONE segment.
local LONG_QUOTE_WORDS = 400

-- Alignments whose bilingual layout is a single page with no chapter subpages.
-- The segments carry no TranslationChapter, so the link is the base page plus
-- the fragment.
local SINGLE_PAGE_BILINGUAL = {
	['013-Tsadra-BCA-Com-Minyak-Ch9'] = true,
	['011-Tsadra-BCA-Com-Pawo'] = true,
}

-- Alignments whose bilingual subpage is named for the work rather than a
-- chapter. Three Bhāvanākramas share one translation, one page each.
local NAMED_BILINGUAL_SUBPAGE = {
	['020-Tsadra-BC-Root-Bhavanakrama-1'] = 'Bhavanakrama-1',
	['020-Tsadra-BC-Root-Bhavanakrama-2'] = 'Bhavanakrama-2',
	['020-Tsadra-BC-Root-Bhavanakrama-3'] = 'Bhavanakrama-3',
}

local function firstValue(v)
	if type(v) == 'table' then return v[1] end
	return v
end

local function trim(s)
	if type(s) ~= 'string' then return nil end
	s = mw.text.trim(s)
	if s == '' then return nil end
	return s
end

-- The RKTS id for this page, e.g. "RKTSK 231". Absent on anything that is not a
-- canonical text, which is what keeps the feature off every other page without
-- needing a flag.
local function rktsId()
	local title = mw.title.getCurrentTitle().prefixedText
	local query = mw.smw.ask{
		'[[' .. title .. ']]',
		'?DrlPageName#=DrlPageName',
		limit = 1
	}
	if not query or not query[1] then return nil, title end
	local id = trim(firstValue(query[1].DrlPageName))
	if id and id:match('^RKTS[KT]%s') then return id, title end
	return nil, title
end

-- Where a quotation links to in its alignment's bilingual layout.
local function bilingualLink(tm, base, chapter, order)
	if not base then return nil end
	local named = NAMED_BILINGUAL_SUBPAGE[tm]
	if named then
		return base .. '/Bilingual/' .. named .. '#seg-' .. order
	end
	if SINGLE_PAGE_BILINGUAL[tm] or not chapter or chapter == '' then
		return base .. '/Bilingual#seg-' .. order
	end
	return base .. '/Bilingual/' .. chapter .. '#seg-' .. order
end

-- Every quote segment naming this id, grouped into quotations, per alignment.
local function fetchByAlignment(id, selfTitle)
	-- EVERY quote segment is fetched, and the exact match happens in Lua.
	--
	-- No wildcard is safe here. "~*RKTSK 12*" over-matches RKTSK 127 and 129;
	-- anchoring on the closing bracket then misses segments where the id is not
	-- last, because a segment may name several sources at once:
	--
	--   Bodhisattvapratimokṣa… (RKTSK 248);Vinayaviniścayopāli… (RKTSK 68)
	--
	-- and SMW's wildcard does not match across a newline, which 86 of these
	-- values contain. Neither ~*id* nor ~*\n* reaches
	-- "…(RKTSK 127)\n(also known as …); \nRatnameghasūtra (RKTSK 231)", so a
	-- narrowed query silently lost quotations.
	--
	-- Fetching all 3,031 rows costs ~2.5s, so the result is cached in the SMW
	-- query cache and reused by the count/works/list calls on the same page.
	-- Correctness first: a citation tool that quietly under-reports is worse than
	-- a slow one.
	local query = mw.smw.ask{
		'[[QuoteSource::+]]',
		'?#-=Page',
		'?TransMemID#=TransMemID',
		'?QuoteSource#=QuoteSource',
		'?SegmentOrder#=SegmentOrder',
		'?TranslationChapter#=TranslationChapter',
		'?TranslationPageNumber#=TranslationPageNumber',
		'?SourcePageNumber#=SourcePageNumber',
		'?SegmentTranslation#=SegmentTranslation',
		'?TranslationWikiPage#=TranslationWikiPage',
		'?SourceWikiPage#=SourceWikiPage',
		limit = SEGMENT_LIMIT
	}
	if not query then return {} end

	-- Matched in Lua, not by the query. A wildcard on QuoteSource cannot do this
	-- correctly: "~*RKTSK 12*" also matches RKTSK 127 and 129, and anchoring on
	-- the closing bracket to fix that then misses segments where the id is not
	-- last — a segment may name several sources at once:
	--
	--   Bodhisattvapratimokṣa… (RKTSK 248);Vinayaviniścayopāli… (RKTSK 68)
	local needle = '(' .. id .. ')'

	local byTm = {}
	for _, row in ipairs(query) do
		local source = firstValue(row.QuoteSource)
		local order = tonumber(firstValue(row.SegmentOrder))
		if order and type(source) == 'string' and source:find(needle, 1, true) then
			local tm = trim(firstValue(row.TransMemID)) or '(unknown)'
			local text = firstValue(row.SegmentTranslation)
			local words = 0
			if type(text) == 'string' then
				for _ in text:gmatch('%S+') do words = words + 1 end
			end
			byTm[tm] = byTm[tm] or { segments = {} }
			byTm[tm].base = byTm[tm].base or trim(firstValue(row.TranslationWikiPage))
			-- The work that does the quoting, under its own title — the source
			-- text rather than the English translation whose page hosts the
			-- bilingual layout. A reader looking for who cites this work wants
			-- "Śikṣāsamuccaya", not "The Training Anthology of Śāntideva".
			byTm[tm].source = byTm[tm].source or trim(firstValue(row.SourceWikiPage))
			table.insert(byTm[tm].segments, {
				order   = order,
				chapter = trim(firstValue(row.TranslationChapter)),
				folio   = trim(firstValue(row.SourcePageNumber)),
				pageNum = trim(firstValue(row.TranslationPageNumber)),
				words   = words,
				text    = type(text) == 'string' and text or nil,
			})
		end
	end

	-- Collapse runs of consecutive SegmentOrder into one quotation, anchored to
	-- the lowest order so a link lands at the start of the passage. A chapter
	-- change breaks a run: two adjacent quotes in different chapters are two
	-- quotations.
	local alignments = {}
	for tm, data in pairs(byTm) do
		table.sort(data.segments, function(a, b) return a.order < b.order end)
		local quotations = {}
		for _, seg in ipairs(data.segments) do
			local last = quotations[#quotations]
			if last and seg.order == last.lastOrder + 1 and seg.chapter == last.chapter then
				last.lastOrder = seg.order
				last.words = last.words + seg.words
			else
				table.insert(quotations, {
					order     = seg.order,
					lastOrder = seg.order,
					chapter   = seg.chapter,
					folio     = seg.folio,
					pageNum   = seg.pageNum,
					words     = seg.words,
					text      = seg.text,
				})
			end
		end

		-- A work does not cite itself: an alignment OF this text quoting this
		-- text is the alignment cross-referencing itself, not a citation.
		-- Compared against BOTH the translation page and the source work: an
		-- alignment of this text quoting this text is the alignment
		-- cross-referencing itself, and it may be recorded under either name.
		local isSelf = (data.base and data.base == selfTitle)
			or (data.source and data.source == selfTitle)
		if #quotations > 0 and not isSelf then
			table.insert(alignments, {
				tm         = tm,
				base       = data.base,
				source     = data.source,
				quotations = quotations,
				count      = #quotations,
				isRoot     = tm:find('%-Root%-') ~= nil or tm:match('%-Root$') ~= nil,
			})
		end
	end

	-- Root texts before commentaries, then most-quoted first. A root text
	-- quoting this work is a stronger fact than a commentary doing so.
	table.sort(alignments, function(a, b)
		if a.isRoot ~= b.isRoot then return a.isRoot end
		if a.count ~= b.count then return a.count > b.count end
		return a.tm < b.tm
	end)

	return alignments
end

-- Cached: the template asks for the count to decide whether to render, then for
-- the body. One query per page rather than two.
local cache = nil
local function alignments()
	if cache == nil then
		local id, title = rktsId()
		cache = id and fetchByAlignment(id, title) or {}
	end
	return cache
end

-- Total quotations across every alignment, for the heading.
function p.count()
	local list = alignments()
	local total = 0
	for _, a in ipairs(list) do total = total + a.count end
	if total == 0 then return '' end
	return tostring(total)
end

-- How many works cite this one, for the heading.
function p.works()
	local list = alignments()
	if #list == 0 then return '' end
	return tostring(#list)
end

-- A one-line taste of the passage, shown beside the folio.
--
-- The text is the FIRST segment of the quotation — the one the anchor lands on,
-- so the excerpt matches what the reader sees on arrival. It carries wiki markup
-- and inline HTML from the translation (<em>, <br>), plus bracketed print-page
-- markers like [216], none of which belong in a single muted line.
--
-- Clipping to one line is CSS's job (line-clamp), not Lua's: a character count
-- would cut mid-word at a width this module cannot know. The cap below only
-- keeps a thousand characters per row out of the page.
local EXCERPT_CHARS = 240

local function excerpt(text)
	if type(text) ~= 'string' then return nil end
	-- Translation notes come through as SMW highlighter tooltips carrying the
	-- ENTIRE note in a nested span. Stripping only the tags glues the note into
	-- the prose ("give upxxvPLV emends parivarja to…"), so the marker and the
	-- note body are removed as elements before any tag stripping happens.
	text = text:gsub('<span[^>]-class="[^"]-smw%-highlighter.-</span></span>', ' ')
	text = text:gsub('<span[^>]-class="[^"]-smwttcontent.-</span>', ' ')
	text = text:gsub('<span[^>]-class="[^"]-smwtticon.-</span>', ' ')
	text = text:gsub('<sup.-</sup>', ' ')
	text = text:gsub('<ref.-</ref>', ' '):gsub('<ref[^>]-/>', ' ')
	-- Remaining inline markup is stripped rather than escaped: this sits in a
	-- row that already carries a link, where stray markup can swallow the line.
	text = text:gsub('<[^>]->', '')
	text = text:gsub("'''", ''):gsub("''", '')
	-- Print-page markers are editorial apparatus, not prose.
	text = text:gsub('%[%s*%d+%s*%]', ' ')
	text = text:gsub('%[%[[^|%]]-|([^%]]-)%]%]', '%1'):gsub('%[%[([^%]]-)%]%]', '%1')
	text = text:gsub('&nbsp;', ' '):gsub('%s+', ' ')
	text = mw.text.trim(text)
	if text == '' then return nil end
	if mw.ustring.len(text) > EXCERPT_CHARS then
		text = mw.ustring.sub(text, 1, EXCERPT_CHARS) .. '…'
	end
	return text
end

local function renderQuotation(out, q, a, index)
	table.insert(out, '<div class="ssc-citation">')
	table.insert(out, '<span class="ssc-citation-index">' .. index .. '</span>')
	table.insert(out, '<span class="ssc-citation-body">')
	if q.chapter and q.chapter ~= '' then
		table.insert(out, '<span class="ssc-citation-chapter">Chapter ' .. q.chapter .. '</span>')
	end
	-- Folio first: SourcePageNumber is the block-print folio ("114a"), which is
	-- what a scholar cites. It can hold several values when a passage spans
	-- folios, so only the first is shown.
	if q.folio then
		table.insert(out, '<span class="ssc-citation-page">folio&nbsp;' ..
			(q.folio:gsub(';.*$', '')) .. '</span>')
	elseif q.pageNum then
		table.insert(out, '<span class="ssc-citation-page">p.&nbsp;' .. q.pageNum .. '</span>')
	end
	if q.words > LONG_QUOTE_WORDS then
		table.insert(out, '<span class="ssc-citation-long">long quotation</span>')
	end
	local snippet = excerpt(q.text)
	if snippet then
		table.insert(out, '<span class="ssc-citation-excerpt">' ..
			mw.text.nowiki(snippet) .. '</span>')
	end
	table.insert(out, '</span>')

	-- External-link syntax on a fullurl, not a [[wikilink]]: wiki links cannot
	-- carry target, and these pages take seconds to load — some over twenty — so
	-- following one and coming back is expensive. $wgExternalLinkTarget supplies
	-- target=_blank and the rel attributes.
	local link = bilingualLink(a.tm, a.base, q.chapter, q.order)
	if link then
		local page, fragment = link:match('^(.-)(#.*)$')
		table.insert(out, '<span class="ssc-citation-link ssc-citation-external">[' ..
			'{{fullurl:' .. page .. '}}' .. fragment .. ' Read the passage]</span>')
	end
	table.insert(out, '</div>')
end

-- One panel, with a section per citing work.
function p.list(frame)
	local list = alignments()
	if #list == 0 then return '' end

	local out = {'<div class="ssc-citations">'}
	for _, a in ipairs(list) do
		-- Heading shows the SOURCE work's title; the link still goes to the
		-- translation, because that is where the bilingual layout lives.
		local label = a.source and a.source:gsub('^[^/]+/', '')
			or (a.base and a.base:gsub('^[^/]+/', ''))
			or a.tm
		-- The three Bhāvanākramas needed distinguishing when the heading came
		-- from the shared translation page. Now that it comes from
		-- SourceWikiPage each already reads "… (1 of 3)", so the suffix is only
		-- added if the label does not distinguish them itself.
		local named = NAMED_BILINGUAL_SUBPAGE[a.tm]
		if named and not label:find('%(%d+ of %d+%)') then
			label = label .. ' — ' .. named:gsub('Bhavanakrama%-', 'Bhāvanākrama ')
		end
		table.insert(out, '<div class="ssc-work">')
		table.insert(out, '<div class="ssc-work-head">')
		-- Source title on top, translation beneath in smaller print, each linking
		-- to its own page: the work that quotes, and the edition a reader can
		-- actually open. Only shown when they differ.
		table.insert(out, '<span class="ssc-work-titles">')
		table.insert(out, '<span class="ssc-work-title">' ..
			(a.source and ('[[' .. a.source .. '|' .. label .. ']]') or label) .. '</span>')
		if a.base and a.base ~= a.source then
			table.insert(out, '<span class="ssc-work-translation">[[' .. a.base .. '|' ..
				a.base:gsub('^[^/]+/', '') .. ']]</span>')
		end
		table.insert(out, '</span>')
		if a.isRoot then
			table.insert(out, '<span class="ssc-work-kind">root text</span>')
		end
		table.insert(out, '<span class="ssc-work-count">' .. a.count ..
			(a.count == 1 and ' quotation' or ' quotations') .. '</span>')
		table.insert(out, '</div>')
		for i, q in ipairs(a.quotations) do
			renderQuotation(out, q, a, i)
		end
		table.insert(out, '</div>')
	end
	table.insert(out, '</div>')

	return frame:preprocess(table.concat(out))
end

return p