Module:TextFit: Difference between revisions

Vergir (talk | contribs)
Lift the parser's inner closures to top-level functions taking an explicit state table, and note in the header why the comment uses level-2 long brackets (with help from vergir-bot LLM)
Vergir (talk | contribs)
Measure everything at bold weight: drop the regular width table and the weight-tracking parser, measure each word once (with help from vergir-bot LLM)
 
Line 4: Line 4:
given width and height, using real per-glyph advance widths rather than a
given width and height, using real per-glyph advance widths rather than a
character count.
character count.
Everything is measured at bold weight. Retail Demo's bold advances run about 4%
wider than its regular ones, so measuring regular text against the bold table
can only make the estimate slightly conservative -- text comes out a hair small
rather than overflowing -- and it removes the need to track font weight through
the markup at all. Both callers render their text bold in any case.


Widths are in em: each font's raw advances divided by its own unitsPerEm, so
Widths are in em: each font's raw advances divided by its own unitsPerEm, so
the tables are directly comparable despite the differing internal grids.
the tables are directly comparable despite the differing internal grids.
Extracted with fonttools from:
Extracted with fonttools from:
   Retail Demo regular [[:File:Retaildemo-regular.woff2]]  unitsPerEm 1000
   Retail Demo bold [[:File:Retaildemo-bold.woff2]]  unitsPerEm 1000
  Retail Demo bold    [[:File:Retaildemo-bold.woff2]]    unitsPerEm 1000
   Open Sans         [[:File:Open_Sans.woff2]]       unitsPerEm 2048
   Open Sans           [[:File:Open_Sans.woff2]]           unitsPerEm 2048


Note the level-2 long brackets on this comment: a plain --[[ ]] comment would
Note the level-2 long brackets on this comment: a plain --[[ ]] comment would
Line 19: Line 24:
local ustring = mw.ustring
local ustring = mw.ustring


-- Retail Demo, normal weight
-- Retail Demo, bold weight. Invisible characters are spelled with
local RD_REGULAR = {
-- ustring.char so they cannot be mistaken for a plain space when edited.
[" "]=0.240, [","]=0.221, ["."]=0.221, ["0"]=0.645, ["1"]=0.268, ["2"]=0.549, ["3"]=0.557, ["4"]=0.601,
local WIDTHS = {
["5"]=0.538, ["6"]=0.583, ["7"]=0.487, ["8"]=0.555, ["9"]=0.580, A=0.624, B=0.585, C=0.641,
D=0.696, E=0.554, F=0.525, G=0.695, H=0.703, I=0.237, J=0.268, K=0.626,
L=0.494, M=0.927, N=0.700, O=0.730, P=0.558, Q=0.734, R=0.614, S=0.527,
T=0.550, U=0.679, V=0.624, W=0.939, X=0.665, Y=0.603, Z=0.647, a=0.553,
b=0.587, c=0.530, d=0.588, e=0.551, f=0.394, g=0.544, h=0.600, i=0.238,
j=0.235, k=0.552, l=0.238, m=0.923, n=0.600, o=0.603, p=0.583, q=0.587,
r=0.369, s=0.477, t=0.413, u=0.600, v=0.509, w=0.810, x=0.540, y=0.513,
z=0.554,
[" "]=0.240, -- U+00A0 no-break space
}
 
-- Retail Demo, bold weight
local RD_BOLD = {
[" "]=0.240, [","]=0.281, ["."]=0.281, ["0"]=0.662, ["1"]=0.312, ["2"]=0.560, ["3"]=0.566, ["4"]=0.612,
[" "]=0.240, [","]=0.281, ["."]=0.281, ["0"]=0.662, ["1"]=0.312, ["2"]=0.560, ["3"]=0.566, ["4"]=0.612,
["5"]=0.562, ["6"]=0.601, ["7"]=0.520, ["8"]=0.583, ["9"]=0.598, A=0.677, B=0.590, C=0.621,
["5"]=0.562, ["6"]=0.601, ["7"]=0.520, ["8"]=0.583, ["9"]=0.598, A=0.677, B=0.590, C=0.621,
Line 44: Line 36:
r=0.392, s=0.496, t=0.404, u=0.602, v=0.549, w=0.776, x=0.571, y=0.550,
r=0.392, s=0.496, t=0.404, u=0.602, v=0.549, w=0.776, x=0.571, y=0.550,
z=0.561,
z=0.561,
[" "]=0.240, -- U+00A0 no-break space
[ustring.char(0xA0)]=0.240, -- no-break space
}
}
-- No bold Open Sans is loaded, so browsers synthesise it. Estimated from
-- Retail Demo's measured bold/regular ratio of 1.037.
local SYNTH_BOLD = 1.04


-- Open Sans, for the characters Retail Demo does not contain
-- Open Sans, for the characters Retail Demo does not contain
Line 54: Line 50:
["_"]=0.438, ["`"]=0.277, ["{"]=0.375, ["|"]=0.549, ["}"]=0.375, ["~"]=0.572, ["¡"]=0.264, ["¢"]=0.572,
["_"]=0.438, ["`"]=0.277, ["{"]=0.375, ["|"]=0.549, ["}"]=0.375, ["~"]=0.572, ["¡"]=0.264, ["¢"]=0.572,
["£"]=0.572, ["¤"]=0.572, ["¥"]=0.572, ["¦"]=0.549, ["§"]=0.514, ["¨"]=0.580, ["©"]=0.832, ["ª"]=0.353,
["£"]=0.572, ["¤"]=0.572, ["¥"]=0.572, ["¦"]=0.549, ["§"]=0.514, ["¨"]=0.580, ["©"]=0.832, ["ª"]=0.353,
["«"]=0.496, ["¬"]=0.572, ["­"]=0.322, ["®"]=0.832, ["¯"]=0.500, ["°"]=0.428, ["±"]=0.572, ["²"]=0.348,
["«"]=0.496, ["¬"]=0.572, [ustring.char(0xAD)]=0.322, -- soft hyphen
["®"]=0.832, ["¯"]=0.500, ["°"]=0.428, ["±"]=0.572, ["²"]=0.348,
["³"]=0.348, ["´"]=0.277, ["µ"]=0.618, ["¶"]=0.655, ["·"]=0.263, ["¸"]=0.222, ["¹"]=0.348, ["º"]=0.374,
["³"]=0.348, ["´"]=0.277, ["µ"]=0.618, ["¶"]=0.655, ["·"]=0.263, ["¸"]=0.222, ["¹"]=0.348, ["º"]=0.374,
["»"]=0.496, ["¼"]=0.740, ["½"]=0.768, ["¾"]=0.778, ["¿"]=0.432, ["À"]=0.632, ["Á"]=0.632, ["Â"]=0.632,
["»"]=0.496, ["¼"]=0.740, ["½"]=0.768, ["¾"]=0.778, ["¿"]=0.432, ["À"]=0.632, ["Á"]=0.632, ["Â"]=0.632,
Line 95: Line 92:
["‘"]=0.169, ["’"]=0.169, ["‚"]=0.245, ["‛"]=0.169, ["“"]=0.349, ["”"]=0.349, ["…"]=0.778, ["€"]=0.572,
["‘"]=0.169, ["’"]=0.169, ["‚"]=0.245, ["‛"]=0.169, ["“"]=0.349, ["”"]=0.349, ["…"]=0.778, ["€"]=0.572,
}
}
-- No bold Open Sans is loaded, so browsers synthesise it
-- Estimated from Retail Demo's measured bold/regular of 1.037.
local SYNTH_BOLD = 1.04


-- Characters in neither face (CJK, rare symbols).
-- Characters in neither face (CJK, rare symbols).
Line 104: Line 97:
local FALLBACK  = 0.55
local FALLBACK  = 0.55


local VOID_TAGS = { br = true, img = true, hr = true, input = true,
local function charWidth(ch)
                    meta = true, link = true, wbr = true }
local w = WIDTHS[ch]
local BOLD_TAGS = { b = true, strong = true }
 
local function charWidth(ch, bold)
local w = (bold and RD_BOLD or RD_REGULAR)[ch]
if w then return w end
if w then return w end


w = FALLBACK_WIDTHS[ch]
w = FALLBACK_WIDTHS[ch]
if w then
if w then return w * SYNTH_BOLD end
return bold and w * SYNTH_BOLD or w
end


local cp = ustring.codepoint(ch) or 0
local cp = ustring.codepoint(ch) or 0
Line 128: Line 115:
local function wordWidth(word)
local function wordWidth(word)
local total = 0
local total = 0
for _, part in ipairs(word) do
for ch in ustring.gmatch(word, ".") do
for ch in ustring.gmatch(part.text, ".") do
total = total + charWidth(ch)
total = total + charWidth(ch, part.bold)
end
end
end
return total
return total
end
-- Parser state. `lines` is the finished output; `line`, `word` and `part` are
-- the containers currently being filled. A part is a run of one weight inside
-- a word, so a word split across tags stays a single word.
local function newState()
return { lines = {}, line = {}, word = {}, part = nil }
end
local function endWord(st)
if st.part then
table.insert(st.word, st.part)
st.part = nil
end
if #st.word > 0 then
table.insert(st.line, st.word)
st.word = {}
end
end
local function endLine(st)
endWord(st)
table.insert(st.lines, st.line)
st.line = {}
end
local function pushChar(st, ch, bold)
if st.part == nil or st.part.bold ~= bold then
if st.part then table.insert(st.word, st.part) end
st.part = { text = ch, bold = bold }
else
st.part.text = st.part.text .. ch
end
end
-- True if any open tag on the stack set bold weight.
local function stackIsBold(stack)
for k = #stack, 1, -1 do
if stack[k] then return true end
end
return false
end
end


Line 187: Line 131:
end
end


-- Apply one HTML tag: push or pop the weight stack, or end the line on <br>.
-- Reduce expanded wikitext/HTML to the text a reader actually sees: a list of
local function applyTag(st, stack, tag)
-- lines, each a list of words.
local closing, name = tag:match("^%s*(/?)%s*(%a+)")
-- Tags are deleted rather than replaced, so a word broken across a tag boundary
if not name then return end
-- stays one word. Splitting on ASCII whitespace only leaves NBSP joining its
name = name:lower()
-- neighbours, which is how the browser wraps.
 
function p.parse(src)
if name == "br" then
endLine(st)
elseif VOID_TAGS[name] then -- no effect on weight
elseif closing == "/" then
table.remove(stack)
elseif not tag:match("/%s*$") then
local lower = tag:lower()
table.insert(stack, BOLD_TAGS[name] == true
or lower:match("font%-weight%s*:%s*bold") ~= nil
or lower:match("font%-weight%s*:%s*[6-9]00") ~= nil)
end
end
 
-- Reduce expanded wikitext/HTML to the text a reader actually sees, keeping
-- track of which runs are bold. Returns a list of lines; each line is a list
-- of words; each word is a list of {text=, bold=} parts.
-- Words break only on real whitespace: NBSP and tag boundaries do not split
-- them, matching how the browser wraps.
-- `baseBold` measures every run as bold, for callers whose text is entirely
-- bold by CSS rather than by markup.
function p.parse(src, baseBold)
src = stripMarkup(src)
src = stripMarkup(src)
src = src:gsub("<%s*[bB][rR]%s*/?%s*>", "\n")
src = src:gsub("<[^<>]*>", "")
-- Decode after the tags are gone, so an escaped &lt; cannot look like one.
src = mw.text.decode(src, true)


local st = newState()
local lines = {}
local stack = {}
for line in (src .. "\n"):gmatch("([^\n]*)\n") do
local i, n = 1, #src
local words = {}
 
for word in line:gmatch("%S+") do
while i <= n do
words[#words + 1] = word
if src:sub(i, i) == "<" then
local j = src:find(">", i, true)
if not j then break end
applyTag(st, stack, src:sub(i + 1, j - 1))
i = j + 1
else
local k = src:find("<", i, true) or (n + 1)
local chunk = mw.text.decode(src:sub(i, k - 1), true)
local bold = baseBold == true or stackIsBold(stack)
for ch in ustring.gmatch(chunk, ".") do
if ch == "\n" then
endLine(st)
elseif ch == " " or ch == "\t" or ch == "\r" then
endWord(st)
else
pushChar(st, ch, bold)  -- NBSP falls here: joins, never breaks
end
end
i = k
end
end
lines[#lines + 1] = words
end
end
 
return lines
endLine(st)
return st.lines
end
end


-- Greedy wrap with the usable width normalised to 1, so `r` is the font size
-- Greedy wrap with the usable width normalised to 1, so `r` is the font size
-- expressed as a fraction of that width. Returns the number of rendered lines.
-- expressed as a fraction of that width. Takes lines of word WIDTHS, already
local function lineCount(lines, r, bold)
-- measured by fit(). Returns the number of rendered lines.
local space = charWidth(" ", bold == true) * r
local function lineCount(lines, r)
local space = charWidth(" ") * r
local total = 0
local total = 0
for _, words in ipairs(lines) do
for _, widths in ipairs(lines) do
local cur = 0
local cur = 0
total = total + 1
total = total + 1
for _, word in ipairs(words) do
for _, ww in ipairs(widths) do
local ww = wordWidth(word) * r
ww = ww * r
if cur == 0 then
if cur == 0 then
cur = ww
cur = ww
Line 280: Line 188:
--  max, min    ceiling and floor, also as a percentage of the width
--  max, min    ceiling and floor, also as a percentage of the width
--  step        search granularity
--  step        search granularity
--  bold        measure every run at bold weight, for callers whose text is
--              entirely bold rather than marked up
function p.fit(text, opts)
function p.fit(text, opts)
local aspect = opts.aspect
local aspect = opts.aspect
Line 288: Line 194:
local minPct = opts.min or 1
local minPct = opts.min or 1
local step  = opts.step or 0.1
local step  = opts.step or 0.1
local bold  = opts.bold == true
local lines = p.parse(text, bold)


local widest = 0
-- Measure every word once; the shrink loop below only scales the results.
for _, words in ipairs(lines) do
local lines, widest = {}, 0
for _, word in ipairs(words) do
for _, words in ipairs(p.parse(text)) do
local ww = wordWidth(word)
local widths = {}
if ww > widest then widest = ww end
for i, word in ipairs(words) do
widths[i] = wordWidth(word)
if widths[i] > widest then widest = widths[i] end
end
end
lines[#lines + 1] = widths
end
end
if widest == 0 then return maxPct end
if widest == 0 then return maxPct end
Line 305: Line 211:


-- Then shrink until the wrapped block fits vertically
-- Then shrink until the wrapped block fits vertically
while pct > minPct and lineCount(lines, pct / 100, bold) * lineH * (pct / 100) > aspect do
while pct > minPct and lineCount(lines, pct / 100) * lineH * (pct / 100) > aspect do
pct = pct - step
pct = pct - step
end
end