Soft launch: Soldier biographies and research notes vary in completeness. Prefer class-roster, NARA, newspapers, and other cited sources for contested facts — treat Research notes as leads, not settled history. Anyone can correct a page; contact us with questions.

Module:Soldier

From Ritchie Boys Wiki
Revision as of 03:45, 19 August 2026 by Admin (talk | contribs) (Do not link or categorize bogus IPW labels)
Jump to navigation Jump to search

Documentation for this module may be created at Module:Soldier/doc

local p = {}

local function trim(s)
	return mw.text.trim(tostring(s or ''))
end

local function countChar(s, ch)
	local _, n = mw.ustring.gsub(s, ch, '')
	return n
end

local function tidyDisplay(s)
	s = trim(s)
	if s == '' then
		return ''
	end
	s = mw.text.decode(s, true)
	s = mw.ustring.gsub(s, mw.ustring.char(160), ' ')
	s = mw.ustring.gsub(s, mw.ustring.char(0xFFFD), '')
	s = mw.ustring.gsub(s, mw.ustring.char(0x200B), '')
	s = trim(s)
	local lower = mw.ustring.lower(s)
	if lower == '#n/a' or lower == 'n/a' or lower == 'none' or lower == 'null' or lower == '-' then
		return ''
	end
	if s == '.' or s == '?' then
		return ''
	end
	local prev
	repeat
		prev = s
		s = mw.ustring.gsub(s, '""', '"')
	until s == prev
	s = trim(s)

	local changed = true
	while changed and mw.ustring.len(s) >= 2 do
		changed = false
		local first = mw.ustring.sub(s, 1, 1)
		local last = mw.ustring.sub(s, -1)
		if first == '"' and last == '"' then
			local inner = mw.ustring.sub(s, 2, -2)
			if countChar(inner, '"') == 0 then
				s = trim(inner)
				changed = true
			elseif countChar(s, '"') % 2 == 1 then
				local second = mw.ustring.find(s, '"', 2, true)
				local slen = mw.ustring.len(s)
				if second and second + 1 < slen then
					local nxt = mw.ustring.sub(s, second + 1, second + 1)
					if mw.ustring.find(nxt, '%w') then
						s = mw.ustring.sub(s, 2)
					else
						s = mw.ustring.sub(s, 1, -2)
					end
				else
					s = mw.ustring.sub(s, 1, -2)
				end
				s = trim(s)
				changed = true
			end
		elseif first == "'" and last == "'" then
			local inner = mw.ustring.sub(s, 2, -2)
			if not mw.ustring.find(inner, "'", 1, true) then
				s = trim(inner)
				changed = true
			end
		end
	end

	local q = countChar(s, '"')
	local first = mw.ustring.sub(s, 1, 1)
	local last = mw.ustring.sub(s, -1)
	if first == '"' and q == 1 and last ~= '"' then
		s = trim(mw.ustring.sub(s, 2))
	end
	last = mw.ustring.sub(s, -1)
	first = mw.ustring.sub(s, 1, 1)
	q = countChar(s, '"')
	if last == '"' and q == 1 and first ~= '"' then
		s = trim(mw.ustring.sub(s, 1, -2))
	end
	s = mw.ustring.gsub(s, '^\\+', '')
	s = trim(s)
	if s == '"' or s == "'" or s == '.' or s == '?' or s == '\\' then
		return ''
	end
	s = mw.ustring.gsub(s, '%s+', ' ')
	return trim(s)
end

local function parentArg(frame, name)
	local parent = frame:getParent()
	local raw
	if parent and parent.args[name] then
		raw = parent.args[name]
	else
		raw = frame.args[name]
	end
	return tidyDisplay(raw)
end

local ACRONYMS = {
	US = true, USA = true, POW = true, IPW = true, MII = true, OSS = true,
	MITC = true, ASN = true, WWII = true, NARA = true, DC = true, DEML = true,
	CSU = true, ASTP = true, CAC = true, KIA = true, MIA = true, NG = true,
	MRBC = true, PACMIRS = true, GED = true, MITU = true, RDC = true,
}

local SMALL_WORDS = {
	of = true, the = true, to = true, a = true, an = true, at = true, by = true,
}
SMALL_WORDS['and'] = true
SMALL_WORDS['for'] = true
SMALL_WORDS['in'] = true
SMALL_WORDS['or'] = true

local STATE_NAME = {
	alabama = 'Alabama', alaska = 'Alaska', arizona = 'Arizona', arkansas = 'Arkansas',
	california = 'California', colorado = 'Colorado', connecticut = 'Connecticut',
	delaware = 'Delaware', florida = 'Florida', georgia = 'Georgia', hawaii = 'Hawaii',
	idaho = 'Idaho', illinois = 'Illinois', indiana = 'Indiana', iowa = 'Iowa',
	kansas = 'Kansas', kentucky = 'Kentucky', louisiana = 'Louisiana', maine = 'Maine',
	maryland = 'Maryland', massachusetts = 'Massachusetts', michigan = 'Michigan',
	minnesota = 'Minnesota', mississippi = 'Mississippi', missouri = 'Missouri',
	montana = 'Montana', nebraska = 'Nebraska', nevada = 'Nevada', ohio = 'Ohio',
	oklahoma = 'Oklahoma', oregon = 'Oregon', pennsylvania = 'Pennsylvania',
	tennessee = 'Tennessee', texas = 'Texas', utah = 'Utah', vermont = 'Vermont',
	virginia = 'Virginia', washington = 'Washington', wisconsin = 'Wisconsin',
	wyoming = 'Wyoming',
}

local STATE_ABBR = {
	al = 'Alabama', ak = 'Alaska', az = 'Arizona', ar = 'Arkansas', ca = 'California',
	co = 'Colorado', ct = 'Connecticut', de = 'Delaware', fl = 'Florida', ga = 'Georgia',
	hi = 'Hawaii', id = 'Idaho', il = 'Illinois', ia = 'Iowa',
	ks = 'Kansas', ky = 'Kentucky', la = 'Louisiana', me = 'Maine', md = 'Maryland',
	ma = 'Massachusetts', mi = 'Michigan', mn = 'Minnesota', ms = 'Mississippi',
	mo = 'Missouri', mt = 'Montana', ne = 'Nebraska', nv = 'Nevada', nh = 'New Hampshire',
	nj = 'New Jersey', nm = 'New Mexico', ny = 'New York', nc = 'North Carolina',
	nd = 'North Dakota', oh = 'Ohio', ok = 'Oklahoma', pa = 'Pennsylvania',
	ri = 'Rhode Island', sc = 'South Carolina', sd = 'South Dakota', tn = 'Tennessee',
	tx = 'Texas', ut = 'Utah', vt = 'Vermont', va = 'Virginia', wa = 'Washington',
	wv = 'West Virginia', wi = 'Wisconsin', wy = 'Wyoming', dc = 'Washington, D.C.',
}
STATE_ABBR['in'] = 'Indiana'
STATE_ABBR['or'] = 'Oregon'

local TWO_WORD_STATE = {
	['new hampshire'] = 'New Hampshire',
	['new jersey'] = 'New Jersey',
	['new mexico'] = 'New Mexico',
	['new york'] = 'New York',
	['north carolina'] = 'North Carolina',
	['north dakota'] = 'North Dakota',
	['rhode island'] = 'Rhode Island',
	['south carolina'] = 'South Carolina',
	['south dakota'] = 'South Dakota',
	['west virginia'] = 'West Virginia',
}

local PLACE_PREP = {
	at = true, from = true, near = true, of = true,
}
PLACE_PREP['in'] = true

local function stateKey(w)
	w = mw.ustring.lower(w)
	w = mw.ustring.gsub(w, '[.]', '')
	w = mw.ustring.gsub(w, '%s+', ' ')
	return trim(w)
end

local function canonicalState(w)
	local k = stateKey(w)
	return STATE_NAME[k] or STATE_ABBR[k] or TWO_WORD_STATE[k]
end

local NEGATIVE_PHRASES = {
	'unsatisfactory', 'failed class', 'failed as', 'failure', 'flunk',
	'washed out', 'wash out', 'washout', 'kicked out', 'kick out',
	'dismissed', 'expelled', 'poor performance', 'poor work', 'poor grades',
	'awol', 'court-martial', 'court martial', 'disciplinary',
	'discharged for cause', 'dropped out', 'dropped from', 'also dropped',
	'physically unfit', 'unfit for', 'relieved prior', 'relieved of',
	'kicked', 'eliminated from', 'not graduated', 'did not graduate',
}

local function splitWords(s)
	local words = {}
	for w in mw.text.gsplit(s, ' ', true) do
		if w ~= '' then
			table.insert(words, w)
		end
	end
	return words
end

local function isAllCapsPhrase(s)
	local letters = mw.ustring.gsub(s, '[^%a]', '')
	if mw.ustring.len(letters) < 2 then
		return false
	end
	return letters == mw.ustring.upper(letters)
end

local function prettyTitle(s, keepSmall)
	s = trim(s)
	if s == '' then
		return s
	end
	local words = splitWords(s)
	local out = {}
	for i, w in ipairs(words) do
		local bare = mw.ustring.gsub(w, '[^%a]', '')
		local up = mw.ustring.upper(bare)
		if ACRONYMS[up] then
			local replaced = mw.ustring.gsub(w, bare, up, 1)
			table.insert(out, replaced)
		else
			local low = mw.ustring.lower(w)
			local lowBare = mw.ustring.lower(bare)
			if keepSmall and i > 1 and SMALL_WORDS[lowBare] then
				table.insert(out, mw.ustring.lower(w))
			else
				local first = mw.ustring.upper(mw.ustring.sub(low, 1, 1))
				table.insert(out, first .. mw.ustring.sub(low, 2))
			end
		end
	end
	return table.concat(out, ' ')
end

local NAME_PARTICLES = {
	von = true, van = true, de = true, der = true, den = true, da = true,
	di = true, du = true, la = true, le = true, del = true, della = true,
	ter = true, ten = true, zu = true, zum = true, bin = true,
}

local NAME_SUFFIX = {
	jr = 'Jr', ['jr.'] = 'Jr.', sr = 'Sr', ['sr.'] = 'Sr.',
	esq = 'Esq', ['esq.'] = 'Esq.',
	ii = 'II', iii = 'III', iv = 'IV',
}

local function isAllCapsLetters(s)
	local letters = mw.ustring.gsub(s, '[^%a]', '')
	if mw.ustring.len(letters) < 1 then
		return false
	end
	return letters == mw.ustring.upper(letters)
end

local function isAllLowerLetters(s)
	local letters = mw.ustring.gsub(s, '[^%a]', '')
	if mw.ustring.len(letters) < 2 then
		return false
	end
	return letters == mw.ustring.lower(letters)
end

local function titleCaseWord(w)
	local low = mw.ustring.lower(w)
	if low == '' then
		return w
	end
	return mw.ustring.upper(mw.ustring.sub(low, 1, 1)) .. mw.ustring.sub(low, 2)
end

local function prettyCapsNameWord(w)
	local trail = ''
	while mw.ustring.len(w) > 0 and mw.ustring.find(mw.ustring.sub(w, -1), '[.,]') do
		trail = mw.ustring.sub(w, -1) .. trail
		w = mw.ustring.sub(w, 1, -2)
	end
	local bare = mw.ustring.gsub(w, '[^%a]', '')
	local up = mw.ustring.upper(bare)
	local len = mw.ustring.len(up)
	if len >= 5 and mw.ustring.sub(up, 1, 2) == 'MC' then
		return 'Mc' .. titleCaseWord(mw.ustring.sub(up, 3)) .. trail
	end
	if len >= 7 and mw.ustring.sub(up, 1, 3) == 'MAC' then
		return 'Mac' .. titleCaseWord(mw.ustring.sub(up, 4)) .. trail
	end
	return titleCaseWord(up) .. trail
end

local function prettyNameToken(w, isLeading)
	w = trim(w)
	if w == '' then
		return w
	end
	if mw.ustring.find(w, '%d') then
		return w
	end
	if mw.ustring.find(w, '-', 1, true) then
		local parts = {}
		local idx = 0
		for part in mw.text.gsplit(w, '-', true) do
			idx = idx + 1
			table.insert(parts, prettyNameToken(part, isLeading and idx == 1))
		end
		return table.concat(parts, '-')
	end
	local apos = mw.ustring.find(w, "'", 1, true)
	if not apos then
		apos = mw.ustring.find(w, '’', 1, true)
	end
	if apos then
		local mark = mw.ustring.sub(w, apos, apos)
		local left = mw.ustring.sub(w, 1, apos - 1)
		local right = mw.ustring.sub(w, apos + 1)
		return prettyNameToken(left, isLeading) .. mark .. prettyNameToken(right, false)
	end
	if mw.ustring.match(w, '^(%a%.)+$') then
		if isAllCapsLetters(w) or isAllLowerLetters(w) then
			return mw.ustring.upper(w)
		end
		return w
	end

	local low = mw.ustring.lower(w)
	local mapped = NAME_SUFFIX[low]
	if mapped then
		return mapped
	end

	local bare = mw.ustring.gsub(w, '[^%a]', '')
	local lowBare = mw.ustring.lower(bare)
	if mw.ustring.len(bare) == 1 then
		local rest = mw.ustring.gsub(w, '%a', '', 1)
		return mw.ustring.upper(bare) .. rest
	end
	if NAME_SUFFIX[lowBare] and mw.ustring.match(w, '^%a+%.?$') then
		local canon = NAME_SUFFIX[lowBare]
		if mw.ustring.sub(w, -1) == '.' and mw.ustring.sub(canon, -1) ~= '.' then
			return canon .. '.'
		end
		return canon
	end

	if NAME_PARTICLES[lowBare] and mw.ustring.match(w, '^%a+$')
		and (isAllCapsLetters(w) or isAllLowerLetters(w)) then
		if isLeading then
			return titleCaseWord(lowBare)
		end
		return lowBare
	end
	if isAllCapsLetters(w) and mw.ustring.len(bare) >= 2 and mw.ustring.len(bare) <= 3
		and not mw.ustring.find(mw.ustring.upper(bare), '[AEIOUY]') then
		return w
	end

	if isAllCapsLetters(w) or isAllLowerLetters(w) then
		return prettyCapsNameWord(w)
	end
	return w
end

local function prettyNamePhrase(s, capitalizeLeadingParticle)
	s = trim(s)
	if s == '' then
		return s
	end
	s = mw.ustring.gsub(s, '%s*;%s*', ' ')
	local clauses = {}
	local clauseIdx = 0
	for clause in mw.text.gsplit(s, ',', true) do
		clause = trim(clause)
		if clause ~= '' then
			clauseIdx = clauseIdx + 1
			local words = splitWords(clause)
			local out = {}
			for i, w in ipairs(words) do
				local leading = (clauseIdx == 1 and i == 1 and capitalizeLeadingParticle) or #words == 1
				local bare = mw.ustring.lower(mw.ustring.gsub(w, '[^%a]', ''))
				if NAME_PARTICLES[bare] and (isAllCapsLetters(w) or isAllLowerLetters(w)) then
					if leading then
						table.insert(out, titleCaseWord(mw.ustring.lower(w)))
					else
						table.insert(out, mw.ustring.lower(w))
					end
				else
					table.insert(out, prettyNameToken(w, i == 1))
				end
			end
			if #out > 0 then
				table.insert(clauses, table.concat(out, ' '))
			end
		end
	end
	return table.concat(clauses, ', ')
end

local function titleCase(s)
	return prettyTitle(s, false)
end

local function looksLikeCode(s)
	s = trim(s)
	if s == '' then
		return true
	end
	if mw.ustring.find(s, '%d') and mw.ustring.len(s) <= 8 and not mw.ustring.find(s, ' ') then
		return true
	end
	if mw.ustring.len(s) <= 3 and not mw.ustring.find(s, ' ') then
		return true
	end
	return false
end

local function joinCityState(cityRaw, stateName, srcAllCaps)
	cityRaw = trim(cityRaw)
	if cityRaw == '' then
		return stateName
	end
	local words = splitWords(cityRaw)
	local prepIdx = nil
	for i, w in ipairs(words) do
		local k = mw.ustring.gsub(mw.ustring.lower(w), '[^%a]', '')
		if PLACE_PREP[k] then
			prepIdx = i
		end
	end
	local prefix = ''
	local city = cityRaw
	if prepIdx and prepIdx < #words then
		local preParts = {}
		for i = 1, prepIdx do
			table.insert(preParts, words[i])
		end
		local cityParts = {}
		for i = prepIdx + 1, #words do
			table.insert(cityParts, words[i])
		end
		prefix = table.concat(preParts, ' ')
		city = table.concat(cityParts, ' ')
	end
	local cityFmt = canonicalState(city)
	if not cityFmt then
		cityFmt = prettyTitle(city, false)
	end
	local place
	if stateName == 'Washington, D.C.' and (cityFmt == '' or mw.ustring.lower(cityFmt) == 'washington') then
		place = 'Washington, D.C.'
	else
		place = cityFmt .. ', ' .. stateName
	end
	if prefix ~= '' then
		local prefixFmt = prefix
		if srcAllCaps or isAllCapsPhrase(prefix) then
			prefixFmt = prettyTitle(prefix, true)
		end
		return prefixFmt .. ' ' .. place
	end
	return place
end

local function prettyPlace(s)
	s = trim(s)
	if s == '' then
		return s
	end
	local srcAllCaps = isAllCapsPhrase(s)
	local compact = mw.ustring.lower(s)
	compact = mw.ustring.gsub(compact, '[.,]', '')
	compact = mw.ustring.gsub(compact, '%s+', ' ')
	compact = trim(compact)
	if compact == 'washington dc' or compact == 'washington d c'
		or compact == 'district of columbia' then
		return 'Washington, D.C.'
	end
	if mw.ustring.find(compact, ' washington dc$') or mw.ustring.find(compact, ' washington d c$') then
		local city = mw.ustring.gsub(s, '[Ww][Aa][Ss][Hh][Ii][Nn][Gg][Tt][Oo][Nn]%s+[Dd]%.?%s*[Cc]%.?$', '')
		return joinCityState(trim(city), 'Washington, D.C.', srcAllCaps)
	end
	if mw.ustring.find(s, ',', 1, true) then
		local bits = {}
		for part in mw.text.gsplit(s, ',', true) do
			part = trim(part)
			if part ~= '' then
				table.insert(bits, part)
			end
		end
		if #bits >= 1 then
			local lastKey = stateKey(bits[#bits])
			if lastKey == 'dc' or lastKey == 'd c' then
				local city = ''
				if #bits >= 2 then
					city = table.concat(bits, ', ', 1, #bits - 1)
				end
				return joinCityState(city, 'Washington, D.C.', srcAllCaps)
			end
		end
		local out = {}
		for _, part in ipairs(bits) do
			table.insert(out, canonicalState(part) or prettyTitle(part, false))
		end
		return table.concat(out, ', ')
	end
	local words = splitWords(s)
	if #words >= 2 then
		local last2 = stateKey(words[#words - 1] .. ' ' .. words[#words])
		if TWO_WORD_STATE[last2] then
			local city = {}
			for i = 1, #words - 2 do
				table.insert(city, words[i])
			end
			return joinCityState(table.concat(city, ' '), TWO_WORD_STATE[last2], srcAllCaps)
		end
	end
	if #words >= 1 then
		local lastKey = stateKey(words[#words])
		local state = STATE_NAME[lastKey] or STATE_ABBR[lastKey]
		if state then
			if #words >= 2 then
				local city = {}
				for i = 1, #words - 1 do
					table.insert(city, words[i])
				end
				return joinCityState(table.concat(city, ' '), state, srcAllCaps)
			end
			return state
		end
	end
	if srcAllCaps then
		return prettyTitle(s, false)
	end
	return s
end

local function prettyProse(s)
	s = trim(s)
	if s == '' or looksLikeCode(s) then
		return s
	end
	if isAllCapsPhrase(s) or prettyPlace(s) ~= s then
		local placed = prettyPlace(s)
		if placed ~= s then
			return placed
		end
	end
	if isAllCapsPhrase(s) then
		return prettyTitle(s, true)
	end
	return s
end

local function prettyNamePart(s)
	return prettyNamePhrase(s, false)
end

local function isNegativeClause(s)
	local lower = mw.ustring.lower(s)
	if lower == 'failed' or mw.ustring.find(lower, '^failed%s') or mw.ustring.find(lower, '%sfailed%s') or mw.ustring.find(lower, '%sfailed$') then
		return true
	end
	for _, phrase in ipairs(NEGATIVE_PHRASES) do
		if mw.ustring.find(lower, phrase, 1, true) then
			return true
		end
	end
	return false
end

local function sanitizeNote(s)
	s = tidyDisplay(s)
	if s == '' then
		return ''
	end
	local lower = mw.ustring.lower(s)
	if mw.ustring.find(lower, 'no notes were in the source', 1, true) then
		return ''
	end
	local kept = {}
	for clause in mw.text.gsplit(s, ';', true) do
		clause = trim(clause)
		if clause ~= '' then
			for sent in mw.text.gsplit(clause, '. ', true) do
				sent = trim(sent)
				sent = mw.ustring.gsub(sent, '%.$', '')
				if sent ~= '' and not isNegativeClause(sent) then
					table.insert(kept, prettyProse(sent))
				end
			end
		end
	end
	return table.concat(kept, '; ')
end

local function containsIgnoreCase(haystack, needle)
	return mw.ustring.find(mw.ustring.lower(haystack), mw.ustring.lower(needle), 1, true) ~= nil
end

local function displayName(frame)
	local first = prettyNamePart(parentArg(frame, 'first'))
	local middle = prettyNamePart(parentArg(frame, 'middle'))
	local surname = prettyNamePart(parentArg(frame, 'surname'))
	local suffix = prettyNamePart(parentArg(frame, 'suffix'))
	local parts = {}
	if first ~= '' then
		table.insert(parts, first)
	end
	if middle ~= '' then
		table.insert(parts, middle)
	end
	if surname ~= '' then
		table.insert(parts, surname)
	end
	local name = table.concat(parts, ' ')
	if suffix ~= '' then
		if name ~= '' then
			name = name .. ', ' .. suffix
		else
			name = suffix
		end
	end
	if name == '' then
		name = 'This soldier'
	end
	return name
end

local function classLinkList(raw, withPrefix)
	local out = {}
	for part in mw.text.gsplit(raw, ';', true) do
		part = trim(part)
		if part ~= '' then
			if withPrefix then
				table.insert(out, string.format('[[Class %s]]', part))
			else
				table.insert(out, string.format('[[Class %s|%s]]', part, part))
			end
		end
	end
	return out
end

local function isJunkIpwItalianLabel(part)
	local lower = mw.ustring.lower(trim(part))
	local spaced = mw.ustring.gsub(lower, '%s+', ' ')
	local junk = {
		['16063'] = true, ['16095'] = true, ['16302'] = true,
		['16320'] = true, ['16695'] = true,
		['1st army'] = true, ['2 day'] = true, ['2/23/44'] = true,
		['468th cic'] = true, ['5/7/45'] = true, ['8/24/44'] = true, ['8/7/43'] = true,
		['apid'] = true, ['i-32'] = true, ['mis-eto'] = true, ['mitc'] = true,
		['pac'] = true, ['sc7'] = true, ['shaef'] = true, ['v-831'] = true,
	}
	return junk[spaced] == true
end

local function isJunkIpwLabel(part)
	local lower = mw.ustring.lower(trim(part))
	if lower == 'csu' or lower == 'fh' or lower == 'fid' or lower == 'illness' then
		return true
	end
	local compact = mw.ustring.gsub(lower, '%s+', '')
	if compact == 'docteam' then
		return true
	end
	return false
end

local function normalizeIpwTeamCode(part)
	part = trim(part)
	if part == '' then
		return nil, nil
	end
	local original = part
	part = mw.ustring.gsub(part, '%s*%?+%s*$', '')
	part = trim(part)
	if part == '' then
		return nil, original
	end
	local lower = mw.ustring.lower(part)
	if mw.ustring.find(lower, 'armd', 1, true) or mw.ustring.find(lower, 'armored', 1, true) then
		return nil, original, nil
	end
	local miiNum, miiLetter = mw.ustring.match(part, '^(%d+)%s*[%-–—]%s*([A-Za-z])$')
	if miiNum then
		local mii = miiNum .. '-' .. mw.ustring.upper(miiLetter)
		return nil, mii, mii
	end
	local num = mw.ustring.match(part, '^(%d+)[Tt]$')
	if num then
		return num, num, nil
	end
	if mw.ustring.match(lower, '^%d+(st|nd|rd|th)$') then
		return nil, original, nil
	end
	if isJunkIpwLabel(part) then
		return nil, original, nil
	end
	return part, part, nil
end

local function teamLinkList(raw, pagePrefix, label)
	local out = {}
	local isIpw = mw.ustring.find(pagePrefix, 'IPW', 1, true) == 1
	for part in mw.text.gsplit(raw, ';', true) do
		part = trim(part)
		part = mw.ustring.gsub(part, '[#<>%[%]|{}]', '')
		part = trim(part)
		if part ~= '' then
			if pagePrefix == 'IPW Italian ' then
				if isJunkIpwItalianLabel(part) then
					table.insert(out, part)
				else
					table.insert(out, string.format('[[Team %s%s|%s %s]]', pagePrefix, part, label, part))
				end
			elseif isIpw then
				local code, shown, mii = normalizeIpwTeamCode(part)
				if mii then
					table.insert(out, string.format('[[Team MII %s|MII %s]]', mii, mii))
				elseif code then
					table.insert(out, string.format('[[Team %s%s|%s %s]]', pagePrefix, code, label, shown))
				elseif shown then
					table.insert(out, shown)
				end
			else
				table.insert(out, string.format('[[Team %s%s|%s %s]]', pagePrefix, part, label, part))
			end
		end
	end
	return table.concat(out, ', ')
end

local function assignmentLink(raw)
	raw = trim(raw)
	if raw == '' then
		return ''
	end
	raw = mw.ustring.gsub(raw, '[#<>%[%]|{}]', '')
	raw = trim(raw)
	if raw == '' then
		return ''
	end
	return string.format('[[Assignment %s|%s]]', raw, raw)
end

local function addHeader(tbl, text)
	tbl:tag('tr'):tag('th')
		:attr('colspan', '2')
		:addClass('infobox-header')
		:wikitext(text)
end

local function addRow(tbl, label, value, dataClass)
	if trim(value) == '' then
		return
	end
	local tr = tbl:tag('tr')
	tr:tag('th')
		:addClass('infobox-label')
		:css('display', 'table-cell')
		:css('width', '38%')
		:css('font-weight', 'bold')
		:css('text-align', 'left')
		:css('vertical-align', 'top')
		:wikitext(label)
	local td = tr:tag('td')
		:addClass('infobox-data')
		:css('display', 'table-cell')
		:css('vertical-align', 'top')
	if dataClass and dataClass ~= '' then
		td:addClass(dataClass)
	end
	td:wikitext(value)
end

local function yearOnly(s)
	s = trim(s)
	if s == '' or s == '?' then
		return ''
	end
	local year = mw.ustring.match(s, '(%d%d%d%d)')
	if year then
		return year
	end
	return ''
end

local RANK_MAP = {
	['PVT'] = 'Private',
	['PVTE'] = 'Private',
	['PRIVATE'] = 'Private',
	['PFC'] = 'Private First Class',
	['P F C'] = 'Private First Class',
	['PRIVATE FIRST CLASS'] = 'Private First Class',
	['CPL'] = 'Corporal',
	['CORPORAL'] = 'Corporal',
	['T/5'] = 'Technician Fifth Grade',
	['TEC5'] = 'Technician Fifth Grade',
	['TEC 5'] = 'Technician Fifth Grade',
	['T5'] = 'Technician Fifth Grade',
	['TECH 5'] = 'Technician Fifth Grade',
	['SGT'] = 'Sergeant',
	['SERGEANT'] = 'Sergeant',
	['T/4'] = 'Technician Fourth Grade',
	['TEC4'] = 'Technician Fourth Grade',
	['TEC 4'] = 'Technician Fourth Grade',
	['T4'] = 'Technician Fourth Grade',
	['TECH 4'] = 'Technician Fourth Grade',
	['S/SGT'] = 'Staff Sergeant',
	['SSGT'] = 'Staff Sergeant',
	['S SGT'] = 'Staff Sergeant',
	['STAFF SERGEANT'] = 'Staff Sergeant',
	['T/3'] = 'Technician Third Grade',
	['TEC3'] = 'Technician Third Grade',
	['TEC 3'] = 'Technician Third Grade',
	['T3'] = 'Technician Third Grade',
	['TECH 3'] = 'Technician Third Grade',
	['T/SGT'] = 'Technical Sergeant',
	['TSGT'] = 'Technical Sergeant',
	['T SGT'] = 'Technical Sergeant',
	['TECHNICAL SERGEANT'] = 'Technical Sergeant',
	['1ST SGT'] = 'First Sergeant',
	['1SGT'] = 'First Sergeant',
	['FIRST SERGEANT'] = 'First Sergeant',
	['M/SGT'] = 'Master Sergeant',
	['MSGT'] = 'Master Sergeant',
	['MASTER SERGEANT'] = 'Master Sergeant',
	['WO'] = 'Warrant Officer',
	['WOJG'] = 'Warrant Officer',
	['WARRANT OFFICER'] = 'Warrant Officer',
	['CWO'] = 'Chief Warrant Officer',
	['2ND LT'] = 'Second Lieutenant',
	['2D LT'] = 'Second Lieutenant',
	['2LT'] = 'Second Lieutenant',
	['SECOND LIEUTENANT'] = 'Second Lieutenant',
	['1ST LT'] = 'First Lieutenant',
	['1LT'] = 'First Lieutenant',
	['FIRST LIEUTENANT'] = 'First Lieutenant',
	['LT'] = 'Lieutenant',
	['LIEUTENANT'] = 'Lieutenant',
	['CAPT'] = 'Captain',
	['CPT'] = 'Captain',
	['CAPTAIN'] = 'Captain',
	['MAJ'] = 'Major',
	['MAJOR'] = 'Major',
	['LT COL'] = 'Lieutenant Colonel',
	['LTCOL'] = 'Lieutenant Colonel',
	['LTC'] = 'Lieutenant Colonel',
	['LIEUTENANT COLONEL'] = 'Lieutenant Colonel',
	['COL'] = 'Colonel',
	['COLONEL'] = 'Colonel',
	['BRIG GEN'] = 'Brigadier General',
	['BG'] = 'Brigadier General',
	['BRIGADIER GENERAL'] = 'Brigadier General',
	['MAJ GEN'] = 'Major General',
	['MG'] = 'Major General',
	['MAJOR GENERAL'] = 'Major General',
	['LT GEN'] = 'Lieutenant General',
	['LIEUTENANT GENERAL'] = 'Lieutenant General',
	['GEN'] = 'General',
	['GENERAL'] = 'General',
	['TECH'] = 'Technician',
	['TEC'] = 'Technician',
}

local function expandRank(raw)
	raw = trim(raw)
	if raw == '' then
		return ''
	end
	local key = mw.ustring.upper(raw)
	key = mw.ustring.gsub(key, '%.', '')
	key = mw.ustring.gsub(key, '%s+', ' ')
	key = trim(key)
	return RANK_MAP[key] or raw
end

local BRANCH_MAP = {
	['INF'] = 'Infantry',
	['INFANTRY'] = 'Infantry',
	['FA'] = 'Field Artillery',
	['FIELD ARTILLERY'] = 'Field Artillery',
	['CAC'] = 'Coast Artillery Corps',
	['CA'] = 'Coast Artillery',
	['COAST ARTILLERY'] = 'Coast Artillery',
	['AC'] = 'Air Corps',
	['AAF'] = 'Army Air Forces',
	['AIR CORPS'] = 'Air Corps',
	['QMC'] = 'Quartermaster Corps',
	['QC'] = 'Quartermaster Corps',
	['QUARTERMASTER'] = 'Quartermaster Corps',
	['CAV'] = 'Cavalry',
	['CAVALRY'] = 'Cavalry',
	['CE'] = 'Corps of Engineers',
	['ENG'] = 'Corps of Engineers',
	['ENGINEERS'] = 'Corps of Engineers',
	['CMP'] = 'Corps of Military Police',
	['MP'] = 'Military Police',
	['CWS'] = 'Chemical Warfare Service',
	['AUS'] = 'Army of the United States',
	['SC'] = 'Signal Corps',
	['SIG'] = 'Signal Corps',
	['SIG C'] = 'Signal Corps',
	['SIGC'] = 'Signal Corps',
	['SIGNAL CORPS'] = 'Signal Corps',
	['ORD'] = 'Ordnance Department',
	['ORDNANCE'] = 'Ordnance Department',
	['MAC'] = 'Medical Administrative Corps',
	['MC'] = 'Medical Corps',
	['MD'] = 'Medical Department',
	['TC'] = 'Transportation Corps',
	['MI'] = 'Military Intelligence',
	['MIS'] = 'Military Intelligence Service',
	['INT'] = 'Intelligence',
	['CIC'] = 'Counter Intelligence Corps',
	['AGD'] = "Adjutant General's Department",
	['USMC'] = 'United States Marine Corps',
	['USMCR'] = 'United States Marine Corps Reserve',
	['AA'] = 'Antiaircraft',
	['AAA'] = 'Antiaircraft Artillery',
	['TD'] = 'Tank Destroyer',
	['FD'] = 'Finance Department',
	['GSC'] = 'General Staff Corps',
	['ARMD F'] = 'Armored Force',
	['ARMD'] = 'Armored',
	['ARMOR'] = 'Armor',
	['SSF'] = 'Special Service Force',
	['PA'] = 'Philippine Army',
	['RCE'] = 'Royal Canadian Engineers',
	['RCA'] = 'Royal Canadian Artillery',
	['CHC'] = 'Chaplain Corps',
	['CHP'] = 'Chaplain Corps',
	['CAO'] = 'Civil Affairs',
	['CAU'] = 'Civil Affairs',
}

local function expandBranchToken(raw)
	raw = trim(raw)
	if raw == '' then
		return ''
	end
	local key = mw.ustring.upper(raw)
	key = mw.ustring.gsub(key, '%.', '')
	key = mw.ustring.gsub(key, '%s+', ' ')
	key = trim(key)
	if BRANCH_MAP[key] then
		return BRANCH_MAP[key]
	end
	local left, right = mw.ustring.match(key, '^(.+) %((.+)%)$')
	if left and right then
		left = trim(left)
		right = trim(right)
		local leftExp = BRANCH_MAP[left] or left
		local rightExp = BRANCH_MAP[right] or right
		if BRANCH_MAP[left] or BRANCH_MAP[right] then
			return leftExp .. ' (' .. rightExp .. ')'
		end
	end
	return raw
end

local function expandBranch(raw)
	raw = trim(raw)
	if raw == '' then
		return ''
	end
	local parts = {}
	for part in mw.text.gsplit(raw, ';', true) do
		part = expandBranchToken(part)
		if part ~= '' then
			table.insert(parts, part)
		end
	end
	return table.concat(parts, '; ')
end

local function lifespan(birth, death)
	birth = yearOnly(birth)
	death = yearOnly(death)
	if birth ~= '' and death ~= '' then
		return ' (' .. birth .. '&ndash;' .. death .. ')'
	end
	if birth ~= '' then
		return ' (born ' .. birth .. ')'
	end
	if death ~= '' then
		return ' (died ' .. death .. ')'
	end
	return ''
end

local function preferredRank(frame)
	local rank = expandRank(parentArg(frame, 'rank_final'))
	if rank == '' then
		rank = expandRank(parentArg(frame, 'class_rank'))
	end
	return rank
end

function p.displayName(frame)
	return displayName(frame)
end

function p.classLinks(frame)
	return table.concat(classLinkList(parentArg(frame, 'class_number'), false), ', ')
end

function p.infobox(frame)
	local name = displayName(frame)
	local tbl = mw.html.create('table')
	tbl:addClass('infobox'):addClass('vcard'):addClass('soldier-infobox')

	tbl:tag('tr'):tag('th')
		:attr('colspan', '2')
		:addClass('infobox-above fn')
		:wikitext(name)

	local photo = parentArg(frame, 'photo')
	if photo ~= '' then
		local size = parentArg(frame, 'image_size')
		if size == '' then
			size = '220px'
		end
		local alt = parentArg(frame, 'alt')
		if alt == '' then
			alt = name
		end
		local td = tbl:tag('tr'):tag('td')
			:attr('colspan', '2')
			:addClass('infobox-image')
		td:wikitext(string.format('[[File:%s|%s|alt=%s]]', photo, size, alt))
		local caption = parentArg(frame, 'caption')
		if caption ~= '' then
			td:tag('div'):addClass('infobox-caption'):wikitext(caption)
		end
	end

	local alias = prettyNamePart(parentArg(frame, 'alias'))
	local birth = yearOnly(parentArg(frame, 'birth_year'))
	local death = yearOnly(parentArg(frame, 'death_year'))
	local nativity = prettyPlace(parentArg(frame, 'nativity'))
	local religion = prettyProse(parentArg(frame, 'religion'))
	local marital = prettyProse(parentArg(frame, 'marital'))
	local education = prettyProse(parentArg(frame, 'education'))
	local occupation = prettyProse(parentArg(frame, 'occupation'))
	local state = prettyPlace(parentArg(frame, 'state'))

	if alias ~= '' or birth ~= '' or death ~= '' or nativity ~= ''
		or religion ~= '' or marital ~= '' or education ~= ''
		or occupation ~= '' or state ~= '' then
		addHeader(tbl, 'Personal details')
	end

	addRow(tbl, 'Nickname', alias, 'nickname')
	addRow(tbl, 'Born', birth)
	addRow(tbl, 'Nativity', nativity)
	addRow(tbl, 'Died', death)
	addRow(tbl, 'Religion', religion)
	addRow(tbl, 'Spouse / marital', marital)
	addRow(tbl, 'Education', education)
	addRow(tbl, 'Occupation', occupation)
	if state ~= '' then
		addRow(tbl, 'State', state)
	end

	addHeader(tbl, 'Military service')
	addRow(tbl, 'Allegiance', 'United States')
	local branch = expandBranch(parentArg(frame, 'class_branch'))
	if branch == '' then
		branch = 'United States Army'
	end
	addRow(tbl, 'Branch', branch)
	addRow(tbl, 'ASN', parentArg(frame, 'asn'), 'serial')

	local rankFinal = expandRank(parentArg(frame, 'rank_final'))
	local classRank = expandRank(parentArg(frame, 'class_rank'))
	if rankFinal ~= '' and classRank ~= '' and classRank ~= rankFinal then
		addRow(tbl, 'Rank', rankFinal)
		addRow(tbl, 'Rank at Ritchie', classRank)
	else
		local rank = rankFinal
		if rank == '' then
			rank = classRank
		end
		addRow(tbl, 'Rank', rank)
	end

	local classRaw = parentArg(frame, 'class_number')
	local classBits = classLinkList(classRaw, false)
	if #classBits > 0 then
		addRow(tbl, 'Camp Ritchie', 'Class ' .. table.concat(classBits, ', '))
	end
	addRow(tbl, 'Section', parentArg(frame, 'class_section'))
	addRow(tbl, 'Other class', parentArg(frame, 'other_class'))

	local enlist = prettyPlace(parentArg(frame, 'enlistment_place'))
	if enlist ~= '' then
		addRow(tbl, 'Enlisted', enlist)
	end
	addRow(tbl, 'Assignment', assignmentLink(parentArg(frame, 'assignment')))
	addRow(tbl, 'Unit / division', prettyProse(parentArg(frame, 'division')))
	addRow(tbl, 'IPW (German)', teamLinkList(parentArg(frame, 'ipw_ge'), 'IPW ', 'IPW'))
	addRow(tbl, 'IPW (Italian)', teamLinkList(parentArg(frame, 'ipw_it'), 'IPW Italian ', 'IPW Italian'))
	addRow(tbl, 'LOC', parentArg(frame, 'loc'))
	addRow(tbl, 'Arrived', parentArg(frame, 'date_arrive'))
	addRow(tbl, 'Departed', parentArg(frame, 'date_depart'))
	addRow(tbl, 'Certificate', parentArg(frame, 'certificate'))

	return tostring(tbl)
end

function p.lead(frame)
	local name = displayName(frame)
	local rank = preferredRank(frame)
	local role = 'United States Army soldier'
	if rank ~= '' then
		role = 'United States Army ' .. rank
	end

	local classLinks = classLinkList(parentArg(frame, 'class_number'), true)
	local trained
	if #classLinks == 1 then
		trained = ' who trained at [[Camp Ritchie]], Maryland, in ' .. classLinks[1]
	elseif #classLinks == 2 then
		trained = ' who trained at [[Camp Ritchie]], Maryland, in ' .. classLinks[1] .. ' and ' .. classLinks[2]
	elseif #classLinks > 2 then
		trained = ' who trained at [[Camp Ritchie]], Maryland, in '
			.. table.concat(classLinks, ', ', 1, #classLinks - 1)
			.. ', and ' .. classLinks[#classLinks]
	else
		trained = ' associated with Camp Ritchie, Maryland'
	end

	local lead = string.format(
		"'''%s'''%s was a %s%s.",
		name,
		lifespan(parentArg(frame, 'birth_year'), parentArg(frame, 'death_year')),
		role,
		trained
	)
	local who = prettyNamePart(parentArg(frame, 'first'))
	if who == '' then
		who = prettyNamePart(parentArg(frame, 'surname'))
	end
	if who == '' then
		who = 'He'
	end
	local enlist = prettyPlace(parentArg(frame, 'enlistment_place'))
	if enlist ~= '' then
		lead = lead .. ' ' .. who .. ' enlisted at ' .. enlist .. '.'
	end
	return lead
end

local function researchNotesSection(content)
	if not content or content == '' then
		return ''
	end
	local startPos = mw.ustring.find(content, '== Research notes ==', 1, true)
	if not startPos then
		return ''
	end
	local after = mw.ustring.sub(content, startPos + 20)
	local nextHead = mw.ustring.find(after, '\n== ', 1, true)
	if nextHead then
		return mw.ustring.sub(after, 1, nextHead - 1)
	end
	return after
end

local function miiPreferredLetter(num)
	local ok, canon = pcall(require, 'Module:TeamCanon')
	if ok and type(canon) == 'table' and type(canon.letter) == 'table' then
		return canon.letter[tostring(num)]
	end
	return nil
end

local function noteTeamNames(text)
	text = mw.ustring.upper(tostring(text or ''))
	text = mw.ustring.gsub(text, mw.ustring.char(0x2013), '-')
	text = mw.ustring.gsub(text, mw.ustring.char(0x2014), '-')
	text = mw.ustring.gsub(text, mw.ustring.char(0x2212), '-')
	local seen = {}
	local names = {}
	local letteredNums = {}
	local function add(name)
		if name ~= '' and not seen[name] then
			seen[name] = true
			table.insert(names, name)
		end
	end
	local work = text
	for num, letter in string.gmatch(work, 'MII[%s%-]+(%d+)%s*%-%s*([A-Z])') do
		add('MII ' .. num .. '-' .. letter)
		letteredNums[num] = true
	end
	work = string.gsub(work, 'MII[%s%-]+%d+%s*%-%s*[A-Z]', ' ')
	for num, letter in string.gmatch(' ' .. text .. ' ', '[^%d]([45]%d%d)%s*%-%s*([A-Z])[^A-Z]') do
		add('MII ' .. num .. '-' .. letter)
		letteredNums[num] = true
	end
	for num in string.gmatch(work, 'MII[%s%-]+(%d+)') do
		if not letteredNums[num] then
			local preferred = miiPreferredLetter(num)
			if preferred then
				add('MII ' .. num .. '-' .. preferred)
			else
				add('MII ' .. num)
			end
		end
	end
	for num, letter in string.gmatch(text, 'IPW%s+(%d+)%s*%-%s*([A-Z])') do
		add('MII ' .. num .. '-' .. letter)
	end
	for num in string.gmatch(text, 'IPW%s+(%d+)[Tt]?%s*%??%f[^%a%d%-]') do
		add('IPW ' .. num)
	end
	return names
end

function p.noteTeamCategories(frame)
	local title = mw.title.getCurrentTitle()
	if not title or not title.getContent then
		return ''
	end
	local ok, content = pcall(function()
		return title:getContent()
	end)
	if not ok or not content then
		return ''
	end
	local bits = {}
	for _, name in ipairs(noteTeamNames(researchNotesSection(content))) do
		table.insert(bits, '[[Category:Team ' .. name .. ']]')
	end
	return table.concat(bits)
end

function p.researchNotes(frame)
	local title = mw.title.getCurrentTitle()
	if not title or not title.getContent then
		return ''
	end
	local ok, content = pcall(function()
		return title:getContent()
	end)
	if not ok or not content or content == '' then
		return ''
	end
	local section = researchNotesSection(content)
	if section == '' then
		return ''
	end
	local items = {}
	for line in mw.text.gsplit(section, '\n', true) do
		line = trim(line)
		if mw.ustring.sub(line, 1, 2) == '* ' then
			local note = sanitizeNote(mw.ustring.sub(line, 3))
			if note ~= '' then
				table.insert(items, note)
			end
		end
	end
	if #items == 0 then
		return ''
	end
	local wrap = mw.html.create('div')
	wrap:addClass('soldier-research-notes')
	wrap:tag('div'):addClass('soldier-research-notes-heading'):wikitext('Research notes')
	local ul = wrap:tag('ul')
	for _, note in ipairs(items) do
		ul:tag('li'):wikitext(note)
	end
	return tostring(wrap)
end

return p