Module:CSVReader

From Terra Invicta Official Wiki
Revision as of 05:58, 27 September 2025 by Redwah (talk | contribs) (Replaced the CommaSplit function with patrik's optimized version?)

Documentation for this module may be created at Module:CSVReader/doc

---
---Convert the specified CSV file on the wiki into a Lua table.
---
---The page must contain properly formatted CSV data and only the CSV data, without any explanation, wiki markup, or other content.
---
---@module CSVReader
local CSVReader = {}

--region Dependencies
-- none
--endregion



--region Private constants

local ERROR_MESSAGE_PARAMETER_MISSING = "CSVReader requires as its parameter that you provide the name of a page on the wiki that contains valid CSV data. Call the main function with a parameter that is the name of the desired page, like this:\n  `CSVReader.main('Template:Data_CSV`)`"

local ERROR_PREFIX_TITLE_INVALID_NAME = "CSVReader failed to create a Mediawiki title object with the page name you provided. Double-check that the page name you provided is a properly formatted string and valid page title: "
local ERROR_PREFIX_TITLE_PAGE_DOES_NOT_EXIST = "CSVReader checked the page name you provided, and the page does not exist. Double-check that the page name you provided is correct: "

local ERROR_PREFIX_CONTENT_NOT_FOUND = "CSVReader did not find content on the page whose name you provided. Double check that the page exists and that it contains content: "
local ERROR_PREFIX_CONTENT_NOT_LONG_ENOUGH = "CSVReader found content on the page whose name you provided, but it is too short to be useful. Double-check that you provided the correct name, and that there is content on the page: "

local MINIMUM_LENGTH_THAT_SUGGESTS_VALID_CONTENT = 4

local tinsert = table.insert
local tconcat = table.concat
local slen = string.len
local sfind = string.find
local ssub = string.sub

--endregion



--region Private member variables
--none
--endregion



--region Private methods

-- Splits the lines of the CSV file on the commas.
-- Special handling made for one set of double quotations.
-- Someone better at coding than I am please generalize this to handle multiple quotations and nested quotations.
-- helper to trim unquoted fields
local function trim(s)
    return (s:gsub("^%s+", ""):gsub("%s+$", ""))
end

-- helper to clean up a field, with explicit quoted-flag
local function finish_field(out, field, was_quoted)
    local str = tconcat(field)
    if was_quoted then
        -- quoted field content already excludes surrounding quotes
        str = str:gsub('""', '"')
    else
        -- unquoted field, trim whitespace
        str = trim(str)
    end
    out[#out + 1] = str
end

-- Splits a CSV line into fields, handling quotes, commas-in-quotes, escaped quotes,
-- optional spaces before an opening quote, and spaces between a closing quote and the separator.
function CSVReader.commaSplit(line, sep)
    sep = sep or ","
    if line == "" then return {""} end

    local out, field = {}, {}
    local i, len = 1, #line
    local in_quotes   = false   -- currently inside a quoted field
    local was_quoted  = false   -- did this field start as quoted?
    local after_quote = false   -- just closed quotes; allow spaces until sep/end

    -- true if field buffer currently contains only whitespace
    local function field_is_whitespace()
        if #field == 0 then return true end
        local s = tconcat(field)
        return s:match("^%s*$") ~= nil
    end

    while i <= len do
        local c = line:sub(i, i)

        if in_quotes then
            if c == '"' then
                local nxt = line:sub(i + 1, i + 1)
                if nxt == '"' then
                    field[#field + 1] = '"'   -- escaped quote
                    i = i + 1
                else
                    in_quotes = false         -- closing quote
                    after_quote = true
                end
            else
                field[#field + 1] = c
            end

        else
            if after_quote then
                -- allow only whitespace until separator or end
                if c == sep then
                    finish_field(out, field, was_quoted)
                    field = {}
                    was_quoted = false
                    after_quote = false
                elseif c == " " or c == "\t" or c == "\r" then
                    -- skip padding
                else
                    -- unexpected char after closing quote, treat as literal continuation
                    field[#field + 1] = c
                    after_quote = false
                end

            else
                if c == '"' then
                    -- allow leading spaces before opening quote
                    if field_is_whitespace() then
                        field = {}            -- drop any leading spaces
                        in_quotes = true
                        was_quoted = true
                    else
                        -- quote in the middle of an unquoted field, keep literally
                        field[#field + 1] = c
                    end

                elseif c == sep then
                    finish_field(out, field, was_quoted)
                    field = {}
                    was_quoted = false

                elseif c ~= "\r" then
                    field[#field + 1] = c
                end
            end
        end

        i = i + 1
    end

    finish_field(out, field, was_quoted)
    return out
end

--endregion

--region Public methods

---Loads CSV data from the specified page on the wiki into a Lua table.
---@param wikiPageName string the title of the wiki page, including
---@return table
function CSVReader.convertCSVToLuaTable(wikiPageName)

	-- If Args Table from #invoke, then convert to String
	if type(wikiPageName) == 'table' then
		wikiPageName = wikiPageName.args[1]
	end

    -- Verify the parameter.
    if not wikiPageName or "" == wikiPageName then
        error(ERROR_MESSAGE_PARAMETER_MISSING)
    end

    -- Verify that the name is valid and that the page exists.
    local titleObject = mw.title.new(wikiPageName)
    if not titleObject then
        error(ERROR_PREFIX_TITLE_INVALID_NAME .. wikiPageName)
    end
    if not titleObject.exists then
        error(ERROR_PREFIX_TITLE_PAGE_DOES_NOT_EXIST .. wikiPageName)
    end

    -- Verify that the page content can be loaded and that there is something there to use.
    local unparsedContent = titleObject:getContent()
    if not unparsedContent then
        error(ERROR_PREFIX_CONTENT_NOT_FOUND .. wikiPageName)
    end
    if #unparsedContent < MINIMUM_LENGTH_THAT_SUGGESTS_VALID_CONTENT then
        error(ERROR_PREFIX_CONTENT_NOT_LONG_ENOUGH .. wikiPageName)
    end

	local decodedTable = {} -- Stores the table to return
	local CSVlines = mw.text.split(unparsedContent, '\n', true) -- split the CSV into separate lines.
	local headerNames = CSVReader.commaSplit(CSVlines[1]) -- get the names of each column from the first line.
	
	-- Most TI template files have a dataName column. Find this dataName column, if it exists. 
	local hasDataNameColumn = false
	local DataNameColumnIndex = 0
	for i, columnName in ipairs(headerNames) do
	    if columnName == 'dataName' then
	    	hasDataNameColumn = true
			DataNameColumnIndex = i
	    end
	end
	
	if hasDataNameColumn then
		for j, CSVline in ipairs(CSVlines) do -- Consider each CSV line one by one:
			if j > 1 then -- Skip the header line.
				-- Create a row with name set to the value in the dataName column.
				local rowDataName = CSVReader.commaSplit(CSVline)[DataNameColumnIndex]
				decodedTable[rowDataName] = {} 
				for i, cellData in ipairs(CSVReader.commaSplit(CSVline)) do
					-- Fill in the cell data for the row corresponding to the dataName and the column corresponding to headerNames[i].
					local columnName = headerNames[i]
					decodedTable[rowDataName][columnName] = cellData
				end
			end
		end
	else
		for j, CSVline in ipairs(CSVlines) do -- Consider each CSV line one by one:
			if j > 1 then -- Skip the header line.
				-- Create a row with index j - 1.
				decodedTable[j-1] = {} 
				for i, cellData in ipairs(CSVReader.commaSplit(CSVline)) do
					-- Fill in the cell data for the j-1-th row and the column corresponding to headerNames[i].
					decodedTable[j-1][headerNames[i]] = cellData 
				end
			end
		end
	end

    return decodedTable
end


---Loads CSV data from the specified page on the wiki into a wiki table.
---@param wikiPageName string the title of the wiki page, including
---@return table
function CSVReader.convertCSVToWikiTable(wikiPageName)
	local decodedTable = CSVReader.convertCSVToLuaTable(wikiPageName)
	local returnString = '{| class="wikitable sortable mw-collapsible mw-collapsed" style="text-align: center;\n|-\n'
	local headersNotDone = true
	for rowID, row in pairs(decodedTable) do
    	if headersNotDone then
    		for column, cellData in pairs(row) do
		    	returnString = returnString .. '! ' .. column .. '\n' 
		    end
		    returnString = returnString .. '|-\n'
		    headersNotDone = false
    	end
	    for column, cellData in pairs(row) do
	    	returnString = returnString .. '| ' .. cellData .. '\n' 
	    end
	    returnString = returnString .. '|-\n'
	end
	returnString = returnString .. '|}'
	return returnString
end

---Loads CSV data from TIBilateralTemplate and turns it into a Lua table of claims.
---@return table
function CSVReader.convertBilateralToClaimTable()

    -- Verify that the name is valid and that the page exists.
    local titleObject = mw.title.new('TIBilateralTemplate_CSV')
    if not titleObject then
        error(ERROR_PREFIX_TITLE_INVALID_NAME .. 'TIBilateralTemplate_CSV')
    end
    if not titleObject.exists then
        error(ERROR_PREFIX_TITLE_PAGE_DOES_NOT_EXIST .. 'TIBilateralTemplate_CSV')
    end

    -- Verify that the page content can be loaded and that there is something there to use.
    local unparsedContent = titleObject:getContent()
    if not unparsedContent then
        error(ERROR_PREFIX_CONTENT_NOT_FOUND .. 'TIBilateralTemplate_CSV')
    end
    if #unparsedContent < MINIMUM_LENGTH_THAT_SUGGESTS_VALID_CONTENT then
        error(ERROR_PREFIX_CONTENT_NOT_LONG_ENOUGH .. 'TIBilateralTemplate_CSV')
    end

	local nationToRegionTable = {} 
	local regionToNationTable = {}
	local claimTypeTable = {}
	local CSVlines = mw.text.split(unparsedContent, '\n', true) -- split the CSV into separate lines.
	local headerNames = CSVReader.commaSplit(CSVlines[1]) -- get the names of each column from the first line.
	local headerNamesIndices = {}
	for i, headerName in ipairs(headerNames) do -- Find the indices of the target headers.
		headerNamesIndices[headerName] = i
	end

	for j, CSVline in ipairs(CSVlines) do -- Consider each CSV line one by one:
		local rowData = CSVReader.commaSplit(CSVline)
		if rowData[headerNamesIndices['relationType']] == 'Claim' then -- Skip any non-claim lines.
			local nationName = rowData[headerNamesIndices['nation1']]
			local regionName = rowData[headerNamesIndices['region1']]
			if not nationToRegionTable[nationName] then
				nationToRegionTable[nationName] = {}
			end
			tinsert(nationToRegionTable[nationName],regionName)
			if not regionToNationTable[regionName] then
				regionToNationTable[regionName]  = {}
			end
			tinsert(regionToNationTable[regionName],nationName)
			if not claimTypeTable[nationName] then
				claimTypeTable[nationName] = {}
			end
			if not claimTypeTable[nationName][regionName] then
				claimTypeTable[nationName][regionName] = {}
			end
			claimTypeTable[nationName][regionName]['initialOwner'] = rowData[headerNamesIndices['initialOwner']]
			claimTypeTable[nationName][regionName]['capitalClaim'] = rowData[headerNamesIndices['capitalClaim']]
			claimTypeTable[nationName][regionName]['projectUnlockName'] = rowData[headerNamesIndices['projectUnlockName']]
			claimTypeTable[nationName][regionName]['initialColony'] = rowData[headerNamesIndices['initialColony']]
		end
	end

    return {nationToRegionTable,regionToNationTable,claimTypeTable}
end

--endregion

return CSVReader