Many changes in DBHandler
new general processing structure just before splitting off the projects folder from this repo
This commit is contained in:
32
Classes/DataBaseHandler/related functions/cleanUpTable.m
Normal file
32
Classes/DataBaseHandler/related functions/cleanUpTable.m
Normal file
@@ -0,0 +1,32 @@
|
||||
function cleanedTable = cleanUpTable(inputTable)
|
||||
% Converts string numbers to numeric, 'NaN' to NaN, and tries to convert date strings to datetime.
|
||||
|
||||
cleanedTable = inputTable;
|
||||
varNames = cleanedTable.Properties.VariableNames;
|
||||
|
||||
for i = 1:numel(varNames)
|
||||
col = cleanedTable.(varNames{i});
|
||||
if iscell(col)
|
||||
numericCol = str2double(col);
|
||||
if all(isnan(numericCol) == strcmpi(col, 'NaN') | cellfun(@isempty, col))
|
||||
cleanedTable.(varNames{i}) = numericCol;
|
||||
else
|
||||
try
|
||||
cleanedTable.(varNames{i}) = datetime(col, 'InputFormat', 'yyyy-MM-dd HH:mm:ss.SSSSSS');
|
||||
catch
|
||||
cleanedTable.(varNames{i}) = string(col);
|
||||
end
|
||||
end
|
||||
elseif isstring(col)
|
||||
numericCol = str2double(col);
|
||||
if all(isnan(numericCol) == strcmpi(col, "NaN"))
|
||||
cleanedTable.(varNames{i}) = numericCol;
|
||||
else
|
||||
try
|
||||
cleanedTable.(varNames{i}) = datetime(col, 'InputFormat', 'yyyy-MM-dd HH:mm:ss');
|
||||
catch
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
53
Classes/DataBaseHandler/related functions/groupIt.m
Normal file
53
Classes/DataBaseHandler/related functions/groupIt.m
Normal file
@@ -0,0 +1,53 @@
|
||||
function resultTable = groupIt(fixedVars, dataTable, aggregationFunction)
|
||||
% groupIt Groups data in a table based on fixedVars and applies aggregationFunction to numeric data.
|
||||
%
|
||||
% resultTable = groupIt(fixedVars, dataTable, aggregationFunction)
|
||||
%
|
||||
% Inputs:
|
||||
% fixedVars - Cell array of variable names to group by
|
||||
% dataTable - Input MATLAB table
|
||||
% aggregationFunction - Function handle (e.g., @mean, @min, @max)
|
||||
%
|
||||
% Output:
|
||||
% resultTable - Grouped and aggregated table
|
||||
|
||||
[G, groupKeys] = findgroups(dataTable(:, fixedVars));
|
||||
varNames = dataTable.Properties.VariableNames;
|
||||
nVars = numel(varNames);
|
||||
aggData = cell(height(groupKeys), nVars);
|
||||
groupCount = zeros(height(groupKeys), 1);
|
||||
|
||||
for i = 1:height(groupKeys)
|
||||
idx = (G == i);
|
||||
groupCount(i) = sum(idx);
|
||||
for j = 1:nVars
|
||||
colData = dataTable.(varNames{j});
|
||||
if isnumeric(colData)
|
||||
if any(idx)
|
||||
aggData{i, j} = aggregationFunction(colData(idx));
|
||||
else
|
||||
aggData{i, j} = NaN;
|
||||
end
|
||||
else
|
||||
if iscell(colData)
|
||||
nonEmptyIdx = find(idx & ~cellfun(@isempty, colData), 1);
|
||||
if ~isempty(nonEmptyIdx)
|
||||
aggData{i, j} = colData{nonEmptyIdx};
|
||||
else
|
||||
aggData{i, j} = [];
|
||||
end
|
||||
else
|
||||
nonEmptyIdx = find(idx, 1);
|
||||
if ~isempty(nonEmptyIdx)
|
||||
aggData{i, j} = colData(nonEmptyIdx);
|
||||
else
|
||||
aggData{i, j} = [];
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
end
|
||||
|
||||
resultTable = cell2table(aggData, 'VariableNames', varNames);
|
||||
resultTable.nRows = groupCount;
|
||||
end
|
||||
@@ -0,0 +1,61 @@
|
||||
function [cleanedTable, outliersTable] = removeGroupOutliers(dataTable, fixedVars, y_var)
|
||||
% removeGroupOutliers removes outliers in y_var within each group defined by fixedVars.
|
||||
%
|
||||
% [cleanedTable, outliersTable] = removeGroupOutliers(dataTable, fixedVars, y_var)
|
||||
%
|
||||
% Inputs:
|
||||
% dataTable - Input MATLAB table
|
||||
% fixedVars - Cell array of variable names to group by
|
||||
% y_var - Name of the variable to check for outliers (string or char)
|
||||
%
|
||||
% Outputs:
|
||||
% cleanedTable - Table with outliers removed
|
||||
% outliersTable - Table of removed outlier rows
|
||||
|
||||
[G, groupKeys] = findgroups(dataTable(:, fixedVars));
|
||||
keepIdx = true(height(dataTable), 1);
|
||||
outlierRecords = [];
|
||||
|
||||
for groupIdx = 1:height(groupKeys)
|
||||
groupRows = (G == groupIdx);
|
||||
y_values = dataTable.(y_var)(groupRows);
|
||||
|
||||
% Skip groups with fewer than 3 points
|
||||
if sum(groupRows) < 3
|
||||
continue;
|
||||
end
|
||||
|
||||
% Detect outliers in log10 space (robust for BER, etc.)
|
||||
y_log = log10(y_values);
|
||||
outlierMask = isoutlier(y_log, 'quartiles', 1);
|
||||
|
||||
if any(outlierMask)
|
||||
groupData = dataTable(groupRows, :);
|
||||
outlierGroupTable = groupData(outlierMask, :);
|
||||
|
||||
% Optionally, add group key values for traceability
|
||||
for k = 1:numel(fixedVars)
|
||||
outlierGroupTable.(['Group_', fixedVars{k}]) = repmat(groupKeys{groupIdx, k}, height(outlierGroupTable), 1);
|
||||
end
|
||||
|
||||
outlierRecords = [outlierRecords; outlierGroupTable]; %#ok<AGROW>
|
||||
end
|
||||
|
||||
% Mark outliers for removal
|
||||
groupRowIdx = find(groupRows);
|
||||
keepIdx(groupRowIdx(outlierMask)) = false;
|
||||
end
|
||||
|
||||
cleanedTable = dataTable(keepIdx, :);
|
||||
|
||||
if isempty(outlierRecords)
|
||||
outliersTable = table();
|
||||
else
|
||||
outliersTable = outlierRecords;
|
||||
end
|
||||
|
||||
nRemoved = sum(~keepIdx);
|
||||
nTotalOriginal = height(dataTable);
|
||||
fprintf('Removed %d outliers from the data table (%.2f%% of total %d entries).\n', ...
|
||||
nRemoved, 100*nRemoved/nTotalOriginal, nTotalOriginal);
|
||||
end
|
||||
Reference in New Issue
Block a user