我正在创建一个分析文件数据质量的工具。所以我需要阅读文件的每一行并分析其中的每一行。我还需要在内存中存储我的文件的所有行,因为用户将能够深入到特定的部分。所以基本上所有的工作都适用于包含数千行的文件。但是,当尝试使用包含超过4百万行的CSV文件时,我会遇到内存不足异常。我认为C#能够处理其内存缓存中的数百万数据,但看起来并不像它。所以我有点卡住,不知道该怎么做。也许我的一段代码不是最高性能的,所以如果你能告诉我一种改进它的方法,那将会很棒吗?只是要记住,我需要在内存中的文件的所有行,因为根据用户的行动,我需要访问特定的行来显示给用户。当读大文件时内存不足
下面是读取每一行
using (FileStream fs = File.Open(this.dlgInput.FileName.ToString(), FileMode.Open, FileAccess.Read, FileShare.Read))
using (BufferedStream bs = new BufferedStream(fs))
using (System.IO.StreamReader sr = new StreamReader(this.dlgInput.FileName.ToString(), Encoding.Default, false, 8192))
{
string line;
if (this.chkSkipHeader.Checked)
{
sr.ReadLine();
}
progressBar1.Visible = true;
int nbOfLines = File.ReadLines(this.dlgInput.FileName.ToString()).Count();
progressBar1.Maximum = nbOfLines;
this.lines = new string[nbOfLines][];
this.patternedLines = new string[nbOfLines][];
for (int i = 0; i < nbOfLines; i++)
{
this.lines[i] = new string[this.dgvFields.Rows.Count];
this.patternedLines[i] = new string[this.dgvFields.Rows.Count];
}
// Read and display lines from the file until the end of
// the file is reached.
while ((line = sr.ReadLine()) != null)
{
this.recordCount += 1;
char[] c = new char[1] { ',' };
System.Text.RegularExpressions.Regex CSVParser = new System.Text.RegularExpressions.Regex(",(?=(?:[^\"]*\"[^\"]*\")*(?![^\"]*\"))");
String[] fields = CSVParser.Split(line);
ParseLine(fields);
this.lines[recordCount - 1] = fields;
progressBar1.PerformStep();
}
}
并且在下面的ParseLine功能也通过阵列一些分析需要保持在存储器中的呼叫:
private void ParseLine(String[] fields2)
{
for (int j = 0; j <= fields2.Length - 1; j++)
{
if ((int)this.dgvFields.Rows[j].Cells["colSelected"].Value == 1)
{
/*' ************************************************
' Save Number of Counts by Value
' ************************************************/
if (this.values[j].ContainsKey(fields2[j]))
{
//values[0] = Dictionary<"TEST", 1> (fields2[0 which is source code] = count])
this.values[j][fields2[j]] += 1;
}
else
{
this.values[j].Add(fields2[j], 1);
}
/* ' ************************************************
' Save Pattern Values/Counts
' ************************************************/
string tmp = System.Text.RegularExpressions.Regex.Replace(fields2[j], "\\p{Lu}", "X");
tmp = System.Text.RegularExpressions.Regex.Replace(tmp, "\\p{Ll}", "x");
tmp = System.Text.RegularExpressions.Regex.Replace(tmp, "[0-9]", "0");
if (this.patterns[j].ContainsKey(tmp))
{
this.patterns[j][tmp] += 1;
}
else
{
this.patterns[j].Add(tmp, 1);
}
this.patternedLines[this.recordCount - 1][j] = tmp;
/* ' ************************************************
' Count Blanks/Alpha/Numeric/Phone/Other
' ************************************************/
if (String.IsNullOrWhiteSpace(fields2[j]))
{
this.blanks[j] += 1;
}
else if (System.Text.RegularExpressions.Regex.IsMatch(fields2[j], "^[0-9]+$"))
{
this.numeric[j] += 1;
}
else if (System.Text.RegularExpressions.Regex.IsMatch(fields2[j].ToUpper().Replace("EXTENSION", "").Replace("EXT", "").Replace("X", ""), "^[0-9()\\- ]+$"))
{
this.phone[j] += 1;
}
else if (System.Text.RegularExpressions.Regex.IsMatch(fields2[j], "^[a-zA-Z ]+$"))
{
this.alpha[j] += 1;
}
else
{
this.other[j] += 1;
}
if (this.recordCount == 1)
{
this.high[j] = fields2[j];
this.low[j] = fields2[j];
}
else
{
if (fields2[j].CompareTo(this.high[j]) > 0)
{
this.high[j] = fields2[j];
}
if (fields2[j].CompareTo(this.low[j]) < 0)
{
this.low[j] = fields2[j];
}
}
}
}
}
更新:新的代码
int nbOfLines = File.ReadLines(this.dlgInput.FileName.ToString()).Count();
//Read file
using (System.IO.StreamReader sr = new StreamReader(this.dlgInput.FileName.ToString(), Encoding.Default, false, 8192))
{
string line;
if (this.chkSkipHeader.Checked)
{ sr.ReadLine(); }
progressBar1.Visible = true;
progressBar1.Maximum = nbOfLines;
this.lines = new string[nbOfLines][];
this.patternedLines = new string[nbOfLines][];
for (int i = 0; i < nbOfLines; i++)
{
this.lines[i] = new string[this.dgvFields.Rows.Count];
this.patternedLines[i] = new string[this.dgvFields.Rows.Count];
}
// Read and display lines from the file until the end of
// the file is reached.
while ((line = sr.ReadLine()) != null)
{
this.recordCount += 1;
char[] c = new char[1] { ',' };
System.Text.RegularExpressions.Regex CSVParser = new System.Text.RegularExpressions.Regex(",(?=(?:[^\"]*\"[^\"]*\")*(?![^\"]*\"))");
String[] fields = CSVParser.Split(line);
ParseLine(fields);
this.lines[recordCount - 1] = fields;
progressBar1.PerformStep();
}
}
请正确格式化您的代码 – byxor
c#无法从无到有创建内存。如果你的数据比适合你的系统内存和/或虚拟内存的数据多,那么你就会陷入困境。要么改变代码的工作方式以减少内存负载,要么获得更多的内存。 –
我有4个内核和16GB的内存 - 对于4百万行文件来说不够吗? –