lamiastella icon

process huge csv files

lamiastella | PRO | 06/22/20 04:02:42 AM UTC | 0 ⭐ | 555 👁️ | Never ⏰ | []
text |

3.24 KB

|

None

|

0 👍

/

0 👎

---------------------------------------------------------------------------
ParserError                               Traceback (most recent call last)
<ipython-input-6-14cf239fd04e> in <module>
      3 import dask.dataframe as dd
      4 
----> 5 df= dd.read_csv("tweets_withheader.csv", quoting=csv.QUOTE_NONE)
      6 
      7 df = df.compute()
 ~/anaconda3/lib/python3.7/site-packages/dask/dataframe/io/csv.py in read(urlpath, blocksize, collection, lineterminator, compression, sample, enforce, assume_missing, storage_options, include_path_column, **kwargs)
    576             storage_options=storage_options,
    577             include_path_column=include_path_column,
--> 578             **kwargs
    579         )
    580 
 ~/anaconda3/lib/python3.7/site-packages/dask/dataframe/io/csv.py in read_pandas(reader, urlpath, blocksize, collection, lineterminator, compression, sample, enforce, assume_missing, storage_options, include_path_column, **kwargs)
    442 
    443     # Use sample to infer dtypes and check for presence of include_path_column
--> 444     head = reader(BytesIO(b_sample), **kwargs)
    445     if include_path_column and (include_path_column in head.columns):
    446         raise ValueError(
 ~/anaconda3/lib/python3.7/site-packages/pandas/io/parsers.py in parser_f(filepath_or_buffer, sep, delimiter, header, names, index_col, usecols, squeeze, prefix, mangle_dupe_cols, dtype, engine, converters, true_values, false_values, skipinitialspace, skiprows, skipfooter, nrows, na_values, keep_default_na, na_filter, verbose, skip_blank_lines, parse_dates, infer_datetime_format, keep_date_col, date_parser, dayfirst, cache_dates, iterator, chunksize, compression, thousands, decimal, lineterminator, quotechar, quoting, doublequote, escapechar, comment, encoding, dialect, error_bad_lines, warn_bad_lines, delim_whitespace, low_memory, memory_map, float_precision)
    674         )
    675 
--> 676         return _read(filepath_or_buffer, kwds)
    677 
    678     parser_f.__name__ = name
 ~/anaconda3/lib/python3.7/site-packages/pandas/io/parsers.py in _read(filepath_or_buffer, kwds)
    452 
    453     try:
--> 454         data = parser.read(nrows)
    455     finally:
    456         parser.close()
 ~/anaconda3/lib/python3.7/site-packages/pandas/io/parsers.py in read(self, nrows)
   1131     def read(self, nrows=None):
   1132         nrows = _validate_integer("nrows", nrows)
-> 1133         ret = self._engine.read(nrows)
   1134 
   1135         # May alter columns / col_dict
 ~/anaconda3/lib/python3.7/site-packages/pandas/io/parsers.py in read(self, nrows)
   2035     def read(self, nrows=None):
   2036         try:
-> 2037             data = self._reader.read(nrows)
   2038         except StopIteration:
   2039             if self._first_chunk:
 pandas/_libs/parsers.pyx in pandas._libs.parsers.TextReader.read()
 pandas/_libs/parsers.pyx in pandas._libs.parsers.TextReader._read_low_memory()
 pandas/_libs/parsers.pyx in pandas._libs.parsers.TextReader._read_rows()
 pandas/_libs/parsers.pyx in pandas._libs.parsers.TextReader._tokenize_rows()
 pandas/_libs/parsers.pyx in pandas._libs.parsers.raise_parser_error()
 ParserError: Error tokenizing data. C error: Expected 25 fields in line 5, saw 26

Comments