| Safe Haskell | None |
|---|---|
| Language | Haskell2010 |
DataFrame.IO.CSV
Description
CSV reading and writing for dataframes. A strict, single-pass RFC 4180 scanner (cassava-compatible) parses fields into typed column builders; ragged rows are padded with nulls and extra fields dropped.
Synopsis
- readCsv :: FilePath -> IO DataFrame
- readTsv :: FilePath -> IO DataFrame
- readCsvWithOpts :: ReadOptions -> FilePath -> IO DataFrame
- readSeparated :: ReadOptions -> FilePath -> IO DataFrame
- readCsvWithSchema :: Schema -> FilePath -> IO DataFrame
- type CsvReader = ReadOptions -> FilePath -> IO DataFrame
- schemaReadOptions :: Schema -> ReadOptions
- decodeSeparated :: ReadOptions -> ByteString -> IO DataFrame
- fromCsv :: String -> IO (Either String DataFrame)
- fromCsvBytes :: ByteString -> IO DataFrame
- data ReadOptions = ReadOptions {
- headerSpec :: HeaderSpec
- typeSpec :: TypeSpec
- safeRead :: SafeReadMode
- safeReadOverrides :: [(Text, SafeReadMode)]
- dateFormat :: String
- columnSeparator :: Char
- numRowsToRead :: Maybe Int
- readColumns :: Maybe [Text]
- missingIndicators :: [Text]
- fastCsvOnRaggedRow :: RaggedRowPolicy
- fastCsvOnUnclosedQuote :: UnclosedQuotePolicy
- fastCsvTrimUnquoted :: Bool
- data TypeSpec
- defaultReadOptions :: ReadOptions
- data HeaderSpec
- = NoHeader
- | UseFirstRow
- | ProvideNames [Text]
- data RaggedRowPolicy
- data UnclosedQuotePolicy
- shouldInferFromSample :: TypeSpec -> Bool
- schemaTypeMap :: TypeSpec -> Map Text SchemaType
- typeInferenceSampleSize :: TypeSpec -> Int
- resolveSelection :: ReadOptions -> [Text] -> [(Text, Int)]
- newtype InvalidReadOptions = InvalidReadOptions [Text]
- readOptionsErrors :: ReadOptions -> [Text]
- validateReadOptions :: ReadOptions -> IO ()
- writeCsv :: FilePath -> DataFrame -> IO ()
- writeTsv :: FilePath -> DataFrame -> IO ()
- writeSeparated :: Char -> FilePath -> DataFrame -> IO ()
- stripQuotes :: Text -> Text
Reading
readCsv :: FilePath -> IO DataFrame Source #
Read CSV file from path and load it into a dataframe.
Example
ghci> D.readCsv "./data/taxi.csv"
readTsv :: FilePath -> IO DataFrame Source #
Read TSV (tab separated) file from path and load it into a dataframe.
Example
ghci> D.readTsv "./data/taxi.tsv"
readCsvWithOpts :: ReadOptions -> FilePath -> IO DataFrame Source #
Read CSV file from path and load it into a dataframe.
Example
ghci> D.readCsvWithOpts "./data/taxi.csv" (D.defaultReadOptions { dateFormat = "%d%-m%-Y" })
readSeparated :: ReadOptions -> FilePath -> IO DataFrame Source #
Read text file with specified delimiter into a dataframe.
Example
ghci> D.readSeparated (D.defaultReadOptions { columnSeparator = ';' }) "./data/taxi.txt"
readCsvWithSchema :: Schema -> FilePath -> IO DataFrame Source #
Schema-driven CSV reader. Coerces each column to the type declared
in Schema; columns absent from the schema fall back to inference and are
still returned. To read only the schema's columns, pass
schemaReadOptions to readCsvWithOpts.
import qualified DataFrame as D df <- D.readCsvWithSchema schema "input.csv"
type CsvReader = ReadOptions -> FilePath -> IO DataFrame Source #
A reader the lazy scan can drive: readSeparated here, or
fastReadCsvWithOpts from dataframe-fastcsv. The scan owns the
ReadOptions, so every reader it accepts honours the projection it asks
for.
Example
myReader :: CsvReader myReader opts path = D.readCsvWithOpts opts path
schemaReadOptions :: Schema -> ReadOptions Source #
Read options a scan derives from its Schema: the schema assigns the
column types and selects the columns, so a scan reads only what its
schema names — the same contract as scanParquet.
Example
ghci> schema = D.makeSchema [("id", D.schemaType @Int), ("name", D.schemaType @Text)]
ghci> D.readCsvWithOpts (schemaReadOptions schema) "./data/customers.csv"
decodeSeparated :: ReadOptions -> ByteString -> IO DataFrame Source #
Decode in-memory CSV bytes into a dataframe. The result is fully
forced. (Note: unlike readSeparated, no UTF-8 BOM is stripped.)
Example
ghci> D.decodeSeparated D.defaultReadOptions "id,name\n1,Ada\n"
fromCsv :: String -> IO (Either String DataFrame) Source #
Parse a CSV string into a DataFrame using default options.
Example
ghci> D.fromCsv "id,name\n1,Ada\n" Right (DataFrame ...)
fromCsvBytes :: ByteString -> IO DataFrame Source #
Parse a lazy ByteString containing CSV data into a DataFrame using
default options.
Example
ghci> D.fromCsvBytes "id,name\n1,Ada\n"
Options
data ReadOptions Source #
CSV read parameters.
Constructors
| ReadOptions | |
Fields
| |
How a reader decides each column's type.
Example
ghci> D.readCsvWithOpts D.defaultReadOptions{D.typeSpec = D.InferFromSample 500} "wide.csv"
Constructors
| InferFromSample Int | Infer each column's type from the first |
| SpecifyTypes [(Text, SchemaType)] TypeSpec | Pin the listed columns to their given types; fall back to the
nested |
| NoInference | Every column is read as |
defaultReadOptions :: ReadOptions Source #
The default ReadOptions: infer types from a 100-row sample, treat the
first row as a header, read every row and every column.
Example
ghci> D.readCsvWithOpts D.defaultReadOptions{D.columnSeparator = ';'} "data.csv"
data HeaderSpec Source #
Where a reader's column names come from.
Example
ghci> D.readCsvWithOpts D.defaultReadOptions{D.headerSpec = D.ProvideNames ["a", "b"]} "no_header.csv"
Constructors
| NoHeader | |
| UseFirstRow | Take names from the first row (the default). |
| ProvideNames [Text] | Name the leading columns; any beyond the given list are numbered. |
Instances
| Show HeaderSpec Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods showsPrec :: Int -> HeaderSpec -> ShowS # show :: HeaderSpec -> String # showList :: [HeaderSpec] -> ShowS # | |
| Eq HeaderSpec Source # | |
Defined in DataFrame.IO.CSV.Internal.Options | |
data RaggedRowPolicy Source #
How the fast reader should treat a row whose field count does not
match the header row. Only consulted by dataframe-fastcsv; the pure
Haskell reader always pads short rows with nulls and drops extras.
Constructors
| PadWithNull | Fill missing cells with nulls; silently drop extras. Matches pandas / polars lenient defaults. |
| Truncate | Fill missing cells with nulls; silently drop extras. Alias
kept for ergonomic naming when the caller only cares about the
"don't raise, just forget the extras" half of |
| RaiseOnRagged | Raise a |
Instances
| Show RaggedRowPolicy Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods showsPrec :: Int -> RaggedRowPolicy -> ShowS # show :: RaggedRowPolicy -> String # showList :: [RaggedRowPolicy] -> ShowS # | |
| Eq RaggedRowPolicy Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods (==) :: RaggedRowPolicy -> RaggedRowPolicy -> Bool # (/=) :: RaggedRowPolicy -> RaggedRowPolicy -> Bool # | |
data UnclosedQuotePolicy Source #
How the fast reader should treat an unclosed quoted field at EOF.
Constructors
| RaiseOnUnclosedQuote | Raise a |
| BestEffort | Return whatever rows were parsed before the stray quote; the remainder is silently dropped. |
Instances
| Show UnclosedQuotePolicy Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods showsPrec :: Int -> UnclosedQuotePolicy -> ShowS # show :: UnclosedQuotePolicy -> String # showList :: [UnclosedQuotePolicy] -> ShowS # | |
| Eq UnclosedQuotePolicy Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods (==) :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool # (/=) :: UnclosedQuotePolicy -> UnclosedQuotePolicy -> Bool # | |
shouldInferFromSample :: TypeSpec -> Bool Source #
Whether a TypeSpec still infers types from a sample, once any
SpecifyTypes fallback chain is followed to its end.
Example
>>>shouldInferFromSample (InferFromSample 100)True>>>shouldInferFromSample NoInferenceFalse
schemaTypeMap :: TypeSpec -> Map Text SchemaType Source #
The explicit column/type pairs a TypeSpec declares, if any.
Example
>>>:set -XTypeApplications>>>M.toList (schemaTypeMap (SpecifyTypes [("id", schemaType @Int)] (InferFromSample 100)))[("id",Int)]
typeInferenceSampleSize :: TypeSpec -> Int Source #
The sample size a TypeSpec infers from, once any SpecifyTypes
fallback chain is followed to its end. 0 when the chain ends in
NoInference.
Example
>>>typeInferenceSampleSize (InferFromSample 100)100>>>typeInferenceSampleSize NoInference0
resolveSelection :: ReadOptions -> [Text] -> [(Text, Int)] Source #
Resolve readColumns against a file's header names: the columns to
build, each paired with its field index within a row. Nothing keeps every
column in file order; a selection keeps the order it was written in.
Example
>>>resolveSelection defaultReadOptions{readColumns = Just ["b", "a"]} ["a", "b", "c"][("b",1),("a",0)]>>>resolveSelection defaultReadOptions ["a", "b"][("a",0),("b",1)]
newtype InvalidReadOptions Source #
A ReadOptions value that contradicts itself, with every problem listed.
Thrown by validateReadOptions; see readOptionsErrors for the check
without the exception.
Example
>>>show (InvalidReadOptions ["readColumns is an empty selection; omit it to read every column"])"Invalid ReadOptions:\n - readColumns is an empty selection; omit it to read every column"
Constructors
| InvalidReadOptions [Text] |
Instances
| Exception InvalidReadOptions Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods toException :: InvalidReadOptions -> SomeException # fromException :: SomeException -> Maybe InvalidReadOptions # | |
| Show InvalidReadOptions Source # | |
Defined in DataFrame.IO.CSV.Internal.Options Methods showsPrec :: Int -> InvalidReadOptions -> ShowS # show :: InvalidReadOptions -> String # showList :: [InvalidReadOptions] -> ShowS # | |
readOptionsErrors :: ReadOptions -> [Text] Source #
Everything self-contradictory about opts, judged without looking at a
file. A type or safe-read override naming an unread column is redundant, not
contradictory, and is not reported.
Example
>>>readOptionsErrors defaultReadOptions{readColumns = Just []}["readColumns is an empty selection; omit it to read every column"]>>>readOptionsErrors defaultReadOptions[]
validateReadOptions :: ReadOptions -> IO () Source #
Throw InvalidReadOptions if opts contradicts itself. Both readers
call this before opening the file.
Example
ghci> D.validateReadOptions D.defaultReadOptions{D.readColumns = Just []}
*** Exception: Invalid ReadOptions:
- readColumns is an empty selection; omit it to read every column
Writing
writeCsv :: FilePath -> DataFrame -> IO () Source #
Write a dataframe to a comma-separated file.
Example
ghci> D.writeCsv "./out.csv" df
writeTsv :: FilePath -> DataFrame -> IO () Source #
Write a dataframe to a tab-separated file.
Example
ghci> D.writeTsv "./out.tsv" df
Write a dataframe to a file using the given field separator.
Example
ghci> D.writeSeparated ';' "./out.txt" df
Helpers
stripQuotes :: Text -> Text Source #
Strip one layer of double-quote delimiters from a field, if present.
Example
>>>stripQuotes "\"hello\"""hello">>>stripQuotes "hello""hello"