| Safe Haskell | None |
|---|---|
| Language | Haskell2010 |
DataFrame.Internal.Parsing.Fast
Description
Fast, non-backtracking field parsers for CSV ingest. Each is a pure
function over a (buf, start, end) byte slice, bit-exact with the reference
parsers in DataFrame.Internal.Parsing. Unboxed # variants avoid boxing.
Synopsis
- parseIntField :: ByteString -> Maybe Int
- parseIntFieldSlice :: ByteString -> Int -> Int -> Maybe Int
- parseIntField# :: ByteString -> Int -> Int -> (# Int#, Int# #)
- parseDoubleField :: ByteString -> Maybe Double
- parseDoubleFieldSlice :: ByteString -> Int -> Int -> Maybe Double
- parseDoubleField# :: ByteString -> Int -> Int -> (# Int#, Double# #)
- parseBoolField :: ByteString -> Maybe Bool
- parseBoolFieldSlice :: ByteString -> Int -> Int -> Maybe Bool
- parseBoolField# :: ByteString -> Int -> Int -> (# Int#, Int# #)
- parseDateField :: ByteString -> Maybe Day
- parseDateFieldSlice :: ByteString -> Int -> Int -> Maybe Day
- isMissingField :: ByteString -> Bool
- isMissingFieldSlice :: ByteString -> Int -> Int -> Bool
- isMissingFieldIn :: [ByteString] -> ByteString -> Bool
Int fields
parseIntField :: ByteString -> Maybe Int Source #
Strip-tolerant Int parse of a whole field; rejects overflow,
matching readByteStringInt exactly.
parseIntFieldSlice :: ByteString -> Int -> Int -> Maybe Int Source #
parseIntField# :: ByteString -> Int -> Int -> (# Int#, Int# #) Source #
Result is (# ok, value #) with ok 0 or 1. Caller guarantees
0 <= start <= end <= length buf.
Double fields
parseDoubleField :: ByteString -> Maybe Double Source #
Strip-tolerant Double parse of a whole field; bit-exact with
readByteStringDouble (falls back to it outside the fast window).
parseDoubleFieldSlice :: ByteString -> Int -> Int -> Maybe Double Source #
parseDoubleField# :: ByteString -> Int -> Int -> (# Int#, Double# #) Source #
Result is (# ok, value #) with ok 0 or 1. Caller guarantees
0 <= start <= end <= length buf.
Bool fields
parseBoolField :: ByteString -> Maybe Bool Source #
Exact-match Bool parse (True|true|TRUE|False|false|FALSE, no strip).
parseBoolFieldSlice :: ByteString -> Int -> Int -> Maybe Bool Source #
parseBoolField# :: ByteString -> Int -> Int -> (# Int#, Int# #) Source #
Exactly True|true|TRUE|False|false|FALSE, no strip (the
readByteStringBool grammar). Result is (# ok, bool #).
Date fields (default %Y-%m-%d format)
parseDateField :: ByteString -> Maybe Day Source #
%Y-%m-%d date parse: byte-level fast path for the padded
10-byte shape, parseTimeM fallback for everything else.
parseDateFieldSlice :: ByteString -> Int -> Int -> Maybe Day Source #
%Y-%m-%d: byte-level fast path for the padded 10-byte shape
(dddd-dd-dd); anything else (unpadded, whitespace-tolerant, long
years, invalid) falls back to readByteStringDate.
Missing-token test
isMissingField :: ByteString -> Bool Source #
Byte-level test against the canonical missing-token list
(Nothing NULL "" " " nan null N/A NaN NAN NA), no Text decode.
isMissingFieldSlice :: ByteString -> Int -> Int -> Bool Source #
Membership in the canonical missing list
["Nothing","NULL",""," ","nan","null","N/A","NaN","NAN","NA"]
(case-sensitive, exact), dispatched on length then first byte.
isMissingFieldIn :: [ByteString] -> ByteString -> Bool Source #
Generic fallback for user-supplied missing-indicator lists.