Compare commits

21 Commits
Author SHA1 Message Date
jlamothe af4066d3db updated package description/synopsis 2022-04-26 20:16:58 -04:00
jlamothe 3bc1be9eb9 moved decodeUTF8 into helpers section 2022-04-25 18:46:57 -04:00
jlamothe d99862b3f4 updated README.md 2022-04-25 18:45:37 -04:00
jlamothe 22b6942900 changed types of encodeCSV and encodeRawCSV
...to be more generic, the type now allows the to take any kind of input
2022-04-24 22:14:30 -04:00
jlamothe c1e9fb7b8e typo 2022-04-24 22:13:07 -04:00
jlamothe 82085eaaf9 note about pull requests 2022-04-24 22:11:28 -04:00
jlamothe 493b7dd9d4 version 0.1.0 2022-04-24 19:03:01 -04:00
jlamothe 723f046ea4 added homepage and bug reports information 2022-04-24 19:00:47 -04:00
jlamothe bb970e4f42 removed warning about API changes 2022-04-24 18:47:53 -04:00
jlamothe 1b20188dfc added executive summary to README.md 2022-04-24 18:47:28 -04:00
jlamothe 886f991ee3 implemented file producers/consumers
- readFromCSV
- readFromCSVRaw
- writeToCSV
- writeToCSVRaw
2022-04-24 18:32:30 -04:00
jlamothe 659b817252 removed writeCSVFromStream and writeRawCSVFromStream 2022-04-24 17:15:27 -04:00
jlamothe d27eb91952 implemented file writing 2022-04-24 16:47:24 -04:00
jlamothe 5c74085ada implemented encodeCSV and encodeRawCSV 2022-04-24 16:05:10 -04:00
jlamothe 3df133147f implemented encodeRows 2022-04-24 16:05:10 -04:00
jlamothe 5d6a7db6c5 implemented encodeRawRows 2022-04-24 14:23:49 -04:00
jlamothe 724dfe0345 implemented slurpRawLabelledCSV 2022-04-21 22:27:48 -04:00
jlamothe 51784123cd implemented slurpLabelledCSV 2022-04-21 22:27:34 -04:00
jlamothe 35130eeae1 documentation for slurpCSV and slurpRawCSV 2022-04-21 22:23:26 -04:00
jlamothe 2af6966192 implemented slurpRawCSV 2022-04-21 21:45:42 -04:00
jlamothe 8533e84caa implemented slurpCSV 2022-04-21 21:39:10 -04:00
6 changed files with 293 additions and 28 deletions
+2
View File
@@ -1,3 +1,5 @@
# Changelog for csv-sip # Changelog for csv-sip
## Unreleased changes ## Unreleased changes
- changed the types of encodeCSV and encodeRawCSV to make them more generic
- slight re-structuring of documentation
+5 -2
View File
@@ -14,5 +14,8 @@ General Public License for more details.
You should have received a copy of the GNU General Public License You should have received a copy of the GNU General Public License
along with this program. If not, see <https://www.gnu.org/licenses/>. along with this program. If not, see <https://www.gnu.org/licenses/>.
## Important Note ## Executive Summary
This library is not yet ready for release. As such, all code should be considered to be unstable and subject to change at any time. This library allows for reading and writing to and from CSV files in a streaming manner. Files can be read and written to on a row-by-row basis allowing larger files to be worked with, since the whole file doesn't have to be loaded to manipulate it. It is based on the [conduit](https://hackage.haskell.org/package/conduit] library.
## Pull Requests
Please make pull requests to the `dev` branch.
+5 -3
View File
@@ -5,10 +5,12 @@ cabal-version: 2.2
-- see: https://github.com/sol/hpack -- see: https://github.com/sol/hpack
name: csv-sip name: csv-sip
version: 0.0.0 version: 0.1.0
synopsis: extracts data from a CSV file synopsis: CSV streaming library
description: extracts data from a CSV file - see README.md for more details description: CSV streaming library - see README.md for more details
category: Data category: Data
homepage: https://codeberg.org/jlamothe/csv-sip
bug-reports: https://codeberg.org/jlamothe/csv-sip/issues
author: Jonathan Lamothe author: Jonathan Lamothe
maintainer: jonathan@jlamothe.net maintainer: jonathan@jlamothe.net
copyright: (C) 2022 Jonathan Lamothe copyright: (C) 2022 Jonathan Lamothe
+5 -3
View File
@@ -1,5 +1,5 @@
name: csv-sip name: csv-sip
version: 0.0.0 version: 0.1.0
license: GPL-3.0-or-later license: GPL-3.0-or-later
author: "Jonathan Lamothe" author: "Jonathan Lamothe"
maintainer: "jonathan@jlamothe.net" maintainer: "jonathan@jlamothe.net"
@@ -10,13 +10,15 @@ extra-source-files:
- ChangeLog.md - ChangeLog.md
# Metadata used when publishing your package # Metadata used when publishing your package
synopsis: extracts data from a CSV file synopsis: CSV streaming library
category: Data category: Data
# To avoid duplicated efforts in documentation and dealing with the # To avoid duplicated efforts in documentation and dealing with the
# complications of embedding Haddock markup inside cabal files, it is # complications of embedding Haddock markup inside cabal files, it is
# common to point users to the README.md file. # common to point users to the README.md file.
description: extracts data from a CSV file - see README.md for more details description: CSV streaming library - see README.md for more details
homepage: https://codeberg.org/jlamothe/csv-sip
bug-reports: https://codeberg.org/jlamothe/csv-sip/issues
ghc-options: ghc-options:
- -Wall - -Wall
+190 -11
View File
@@ -26,25 +26,179 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
{-# LANGUAGE LambdaCase, OverloadedStrings #-} {-# LANGUAGE LambdaCase, OverloadedStrings #-}
module Data.CSV.Sip ( module Data.CSV.Sip (
-- * Working with Files
-- ** Read an entire CSV file
slurpCSV,
slurpRawCSV,
slurpLabelledCSV,
slurpRawLabelledCSV,
-- ** Write an entire CSV file
writeCSV,
writeRawCSV,
-- * Conduits
-- ** Producers
readFromCSV,
readFromCSVRaw,
encodeCSV,
encodeRawCSV,
-- ** Consumers
writeToCSV,
writeToCSVRaw,
-- ** Transformers
-- *** Encoding
encodeRows,
encodeRawRows,
-- *** Decoding
labelFields, labelFields,
decodeRows, decodeRows,
decodeRawRows, decodeRawRows,
decodeUTF8,
toBytes, toBytes,
-- * Helper Functions
decodeUTF8,
) where ) where
import Conduit (ConduitT, await, mapC, yield, (.|)) import Conduit
( ConduitT
, MonadResource
, await
, mapC
, runConduit
, sinkFile
, sourceFile
, yield
, (.|)
)
import Control.Monad (unless) import Control.Monad (unless)
import Control.Monad.Trans.Class (lift) import Control.Monad.Trans.Class (lift)
import Control.Monad.Trans.State (StateT, evalStateT, get, gets, modify) import Control.Monad.Trans.State (StateT, evalStateT, get, gets, modify)
import qualified Data.ByteString as BS import qualified Data.ByteString as BS
import Data.Conduit.List (consume, sourceList)
import qualified Data.Map as M import qualified Data.Map as M
import Data.Maybe (fromMaybe) import Data.Maybe (fromMaybe)
import qualified Data.Text as T import qualified Data.Text as T
import Data.Text.Encoding (decodeUtf8') import Data.Text.Encoding (decodeUtf8', encodeUtf8)
import Data.Word (Word8) import Data.Word (Word8)
-- | read a CSV stream, using the first row as a header containing field labels -- | read an entire CSV file
slurpCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to read
-> m [[T.Text]]
slurpCSV file = runConduit $ sourceFile file .| decodeRows .| consume
-- | read an entire CSV file in raw mode
slurpRawCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to read
-> m [[BS.ByteString]]
slurpRawCSV file = runConduit $ sourceFile file .| decodeRawRows .| consume
-- | read an entire CSV file with a header
slurpLabelledCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to read
-> m [M.Map T.Text T.Text]
slurpLabelledCSV file = runConduit $
sourceFile file .| decodeRows .| labelFields .|consume
-- | read an entire CSV file with a header
slurpRawLabelledCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to read
-> m [M.Map BS.ByteString BS.ByteString]
slurpRawLabelledCSV file = runConduit $
sourceFile file .| decodeRawRows .| labelFields .|consume
-- | write a CSV file from Text-based rows
writeCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to write to
-> [[T.Text]]
-- ^ the fields/rows being written
-> m ()
writeCSV file csv = runConduit $ encodeCSV csv .| sinkFile file
-- | write a CSV file from raw ByteString-based rows
writeRawCSV
:: MonadResource m
=> FilePath
-- ^ the path to the file to write to
-> [[BS.ByteString]]
-- ^ the fields/rows being written
-> m ()
writeRawCSV file csv = runConduit $ encodeRawCSV csv .| sinkFile file
-- | reads a stream of Text-based rows from a CSV file
readFromCSV
:: MonadResource m
=> FilePath
-- ^ the path to the CSV file to read from
-> ConduitT i [T.Text] m ()
readFromCSV file = sourceFile file .| decodeRows
-- | reads a stream of ByteString-based rows from a CSV file
readFromCSVRaw
:: MonadResource m
=> FilePath
-- ^ the path to the CSV file to read from
-> ConduitT i [BS.ByteString] m ()
readFromCSVRaw file = sourceFile file .| decodeRawRows
-- | encode an entire CSV file
encodeCSV
:: Monad m
=> [[T.Text]]
-- ^ the data being encoded, organized into rows and fields
-> ConduitT o BS.ByteString m ()
encodeCSV csv = sourceList csv .| encodeRows
-- | encode an entire CSV file
encodeRawCSV
:: Monad m
=> [[BS.ByteString]]
-- ^ the data being encoded, organized into rows and fields
-> ConduitT o BS.ByteString m ()
encodeRawCSV csv = sourceList csv .| encodeRawRows
-- | Writes a stream of Text-based rows to a CSV file
writeToCSV
:: MonadResource m
=> FilePath
-- ^ the path to the CSV file to write to
-> ConduitT [T.Text] o m ()
writeToCSV file = encodeRows .| sinkFile file
-- | Writes a stream of ByteString-based rows to a CSV file
writeToCSVRaw
:: MonadResource m
=> FilePath
-- ^ the path to the CSV file to write to
-> ConduitT [BS.ByteString] o m ()
writeToCSVRaw file = encodeRawRows .| sinkFile file
-- | encode a CSV stream row by row, each element in the list read
-- represents a field, with the entire list representing a row
encodeRows :: Monad m => ConduitT [T.Text] BS.ByteString m ()
encodeRows = mapC (map encodeUtf8) .| encodeRawRows
-- | encode raw CSV stream row by row, each element in the list read
-- represents a field, with the entire list representing a row
encodeRawRows :: Monad m => ConduitT [BS.ByteString] BS.ByteString m ()
encodeRawRows = await >>= \case
Just fs-> do
encodeFields fs
encodeRawRows
Nothing -> return ()
-- | read a CSV stream, using the first row as a header containing
-- field labels
labelFields :: (Monad m, Ord a) => ConduitT [a] (M.Map a a) m () labelFields :: (Monad m, Ord a) => ConduitT [a] (M.Map a a) m ()
labelFields = await >>= \case labelFields = await >>= \case
Just headers -> labelLoop headers Just headers -> labelLoop headers
@@ -58,13 +212,7 @@ decodeRows = decodeRawRows .| mapC (map $ fromMaybe "" . decodeUTF8)
decodeRawRows :: Monad m => ConduitT BS.ByteString [BS.ByteString] m () decodeRawRows :: Monad m => ConduitT BS.ByteString [BS.ByteString] m ()
decodeRawRows = toBytes .| evalStateT decodeLoop newDecodeState decodeRawRows = toBytes .| evalStateT decodeLoop newDecodeState
-- | decode a raw ByteString into Text (if possible) -- | convert a stream to ByteStrings to a stream of bytes
decodeUTF8 :: BS.ByteString -> Maybe T.Text
decodeUTF8 bs = case decodeUtf8' bs of
Left _ -> Nothing
Right txt -> Just txt
-- | convert a stream to ByteStrings to a string of bytes
toBytes :: Monad m => ConduitT BS.ByteString Word8 m () toBytes :: Monad m => ConduitT BS.ByteString Word8 m ()
toBytes = await >>= \case toBytes = await >>= \case
Just bs -> do Just bs -> do
@@ -73,6 +221,12 @@ toBytes = await >>= \case
toBytes toBytes
Nothing -> return () Nothing -> return ()
-- | decode a raw ByteString into Text (if possible)
decodeUTF8 :: BS.ByteString -> Maybe T.Text
decodeUTF8 bs = case decodeUtf8' bs of
Left _ -> Nothing
Right txt -> Just txt
-- Internal -- Internal
data DecodeState = DecodeState data DecodeState = DecodeState
@@ -93,6 +247,16 @@ newDecodeState = DecodeState
-- Conduits -- Conduits
encodeFields
:: Monad m
=> [BS.ByteString]
-> ConduitT [BS.ByteString] BS.ByteString m ()
encodeFields [] = yield "\r\n"
encodeFields [f] = yield $ escapeField f `BS.append` "\r\n"
encodeFields (f:fs) = do
yield $ escapeField f `BS.append` ","
encodeFields fs
labelLoop :: (Monad m, Ord a) => [a] -> ConduitT [a] (M.Map a a) m () labelLoop :: (Monad m, Ord a) => [a] -> ConduitT [a] (M.Map a a) m ()
labelLoop headers = await >>= \case labelLoop headers = await >>= \case
Just values -> do Just values -> do
@@ -219,4 +383,19 @@ dropField s = s
setQuoted :: Modifier setQuoted :: Modifier
setQuoted s = s { isQuoted = True } setQuoted s = s { isQuoted = True }
-- Helpers
escapeField :: BS.ByteString -> BS.ByteString
escapeField field = let
bytes = BS.unpack field
in BS.concat
[ "\""
, BS.pack $ escapeLoop bytes
, "\""
]
escapeLoop :: [Word8] -> [Word8]
escapeLoop [] = []
escapeLoop (0x22:bs) = [0x22, 0x22] ++ escapeLoop bs -- escape quote
escapeLoop (b:bs) = b : escapeLoop bs
--jl --jl
+86 -9
View File
@@ -23,6 +23,7 @@ along with this program. If not, see <https://www.gnu.org/licenses/>.
module Data.CSV.SipSpec (spec) where module Data.CSV.SipSpec (spec) where
import Conduit (runConduit, (.|)) import Conduit (runConduit, (.|))
import qualified Data.ByteString as BS
import Data.Char (ord) import Data.Char (ord)
import Data.Conduit.List (consume, sourceList) import Data.Conduit.List (consume, sourceList)
import qualified Data.Map as M import qualified Data.Map as M
@@ -32,11 +33,87 @@ import Data.CSV.Sip
spec :: Spec spec :: Spec
spec = describe "Data.CSV.Sip" $ do spec = describe "Data.CSV.Sip" $ do
encodeCSVSpec
encodeRawCSVSpec
encodeRowsSpec
encodeRawRowsSpec
labelFieldsSpec labelFieldsSpec
decodeRowsSpec decodeRowsSpec
decodeRawRowsSpec decodeRawRowsSpec
decodeUTF8Spec
toBytesSpec toBytesSpec
decodeUTF8Spec
encodeCSVSpec :: Spec
encodeCSVSpec = describe "encodeCSV" $ do
result <- BS.concat <$> runConduit
(encodeCSV input .| consume)
it ("shouldBe " ++ show expected) $
result `shouldBe` expected
where
input =
[ [ "foo", "a\"b" ]
, [ "a\rb", "a\nb" ]
]
expected = BS.concat
[ "\"foo\",\"a\"\"b\"\r\n"
, "\"a\rb\",\"a\nb\"\r\n"
]
encodeRawCSVSpec :: Spec
encodeRawCSVSpec = describe "encodeRawCSV" $ do
result <- BS.concat <$> runConduit
(encodeRawCSV input .| consume)
it ("shouldBe " ++ show expected) $
result `shouldBe` expected
where
input =
[ [ "foo", "a\"b" ]
, [ "a\rb", "a\nb" ]
]
expected = BS.concat
[ "\"foo\",\"a\"\"b\"\r\n"
, "\"a\rb\",\"a\nb\"\r\n"
]
encodeRowsSpec :: Spec
encodeRowsSpec = describe "encodeRows" $ do
result <- BS.concat <$> runConduit
(sourceList input .| encodeRows .| consume)
it ("shouldBe " ++ show expected) $
result `shouldBe` expected
where
input =
[ [ "foo", "a\"b" ]
, [ "a\rb", "a\nb" ]
]
expected = BS.concat
[ "\"foo\",\"a\"\"b\"\r\n"
, "\"a\rb\",\"a\nb\"\r\n"
]
encodeRawRowsSpec :: Spec
encodeRawRowsSpec = describe "encodeRawRows" $ do
result <- BS.concat <$> runConduit
(sourceList input .| encodeRawRows .| consume)
it ("should be " ++ show expected) $
result `shouldBe` expected
where
input =
[ [ "foo", "a\"b" ]
, [ "a\rb", "a\nb" ]
]
expected = BS.concat
[ "\"foo\",\"a\"\"b\"\r\n"
, "\"a\rb\",\"a\nb\"\r\n"
]
labelFieldsSpec :: Spec labelFieldsSpec :: Spec
labelFieldsSpec = describe "labelFields" $ mapM_ labelFieldsSpec = describe "labelFields" $ mapM_
@@ -250,6 +327,14 @@ decodeRawRowsSpec = describe "decodeRawRows" $ mapM_
, ["baz", "quux"] , ["baz", "quux"]
] ]
toBytesSpec :: Spec
toBytesSpec = describe "toBytes" $ let
input = ["ab", "cd"]
expected = map (fromIntegral . ord) "abcd"
in it ("should be " ++ show expected) $ do
result <- runConduit $ sourceList input .| toBytes .| consume
result `shouldBe` expected
decodeUTF8Spec :: Spec decodeUTF8Spec :: Spec
decodeUTF8Spec = describe "decodeUTF8" $ mapM_ decodeUTF8Spec = describe "decodeUTF8" $ mapM_
( \(label, input, expected) -> context label $ ( \(label, input, expected) -> context label $
@@ -264,12 +349,4 @@ decodeUTF8Spec = describe "decodeUTF8" $ mapM_
, ( "blank", "", Just "" ) , ( "blank", "", Just "" )
] ]
toBytesSpec :: Spec
toBytesSpec = describe "toBytes" $ let
input = ["ab", "cd"]
expected = map (fromIntegral . ord) "abcd"
in it ("should be " ++ show expected) $ do
result <- runConduit $ sourceList input .| toBytes .| consume
result `shouldBe` expected
--jl --jl