2015-12-15 17:22:45 +05:30
|
|
|
module Ringo.Generator
|
|
|
|
( tableDefnSQL
|
2015-12-17 23:40:56 +05:30
|
|
|
, factTableDefnSQL
|
2015-12-19 11:55:08 +05:30
|
|
|
, dimensionTablePopulateSQL
|
2015-12-20 18:25:14 +05:30
|
|
|
, factTablePopulateSQL
|
2015-12-15 17:22:45 +05:30
|
|
|
) where
|
|
|
|
|
2015-12-28 19:28:35 +05:30
|
|
|
import qualified Data.Map as Map
|
2015-12-15 17:22:45 +05:30
|
|
|
import qualified Data.Text as Text
|
|
|
|
|
2015-12-18 02:37:17 +05:30
|
|
|
#if MIN_VERSION_base(4,8,0)
|
|
|
|
#else
|
|
|
|
import Control.Applicative ((<$>))
|
|
|
|
#endif
|
|
|
|
|
2015-12-16 03:03:47 +05:30
|
|
|
import Control.Monad.Reader (Reader, asks)
|
2015-12-28 18:09:02 +05:30
|
|
|
import Data.List (nub, find)
|
|
|
|
import Data.Maybe (fromJust, fromMaybe, mapMaybe)
|
2015-12-16 03:03:47 +05:30
|
|
|
import Data.Monoid ((<>))
|
|
|
|
import Data.Text (Text)
|
2015-12-15 17:22:45 +05:30
|
|
|
|
2015-12-15 18:22:51 +05:30
|
|
|
import Ringo.Extractor.Internal
|
2015-12-15 17:22:45 +05:30
|
|
|
import Ringo.Types
|
2015-12-16 16:57:10 +05:30
|
|
|
import Ringo.Utils
|
2015-12-15 17:22:45 +05:30
|
|
|
|
|
|
|
nullableDefnSQL :: Nullable -> Text
|
|
|
|
nullableDefnSQL Null = "NULL"
|
|
|
|
nullableDefnSQL NotNull = "NOT NULL"
|
|
|
|
|
|
|
|
columnDefnSQL :: Column -> Text
|
|
|
|
columnDefnSQL Column {..} =
|
|
|
|
columnName <> " " <> columnType <> " " <> nullableDefnSQL columnNullable
|
|
|
|
|
2015-12-28 18:09:02 +05:30
|
|
|
joinColumnNames :: [ColumnName] -> Text
|
|
|
|
joinColumnNames = Text.intercalate ",\n"
|
2015-12-19 11:55:08 +05:30
|
|
|
|
|
|
|
fullColName :: TableName -> ColumnName -> ColumnName
|
|
|
|
fullColName tName cName = tName <> "." <> cName
|
|
|
|
|
|
|
|
constraintDefnSQL :: Table -> TableConstraint -> [Text]
|
|
|
|
constraintDefnSQL Table {..} constraint =
|
|
|
|
let alterTableSQL = "ALTER TABLE ONLY " <> tableName <> " ADD "
|
|
|
|
in case constraint of
|
|
|
|
PrimaryKey cName -> [ alterTableSQL <> "PRIMARY KEY (" <> cName <> ")" ]
|
|
|
|
ForeignKey oTableName cNamePairs ->
|
2015-12-28 18:09:02 +05:30
|
|
|
[ alterTableSQL <> "FOREIGN KEY (" <> joinColumnNames (map fst cNamePairs) <> ") REFERENCES "
|
|
|
|
<> oTableName <> " (" <> joinColumnNames (map snd cNamePairs) <> ")" ]
|
|
|
|
UniqueKey cNames -> ["CREATE UNIQUE INDEX ON " <> tableName <> " (" <> joinColumnNames cNames <> ")"]
|
2015-12-15 17:22:45 +05:30
|
|
|
|
|
|
|
tableDefnSQL :: Table -> [Text]
|
2015-12-19 11:55:08 +05:30
|
|
|
tableDefnSQL table@Table {..} =
|
|
|
|
tableSQL : concatMap (constraintDefnSQL table) tableConstraints
|
2015-12-15 17:22:45 +05:30
|
|
|
where
|
|
|
|
tableSQL = "CREATE TABLE " <> tableName <> " (\n"
|
2015-12-28 18:09:02 +05:30
|
|
|
<> (joinColumnNames . map columnDefnSQL $ tableColumns)
|
2015-12-15 17:22:45 +05:30
|
|
|
<> "\n)"
|
2015-12-15 18:22:51 +05:30
|
|
|
|
2015-12-17 23:40:56 +05:30
|
|
|
factTableDefnSQL :: Fact -> Table -> Reader Env [Text]
|
|
|
|
factTableDefnSQL fact table = do
|
|
|
|
Settings {..} <- asks envSettings
|
|
|
|
allDims <- extractAllDimensionTables fact
|
|
|
|
|
2015-12-18 13:20:35 +05:30
|
|
|
let factCols = forMaybe (factColumns fact) $ \col -> case col of
|
2015-12-18 01:00:32 +05:30
|
|
|
DimTime cName -> Just $ timeUnitColumnName settingDimTableIdColumnName cName settingTimeUnit
|
2015-12-17 23:40:56 +05:30
|
|
|
NoDimId cName -> Just cName
|
|
|
|
_ -> Nothing
|
|
|
|
|
2015-12-18 13:20:35 +05:30
|
|
|
dimCols = [ factDimFKIdColumnName settingDimPrefix settingDimTableIdColumnName tableName
|
|
|
|
| (_, Table {..}) <- allDims ]
|
|
|
|
|
|
|
|
indexSQLs = [ "CREATE INDEX ON " <> tableName table <> " USING btree (" <> col <> ")"
|
|
|
|
| col <- factCols ++ dimCols ]
|
2015-12-17 23:40:56 +05:30
|
|
|
|
|
|
|
return $ tableDefnSQL table ++ indexSQLs
|
|
|
|
|
2015-12-16 16:57:10 +05:30
|
|
|
dimColumnMapping :: Text -> Fact -> TableName -> [(ColumnName, ColumnName)]
|
|
|
|
dimColumnMapping dimPrefix fact dimTableName =
|
2015-12-19 11:55:08 +05:30
|
|
|
[ (dimColumnName dName cName, cName)
|
|
|
|
| DimVal dName cName <- factColumns fact , dimPrefix <> dName == dimTableName]
|
|
|
|
|
2015-12-28 19:28:35 +05:30
|
|
|
coalesceColumn :: TypeDefaults -> TableName -> Column -> Text
|
|
|
|
coalesceColumn defaults tName Column{..} =
|
2015-12-24 17:42:47 +05:30
|
|
|
if columnNullable == Null
|
2015-12-28 18:09:02 +05:30
|
|
|
then "coalesce(" <> fqColName <> "," <> defVal columnType <> ")"
|
|
|
|
else fqColName
|
2015-12-24 17:42:47 +05:30
|
|
|
where
|
2015-12-28 18:09:02 +05:30
|
|
|
fqColName = fullColName tName columnName
|
|
|
|
|
2015-12-28 19:28:35 +05:30
|
|
|
defVal colType =
|
|
|
|
fromMaybe (error $ "Default value not known for column type: " ++ Text.unpack colType)
|
|
|
|
. fmap snd
|
|
|
|
. find (\(k, _) -> k `Text.isPrefixOf` colType)
|
|
|
|
. Map.toList
|
|
|
|
$ defaults
|
2015-12-24 17:42:47 +05:30
|
|
|
|
2015-12-19 11:55:08 +05:30
|
|
|
dimensionTablePopulateSQL :: TablePopulationMode -> Fact -> TableName -> Reader Env Text
|
|
|
|
dimensionTablePopulateSQL popMode fact dimTableName = do
|
|
|
|
dimPrefix <- settingDimPrefix <$> asks envSettings
|
2015-12-24 17:42:47 +05:30
|
|
|
tables <- asks envTables
|
2015-12-28 19:28:35 +05:30
|
|
|
defaults <- asks envTypeDefaults
|
2015-12-24 17:42:47 +05:30
|
|
|
let factTable = fromJust $ findTable (factTableName fact) tables
|
|
|
|
colMapping = dimColumnMapping dimPrefix fact dimTableName
|
|
|
|
baseSelectC = "SELECT DISTINCT\n"
|
2015-12-28 18:09:02 +05:30
|
|
|
<> joinColumnNames
|
2015-12-28 18:43:49 +05:30
|
|
|
(map (\(_, cName) ->
|
|
|
|
let col = fromJust . findColumn cName $ tableColumns factTable
|
2015-12-28 19:28:35 +05:30
|
|
|
in coalesceColumn defaults (factTableName fact) col <> " AS " <> cName)
|
2015-12-24 17:42:47 +05:30
|
|
|
colMapping)
|
|
|
|
<> "\n"
|
2015-12-19 11:55:08 +05:30
|
|
|
<> "FROM " <> factTableName fact
|
|
|
|
insertC selectC = "INSERT INTO " <> dimTableName
|
2015-12-28 18:09:02 +05:30
|
|
|
<> " (\n" <> joinColumnNames (map fst colMapping) <> "\n) "
|
2015-12-19 11:55:08 +05:30
|
|
|
<> "SELECT x.* FROM (\n" <> selectC <> ") x"
|
|
|
|
timeCol = head [ cName | DimTime cName <- factColumns fact ]
|
|
|
|
return $ case popMode of
|
|
|
|
FullPopulation -> insertC baseSelectC
|
|
|
|
IncrementalPopulation ->
|
|
|
|
insertC (baseSelectC <> "\nWHERE "
|
|
|
|
<> timeCol <> " > ? AND " <> timeCol <> " <= ?"
|
|
|
|
<> " AND (\n"
|
|
|
|
<> Text.intercalate "\nOR " [ c <> " IS NOT NULL" | (_, c) <- colMapping ]
|
|
|
|
<> "\n)")
|
|
|
|
<> "\nLEFT JOIN " <> dimTableName <> " ON\n"
|
|
|
|
<> Text.intercalate " \nAND "
|
|
|
|
[ fullColName dimTableName c1 <> " IS NOT DISTINCT FROM " <> fullColName "x" c2
|
|
|
|
| (c1, c2) <- colMapping ]
|
|
|
|
<> "\nWHERE " <> Text.intercalate " \nAND "
|
|
|
|
[ fullColName dimTableName c <> " IS NULL" | (c, _) <- colMapping ]
|
2015-12-16 16:57:10 +05:30
|
|
|
|
2015-12-22 19:46:37 +05:30
|
|
|
data FactTablePopulateSelectSQL = FactTablePopulateSelectSQL
|
|
|
|
{ ftpsSelectCols :: ![(Text, Text)]
|
|
|
|
, ftpsSelectTable :: !Text
|
|
|
|
, ftpsJoinClauses :: ![Text]
|
|
|
|
, ftpsWhereClauses :: ![Text]
|
|
|
|
, ftpsGroupByCols :: ![Text]
|
|
|
|
} deriving (Show, Eq)
|
|
|
|
|
|
|
|
factTablePopulateSQL :: TablePopulationMode -> Fact -> Reader Env [Text]
|
2015-12-20 18:25:14 +05:30
|
|
|
factTablePopulateSQL popMode fact = do
|
2015-12-22 19:46:37 +05:30
|
|
|
Settings {..} <- asks envSettings
|
|
|
|
allDims <- extractAllDimensionTables fact
|
|
|
|
tables <- asks envTables
|
2015-12-28 19:28:35 +05:30
|
|
|
defaults <- asks envTypeDefaults
|
2015-12-22 19:46:37 +05:30
|
|
|
let fTableName = factTableName fact
|
2015-12-28 18:09:02 +05:30
|
|
|
fTable = fromJust . findTable fTableName $ tables
|
2015-12-22 19:46:37 +05:30
|
|
|
dimIdColName = settingDimTableIdColumnName
|
2015-12-28 18:09:02 +05:30
|
|
|
tablePKColName = head [ cName | PrimaryKey cName <- tableConstraints fTable ]
|
2015-12-16 16:57:10 +05:30
|
|
|
|
2015-12-18 13:20:35 +05:30
|
|
|
timeUnitColumnInsertSQL cName =
|
2015-12-18 01:00:32 +05:30
|
|
|
let colName = timeUnitColumnName dimIdColName cName settingTimeUnit
|
2015-12-18 17:00:46 +05:30
|
|
|
in ( colName
|
2015-12-28 18:09:02 +05:30
|
|
|
, "extract(epoch from " <> fullColName fTableName cName <> ")::bigint/"
|
|
|
|
<> Text.pack (show $ timeUnitToSeconds settingTimeUnit)
|
2015-12-18 17:00:46 +05:30
|
|
|
, True
|
|
|
|
)
|
2015-12-16 16:57:10 +05:30
|
|
|
|
2015-12-18 13:20:35 +05:30
|
|
|
factColMap = concatFor (factColumns fact) $ \col -> case col of
|
2015-12-22 19:46:37 +05:30
|
|
|
DimTime cName -> [ timeUnitColumnInsertSQL cName ]
|
2015-12-28 18:09:02 +05:30
|
|
|
NoDimId cName ->
|
|
|
|
let sCol = fromJust . findColumn cName $ tableColumns fTable
|
2015-12-28 19:28:35 +05:30
|
|
|
in [ (cName, coalesceColumn defaults fTableName sCol, True) ]
|
2015-12-22 19:46:37 +05:30
|
|
|
FactCount scName cName ->
|
2015-12-21 22:19:54 +05:30
|
|
|
[ (cName, "count(" <> maybe "*" (fullColName fTableName) scName <> ")", False) ]
|
2015-12-22 19:46:37 +05:30
|
|
|
FactSum scName cName ->
|
2015-12-21 22:19:54 +05:30
|
|
|
[ (cName, "sum(" <> fullColName fTableName scName <> ")", False) ]
|
2015-12-22 19:46:37 +05:30
|
|
|
FactAverage scName cName ->
|
2015-12-18 17:00:46 +05:30
|
|
|
[ ( cName <> settingAvgCountColumSuffix
|
|
|
|
, "count(" <> fullColName fTableName scName <> ")"
|
|
|
|
, False
|
|
|
|
)
|
|
|
|
, ( cName <> settingAvgSumColumnSuffix
|
|
|
|
, "sum(" <> fullColName fTableName scName <> ")"
|
|
|
|
, False
|
|
|
|
)
|
|
|
|
]
|
2015-12-22 19:46:37 +05:30
|
|
|
FactCountDistinct _ cName -> [ (cName, "'{}'::json", False)]
|
2015-12-16 16:57:10 +05:30
|
|
|
_ -> []
|
|
|
|
|
2015-12-28 18:09:02 +05:30
|
|
|
dimColMap = for allDims $ \(dimFact, factTable@Table {tableName}) ->
|
2015-12-18 01:00:32 +05:30
|
|
|
let colName = factDimFKIdColumnName settingDimPrefix dimIdColName tableName
|
2015-12-28 18:09:02 +05:30
|
|
|
col = fromJust . findColumn colName $ tableColumns factSourceTable
|
2015-12-16 16:57:10 +05:30
|
|
|
factSourceTableName = factTableName dimFact
|
2015-12-28 18:09:02 +05:30
|
|
|
factSourceTable = fromJust . findTable factSourceTableName $ tables
|
|
|
|
insertSQL = if factTable `elem` tables -- existing dimension table
|
|
|
|
then (if columnNullable col == Null then coalesceFKId else id)
|
|
|
|
$ fullColName factSourceTableName colName
|
2015-12-18 13:20:35 +05:30
|
|
|
else let
|
|
|
|
dimLookupWhereClauses =
|
2015-12-28 19:28:35 +05:30
|
|
|
[ fullColName tableName c1 <> " = " <> coalesceColumn defaults factSourceTableName col2
|
2015-12-28 18:09:02 +05:30
|
|
|
| (c1, c2) <- dimColumnMapping settingDimPrefix dimFact tableName
|
|
|
|
, let col2 = fromJust . findColumn c2 $ tableColumns factSourceTable ]
|
2015-12-18 01:00:32 +05:30
|
|
|
in "SELECT " <> dimIdColName <> " FROM " <> tableName <> "\nWHERE "
|
2015-12-28 18:09:02 +05:30
|
|
|
<> Text.intercalate "\n AND " dimLookupWhereClauses
|
|
|
|
insertSQL' = if factSourceTableName == fTableName
|
|
|
|
then insertSQL
|
|
|
|
else coalesceFKId insertSQL
|
|
|
|
|
|
|
|
in (colName, insertSQL', True)
|
2015-12-16 16:57:10 +05:30
|
|
|
|
2015-12-22 19:46:37 +05:30
|
|
|
colMap = [ (cName, (sql, groupByColPrefix <> cName), addAs)
|
2015-12-18 17:00:46 +05:30
|
|
|
| (cName, sql, addAs) <- factColMap ++ dimColMap ]
|
2015-12-16 16:57:10 +05:30
|
|
|
|
|
|
|
joinClauses =
|
2015-12-28 18:09:02 +05:30
|
|
|
mapMaybe (\tName -> (\p -> "LEFT JOIN " <> tName <> "\nON "<> p) <$> joinClausePreds fTable tName)
|
2015-12-16 16:57:10 +05:30
|
|
|
. nub
|
2015-12-18 17:00:46 +05:30
|
|
|
. map (factTableName . fst)
|
2015-12-16 16:57:10 +05:30
|
|
|
$ allDims
|
|
|
|
|
2015-12-20 18:25:14 +05:30
|
|
|
timeCol = fullColName fTableName $ head [ cName | DimTime cName <- factColumns fact ]
|
|
|
|
|
2015-12-22 19:46:37 +05:30
|
|
|
extFactTableName =
|
|
|
|
extractedFactTableName settingFactPrefix settingFactInfix (factName fact) settingTimeUnit
|
|
|
|
|
|
|
|
insertIntoSelectSQL =
|
|
|
|
FactTablePopulateSelectSQL
|
|
|
|
{ ftpsSelectCols = map snd3 colMap
|
|
|
|
, ftpsSelectTable = fTableName
|
|
|
|
, ftpsJoinClauses = joinClauses
|
|
|
|
, ftpsWhereClauses = if popMode == IncrementalPopulation
|
|
|
|
then [timeCol <> " > ?", timeCol <> " <= ?"]
|
|
|
|
else []
|
|
|
|
, ftpsGroupByCols = map ((groupByColPrefix <>) . fst3) . filter thd3 $ colMap
|
|
|
|
}
|
|
|
|
|
|
|
|
insertIntoInsertSQL = "INSERT INTO " <> extFactTableName
|
|
|
|
<> " (\n" <> Text.intercalate ",\n " (map fst3 colMap) <> "\n)"
|
|
|
|
|
|
|
|
countDistinctCols = [ col | col@(FactCountDistinct _ _) <- factColumns fact]
|
|
|
|
|
|
|
|
updateSQLs =
|
|
|
|
let origGroupByCols = ftpsGroupByCols insertIntoSelectSQL
|
|
|
|
origSelectCols = ftpsSelectCols insertIntoSelectSQL
|
|
|
|
|
|
|
|
in for countDistinctCols $ \(FactCountDistinct scName cName) ->
|
|
|
|
let unqCol = fullColName fTableName (fromMaybe tablePKColName scName) <> "::text"
|
|
|
|
|
|
|
|
bucketSelectCols =
|
|
|
|
[ ( "hashtext(" <> unqCol <> ") & "
|
|
|
|
<> Text.pack (show $ bucketCount settingFactCountDistinctErrorRate - 1)
|
|
|
|
, cName <> "_bnum")
|
|
|
|
, ( "31 - floor(log(2, min(hashtext(" <> unqCol <> ") & ~(1 << 31))))::int"
|
|
|
|
, cName <> "_bhash"
|
|
|
|
)
|
|
|
|
]
|
|
|
|
|
|
|
|
selectSQL = toSelectSQL $
|
|
|
|
insertIntoSelectSQL
|
|
|
|
{ ftpsSelectCols = filter ((`elem` origGroupByCols) . snd) origSelectCols ++ bucketSelectCols
|
|
|
|
, ftpsGroupByCols = origGroupByCols ++ [cName <> "_bnum"]
|
|
|
|
, ftpsWhereClauses = ftpsWhereClauses insertIntoSelectSQL ++ [ unqCol <> " IS NOT NULL" ]
|
|
|
|
}
|
|
|
|
|
|
|
|
aggSelectClause =
|
|
|
|
"json_object_agg(" <> cName <> "_bnum, " <> cName <> "_bhash) AS " <> cName
|
|
|
|
|
|
|
|
in "UPDATE " <> extFactTableName
|
|
|
|
<> "\nSET " <> cName <> " = " <> fullColName "xyz" cName
|
|
|
|
<> "\nFROM ("
|
2015-12-28 18:09:02 +05:30
|
|
|
<> "\nSELECT " <> joinColumnNames (origGroupByCols ++ [aggSelectClause])
|
2015-12-22 19:46:37 +05:30
|
|
|
<> "\nFROM (\n" <> selectSQL <> "\n) zyx"
|
2015-12-28 18:09:02 +05:30
|
|
|
<> "\nGROUP BY \n" <> joinColumnNames origGroupByCols
|
2015-12-22 19:46:37 +05:30
|
|
|
<> "\n) xyz"
|
|
|
|
<> "\n WHERE\n"
|
|
|
|
<> Text.intercalate "\nAND "
|
2015-12-28 18:09:02 +05:30
|
|
|
[ fullColName extFactTableName .fromJust . Text.stripPrefix groupByColPrefix $ col
|
|
|
|
<> " = " <> fullColName "xyz" col
|
2015-12-22 19:46:37 +05:30
|
|
|
| col <- origGroupByCols ]
|
|
|
|
|
|
|
|
return $ insertIntoInsertSQL <> "\n" <> toSelectSQL insertIntoSelectSQL :
|
|
|
|
if null countDistinctCols then [] else updateSQLs
|
2015-12-16 16:57:10 +05:30
|
|
|
where
|
2015-12-22 19:46:37 +05:30
|
|
|
groupByColPrefix = "xxff_"
|
2015-12-16 16:57:10 +05:30
|
|
|
|
|
|
|
joinClausePreds table oTableName =
|
|
|
|
fmap (\(ForeignKey _ colPairs) ->
|
2015-12-19 11:55:08 +05:30
|
|
|
Text.intercalate " AND "
|
2015-12-16 16:57:10 +05:30
|
|
|
. map (\(c1, c2) -> fullColName (tableName table) c1 <> " = " <> fullColName oTableName c2)
|
|
|
|
$ colPairs )
|
|
|
|
. find (\cons -> case cons of
|
|
|
|
ForeignKey tName _ -> tName == oTableName
|
|
|
|
_ -> False)
|
|
|
|
. tableConstraints
|
|
|
|
$ table
|
2015-12-22 19:46:37 +05:30
|
|
|
|
|
|
|
toSelectSQL FactTablePopulateSelectSQL {..} =
|
2015-12-28 18:09:02 +05:30
|
|
|
"SELECT \n" <> joinColumnNames (map (uncurry asName) ftpsSelectCols)
|
2015-12-22 19:46:37 +05:30
|
|
|
<> "\nFROM " <> ftpsSelectTable
|
|
|
|
<> (if not . null $ ftpsJoinClauses
|
2015-12-28 18:09:02 +05:30
|
|
|
then "\n" <> Text.intercalate "\n" ftpsJoinClauses
|
2015-12-22 19:46:37 +05:30
|
|
|
else "")
|
|
|
|
<> (if not . null $ ftpsWhereClauses
|
|
|
|
then "\nWHERE " <> Text.intercalate "\nAND " ftpsWhereClauses
|
|
|
|
else "")
|
|
|
|
<> "\nGROUP BY \n"
|
2015-12-28 18:09:02 +05:30
|
|
|
<> joinColumnNames ftpsGroupByCols
|
2015-12-22 19:46:37 +05:30
|
|
|
where
|
|
|
|
asName sql alias = "(" <> sql <> ")" <> " as " <> alias
|
|
|
|
|
2015-12-28 18:09:02 +05:30
|
|
|
coalesceFKId col =
|
|
|
|
if "coalesce" `Text.isPrefixOf` col
|
|
|
|
then col
|
|
|
|
else "coalesce((" <> col <> "), -1)"
|
2015-12-22 19:46:37 +05:30
|
|
|
|
|
|
|
bucketCount :: Double -> Integer
|
|
|
|
bucketCount errorRate =
|
|
|
|
let power :: Double = fromIntegral (ceiling . logBase 2 $ (1.04 / errorRate) ** 2 :: Integer)
|
|
|
|
in ceiling $ 2 ** power
|
|
|
|
|