[ 
https://issues.apache.org/jira/browse/HIVE-27158?focusedWorklogId=853110&page=com.atlassian.jira.plugin.system.issuetabpanels:worklog-tabpanel#worklog-853110
 ]

ASF GitHub Bot logged work on HIVE-27158:
-----------------------------------------

                Author: ASF GitHub Bot
            Created on: 27/Mar/23 09:13
            Start Date: 27/Mar/23 09:13
    Worklog Time Spent: 10m 
      Work Description: deniskuzZ commented on code in PR #4131:
URL: https://github.com/apache/hive/pull/4131#discussion_r1149022350


##########
iceberg/iceberg-handler/src/main/java/org/apache/iceberg/mr/hive/HiveIcebergStorageHandler.java:
##########
@@ -349,6 +365,96 @@ public Map<String, String> getBasicStatistics(Partish 
partish) {
     return stats;
   }
 
+
+  @Override
+  public boolean canSetColStatistics() {
+    String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_USE_STATS_FROM).toLowerCase();
+    return statsSource.equals(PUFFIN);
+  }
+
+  @Override
+  public boolean 
canProvideColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) {
+
+    org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl;
+    TableDesc tableDesc = Utilities.getTableDesc(hmsTable);
+    Table table = Catalogs.loadTable(conf, tableDesc.getProperties());
+    if (table.currentSnapshot() != null) {
+      String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase();
+      String statsPath = table.location() + STATS + table.name() + 
table.currentSnapshot().snapshotId();
+      if (statsSource.equals(PUFFIN)) {
+        try (FileSystem fs = new Path(table.location()).getFileSystem(conf)) {
+          if (fs.exists(new Path(statsPath))) {
+            return true;
+          }
+        } catch (IOException e) {
+          LOG.warn(e.getMessage());
+        }
+      }
+    }
+    return false;
+  }
+
+  @Override
+  public List<ColumnStatisticsObj> 
getColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) {
+
+    org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl;
+    TableDesc tableDesc = Utilities.getTableDesc(hmsTable);
+    Table table = Catalogs.loadTable(conf, tableDesc.getProperties());
+    String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase();
+    switch (statsSource) {
+      case ICEBERG:
+        // Place holder for iceberg stats
+        break;
+      case PUFFIN:
+        String snapshotId = table.name() + 
table.currentSnapshot().snapshotId();
+        String statsPath = table.location() + STATS + snapshotId;

Review Comment:
   could we extract path construction into a helper method?



##########
iceberg/iceberg-handler/src/main/java/org/apache/iceberg/mr/hive/HiveIcebergStorageHandler.java:
##########
@@ -349,6 +365,96 @@ public Map<String, String> getBasicStatistics(Partish 
partish) {
     return stats;
   }
 
+
+  @Override
+  public boolean canSetColStatistics() {
+    String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_USE_STATS_FROM).toLowerCase();
+    return statsSource.equals(PUFFIN);
+  }
+
+  @Override
+  public boolean 
canProvideColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) {
+
+    org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl;
+    TableDesc tableDesc = Utilities.getTableDesc(hmsTable);
+    Table table = Catalogs.loadTable(conf, tableDesc.getProperties());
+    if (table.currentSnapshot() != null) {
+      String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase();
+      String statsPath = table.location() + STATS + table.name() + 
table.currentSnapshot().snapshotId();
+      if (statsSource.equals(PUFFIN)) {
+        try (FileSystem fs = new Path(table.location()).getFileSystem(conf)) {
+          if (fs.exists(new Path(statsPath))) {
+            return true;
+          }
+        } catch (IOException e) {
+          LOG.warn(e.getMessage());
+        }
+      }
+    }
+    return false;
+  }
+
+  @Override
+  public List<ColumnStatisticsObj> 
getColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) {
+
+    org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl;
+    TableDesc tableDesc = Utilities.getTableDesc(hmsTable);
+    Table table = Catalogs.loadTable(conf, tableDesc.getProperties());
+    String statsSource = HiveConf.getVar(conf, 
HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase();
+    switch (statsSource) {
+      case ICEBERG:
+        // Place holder for iceberg stats
+        break;
+      case PUFFIN:
+        String snapshotId = table.name() + 
table.currentSnapshot().snapshotId();
+        String statsPath = table.location() + STATS + snapshotId;

Review Comment:
   could we extract path construction into a helper method and reuse?





Issue Time Tracking
-------------------

    Worklog Id:     (was: 853110)
    Time Spent: 3h 50m  (was: 3h 40m)

> Store hive columns stats in puffin files for iceberg tables
> -----------------------------------------------------------
>
>                 Key: HIVE-27158
>                 URL: https://issues.apache.org/jira/browse/HIVE-27158
>             Project: Hive
>          Issue Type: Improvement
>            Reporter: Simhadri Govindappa
>            Assignee: Simhadri Govindappa
>            Priority: Major
>              Labels: pull-request-available
>          Time Spent: 3h 50m
>  Remaining Estimate: 0h
>




--
This message was sent by Atlassian Jira
(v8.20.10#820010)

Reply via email to