deniskuzZ commented on code in PR #4131: URL: https://github.com/apache/hive/pull/4131#discussion_r1149033188
########## iceberg/iceberg-handler/src/main/java/org/apache/iceberg/mr/hive/HiveIcebergStorageHandler.java: ########## @@ -349,6 +365,96 @@ public Map<String, String> getBasicStatistics(Partish partish) { return stats; } + + @Override + public boolean canSetColStatistics() { + String statsSource = HiveConf.getVar(conf, HiveConf.ConfVars.HIVE_USE_STATS_FROM).toLowerCase(); + return statsSource.equals(PUFFIN); + } + + @Override + public boolean canProvideColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) { + + org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl; + TableDesc tableDesc = Utilities.getTableDesc(hmsTable); + Table table = Catalogs.loadTable(conf, tableDesc.getProperties()); + if (table.currentSnapshot() != null) { + String statsSource = HiveConf.getVar(conf, HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase(); + String statsPath = table.location() + STATS + table.name() + table.currentSnapshot().snapshotId(); + if (statsSource.equals(PUFFIN)) { + try (FileSystem fs = new Path(table.location()).getFileSystem(conf)) { + if (fs.exists(new Path(statsPath))) { + return true; + } + } catch (IOException e) { + LOG.warn(e.getMessage()); + } + } + } + return false; + } + + @Override + public List<ColumnStatisticsObj> getColStatistics(org.apache.hadoop.hive.ql.metadata.Table tbl) { + + org.apache.hadoop.hive.ql.metadata.Table hmsTable = tbl; + TableDesc tableDesc = Utilities.getTableDesc(hmsTable); + Table table = Catalogs.loadTable(conf, tableDesc.getProperties()); + String statsSource = HiveConf.getVar(conf, HiveConf.ConfVars.HIVE_COL_STATS_SOURCE).toLowerCase(); + switch (statsSource) { + case ICEBERG: + // Place holder for iceberg stats + break; + case PUFFIN: + String snapshotId = table.name() + table.currentSnapshot().snapshotId(); + String statsPath = table.location() + STATS + snapshotId; + LOG.info("Using stats from puffin file at:" + statsPath); + try (PuffinReader reader = Puffin.read(table.io().newInputFile(statsPath)).build()) { + BlobMetadata blobMetadata = reader.fileMetadata().blobs().get(0); + Map<BlobMetadata, List<ColumnStatistics>> collect = + Streams.stream(reader.readAll(ImmutableList.of(blobMetadata))).collect(Collectors.toMap(Pair::first, + blobMetadataByteBufferPair -> SerializationUtils.deserialize( + ByteBuffers.toByteArray(blobMetadataByteBufferPair.second())))); + + return collect.entrySet().stream().iterator().next().getValue().get(0).getStatsObj(); + } catch (IOException e) { + LOG.info(String.valueOf(e)); + } + break; + default: + // fall back to metastore + } + return null; + } + + + @Override + public boolean setColStatistics(org.apache.hadoop.hive.ql.metadata.Table table, + List<ColumnStatistics> colStats) { + TableDesc tableDesc = Utilities.getTableDesc(table); + Table tbl = Catalogs.loadTable(conf, tableDesc.getProperties()); + String snapshotId = tbl.name() + tbl.currentSnapshot().snapshotId(); + byte[] serializeColStats = SerializationUtils.serialize((Serializable) colStats); Review Comment: should we check if for NULL here? -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: gitbox-unsubscr...@hive.apache.org For queries about this service, please contact Infrastructure at: us...@infra.apache.org --------------------------------------------------------------------- To unsubscribe, e-mail: gitbox-unsubscr...@hive.apache.org For additional commands, e-mail: gitbox-h...@hive.apache.org