diff --git a/assemblies/debug/pom.xml b/assemblies/debug/pom.xml index 681a4451f3e..10fd5c16bcf 100644 --- a/assemblies/debug/pom.xml +++ b/assemblies/debug/pom.xml @@ -409,6 +409,12 @@ ${project.version} provided + + org.apache.hop + hop-tech-lakehouse + ${project.version} + provided + org.apache.hop hop-tech-parquet diff --git a/assemblies/plugins/pom.xml b/assemblies/plugins/pom.xml index 8f3d5668a75..73a6951f74c 100644 --- a/assemblies/plugins/pom.xml +++ b/assemblies/plugins/pom.xml @@ -1767,6 +1767,12 @@ ${project.version} zip + + org.apache.hop + hop-tech-lakehouse + ${project.version} + zip + org.apache.hop hop-tech-minio diff --git a/plugins/engines/spark/pom.xml b/plugins/engines/spark/pom.xml index 8de3dc3f446..e74e0a1bcd7 100644 --- a/plugins/engines/spark/pom.xml +++ b/plugins/engines/spark/pom.xml @@ -120,6 +120,14 @@ provided + + + org.apache.hop + hop-tech-lakehouse + ${project.version} + provided + + org.apache.hop diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/engines/SparkPipelineEngine.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/engines/SparkPipelineEngine.java index 47b23ed6ed2..d450fd7863a 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/engines/SparkPipelineEngine.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/engines/SparkPipelineEngine.java @@ -1118,7 +1118,7 @@ private SparkSession buildSparkSession() throws HopException { } } - // Lakehouse: Delta/Iceberg extensions, hop_iceberg PATH catalog, SparkCatalog metadata. + // Lakehouse: Delta/Iceberg extensions, hop_iceberg PATH catalog, LakeCatalog metadata. // Applied after run-config sparkConfigs so lake defaults fill gaps; explicit run-config // keys already set above win if the user overrode them (except catalog apply overwrites). if (!lakePlan.isEmpty()) { diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableInputHandler.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableInputHandler.java index 21212dd224d..a4829bcece2 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableInputHandler.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableInputHandler.java @@ -24,13 +24,13 @@ import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; import org.apache.hop.spark.core.SparkNativeMetrics; import org.apache.hop.spark.engines.ISparkPipelineEngineRunConfiguration; import org.apache.hop.spark.table.SparkLakeTableSupport; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; import org.apache.spark.sql.SparkSession; @@ -60,7 +60,7 @@ public void handleTransform( Dataset input) throws HopException { - SparkLakeTableInputMeta meta = new SparkLakeTableInputMeta(); + LakeTableInputMeta meta = new LakeTableInputMeta(); loadTransformMetadata(meta, transformMeta, metadataProvider, pipelineMeta); String pathSchemeMap = runConfiguration != null ? runConfiguration.getPathSchemeMap() : null; @@ -71,8 +71,7 @@ public void handleTransform( transformDatasetMap.put(transformMeta.getName(), dataset); String target = - SparkLakeTableInputMeta.MODE_TABLE.equalsIgnoreCase( - String.valueOf(meta.getIdentifierMode())) + LakeTableInputMeta.MODE_TABLE.equalsIgnoreCase(String.valueOf(meta.getIdentifierMode())) ? "table=" + variables.resolve(meta.getTableIdentifier()) : "path=" + variables.resolve(meta.getTablePath()); log.logBasic( diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMaintenanceHandler.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMaintenanceHandler.java index a64f736fa4b..3afb508ad89 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMaintenanceHandler.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMaintenanceHandler.java @@ -24,6 +24,7 @@ import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -32,7 +33,6 @@ import org.apache.hop.spark.table.SparkLakeTableSupport; import org.apache.hop.spark.table.SparkLakeTableSupport.MaintenanceTarget; import org.apache.hop.spark.table.SparkMaintenanceSqlBuilder; -import org.apache.hop.spark.transforms.table.SparkLakeTableMaintenanceMeta; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; import org.apache.spark.sql.SparkSession; @@ -66,7 +66,7 @@ public void handleTransform( throws HopException { // Zero-input: ignore upstream if hop-connected (KD-20) - SparkLakeTableMaintenanceMeta meta = new SparkLakeTableMaintenanceMeta(); + LakeTableMaintenanceMeta meta = new LakeTableMaintenanceMeta(); loadTransformMetadata(meta, transformMeta, metadataProvider, pipelineMeta); String operation = diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMergeHandler.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMergeHandler.java index 5ef6a7fc7a1..c30fe7c2185 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMergeHandler.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableMergeHandler.java @@ -24,6 +24,7 @@ import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.transforms.LakeTableMergeMeta; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -31,7 +32,6 @@ import org.apache.hop.spark.table.SparkLakeActionSupport; import org.apache.hop.spark.table.SparkLakeTableSupport; import org.apache.hop.spark.table.SparkMergeSqlBuilder; -import org.apache.hop.spark.transforms.table.SparkLakeTableMergeMeta; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; import org.apache.spark.sql.SparkSession; @@ -71,7 +71,7 @@ public void handleTransform( + "' requires exactly one upstream Dataset (source rows for USING)."); } - SparkLakeTableMergeMeta meta = new SparkLakeTableMergeMeta(); + LakeTableMergeMeta meta = new LakeTableMergeMeta(); loadTransformMetadata(meta, transformMeta, metadataProvider, pipelineMeta); String pathSchemeMap = runConfiguration != null ? runConfiguration.getPathSchemeMap() : null; diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableOutputHandler.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableOutputHandler.java index 94244fb350b..92f19381c72 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableOutputHandler.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/pipeline/handler/SparkLakeTableOutputHandler.java @@ -23,6 +23,8 @@ import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -30,8 +32,6 @@ import org.apache.hop.spark.engines.ISparkPipelineEngineRunConfiguration; import org.apache.hop.spark.table.SparkLakeActionSupport; import org.apache.hop.spark.table.SparkLakeTableSupport; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; import org.apache.spark.sql.SparkSession; @@ -72,7 +72,7 @@ public void handleTransform( + " (disabled hops and disconnected transforms are not executed on native Spark)."); } - SparkLakeTableOutputMeta meta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta meta = new LakeTableOutputMeta(); loadTransformMetadata(meta, transformMeta, metadataProvider, pipelineMeta); Dataset toWrite = trackMetrics(input, transformMeta, SparkNativeMetrics.Role.OUTPUT); @@ -82,8 +82,7 @@ public void handleTransform( SparkLakeActionSupport.putEmptyLeaf(transformDatasetMap, transformMeta.getName(), spark); String target = - SparkLakeTableInputMeta.MODE_TABLE.equalsIgnoreCase( - String.valueOf(meta.getIdentifierMode())) + LakeTableInputMeta.MODE_TABLE.equalsIgnoreCase(String.valueOf(meta.getIdentifierMode())) ? "table=" + variables.resolve(meta.getTableIdentifier()) : "path=" + variables.resolve(meta.getTablePath()); log.logBasic( diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/LakeSessionPlan.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/LakeSessionPlan.java index 712ac39de13..145864d7ff1 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/LakeSessionPlan.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/LakeSessionPlan.java @@ -29,23 +29,24 @@ import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.variables.IVariables; import org.apache.hop.core.xml.XmlHandler; +import org.apache.hop.lakehouse.metadata.LakeCatalog; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta; +import org.apache.hop.lakehouse.transforms.LakeTableMergeMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.api.IHopMetadataProvider; +import org.apache.hop.metadata.serializer.xml.XmlMetadataUtil; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.ITransformMeta; import org.apache.hop.pipeline.transform.TransformMeta; -import org.apache.hop.spark.metadata.SparkCatalog; import org.apache.hop.spark.pipeline.HopPipelineMetaToSparkConverter; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableMaintenanceMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableMergeMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.SparkSession; import org.w3c.dom.Node; /** - * Pre-session scan of active lake transforms: formats, referenced {@link SparkCatalog} metadata, - * and session conf for Delta / Iceberg PATH + TABLE modes. + * Pre-session scan of active lake transforms: formats, referenced {@link LakeCatalog} metadata, and + * session conf for Delta / Iceberg PATH + TABLE modes. */ public final class LakeSessionPlan { @@ -57,7 +58,7 @@ public final class LakeSessionPlan { SparkConst.SPARK_LAKE_TABLE_MAINTENANCE_PLUGIN_ID); private final Set formatsNeeded = new LinkedHashSet<>(); - private final Map catalogsByMetaName = new LinkedHashMap<>(); + private final Map catalogsByMetaName = new LinkedHashMap<>(); private boolean needsTableMode; private LakeSessionPlan() {} @@ -66,7 +67,7 @@ public Set getFormatsNeeded() { return formatsNeeded; } - public Map getCatalogsByMetaName() { + public Map getCatalogsByMetaName() { return catalogsByMetaName; } @@ -86,7 +87,7 @@ public boolean needsTableMode() { return needsTableMode; } - /** Scan active transforms for lake plugin ids, formats, and SparkCatalog metadata references. */ + /** Scan active transforms for lake plugin ids, formats, and LakeCatalog metadata references. */ public static LakeSessionPlan from( PipelineMeta pipelineMeta, IHopMetadataProvider metadataProvider) throws HopException { return from(pipelineMeta, metadataProvider, null); @@ -107,7 +108,7 @@ public static LakeSessionPlan from( continue; } if (SparkConst.SPARK_LAKE_TABLE_INPUT_PLUGIN_ID.equals(id)) { - SparkLakeTableInputMeta meta = new SparkLakeTableInputMeta(); + LakeTableInputMeta meta = new LakeTableInputMeta(); loadMeta(meta, tm, metadataProvider); SparkLakeTableSupport.collectFormat(meta.getFormat(), plan.formatsNeeded); collectCatalogRef( @@ -117,7 +118,7 @@ public static LakeSessionPlan from( meta.getIdentifierMode(), meta.getCatalogMetadataName()); } else if (SparkConst.SPARK_LAKE_TABLE_OUTPUT_PLUGIN_ID.equals(id)) { - SparkLakeTableOutputMeta meta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta meta = new LakeTableOutputMeta(); loadMeta(meta, tm, metadataProvider); SparkLakeTableSupport.collectFormat(meta.getFormat(), plan.formatsNeeded); collectCatalogRef( @@ -127,7 +128,7 @@ public static LakeSessionPlan from( meta.getIdentifierMode(), meta.getCatalogMetadataName()); } else if (SparkConst.SPARK_LAKE_TABLE_MERGE_PLUGIN_ID.equals(id)) { - SparkLakeTableMergeMeta meta = new SparkLakeTableMergeMeta(); + LakeTableMergeMeta meta = new LakeTableMergeMeta(); loadMeta(meta, tm, metadataProvider); SparkLakeTableSupport.collectFormat(meta.getFormat(), plan.formatsNeeded); collectCatalogRef( @@ -137,7 +138,7 @@ public static LakeSessionPlan from( meta.getIdentifierMode(), meta.getCatalogMetadataName()); } else if (SparkConst.SPARK_LAKE_TABLE_MAINTENANCE_PLUGIN_ID.equals(id)) { - SparkLakeTableMaintenanceMeta meta = new SparkLakeTableMaintenanceMeta(); + LakeTableMaintenanceMeta meta = new LakeTableMaintenanceMeta(); loadMeta(meta, tm, metadataProvider); SparkLakeTableSupport.collectFormat(meta.getFormat(), plan.formatsNeeded); collectCatalogRef( @@ -162,9 +163,9 @@ private static void collectCatalogRef( try { mode = SparkLakeTableSupport.normalizeIdentifierMode(identifierMode); } catch (HopException e) { - mode = SparkLakeTableInputMeta.MODE_PATH; + mode = LakeTableInputMeta.MODE_PATH; } - if (SparkLakeTableInputMeta.MODE_TABLE.equals(mode)) { + if (LakeTableInputMeta.MODE_TABLE.equals(mode)) { plan.needsTableMode = true; } if (StringUtils.isEmpty(catalogMetadataName) || metadataProvider == null) { @@ -176,7 +177,7 @@ private static void collectCatalogRef( return; } try { - SparkCatalog catalog = metadataProvider.getSerializer(SparkCatalog.class).load(metaName); + LakeCatalog catalog = metadataProvider.getSerializer(LakeCatalog.class).load(metaName); if (catalog == null) { throw new HopException( "SparkCatalog metadata '" @@ -200,7 +201,7 @@ public void verifyClasspath(ClassLoader classLoader) throws HopException { } /** - * Apply Delta / Iceberg session conf and referenced SparkCatalog entries on a new hop-run session + * Apply Delta / Iceberg session conf and referenced LakeCatalog entries on a new hop-run session * builder. */ public void applyToBuilder(SparkSession.Builder builder) throws HopException { @@ -230,14 +231,14 @@ public void applyToBuilder(SparkSession.Builder builder, IVariables variables) if (!extensions.isEmpty()) { builder.config(SparkLakeFormats.SPARK_CONF_EXTENSIONS, String.join(",", extensions)); } - for (SparkCatalog catalog : catalogsByMetaName.values()) { + for (LakeCatalog catalog : catalogsByMetaName.values()) { SparkCatalogApplier.applyToBuilder(builder, catalog, variables); } } /** * For an active/reused session (spark-submit): require connector classes and Delta conf; register - * missing Iceberg path catalog and user SparkCatalog entries when possible. + * missing Iceberg path catalog and user LakeCatalog entries when possible. */ public void verifyActiveSession(SparkSession session, ILogChannel log) throws HopException { verifyActiveSession(session, log, null); @@ -313,9 +314,9 @@ public void verifyActiveSession(SparkSession session, ILogChannel log, IVariable } } - // Apply / verify user SparkCatalog metadata (TABLE mode) - for (Map.Entry e : catalogsByMetaName.entrySet()) { - SparkCatalog cat = e.getValue(); + // Apply / verify user LakeCatalog metadata (TABLE mode) + for (Map.Entry e : catalogsByMetaName.entrySet()) { + LakeCatalog cat = e.getValue(); String sparkName = StringUtils.defaultIfBlank(cat.getCatalogName(), cat.getName()); if (variables != null) { sparkName = variables.resolve(sparkName); @@ -363,65 +364,32 @@ private static void loadMeta( Node node = XmlHandler.getSubNode(XmlHandler.loadXmlString(xml), TransformMeta.XML_TAG); meta.loadXml(node, metadataProvider); } catch (Exception e) { - ITransformMeta live = transformMeta.getTransform(); - if (meta instanceof SparkLakeTableInputMeta target - && live instanceof SparkLakeTableInputMeta in) { - target.setFormat(in.getFormat()); - target.setIdentifierMode(in.getIdentifierMode()); - target.setTablePath(in.getTablePath()); - target.setTableIdentifier(in.getTableIdentifier()); - target.setCatalogMetadataName(in.getCatalogMetadataName()); - target.setTimeTravelType(in.getTimeTravelType()); - target.setTimeTravelVersion(in.getTimeTravelVersion()); - target.setTimeTravelTimestamp(in.getTimeTravelTimestamp()); - target.setExtraOptions(in.getExtraOptions()); - target.setFields(in.getFields()); - return; - } - if (meta instanceof SparkLakeTableOutputMeta target - && live instanceof SparkLakeTableOutputMeta out) { - target.setFormat(out.getFormat()); - target.setIdentifierMode(out.getIdentifierMode()); - target.setTablePath(out.getTablePath()); - target.setTableIdentifier(out.getTableIdentifier()); - target.setCatalogMetadataName(out.getCatalogMetadataName()); - target.setSaveMode(out.getSaveMode()); - target.setExtraOptions(out.getExtraOptions()); - target.setPartitionByColumns(out.getPartitionByColumns()); - target.setCoalescePartitions(out.getCoalescePartitions()); - return; - } - if (meta instanceof SparkLakeTableMergeMeta target - && live instanceof SparkLakeTableMergeMeta merge) { - target.setFormat(merge.getFormat()); - target.setIdentifierMode(merge.getIdentifierMode()); - target.setTablePath(merge.getTablePath()); - target.setTableIdentifier(merge.getTableIdentifier()); - target.setCatalogMetadataName(merge.getCatalogMetadataName()); - target.setMergeCondition(merge.getMergeCondition()); - target.setMatchedAction(merge.getMatchedAction()); - target.setNotMatchedAction(merge.getNotMatchedAction()); - target.setNotMatchedBySourceAction(merge.getNotMatchedBySourceAction()); - target.setRawMergeSql(merge.getRawMergeSql()); - return; - } - if (meta instanceof SparkLakeTableMaintenanceMeta target - && live instanceof SparkLakeTableMaintenanceMeta maint) { - target.setFormat(maint.getFormat()); - target.setIdentifierMode(maint.getIdentifierMode()); - target.setTablePath(maint.getTablePath()); - target.setTableIdentifier(maint.getTableIdentifier()); - target.setCatalogMetadataName(maint.getCatalogMetadataName()); - target.setOperation(maint.getOperation()); - target.setRetentionHours(maint.getRetentionHours()); - target.setRetainLast(maint.getRetainLast()); - target.setWhereClause(maint.getWhereClause()); - target.setZOrderColumns(maint.getZOrderColumns()); - target.setAcknowledgeDestructive(maint.isAcknowledgeDestructive()); - return; + try { + copyFromLive(meta, transformMeta.getTransform(), metadataProvider); + } catch (Exception copyFailure) { + e.addSuppressed(copyFailure); + throw new HopException( + "Unable to load lake transform metadata for '" + transformMeta.getName() + "'", e); } + } + } + + /** + * Copies the settings of a transform's in-memory meta into {@code meta}. The lake transforms + * belong to the lakehouse plugin, so the in-memory meta can come from another class loader than + * {@code meta}: it is copied through its serialized form instead of being cast. + */ + static void copyFromLive( + ITransformMeta meta, ITransformMeta live, IHopMetadataProvider metadataProvider) + throws HopException { + if (live == null || !live.getClass().getName().equals(meta.getClass().getName())) { throw new HopException( - "Unable to load lake transform metadata for '" + transformMeta.getName() + "'", e); + "No " + meta.getClass().getSimpleName() + " settings to copy from the transform"); } + String xml = XmlMetadataUtil.serializeObjectToXml(live); + Node node = + XmlHandler.getSubNode( + XmlHandler.loadXmlString("" + xml + ""), "transform"); + XmlMetadataUtil.deSerializeFromXml(node, meta.getClass(), meta, metadataProvider); } } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkCatalogApplier.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkCatalogApplier.java index 6d7870acba2..acb47106d2e 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkCatalogApplier.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkCatalogApplier.java @@ -23,11 +23,11 @@ import org.apache.commons.lang3.StringUtils; import org.apache.hop.core.exception.HopException; import org.apache.hop.core.variables.IVariables; -import org.apache.hop.spark.metadata.SparkCatalog; +import org.apache.hop.lakehouse.metadata.LakeCatalog; import org.apache.spark.sql.SparkSession; /** - * Expands {@link SparkCatalog} metadata into {@code spark.sql.catalog..*} configuration for + * Expands {@link LakeCatalog} metadata into {@code spark.sql.catalog..*} configuration for * hop-run builders and active sessions. */ public final class SparkCatalogApplier { @@ -38,7 +38,7 @@ private SparkCatalogApplier() {} * Build Spark conf key/value pairs for a catalog definition. Does not log credential or confExtra * values. */ - public static Map toSparkConfigs(SparkCatalog catalog, IVariables variables) + public static Map toSparkConfigs(LakeCatalog catalog, IVariables variables) throws HopException { if (catalog == null) { throw new HopException("SparkCatalog is null"); @@ -62,7 +62,7 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable } String type = - StringUtils.defaultIfBlank(catalog.getCatalogType(), SparkCatalog.TYPE_HADOOP) + StringUtils.defaultIfBlank(catalog.getCatalogType(), LakeCatalog.TYPE_HADOOP) .trim() .toLowerCase(Locale.ROOT); @@ -75,7 +75,7 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable } switch (type) { - case SparkCatalog.TYPE_HADOOP -> { + case LakeCatalog.TYPE_HADOOP -> { conf.put(prefix, impl); conf.put(prefix + ".type", "hadoop"); String warehouse = resolve(variables, catalog.getWarehouse()); @@ -85,7 +85,7 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable } conf.put(prefix + ".warehouse", toUriIfLocalPath(warehouse)); } - case SparkCatalog.TYPE_REST -> { + case LakeCatalog.TYPE_REST -> { conf.put(prefix, impl); conf.put(prefix + ".type", "rest"); String uri = resolve(variables, catalog.getUri()); @@ -99,10 +99,10 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable conf.put(prefix + ".warehouse", toUriIfLocalPath(warehouse)); } } - case SparkCatalog.TYPE_CUSTOM -> { + case LakeCatalog.TYPE_CUSTOM -> { conf.put(prefix, impl); } - case SparkCatalog.TYPE_HIVE, SparkCatalog.TYPE_GLUE -> { + case LakeCatalog.TYPE_HIVE, LakeCatalog.TYPE_GLUE -> { // Advanced: operator supplies full conf via confExtra / implementation if (StringUtils.isNotEmpty(impl)) { conf.put(prefix, impl); @@ -121,7 +121,7 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable // Optional credential — only if operator maps it via confExtra typically; expose as token // property for REST when set String credential = resolve(variables, catalog.getCredential()); - if (StringUtils.isNotEmpty(credential) && SparkCatalog.TYPE_REST.equals(type)) { + if (StringUtils.isNotEmpty(credential) && LakeCatalog.TYPE_REST.equals(type)) { conf.putIfAbsent(prefix + ".token", credential); } @@ -155,15 +155,14 @@ public static Map toSparkConfigs(SparkCatalog catalog, IVariable } public static void applyToBuilder( - SparkSession.Builder builder, SparkCatalog catalog, IVariables variables) - throws HopException { + SparkSession.Builder builder, LakeCatalog catalog, IVariables variables) throws HopException { for (Map.Entry e : toSparkConfigs(catalog, variables).entrySet()) { builder.config(e.getKey(), e.getValue()); } } - public static void applyToSession( - SparkSession session, SparkCatalog catalog, IVariables variables) throws HopException { + public static void applyToSession(SparkSession session, LakeCatalog catalog, IVariables variables) + throws HopException { for (Map.Entry e : toSparkConfigs(catalog, variables).entrySet()) { session.conf().set(e.getKey(), e.getValue()); } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeFormats.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeFormats.java index 74296b1a9f7..d5a80b151e3 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeFormats.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeFormats.java @@ -17,14 +17,16 @@ package org.apache.hop.spark.table; +import org.apache.hop.lakehouse.LakeFormats; + /** * Format identifiers and well-known class / Spark conf names for open table formats on the native * Spark engine. Connector JARs are optional at runtime (see {@link SparkLakeConnectorProbe}). */ public final class SparkLakeFormats { - public static final String FORMAT_DELTA = "delta"; - public static final String FORMAT_ICEBERG = "iceberg"; + public static final String FORMAT_DELTA = LakeFormats.FORMAT_DELTA; + public static final String FORMAT_ICEBERG = LakeFormats.FORMAT_ICEBERG; /** Delta SparkSession extension (must be set at session build when Delta is used). */ public static final String DELTA_EXTENSION = "io.delta.sql.DeltaSparkSessionExtension"; @@ -33,7 +35,7 @@ public final class SparkLakeFormats { * Delta catalog implementation for {@code spark.sql.catalog.spark_catalog}. Required for modern * Delta 4.x on Spark 4 (SQL / MERGE / OPTIMIZE); hop-run always sets it when Delta is needed. */ - public static final String DELTA_CATALOG = "org.apache.spark.sql.delta.catalog.DeltaCatalog"; + public static final String DELTA_CATALOG = LakeFormats.DELTA_CATALOG; public static final String SPARK_CONF_EXTENSIONS = "spark.sql.extensions"; public static final String SPARK_CONF_SPARK_CATALOG = "spark.sql.catalog.spark_catalog"; @@ -43,7 +45,7 @@ public final class SparkLakeFormats { "org.apache.iceberg.spark.extensions.IcebergSparkSessionExtensions"; /** Default Iceberg catalog implementation (Hadoop / REST via catalog conf). */ - public static final String ICEBERG_CATALOG = "org.apache.iceberg.spark.SparkCatalog"; + public static final String ICEBERG_CATALOG = LakeFormats.ICEBERG_CATALOG; /** * Built-in Hadoop catalog name used for Iceberg PATH mode ({@code hop_iceberg.`file:///…`}). diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeTableSupport.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeTableSupport.java index aabd0f81abb..7cc0bf747b9 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeTableSupport.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkLakeTableSupport.java @@ -28,10 +28,12 @@ import org.apache.hop.core.exception.HopException; import org.apache.hop.core.logging.ILogChannel; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.LakeField; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.spark.pipeline.handler.SparkFileInputHandler; import org.apache.hop.spark.pipeline.handler.SparkFileIoSupport; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; +import org.apache.hop.spark.transforms.io.SparkField; import org.apache.hop.spark.util.SparkPathDialect; import org.apache.spark.sql.DataFrameReader; import org.apache.spark.sql.Dataset; @@ -72,20 +74,19 @@ public static String normalizeFormat(String format) throws HopException { public static String normalizeIdentifierMode(String mode) throws HopException { if (StringUtils.isEmpty(mode)) { - return SparkLakeTableInputMeta.MODE_PATH; + return LakeTableInputMeta.MODE_PATH; } String m = mode.trim().toUpperCase(Locale.ROOT); - if (SparkLakeTableInputMeta.MODE_PATH.equals(m) - || SparkLakeTableInputMeta.MODE_TABLE.equals(m)) { + if (LakeTableInputMeta.MODE_PATH.equals(m) || LakeTableInputMeta.MODE_TABLE.equals(m)) { return m; } throw new HopException( "Unsupported identifier mode '" + mode + "'. Supported: " - + SparkLakeTableInputMeta.MODE_PATH + + LakeTableInputMeta.MODE_PATH + ", " - + SparkLakeTableInputMeta.MODE_TABLE + + LakeTableInputMeta.MODE_TABLE + "."); } @@ -119,23 +120,23 @@ public static String icebergPathSqlIdentifier(String path) { public static String normalizeTimeTravelType(String type) throws HopException { if (StringUtils.isEmpty(type)) { - return SparkLakeTableInputMeta.TIME_TRAVEL_NONE; + return LakeTableInputMeta.TIME_TRAVEL_NONE; } String t = type.trim().toUpperCase(Locale.ROOT); - if (SparkLakeTableInputMeta.TIME_TRAVEL_NONE.equals(t) - || SparkLakeTableInputMeta.TIME_TRAVEL_VERSION.equals(t) - || SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP.equals(t)) { + if (LakeTableInputMeta.TIME_TRAVEL_NONE.equals(t) + || LakeTableInputMeta.TIME_TRAVEL_VERSION.equals(t) + || LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP.equals(t)) { return t; } throw new HopException( "Unsupported time travel type '" + type + "'. Supported: " - + SparkLakeTableInputMeta.TIME_TRAVEL_NONE + + LakeTableInputMeta.TIME_TRAVEL_NONE + ", " - + SparkLakeTableInputMeta.TIME_TRAVEL_VERSION + + LakeTableInputMeta.TIME_TRAVEL_VERSION + ", " - + SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP + + LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP + "."); } @@ -155,10 +156,10 @@ public static Map timeTravelOptionMap( Map map = new java.util.LinkedHashMap<>(); String fmt = normalizeFormat(format); String tt = normalizeTimeTravelType(timeTravelType); - if (SparkLakeTableInputMeta.TIME_TRAVEL_NONE.equals(tt)) { + if (LakeTableInputMeta.TIME_TRAVEL_NONE.equals(tt)) { return map; } - if (SparkLakeTableInputMeta.TIME_TRAVEL_VERSION.equals(tt)) { + if (LakeTableInputMeta.TIME_TRAVEL_VERSION.equals(tt)) { if (StringUtils.isEmpty(version)) { throw new HopException("Time travel type VERSION requires a version / snapshot id value"); } @@ -193,7 +194,7 @@ public static Map timeTravelOptionMap( public static String resolveMergeTargetSqlId( SparkSession spark, IVariables variables, - org.apache.hop.spark.transforms.table.SparkLakeTableMergeMeta meta, + org.apache.hop.lakehouse.transforms.LakeTableMergeMeta meta, String transformName) throws HopException { return resolveMergeTargetSqlId(spark, variables, meta, transformName, null); @@ -202,7 +203,7 @@ public static String resolveMergeTargetSqlId( public static String resolveMergeTargetSqlId( SparkSession spark, IVariables variables, - org.apache.hop.spark.transforms.table.SparkLakeTableMergeMeta meta, + org.apache.hop.lakehouse.transforms.LakeTableMergeMeta meta, String transformName, String pathSchemeMap) throws HopException { @@ -224,7 +225,7 @@ public static String resolveMergeTargetSqlId( public static MaintenanceTarget resolveMaintenanceTarget( SparkSession spark, IVariables variables, - org.apache.hop.spark.transforms.table.SparkLakeTableMaintenanceMeta meta, + org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta meta, String transformName) throws HopException { return resolveMaintenanceTarget(spark, variables, meta, transformName, null); @@ -233,7 +234,7 @@ public static MaintenanceTarget resolveMaintenanceTarget( public static MaintenanceTarget resolveMaintenanceTarget( SparkSession spark, IVariables variables, - org.apache.hop.spark.transforms.table.SparkLakeTableMaintenanceMeta meta, + org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta meta, String transformName, String pathSchemeMap) throws HopException { @@ -254,7 +255,7 @@ public static MaintenanceTarget resolveMaintenanceTarget( String procedureCatalog = null; String tableRefForCall = null; if (SparkLakeFormats.FORMAT_ICEBERG.equals(format)) { - if (SparkLakeTableInputMeta.MODE_PATH.equals(mode)) { + if (LakeTableInputMeta.MODE_PATH.equals(mode)) { procedureCatalog = SparkLakeFormats.ICEBERG_PATH_CATALOG_NAME; tableRefForCall = toTableLocationUri( @@ -317,7 +318,7 @@ public static String resolveLakeTargetSqlId( String format = normalizeFormat(formatRaw); String mode = normalizeIdentifierMode(modeRaw); - if (SparkLakeTableInputMeta.MODE_TABLE.equals(mode)) { + if (LakeTableInputMeta.MODE_TABLE.equals(mode)) { return resolveTableIdentifier(tableIdentifier, null, variables); } @@ -378,7 +379,7 @@ public static Dataset resolveRead( IVariables variables, ILogChannel log, String transformName, - SparkLakeTableInputMeta meta) + LakeTableInputMeta meta) throws HopException { return resolveRead(spark, variables, log, transformName, meta, null); } @@ -392,7 +393,7 @@ public static Dataset resolveRead( IVariables variables, ILogChannel log, String transformName, - SparkLakeTableInputMeta meta, + LakeTableInputMeta meta, String pathSchemeMap) throws HopException { @@ -415,7 +416,7 @@ public static Dataset resolveRead( options.putAll(ttOptions); Dataset dataset; - if (SparkLakeTableInputMeta.MODE_TABLE.equals(mode)) { + if (LakeTableInputMeta.MODE_TABLE.equals(mode)) { String tableId = resolveTableIdentifier(meta.getTableIdentifier(), null, variables); // Prefer catalogMetadataName only for session plan; identifier is full Spark id dataset = @@ -436,18 +437,31 @@ public static Dataset resolveRead( if (meta.getFields() != null && !meta.getFields().isEmpty()) { dataset = - SparkFileInputHandler.projectAndCastByName(log, transformName, dataset, meta.getFields()); + SparkFileInputHandler.projectAndCastByName( + log, transformName, dataset, toSparkFields(meta.getFields())); } return dataset; } + static List toSparkFields(List fields) { + List sparkFields = new ArrayList<>(fields.size()); + for (LakeField field : fields) { + SparkField sparkField = + new SparkField( + field.getName(), field.getHopType(), field.getLength(), field.getPrecision()); + sparkField.setFormatMask(field.getFormatMask()); + sparkFields.add(sparkField); + } + return sparkFields; + } + /** Write a Dataset to a lake table (PATH + TABLE). Action runs immediately. */ public static void resolveWrite( SparkSession spark, Dataset dataset, IVariables variables, String transformName, - SparkLakeTableOutputMeta meta) + LakeTableOutputMeta meta) throws HopException { resolveWrite(spark, dataset, variables, transformName, meta, null); } @@ -461,7 +475,7 @@ public static void resolveWrite( Dataset dataset, IVariables variables, String transformName, - SparkLakeTableOutputMeta meta, + LakeTableOutputMeta meta, String pathSchemeMap) throws HopException { @@ -490,7 +504,7 @@ public static void resolveWrite( toWrite = toWrite.coalesce(coalesce); } - if (SparkLakeTableInputMeta.MODE_TABLE.equals(mode)) { + if (LakeTableInputMeta.MODE_TABLE.equals(mode)) { String tableId = resolveTableIdentifier(meta.getTableIdentifier(), null, variables); writeTable( spark, toWrite, format, tableId, saveMode, options, partitionColumns, transformName); @@ -697,10 +711,10 @@ static String buildIcebergTimeTravelSql( String sqlIdentifier, String timeTravelType, String version, String timestamp) throws HopException { String tt = normalizeTimeTravelType(timeTravelType); - if (SparkLakeTableInputMeta.TIME_TRAVEL_NONE.equals(tt)) { + if (LakeTableInputMeta.TIME_TRAVEL_NONE.equals(tt)) { return "SELECT * FROM " + sqlIdentifier; } - if (SparkLakeTableInputMeta.TIME_TRAVEL_VERSION.equals(tt)) { + if (LakeTableInputMeta.TIME_TRAVEL_VERSION.equals(tt)) { if (StringUtils.isEmpty(version)) { throw new HopException("Iceberg VERSION time travel requires a snapshot id"); } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMaintenanceSqlBuilder.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMaintenanceSqlBuilder.java index 9b370d3e6d4..d0b3fdb7a76 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMaintenanceSqlBuilder.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMaintenanceSqlBuilder.java @@ -20,6 +20,7 @@ import java.util.Locale; import org.apache.commons.lang3.StringUtils; import org.apache.hop.core.exception.HopException; +import org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta; /** * Builds Spark SQL for lakehouse maintenance (OPTIMIZE / VACUUM / expire / rewrite / DELETE). Does @@ -27,11 +28,11 @@ */ public final class SparkMaintenanceSqlBuilder { - public static final String OP_OPTIMIZE = "OPTIMIZE"; - public static final String OP_VACUUM = "VACUUM"; - public static final String OP_EXPIRE_SNAPSHOTS = "EXPIRE_SNAPSHOTS"; - public static final String OP_REWRITE_MANIFESTS = "REWRITE_MANIFESTS"; - public static final String OP_DELETE_WHERE = "DELETE_WHERE"; + public static final String OP_OPTIMIZE = LakeTableMaintenanceMeta.OP_OPTIMIZE; + public static final String OP_VACUUM = LakeTableMaintenanceMeta.OP_VACUUM; + public static final String OP_EXPIRE_SNAPSHOTS = LakeTableMaintenanceMeta.OP_EXPIRE_SNAPSHOTS; + public static final String OP_REWRITE_MANIFESTS = LakeTableMaintenanceMeta.OP_REWRITE_MANIFESTS; + public static final String OP_DELETE_WHERE = LakeTableMaintenanceMeta.OP_DELETE_WHERE; private SparkMaintenanceSqlBuilder() {} diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMergeSqlBuilder.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMergeSqlBuilder.java index 010a9e6a2a4..6c910852af4 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMergeSqlBuilder.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/table/SparkMergeSqlBuilder.java @@ -20,6 +20,7 @@ import java.util.Locale; import org.apache.commons.lang3.StringUtils; import org.apache.hop.core.exception.HopException; +import org.apache.hop.lakehouse.transforms.LakeTableMergeMeta; /** * Builds Spark SQL {@code MERGE INTO} statements for lakehouse upserts. Does not execute SQL. @@ -30,15 +31,17 @@ */ public final class SparkMergeSqlBuilder { - public static final String MATCHED_UPDATE_ALL = "UPDATE_ALL"; - public static final String MATCHED_DELETE = "DELETE"; - public static final String MATCHED_NONE = "NONE"; + public static final String MATCHED_UPDATE_ALL = LakeTableMergeMeta.MATCHED_UPDATE_ALL; + public static final String MATCHED_DELETE = LakeTableMergeMeta.MATCHED_DELETE; + public static final String MATCHED_NONE = LakeTableMergeMeta.MATCHED_NONE; - public static final String NOT_MATCHED_INSERT_ALL = "INSERT_ALL"; - public static final String NOT_MATCHED_NONE = "NONE"; + public static final String NOT_MATCHED_INSERT_ALL = LakeTableMergeMeta.NOT_MATCHED_INSERT_ALL; + public static final String NOT_MATCHED_NONE = LakeTableMergeMeta.NOT_MATCHED_NONE; - public static final String NOT_MATCHED_BY_SOURCE_DELETE = "DELETE"; - public static final String NOT_MATCHED_BY_SOURCE_NONE = "NONE"; + public static final String NOT_MATCHED_BY_SOURCE_DELETE = + LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_DELETE; + public static final String NOT_MATCHED_BY_SOURCE_NONE = + LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_NONE; private SparkMergeSqlBuilder() {} diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/util/SparkConst.java b/plugins/engines/spark/src/main/java/org/apache/hop/spark/util/SparkConst.java index 7b6d21088ff..2d72ae47319 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/util/SparkConst.java +++ b/plugins/engines/spark/src/main/java/org/apache/hop/spark/util/SparkConst.java @@ -17,9 +17,11 @@ package org.apache.hop.spark.util; +import org.apache.hop.lakehouse.LakehouseConst; + public final class SparkConst { - public static final String PLUGIN_ID = "SparkPipelineEngine"; + public static final String PLUGIN_ID = LakehouseConst.SPARK_ENGINE_ID; public static final String PLUGIN_NAME = "Native Spark pipeline engine"; public static final String INJECTOR_TRANSFORM_NAME = "_INJECTOR_"; @@ -45,13 +47,17 @@ public final class SparkConst { public static final String SPARK_FILE_OUTPUT_PLUGIN_ID = "SparkFileOutput"; /** Open table format (Delta / Iceberg) path and catalog I/O — native Spark only. */ - public static final String SPARK_LAKE_TABLE_INPUT_PLUGIN_ID = "SparkLakeTableInput"; + public static final String SPARK_LAKE_TABLE_INPUT_PLUGIN_ID = + LakehouseConst.LAKE_TABLE_INPUT_PLUGIN_ID; - public static final String SPARK_LAKE_TABLE_OUTPUT_PLUGIN_ID = "SparkLakeTableOutput"; + public static final String SPARK_LAKE_TABLE_OUTPUT_PLUGIN_ID = + LakehouseConst.LAKE_TABLE_OUTPUT_PLUGIN_ID; - public static final String SPARK_LAKE_TABLE_MERGE_PLUGIN_ID = "SparkLakeTableMerge"; + public static final String SPARK_LAKE_TABLE_MERGE_PLUGIN_ID = + LakehouseConst.LAKE_TABLE_MERGE_PLUGIN_ID; - public static final String SPARK_LAKE_TABLE_MAINTENANCE_PLUGIN_ID = "SparkLakeTableMaintenance"; + public static final String SPARK_LAKE_TABLE_MAINTENANCE_PLUGIN_ID = + LakehouseConst.LAKE_TABLE_MAINTENANCE_PLUGIN_ID; /** Spark SQL over the Datasets of the incoming transforms — native Spark only. */ public static final String SPARK_SQL_PLUGIN_ID = "SparkSql"; diff --git a/plugins/engines/spark/src/main/resources/dependencies.xml b/plugins/engines/spark/src/main/resources/dependencies.xml index 55ac030df2e..c3b8fc78473 100644 --- a/plugins/engines/spark/src/main/resources/dependencies.xml +++ b/plugins/engines/spark/src/main/resources/dependencies.xml @@ -17,6 +17,7 @@ --> + ../../tech/lakehouse ../../transforms/memgroupby ../../transforms/mergejoin ../../transforms/uniquerows diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/LakeSessionPlanCopyTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/LakeSessionPlanCopyTest.java new file mode 100644 index 00000000000..baedb12b2e4 --- /dev/null +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/LakeSessionPlanCopyTest.java @@ -0,0 +1,67 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.hop.spark.table; + +import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertThrows; + +import org.apache.hop.core.HopEnvironment; +import org.apache.hop.core.exception.HopException; +import org.apache.hop.lakehouse.LakeField; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableMergeMeta; +import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; +import org.junit.jupiter.api.BeforeAll; +import org.junit.jupiter.api.Test; + +class LakeSessionPlanCopyTest { + + @BeforeAll + static void init() throws Exception { + HopEnvironment.init(); + } + + @Test + void copiesEverySettingOfTheLiveMeta() throws Exception { + LakeTableInputMeta live = new LakeTableInputMeta(); + live.setFormat(LakeFormats.FORMAT_ICEBERG); + live.setIdentifierMode(LakeTableInputMeta.MODE_TABLE); + live.setTableIdentifier("lake.sales.orders"); + live.setCatalogMetadataName("lake"); + live.setTimeTravelType(LakeTableInputMeta.TIME_TRAVEL_VERSION); + live.setTimeTravelVersion("42"); + live.getFields().add(new LakeField("id", "Integer")); + + LakeTableInputMeta copy = new LakeTableInputMeta(); + LakeSessionPlan.copyFromLive(copy, live, new MemoryMetadataProvider()); + + assertEquals(live.getXml(), copy.getXml()); + assertEquals("lake.sales.orders", copy.getTableIdentifier()); + assertEquals("id", copy.getFields().get(0).getName()); + } + + @Test + void refusesAnotherTransformType() { + assertThrows( + HopException.class, + () -> + LakeSessionPlan.copyFromLive( + new LakeTableInputMeta(), new LakeTableMergeMeta(), new MemoryMetadataProvider())); + } +} diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkCatalogApplierTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkCatalogApplierTest.java index b2ea45c754b..117814961bb 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkCatalogApplierTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkCatalogApplierTest.java @@ -24,17 +24,17 @@ import java.util.Map; import org.apache.hop.core.exception.HopException; import org.apache.hop.core.variables.Variables; -import org.apache.hop.spark.metadata.SparkCatalog; +import org.apache.hop.lakehouse.metadata.LakeCatalog; import org.junit.jupiter.api.Test; class SparkCatalogApplierTest { @Test void hadoopCatalogExpandsConf() throws Exception { - SparkCatalog cat = new SparkCatalog(); + LakeCatalog cat = new LakeCatalog(); cat.setName("my-lake"); cat.setCatalogName("lake"); - cat.setCatalogType(SparkCatalog.TYPE_HADOOP); + cat.setCatalogType(LakeCatalog.TYPE_HADOOP); cat.setWarehouse("/tmp/warehouse"); cat.setConfExtra("io-impl=org.apache.iceberg.hadoop.HadoopFileIO\n# comment\n"); @@ -48,20 +48,20 @@ void hadoopCatalogExpandsConf() throws Exception { @Test void restCatalogRequiresUri() { - SparkCatalog cat = new SparkCatalog(); + LakeCatalog cat = new LakeCatalog(); cat.setName("rest"); cat.setCatalogName("remote"); - cat.setCatalogType(SparkCatalog.TYPE_REST); + cat.setCatalogType(LakeCatalog.TYPE_REST); assertThrows( HopException.class, () -> SparkCatalogApplier.toSparkConfigs(cat, new Variables())); } @Test void restCatalogWithToken() throws Exception { - SparkCatalog cat = new SparkCatalog(); + LakeCatalog cat = new LakeCatalog(); cat.setName("rest"); cat.setCatalogName("remote"); - cat.setCatalogType(SparkCatalog.TYPE_REST); + cat.setCatalogType(LakeCatalog.TYPE_REST); cat.setUri("https://catalog.example.com/iceberg"); cat.setCredential("secret-token"); Map conf = SparkCatalogApplier.toSparkConfigs(cat, new Variables()); @@ -72,10 +72,10 @@ void restCatalogWithToken() throws Exception { @Test void fullSparkKeyInConfExtra() throws Exception { - SparkCatalog cat = new SparkCatalog(); + LakeCatalog cat = new LakeCatalog(); cat.setName("c"); cat.setCatalogName("c"); - cat.setCatalogType(SparkCatalog.TYPE_CUSTOM); + cat.setCatalogType(LakeCatalog.TYPE_CUSTOM); cat.setImplementation("com.example.MyCatalog"); cat.setConfExtra("spark.sql.defaultCatalog=c"); Map conf = SparkCatalogApplier.toSparkConfigs(cat, new Variables()); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableDeltaPathTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableDeltaPathTest.java index d0963c846f5..28bb7292069 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableDeltaPathTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableDeltaPathTest.java @@ -29,6 +29,8 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -36,8 +38,6 @@ import org.apache.hop.spark.pipeline.handler.SparkLakeTableInputHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -100,9 +100,9 @@ void deltaPathOutputThenInputRoundTrip() throws Exception { Dataset source = spark.range(0, 20).toDF("id"); - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); outMeta.setTablePath(tablePath.toString()); // Overwrite so the test is idempotent outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); @@ -132,9 +132,9 @@ void deltaPathOutputThenInputRoundTrip() throws Exception { // Empty leaf after write assertEquals(0, map.get("lake_out").count()); - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); - inMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + inMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); inMeta.setTablePath(tablePath.toString()); TransformMeta inTm = new TransformMeta("lake_in", inMeta); @@ -162,7 +162,7 @@ void deltaPathOutputThenInputRoundTrip() throws Exception { @Test void lakeSessionPlanCollectsDeltaFormat() throws Exception { - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); inMeta.setTablePath("/tmp/x"); TransformMeta inTm = new TransformMeta("in", inMeta); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableIcebergPathTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableIcebergPathTest.java index e9bf9b96dbb..d4b5e532475 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableIcebergPathTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableIcebergPathTest.java @@ -29,6 +29,8 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -36,8 +38,6 @@ import org.apache.hop.spark.pipeline.handler.SparkLakeTableInputHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -106,9 +106,9 @@ void icebergPathOutputThenInputRoundTrip() throws Exception { Dataset source = spark.range(0, 18).toDF("id"); - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); outMeta.setTablePath(tablePath.toString()); outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); @@ -136,9 +136,9 @@ void icebergPathOutputThenInputRoundTrip() throws Exception { assertEquals(0, map.get("ice_out").count()); - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); - inMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + inMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); inMeta.setTablePath(tablePath.toString()); TransformMeta inTm = new TransformMeta("ice_in", inMeta); @@ -166,7 +166,7 @@ void icebergPathOutputThenInputRoundTrip() throws Exception { @Test void lakeSessionPlanCollectsIcebergFormat() throws Exception { - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); inMeta.setTablePath("/tmp/x"); TransformMeta inTm = new TransformMeta("in", inMeta); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMaintenanceTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMaintenanceTest.java index 05998f07aea..a2bf84b0c17 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMaintenanceTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMaintenanceTest.java @@ -31,6 +31,9 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableMaintenanceMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -38,9 +41,6 @@ import org.apache.hop.spark.pipeline.handler.SparkLakeTableMaintenanceHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableMaintenanceMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -83,9 +83,9 @@ void stopSpark() { @Test void vacuumRequiresAcknowledge() { - SparkLakeTableMaintenanceMeta meta = new SparkLakeTableMaintenanceMeta(); + LakeTableMaintenanceMeta meta = new LakeTableMaintenanceMeta(); meta.setFormat(SparkLakeFormats.FORMAT_DELTA); - meta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + meta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); meta.setTablePath("/tmp/t"); meta.setOperation(SparkMaintenanceSqlBuilder.OP_VACUUM); meta.setRetentionHours("168"); @@ -139,9 +139,9 @@ void deltaOptimizeSmoke() throws Exception { // Seed small table writeDelta(tablePath.toString(), spark.range(0, 50).toDF("id")); - SparkLakeTableMaintenanceMeta meta = new SparkLakeTableMaintenanceMeta(); + LakeTableMaintenanceMeta meta = new LakeTableMaintenanceMeta(); meta.setFormat(SparkLakeFormats.FORMAT_DELTA); - meta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + meta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); meta.setTablePath(tablePath.toString()); meta.setOperation(SparkMaintenanceSqlBuilder.OP_OPTIMIZE); meta.setAcknowledgeDestructive(false); @@ -177,9 +177,9 @@ private static void assertTrue(boolean cond) { } private void writeDelta(String path, Dataset data) throws Exception { - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); outMeta.setTablePath(path); outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); TransformMeta outTm = new TransformMeta("seed", outMeta); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMergeTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMergeTest.java index 09f89997c90..8b586592e7e 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMergeTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableMergeTest.java @@ -29,6 +29,9 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableMergeMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -36,9 +39,6 @@ import org.apache.hop.spark.pipeline.handler.SparkLakeTableMergeHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableMergeMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -119,9 +119,9 @@ void deltaPathMergeUpsert() throws Exception { spark.createDataFrame( List.of(RowFactory.create(1L, "a-updated"), RowFactory.create(3L, "c")), schema); - SparkLakeTableMergeMeta mergeMeta = new SparkLakeTableMergeMeta(); + LakeTableMergeMeta mergeMeta = new LakeTableMergeMeta(); mergeMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); - mergeMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + mergeMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); mergeMeta.setTablePath(tablePath.toString()); mergeMeta.setMergeCondition("t.id = s.id"); mergeMeta.setMatchedAction(SparkMergeSqlBuilder.MATCHED_UPDATE_ALL); @@ -165,18 +165,18 @@ void deltaPathMergeUpsert() throws Exception { @Test void resolveMergeTargetDeltaPath() throws Exception { - SparkLakeTableMergeMeta meta = new SparkLakeTableMergeMeta(); + LakeTableMergeMeta meta = new LakeTableMergeMeta(); meta.setFormat(SparkLakeFormats.FORMAT_DELTA); - meta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + meta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); meta.setTablePath("/tmp/orders"); String id = SparkLakeTableSupport.resolveMergeTargetSqlId(null, new Variables(), meta, "m"); assertEquals("delta.`/tmp/orders`", id); } private void writeDelta(String path, Dataset data) throws Exception { - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(SparkLakeFormats.FORMAT_DELTA); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); outMeta.setTablePath(path); outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); TransformMeta outTm = new TransformMeta("seed", outMeta); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableSupportTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableSupportTest.java index 59889026b79..2470c825a30 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableSupportTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableSupportTest.java @@ -26,8 +26,8 @@ import java.util.Set; import org.apache.hop.core.exception.HopException; import org.apache.hop.core.variables.Variables; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.junit.jupiter.api.Test; /** Unit tests that do not require Delta/Iceberg connectors on the classpath. */ @@ -45,10 +45,9 @@ void normalizeFormatDefaultsAndValidates() throws Exception { @Test void normalizeIdentifierMode() throws Exception { + assertEquals(LakeTableInputMeta.MODE_PATH, SparkLakeTableSupport.normalizeIdentifierMode(null)); assertEquals( - SparkLakeTableInputMeta.MODE_PATH, SparkLakeTableSupport.normalizeIdentifierMode(null)); - assertEquals( - SparkLakeTableInputMeta.MODE_TABLE, SparkLakeTableSupport.normalizeIdentifierMode("table")); + LakeTableInputMeta.MODE_TABLE, SparkLakeTableSupport.normalizeIdentifierMode("table")); assertThrows(HopException.class, () -> SparkLakeTableSupport.normalizeIdentifierMode("uri")); } @@ -81,7 +80,7 @@ void resolveLakeTargetAppliesPathSchemeMap() throws Exception { null, new Variables(), SparkLakeFormats.FORMAT_DELTA, - SparkLakeTableInputMeta.MODE_PATH, + LakeTableInputMeta.MODE_PATH, "s3://bucket/table", null, "merge", @@ -109,9 +108,9 @@ void resolveTableIdentifierRequiresValue() { @Test void resolveWriteRejectsMissingPath() { - SparkLakeTableOutputMeta meta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta meta = new LakeTableOutputMeta(); meta.setFormat(SparkLakeFormats.FORMAT_DELTA); - meta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + meta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); meta.setTablePath(""); HopException ex = assertThrows( @@ -122,7 +121,7 @@ void resolveWriteRejectsMissingPath() { @Test void defaultSaveModeIsErrorIfExists() { - SparkLakeTableOutputMeta meta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta meta = new LakeTableOutputMeta(); assertEquals( org.apache.hop.spark.transforms.io.SparkFileOutputMeta.MODE_ERROR, meta.getSaveMode()); } @@ -131,14 +130,14 @@ void defaultSaveModeIsErrorIfExists() { void timeTravelOptionMapDelta() throws Exception { Map v = SparkLakeTableSupport.timeTravelOptionMap( - SparkLakeFormats.FORMAT_DELTA, SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, "12", null); + SparkLakeFormats.FORMAT_DELTA, LakeTableInputMeta.TIME_TRAVEL_VERSION, "12", null); assertEquals("12", v.get("versionAsOf")); assertEquals(1, v.size()); Map t = SparkLakeTableSupport.timeTravelOptionMap( SparkLakeFormats.FORMAT_DELTA, - SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, + LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, null, "2024-01-15 10:00:00"); assertEquals("2024-01-15 10:00:00", t.get("timestampAsOf")); @@ -148,16 +147,13 @@ void timeTravelOptionMapDelta() throws Exception { void timeTravelOptionMapIceberg() throws Exception { Map v = SparkLakeTableSupport.timeTravelOptionMap( - SparkLakeFormats.FORMAT_ICEBERG, - SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, - "999", - null); + SparkLakeFormats.FORMAT_ICEBERG, LakeTableInputMeta.TIME_TRAVEL_VERSION, "999", null); assertEquals("999", v.get("snapshot-id")); Map t = SparkLakeTableSupport.timeTravelOptionMap( SparkLakeFormats.FORMAT_ICEBERG, - SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, + LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, null, "2024-06-01 12:00:00"); assertEquals("2024-06-01 12:00:00", t.get("as-of-timestamp")); @@ -167,7 +163,7 @@ void timeTravelOptionMapIceberg() throws Exception { void timeTravelNoneYieldsEmptyMap() throws Exception { assertTrue( SparkLakeTableSupport.timeTravelOptionMap( - SparkLakeFormats.FORMAT_DELTA, SparkLakeTableInputMeta.TIME_TRAVEL_NONE, "1", "t") + SparkLakeFormats.FORMAT_DELTA, LakeTableInputMeta.TIME_TRAVEL_NONE, "1", "t") .isEmpty()); } @@ -177,10 +173,7 @@ void timeTravelVersionRequiresValue() { HopException.class, () -> SparkLakeTableSupport.timeTravelOptionMap( - SparkLakeFormats.FORMAT_DELTA, - SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, - "", - null)); + SparkLakeFormats.FORMAT_DELTA, LakeTableInputMeta.TIME_TRAVEL_VERSION, "", null)); } @Test @@ -189,14 +182,14 @@ void icebergTimeTravelSql() throws Exception { assertEquals( "SELECT * FROM " + id, SparkLakeTableSupport.buildIcebergTimeTravelSql( - id, SparkLakeTableInputMeta.TIME_TRAVEL_NONE, null, null)); + id, LakeTableInputMeta.TIME_TRAVEL_NONE, null, null)); assertEquals( "SELECT * FROM " + id + " VERSION AS OF 42", SparkLakeTableSupport.buildIcebergTimeTravelSql( - id, SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, "42", null)); + id, LakeTableInputMeta.TIME_TRAVEL_VERSION, "42", null)); assertEquals( "SELECT * FROM " + id + " TIMESTAMP AS OF TIMESTAMP '2024-01-01 00:00:00'", SparkLakeTableSupport.buildIcebergTimeTravelSql( - id, SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, null, "2024-01-01 00:00:00")); + id, LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP, null, "2024-01-01 00:00:00")); } } diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTableModeTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTableModeTest.java index d1d19acec55..a04d99f11a8 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTableModeTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTableModeTest.java @@ -30,16 +30,16 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.metadata.LakeCatalog; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; import org.apache.hop.spark.engines.SparkPipelineRunConfiguration; -import org.apache.hop.spark.metadata.SparkCatalog; import org.apache.hop.spark.pipeline.handler.SparkLakeTableInputHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -49,7 +49,7 @@ import org.junit.jupiter.api.Test; import org.junit.jupiter.api.io.TempDir; -/** Iceberg TABLE mode via SparkCatalog (Hadoop warehouse; skipped if connectors missing). */ +/** Iceberg TABLE mode via LakeCatalog (Hadoop warehouse; skipped if connectors missing). */ class SparkLakeTableTableModeTest { @TempDir Path tempDir; @@ -87,14 +87,14 @@ void icebergTableModeRoundTripWithSparkCatalog() throws Exception { "Iceberg connector not on classpath; connectors missing from test classpath"); Path warehouse = tempDir.resolve("wh"); - SparkCatalog catalogMeta = new SparkCatalog(); + LakeCatalog catalogMeta = new LakeCatalog(); catalogMeta.setName("lake-meta"); catalogMeta.setCatalogName("lake"); - catalogMeta.setCatalogType(SparkCatalog.TYPE_HADOOP); + catalogMeta.setCatalogType(LakeCatalog.TYPE_HADOOP); catalogMeta.setWarehouse(warehouse.toUri().toString()); MemoryMetadataProvider provider = new MemoryMetadataProvider(); - provider.getSerializer(SparkCatalog.class).save(catalogMeta); + provider.getSerializer(LakeCatalog.class).save(catalogMeta); // Build session as LakeSessionPlan would for hop-run SparkSession.Builder builder = @@ -110,9 +110,9 @@ void icebergTableModeRoundTripWithSparkCatalog() throws Exception { String tableId = "lake.db.orders"; - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_TABLE); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_TABLE); outMeta.setTableIdentifier(tableId); outMeta.setCatalogMetadataName("lake-meta"); outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); @@ -142,9 +142,9 @@ void icebergTableModeRoundTripWithSparkCatalog() throws Exception { assertEquals(0, map.get("out").count()); assertEquals(12L, spark.table(tableId).count()); - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); - inMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_TABLE); + inMeta.setIdentifierMode(LakeTableInputMeta.MODE_TABLE); inMeta.setTableIdentifier(tableId); inMeta.setCatalogMetadataName("lake-meta"); @@ -173,18 +173,18 @@ void icebergTableModeRoundTripWithSparkCatalog() throws Exception { @Test void lakeSessionPlanLoadsSparkCatalog() throws Exception { - SparkCatalog catalogMeta = new SparkCatalog(); + LakeCatalog catalogMeta = new LakeCatalog(); catalogMeta.setName("lake-meta"); catalogMeta.setCatalogName("lake"); - catalogMeta.setCatalogType(SparkCatalog.TYPE_HADOOP); + catalogMeta.setCatalogType(LakeCatalog.TYPE_HADOOP); catalogMeta.setWarehouse("/tmp/wh"); MemoryMetadataProvider provider = new MemoryMetadataProvider(); - provider.getSerializer(SparkCatalog.class).save(catalogMeta); + provider.getSerializer(LakeCatalog.class).save(catalogMeta); - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(SparkLakeFormats.FORMAT_ICEBERG); - inMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_TABLE); + inMeta.setIdentifierMode(LakeTableInputMeta.MODE_TABLE); inMeta.setTableIdentifier("lake.db.t"); inMeta.setCatalogMetadataName("lake-meta"); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTimeTravelTest.java b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTimeTravelTest.java index 9b9182b8541..5a798ad3d50 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTimeTravelTest.java +++ b/plugins/engines/spark/src/test/java/org/apache/hop/spark/table/SparkLakeTableTimeTravelTest.java @@ -29,6 +29,8 @@ import org.apache.hop.core.logging.LogChannel; import org.apache.hop.core.row.RowMeta; import org.apache.hop.core.variables.Variables; +import org.apache.hop.lakehouse.transforms.LakeTableInputMeta; +import org.apache.hop.lakehouse.transforms.LakeTableOutputMeta; import org.apache.hop.metadata.serializer.memory.MemoryMetadataProvider; import org.apache.hop.pipeline.PipelineMeta; import org.apache.hop.pipeline.transform.TransformMeta; @@ -36,8 +38,6 @@ import org.apache.hop.spark.pipeline.handler.SparkLakeTableInputHandler; import org.apache.hop.spark.pipeline.handler.SparkLakeTableOutputHandler; import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableInputMeta; -import org.apache.hop.spark.transforms.table.SparkLakeTableOutputMeta; import org.apache.hop.spark.util.SparkConst; import org.apache.spark.sql.Dataset; import org.apache.spark.sql.Row; @@ -108,7 +108,7 @@ void deltaVersionAsOfFirstWrite() throws Exception { readLake( tablePath.toString(), SparkLakeFormats.FORMAT_DELTA, - SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, + LakeTableInputMeta.TIME_TRAVEL_VERSION, "0")); } @@ -151,14 +151,14 @@ void icebergSnapshotAsOfFirstWrite() throws Exception { readLake( tablePath.toString(), SparkLakeFormats.FORMAT_ICEBERG, - SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, + LakeTableInputMeta.TIME_TRAVEL_VERSION, Long.toString(firstSnap))); } private void writeLake(String path, String format, Dataset data) throws Exception { - SparkLakeTableOutputMeta outMeta = new SparkLakeTableOutputMeta(); + LakeTableOutputMeta outMeta = new LakeTableOutputMeta(); outMeta.setFormat(format); - outMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + outMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); outMeta.setTablePath(path); outMeta.setSaveMode(SparkFileOutputMeta.MODE_OVERWRITE); @@ -187,9 +187,9 @@ private void writeLake(String path, String format, Dataset data) throws Exc private long readLake(String path, String format, String ttType, String version) throws Exception { - SparkLakeTableInputMeta inMeta = new SparkLakeTableInputMeta(); + LakeTableInputMeta inMeta = new LakeTableInputMeta(); inMeta.setFormat(format); - inMeta.setIdentifierMode(SparkLakeTableInputMeta.MODE_PATH); + inMeta.setIdentifierMode(LakeTableInputMeta.MODE_PATH); inMeta.setTablePath(path); if (ttType != null) { inMeta.setTimeTravelType(ttType); diff --git a/plugins/tech/lakehouse/pom.xml b/plugins/tech/lakehouse/pom.xml new file mode 100644 index 00000000000..1a267c8f0ac --- /dev/null +++ b/plugins/tech/lakehouse/pom.xml @@ -0,0 +1,31 @@ + + + + 4.0.0 + + + org.apache.hop + hop-plugins-tech + 2.20.0-SNAPSHOT + + + hop-tech-lakehouse + jar + Hop Plugins Technology Lakehouse + + diff --git a/plugins/tech/lakehouse/src/assembly/assembly.xml b/plugins/tech/lakehouse/src/assembly/assembly.xml new file mode 100644 index 00000000000..7fddda65208 --- /dev/null +++ b/plugins/tech/lakehouse/src/assembly/assembly.xml @@ -0,0 +1,45 @@ + + + + hop-tech-lakehouse + + zip + + . + + + ${project.basedir}/src/main/resources/version.xml + ${hop.plugin.libdir} + true + + + + + + ${project.basedir}/src/main/samples + config/projects/samples/ + + + + + + ${maven.multiModuleProjectDirectory}/assemblies/shared/hop-plugin-libs.xml + + diff --git a/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeField.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeField.java new file mode 100644 index 00000000000..8992336f3fa --- /dev/null +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeField.java @@ -0,0 +1,72 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.hop.lakehouse; + +import java.io.Serializable; +import lombok.Getter; +import lombok.Setter; +import org.apache.hop.core.exception.HopPluginException; +import org.apache.hop.core.row.IValueMeta; +import org.apache.hop.core.row.value.ValueMetaFactory; +import org.apache.hop.metadata.api.HopMetadataProperty; + +/** Field definition for Spark File Input schema. */ +@Getter +@Setter +public class LakeField implements Serializable { + private static final long serialVersionUID = 1L; + + @HopMetadataProperty(key = "name", injectionKey = "NAME") + private String name; + + /** Hop type description, e.g. String, Integer, Number, Boolean, Date, BigNumber, Binary */ + @HopMetadataProperty(key = "type", injectionKey = "TYPE") + private String hopType = "String"; + + @HopMetadataProperty(key = "length", injectionKey = "LENGTH") + private int length = -1; + + @HopMetadataProperty(key = "precision", injectionKey = "PRECISION") + private int precision = -1; + + @HopMetadataProperty(key = "format", injectionKey = "FORMAT_MASK") + private String formatMask; + + public LakeField() {} + + public LakeField(String name, String hopType) { + this.name = name; + this.hopType = hopType; + } + + public LakeField(String name, String hopType, int length, int precision) { + this.name = name; + this.hopType = hopType; + this.length = length; + this.precision = precision; + } + + public IValueMeta createValueMeta() throws HopPluginException { + int type = ValueMetaFactory.getIdForValueMeta(hopType); + IValueMeta valueMeta = ValueMetaFactory.createValueMeta(name, type, length, precision); + if (formatMask != null && !formatMask.isEmpty()) { + valueMeta.setConversionMask(formatMask); + } + return valueMeta; + } +} diff --git a/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeFormats.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeFormats.java new file mode 100644 index 00000000000..ea74a6c1874 --- /dev/null +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakeFormats.java @@ -0,0 +1,33 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.hop.lakehouse; + +/** Open table format identifiers and the catalog implementation classes they use. */ +public final class LakeFormats { + + public static final String FORMAT_DELTA = "delta"; + public static final String FORMAT_ICEBERG = "iceberg"; + + /** Delta catalog implementation (Spark {@code spark_catalog}). */ + public static final String DELTA_CATALOG = "org.apache.spark.sql.delta.catalog.DeltaCatalog"; + + /** Default Iceberg catalog implementation (Hadoop / REST via catalog conf). */ + public static final String ICEBERG_CATALOG = "org.apache.iceberg.spark.SparkCatalog"; + + private LakeFormats() {} +} diff --git a/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakehouseConst.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakehouseConst.java new file mode 100644 index 00000000000..cae1c9e9c76 --- /dev/null +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/LakehouseConst.java @@ -0,0 +1,35 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.hop.lakehouse; + +/** + * Plugin IDs of the lake table transforms. They keep their original values so that existing + * pipelines open unchanged. + */ +public final class LakehouseConst { + + public static final String LAKE_TABLE_INPUT_PLUGIN_ID = "SparkLakeTableInput"; + public static final String LAKE_TABLE_OUTPUT_PLUGIN_ID = "SparkLakeTableOutput"; + public static final String LAKE_TABLE_MERGE_PLUGIN_ID = "SparkLakeTableMerge"; + public static final String LAKE_TABLE_MAINTENANCE_PLUGIN_ID = "SparkLakeTableMaintenance"; + + /** ID of the native Spark pipeline engine plugin. */ + public static final String SPARK_ENGINE_ID = "SparkPipelineEngine"; + + private LakehouseConst() {} +} diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalog.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalog.java similarity index 89% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalog.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalog.java index 7ca4f067a70..5934cbb65cb 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalog.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalog.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.metadata; +package org.apache.hop.lakehouse.metadata; import java.io.Serializable; import lombok.Getter; @@ -24,13 +24,13 @@ import org.apache.hop.core.gui.plugin.GuiPlugin; import org.apache.hop.core.gui.plugin.GuiWidgetElement; import org.apache.hop.i18n.BaseMessages; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.metadata.template.LakeCatalogTemplate; import org.apache.hop.metadata.api.HopMetadata; import org.apache.hop.metadata.api.HopMetadataBase; import org.apache.hop.metadata.api.HopMetadataCategory; import org.apache.hop.metadata.api.HopMetadataProperty; import org.apache.hop.metadata.api.IHopMetadata; -import org.apache.hop.spark.metadata.template.SparkCatalogTemplate; -import org.apache.hop.spark.table.SparkLakeFormats; import org.apache.hop.ui.core.dialog.EnterSelectionDialog; import org.apache.hop.ui.hopgui.HopGui; import org.eclipse.swt.SWT; @@ -52,9 +52,9 @@ image = "spark-catalog.svg", category = HopMetadataCategory.CONNECTIONS, documentationUrl = "/metadata-types/spark-catalog.html") -public class SparkCatalog extends HopMetadataBase implements Serializable, IHopMetadata { +public class LakeCatalog extends HopMetadataBase implements Serializable, IHopMetadata { - private static final Class PKG = SparkCatalog.class; + private static final Class PKG = LakeCatalog.class; public static final String TYPE_HADOOP = "hadoop"; public static final String TYPE_REST = "rest"; @@ -66,7 +66,7 @@ public class SparkCatalog extends HopMetadataBase implements Serializable, IHopM /** Advanced — confExtra only; not packaged by default. */ public static final String TYPE_GLUE = "glue"; - private static final String PARENT = SparkCatalogEditor.GUI_WIDGETS_PARENT_ID; + private static final String PARENT = LakeCatalogEditor.GUI_WIDGETS_PARENT_ID; /** * Fill catalog fields from a named scenario (Iceberg Hadoop/REST, Hive, Glue, …). Mutates {@code @@ -80,11 +80,11 @@ public class SparkCatalog extends HopMetadataBase implements Serializable, IHopM label = "i18n::SparkCatalog.LoadTemplate.Label", toolTip = "i18n::SparkCatalog.LoadTemplate.ToolTip") public void loadCatalogTemplate(Object object) { - if (!(object instanceof SparkCatalog catalog)) { + if (!(object instanceof LakeCatalog catalog)) { return; } Shell shell = HopGui.getInstance().getShell(); - String[] labels = SparkCatalogTemplate.displayNames(); + String[] labels = LakeCatalogTemplate.displayNames(); EnterSelectionDialog dialog = new EnterSelectionDialog( shell, @@ -96,11 +96,11 @@ public void loadCatalogTemplate(Object object) { if (choice == null) { return; } - SparkCatalogTemplate template = SparkCatalogTemplate.fromDisplayName(choice); + LakeCatalogTemplate template = LakeCatalogTemplate.fromDisplayName(choice); if (template == null) { return; } - if (SparkCatalogTemplate.looksCustomized(catalog)) { + if (LakeCatalogTemplate.looksCustomized(catalog)) { MessageBox box = new MessageBox(shell, SWT.YES | SWT.NO | SWT.ICON_QUESTION); box.setText(BaseMessages.getString(PKG, "SparkCatalog.LoadTemplate.Confirm.Title")); box.setMessage( @@ -179,9 +179,9 @@ public void loadCatalogTemplate(Object object) { @HopMetadataProperty private String confExtra; - public SparkCatalog() { + public LakeCatalog() { this.catalogType = TYPE_HADOOP; - this.implementation = SparkLakeFormats.ICEBERG_CATALOG; + this.implementation = LakeFormats.ICEBERG_CATALOG; } /** Combo values for catalog type widget (GuiCompositeWidgets signature). */ diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalogEditor.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalogEditor.java similarity index 94% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalogEditor.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalogEditor.java index 3103f5a6944..f0a6dc320c2 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/SparkCatalogEditor.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/LakeCatalogEditor.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.metadata; +package org.apache.hop.lakehouse.metadata; import org.apache.hop.core.Const; import org.apache.hop.core.gui.plugin.GuiPlugin; @@ -38,9 +38,9 @@ import org.eclipse.swt.widgets.Label; @GuiPlugin(description = "Editor for Spark catalog metadata") -public class SparkCatalogEditor extends MetadataEditor { +public class LakeCatalogEditor extends MetadataEditor { - private static final Class PKG = SparkCatalog.class; + private static final Class PKG = LakeCatalog.class; public static final String GUI_WIDGETS_PARENT_ID = "SparkCatalogEditor-GuiWidgetsParent"; @@ -48,8 +48,8 @@ public class SparkCatalogEditor extends MetadataEditor { private Composite wWidgetsComposite; private GuiCompositeWidgets guiCompositeWidgets; - public SparkCatalogEditor( - HopGui hopGui, MetadataManager manager, SparkCatalog metadata) { + public LakeCatalogEditor( + HopGui hopGui, MetadataManager manager, LakeCatalog metadata) { super(hopGui, manager, metadata); } @@ -138,13 +138,13 @@ public void afterButtonPressed(Object sourceObject) { @Override public void setWidgetsContent() { - SparkCatalog meta = this.getMetadata(); + LakeCatalog meta = this.getMetadata(); wName.setText(Const.NVL(meta.getName(), "")); guiCompositeWidgets.setWidgetsContents(metadata, wWidgetsComposite, GUI_WIDGETS_PARENT_ID); } @Override - public void getWidgetsContent(SparkCatalog meta) { + public void getWidgetsContent(LakeCatalog meta) { meta.setName(wName.getText()); guiCompositeWidgets.getWidgetsContents(metadata, GUI_WIDGETS_PARENT_ID); } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplate.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplate.java similarity index 85% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplate.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplate.java index adb989395d4..92711ed4afd 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplate.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplate.java @@ -15,28 +15,28 @@ * limitations under the License. */ -package org.apache.hop.spark.metadata.template; +package org.apache.hop.lakehouse.metadata.template; import java.util.Arrays; import java.util.Objects; import org.apache.commons.lang3.StringUtils; -import org.apache.hop.spark.metadata.SparkCatalog; -import org.apache.hop.spark.table.SparkLakeFormats; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.metadata.LakeCatalog; /** - * Named presets for {@link SparkCatalog} fields. Pure data — no SWT. Apply via {@link - * #applyTo(SparkCatalog)}. + * Named presets for {@link LakeCatalog} fields. Pure data — no SWT. Apply via {@link + * #applyTo(LakeCatalog)}. * *

Advanced presets include a commented {@code # docs: …} line in conf extra (ignored by {@code * SparkCatalogApplier}) so operators can open vendor documentation without leaving Hop. */ -public enum SparkCatalogTemplate { +public enum LakeCatalogTemplate { ICEBERG_HADOOP_LOCAL( "Iceberg Hadoop (local)", "Named Iceberg Hadoop catalog on a local warehouse path. Use TABLE mode as lake.db.table.", "lake", - SparkCatalog.TYPE_HADOOP, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_HADOOP, + LakeFormats.ICEBERG_CATALOG, "file:///tmp/hop-warehouse", "", confWithDocs(Docs.ICEBERG_SPARK, "")), @@ -44,8 +44,8 @@ public enum SparkCatalogTemplate { "Iceberg Hadoop (object store)", "Iceberg Hadoop warehouse on object storage (replace bucket). Credentials via env/Hadoop conf.", "lake", - SparkCatalog.TYPE_HADOOP, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_HADOOP, + LakeFormats.ICEBERG_CATALOG, "s3a://bucket/warehouse", "", confWithDocs(Docs.ICEBERG_AWS, "io-impl=org.apache.iceberg.aws.s3.S3FileIO")), @@ -53,8 +53,8 @@ public enum SparkCatalogTemplate { "Iceberg REST", "Iceberg REST catalog. Replace the URI with your catalog service endpoint.", "lake", - SparkCatalog.TYPE_REST, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_REST, + LakeFormats.ICEBERG_CATALOG, "", "https://catalog.example.com/v1", confWithDocs(Docs.ICEBERG_REST, "")), @@ -62,8 +62,8 @@ public enum SparkCatalogTemplate { "Iceberg REST (authenticated)", "Iceberg REST with token auth. Put the secret in Credential (maps to catalog .token).", "lake", - SparkCatalog.TYPE_REST, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_REST, + LakeFormats.ICEBERG_CATALOG, "", "https://catalog.example.com/v1", confWithDocs( @@ -73,8 +73,8 @@ public enum SparkCatalogTemplate { "Hive Metastore (advanced)", "Iceberg via Hive Metastore. Requires Hive/Iceberg deps on the cluster (not packaged by default).", "hive", - SparkCatalog.TYPE_HIVE, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_HIVE, + LakeFormats.ICEBERG_CATALOG, "", "", confWithDocs(Docs.ICEBERG_HIVE, "uri=thrift://hive-metastore:9083")), @@ -82,8 +82,8 @@ public enum SparkCatalogTemplate { "AWS Glue (advanced)", "Iceberg via AWS Glue. Requires AWS/Iceberg deps and IAM on the cluster (not packaged by default).", "glue", - SparkCatalog.TYPE_GLUE, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_GLUE, + LakeFormats.ICEBERG_CATALOG, "s3a://bucket/warehouse", "", confWithDocs( @@ -93,8 +93,8 @@ public enum SparkCatalogTemplate { "Nessie (advanced)", "Iceberg Nessie catalog. Requires Nessie/Iceberg deps; replace URI, ref, and warehouse.", "nessie", - SparkCatalog.TYPE_CUSTOM, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_CUSTOM, + LakeFormats.ICEBERG_CATALOG, "file:///tmp/nessie-warehouse", "http://localhost:19120/api/v1", confWithDocs( @@ -107,8 +107,8 @@ public enum SparkCatalogTemplate { "Databricks Unity Catalog (advanced)", "Skeleton for Unity Catalog on Databricks. Prefer workspace-managed conf; adjust URI/auth for your env.", "unity", - SparkCatalog.TYPE_CUSTOM, - SparkLakeFormats.ICEBERG_CATALOG, + LakeCatalog.TYPE_CUSTOM, + LakeFormats.ICEBERG_CATALOG, "", "", confWithDocs( @@ -121,8 +121,8 @@ public enum SparkCatalogTemplate { "Delta named catalog (advanced)", "Named DeltaCatalog (not spark_catalog). Prefer spark_catalog for Delta when coexisting with Iceberg.", "delta_cat", - SparkCatalog.TYPE_CUSTOM, - SparkLakeFormats.DELTA_CATALOG, + LakeCatalog.TYPE_CUSTOM, + LakeFormats.DELTA_CATALOG, "", "", confWithDocs( @@ -154,7 +154,7 @@ public interface Docs { private final String uri; private final String confExtra; - SparkCatalogTemplate( + LakeCatalogTemplate( String displayName, String description, String catalogName, @@ -195,14 +195,14 @@ public String getDescription() { /** Labels for selection dialogs (same order as {@link #values()}). */ public static String[] displayNames() { - return Arrays.stream(values()).map(SparkCatalogTemplate::getDisplayName).toArray(String[]::new); + return Arrays.stream(values()).map(LakeCatalogTemplate::getDisplayName).toArray(String[]::new); } - public static SparkCatalogTemplate fromDisplayName(String name) { + public static LakeCatalogTemplate fromDisplayName(String name) { if (name == null) { return null; } - for (SparkCatalogTemplate t : values()) { + for (LakeCatalogTemplate t : values()) { if (t.displayName.equals(name)) { return t; } @@ -214,7 +214,7 @@ public static SparkCatalogTemplate fromDisplayName(String name) { * Apply this template onto {@code catalog}. Does not change the Hop metadata object name; leaves * credential empty (never put secrets in presets). */ - public void applyTo(SparkCatalog catalog) { + public void applyTo(LakeCatalog catalog) { Objects.requireNonNull(catalog, "catalog"); catalog.setCatalogName(catalogName); catalog.setCatalogType(catalogType); @@ -226,14 +226,14 @@ public void applyTo(SparkCatalog catalog) { } /** - * True if the catalog looks customized relative to a fresh {@link SparkCatalog} default (used for + * True if the catalog looks customized relative to a fresh {@link LakeCatalog} default (used for * overwrite confirmation). */ - public static boolean looksCustomized(SparkCatalog catalog) { + public static boolean looksCustomized(LakeCatalog catalog) { if (catalog == null) { return false; } - SparkCatalog def = new SparkCatalog(); + LakeCatalog def = new LakeCatalog(); if (StringUtils.isNotEmpty(catalog.getCatalogName())) { return true; } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInput.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInput.java similarity index 84% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInput.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInput.java index 741de475a1c..90ad3984e47 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInput.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInput.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.exception.HopException; import org.apache.hop.pipeline.Pipeline; @@ -26,13 +26,12 @@ /** * Metadata-only on Local engine. On the native Spark engine this becomes a lake table Dataset read. */ -public class SparkLakeTableInput - extends BaseTransform { +public class LakeTableInput extends BaseTransform { - public SparkLakeTableInput( + public LakeTableInput( TransformMeta transformMeta, - SparkLakeTableInputMeta meta, - SparkLakeTableInputData data, + LakeTableInputMeta meta, + LakeTableInputData data, int copyNr, PipelineMeta pipelineMeta, Pipeline pipeline) { diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputData.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputData.java similarity index 84% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputData.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputData.java index f89ee00e6ed..58fa976d845 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputData.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputData.java @@ -15,13 +15,13 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.pipeline.transform.BaseTransformData; import org.apache.hop.pipeline.transform.ITransformData; -public class SparkLakeTableInputData extends BaseTransformData implements ITransformData { - public SparkLakeTableInputData() { +public class LakeTableInputData extends BaseTransformData implements ITransformData { + public LakeTableInputData() { super(); } } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputDialog.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputDialog.java similarity index 91% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputDialog.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputDialog.java index d6b2a1d65cf..7a5992e5add 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputDialog.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputDialog.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import java.util.ArrayList; import java.util.List; @@ -24,9 +24,9 @@ import org.apache.hop.core.util.Utils; import org.apache.hop.core.variables.IVariables; import org.apache.hop.i18n.BaseMessages; +import org.apache.hop.lakehouse.LakeField; +import org.apache.hop.lakehouse.LakeFormats; import org.apache.hop.pipeline.PipelineMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.transforms.io.SparkField; import org.apache.hop.ui.core.PropsUi; import org.apache.hop.ui.core.dialog.BaseDialog; import org.apache.hop.ui.core.widget.ColumnInfo; @@ -46,10 +46,10 @@ import org.eclipse.swt.widgets.TableItem; import org.eclipse.swt.widgets.Text; -public class SparkLakeTableInputDialog extends BaseTransformDialog { - private static final Class PKG = SparkLakeTableInputMeta.class; +public class LakeTableInputDialog extends BaseTransformDialog { + private static final Class PKG = LakeTableInputMeta.class; - private final SparkLakeTableInputMeta input; + private final LakeTableInputMeta input; private CCombo wFormat; private CCombo wIdentifierMode; @@ -62,10 +62,10 @@ public class SparkLakeTableInputDialog extends BaseTransformDialog { private Text wExtraOptions; private TableView wFields; - public SparkLakeTableInputDialog( + public LakeTableInputDialog( Shell parent, IVariables variables, - SparkLakeTableInputMeta transformMeta, + LakeTableInputMeta transformMeta, PipelineMeta pipelineMeta) { super(parent, variables, transformMeta, pipelineMeta); this.input = transformMeta; @@ -110,13 +110,13 @@ public String open() { last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableInputDialog.Format", true); wFormat = (CCombo) last; - wFormat.setItems(new String[] {SparkLakeFormats.FORMAT_DELTA, SparkLakeFormats.FORMAT_ICEBERG}); + wFormat.setItems(new String[] {LakeFormats.FORMAT_DELTA, LakeFormats.FORMAT_ICEBERG}); last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableInputDialog.IdentifierMode", true); wIdentifierMode = (CCombo) last; wIdentifierMode.setItems( - new String[] {SparkLakeTableInputMeta.MODE_PATH, SparkLakeTableInputMeta.MODE_TABLE}); + new String[] {LakeTableInputMeta.MODE_PATH, LakeTableInputMeta.MODE_TABLE}); last = labeledTextVar(lsMod, middle, margin, last, "SparkLakeTableInputDialog.TablePath"); wTablePath = (TextVar) last; @@ -134,9 +134,9 @@ public String open() { wTimeTravelType = (CCombo) last; wTimeTravelType.setItems( new String[] { - SparkLakeTableInputMeta.TIME_TRAVEL_NONE, - SparkLakeTableInputMeta.TIME_TRAVEL_VERSION, - SparkLakeTableInputMeta.TIME_TRAVEL_TIMESTAMP + LakeTableInputMeta.TIME_TRAVEL_NONE, + LakeTableInputMeta.TIME_TRAVEL_VERSION, + LakeTableInputMeta.TIME_TRAVEL_TIMESTAMP }); last = @@ -280,20 +280,19 @@ private TextVar labeledTextVar( private void getData() { wTransformName.setText(Const.NVL(transformName, "")); - wFormat.setText(Const.NVL(input.getFormat(), SparkLakeFormats.FORMAT_DELTA)); - wIdentifierMode.setText( - Const.NVL(input.getIdentifierMode(), SparkLakeTableInputMeta.MODE_PATH)); + wFormat.setText(Const.NVL(input.getFormat(), LakeFormats.FORMAT_DELTA)); + wIdentifierMode.setText(Const.NVL(input.getIdentifierMode(), LakeTableInputMeta.MODE_PATH)); wTablePath.setText(Const.NVL(input.getTablePath(), "")); wTableIdentifier.setText(Const.NVL(input.getTableIdentifier(), "")); wCatalogMetadataName.setText(Const.NVL(input.getCatalogMetadataName(), "")); wTimeTravelType.setText( - Const.NVL(input.getTimeTravelType(), SparkLakeTableInputMeta.TIME_TRAVEL_NONE)); + Const.NVL(input.getTimeTravelType(), LakeTableInputMeta.TIME_TRAVEL_NONE)); wTimeTravelVersion.setText(Const.NVL(input.getTimeTravelVersion(), "")); wTimeTravelTimestamp.setText(Const.NVL(input.getTimeTravelTimestamp(), "")); wExtraOptions.setText(Const.NVL(input.getExtraOptions(), "")); if (input.getFields() != null) { int i = 0; - for (SparkField f : input.getFields()) { + for (LakeField f : input.getFields()) { TableItem item = wFields.table.getItem(i); if (item == null) { item = new TableItem(wFields.table, SWT.NONE); @@ -331,10 +330,10 @@ private void ok() { input.setTimeTravelVersion(wTimeTravelVersion.getText()); input.setTimeTravelTimestamp(wTimeTravelTimestamp.getText()); input.setExtraOptions(wExtraOptions.getText()); - List fields = new ArrayList<>(); + List fields = new ArrayList<>(); for (int i = 0; i < wFields.nrNonEmpty(); i++) { TableItem item = wFields.getNonEmpty(i); - SparkField f = new SparkField(); + LakeField f = new LakeField(); f.setName(item.getText(1)); f.setHopType(item.getText(2)); f.setLength(Const.toInt(item.getText(3), -1)); diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMeta.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputMeta.java similarity index 85% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMeta.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputMeta.java index 78df4b7c1b2..7afc107068f 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMeta.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableInputMeta.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import java.util.ArrayList; import java.util.List; @@ -26,27 +26,26 @@ import org.apache.hop.core.exception.HopTransformException; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.LakeField; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.LakehouseConst; import org.apache.hop.metadata.api.HopMetadataProperty; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.transform.BaseTransformMeta; import org.apache.hop.pipeline.transform.TransformMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.transforms.io.SparkField; -import org.apache.hop.spark.util.SparkConst; @Transform( - id = SparkConst.SPARK_LAKE_TABLE_INPUT_PLUGIN_ID, + id = LakehouseConst.LAKE_TABLE_INPUT_PLUGIN_ID, name = "i18n::SparkLakeTableInput.Name", description = "i18n::SparkLakeTableInput.Description", image = "spark-lake-table-input.svg", categoryDescription = "i18n:org.apache.hop.pipeline.transform:BaseTransform.Category.BigData", keywords = "i18n::SparkLakeTableInput.Keyword", documentationUrl = "/pipeline/transforms/spark-lake-table-input.html", - supportedEngines = {SparkConst.PLUGIN_ID}) + supportedEngines = {LakehouseConst.SPARK_ENGINE_ID}) @Getter @Setter -public class SparkLakeTableInputMeta - extends BaseTransformMeta { +public class LakeTableInputMeta extends BaseTransformMeta { public static final String MODE_PATH = "PATH"; public static final String MODE_TABLE = "TABLE"; @@ -55,9 +54,9 @@ public class SparkLakeTableInputMeta public static final String TIME_TRAVEL_VERSION = "VERSION"; public static final String TIME_TRAVEL_TIMESTAMP = "TIMESTAMP"; - /** {@link SparkLakeFormats#FORMAT_DELTA} or {@link SparkLakeFormats#FORMAT_ICEBERG} */ + /** {@link LakeFormats#FORMAT_DELTA} or {@link LakeFormats#FORMAT_ICEBERG} */ @HopMetadataProperty(key = "format", injectionKey = "FORMAT") - private String format = SparkLakeFormats.FORMAT_DELTA; + private String format = LakeFormats.FORMAT_DELTA; /** PATH (v1 primary) or TABLE (catalog — later PRs) */ @HopMetadataProperty(key = "identifier_mode", injectionKey = "IDENTIFIER_MODE") @@ -70,7 +69,7 @@ public class SparkLakeTableInputMeta @HopMetadataProperty(key = "table_identifier", injectionKey = "TABLE_IDENTIFIER") private String tableIdentifier; - /** Hop SparkCatalog metadata name when mode is TABLE. */ + /** Hop LakeCatalog metadata name when mode is TABLE. */ @HopMetadataProperty(key = "catalog_metadata_name", injectionKey = "CATALOG_METADATA_NAME") private String catalogMetadataName; @@ -101,9 +100,9 @@ public class SparkLakeTableInputMeta key = "field", injectionGroupKey = "FIELDS", injectionGroupDescription = "SparkLakeTableInput.Injection.Group.Fields") - private List fields = new ArrayList<>(); + private List fields = new ArrayList<>(); - public SparkLakeTableInputMeta() { + public LakeTableInputMeta() { super(); } @@ -119,7 +118,7 @@ public boolean canStartWithoutInput() { @Override public String getDialogClassName() { - return SparkLakeTableInputDialog.class.getName(); + return LakeTableInputDialog.class.getName(); } @Override @@ -136,7 +135,7 @@ public void getFields( return; } try { - for (SparkField field : fields) { + for (LakeField field : fields) { if (field.getName() != null && !field.getName().isEmpty()) { inputRowMeta.addValueMeta(field.createValueMeta()); } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenance.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenance.java similarity index 83% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenance.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenance.java index 894de666d32..6cc28e20bca 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenance.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenance.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.exception.HopException; import org.apache.hop.pipeline.Pipeline; @@ -27,13 +27,13 @@ * Metadata-only on Local engine. On native Spark this runs OPTIMIZE / VACUUM / expire / DELETE as a * zero-input action sink. */ -public class SparkLakeTableMaintenance - extends BaseTransform { +public class LakeTableMaintenance + extends BaseTransform { - public SparkLakeTableMaintenance( + public LakeTableMaintenance( TransformMeta transformMeta, - SparkLakeTableMaintenanceMeta meta, - SparkLakeTableMaintenanceData data, + LakeTableMaintenanceMeta meta, + LakeTableMaintenanceData data, int copyNr, PipelineMeta pipelineMeta, Pipeline pipeline) { diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputData.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceData.java similarity index 86% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputData.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceData.java index f756b321288..adf5bc09a89 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputData.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceData.java @@ -15,13 +15,13 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.pipeline.transform.BaseTransformData; import org.apache.hop.pipeline.transform.ITransformData; -public class SparkLakeTableOutputData extends BaseTransformData implements ITransformData { - public SparkLakeTableOutputData() { +public class LakeTableMaintenanceData extends BaseTransformData implements ITransformData { + public LakeTableMaintenanceData() { super(); } } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceDialog.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceDialog.java similarity index 88% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceDialog.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceDialog.java index 8fce987c9d2..80d46154c64 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceDialog.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceDialog.java @@ -15,15 +15,14 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.Const; import org.apache.hop.core.util.Utils; import org.apache.hop.core.variables.IVariables; import org.apache.hop.i18n.BaseMessages; +import org.apache.hop.lakehouse.LakeFormats; import org.apache.hop.pipeline.PipelineMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.table.SparkMaintenanceSqlBuilder; import org.apache.hop.ui.core.PropsUi; import org.apache.hop.ui.core.dialog.BaseDialog; import org.apache.hop.ui.core.widget.TextVar; @@ -39,10 +38,10 @@ import org.eclipse.swt.widgets.Label; import org.eclipse.swt.widgets.Shell; -public class SparkLakeTableMaintenanceDialog extends BaseTransformDialog { - private static final Class PKG = SparkLakeTableMaintenanceMeta.class; +public class LakeTableMaintenanceDialog extends BaseTransformDialog { + private static final Class PKG = LakeTableMaintenanceMeta.class; - private final SparkLakeTableMaintenanceMeta input; + private final LakeTableMaintenanceMeta input; private CCombo wFormat; private CCombo wIdentifierMode; @@ -56,10 +55,10 @@ public class SparkLakeTableMaintenanceDialog extends BaseTransformDialog { private TextVar wZOrderColumns; private Button wAcknowledgeDestructive; - public SparkLakeTableMaintenanceDialog( + public LakeTableMaintenanceDialog( Shell parent, IVariables variables, - SparkLakeTableMaintenanceMeta transformMeta, + LakeTableMaintenanceMeta transformMeta, PipelineMeta pipelineMeta) { super(parent, variables, transformMeta, pipelineMeta); this.input = transformMeta; @@ -104,13 +103,13 @@ public String open() { last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableMaintenanceDialog.Format"); wFormat = (CCombo) last; - wFormat.setItems(new String[] {SparkLakeFormats.FORMAT_DELTA, SparkLakeFormats.FORMAT_ICEBERG}); + wFormat.setItems(new String[] {LakeFormats.FORMAT_DELTA, LakeFormats.FORMAT_ICEBERG}); last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableMaintenanceDialog.IdentifierMode"); wIdentifierMode = (CCombo) last; wIdentifierMode.setItems( - new String[] {SparkLakeTableInputMeta.MODE_PATH, SparkLakeTableInputMeta.MODE_TABLE}); + new String[] {LakeTableInputMeta.MODE_PATH, LakeTableInputMeta.MODE_TABLE}); last = labeledTextVar(lsMod, middle, margin, last, "SparkLakeTableMaintenanceDialog.TablePath"); wTablePath = (TextVar) last; @@ -129,11 +128,11 @@ public String open() { wOperation = (CCombo) last; wOperation.setItems( new String[] { - SparkMaintenanceSqlBuilder.OP_OPTIMIZE, - SparkMaintenanceSqlBuilder.OP_VACUUM, - SparkMaintenanceSqlBuilder.OP_EXPIRE_SNAPSHOTS, - SparkMaintenanceSqlBuilder.OP_REWRITE_MANIFESTS, - SparkMaintenanceSqlBuilder.OP_DELETE_WHERE + LakeTableMaintenanceMeta.OP_OPTIMIZE, + LakeTableMaintenanceMeta.OP_VACUUM, + LakeTableMaintenanceMeta.OP_EXPIRE_SNAPSHOTS, + LakeTableMaintenanceMeta.OP_REWRITE_MANIFESTS, + LakeTableMaintenanceMeta.OP_DELETE_WHERE }); last = @@ -228,13 +227,12 @@ private TextVar labeledTextVar( private void getData() { wTransformName.setText(Const.NVL(transformName, "")); - wFormat.setText(Const.NVL(input.getFormat(), SparkLakeFormats.FORMAT_DELTA)); - wIdentifierMode.setText( - Const.NVL(input.getIdentifierMode(), SparkLakeTableInputMeta.MODE_PATH)); + wFormat.setText(Const.NVL(input.getFormat(), LakeFormats.FORMAT_DELTA)); + wIdentifierMode.setText(Const.NVL(input.getIdentifierMode(), LakeTableInputMeta.MODE_PATH)); wTablePath.setText(Const.NVL(input.getTablePath(), "")); wTableIdentifier.setText(Const.NVL(input.getTableIdentifier(), "")); wCatalogMetadataName.setText(Const.NVL(input.getCatalogMetadataName(), "")); - wOperation.setText(Const.NVL(input.getOperation(), SparkMaintenanceSqlBuilder.OP_OPTIMIZE)); + wOperation.setText(Const.NVL(input.getOperation(), LakeTableMaintenanceMeta.OP_OPTIMIZE)); wRetentionHours.setText(Const.NVL(input.getRetentionHours(), "")); wRetainLast.setText(Const.NVL(input.getRetainLast(), "1")); wWhereClause.setText(Const.NVL(input.getWhereClause(), "")); diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMeta.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMeta.java similarity index 78% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMeta.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMeta.java index e3632ec1977..98a5c35f384 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMeta.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMeta.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import lombok.Getter; import lombok.Setter; @@ -23,33 +23,38 @@ import org.apache.hop.core.exception.HopTransformException; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.LakehouseConst; import org.apache.hop.metadata.api.HopMetadataProperty; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.transform.BaseTransformMeta; import org.apache.hop.pipeline.transform.TransformMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.table.SparkMaintenanceSqlBuilder; -import org.apache.hop.spark.util.SparkConst; @Transform( - id = SparkConst.SPARK_LAKE_TABLE_MAINTENANCE_PLUGIN_ID, + id = LakehouseConst.LAKE_TABLE_MAINTENANCE_PLUGIN_ID, name = "i18n::SparkLakeTableMaintenance.Name", description = "i18n::SparkLakeTableMaintenance.Description", image = "spark-lake-table-maintenance.svg", categoryDescription = "i18n:org.apache.hop.pipeline.transform:BaseTransform.Category.BigData", keywords = "i18n::SparkLakeTableMaintenance.Keyword", documentationUrl = "/pipeline/transforms/spark-lake-table-maintenance.html", - supportedEngines = {SparkConst.PLUGIN_ID}) + supportedEngines = {LakehouseConst.SPARK_ENGINE_ID}) @Getter @Setter -public class SparkLakeTableMaintenanceMeta - extends BaseTransformMeta { +public class LakeTableMaintenanceMeta + extends BaseTransformMeta { + + public static final String OP_OPTIMIZE = "OPTIMIZE"; + public static final String OP_VACUUM = "VACUUM"; + public static final String OP_EXPIRE_SNAPSHOTS = "EXPIRE_SNAPSHOTS"; + public static final String OP_REWRITE_MANIFESTS = "REWRITE_MANIFESTS"; + public static final String OP_DELETE_WHERE = "DELETE_WHERE"; @HopMetadataProperty(key = "format", injectionKey = "FORMAT") - private String format = SparkLakeFormats.FORMAT_DELTA; + private String format = LakeFormats.FORMAT_DELTA; @HopMetadataProperty(key = "identifier_mode", injectionKey = "IDENTIFIER_MODE") - private String identifierMode = SparkLakeTableInputMeta.MODE_PATH; + private String identifierMode = LakeTableInputMeta.MODE_PATH; @HopMetadataProperty(key = "table_path", injectionKey = "TABLE_PATH") private String tablePath; @@ -60,9 +65,9 @@ public class SparkLakeTableMaintenanceMeta @HopMetadataProperty(key = "catalog_metadata_name", injectionKey = "CATALOG_METADATA_NAME") private String catalogMetadataName; - /** {@link SparkMaintenanceSqlBuilder} operation constants. */ + /** The OP_* operation constants above. */ @HopMetadataProperty(key = "operation", injectionKey = "OPERATION") - private String operation = SparkMaintenanceSqlBuilder.OP_OPTIMIZE; + private String operation = LakeTableMaintenanceMeta.OP_OPTIMIZE; /** Required for VACUUM / EXPIRE_SNAPSHOTS (hours). No silent default. */ @HopMetadataProperty(key = "retention_hours", injectionKey = "RETENTION_HOURS") @@ -87,13 +92,13 @@ public class SparkLakeTableMaintenanceMeta @HopMetadataProperty(key = "acknowledge_destructive", injectionKey = "ACKNOWLEDGE_DESTRUCTIVE") private boolean acknowledgeDestructive; - public SparkLakeTableMaintenanceMeta() { + public LakeTableMaintenanceMeta() { super(); } @Override public String getDialogClassName() { - return SparkLakeTableMaintenanceDialog.class.getName(); + return LakeTableMaintenanceDialog.class.getName(); } @Override diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMerge.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMerge.java similarity index 84% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMerge.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMerge.java index 5de744c18fb..bb022e9f137 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMerge.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMerge.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.exception.HopException; import org.apache.hop.pipeline.Pipeline; @@ -27,13 +27,12 @@ * Metadata-only on Local engine. On native Spark this becomes a MERGE INTO action against a lake * table. */ -public class SparkLakeTableMerge - extends BaseTransform { +public class LakeTableMerge extends BaseTransform { - public SparkLakeTableMerge( + public LakeTableMerge( TransformMeta transformMeta, - SparkLakeTableMergeMeta meta, - SparkLakeTableMergeData data, + LakeTableMergeMeta meta, + LakeTableMergeData data, int copyNr, PipelineMeta pipelineMeta, Pipeline pipeline) { diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeData.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeData.java similarity index 84% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeData.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeData.java index f8c59640965..a59aeed45e6 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeData.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeData.java @@ -15,13 +15,13 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.pipeline.transform.BaseTransformData; import org.apache.hop.pipeline.transform.ITransformData; -public class SparkLakeTableMergeData extends BaseTransformData implements ITransformData { - public SparkLakeTableMergeData() { +public class LakeTableMergeData extends BaseTransformData implements ITransformData { + public LakeTableMergeData() { super(); } } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeDialog.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeDialog.java similarity index 87% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeDialog.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeDialog.java index 6768b4a98d0..c384e168fbd 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeDialog.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeDialog.java @@ -15,15 +15,14 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.Const; import org.apache.hop.core.util.Utils; import org.apache.hop.core.variables.IVariables; import org.apache.hop.i18n.BaseMessages; +import org.apache.hop.lakehouse.LakeFormats; import org.apache.hop.pipeline.PipelineMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.table.SparkMergeSqlBuilder; import org.apache.hop.ui.core.PropsUi; import org.apache.hop.ui.core.dialog.BaseDialog; import org.apache.hop.ui.core.widget.TextVar; @@ -40,10 +39,10 @@ import org.eclipse.swt.widgets.Shell; import org.eclipse.swt.widgets.Text; -public class SparkLakeTableMergeDialog extends BaseTransformDialog { - private static final Class PKG = SparkLakeTableMergeMeta.class; +public class LakeTableMergeDialog extends BaseTransformDialog { + private static final Class PKG = LakeTableMergeMeta.class; - private final SparkLakeTableMergeMeta input; + private final LakeTableMergeMeta input; private CCombo wFormat; private CCombo wIdentifierMode; @@ -56,10 +55,10 @@ public class SparkLakeTableMergeDialog extends BaseTransformDialog { private CCombo wNotMatchedBySourceAction; private Text wRawMergeSql; - public SparkLakeTableMergeDialog( + public LakeTableMergeDialog( Shell parent, IVariables variables, - SparkLakeTableMergeMeta transformMeta, + LakeTableMergeMeta transformMeta, PipelineMeta pipelineMeta) { super(parent, variables, transformMeta, pipelineMeta); this.input = transformMeta; @@ -104,12 +103,12 @@ public String open() { last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableMergeDialog.Format"); wFormat = (CCombo) last; - wFormat.setItems(new String[] {SparkLakeFormats.FORMAT_DELTA, SparkLakeFormats.FORMAT_ICEBERG}); + wFormat.setItems(new String[] {LakeFormats.FORMAT_DELTA, LakeFormats.FORMAT_ICEBERG}); last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableMergeDialog.IdentifierMode"); wIdentifierMode = (CCombo) last; wIdentifierMode.setItems( - new String[] {SparkLakeTableInputMeta.MODE_PATH, SparkLakeTableInputMeta.MODE_TABLE}); + new String[] {LakeTableInputMeta.MODE_PATH, LakeTableInputMeta.MODE_TABLE}); last = labeledTextVar(lsMod, middle, margin, last, "SparkLakeTableMergeDialog.TablePath"); wTablePath = (TextVar) last; @@ -129,16 +128,16 @@ public String open() { wMatchedAction = (CCombo) last; wMatchedAction.setItems( new String[] { - SparkMergeSqlBuilder.MATCHED_UPDATE_ALL, - SparkMergeSqlBuilder.MATCHED_DELETE, - SparkMergeSqlBuilder.MATCHED_NONE + LakeTableMergeMeta.MATCHED_UPDATE_ALL, + LakeTableMergeMeta.MATCHED_DELETE, + LakeTableMergeMeta.MATCHED_NONE }); last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableMergeDialog.NotMatchedAction"); wNotMatchedAction = (CCombo) last; wNotMatchedAction.setItems( new String[] { - SparkMergeSqlBuilder.NOT_MATCHED_INSERT_ALL, SparkMergeSqlBuilder.NOT_MATCHED_NONE + LakeTableMergeMeta.NOT_MATCHED_INSERT_ALL, LakeTableMergeMeta.NOT_MATCHED_NONE }); last = @@ -147,8 +146,8 @@ public String open() { wNotMatchedBySourceAction = (CCombo) last; wNotMatchedBySourceAction.setItems( new String[] { - SparkMergeSqlBuilder.NOT_MATCHED_BY_SOURCE_NONE, - SparkMergeSqlBuilder.NOT_MATCHED_BY_SOURCE_DELETE + LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_NONE, + LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_DELETE }); Label wlRaw = new Label(shell, SWT.RIGHT); @@ -233,20 +232,19 @@ private TextVar labeledTextVar( private void getData() { wTransformName.setText(Const.NVL(transformName, "")); - wFormat.setText(Const.NVL(input.getFormat(), SparkLakeFormats.FORMAT_DELTA)); - wIdentifierMode.setText( - Const.NVL(input.getIdentifierMode(), SparkLakeTableInputMeta.MODE_PATH)); + wFormat.setText(Const.NVL(input.getFormat(), LakeFormats.FORMAT_DELTA)); + wIdentifierMode.setText(Const.NVL(input.getIdentifierMode(), LakeTableInputMeta.MODE_PATH)); wTablePath.setText(Const.NVL(input.getTablePath(), "")); wTableIdentifier.setText(Const.NVL(input.getTableIdentifier(), "")); wCatalogMetadataName.setText(Const.NVL(input.getCatalogMetadataName(), "")); wMergeCondition.setText(Const.NVL(input.getMergeCondition(), "t.id = s.id")); wMatchedAction.setText( - Const.NVL(input.getMatchedAction(), SparkMergeSqlBuilder.MATCHED_UPDATE_ALL)); + Const.NVL(input.getMatchedAction(), LakeTableMergeMeta.MATCHED_UPDATE_ALL)); wNotMatchedAction.setText( - Const.NVL(input.getNotMatchedAction(), SparkMergeSqlBuilder.NOT_MATCHED_INSERT_ALL)); + Const.NVL(input.getNotMatchedAction(), LakeTableMergeMeta.NOT_MATCHED_INSERT_ALL)); wNotMatchedBySourceAction.setText( Const.NVL( - input.getNotMatchedBySourceAction(), SparkMergeSqlBuilder.NOT_MATCHED_BY_SOURCE_NONE)); + input.getNotMatchedBySourceAction(), LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_NONE)); wRawMergeSql.setText(Const.NVL(input.getRawMergeSql(), "")); wTransformName.selectAll(); wTransformName.setFocus(); diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMeta.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMeta.java similarity index 69% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMeta.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMeta.java index c2bbcd95da8..f790d21cd6c 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMeta.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMeta.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import lombok.Getter; import lombok.Setter; @@ -23,33 +23,41 @@ import org.apache.hop.core.exception.HopTransformException; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.LakehouseConst; import org.apache.hop.metadata.api.HopMetadataProperty; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.transform.BaseTransformMeta; import org.apache.hop.pipeline.transform.TransformMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.table.SparkMergeSqlBuilder; -import org.apache.hop.spark.util.SparkConst; @Transform( - id = SparkConst.SPARK_LAKE_TABLE_MERGE_PLUGIN_ID, + id = LakehouseConst.LAKE_TABLE_MERGE_PLUGIN_ID, name = "i18n::SparkLakeTableMerge.Name", description = "i18n::SparkLakeTableMerge.Description", image = "spark-lake-table-merge.svg", categoryDescription = "i18n:org.apache.hop.pipeline.transform:BaseTransform.Category.BigData", keywords = "i18n::SparkLakeTableMerge.Keyword", documentationUrl = "/pipeline/transforms/spark-lake-table-merge.html", - supportedEngines = {SparkConst.PLUGIN_ID}) + supportedEngines = {LakehouseConst.SPARK_ENGINE_ID}) @Getter @Setter -public class SparkLakeTableMergeMeta - extends BaseTransformMeta { +public class LakeTableMergeMeta extends BaseTransformMeta { + + public static final String MATCHED_UPDATE_ALL = "UPDATE_ALL"; + public static final String MATCHED_DELETE = "DELETE"; + public static final String MATCHED_NONE = "NONE"; + + public static final String NOT_MATCHED_INSERT_ALL = "INSERT_ALL"; + public static final String NOT_MATCHED_NONE = "NONE"; + + public static final String NOT_MATCHED_BY_SOURCE_DELETE = "DELETE"; + public static final String NOT_MATCHED_BY_SOURCE_NONE = "NONE"; @HopMetadataProperty(key = "format", injectionKey = "FORMAT") - private String format = SparkLakeFormats.FORMAT_DELTA; + private String format = LakeFormats.FORMAT_DELTA; @HopMetadataProperty(key = "identifier_mode", injectionKey = "IDENTIFIER_MODE") - private String identifierMode = SparkLakeTableInputMeta.MODE_PATH; + private String identifierMode = LakeTableInputMeta.MODE_PATH; @HopMetadataProperty(key = "table_path", injectionKey = "TABLE_PATH") private String tablePath; @@ -64,22 +72,21 @@ public class SparkLakeTableMergeMeta @HopMetadataProperty(key = "merge_condition", injectionKey = "MERGE_CONDITION") private String mergeCondition; - /** {@link SparkMergeSqlBuilder#MATCHED_UPDATE_ALL}, DELETE, or NONE */ + /** {@link #MATCHED_UPDATE_ALL}, DELETE, or NONE */ @HopMetadataProperty(key = "matched_action", injectionKey = "MATCHED_ACTION") - private String matchedAction = SparkMergeSqlBuilder.MATCHED_UPDATE_ALL; + private String matchedAction = LakeTableMergeMeta.MATCHED_UPDATE_ALL; - /** {@link SparkMergeSqlBuilder#NOT_MATCHED_INSERT_ALL} or NONE */ + /** {@link #NOT_MATCHED_INSERT_ALL} or NONE */ @HopMetadataProperty(key = "not_matched_action", injectionKey = "NOT_MATCHED_ACTION") - private String notMatchedAction = SparkMergeSqlBuilder.NOT_MATCHED_INSERT_ALL; + private String notMatchedAction = LakeTableMergeMeta.NOT_MATCHED_INSERT_ALL; /** - * Optional {@link SparkMergeSqlBuilder#NOT_MATCHED_BY_SOURCE_DELETE} (Delta; Iceberg support - * varies). Default NONE. + * Optional {@link #NOT_MATCHED_BY_SOURCE_DELETE} (Delta; Iceberg support varies). Default NONE. */ @HopMetadataProperty( key = "not_matched_by_source_action", injectionKey = "NOT_MATCHED_BY_SOURCE_ACTION") - private String notMatchedBySourceAction = SparkMergeSqlBuilder.NOT_MATCHED_BY_SOURCE_NONE; + private String notMatchedBySourceAction = LakeTableMergeMeta.NOT_MATCHED_BY_SOURCE_NONE; /** * Advanced: full MERGE SQL. When non-empty, overrides structured fields. Operator is trusted; @@ -88,13 +95,13 @@ public class SparkLakeTableMergeMeta @HopMetadataProperty(key = "raw_merge_sql", injectionKey = "RAW_MERGE_SQL") private String rawMergeSql; - public SparkLakeTableMergeMeta() { + public LakeTableMergeMeta() { super(); } @Override public String getDialogClassName() { - return SparkLakeTableMergeDialog.class.getName(); + return LakeTableMergeDialog.class.getName(); } @Override diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutput.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutput.java similarity index 84% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutput.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutput.java index d94c2b30947..b2802895e33 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutput.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutput.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.exception.HopException; import org.apache.hop.pipeline.Pipeline; @@ -26,13 +26,12 @@ /** * Metadata-only on Local engine. On the native Spark engine this becomes a lake table write action. */ -public class SparkLakeTableOutput - extends BaseTransform { +public class LakeTableOutput extends BaseTransform { - public SparkLakeTableOutput( + public LakeTableOutput( TransformMeta transformMeta, - SparkLakeTableOutputMeta meta, - SparkLakeTableOutputData data, + LakeTableOutputMeta meta, + LakeTableOutputData data, int copyNr, PipelineMeta pipelineMeta, Pipeline pipeline) { diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceData.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputData.java similarity index 83% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceData.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputData.java index 35bcb120374..f5510ddb101 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceData.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputData.java @@ -15,13 +15,13 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.pipeline.transform.BaseTransformData; import org.apache.hop.pipeline.transform.ITransformData; -public class SparkLakeTableMaintenanceData extends BaseTransformData implements ITransformData { - public SparkLakeTableMaintenanceData() { +public class LakeTableOutputData extends BaseTransformData implements ITransformData { + public LakeTableOutputData() { super(); } } diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputDialog.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputDialog.java similarity index 89% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputDialog.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputDialog.java index bbe24a6245a..dacfab15784 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputDialog.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputDialog.java @@ -15,15 +15,14 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.Const; import org.apache.hop.core.util.Utils; import org.apache.hop.core.variables.IVariables; import org.apache.hop.i18n.BaseMessages; +import org.apache.hop.lakehouse.LakeFormats; import org.apache.hop.pipeline.PipelineMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; import org.apache.hop.ui.core.PropsUi; import org.apache.hop.ui.core.dialog.BaseDialog; import org.apache.hop.ui.core.widget.TextVar; @@ -40,10 +39,10 @@ import org.eclipse.swt.widgets.Shell; import org.eclipse.swt.widgets.Text; -public class SparkLakeTableOutputDialog extends BaseTransformDialog { - private static final Class PKG = SparkLakeTableOutputMeta.class; +public class LakeTableOutputDialog extends BaseTransformDialog { + private static final Class PKG = LakeTableOutputMeta.class; - private final SparkLakeTableOutputMeta input; + private final LakeTableOutputMeta input; private CCombo wFormat; private CCombo wIdentifierMode; @@ -55,10 +54,10 @@ public class SparkLakeTableOutputDialog extends BaseTransformDialog { private TextVar wCoalesce; private Text wExtraOptions; - public SparkLakeTableOutputDialog( + public LakeTableOutputDialog( Shell parent, IVariables variables, - SparkLakeTableOutputMeta transformMeta, + LakeTableOutputMeta transformMeta, PipelineMeta pipelineMeta) { super(parent, variables, transformMeta, pipelineMeta); this.input = transformMeta; @@ -103,12 +102,12 @@ public String open() { last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableOutputDialog.Format"); wFormat = (CCombo) last; - wFormat.setItems(new String[] {SparkLakeFormats.FORMAT_DELTA, SparkLakeFormats.FORMAT_ICEBERG}); + wFormat.setItems(new String[] {LakeFormats.FORMAT_DELTA, LakeFormats.FORMAT_ICEBERG}); last = labeledCombo(lsMod, middle, margin, last, "SparkLakeTableOutputDialog.IdentifierMode"); wIdentifierMode = (CCombo) last; wIdentifierMode.setItems( - new String[] {SparkLakeTableInputMeta.MODE_PATH, SparkLakeTableInputMeta.MODE_TABLE}); + new String[] {LakeTableInputMeta.MODE_PATH, LakeTableInputMeta.MODE_TABLE}); last = labeledTextVar(lsMod, middle, margin, last, "SparkLakeTableOutputDialog.TablePath"); wTablePath = (TextVar) last; @@ -126,10 +125,10 @@ public String open() { wSaveMode = (CCombo) last; wSaveMode.setItems( new String[] { - SparkFileOutputMeta.MODE_ERROR, - SparkFileOutputMeta.MODE_APPEND, - SparkFileOutputMeta.MODE_OVERWRITE, - SparkFileOutputMeta.MODE_IGNORE + LakeTableOutputMeta.MODE_ERROR, + LakeTableOutputMeta.MODE_APPEND, + LakeTableOutputMeta.MODE_OVERWRITE, + LakeTableOutputMeta.MODE_IGNORE }); last = labeledTextVar(lsMod, middle, margin, last, "SparkLakeTableOutputDialog.PartitionBy"); @@ -219,13 +218,12 @@ private TextVar labeledTextVar( private void getData() { wTransformName.setText(Const.NVL(transformName, "")); - wFormat.setText(Const.NVL(input.getFormat(), SparkLakeFormats.FORMAT_DELTA)); - wIdentifierMode.setText( - Const.NVL(input.getIdentifierMode(), SparkLakeTableInputMeta.MODE_PATH)); + wFormat.setText(Const.NVL(input.getFormat(), LakeFormats.FORMAT_DELTA)); + wIdentifierMode.setText(Const.NVL(input.getIdentifierMode(), LakeTableInputMeta.MODE_PATH)); wTablePath.setText(Const.NVL(input.getTablePath(), "")); wTableIdentifier.setText(Const.NVL(input.getTableIdentifier(), "")); wCatalogMetadataName.setText(Const.NVL(input.getCatalogMetadataName(), "")); - wSaveMode.setText(Const.NVL(input.getSaveMode(), SparkFileOutputMeta.MODE_ERROR)); + wSaveMode.setText(Const.NVL(input.getSaveMode(), LakeTableOutputMeta.MODE_ERROR)); wPartitionBy.setText(Const.NVL(input.getPartitionByColumns(), "")); wCoalesce.setText(Const.NVL(input.getCoalescePartitions(), "")); wExtraOptions.setText(Const.NVL(input.getExtraOptions(), "")); diff --git a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMeta.java b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMeta.java similarity index 73% rename from plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMeta.java rename to plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMeta.java index 70586b5c970..02fd1abf6e5 100644 --- a/plugins/engines/spark/src/main/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMeta.java +++ b/plugins/tech/lakehouse/src/main/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMeta.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import lombok.Getter; import lombok.Setter; @@ -23,34 +23,39 @@ import org.apache.hop.core.exception.HopTransformException; import org.apache.hop.core.row.IRowMeta; import org.apache.hop.core.variables.IVariables; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.LakehouseConst; import org.apache.hop.metadata.api.HopMetadataProperty; import org.apache.hop.metadata.api.IHopMetadataProvider; import org.apache.hop.pipeline.transform.BaseTransformMeta; import org.apache.hop.pipeline.transform.TransformMeta; -import org.apache.hop.spark.table.SparkLakeFormats; -import org.apache.hop.spark.transforms.io.SparkFileOutputMeta; -import org.apache.hop.spark.util.SparkConst; @Transform( - id = SparkConst.SPARK_LAKE_TABLE_OUTPUT_PLUGIN_ID, + id = LakehouseConst.LAKE_TABLE_OUTPUT_PLUGIN_ID, name = "i18n::SparkLakeTableOutput.Name", description = "i18n::SparkLakeTableOutput.Description", image = "spark-lake-table-output.svg", categoryDescription = "i18n:org.apache.hop.pipeline.transform:BaseTransform.Category.BigData", keywords = "i18n::SparkLakeTableOutput.Keyword", documentationUrl = "/pipeline/transforms/spark-lake-table-output.html", - supportedEngines = {SparkConst.PLUGIN_ID}) + supportedEngines = {LakehouseConst.SPARK_ENGINE_ID}) @Getter @Setter -public class SparkLakeTableOutputMeta - extends BaseTransformMeta { +public class LakeTableOutputMeta extends BaseTransformMeta { - /** {@link SparkLakeFormats#FORMAT_DELTA} or {@link SparkLakeFormats#FORMAT_ICEBERG} */ + /** Save modes, using the names of Spark's {@code SaveMode}. */ + public static final String MODE_OVERWRITE = "Overwrite"; + + public static final String MODE_APPEND = "Append"; + public static final String MODE_IGNORE = "Ignore"; + public static final String MODE_ERROR = "ErrorIfExists"; + + /** {@link LakeFormats#FORMAT_DELTA} or {@link LakeFormats#FORMAT_ICEBERG} */ @HopMetadataProperty(key = "format", injectionKey = "FORMAT") - private String format = SparkLakeFormats.FORMAT_DELTA; + private String format = LakeFormats.FORMAT_DELTA; @HopMetadataProperty(key = "identifier_mode", injectionKey = "IDENTIFIER_MODE") - private String identifierMode = SparkLakeTableInputMeta.MODE_PATH; + private String identifierMode = LakeTableInputMeta.MODE_PATH; @HopMetadataProperty(key = "table_path", injectionKey = "TABLE_PATH") private String tablePath; @@ -62,11 +67,11 @@ public class SparkLakeTableOutputMeta private String catalogMetadataName; /** - * Default {@link SparkFileOutputMeta#MODE_ERROR} (ErrorIfExists) — ACID tables must not default - * to destructive overwrite. + * Default {@link #MODE_ERROR} (ErrorIfExists) — ACID tables must not default to destructive + * overwrite. */ @HopMetadataProperty(key = "save_mode", injectionKey = "SAVE_MODE") - private String saveMode = SparkFileOutputMeta.MODE_ERROR; + private String saveMode = MODE_ERROR; @HopMetadataProperty(key = "partition_by", injectionKey = "PARTITION_BY") private String partitionByColumns; @@ -78,13 +83,13 @@ public class SparkLakeTableOutputMeta @HopMetadataProperty(key = "extra_options", injectionKey = "EXTRA_OPTIONS") private String extraOptions; - public SparkLakeTableOutputMeta() { + public LakeTableOutputMeta() { super(); } @Override public String getDialogClassName() { - return SparkLakeTableOutputDialog.class.getName(); + return LakeTableOutputDialog.class.getName(); } @Override diff --git a/plugins/engines/spark/src/main/resources/org/apache/hop/spark/metadata/messages/messages_en_US.properties b/plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/metadata/messages/messages_en_US.properties similarity index 100% rename from plugins/engines/spark/src/main/resources/org/apache/hop/spark/metadata/messages/messages_en_US.properties rename to plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/metadata/messages/messages_en_US.properties diff --git a/plugins/engines/spark/src/main/resources/org/apache/hop/spark/metadata/messages/messages_pt_BR.properties b/plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/metadata/messages/messages_pt_BR.properties similarity index 100% rename from plugins/engines/spark/src/main/resources/org/apache/hop/spark/metadata/messages/messages_pt_BR.properties rename to plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/metadata/messages/messages_pt_BR.properties diff --git a/plugins/engines/spark/src/main/resources/org/apache/hop/spark/transforms/table/messages/messages_en_US.properties b/plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/transforms/messages/messages_en_US.properties similarity index 100% rename from plugins/engines/spark/src/main/resources/org/apache/hop/spark/transforms/table/messages/messages_en_US.properties rename to plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/transforms/messages/messages_en_US.properties diff --git a/plugins/engines/spark/src/main/resources/org/apache/hop/spark/transforms/table/messages/messages_pt_BR.properties b/plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/transforms/messages/messages_pt_BR.properties similarity index 100% rename from plugins/engines/spark/src/main/resources/org/apache/hop/spark/transforms/table/messages/messages_pt_BR.properties rename to plugins/tech/lakehouse/src/main/resources/org/apache/hop/lakehouse/transforms/messages/messages_pt_BR.properties diff --git a/plugins/engines/spark/src/main/resources/spark-catalog.svg b/plugins/tech/lakehouse/src/main/resources/spark-catalog.svg similarity index 100% rename from plugins/engines/spark/src/main/resources/spark-catalog.svg rename to plugins/tech/lakehouse/src/main/resources/spark-catalog.svg diff --git a/plugins/engines/spark/src/main/resources/spark-lake-table-input.svg b/plugins/tech/lakehouse/src/main/resources/spark-lake-table-input.svg similarity index 100% rename from plugins/engines/spark/src/main/resources/spark-lake-table-input.svg rename to plugins/tech/lakehouse/src/main/resources/spark-lake-table-input.svg diff --git a/plugins/engines/spark/src/main/resources/spark-lake-table-maintenance.svg b/plugins/tech/lakehouse/src/main/resources/spark-lake-table-maintenance.svg similarity index 100% rename from plugins/engines/spark/src/main/resources/spark-lake-table-maintenance.svg rename to plugins/tech/lakehouse/src/main/resources/spark-lake-table-maintenance.svg diff --git a/plugins/engines/spark/src/main/resources/spark-lake-table-merge.svg b/plugins/tech/lakehouse/src/main/resources/spark-lake-table-merge.svg similarity index 100% rename from plugins/engines/spark/src/main/resources/spark-lake-table-merge.svg rename to plugins/tech/lakehouse/src/main/resources/spark-lake-table-merge.svg diff --git a/plugins/engines/spark/src/main/resources/spark-lake-table-output.svg b/plugins/tech/lakehouse/src/main/resources/spark-lake-table-output.svg similarity index 100% rename from plugins/engines/spark/src/main/resources/spark-lake-table-output.svg rename to plugins/tech/lakehouse/src/main/resources/spark-lake-table-output.svg diff --git a/plugins/tech/lakehouse/src/main/resources/version.xml b/plugins/tech/lakehouse/src/main/resources/version.xml new file mode 100644 index 00000000000..ee1c2377fe3 --- /dev/null +++ b/plugins/tech/lakehouse/src/main/resources/version.xml @@ -0,0 +1,19 @@ + + + +${project.version} \ No newline at end of file diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplateTest.java b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplateTest.java similarity index 53% rename from plugins/engines/spark/src/test/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplateTest.java rename to plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplateTest.java index 604e87e4d43..66fb87558b5 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/metadata/template/SparkCatalogTemplateTest.java +++ b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/metadata/template/LakeCatalogTemplateTest.java @@ -15,26 +15,26 @@ * limitations under the License. */ -package org.apache.hop.spark.metadata.template; +package org.apache.hop.lakehouse.metadata.template; import static org.junit.jupiter.api.Assertions.assertEquals; import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNotNull; import static org.junit.jupiter.api.Assertions.assertTrue; -import org.apache.hop.spark.metadata.SparkCatalog; -import org.apache.hop.spark.table.SparkLakeFormats; +import org.apache.hop.lakehouse.LakeFormats; +import org.apache.hop.lakehouse.metadata.LakeCatalog; import org.junit.jupiter.api.Test; -class SparkCatalogTemplateTest { +class LakeCatalogTemplateTest { @Test void icebergHadoopLocalSetsWarehouseAndType() { - SparkCatalog cat = new SparkCatalog(); - SparkCatalogTemplate.ICEBERG_HADOOP_LOCAL.applyTo(cat); + LakeCatalog cat = new LakeCatalog(); + LakeCatalogTemplate.ICEBERG_HADOOP_LOCAL.applyTo(cat); assertEquals("lake", cat.getCatalogName()); - assertEquals(SparkCatalog.TYPE_HADOOP, cat.getCatalogType()); - assertEquals(SparkLakeFormats.ICEBERG_CATALOG, cat.getImplementation()); + assertEquals(LakeCatalog.TYPE_HADOOP, cat.getCatalogType()); + assertEquals(LakeFormats.ICEBERG_CATALOG, cat.getImplementation()); assertEquals("file:///tmp/hop-warehouse", cat.getWarehouse()); assertEquals("", cat.getUri()); assertEquals("", cat.getCredential()); @@ -42,71 +42,71 @@ void icebergHadoopLocalSetsWarehouseAndType() { @Test void icebergRestSetsUri() { - SparkCatalog cat = new SparkCatalog(); - SparkCatalogTemplate.ICEBERG_REST.applyTo(cat); - assertEquals(SparkCatalog.TYPE_REST, cat.getCatalogType()); + LakeCatalog cat = new LakeCatalog(); + LakeCatalogTemplate.ICEBERG_REST.applyTo(cat); + assertEquals(LakeCatalog.TYPE_REST, cat.getCatalogType()); assertEquals("https://catalog.example.com/v1", cat.getUri()); assertTrue(cat.getWarehouse() == null || cat.getWarehouse().isEmpty()); } @Test void icebergRestAuthLeavesCredentialEmptyAndDocumentsToken() { - SparkCatalog cat = new SparkCatalog(); + LakeCatalog cat = new LakeCatalog(); cat.setCredential("should-be-cleared"); - SparkCatalogTemplate.ICEBERG_REST_AUTH.applyTo(cat); - assertEquals(SparkCatalog.TYPE_REST, cat.getCatalogType()); + LakeCatalogTemplate.ICEBERG_REST_AUTH.applyTo(cat); + assertEquals(LakeCatalog.TYPE_REST, cat.getCatalogType()); assertEquals("", cat.getCredential()); assertTrue(cat.getConfExtra().contains("Credential")); } @Test void objectStoreSetsS3aAndIoImpl() { - SparkCatalog cat = new SparkCatalog(); - SparkCatalogTemplate.ICEBERG_HADOOP_OBJECT_STORE.applyTo(cat); + LakeCatalog cat = new LakeCatalog(); + LakeCatalogTemplate.ICEBERG_HADOOP_OBJECT_STORE.applyTo(cat); assertEquals("s3a://bucket/warehouse", cat.getWarehouse()); assertTrue(cat.getConfExtra().contains("io-impl=")); } @Test void hiveAndGlueAreAdvancedTypes() { - SparkCatalog hive = new SparkCatalog(); - SparkCatalogTemplate.HIVE_METASTORE.applyTo(hive); - assertEquals(SparkCatalog.TYPE_HIVE, hive.getCatalogType()); + LakeCatalog hive = new LakeCatalog(); + LakeCatalogTemplate.HIVE_METASTORE.applyTo(hive); + assertEquals(LakeCatalog.TYPE_HIVE, hive.getCatalogType()); assertEquals("hive", hive.getCatalogName()); assertTrue(hive.getConfExtra().contains("thrift://")); assertTrue(hive.getConfExtra().startsWith("# docs:")); - SparkCatalog glue = new SparkCatalog(); - SparkCatalogTemplate.AWS_GLUE.applyTo(glue); - assertEquals(SparkCatalog.TYPE_GLUE, glue.getCatalogType()); + LakeCatalog glue = new LakeCatalog(); + LakeCatalogTemplate.AWS_GLUE.applyTo(glue); + assertEquals(LakeCatalog.TYPE_GLUE, glue.getCatalogType()); assertEquals("glue", glue.getCatalogName()); assertTrue(glue.getConfExtra().contains("# docs:")); } @Test void nessieUnityAndDeltaAdvancedTemplates() { - SparkCatalog nessie = new SparkCatalog(); - SparkCatalogTemplate.NESSIE.applyTo(nessie); - assertEquals(SparkCatalog.TYPE_CUSTOM, nessie.getCatalogType()); + LakeCatalog nessie = new LakeCatalog(); + LakeCatalogTemplate.NESSIE.applyTo(nessie); + assertEquals(LakeCatalog.TYPE_CUSTOM, nessie.getCatalogType()); assertEquals("nessie", nessie.getCatalogName()); assertTrue(nessie.getConfExtra().contains("NessieCatalog")); - assertTrue(nessie.getConfExtra().contains(SparkCatalogTemplate.Docs.NESSIE_SPARK)); + assertTrue(nessie.getConfExtra().contains(LakeCatalogTemplate.Docs.NESSIE_SPARK)); - SparkCatalog unity = new SparkCatalog(); - SparkCatalogTemplate.DATABRICKS_UNITY.applyTo(unity); + LakeCatalog unity = new LakeCatalog(); + LakeCatalogTemplate.DATABRICKS_UNITY.applyTo(unity); assertEquals("unity", unity.getCatalogName()); - assertTrue(unity.getConfExtra().contains(SparkCatalogTemplate.Docs.DATABRICKS_UNITY)); + assertTrue(unity.getConfExtra().contains(LakeCatalogTemplate.Docs.DATABRICKS_UNITY)); - SparkCatalog delta = new SparkCatalog(); - SparkCatalogTemplate.DELTA_NAMED_CATALOG.applyTo(delta); - assertEquals(SparkLakeFormats.DELTA_CATALOG, delta.getImplementation()); - assertTrue(delta.getConfExtra().contains(SparkCatalogTemplate.Docs.DELTA)); + LakeCatalog delta = new LakeCatalog(); + LakeCatalogTemplate.DELTA_NAMED_CATALOG.applyTo(delta); + assertEquals(LakeFormats.DELTA_CATALOG, delta.getImplementation()); + assertTrue(delta.getConfExtra().contains(LakeCatalogTemplate.Docs.DELTA)); } @Test void everyTemplateIncludesDocsCommentInConfExtra() { - for (SparkCatalogTemplate t : SparkCatalogTemplate.values()) { - SparkCatalog cat = new SparkCatalog(); + for (LakeCatalogTemplate t : LakeCatalogTemplate.values()) { + LakeCatalog cat = new LakeCatalog(); t.applyTo(cat); assertTrue( cat.getConfExtra() != null && cat.getConfExtra().contains("# docs:"), @@ -117,39 +117,38 @@ void everyTemplateIncludesDocsCommentInConfExtra() { @Test void confWithDocsFormatsHeaderAndBody() { assertEquals( - "# docs: https://example.com", - SparkCatalogTemplate.confWithDocs("https://example.com", "")); + "# docs: https://example.com", LakeCatalogTemplate.confWithDocs("https://example.com", "")); assertEquals( "# docs: https://example.com\nuri=thrift://x", - SparkCatalogTemplate.confWithDocs("https://example.com", "uri=thrift://x")); + LakeCatalogTemplate.confWithDocs("https://example.com", "uri=thrift://x")); } @Test void fromDisplayNameRoundTrip() { - for (SparkCatalogTemplate t : SparkCatalogTemplate.values()) { - assertEquals(t, SparkCatalogTemplate.fromDisplayName(t.getDisplayName())); + for (LakeCatalogTemplate t : LakeCatalogTemplate.values()) { + assertEquals(t, LakeCatalogTemplate.fromDisplayName(t.getDisplayName())); } - assertEquals(SparkCatalogTemplate.values().length, SparkCatalogTemplate.displayNames().length); + assertEquals(LakeCatalogTemplate.values().length, LakeCatalogTemplate.displayNames().length); } @Test void looksCustomizedDetectsEdits() { - SparkCatalog fresh = new SparkCatalog(); - assertFalse(SparkCatalogTemplate.looksCustomized(fresh)); + LakeCatalog fresh = new LakeCatalog(); + assertFalse(LakeCatalogTemplate.looksCustomized(fresh)); - SparkCatalog withName = new SparkCatalog(); + LakeCatalog withName = new LakeCatalog(); withName.setCatalogName("lake"); - assertTrue(SparkCatalogTemplate.looksCustomized(withName)); + assertTrue(LakeCatalogTemplate.looksCustomized(withName)); - SparkCatalog withWarehouse = new SparkCatalog(); + LakeCatalog withWarehouse = new LakeCatalog(); withWarehouse.setWarehouse("file:///tmp/wh"); - assertTrue(SparkCatalogTemplate.looksCustomized(withWarehouse)); + assertTrue(LakeCatalogTemplate.looksCustomized(withWarehouse)); } @Test void everyTemplateAppliesWithoutNpe() { - for (SparkCatalogTemplate t : SparkCatalogTemplate.values()) { - SparkCatalog cat = new SparkCatalog(); + for (LakeCatalogTemplate t : LakeCatalogTemplate.values()) { + LakeCatalog cat = new LakeCatalog(); t.applyTo(cat); assertNotNull(cat.getCatalogType()); assertNotNull(cat.getImplementation()); diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMetaInjectionTest.java b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableInputMetaInjectionTest.java similarity index 86% rename from plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMetaInjectionTest.java rename to plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableInputMetaInjectionTest.java index 96751c07e23..fbd16f28e94 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableInputMetaInjectionTest.java +++ b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableInputMetaInjectionTest.java @@ -15,28 +15,27 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import java.util.ArrayList; import java.util.List; import org.apache.hop.core.injection.BaseMetadataInjectionTestJunit5; import org.apache.hop.junit.rules.RestoreHopEngineEnvironmentExtension; -import org.apache.hop.spark.transforms.io.SparkField; +import org.apache.hop.lakehouse.LakeField; import org.junit.jupiter.api.BeforeEach; import org.junit.jupiter.api.Test; import org.junit.jupiter.api.extension.RegisterExtension; -class SparkLakeTableInputMetaInjectionTest - extends BaseMetadataInjectionTestJunit5 { +class LakeTableInputMetaInjectionTest extends BaseMetadataInjectionTestJunit5 { @RegisterExtension static RestoreHopEngineEnvironmentExtension env = new RestoreHopEngineEnvironmentExtension(); @BeforeEach void setup() throws Exception { - SparkLakeTableInputMeta meta = new SparkLakeTableInputMeta(); - List fields = new ArrayList<>(); - fields.add(new SparkField()); + LakeTableInputMeta meta = new LakeTableInputMeta(); + List fields = new ArrayList<>(); + fields.add(new LakeField()); meta.setFields(fields); setup(meta); } diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMetaInjectionTest.java b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMetaInjectionTest.java similarity index 89% rename from plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMetaInjectionTest.java rename to plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMetaInjectionTest.java index abb2adcf80b..ed020b3a0d8 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMaintenanceMetaInjectionTest.java +++ b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMaintenanceMetaInjectionTest.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.injection.BaseMetadataInjectionTestJunit5; import org.apache.hop.junit.rules.RestoreHopEngineEnvironmentExtension; @@ -23,15 +23,15 @@ import org.junit.jupiter.api.Test; import org.junit.jupiter.api.extension.RegisterExtension; -class SparkLakeTableMaintenanceMetaInjectionTest - extends BaseMetadataInjectionTestJunit5 { +class LakeTableMaintenanceMetaInjectionTest + extends BaseMetadataInjectionTestJunit5 { @RegisterExtension static RestoreHopEngineEnvironmentExtension env = new RestoreHopEngineEnvironmentExtension(); @BeforeEach void setup() throws Exception { - setup(new SparkLakeTableMaintenanceMeta()); + setup(new LakeTableMaintenanceMeta()); } @Test diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMetaInjectionTest.java b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMetaInjectionTest.java similarity index 90% rename from plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMetaInjectionTest.java rename to plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMetaInjectionTest.java index 23183d79795..c8cc89383c3 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableMergeMetaInjectionTest.java +++ b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableMergeMetaInjectionTest.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.injection.BaseMetadataInjectionTestJunit5; import org.apache.hop.junit.rules.RestoreHopEngineEnvironmentExtension; @@ -23,15 +23,14 @@ import org.junit.jupiter.api.Test; import org.junit.jupiter.api.extension.RegisterExtension; -class SparkLakeTableMergeMetaInjectionTest - extends BaseMetadataInjectionTestJunit5 { +class LakeTableMergeMetaInjectionTest extends BaseMetadataInjectionTestJunit5 { @RegisterExtension static RestoreHopEngineEnvironmentExtension env = new RestoreHopEngineEnvironmentExtension(); @BeforeEach void setup() throws Exception { - setup(new SparkLakeTableMergeMeta()); + setup(new LakeTableMergeMeta()); } @Test diff --git a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMetaInjectionTest.java b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMetaInjectionTest.java similarity index 89% rename from plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMetaInjectionTest.java rename to plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMetaInjectionTest.java index 2dc9f607409..b00472635fc 100644 --- a/plugins/engines/spark/src/test/java/org/apache/hop/spark/transforms/table/SparkLakeTableOutputMetaInjectionTest.java +++ b/plugins/tech/lakehouse/src/test/java/org/apache/hop/lakehouse/transforms/LakeTableOutputMetaInjectionTest.java @@ -15,7 +15,7 @@ * limitations under the License. */ -package org.apache.hop.spark.transforms.table; +package org.apache.hop.lakehouse.transforms; import org.apache.hop.core.injection.BaseMetadataInjectionTestJunit5; import org.apache.hop.junit.rules.RestoreHopEngineEnvironmentExtension; @@ -23,15 +23,15 @@ import org.junit.jupiter.api.Test; import org.junit.jupiter.api.extension.RegisterExtension; -class SparkLakeTableOutputMetaInjectionTest - extends BaseMetadataInjectionTestJunit5 { +class LakeTableOutputMetaInjectionTest + extends BaseMetadataInjectionTestJunit5 { @RegisterExtension static RestoreHopEngineEnvironmentExtension env = new RestoreHopEngineEnvironmentExtension(); @BeforeEach void setup() throws Exception { - setup(new SparkLakeTableOutputMeta()); + setup(new LakeTableOutputMeta()); } @Test diff --git a/plugins/tech/pom.xml b/plugins/tech/pom.xml index 08deec71be7..0d53916f8e4 100644 --- a/plugins/tech/pom.xml +++ b/plugins/tech/pom.xml @@ -43,6 +43,7 @@ git-vfs google hadoop + lakehouse minio mongodb neo4j