This is an automated email from the ASF dual-hosted git repository.

github-merge-queue[bot] pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/texera.git


The following commit(s) were added to refs/heads/main by this push:
     new 69c66a1764 feat(csv-scan): declare the Delimiter as the single 
character it is (#7261)
69c66a1764 is described below

commit 69c66a17647f3e530a90486d7acde5b0120cda74
Author: Kary Zheng <[email protected]>
AuthorDate: Wed Aug 5 17:01:01 2026 -0700

    feat(csv-scan): declare the Delimiter as the single character it is (#7261)
    
    ### What changes were proposed in this PR?
    
    `customDelimiter` on the three CSV scans is declared as a string with no
    length constraint, while every reader narrows it with `charAt(0)` —
    univocity's `setDelimiter` and scala-csv's `DefaultCSVFormat.delimiter`
    both take a `Char`. Each now declares `maxLength: 1`, and the
    description says a single character rather than leaving it implied.
    
    ### Why are the changes needed?
    
    Characters past the first are dropped without a word. Driving each
    operator's own `sourceSchema()` over a three-column `;`-separated file,
    `;abc` infers the same three columns as `;`, and `,;` collapses to one
    column exactly as a bare `,` does — so the second character never
    participates. A workflow typed that way runs to completion and looks
    right, and nothing ever says part of the field was discarded.
    
    ### Any related issues, documentation, discussions?
    
    Closes #7211
    
    ### How was this PR tested?
    
    `WorkflowOperator/compile`, `scalafmtCheckAll`, `scalafixAll --check`,
    and the three operators' descriptor specs (22 tests, three of them added
    here for the constraint itself).
    
    ### Does this PR introduce any user-facing change?
    
    Yes, and one worth calling out. A saved workflow whose Delimiter holds
    more than one character will now show as invalid in the property editor,
    since the form validates the stored properties against the schema. It
    still runs — the backend does not enforce the schema — and the run
    behaves exactly as before, because only the first character ever
    counted. Surfacing those workflows is the point: today they are silently
    truncating.
    
    ### Was this PR authored or co-authored using generative AI tooling?
    
    Generated-by: Claude Code (Claude Opus 5)
    
    ---------
    
    Co-authored-by: Claude Opus 5 (1M context) <[email protected]>
---
 .../source/scan/csv/CSVScanSourceOpDesc.scala      |  7 +-
 .../scan/csv/ParallelCSVScanSourceOpDesc.scala     |  6 +-
 .../scan/csvOld/CSVOldScanSourceOpDesc.scala       |  6 +-
 .../source/scan/csv/CSVScanSourceOpDescSpec.scala  | 89 +++++++++++++++++++++-
 4 files changed, 101 insertions(+), 7 deletions(-)

diff --git 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDesc.scala
 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDesc.scala
index 328e3b6c1f..1a784a44ee 100644
--- 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDesc.scala
+++ 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDesc.scala
@@ -20,7 +20,7 @@
 package org.apache.texera.amber.operator.source.scan.csv
 
 import com.fasterxml.jackson.annotation.{JsonInclude, JsonProperty, 
JsonPropertyDescription}
-import com.kjetland.jackson.jsonSchema.annotations.JsonSchemaTitle
+import com.kjetland.jackson.jsonSchema.annotations.{JsonSchemaInject, 
JsonSchemaTitle}
 import com.univocity.parsers.csv.{CsvFormat, CsvParser, CsvParserSettings}
 import org.apache.texera.amber.core.executor.OpExecWithClassName
 import org.apache.texera.amber.core.storage.DocumentFactory
@@ -37,10 +37,13 @@ import java.net.URI
 
 class CSVScanSourceOpDesc extends ScanSourceOpDesc {
 
+  // One character: every reader narrows this with charAt(0), because 
univocity's
+  // setDelimiter and scala-csv's DefaultCSVFormat both take a Char.
   @JsonProperty(defaultValue = ",")
   @JsonSchemaTitle("Delimiter")
-  @JsonPropertyDescription("delimiter to separate each line into fields")
+  @JsonPropertyDescription("single character separating the fields on each 
line")
   @JsonInclude(JsonInclude.Include.NON_ABSENT)
+  @JsonSchemaInject(json = """{ "maxLength": 1 }""")
   var customDelimiter: Option[String] = None
 
   @JsonProperty(defaultValue = "true")
diff --git 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/ParallelCSVScanSourceOpDesc.scala
 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/ParallelCSVScanSourceOpDesc.scala
index b3d1071b86..cdade62d1e 100644
--- 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/ParallelCSVScanSourceOpDesc.scala
+++ 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csv/ParallelCSVScanSourceOpDesc.scala
@@ -22,7 +22,7 @@ package org.apache.texera.amber.operator.source.scan.csv
 import com.fasterxml.jackson.annotation.{JsonProperty, JsonPropertyDescription}
 import com.fasterxml.jackson.databind.annotation.JsonDeserialize
 import com.github.tototoshi.csv.{CSVReader, DefaultCSVFormat}
-import com.kjetland.jackson.jsonSchema.annotations.JsonSchemaTitle
+import com.kjetland.jackson.jsonSchema.annotations.{JsonSchemaInject, 
JsonSchemaTitle}
 import org.apache.texera.amber.core.executor.OpExecWithClassName
 import org.apache.texera.amber.core.storage.DocumentFactory
 import 
org.apache.texera.amber.core.tuple.AttributeTypeUtils.inferSchemaFromRows
@@ -37,10 +37,12 @@ import java.net.URI
 
 class ParallelCSVScanSourceOpDesc extends ScanSourceOpDesc {
 
+  // One character -- see CSVScanSourceOpDesc.
   @JsonProperty(defaultValue = ",")
   @JsonSchemaTitle("Delimiter")
-  @JsonPropertyDescription("delimiter to separate each line into fields")
+  @JsonPropertyDescription("single character separating the fields on each 
line")
   @JsonDeserialize(contentAs = classOf[java.lang.String])
+  @JsonSchemaInject(json = """{ "maxLength": 1 }""")
   var customDelimiter: Option[String] = None
 
   @JsonProperty(defaultValue = "true")
diff --git 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csvOld/CSVOldScanSourceOpDesc.scala
 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csvOld/CSVOldScanSourceOpDesc.scala
index 2a18d431ff..57dec76c55 100644
--- 
a/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csvOld/CSVOldScanSourceOpDesc.scala
+++ 
b/common/workflow-operator/src/main/scala/org/apache/texera/amber/operator/source/scan/csvOld/CSVOldScanSourceOpDesc.scala
@@ -21,7 +21,7 @@ package org.apache.texera.amber.operator.source.scan.csvOld
 
 import com.fasterxml.jackson.annotation.{JsonProperty, JsonPropertyDescription}
 import com.github.tototoshi.csv.{CSVReader, DefaultCSVFormat}
-import com.kjetland.jackson.jsonSchema.annotations.JsonSchemaTitle
+import com.kjetland.jackson.jsonSchema.annotations.{JsonSchemaInject, 
JsonSchemaTitle}
 import org.apache.texera.amber.core.executor.OpExecWithClassName
 import org.apache.texera.amber.core.storage.DocumentFactory
 import 
org.apache.texera.amber.core.tuple.AttributeTypeUtils.inferSchemaFromRows
@@ -36,9 +36,11 @@ import java.net.URI
 
 class CSVOldScanSourceOpDesc extends ScanSourceOpDesc {
 
+  // One character -- see CSVScanSourceOpDesc.
   @JsonProperty(defaultValue = ",")
   @JsonSchemaTitle("Delimiter")
-  @JsonPropertyDescription("delimiter to separate each line into fields")
+  @JsonPropertyDescription("single character separating the fields on each 
line")
+  @JsonSchemaInject(json = """{ "maxLength": 1 }""")
   var customDelimiter: Option[String] = Some(",")
 
   @JsonProperty(defaultValue = "true")
diff --git 
a/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDescSpec.scala
 
b/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDescSpec.scala
index 87b86fea6d..3049d7f02a 100644
--- 
a/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDescSpec.scala
+++ 
b/common/workflow-operator/src/test/scala/org/apache/texera/amber/operator/source/scan/csv/CSVScanSourceOpDescSpec.scala
@@ -19,13 +19,18 @@
 
 package org.apache.texera.amber.operator.source.scan.csv
 
+import com.fasterxml.jackson.databind.JsonNode
+import com.fasterxml.jackson.databind.node.TextNode
+import com.github.fge.jsonschema.main.JsonSchemaFactory
 import org.apache.texera.amber.core.storage.FileResolver
 import org.apache.texera.amber.core.tuple.{AttributeType, Schema}
 import org.apache.texera.amber.core.workflow.WorkflowContext.{
   DEFAULT_EXECUTION_ID,
   DEFAULT_WORKFLOW_ID
 }
-import org.apache.texera.amber.operator.TestOperators
+import org.apache.texera.amber.operator.{LogicalOp, TestOperators}
+import org.apache.texera.amber.operator.metadata.OperatorMetadataGenerator
+import org.apache.texera.amber.operator.source.scan.ScanSourceOpDesc
 import 
org.apache.texera.amber.operator.source.scan.csvOld.CSVOldScanSourceOpDesc
 import org.scalatest.BeforeAndAfter
 import org.scalatest.flatspec.AnyFlatSpec
@@ -42,6 +47,27 @@ class CSVScanSourceOpDescSpec extends AnyFlatSpec with 
BeforeAndAfter {
     parallelCsvScanSourceOpDesc = new ParallelCSVScanSourceOpDesc()
   }
 
+  private val delimiterOwners: List[(String, Class[_ <: LogicalOp])] = List(
+    "CSV" -> classOf[CSVScanSourceOpDesc],
+    "parallel CSV" -> classOf[ParallelCSVScanSourceOpDesc],
+    "old CSV" -> classOf[CSVOldScanSourceOpDesc]
+  )
+
+  private def delimiterSchema(opDescClass: Class[_ <: LogicalOp]): JsonNode =
+    OperatorMetadataGenerator
+      .generateOperatorJsonSchema(opDescClass)
+      .path("properties")
+      .path("customDelimiter")
+
+  // The property editor validates a stored delimiter against the schema, so 
validate
+  // the same way rather than restating the bound the schema declares.
+  private def schemaValidates(propertySchema: JsonNode, delimiter: String): 
Boolean =
+    JsonSchemaFactory
+      .byDefault()
+      .getJsonSchema(propertySchema)
+      .validate(TextNode.valueOf(delimiter))
+      .isSuccess
+
   // Writes a CSV whose header row has an empty column (the third position),
   // e.g. `id,name,,age`, and returns the absolute path.
   private def writeCsvWithEmptyHeader(): String = {
@@ -54,6 +80,23 @@ class CSVScanSourceOpDescSpec extends AnyFlatSpec with 
BeforeAndAfter {
     tmpFile.toString
   }
 
+  // Writes a three-column `;`-separated CSV and returns the absolute path.
+  private def writeSemicolonCsv(): String = {
+    val tmpFile = Files.createTempFile("semicolon-", ".csv")
+    tmpFile.toFile.deleteOnExit()
+    Files.write(
+      tmpFile,
+      "id;name;age\n1;Alice;30\n2;Bob;25\n".getBytes(StandardCharsets.UTF_8)
+    )
+    tmpFile.toString
+  }
+
+  private def columnNames(opDesc: ScanSourceOpDesc, path: String): 
List[String] = {
+    opDesc.fileName = Some(path)
+    opDesc.setResolvedFileName(FileResolver.resolve(path))
+    opDesc.sourceSchema().getAttributes.map(_.getName).toList
+  }
+
   it should "infer schema from single-line-data csv" in {
 
     parallelCsvScanSourceOpDesc.fileName = 
Some(TestOperators.CountrySalesSmallCsvPath)
@@ -210,4 +253,48 @@ class CSVScanSourceOpDescSpec extends AnyFlatSpec with 
BeforeAndAfter {
     assert(names == List("id", "name", "column-3", "age"))
   }
 
+  it should "declare the delimiter as a single character on every CSV scan" in 
{
+    delimiterOwners.foreach {
+      case (name, opDescClass) =>
+        withClue(s"$name: ") {
+          val propertySchema = delimiterSchema(opDescClass)
+          assert(propertySchema.path("type").asText() == "string")
+          assert(propertySchema.path("maxLength").asInt() == 1)
+        }
+    }
+  }
+
+  it should "validate an empty or one-character delimiter and refuse a longer 
one" in {
+    delimiterOwners.foreach {
+      case (name, opDescClass) =>
+        withClue(s"$name: ") {
+          val propertySchema = delimiterSchema(opDescClass)
+          // Empty stays valid: the field is optional and every reader 
resolves an
+          // empty delimiter to a comma, so clearing it must not be an error.
+          assert(schemaValidates(propertySchema, ""))
+          assert(schemaValidates(propertySchema, ","))
+          assert(schemaValidates(propertySchema, ";"))
+          assert(!schemaValidates(propertySchema, ",;"))
+          assert(!schemaValidates(propertySchema, ";abc"))
+        }
+    }
+  }
+
+  it should "read a multi-character delimiter as its first character" in {
+    // What the constraint gives up: a saved workflow holding a longer 
delimiter now
+    // shows as invalid in the property editor. Its run is unchanged, which is 
what
+    // this pins -- the characters past the first never reached a parser.
+    val path = writeSemicolonCsv()
+    val csv = new CSVScanSourceOpDesc()
+    csv.customDelimiter = Some(";abc")
+    val parallelCsv = new ParallelCSVScanSourceOpDesc()
+    parallelCsv.customDelimiter = Some(";abc")
+    val oldCsv = new CSVOldScanSourceOpDesc()
+    oldCsv.customDelimiter = Some(";abc")
+
+    assert(columnNames(csv, path) == List("id", "name", "age"))
+    assert(columnNames(parallelCsv, path) == List("id", "name", "age"))
+    assert(columnNames(oldCsv, path) == List("id", "name", "age"))
+  }
+
 }

Reply via email to