This is an automated email from the ASF dual-hosted git repository.

laskoviymishka pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/iceberg-go.git


The following commit(s) were added to refs/heads/main by this push:
     new 55f384b8e fix(schema): match Java column name sanitization (#1576)
55f384b8e is described below

commit 55f384b8e12272bf576e8ed78fb532f4805826e7
Author: Minh Vu <[email protected]>
AuthorDate: Fri Jul 31 16:19:53 2026 +0200

    fix(schema): match Java column name sanitization (#1576)
    
    ## What changed
    
    Align column-name sanitization with Java Iceberg's Unicode behavior:
    
    - process names rune by rune instead of treating the first UTF-8 byte as
    a character
    - preserve valid BMP Unicode letters and digits
    - encode supplementary characters as Java-compatible UTF-16 surrogate
    escapes
    - reject malformed UTF-8 and sibling-name collisions introduced by
    sanitization
    
    Initial digits retain the existing readable `_<digit>` form.
    
    ## Why
    
    The previous implementation handled the first byte of a multibyte
    character separately, then resumed iteration in the middle of its UTF-8
    encoding. This could produce malformed output and behavior that differed
    from Java Iceberg.
    
    Sanitization can also map distinct source names to the same result.
    Rejecting those collisions prevents schemas with ambiguous sibling
    fields.
    
    ## Testing
    
    Coverage includes valid ASCII names, BMP and supplementary Unicode
    letters and digits, emoji, punctuation, malformed UTF-8, top-level and
    nested collisions, and collision scoping across separate records.
    
    - `go test .`
    - `go vet .`
    
    ---------
    
    Signed-off-by: Minh Vu <[email protected]>
    Signed-off-by: Hoang Minh Vu <[email protected]>
---
 schema.go      |  69 ++++++++++++++++++++++++-------
 schema_test.go | 128 +++++++++++++++++++++++++++++++++++++++++++++++++++++++++
 2 files changed, 183 insertions(+), 14 deletions(-)

diff --git a/schema.go b/schema.go
index f4c2095aa..9f51474ed 100644
--- a/schema.go
+++ b/schema.go
@@ -27,6 +27,8 @@ import (
        "sync"
        "sync/atomic"
        "unicode"
+       "unicode/utf16"
+       "unicode/utf8"
 )
 
 // Schema is an Iceberg table schema, represented as a struct with
@@ -1649,12 +1651,16 @@ func validAvroName(n string) bool {
                return false
        }
 
-       if !unicode.IsLetter(rune(n[0])) && n[0] != '_' {
-               return false
-       }
+       for i, r := range n {
+               if i == 0 {
+                       if !isAvroNameStart(r) {
+                               return false
+                       }
 
-       for _, r := range n[1:] {
-               if !unicode.In(r, unicode.Number, unicode.Letter) && r != '_' {
+                       continue
+               }
+
+               if !isAvroNamePart(r) {
                        return false
                }
        }
@@ -1662,7 +1668,23 @@ func validAvroName(n string) bool {
        return true
 }
 
+const maxBMPRune rune = 0xFFFF
+
+func isAvroNameStart(r rune) bool {
+       return r <= maxBMPRune && (r == '_' || unicode.IsLetter(r))
+}
+
+func isAvroNamePart(r rune) bool {
+       return r <= maxBMPRune && (isAvroNameStart(r) || unicode.IsDigit(r))
+}
+
 func sanitize(r rune) string {
+       if r > maxBMPRune {
+               high, low := utf16.EncodeRune(r)
+
+               return fmt.Sprintf("_x%X_x%X", high, low)
+       }
+
        if unicode.IsDigit(r) {
                return "_" + string(r)
        }
@@ -1676,17 +1698,15 @@ func sanitizeName(n string) string {
        }
 
        var b strings.Builder
-       b.Grow(len(n))
+       b.Grow(len(n) * 3)
 
-       first := n[0]
-       if !unicode.IsLetter(rune(first)) && first != '_' {
-               b.WriteString(sanitize(rune(first)))
-       } else {
-               b.WriteByte(first)
-       }
+       for i, r := range n {
+               valid := isAvroNamePart(r)
+               if i == 0 {
+                       valid = isAvroNameStart(r)
+               }
 
-       for _, r := range n[1:] {
-               if !unicode.In(r, unicode.Number, unicode.Letter) && r != '_' {
+               if !valid {
                        b.WriteString(sanitize(r))
                } else {
                        b.WriteRune(r)
@@ -1696,6 +1716,14 @@ func sanitizeName(n string) string {
        return b.String()
 }
 
+// SanitizeColumnNames returns a copy of sc whose field names are compatible
+// with Java Iceberg's Avro name sanitization. Characters outside the BMP are
+// escaped as UTF-16 surrogate pairs, and only Unicode decimal digits are
+// treated as digits; other numeric categories are escaped.
+//
+// Empty or invalid UTF-8 field names and names that collide after sanitization
+// return an error wrapping ErrInvalidSchema. Collision errors are reported 
here
+// before a downstream Avro schema builder encounters the duplicate field name.
 func SanitizeColumnNames(sc *Schema) (*Schema, error) {
        result, err := Visit(sc, sanitizeColumnNameVisitor{})
        if err != nil {
@@ -1718,6 +1746,9 @@ func (sanitizeColumnNameVisitor) Field(field NestedField, 
fieldResult NestedFiel
        if field.Name == "" {
                panic(fmt.Errorf("%w: field name cannot be empty", 
ErrInvalidSchema))
        }
+       if !utf8.ValidString(field.Name) {
+               panic(fmt.Errorf("%w: field %d name is not valid UTF-8", 
ErrInvalidSchema, field.ID))
+       }
 
        field.Name = makeCompatibleName(field.Name)
 
@@ -1725,6 +1756,16 @@ func (sanitizeColumnNameVisitor) Field(field 
NestedField, fieldResult NestedFiel
 }
 
 func (sanitizeColumnNameVisitor) Struct(_ StructType, fieldResults 
[]NestedField) NestedField {
+       seen := make(map[string]int, len(fieldResults))
+       for _, field := range fieldResults {
+               if previousID, ok := seen[field.Name]; ok {
+                       panic(fmt.Errorf(
+                               "%w: fields %d and %d produce duplicate 
sanitized name %q",
+                               ErrInvalidSchema, previousID, field.ID, 
field.Name))
+               }
+               seen[field.Name] = field.ID
+       }
+
        return NestedField{Type: &StructType{FieldList: fieldResults}}
 }
 
diff --git a/schema_test.go b/schema_test.go
index e9102b81e..34dbe7414 100644
--- a/schema_test.go
+++ b/schema_test.go
@@ -1536,6 +1536,134 @@ func TestSanitizeColumnNamesEmptyFieldName(t 
*testing.T) {
        assert.ErrorContains(t, err, "field name cannot be empty")
 }
 
+func TestSanitizeColumnNamesMatchesJavaIceberg(t *testing.T) {
+       t.Parallel()
+
+       tests := []struct {
+               name  string
+               input string
+               want  string
+       }{
+               {name: "ASCII letter", input: "Field_9", want: "Field_9"},
+               {name: "underscore", input: "_field", want: "_field"},
+               {name: "ASCII digit first", input: "1field", want: "_1field"},
+               {name: "latin letter", input: "éclair", want: "éclair"},
+               {name: "CJK letters", input: "你好", want: "你好"},
+               {name: "extended latin letter", input: "Łacinka", want: 
"Łacinka"},
+               {name: "Unicode digit first", input: "١field", want: "_١field"},
+               {name: "Unicode digit later", input: "a١field", want: 
"a١field"},
+               {name: "supplementary letter", input: "𐐀field", want: 
"_xD801_xDC00field"},
+               {name: "supplementary digit first", input: "𝟎field", want: 
"_xD835_xDFCEfield"},
+               {name: "supplementary digit later", input: "a𝟎field", want: 
"a_xD835_xDFCEfield"},
+               {name: "superscript number", input: "a²", want: "a_xB2"},
+               // Java Character.isLetterOrDigit excludes Unicode letter 
numbers (Nl).
+               {name: "letter number", input: "aⅡ", want: "a_x2161"},
+               {name: "combining mark", input: "e\u0301", want: "e_x301"},
+               {name: "emoji first", input: "😀field", want: 
"_xD83D_xDE00field"},
+               {name: "emoji later", input: "a😀field", want: 
"a_xD83D_xDE00field"},
+               {name: "punctuation first", input: "-field", want: "_x2Dfield"},
+               {name: "punctuation later", input: "a-field", want: 
"a_x2Dfield"},
+       }
+
+       for _, test := range tests {
+               t.Run(test.name, func(t *testing.T) {
+                       t.Parallel()
+
+                       schema := iceberg.NewSchema(1, iceberg.NestedField{ID: 
1, Name: test.input, Type: iceberg.PrimitiveTypes.String})
+                       sanitized, err := iceberg.SanitizeColumnNames(schema)
+                       require.NoError(t, err)
+                       got := sanitized.Field(0).Name
+                       assert.Equal(t, test.want, got)
+               })
+       }
+}
+
+func TestSanitizeColumnNamesRejectsCollisions(t *testing.T) {
+       t.Parallel()
+
+       tests := []struct {
+               name   string
+               fields []iceberg.NestedField
+               want   string
+       }{
+               {
+                       name: "leading ASCII digit collides with underscored 
name",
+                       fields: []iceberg.NestedField{
+                               {ID: 1, Name: "1field", Type: 
iceberg.PrimitiveTypes.String},
+                               {ID: 2, Name: "_1field", Type: 
iceberg.PrimitiveTypes.String},
+                       },
+                       want: `fields 1 and 2 produce duplicate sanitized name 
"_1field"`,
+               },
+               {
+                       name: "emoji escape collides with existing name",
+                       fields: []iceberg.NestedField{
+                               {ID: 3, Name: "😀", Type: 
iceberg.PrimitiveTypes.String},
+                               {ID: 4, Name: "_xD83D_xDE00", Type: 
iceberg.PrimitiveTypes.String},
+                       },
+                       want: `fields 3 and 4 produce duplicate sanitized name 
"_xD83D_xDE00"`,
+               },
+       }
+
+       for _, test := range tests {
+               t.Run(test.name, func(t *testing.T) {
+                       t.Parallel()
+
+                       _, err := 
iceberg.SanitizeColumnNames(iceberg.NewSchema(1, test.fields...))
+                       require.ErrorIs(t, err, iceberg.ErrInvalidSchema)
+                       assert.ErrorContains(t, err, test.want)
+               })
+       }
+}
+
+func TestSanitizeColumnNamesScopesCollisionChecksToStruct(t *testing.T) {
+       t.Parallel()
+
+       schema := iceberg.NewSchema(1,
+               iceberg.NestedField{
+                       ID: 1, Name: "customer", Type: 
&iceberg.StructType{FieldList: []iceberg.NestedField{
+                               {ID: 2, Name: "😀", Type: 
iceberg.PrimitiveTypes.String},
+                       }},
+               },
+               iceberg.NestedField{
+                       ID: 3, Name: "address", Type: 
&iceberg.StructType{FieldList: []iceberg.NestedField{
+                               {ID: 4, Name: "_xD83D_xDE00", Type: 
iceberg.PrimitiveTypes.String},
+                       }},
+               },
+       )
+
+       sanitized, err := iceberg.SanitizeColumnNames(schema)
+       require.NoError(t, err)
+       assert.Equal(t, "_xD83D_xDE00", 
sanitized.Field(0).Type.(*iceberg.StructType).FieldList[0].Name)
+       assert.Equal(t, "_xD83D_xDE00", 
sanitized.Field(1).Type.(*iceberg.StructType).FieldList[0].Name)
+}
+
+func TestSanitizeColumnNamesRejectsNestedCollision(t *testing.T) {
+       t.Parallel()
+
+       schema := iceberg.NewSchema(1, iceberg.NestedField{
+               ID: 1, Name: "record", Type: &iceberg.StructType{FieldList: 
[]iceberg.NestedField{
+                       {ID: 2, Name: "1field", Type: 
iceberg.PrimitiveTypes.String},
+                       {ID: 3, Name: "_1field", Type: 
iceberg.PrimitiveTypes.String},
+               }},
+       })
+
+       _, err := iceberg.SanitizeColumnNames(schema)
+       require.ErrorIs(t, err, iceberg.ErrInvalidSchema)
+       assert.ErrorContains(t, err, `fields 2 and 3 produce duplicate 
sanitized name "_1field"`)
+}
+
+func TestSanitizeColumnNamesRejectsInvalidUTF8(t *testing.T) {
+       t.Parallel()
+
+       schema := iceberg.NewSchema(1, iceberg.NestedField{
+               ID: 7, Name: string([]byte{0xff, 'a'}), Type: 
iceberg.PrimitiveTypes.String,
+       })
+
+       _, err := iceberg.SanitizeColumnNames(schema)
+       require.ErrorIs(t, err, iceberg.ErrInvalidSchema)
+       assert.ErrorContains(t, err, "field 7 name is not valid UTF-8")
+}
+
 func TestSchemaSelectCaseSensitiveSuccess(t *testing.T) {
        selected, err := tableSchemaSimple.Select(true, "foo", "bar")
        require.NoError(t, err)

Reply via email to