FelixYBW commented on issue #13140:
URL: https://github.com/apache/gluten/issues/13140#issuecomment-5852401024

   the classes related to Arrow after implementation:
   
   <img width="1576" height="665" alt="Image" 
src="https://github.com/user-attachments/assets/28691920-dac2-4f3d-814f-0a9c6bb6530e";
 />
   
   <details>
   <summary>plantuml</summary>
   
   ```puml
   @startuml arrow_vectors_after
   set separator none
   skinparam linetype ortho
   skinparam nodesep 45
   skinparam ranksep 45
   
   !define ARROW_COLOR #E1F5FE
   !define SPARK_COLOR #FFF9C4
   !define PROPOSED_COLOR #FFE0B2
   !define GLUTEN_COLOR #E8F5E9
   !define REMOVED_COLOR #afabab
   
   title After: Gluten reuses Spark's ArrowWriter + ArrowColumnVector
   
   package "Arrow (org.apache.arrow)" as PArrow #EAF6FD {
     interface "vector.ValueVector" as ValueVector ARROW_COLOR
     interface "vector.FieldVector" as FieldVector ARROW_COLOR
     class "vector.util.TransferPair" as TransferPair ARROW_COLOR {
       +transfer()
     }
     interface "memory.BufferAllocator" as BufferAllocator ARROW_COLOR
     class "vector.VectorSchemaRoot" as VectorSchemaRoot ARROW_COLOR {
       +create(Schema, BufferAllocator)
       +of(FieldVector...)
       +getFieldVectors()
     }
     class "c.Data (ArrowArray / ArrowSchema)" as CData ARROW_COLOR {
       +importVectorSchemaRoot()
       +exportVectorSchemaRoot()
     }
   }
   
   package "Spark (org.apache.spark.sql)" as PSpark #FFFDE7 {
     class "vectorized.ArrowColumnVector" as ArrowColumnVector SPARK_COLOR {
       +ArrowColumnVector(ValueVector)
       +getValueVector(): ValueVector
       +close()
     }
     abstract class "vectorized.ColumnVector" as ColumnVector SPARK_COLOR
     class "vectorized.ColumnarBatch" as ColumnarBatch SPARK_COLOR
     abstract class "execution.arrow.ArrowFieldWriter" as ArrowFieldWriter 
SPARK_COLOR
     class "execution.arrow.ArrowWriter" as ArrowWriter SPARK_COLOR {
       +create(root): ArrowWriter
       +write(InternalRow)
       +finish() / reset()
     }
     class "util.ArrowUtils" as ArrowUtils SPARK_COLOR {
       +toArrowSchema(StructType, ...)
       +isCompatibleWithDeclaredField()
     }
     class "execution.arrow.ArrowBatchUtils" as ArrowBatchUtils <<proposed: 
from #55120 helpers>> PROPOSED_COLOR {
       +isArrowBacked(batch, declaredSchema)
       +toVectorSchemaRoot(batch)
     }
     class "execution.convention.BatchType.ArrowBatchType" as SArrowBatch 
<<proposed: SPARK-57468>> PROPOSED_COLOR {
       batches of ArrowColumnVector
       canonical layout for the plan's schema
       valid until next(); transfer to keep
     }
     class "execution.RowToArrowColumnarExec" as RowToArrow <<proposed: 
SPARK-37124>> PROPOSED_COLOR {
       ArrowBatchType.fromRow(VanillaRowType)
       uses ArrowWriter
     }
     class "execution.python.ArrowEvalPythonExec\n/ ArrowEvalPythonUDTFExec" as 
SparkPython SPARK_COLOR {
       requires ArrowBatchType
       (#55120, SPARK-59792)
     }
   }
   
   package "Gluten (org.apache.gluten)" as PGluten #F1F8E9 {
     class "backendsapi.arrow.ArrowBatchTypes.ArrowJavaBatchType" as ArrowJava 
GLUTEN_COLOR {
       same type as Spark ArrowBatchType
       (bound via SparkConventions)
     }
     class "extension.columnar.transition.SparkConventions" as SparkConventions 
GLUTEN_COLOR {
       +bind(ArrowBatchType, ArrowJavaBatchType)
     }
     class "utils.ArrowAbiUtil" as ArrowAbiUtil GLUTEN_COLOR {
       +importToArrowColumnarBatch()
       +exportFromArrowColumnarBatch()
     }
     class "columnarbatch.ColumnarBatches" as ColumnarBatches GLUTEN_COLOR {
       +load(allocator, nativeBatch)
       +offload(allocator, arrowBatch)
       ownership transfer, no ref count
     }
     class "execution.LoadArrowDataExec\n/ OffloadArrowDataExec" as LoadOffload 
GLUTEN_COLOR {
       ArrowNative <-> ArrowJava
     }
     class "execution: 
ColumnarRangeExec,\nColumnarPartialProjectExec,\nColumnarPartialGenerateExec" 
as WriteOps GLUTEN_COLOR {
       write rows via ArrowWriter
     }
     class "memory.arrow.alloc.ArrowBufferAllocators" as GlutenAlloc 
GLUTEN_COLOR {
       +contextInstance(): BufferAllocator
     }
   }
   
   package "Removed from Gluten" as PRemoved #F5F5F5 {
     class "vectorized.ArrowWritableColumnVector" as R1 REMOVED_COLOR
     class "WritableColumnVectorShim\n(spark34 / 35 / 40 / 41)" as R2 
REMOVED_COLOR
     class 
"SparkArrowBatchType,\nArrowJavaToSparkArrowExec,\nSparkArrowToArrowJavaExec" 
as R3 REMOVED_COLOR
     class "ColumnarArrowEvalPythonExec,\nColumnarArrowEvalPythonUDTFExec 
(#12276)" as R4 REMOVED_COLOR
   }
   
   ' --- Arrow internals (top-to-bottom order) ---
   ValueVector <-- FieldVector
   FieldVector "many" --* VectorSchemaRoot
   BufferAllocator <.. VectorSchemaRoot
   ValueVector <.. TransferPair
   VectorSchemaRoot <.. CData
   
   ' --- Arrow (top) -> Spark (below) ---
   ValueVector "1" <--o ArrowColumnVector : wraps (read)
   ValueVector "1" <--o ArrowFieldWriter : setSafe()
   VectorSchemaRoot "1" <--o ArrowWriter : writes into
   
   ' --- Spark internals ---
   ArrowColumnVector --|> ColumnVector
   ColumnVector --* "many" ColumnarBatch
   ArrowFieldWriter --* "many" ArrowWriter
   ArrowWriter ..> ArrowUtils
   ArrowBatchUtils ..> ArrowUtils : layout check
   ArrowBatchUtils ..> VectorSchemaRoot : of(getValueVector...)
   SArrowBatch ..> ColumnarBatch : of ArrowColumnVector
   RowToArrow ..> ArrowWriter
   RowToArrow ..> SArrowBatch : produces
   SparkPython ..> SArrowBatch : consumes
   SparkPython ..> ArrowBatchUtils
   
   ' --- Spark (top) -> Gluten (bottom) ---
   ArrowColumnVector <.. ArrowAbiUtil : wrap imported vectors
   ArrowBatchUtils <.. ArrowAbiUtil : unwrap for export
   SArrowBatch <.. ArrowJava : same type
   SArrowBatch <.. SparkConventions
   ArrowWriter <.. WriteOps : create(root)
   ArrowColumnVector <.. WriteOps : wrap root vectors
   
   ' --- Gluten & Arrow interactions ---
   CData <.. ArrowAbiUtil
   BufferAllocator <.. GlutenAlloc : task-scoped,\nmemory-managed
   GlutenAlloc <.. WriteOps : root allocator
   ArrowJava <.. SparkConventions
   ArrowAbiUtil <.. ColumnarBatches
   TransferPair <.. ColumnarBatches : take ownership
   ColumnarBatches <.. LoadOffload
   ArrowJava <.. LoadOffload
   
   ' --- Strict vertical hierarchy ---
   CData -[hidden]down-> ArrowColumnVector
   CData -[hidden]down-> ArrowFieldWriter
   CData -[hidden]down-> SparkPython
   
   ColumnarBatch -[hidden]down-> ArrowJava
   ArrowUtils -[hidden]down-> GlutenAlloc
   ColumnarBatch -[hidden]left-> R1
   
   note top of ColumnarBatch
     One Java Arrow batch type in Gluten and Spark:
     a Spark ColumnarBatch of ArrowColumnVector
     wrapping Arrow FieldVectors allocated by
     Gluten's managed allocator (or transferred
     into it).
   end note
   
   note bottom of WriteOps
     rows -> Arrow:
     root = VectorSchemaRoot.create(schema, contextInstance())
     w = ArrowWriter.create(root); w.write(row)...; w.finish()
     new ColumnarBatch(root.getFieldVectors.map(new ArrowColumnVector(_)))
   end note
   
   note bottom of ColumnarBatches
     native -> Java: Data.importVectorSchemaRoot -> ArrowColumnVector
     Java -> native: getValueVector -> VectorSchemaRoot.of
                     -> Data.exportVectorSchemaRoot
     keep a batch: TransferPair.transfer()
   end note
   @enduml
   
   ```
   
   </details>


-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]


---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]

Reply via email to