This is an automated email from the ASF dual-hosted git repository.

github-merge-queue[bot] pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/texera.git


The following commit(s) were added to refs/heads/main by this push:
     new 6defa2c464 test(verify): give every operator the same table to read 
(#8360)
6defa2c464 is described below

commit 6defa2c464e7785996a3c30f8ea5154e41fa30f1
Author: Kary Zheng <[email protected]>
AuthorDate: Mon Sep 21 22:09:51 2026 +0000

    test(verify): give every operator the same table to read (#8360)
    
    ### What changes were proposed in this PR?
    
    One table, thirty-four columns and fifteen rows, holding a column of
    every
    type an operator can ask for and the awkward values each type admits: an
    empty
    cell, a zero, a negative, a string that looks like a number, a date at a
    boundary.
    
    An operator that needs something the table cannot hold — a pair of
    tables
    sharing a key, a numeric projection — names a projection of it rather
    than a
    fixture of its own, so what every operator reads stays one thing.
    
    ### Any related issues, documentation, discussions?
    
    Part of #8325, 8 of 27; that issue lists the set in order.
    
    Closes #8412, the task this change is the whole of.
    
    ### How was this PR tested?
    
    The tests in this change cover it. The whole set is exercised together
    once the last piece lands: every operator run through the engine and
    through its generated script, and the two answers compared.
    
    ### Was this PR authored or co-authored using generative AI tooling?
    
    Generated-by: Claude Code (Opus 5)
    
    ---------
    
    Co-authored-by: Claude Opus 5 (1M context) <[email protected]>
---
 .../test/resources/verify/canonical_fixture.json   | 557 +++++++++++++++++++++
 .../amber/translator/verify/CanonicalFixture.scala | 251 ++++++++++
 .../translator/verify/CanonicalFixtureSpec.scala   | 315 ++++++++++++
 .../amber/translator/verify/SharedFixture.scala    | 225 +++++++++
 4 files changed, 1348 insertions(+)

diff --git 
a/workflow-compiling-service/src/test/resources/verify/canonical_fixture.json 
b/workflow-compiling-service/src/test/resources/verify/canonical_fixture.json
new file mode 100644
index 0000000000..98789459b1
--- /dev/null
+++ 
b/workflow-compiling-service/src/test/resources/verify/canonical_fixture.json
@@ -0,0 +1,557 @@
+[
+  {
+    "id": 7,
+    "name": "eve",
+    "score": 0.9,
+    "open": 7.0,
+    "high": 8.5,
+    "low": 6.0,
+    "close": 7.5,
+    "iso_country": "IND",
+    "trade_date": "2024-01-07",
+    "pvalue": 0.2597402597402597,
+    "log2fc": 1.4,
+    "comp_a": 3.0,
+    "comp_b": 1.0,
+    "comp_c": 2.0,
+    "uvec": -3.0,
+    "edge_pair": "[0, 7]",
+    "node_src": "n3",
+    "node_dst": "n4",
+    "start_ts": "2024-01-07 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-07 08:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_7",
+    "a\"b\\c_simplex_a": 35.0,
+    "a\"b\\c_simplex_b": 30.0,
+    "a\"b\\c_simplex_c": 35.0,
+    "short_text": "The meeting is scheduled for three o'clock tomorrow.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 3.9000000000000004,
+    "petal_width": 1.3,
+    "a\"b\\c_species": 1,
+    "csv_list": "a7",
+    "mixed_case": "abacus",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 3,
+    "name": "bob",
+    "score": 1.2,
+    "open": 3.0,
+    "high": 4.5,
+    "low": 2.0,
+    "close": 3.5,
+    "iso_country": "JPN",
+    "trade_date": "2024-01-03",
+    "pvalue": 0.11188811188811189,
+    "log2fc": -1.4,
+    "comp_a": 4.0,
+    "comp_b": 4.0,
+    "comp_c": 1.0,
+    "uvec": 0.0,
+    "edge_pair": "[0, 3]",
+    "node_src": "n3",
+    "node_dst": "n4",
+    "start_ts": "2024-01-03 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-03 04:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_3",
+    "a\"b\\c_simplex_a": 35.0,
+    "a\"b\\c_simplex_b": 25.0,
+    "a\"b\\c_simplex_c": 40.0,
+    "short_text": "The meeting is scheduled for three o'clock tomorrow.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 5.2,
+    "petal_width": 1.85,
+    "a\"b\\c_species": 1,
+    "csv_list": "a3,b3",
+    "mixed_case": "ABBEY",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 11,
+    "name": "1",
+    "score": 1.7,
+    "open": 11.0,
+    "high": 12.5,
+    "low": 10.0,
+    "close": 11.5,
+    "iso_country": "USA",
+    "trade_date": "2024-01-11",
+    "pvalue": 0.4075924075924076,
+    "log2fc": -3.5,
+    "comp_a": 2.0,
+    "comp_b": 5.0,
+    "comp_c": 3.0,
+    "uvec": 1.0,
+    "edge_pair": "[0, 11]",
+    "node_src": "n3",
+    "node_dst": "n4",
+    "start_ts": "2024-01-11 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-11 04:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_11",
+    "a\"b\\c_simplex_a": 35.0,
+    "a\"b\\c_simplex_b": 35.0,
+    "a\"b\\c_simplex_c": 30.0,
+    "short_text": "URGENT: your account needs verification, click the link 
immediately.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 2.6,
+    "petal_width": 0.75,
+    "a\"b\\c_species": 0,
+    "csv_list": "a11,b11,c11",
+    "mixed_case": "101",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 0,
+    "species_name": "setosa",
+    "species_name_pred": "setosa"
+  },
+  {
+    "id": 1,
+    "name": "1",
+    "score": 0.5,
+    "open": 1.0,
+    "high": 2.5,
+    "low": 0.0,
+    "close": 1.5,
+    "iso_country": "USA",
+    "trade_date": "2024-01-01",
+    "pvalue": 0.03796203796203796,
+    "log2fc": -2.8,
+    "comp_a": 2.0,
+    "comp_b": 2.0,
+    "comp_c": 2.0,
+    "uvec": -2.0,
+    "edge_pair": "[0, 1]",
+    "node_src": "n1",
+    "node_dst": "n2",
+    "start_ts": "2024-01-01 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-01 02:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_1",
+    "a\"b\\c_simplex_a": 25.0,
+    "a\"b\\c_simplex_b": 30.0,
+    "a\"b\\c_simplex_c": 45.0,
+    "short_text": "I'm really not sure how I feel about it.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 2.6,
+    "petal_width": 0.75,
+    "a\"b\\c_species": 0,
+    "csv_list": "a1,b1,c1,d1",
+    "mixed_case": "abdomen",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "setosa",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 14,
+    "name": "carol",
+    "score": 4.5,
+    "open": 14.0,
+    "high": 15.5,
+    "low": 13.0,
+    "close": 14.5,
+    "iso_country": "DEU",
+    "trade_date": "2024-01-14",
+    "pvalue": 0.5184815184815185,
+    "log2fc": -1.4,
+    "comp_a": 5.0,
+    "comp_b": 1.0,
+    "comp_c": 3.0,
+    "uvec": -3.0,
+    "edge_pair": "[0, 14]",
+    "node_src": "n2",
+    "node_dst": "n3",
+    "start_ts": "2024-01-14 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-14 07:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_14",
+    "a\"b\\c_simplex_a": 30.0,
+    "a\"b\\c_simplex_b": 35.0,
+    "a\"b\\c_simplex_c": 35.0,
+    "short_text": "I absolutely love this, it completely made my day!",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 6.5,
+    "petal_width": 2.4000000000000004,
+    "a\"b\\c_species": 1,
+    "csv_list": "a14",
+    "mixed_case": "ABILITY",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 5,
+    "name": "dave",
+    "score": 2.0,
+    "open": 5.0,
+    "high": 6.5,
+    "low": 4.0,
+    "close": 5.5,
+    "iso_country": "GBR",
+    "trade_date": "2024-01-05",
+    "pvalue": 0.18581418581418582,
+    "log2fc": 0.0,
+    "comp_a": 1.0,
+    "comp_b": 6.0,
+    "comp_c": 3.0,
+    "uvec": 2.0,
+    "edge_pair": "[0, 5]",
+    "node_src": "n1",
+    "node_dst": "n2",
+    "start_ts": "2024-01-05 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-05 06:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_5",
+    "a\"b\\c_simplex_a": 25.0,
+    "a\"b\\c_simplex_b": 35.0,
+    "a\"b\\c_simplex_c": 40.0,
+    "short_text": "This is the worst experience I have ever had.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 1.3,
+    "petal_width": 0.2,
+    "a\"b\\c_species": 0,
+    "csv_list": "a5,b5",
+    "mixed_case": "202",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 0,
+    "species_name": "setosa",
+    "species_name_pred": "setosa"
+  },
+  {
+    "id": 9,
+    "name": "grace",
+    "score": 2.4,
+    "open": 9.0,
+    "high": 10.5,
+    "low": 8.0,
+    "close": 9.5,
+    "iso_country": "CAN",
+    "trade_date": "2024-01-09",
+    "pvalue": 0.3336663336663337,
+    "log2fc": 2.8,
+    "comp_a": 5.0,
+    "comp_b": 3.0,
+    "comp_c": 1.0,
+    "uvec": -1.0,
+    "edge_pair": "[0, 9]",
+    "node_src": "n1",
+    "node_dst": "n2",
+    "start_ts": "2024-01-09 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-09 02:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_9",
+    "a\"b\\c_simplex_a": 25.0,
+    "a\"b\\c_simplex_b": 25.0,
+    "a\"b\\c_simplex_c": 50.0,
+    "short_text": "I absolutely love this, it completely made my day!",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 6.5,
+    "petal_width": 2.4000000000000004,
+    "a\"b\\c_species": 1,
+    "csv_list": "a9,b9,c9",
+    "mixed_case": "abstract",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 0,
+    "species_name": "versicolor",
+    "species_name_pred": "setosa"
+  },
+  {
+    "id": 2,
+    "name": "alice",
+    "score": 3.1,
+    "open": 2.0,
+    "high": 3.5,
+    "low": 1.0,
+    "close": 2.5,
+    "iso_country": "CHN",
+    "trade_date": "2024-01-02",
+    "pvalue": 0.07492507492507493,
+    "log2fc": -2.0999999999999996,
+    "comp_a": 3.0,
+    "comp_b": 3.0,
+    "comp_c": 3.0,
+    "uvec": -1.0,
+    "edge_pair": "[0, 2]",
+    "node_src": "n2",
+    "node_dst": "n3",
+    "start_ts": "2024-01-02 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-02 03:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_2",
+    "a\"b\\c_simplex_a": 30.0,
+    "a\"b\\c_simplex_b": 35.0,
+    "a\"b\\c_simplex_c": 35.0,
+    "short_text": "Thank you so much, everything was absolutely perfect.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 3.9000000000000004,
+    "petal_width": 1.3,
+    "a\"b\\c_species": 1,
+    "csv_list": "a2,b2,c2,d2",
+    "mixed_case": "ABACUS",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 13,
+    "name": "bob",
+    "score": 2.8,
+    "open": 13.0,
+    "high": 14.5,
+    "low": 12.0,
+    "close": 13.5,
+    "iso_country": "JPN",
+    "trade_date": "2024-01-13",
+    "pvalue": 0.48151848151848153,
+    "log2fc": -2.0999999999999996,
+    "comp_a": 4.0,
+    "comp_b": 7.0,
+    "comp_c": 2.0,
+    "uvec": 3.0,
+    "edge_pair": "[0, 13]",
+    "node_src": "n1",
+    "node_dst": "n2",
+    "start_ts": "2024-01-13 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-13 06:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_13",
+    "a\"b\\c_simplex_a": 25.0,
+    "a\"b\\c_simplex_b": 30.0,
+    "a\"b\\c_simplex_c": 45.0,
+    "short_text": "Thank you so much, everything was absolutely perfect.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 5.2,
+    "petal_width": 1.85,
+    "a\"b\\c_species": 1,
+    "csv_list": "a13",
+    "mixed_case": "303",
+    "a\"b\\c_high_score": false,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 6,
+    "name": "1",
+    "score": 5.5,
+    "open": 6.0,
+    "high": 7.5,
+    "low": 5.0,
+    "close": 6.5,
+    "iso_country": "FRA",
+    "trade_date": "2024-01-06",
+    "pvalue": 0.22277722277722278,
+    "log2fc": 0.7,
+    "comp_a": 2.0,
+    "comp_b": 7.0,
+    "comp_c": 1.0,
+    "uvec": 3.0,
+    "edge_pair": "[0, 6]",
+    "node_src": "n2",
+    "node_dst": "n3",
+    "start_ts": "2024-01-06 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-06 07:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_6",
+    "a\"b\\c_simplex_a": 30.0,
+    "a\"b\\c_simplex_b": 25.0,
+    "a\"b\\c_simplex_c": 45.0,
+    "short_text": "I'm really not sure how I feel about it.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 2.6,
+    "petal_width": 0.75,
+    "a\"b\\c_species": 0,
+    "csv_list": "a6,b6",
+    "mixed_case": "abbey",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 0,
+    "species_name": "setosa",
+    "species_name_pred": "setosa"
+  },
+  {
+    "id": 10,
+    "name": "heidi",
+    "score": 4.2,
+    "open": 10.0,
+    "high": 11.5,
+    "low": 9.0,
+    "close": 10.5,
+    "iso_country": "AUS",
+    "trade_date": "2024-01-10",
+    "pvalue": 0.3706293706293706,
+    "log2fc": 3.5,
+    "comp_a": 1.0,
+    "comp_b": 4.0,
+    "comp_c": 2.0,
+    "uvec": 0.0,
+    "edge_pair": "[0, 10]",
+    "node_src": "n2",
+    "node_dst": "n3",
+    "start_ts": "2024-01-10 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-10 03:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_10",
+    "a\"b\\c_simplex_a": 30.0,
+    "a\"b\\c_simplex_b": 30.0,
+    "a\"b\\c_simplex_c": 40.0,
+    "short_text": "This is the worst experience I have ever had.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 1.3,
+    "petal_width": 0.2,
+    "a\"b\\c_species": 0,
+    "csv_list": "a10,b10,c10",
+    "mixed_case": "ABDOMEN",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "setosa",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 4,
+    "name": "carol",
+    "score": 4.8,
+    "open": 4.0,
+    "high": 5.5,
+    "low": 3.0,
+    "close": 4.5,
+    "iso_country": "DEU",
+    "trade_date": "2024-01-04",
+    "pvalue": 0.14885114885114886,
+    "log2fc": -0.7,
+    "comp_a": 5.0,
+    "comp_b": 5.0,
+    "comp_c": 2.0,
+    "uvec": 1.0,
+    "edge_pair": "[0, 4]",
+    "node_src": "n0",
+    "node_dst": "n1",
+    "start_ts": "2024-01-04 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-04 05:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_4",
+    "a\"b\\c_simplex_a": 20.0,
+    "a\"b\\c_simplex_b": 30.0,
+    "a\"b\\c_simplex_c": 50.0,
+    "short_text": "Congratulations! You have won a free prize, reply now to 
claim it.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 6.5,
+    "petal_width": 2.4000000000000004,
+    "a\"b\\c_species": 1,
+    "csv_list": "a4,b4,c4,d4",
+    "mixed_case": "404",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 15,
+    "name": "dave",
+    "score": 3.3,
+    "open": 15.0,
+    "high": 16.5,
+    "low": 14.0,
+    "close": 15.5,
+    "iso_country": "GBR",
+    "trade_date": "2024-01-15",
+    "pvalue": 0.5554445554445554,
+    "log2fc": -0.7,
+    "comp_a": 1.0,
+    "comp_b": 2.0,
+    "comp_c": 1.0,
+    "uvec": -2.0,
+    "edge_pair": "[0, 15]",
+    "node_src": "n3",
+    "node_dst": "n4",
+    "start_ts": "2024-01-15 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-15 08:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_15",
+    "a\"b\\c_simplex_a": 35.0,
+    "a\"b\\c_simplex_b": 25.0,
+    "a\"b\\c_simplex_c": 40.0,
+    "short_text": "URGENT: your account needs verification, click the link 
immediately.",
+    "long_text": "The city council voted on Tuesday to approve a new public 
transit plan. The proposal adds three bus routes and extends weekend service 
hours. Officials estimate the changes will serve ten thousand additional riders 
each month. The measure passed by a vote of seven to two after a long public 
comment period.",
+    "petal_length": 1.3,
+    "petal_width": 0.2,
+    "a\"b\\c_species": 0,
+    "csv_list": "a15",
+    "mixed_case": "ability",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 0,
+    "species_name": "setosa",
+    "species_name_pred": "setosa"
+  },
+  {
+    "id": 8,
+    "name": "frank",
+    "score": 3.7,
+    "open": 8.0,
+    "high": 9.5,
+    "low": 7.0,
+    "close": 8.5,
+    "iso_country": "BRA",
+    "trade_date": "2024-01-08",
+    "pvalue": 0.2967032967032967,
+    "log2fc": 2.0999999999999996,
+    "comp_a": 4.0,
+    "comp_b": 2.0,
+    "comp_c": 3.0,
+    "uvec": -2.0,
+    "edge_pair": "[0, 8]",
+    "node_src": "n0",
+    "node_dst": "n1",
+    "start_ts": "2024-01-08 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-08 01:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_8",
+    "a\"b\\c_simplex_a": 20.0,
+    "a\"b\\c_simplex_b": 35.0,
+    "a\"b\\c_simplex_c": 45.0,
+    "short_text": "See you at lunch, save me a seat by the window.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 5.2,
+    "petal_width": 1.85,
+    "a\"b\\c_species": 1,
+    "csv_list": "a8,b8",
+    "mixed_case": "ABSTRACT",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  },
+  {
+    "id": 12,
+    "name": "alice",
+    "score": 5.1,
+    "open": 12.0,
+    "high": 13.5,
+    "low": 11.0,
+    "close": 12.5,
+    "iso_country": "CHN",
+    "trade_date": "2024-01-12",
+    "pvalue": 0.44455544455544455,
+    "log2fc": -2.8,
+    "comp_a": 3.0,
+    "comp_b": 6.0,
+    "comp_c": 1.0,
+    "uvec": 2.0,
+    "edge_pair": "[0, 12]",
+    "node_src": "n0",
+    "node_dst": "n1",
+    "start_ts": "2024-01-12 00:00:00.0",
+    "a\"b\\c_finish_ts": "2024-01-12 05:00:00.0",
+    "a\"b'c\\d\ne_uniq_name": "cat_12",
+    "a\"b\\c_simplex_a": 20.0,
+    "a\"b\\c_simplex_b": 25.0,
+    "a\"b\\c_simplex_c": 55.0,
+    "short_text": "Congratulations! You have won a free prize, reply now to 
claim it.",
+    "long_text": "Researchers published a study this week describing a new way 
to recycle lithium batteries. The technique recovers most of the metal at a 
lower cost than existing methods. The team says a pilot plant could open within 
two years if funding is secured. Analysts called the results promising but 
cautioned that scaling up remains difficult.",
+    "petal_length": 3.9000000000000004,
+    "petal_width": 1.3,
+    "a\"b\\c_species": 1,
+    "csv_list": "a12,b12,c12",
+    "mixed_case": "505",
+    "a\"b\\c_high_score": true,
+    "a\"b\\c_species_pred": 1,
+    "species_name": "versicolor",
+    "species_name_pred": "versicolor"
+  }
+]
diff --git 
a/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixture.scala
 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixture.scala
new file mode 100644
index 0000000000..373e059641
--- /dev/null
+++ 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixture.scala
@@ -0,0 +1,251 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.texera.amber.translator.verify
+
+import com.fasterxml.jackson.databind.ObjectMapper
+import org.apache.texera.amber.core.tuple.{Attribute, AttributeType, Schema, 
Tuple}
+
+import java.sql.Timestamp
+import scala.jdk.CollectionConverters._
+
+/**
+  * The shared input dataset for auto-configured transform verification.
+  * Properties are deliberate (see CanonicalFixtureSpec): enough rows and
+  * partial port-0/port-1 overlap to defeat hash-coincidence false passes on
+  * set ops and joins, and the canonical value "1" present in some-but-not-all
+  * rows so ConfigGenerator-filled free-form predicates match a proper subset.
+  */
+object CanonicalFixture extends SharedFixture {
+
+  // Columns are semantically named and type-correct so @SampleColumn-tagged or
+  // type-constrained fields can be filled with realistic input (a valid OHLC
+  // block, real ISO country codes, real dates) instead of a degenerate
+  // first-column pick. Ordering is deliberate: id/name/score lead so the
+  // first-column fallback AND the type-rule tier ("first column of a matching
+  // type") are unchanged for un-annotated fields — the domain-specific columns
+  // that follow are only reached via an explicit @SampleColumn.
+  val schema: Schema = new Schema(
+    new Attribute("id", AttributeType.INTEGER),
+    new Attribute("name", AttributeType.STRING),
+    new Attribute("score", AttributeType.DOUBLE),
+    new Attribute("open", AttributeType.DOUBLE),
+    new Attribute("high", AttributeType.DOUBLE),
+    new Attribute("low", AttributeType.DOUBLE),
+    new Attribute("close", AttributeType.DOUBLE),
+    new Attribute("iso_country", AttributeType.STRING),
+    new Attribute("trade_date", AttributeType.STRING),
+    // --- Domain-specific columns (reached only via @SampleColumn) ---
+    new Attribute("pvalue", AttributeType.DOUBLE), // strictly in (0,1): 
p-values
+    new Attribute("log2fc", AttributeType.DOUBLE), // signed, centered on 0: 
fold-change
+    new Attribute("comp_a", AttributeType.DOUBLE), // >0 ternary simplex 
component
+    new Attribute("comp_b", AttributeType.DOUBLE), // >0 ternary simplex 
component
+    new Attribute("comp_c", AttributeType.DOUBLE), // >0 ternary simplex 
component
+    new Attribute("uvec", AttributeType.DOUBLE), // any real: a 4th numeric 
(Quiver u/v)
+    new Attribute(
+      "edge_pair",
+      AttributeType.STRING
+    ), // "[parent, child]" literals, single-rooted tree
+    new Attribute("node_src", AttributeType.STRING), // edge source id 
(Sankey/Network)
+    new Attribute("node_dst", AttributeType.STRING), // edge target id, 
overlaps node_src (a DAG)
+    new Attribute(
+      "start_ts",
+      AttributeType.TIMESTAMP
+    ), // real timestamp; Gantt start / TimeSeries axis
+    new Attribute(
+      "a\"b\\c_finish_ts",
+      AttributeType.TIMESTAMP
+    ), // always > start_ts; Gantt finish (bar width)
+    new Attribute(
+      "a\"b'c\\d\ne_uniq_name",
+      AttributeType.STRING
+    ), // distinct per row: Pie/name-keyed ops need no duplicates
+    new Attribute(
+      "a\"b\\c_simplex_a",
+      AttributeType.DOUBLE
+    ), // >0, and the three sum to 100 in every row (ternary-contour)
+    new Attribute("a\"b\\c_simplex_b", AttributeType.DOUBLE), // >0 simplex 
component summing to 100
+    new Attribute("a\"b\\c_simplex_c", AttributeType.DOUBLE), // >0 simplex 
component summing to 100
+    // ── text + iris-numeric columns for Hugging Face model operators ──
+    new Attribute(
+      "short_text",
+      AttributeType.STRING
+    ), // one sentence: sentiment / spam-detection input
+    new Attribute(
+      "long_text",
+      AttributeType.STRING
+    ), // a multi-sentence paragraph: summarization input
+    new Attribute("petal_length", AttributeType.DOUBLE), // iris petal length 
in cm (~1.3–6.5)
+    new Attribute("petal_width", AttributeType.DOUBLE), // iris petal width in 
cm (~0.2–2.4)
+    new Attribute(
+      "a\"b\\c_species",
+      AttributeType.INTEGER
+    ), // the 0/1 iris class, exactly `petal_length >= 3.9`: the label the 
sklearn
+    // families fit against, separable by the two petal columns above
+    new Attribute(
+      "csv_list",
+      AttributeType.STRING
+    ), // comma-delimited, 1–4 tokens per row: split/explode ops need real 
fan-out
+    new Attribute(
+      "mixed_case",
+      AttributeType.STRING
+    ), // a third lower-case, a third upper, a third letterless: a case flag 
has to
+    // change WHICH rows match, and on any other column it changes nothing
+    new Attribute(
+      "a\"b\\c_high_score",
+      AttributeType.BOOLEAN
+    ), // `score >= 3.0`, the table's only boolean. Its TEXT is what no other 
column
+    // can ask: the engine writes "true" where Python's str() writes "True". 
Six
+    // false and nine true, both present in either port window
+    new Attribute(
+      "a\"b\\c_species_pred",
+      AttributeType.INTEGER
+    ), // a predictor's guess at the label: the same 0/1 domain, wrong on a 
few rows.
+    // Scoring compares a PAIR of columns, and no single label column supplies 
one.
+    // Last so the first-unused fallback reaches it only after every other 
column
+    new Attribute("species_name", AttributeType.STRING), // the label spelled 
out
+    new Attribute(
+      "species_name_pred",
+      AttributeType.STRING
+    ) // the prediction spelled out. A scorer takes a string label as readily 
as a
+    // numeric one and names the class after it rather than after its 
position, so
+    // the pair exists a second time in text
+  )
+
+  // ── Data source ──
+  // The rows are NOT generated at runtime — they live in a single, checked-in,
+  // human-readable JSON file that IS the source of truth:
+  //   src/test/resources/verify/canonical_fixture.json   (15 rows, ids 1..15)
+  // Open it to see the exact table; edit it to change the data. The
+  // CanonicalFixtureSpec invariants guard every semantic constraint (valid 
OHLC
+  // block, pvalue ∈ (0,1), ternary parts summing to 100, each finish after 
its start,
+  // etc.), so a hand-edit that breaks one fails the build. `schema` above 
stays
+  // authoritative for column types: JSON has no TIMESTAMP, so the timestamps
+  // are stored as JDBC strings ("2024-01-01 00:00:00.0") and coerced back 
here.
+  private val fixtureResource = "/verify/canonical_fixture.json"
+
+  override val allRows: Vector[Tuple] = {
+    val stream = Option(getClass.getResourceAsStream(fixtureResource))
+      .getOrElse(sys.error(s"canonical fixture not found on classpath: 
$fixtureResource"))
+    val root =
+      try new ObjectMapper().readTree(stream)
+      finally stream.close()
+    root
+      .elements()
+      .asScala
+      .map { node =>
+        val b = Tuple.builder(schema)
+        schema.getAttributes.foreach { attr =>
+          val cell = node.get(attr.getName)
+          require(cell != null, s"fixture row missing column 
'${attr.getName}'")
+          val value: AnyRef = attr.getType match {
+            case AttributeType.INTEGER   => Int.box(cell.asInt())
+            case AttributeType.LONG      => Long.box(cell.asLong())
+            case AttributeType.DOUBLE    => Double.box(cell.asDouble())
+            case AttributeType.BOOLEAN   => Boolean.box(cell.asBoolean())
+            case AttributeType.TIMESTAMP => Timestamp.valueOf(cell.asText())
+            case _                       => cell.asText() // STRING
+          }
+          b.add(attr, value)
+        }
+        b.build()
+      }
+      .toVector
+  }
+
+  // Each port takes two thirds of the table, from opposite ends, so the ports
+  // overlap by the remaining third and no row sits outside both. The overlap 
is
+  // what stops joins and set ops passing by hash coincidence: ports holding 
the
+  // same rows make intersect and union the same answer, and disjoint ports 
make
+  // both empty, which a broken operator produces too. Two thirds rather than a
+  // fixed count so a row added to the JSON widens the windows instead of 
falling
+  // off the end, and a port stays well under the whole table — these windows 
are
+  // what every per-test Python run is sized by. At 15 rows this is positions 
0-9
+  // and 5-14. Rows sit out of id order in the file, so the windows are
+  // positional — not id ranges.
+  private def windowSize: Int = allRows.size * 2 / 3
+  def port0Rows: Seq[Tuple] = allRows.take(windowSize)
+  def port1Rows: Seq[Tuple] = allRows.takeRight(windowSize)
+
+  override def rowsFor(port: Int): Seq[Tuple] = if (port == 0) port0Rows else 
port1Rows
+
+  /** `id` keeps every value: it is what joins and set operations match on, and
+    * emptying it would change which rows pair up rather than what a null does.
+    */
+  override val keepFilled: Set[String] = Set("id")
+
+  /** This table as the sklearn families read it: the two petal columns and the
+    * species label, and nothing else, because `X = table.drop(target, axis=1)`
+    * hands `fit` every column that is not the target. The two features 
separate
+    * the classes exactly, so an estimator fits them without a tie to break.
+    */
+  val sklearnNumeric: SharedFixture = ProjectedFixture(
+    this,
+    Seq("petal_length", "petal_width", "a\"b\\c_species"),
+    keepFilled = Set("a\"b\\c_species")
+  )
+
+  /** This table minus `score`. An operator whose output column is named 
`score`
+    * by default cannot run here otherwise: it would create a column the input
+    * already holds, and the schema refuses the duplicate before the operator
+    * runs. Dropping the one column puts the DEFAULT config under test, which 
is
+    * the config a user gets, rather than a hand-written name chosen to dodge 
the
+    * clash.
+    */
+  val withoutScore: SharedFixture = ProjectedFixture(
+    this,
+    schema.getAttributeNames.filterNot(_ == "score"),
+    keepFilled = keepFilled
+  )
+
+  /** This table as a scorer reads it when the labels are text: the same pair 
as
+    * the numeric label and its prediction, spelled out, and nothing else. The 
scenario
+    * that takes it names the two columns itself, since the operator's 
`@SampleColumn`s
+    * name the numeric pair this projection does not carry.
+    */
+  val scorerTextLabels: SharedFixture = ProjectedFixture(
+    this,
+    Seq("species_name", "species_name_pred"),
+    keepFilled = Set.empty
+  )
+
+  /** [[sklearnNumeric]] plus a column an estimator cannot fit, so the 
narrowing to
+    * the fittable columns has something to do. The two paths narrow in 
different
+    * places, the operator once ahead of the port branch and the standalone 
script
+    * once per port, so each has to drop it on its own. The text carries no 
signal
+    * about the label, so the fit is the one the two petal columns already 
give.
+    */
+  val sklearnNumericWithText: SharedFixture = ProjectedFixture(
+    this,
+    Seq("petal_length", "petal_width", "short_text", "a\"b\\c_species"),
+    keepFilled = Set("a\"b\\c_species")
+  )
+
+  /** This table as the `countVectorizer=true` path reads it: one text column 
and
+    * the same label. `short_text` leads because the model probe feeds a text
+    * pipeline the frame's first column as a Series. Every row carrying a given
+    * sentence carries the same label (an invariant of the table), so the
+    * vectorized classes separate exactly, as the numeric pair does.
+    */
+  val sklearnText: SharedFixture = ProjectedFixture(
+    this,
+    Seq("short_text", "long_text", "a\"b\\c_species"),
+    keepFilled = Set("a\"b\\c_species")
+  )
+}
diff --git 
a/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixtureSpec.scala
 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixtureSpec.scala
new file mode 100644
index 0000000000..ba74f3df0d
--- /dev/null
+++ 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/CanonicalFixtureSpec.scala
@@ -0,0 +1,315 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.texera.amber.translator.verify
+
+import org.apache.texera.amber.core.workflow.PortIdentity
+import org.scalatest.flatspec.AnyFlatSpec
+import org.scalatest.matchers.should.Matchers
+
+import java.nio.file.Files
+
+class CanonicalFixtureSpec extends AnyFlatSpec with Matchers {
+
+  "CanonicalFixture" should "have at least 10 rows per port with partial id 
overlap" in {
+    CanonicalFixture.port0Rows.size should be >= 10
+    CanonicalFixture.port1Rows.size should be >= 10
+    val ids0 = CanonicalFixture.port0Rows.map(_.getField[Integer]("id")).toSet
+    val ids1 = CanonicalFixture.port1Rows.map(_.getField[Integer]("id")).toSet
+    (ids0 intersect ids1) should not be empty
+    (ids0 diff ids1) should not be empty
+    (ids1 diff ids0) should not be empty
+  }
+
+  // The windows are a rule over the table's size rather than fixed indices, so
+  // that a row added to the JSON is read by a port instead of falling off the
+  // end. Fixed indices made that silent: the row was simply never fed to
+  // anything, and every assertion above still held.
+  it should "leave no row outside both ports" in {
+    val onAPort = (CanonicalFixture.port0Rows ++ CanonicalFixture.port1Rows)
+      .map(_.getField[Integer]("id"))
+      .toSet
+    onAPort shouldBe 
CanonicalFixture.allRows.map(_.getField[Integer]("id")).toSet
+  }
+
+  it should "contain the canonical value \"1\" in some but not all name cells" 
in {
+    val names = CanonicalFixture.port0Rows.map(_.getField[String]("name"))
+    names.count(_ == "1") should be > 0
+    names.count(_ == "1") should be < names.size
+  }
+
+  it should "expose a valid OHLC block (high >= open/close >= low) for 
candlestick-style ops" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val o = t.getField[java.lang.Double]("open").doubleValue
+      val h = t.getField[java.lang.Double]("high").doubleValue
+      val l = t.getField[java.lang.Double]("low").doubleValue
+      val c = t.getField[java.lang.Double]("close").doubleValue
+      h should be >= math.max(o, c)
+      l should be <= math.min(o, c)
+    }
+  }
+
+  it should "keep pvalue strictly inside (0, 1) for probability-domain fields" 
in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val p = t.getField[java.lang.Double]("pvalue").doubleValue
+      p should (be > 0.0 and be < 1.0)
+    }
+  }
+
+  it should "keep log2fc signed and centered (both a negative and a positive 
present)" in {
+    val vals = 
CanonicalFixture.port0Rows.map(_.getField[java.lang.Double]("log2fc").doubleValue)
+    vals.min should be < 0.0
+    vals.max should be > 0.0
+  }
+
+  it should "keep ternary components strictly positive" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      t.getField[java.lang.Double]("comp_a").doubleValue should be > 0.0
+      t.getField[java.lang.Double]("comp_b").doubleValue should be > 0.0
+      t.getField[java.lang.Double]("comp_c").doubleValue should be > 0.0
+    }
+  }
+
+  it should "expose uniq_name as globally distinct so name-keyed ops have no 
duplicates" in {
+    val names = 
CanonicalFixture.port0Rows.map(_.getField[String]("a\"b'c\\d\ne_uniq_name"))
+    names.distinct.size shouldBe names.size
+  }
+
+  it should "expose a valid ternary simplex (positive parts summing to 100)" 
in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val a = t.getField[java.lang.Double]("a\"b\\c_simplex_a").doubleValue
+      val b = t.getField[java.lang.Double]("a\"b\\c_simplex_b").doubleValue
+      val c = t.getField[java.lang.Double]("a\"b\\c_simplex_c").doubleValue
+      a should be > 0.0
+      b should be > 0.0
+      c should be > 0.0
+      (a + b + c) shouldBe 100.0 +- 1e-9
+    }
+  }
+
+  it should "expose trade_date as a real ISO-8601 date (parseable, not the old 
day-N)" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val d = t.getField[String]("trade_date")
+      noException should be thrownBy java.time.LocalDate.parse(d)
+    }
+  }
+
+  it should "expose edge_pair as single-rooted 2-element list literals" in {
+    // Every cell is "[0, child]" → parses to a 2-list rooted at 0, so TreePlot
+    // builds one connected tree instead of an error page.
+    CanonicalFixture.port0Rows.foreach { t =>
+      t.getField[String]("edge_pair") should fullyMatch regex """\[0, \d+\]"""
+    }
+  }
+
+  it should "expose overlapping node_src/node_dst so graph ops have drawable 
edges" in {
+    val src = 
CanonicalFixture.port0Rows.map(_.getField[String]("node_src")).toSet
+    val dst = 
CanonicalFixture.port0Rows.map(_.getField[String]("node_dst")).toSet
+    (src intersect dst) should not be empty
+  }
+
+  it should "expose finish_ts strictly after start_ts (non-degenerate Gantt 
bar)" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val s = t.getField[java.sql.Timestamp]("start_ts")
+      val f = t.getField[java.sql.Timestamp]("a\"b\\c_finish_ts")
+      f.after(s) shouldBe true
+    }
+  }
+
+  it should "round-trip TIMESTAMP columns losslessly through TupleIO (write 
then read)" in {
+    val root = Files.createTempDirectory("canonical-fixture-ts-")
+    val path = CanonicalFixture.writeInputs(root, inputPortCount = 
1)(PortIdentity(0))
+    val schema = TupleIO.readSchemaSidecar(path)
+    val rows = TupleIO.readTuples(path, schema).toList
+    rows should not be empty
+    val read = rows.head
+    val orig = CanonicalFixture.port0Rows.head
+    // The JDBC-string codec is the exact inverse of Timestamp.toString, so the
+    // value read back equals the value written — no timezone drift.
+    read.getField[java.sql.Timestamp]("start_ts") shouldBe 
orig.getField[java.sql.Timestamp](
+      "start_ts"
+    )
+    read.getField[java.sql.Timestamp]("a\"b\\c_finish_ts") shouldBe orig
+      .getField[java.sql.Timestamp](
+        "a\"b\\c_finish_ts"
+      )
+  }
+
+  it should "expose non-empty short_text sentences for text-classification 
ops" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      t.getField[String]("short_text").trim should not be empty
+    }
+  }
+
+  it should "expose long_text with several sentences so summarization is 
non-trivial" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val txt = t.getField[String]("long_text")
+      // multiple sentence-terminating periods → real content to condense
+      txt.count(_ == '.') should be >= 2
+    }
+  }
+
+  // The sklearn families fit the label against the petal columns, so the table
+  // has to hold up as training data. The estimators that cross-validate pass 
no
+  // fold count and so take sklearn's default of five, and a class of fewer 
than
+  // five rows leaves a fold holding none of it — no error, just a warning and 
a
+  // fold that asks nothing. A label the features cannot separate is the other
+  // half: it leaves the fit breaking ties, which is where two paths drift 
apart.
+  it should "expose species as a petal-separable label with enough members to 
fold on" in {
+    val rows = CanonicalFixture.allRows
+    val byClass = 
rows.groupBy(_.getField[java.lang.Integer]("a\"b\\c_species").intValue)
+    byClass.keySet shouldBe Set(0, 1)
+    byClass.values.foreach(_.size should be >= 5)
+    rows.foreach { t =>
+      val large = t.getField[java.lang.Double]("petal_length").doubleValue >= 
3.9
+      t.getField[java.lang.Integer]("a\"b\\c_species").intValue shouldBe (if 
(large) 1 else 0)
+    }
+  }
+
+  // The prediction column exists so the scorer has a real pair to compare. 
All four
+  // cells of the confusion matrix have to be occupied: a perfect prediction
+  // scores every metric at 1.0, and one that never calls a class leaves that
+  // class's precision undefined — either way the metrics stop telling the two
+  // code paths apart.
+  it should "hold a species_pred that is right on most rows and wrong on some" 
in {
+    val cells = CanonicalFixture.allRows
+      .map(t =>
+        (
+          t.getField[java.lang.Integer]("a\"b\\c_species").intValue,
+          t.getField[java.lang.Integer]("a\"b\\c_species_pred").intValue
+        )
+      )
+      .distinct
+    cells should contain theSameElementsAs Seq((0, 0), (0, 1), (1, 0), (1, 1))
+  }
+
+  // The text pair exists to run the scorer's string-label path on the same
+  // arrangement the numeric pair gives it. Spelling out a different prediction
+  // would make the two paths score differently for a reason that has nothing 
to
+  // do with the label being text.
+  it should "spell out the species pair without changing what it says" in {
+    val name = Map(0 -> "setosa", 1 -> "versicolor")
+    CanonicalFixture.allRows.foreach { t =>
+      t.getField[String]("species_name") shouldBe name(
+        t.getField[java.lang.Integer]("a\"b\\c_species").intValue
+      )
+      t.getField[String]("species_name_pred") shouldBe name(
+        t.getField[java.lang.Integer]("a\"b\\c_species_pred").intValue
+      )
+    }
+  }
+
+  // The countVectorizer=true path fits the same label on short_text alone, 
and a
+  // sentence appearing under both labels makes that set unlearnable.
+  it should "keep every short_text sentence inside one species" in {
+    CanonicalFixture.allRows
+      .groupBy(_.getField[String]("short_text"))
+      .foreach {
+        case (sentence, rows) =>
+          withClue(s"$sentence: ") {
+            
rows.map(_.getField[java.lang.Integer]("a\"b\\c_species")).distinct.size 
shouldBe 1
+          }
+      }
+  }
+
+  it should "expose iris petal columns in a realistic centimetre range" in {
+    CanonicalFixture.port0Rows.foreach { t =>
+      val len = t.getField[java.lang.Double]("petal_length").doubleValue
+      val wid = t.getField[java.lang.Double]("petal_width").doubleValue
+      len should (be > 0.0 and be < 8.0)
+      wid should (be > 0.0 and be < 3.0)
+    }
+  }
+
+  // A single-token row would make split/explode a no-op, so both windows need
+  // rows that fan out AND a row that doesn't — the two branches of an unnest.
+  it should "expose csv_list as a clean delimited list with varying token 
counts" in {
+    Seq(CanonicalFixture.port0Rows, CanonicalFixture.port1Rows).foreach { rows 
=>
+      val tokenCounts = rows.map { t =>
+        val raw = t.getField[String]("csv_list")
+        raw should not startWith ","
+        raw should not endWith ","
+        val tokens = raw.split(",", -1)
+        tokens.foreach(_.trim should not be empty)
+        tokens.length
+      }
+      tokenCounts.min shouldBe 1
+      tokenCounts.max should be > 1
+    }
+  }
+
+  // A case flag is only worth sweeping where flipping it changes WHICH rows 
match,
+  // which needs rows of all three kinds in EVERY window a test reads — hence 
the
+  // per-port check rather than one over the whole table.
+  it should "expose mixed_case with lower, upper and letterless rows on every 
port" in {
+    Seq(CanonicalFixture.port0Rows, CanonicalFixture.port1Rows).foreach { rows 
=>
+      val values = rows.map(_.getField[String]("mixed_case"))
+      values.count(v => v.exists(_.isLower)) should be > 0
+      values.count(v => v.exists(_.isUpper) && !v.exists(_.isLower)) should be 
> 0
+      values.count(v => !v.exists(_.isLetter)) should be > 0
+    }
+  }
+
+  it should "hold high_score as score >= 3.0, both values present on every 
port" in {
+    CanonicalFixture.allRows.foreach { t =>
+      t.getField[java.lang.Boolean]("a\"b\\c_high_score") shouldBe
+        (t.getField[java.lang.Double]("score") >= 3.0)
+    }
+    Seq(CanonicalFixture.port0Rows, CanonicalFixture.port1Rows).foreach { rows 
=>
+      val values = 
rows.map(_.getField[java.lang.Boolean]("a\"b\\c_high_score"))
+      values.count(_ == true) should be > 0
+      values.count(_ == false) should be > 0
+    }
+  }
+
+  it should "write one JSONL fixture per requested input port" in {
+    val root = Files.createTempDirectory("canonical-fixture-")
+    val inputs = CanonicalFixture.writeInputs(root, inputPortCount = 2)
+    inputs.keySet shouldBe Set(PortIdentity(0), PortIdentity(1))
+    inputs.values.foreach(p => Files.size(p) should be > 0L)
+  }
+
+  it should "reject unsupported port counts" in {
+    val root = Files.createTempDirectory("canonical-fixture-")
+    an[IllegalArgumentException] should be thrownBy
+      CanonicalFixture.writeInputs(root, inputPortCount = 3)
+  }
+
+  it should "empty every column but id exactly once in the gapped table" in {
+    val rows = 
CanonicalFixture.emptyOneCellPerColumn(CanonicalFixture.port0Rows)
+    rows.size shouldBe CanonicalFixture.port0Rows.size
+
+    CanonicalFixture.schema.getAttributes.foreach { attr =>
+      val empties = rows.count(_.getField[AnyRef](attr.getName) == null)
+      // id carries the joins, so it keeps every value; everything else gets 
one
+      // hole, which is what makes the case a null case rather than an empty 
table.
+      if (attr.getName == "id") empties shouldBe 0
+      else empties shouldBe 1
+    }
+  }
+
+  it should "leave every row in the gapped table with something in it" in {
+    val names = CanonicalFixture.schema.getAttributes.map(_.getName)
+    CanonicalFixture.emptyOneCellPerColumn(CanonicalFixture.port0Rows).foreach 
{ t =>
+      // A wholly empty row would test the operator's handling of an empty 
table
+      // instead, and would say nothing about a null beside a filled neighbour.
+      names.count(n => t.getField[AnyRef](n) != null) should be > 0
+    }
+  }
+}
diff --git 
a/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/SharedFixture.scala
 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/SharedFixture.scala
new file mode 100644
index 0000000000..b6e773e1ca
--- /dev/null
+++ 
b/workflow-compiling-service/src/test/scala/org/apache/texera/amber/translator/verify/SharedFixture.scala
@@ -0,0 +1,225 @@
+/*
+ * Licensed to the Apache Software Foundation (ASF) under one
+ * or more contributor license agreements.  See the NOTICE file
+ * distributed with this work for additional information
+ * regarding copyright ownership.  The ASF licenses this file
+ * to you under the Apache License, Version 2.0 (the
+ * "License"); you may not use this file except in compliance
+ * with the License.  You may obtain a copy of the License at
+ *
+ *   http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing,
+ * software distributed under the License is distributed on an
+ * "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
+ * KIND, either express or implied.  See the License for the
+ * specific language governing permissions and limitations
+ * under the License.
+ */
+
+package org.apache.texera.amber.translator.verify
+
+import org.apache.texera.amber.core.tuple.{Schema, Tuple}
+import org.apache.texera.amber.core.workflow.PortIdentity
+
+import java.nio.file.Path
+
+/**
+  * A checked-in table a whole FAMILY of operators runs on, as opposed to one
+  * written for a single operator.
+  *
+  * Which table an operator runs on is its own axis, separate from who writes 
its
+  * config. [[CanonicalFixture]] is the wide mixed-type table every operator
+  * takes; the sklearn families take [[ProjectedFixture]] views of it.
+  */
+trait SharedFixture {
+
+  def schema: Schema
+
+  /** The rows port `port` gets. Ports may take different windows of the table
+    * (canonical overlaps them partially, to defeat hash-coincidence passes on
+    * joins) or the same rows twice.
+    */
+  def rowsFor(port: Int): Seq[Tuple]
+
+  /** Every row of the table, ports aside — what a [[ProjectedFixture]] of it
+    * narrows. A table whose ports read windows says so by overriding; by 
default
+    * a port already sees the whole of it.
+    */
+  def allRows: Seq[Tuple] = rowsFor(0)
+
+  /** Columns [[write]] never empties, because their VALUE is what the table 
was
+    * built to arrange rather than data under test: canonical's `id` is what 
joins
+    * and set operations pair rows on, and a sklearn table's label is what its
+    * estimator fits against. Emptying one of those changes what the test asks
+    * instead of asking what an operator does with a null.
+    */
+  def keepFilled: Set[String]
+
+  /** Write one JSONL file per 0-based input port under `dir`. At most 2 
ports. */
+  final def write(
+      dir: Path,
+      inputPortCount: Int,
+      withGaps: Boolean
+  ): Map[PortIdentity, Path] =
+    writeShaped(dir, inputPortCount, rows => if (withGaps) 
emptyOneCellPerColumn(rows) else rows)
+
+  /** [[write]] with the rows put through `shape` first, so a caller that 
wants a
+    * differently shaped table writes it the same way and gets the same 
sidecars.
+    */
+  private def writeShaped(
+      dir: Path,
+      inputPortCount: Int,
+      shape: Seq[Tuple] => Seq[Tuple]
+  ): Map[PortIdentity, Path] = {
+    require(
+      inputPortCount >= 1 && inputPortCount <= 2,
+      s"unsupported input port count: $inputPortCount"
+    )
+    (0 until inputPortCount).map { port =>
+      val path = dir.resolve(s"input_port_$port.jsonl")
+      TupleIO.writeTuples(path, shape(rowsFor(port)).iterator, schema)
+      PortIdentity(port) -> path
+    }.toMap
+  }
+
+  /** Schemas ConfigGenerator resolves @AutofillAttributeName fields against.
+    * Every port sees the same columns: a fixture's ports differ in which ROWS
+    * they get, not in shape.
+    */
+  final def schemasByPort: Map[Int, Schema] = Map(0 -> schema, 1 -> schema)
+
+  /** Write one JSONL fixture per 0-based input port, every cell filled. */
+  final def writeInputs(dir: Path, inputPortCount: Int): Map[PortIdentity, 
Path] =
+    write(dir, inputPortCount, withGaps = false)
+
+  /** The same columns with no rows under them. An upstream filter that matches
+    * nothing hands an operator exactly this, and it is not the same table as 
one
+    * with holes in it: a column with no values has no minimum, no quantile 
and no
+    * inferable type, so the code that reads one either answers with an empty
+    * result or raises.
+    */
+  final def writeEmptyInputs(dir: Path, inputPortCount: Int): 
Map[PortIdentity, Path] =
+    writeShaped(dir, inputPortCount, _ => Seq.empty)
+
+  /** How many rows port 0 gets — what a row-count-sensitive knob (`limit`,
+    * `offset`) is sized against so its value keeps some rows and drops some.
+    */
+  final def port0RowCount: Int = rowsFor(0).size
+
+  /** This table's rows with [[SharedFixture.emptyOneCellPerColumn]] applied. 
*/
+  private[verify] def emptyOneCellPerColumn(rows: Seq[Tuple]): Seq[Tuple] =
+    SharedFixture.emptyOneCellPerColumn(rows, schema, keepFilled)
+}
+
+/**
+  * A column subset of another table: the same rows in the same order, keeping
+  * only the named columns, in the order named.
+  *
+  * The sklearn families need one. Their generated code is
+  * `X = table.drop(target, axis=1)`, so every column that is not the target
+  * reaches `fit`, and a string or a timestamp ends it. A projection hands 
them a
+  * table an estimator can fit without a second dataset to keep in step: the 
rows
+  * are still [[CanonicalFixture]]'s, only narrower.
+  */
+final case class ProjectedFixture(
+    source: SharedFixture,
+    columns: Seq[String],
+    keepFilled: Set[String]
+) extends SharedFixture {
+
+  val schema: Schema = new Schema(columns.map(c => 
source.schema.getAttribute(c)): _*)
+
+  private val rows: Vector[Tuple] = source.allRows.map { t =>
+    val b = Tuple.builder(schema)
+    schema.getAttributes.foreach(a => b.add(a, t.getField[AnyRef](a.getName)))
+    b.build()
+  }.toVector
+
+  /** Every port gets the whole table. What the comparison sees is the fitted
+    * model, which port 1 has no hand in, so giving the ports different rows 
buys
+    * nothing.
+    *
+    * The whole table rather than the source's ten-row window, because a
+    * cross-validating estimator passes no fold count and takes sklearn's 
default
+    * of five: the window would leave the smaller class at four, and a fold
+    * holding none of a class asks nothing.
+    */
+  override def rowsFor(port: Int): Seq[Tuple] = rows
+}
+
+object HostileColumn {
+
+  /** Columns of [[CanonicalFixture]] whose NAMES hold the characters that end 
a
+    * Python string literal, so an operator pointed at one has to escape the 
name
+    * it writes. They are the table's own columns renamed rather than columns 
added
+    * beside them: the arrangements an operator needs already live here, and a
+    * second set under hostile names would be the same table twice.
+    */
+  private val prefix = "a\"b\\c_"
+
+  /** Three components summing to 100 in every row (the ternary family). */
+  val Numeric: Seq[String] = Seq("simplex_a", "simplex_b", 
"simplex_c").map(prefix + _)
+
+  val IntegerLike: Seq[String] = Seq(prefix + "species_pred")
+
+  /** Carries a single quote and a newline besides, the two that end a `'...'`
+    * literal and a `#` comment. Only one column needs them to put the 
question.
+    */
+  val Text: Seq[String] = Seq("a\"b'c\\d\ne_uniq_name")
+
+  val Timestamp: Seq[String] = Seq(prefix + "finish_ts")
+
+  val all: Seq[String] = Numeric ++ IntegerLike ++ Text ++ Timestamp
+
+  /** The ones carrying `t`, in the order a caller hands them to siblings. */
+  def forType(t: org.apache.texera.amber.core.tuple.AttributeType): 
Seq[String] = {
+    import org.apache.texera.amber.core.tuple.AttributeType._
+    t match {
+      case STRING         => Text
+      case TIMESTAMP      => Timestamp
+      case INTEGER | LONG => IntegerLike
+      case DOUBLE         => Numeric
+      case _              => Seq.empty
+    }
+  }
+}
+
+object SharedFixture {
+
+  /** One empty cell per column, spread across rows so no row is wholly empty 
— an
+    * operator that reads two columns should still meet a row where one is 
filled
+    * and the other is not. Placement is by column position, so it is the same 
on
+    * every run.
+    *
+    * Free-standing rather than a member, because a curated handler's table 
has no
+    * [[SharedFixture]] behind it: the runner reads back the rows the handler 
wrote
+    * and punches the holes here.
+    */
+  def emptyOneCellPerColumn(
+      rows: Seq[Tuple],
+      schema: Schema,
+      keepFilled: Set[String]
+  ): Seq[Tuple] = {
+    if (rows.isEmpty) return rows
+    val holes: Map[Int, Set[String]] = schema.getAttributes.zipWithIndex
+      .filterNot { case (attr, _) => keepFilled.contains(attr.getName) }
+      .map { case (attr, i) => (i % rows.size) -> attr.getName }
+      .groupBy(_._1)
+      .map { case (row, pairs) => row -> pairs.map(_._2).toSet }
+    rows.zipWithIndex.map {
+      case (t, rowIdx) =>
+        val emptied = holes.getOrElse(rowIdx, Set.empty)
+        if (emptied.isEmpty) t
+        else {
+          val b = Tuple.builder(schema)
+          schema.getAttributes.foreach { attr =>
+            val v: AnyRef =
+              if (emptied.contains(attr.getName)) null else 
t.getField[AnyRef](attr.getName)
+            b.add(attr, v)
+          }
+          b.build()
+        }
+    }
+  }
+}

Reply via email to