This is an automated email from the ASF dual-hosted git repository.

potiuk pushed a commit to branch main
in repository https://gitbox.apache.org/repos/asf/airflow.git


The following commit(s) were added to refs/heads/main by this push:
     new b3bcb388742 Recognize grouped PR references in provider changelog 
tooling (#73855)
b3bcb388742 is described below

commit b3bcb388742c645b5edc5dba3990f3e8a8a25584
Author: Shahar Epstein <[email protected]>
AuthorDate: Tue Sep 29 00:20:09 2026 +0300

    Recognize grouped PR references in provider changelog tooling (#73855)
    
    * Recognise grouped PR references in provider changelog tooling
    
    Changelog entries sometimes credit several PRs at once, e.g.
    ``(#73367, #73368)``. The provider changelog generator only looked for
    the literal ``(#NNN)`` when deciding whether a commit was already
    listed, so such entries were appended again on the next run for the
    same version. The duplicate-entries prek hook had the mirror-image
    gap and skipped those lines entirely, so a PR repeated across a
    grouped and a standalone entry went unreported.
    
    * Allow the intentional double listing of #73368 in the common.ai changelog
    
    #73850 lists #73368 both under breaking changes and inside a grouped
    feature entry, which the grouped-reference parsing now detects.
    
    * Count only bracketed reference groups as listed PRs in the changelog 
generator
    
    Matching any "#N" before a comma or closing bracket also picked up prose
    such as "(continuation of #57320)", so a commit with that number would be
    silently dropped from the regenerated changelog. The generator now uses
    the same reference-group shape as the changelog duplicates hook.
    
    * Keep an older version section in the grouped-reference changelog fixture
---
 .../prepare_providers/provider_documentation.py    | 11 +++--
 dev/breeze/tests/test_provider_documentation.py    | 48 ++++++++++++++++++++++
 scripts/ci/prek/changelog_duplicates.py            | 11 ++++-
 scripts/tests/ci/prek/test_changelog_duplicates.py | 36 ++++++++++------
 4 files changed, 88 insertions(+), 18 deletions(-)

diff --git 
a/dev/breeze/src/airflow_breeze/prepare_providers/provider_documentation.py 
b/dev/breeze/src/airflow_breeze/prepare_providers/provider_documentation.py
index 4f16be2471e..b4a3eab67d8 100644
--- a/dev/breeze/src/airflow_breeze/prepare_providers/provider_documentation.py
+++ b/dev/breeze/src/airflow_breeze/prepare_providers/provider_documentation.py
@@ -1171,9 +1171,14 @@ def _generate_new_changelog(
                 "has first release. Not updating the changelog.[/]"
             )
             return
-        new_changes = [
-            change for change in changes[0] if change.pr and "(#" + change.pr 
+ ")" not in current_changelog
-        ]
+        # Entries may reference several PRs at once, e.g. ``(#123, #456)``. 
Only bracketed
+        # reference groups count: prose like ``(continuation of #123)`` does 
not list a PR.
+        existing_prs = {
+            pr
+            for group in re.findall(r"\(((?:#\d+, )*#\d+)\)", 
current_changelog)
+            for pr in group.replace("#", "").split(", ")
+        }
+        new_changes = [change for change in changes[0] if change.pr and 
change.pr not in existing_prs]
         if not new_changes:
             console_print(
                 f"[success]The provider {package_id} changelog for 
`{latest_version}` "
diff --git a/dev/breeze/tests/test_provider_documentation.py 
b/dev/breeze/tests/test_provider_documentation.py
index f8203b2fc6c..851ff7f79df 100644
--- a/dev/breeze/tests/test_provider_documentation.py
+++ b/dev/breeze/tests/test_provider_documentation.py
@@ -34,6 +34,7 @@ from airflow_breeze.prepare_providers.provider_documentation 
import (
     TypeOfChange,
     _convert_git_changes_to_table,
     _find_insertion_index_for_version,
+    _generate_new_changelog,
     _get_change_from_line,
     _get_changes_classified,
     _get_git_log_command,
@@ -100,6 +101,53 @@ def test_find_insertion_index_insert_new_changelog():
     assert index == 3
 
 
+def test_generate_new_changelog_recognises_grouped_pr_references(tmp_path):
+    changelog_path = tmp_path / "changelog.rst"
+    changelog_path.write_text(
+        """
+Changelog
+---------
+
+5.0.0
+.....
+
+Features
+~~~~~~~~
+
+* ``Add X (#1001, #1002)``
+* ``Add Y (#1003)``
+* ``Add Z (continuation of #1005) (#1006)``
+
+4.7.0
+.....
+
+* ``Old (#900)``
+"""
+    )
+    provider_details = mock.MagicMock(
+        spec=ProviderPackageDetails, versions=["5.0.0"], 
changelog_path=changelog_path
+    )
+    changes = [
+        Change("hash", "short", "2024-01-01", "5.0.0", f"Fix (#{pr})", f"Fix 
(#{pr})", pr)
+        for pr in ("1001", "1002", "1003", "1004", "1005")
+    ]
+
+    _generate_new_changelog(
+        package_id="asana",
+        provider_details=provider_details,
+        changes=[changes],
+        context={},
+        with_breaking_changes=False,
+        maybe_with_new_features=False,
+    )
+
+    new_changelog = changelog_path.read_text()
+    assert "* ``Fix (#1004)``" in new_changelog
+    assert "* ``Fix (#1005)``" in new_changelog
+    for pr in ("1001", "1002", "1003"):
+        assert new_changelog.count(f"#{pr}") == 1
+
+
 @pytest.mark.parametrize(
     ("version", "provider_id", "suffix", "tag"),
     [
diff --git a/scripts/ci/prek/changelog_duplicates.py 
b/scripts/ci/prek/changelog_duplicates.py
index e3b7f636911..fadcedccb46 100755
--- a/scripts/ci/prek/changelog_duplicates.py
+++ b/scripts/ci/prek/changelog_duplicates.py
@@ -34,9 +34,16 @@ known_exceptions = [
     "3946",  # Commits tagged to the same PR for both 1.10.2 and 1.10.3
     "4260",  # Commits tagged to the same PR for both 1.10.2 and 1.10.3
     "13153",  # Both a bugfix and a feature
+    "73368",  # Both a breaking change and part of a grouped feature entry in 
common.ai 0.10.0
 ]
 
-pr_number_re = re.compile(r".*\(#([0-9]{1,6})\)`?`?$")
+pr_numbers_re = re.compile(r"\(((?:#[0-9]{1,6}, )*#[0-9]{1,6})\)`?`?$")
+
+
+def extract_pr_numbers(line: str) -> list[str]:
+    """Return the PR numbers ending a changelog entry: ``(#123)`` or ``(#123, 
#456)``."""
+    match = pr_numbers_re.search(line)
+    return match.group(1).replace("#", "").split(", ") if match else []
 
 
 def find_duplicates(lines: list[str]) -> list[str]:
@@ -44,7 +51,7 @@ def find_duplicates(lines: list[str]) -> list[str]:
     seen: list[str] = []
     dups: list[str] = []
     for line in lines:
-        if (match := pr_number_re.search(line)) and (pr := match.group(1)):
+        for pr in extract_pr_numbers(line):
             if pr not in seen:
                 seen.append(pr)
             elif pr not in known_exceptions:
diff --git a/scripts/tests/ci/prek/test_changelog_duplicates.py 
b/scripts/tests/ci/prek/test_changelog_duplicates.py
index 8cb097e3b28..17798175136 100644
--- a/scripts/tests/ci/prek/test_changelog_duplicates.py
+++ b/scripts/tests/ci/prek/test_changelog_duplicates.py
@@ -17,24 +17,24 @@
 from __future__ import annotations
 
 import pytest
-from ci.prek.changelog_duplicates import find_duplicates, known_exceptions, 
pr_number_re
+from ci.prek.changelog_duplicates import extract_pr_numbers, find_duplicates, 
known_exceptions
 
 
-class TestPrNumberRegex:
+class TestExtractPrNumbers:
     @pytest.mark.parametrize(
-        "line, expected_pr",
+        "line, expected_prs",
         [
-            ("* Fix something (#12345)", "12345"),
-            ("* Fix something (#1)", "1"),
-            ("* Fix something (#123456)", "123456"),
-            ("Some change (#99999)`", "99999"),
-            ("Some change (#99999)``", "99999"),
+            ("* Fix something (#12345)", ["12345"]),
+            ("* Fix something (#1)", ["1"]),
+            ("* Fix something (#123456)", ["123456"]),
+            ("Some change (#99999)`", ["99999"]),
+            ("Some change (#99999)``", ["99999"]),
+            ("* ``Fix something (#12345, #67890)``", ["12345", "67890"]),
+            ("* Fix something (#1, #2, #3)", ["1", "2", "3"]),
         ],
     )
-    def test_matches_valid_pr_numbers(self, line, expected_pr):
-        match = pr_number_re.search(line)
-        assert match is not None
-        assert match.group(1) == expected_pr
+    def test_extracts_pr_numbers(self, line, expected_prs):
+        assert extract_pr_numbers(line) == expected_prs
 
     @pytest.mark.parametrize(
         "line",
@@ -47,7 +47,7 @@ class TestPrNumberRegex:
         ],
     )
     def test_no_match(self, line):
-        assert pr_number_re.search(line) is None
+        assert extract_pr_numbers(line) == []
 
 
 class TestFindDuplicates:
@@ -95,6 +95,16 @@ class TestFindDuplicates:
     def test_empty_input(self):
         assert find_duplicates([]) == []
 
+    @pytest.mark.parametrize(
+        "lines, expected",
+        [
+            (["* Fix A (#1001, #1002)", "* Fix B (#1002)"], ["1002"]),
+            (["* Fix A (#1001, #1002)", "* Fix B (#1001, #1003)"], ["1001"]),
+        ],
+    )
+    def test_grouped_pr_numbers(self, lines, expected):
+        assert find_duplicates(lines) == expected
+
     def test_all_known_exceptions_are_strings(self):
         for exc in known_exceptions:
             assert isinstance(exc, str)

Reply via email to