From 3b4e925eee668f5892cd087846325d03a88b569a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Lehmann?= Date: Thu, 8 Oct 2026 09:41:12 +0200 Subject: [PATCH 1/2] pkg_in_pipe: log the progress, the cache usage and the source problems during the generation (#849) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Fix plane connection error blocking the script execution It's meant to generate a message in the report, not block the report generation Signed-off-by: Gaëtan Lehmann * pkg_in_pipe: log the progress, the cache usage and the source problems during the generation Add a --debug option and log the koji tag and the build being processed, whether values come from the cache or are fetched again, and warn when the plane tickets or github can't be reached, using the same logger setup as gen-dnf-proxy.py. Signed-off-by: Gaëtan Lehmann --------- Signed-off-by: Gaëtan Lehmann --- scripts/pkg_in_pipe/README.md | 2 + scripts/pkg_in_pipe/pkg_in_pipe.py | 81 +++++++++++++++++++++++++----- 2 files changed, 71 insertions(+), 12 deletions(-) diff --git a/scripts/pkg_in_pipe/README.md b/scripts/pkg_in_pipe/README.md index 66b4d31d..8a69a086 100644 --- a/scripts/pkg_in_pipe/README.md +++ b/scripts/pkg_in_pipe/README.md @@ -16,6 +16,8 @@ environment variable or the `--plane-token` command line option. An extra `--generated-info` command line option may be used to add some info about the report generation process. +The `--debug` option logs more details about the generation on stderr. + A machine readable version of the report can also be generated with the `--json-output` option: ```sh diff --git a/scripts/pkg_in_pipe/pkg_in_pipe.py b/scripts/pkg_in_pipe/pkg_in_pipe.py index 7451c89a..adc1ffa5 100755 --- a/scripts/pkg_in_pipe/pkg_in_pipe.py +++ b/scripts/pkg_in_pipe/pkg_in_pipe.py @@ -2,8 +2,10 @@ import argparse import io import json +import logging import os import re +import sys import tomllib from collections import defaultdict from datetime import datetime @@ -19,6 +21,31 @@ from github.GithubException import BadCredentialsException from github.PullRequest import PullRequest +log = logging.getLogger(__name__) + + +class ColorFormatter(logging.Formatter): + colors = { + logging.CRITICAL: '\033[31m', + logging.ERROR: '\033[31m', + logging.WARNING: '\033[33m', + logging.DEBUG: '\033[34m', + } + + def __init__(self, use_color: bool) -> None: + super().__init__(fmt='{message}', style='{') + self.use_color = use_color + + def format(self, record: logging.LogRecord) -> str: + message = super().format(record) + if record.levelno == logging.INFO: + return message + + prefix = f'{record.levelname.lower()}: ' + if color := self.colors.get(record.levelno) if self.use_color else None: + prefix = f'{color}{prefix}\033[0m' + return '\n'.join(f'{prefix}{line}' for line in message.splitlines() or ['']) + with open(os.path.join(os.path.dirname(__file__), 'missing_sources.toml'), 'rb') as missing_sources_file: MISSING_SOURCES = tomllib.load(missing_sources_file) @@ -254,6 +281,21 @@ def find_released_build(build_tag, package_name): return get_koji_build(max(tagged, key=lambda t: t['build_id'])['build_id']) return None +def cached(key): + """Return the value cached for that key, or None when it must be fetched again.""" + if args.re_cache: + log.debug("cache ignored because of --re-cache: %s", key) + return None + if key in CACHE: + log.debug("cache hit: %s", key) + return CACHE[key] + log.debug("cache miss: %s", key) + return None + +def cache(key, value): + log.debug("caching: %s", key) + CACHE.set(key, value, expire=RETENTION_TIME) + def find_commits(gh, repo, start_sha, end_sha) -> list[Commit]: """ List the commits in the range [start_sha,end_sha[. @@ -262,15 +304,15 @@ def find_commits(gh, repo, start_sha, end_sha) -> list[Commit]: A commit older that the end_sha commit and added by a merge commit won't appear in this list. """ cache_key = f'commits-2-{start_sha}-{end_sha}' - if not args.re_cache and cache_key in CACHE: - return cast(list[Commit], CACHE[cache_key]) + if (commits := cast(list[Commit], cached(cache_key))) is not None: + return commits commits = [] if gh: for commit in gh.get_repo(repo).get_commits(start_sha): if commit.sha == end_sha: break commits.append(commit) - CACHE.set(cache_key, commits, expire=RETENTION_TIME) + cache(cache_key, commits) return commits def find_pull_requests(repo, start_sha, end_sha): @@ -278,8 +320,8 @@ def find_pull_requests(repo, start_sha, end_sha): prs = set() for commit in find_commits(GITHUB, repo, start_sha, end_sha): cache_key = f'commit-prs-6-{commit.sha}' - if not args.re_cache and cache_key in CACHE: - prs.update(cast(list[PullRequest], CACHE[cache_key])) + if (commit_prs := cast(list[PullRequest], cached(cache_key))) is not None: + prs.update(commit_prs) elif GITHUB: commit_prs = list(commit.get_pulls()) if not commit_prs: @@ -295,17 +337,17 @@ def find_pull_requests(repo, start_sha, end_sha): # commit. The latter is needed for the PRs merged by rebase or squash, whose merge # commit on the base branch is not one of the PR commits. commit_prs = [pr for pr in commit_prs if commit in pr.get_commits() or commit.sha == pr.merge_commit_sha] - CACHE.set(cache_key, commit_prs, expire=RETENTION_TIME) + cache(cache_key, commit_prs) prs.update(commit_prs) return sorted(prs, key=lambda p: p.number, reverse=True) def get_koji_build(build_id) -> dict: cache_key = f'koji-build-{build_id}' - if not args.re_cache and cache_key in CACHE: - return cast(dict, CACHE[cache_key]) + if (build := cast(dict, cached(cache_key))) is not None: + return build else: build = KOJI.getBuild(build_id) - CACHE.set(cache_key, build, expire=RETENTION_TIME) + cache(cache_key, build) return build def get_plane_issues(plane_token): @@ -409,8 +451,15 @@ def get_plane_issues_with_milestones(plane_token): ) parser.add_argument('--re-cache', help="Refresh the cache", action='store_true') parser.add_argument('--json-output', help="Also write a machine readable report in json format to this path") +parser.add_argument('--debug', help="Log more details on stderr", action='store_true') args = parser.parse_args() +handler = logging.StreamHandler() +handler.setFormatter(ColorFormatter( + not os.environ.get('NO_COLOR') and (bool(os.environ.get('FORCE_COLOR')) or sys.stderr.isatty()), +)) +logging.basicConfig(level=logging.DEBUG if args.debug else logging.INFO, handlers=[handler]) + CACHE = diskcache.Cache(args.cache) RETENTION_TIME = 24 * 60 * 60 # 24 hours @@ -426,9 +475,9 @@ def get_plane_issues_with_milestones(plane_token): # load the issues from plane, so we can search for the plane card related to a build try: issues = get_plane_issues_with_milestones(args.plane_token) -except Exception: +except Exception as e: issues = [] - raise + log.warning("failed to load the tickets from plane: %s", e) # connect to github GITHUB = None @@ -438,6 +487,12 @@ def get_plane_issues_with_milestones(plane_token): GITHUB.get_repo('xcp-ng/xcp') # check that the token is valid except BadCredentialsException: GITHUB = None + log.warning("the github token is invalid") + except github.GithubException as e: + GITHUB = None + log.warning("failed to connect to github: %s", e) +else: + log.warning("no github token, the pull requests come from the cache") # load the packages maintainers with urlopen('https://github.com/xcp-ng/xcp/raw/refs/heads/master/scripts/rpm_owners/packages.json') as f: @@ -464,6 +519,7 @@ def get_plane_issues_with_milestones(plane_token): KOJI = koji.ClientSession('https://kojihub.xcp-ng.org', config) KOJI.ssl_login(config['cert'], None, config['serverca']) for tag in tags: + log.info('processing tag %s', tag) tag_data = {'tag': tag, 'builds': []} report_data['tags'].append(tag_data) tag_history = dict( @@ -475,6 +531,8 @@ def get_plane_issues_with_milestones(plane_token): taggeds = (t for t in taggeds if t['package_name'] in args.packages or args.packages == []) taggeds = sorted(taggeds, key=lambda t: (tag_history[t['build_id']], t['build_id']), reverse=True) for tagged in taggeds: + build_url = f'https://koji.xcp-ng.org/buildinfo?buildID={tagged["build_id"]}' + log.info(' processing build %s (%s)', tagged['nvr'], build_url) build = get_koji_build(tagged['build_id']) prs: list[PullRequest] = [] maintained_by = None @@ -484,7 +542,6 @@ def get_plane_issues_with_milestones(plane_token): (repo, sha) = parse_source(build['source']) prs = find_pull_requests(repo, sha, previous_build_sha) maintained_by = PACKAGES.get(tagged['package_name'], {}).get('maintainer') - build_url = f'https://koji.xcp-ng.org/buildinfo?buildID={tagged["build_id"]}' build_issues = filter_issues(issues, [build_url] + [pr.html_url for pr in prs]) print_table_line( temp_out, tagged['nvr'], build_url, build_issues, tagged['owner_name'], prs, maintained_by From c5a1fc0444c211cee0e46250651a435478bdc839 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ga=C3=ABtan=20Lehmann?= Date: Wed, 26 Aug 2026 15:12:41 +0200 Subject: [PATCH 2/2] pkg_in_pipe: generate a draft of the release post from the report MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add release_post.py, which writes the "What changed" and "Versions" sections of the XCP-ng release post from the json report, with the change descriptions for the users taken from the related pull requests. Signed-off-by: Gaëtan Lehmann --- scripts/pkg_in_pipe/README.md | 45 ++++ scripts/pkg_in_pipe/release_post.py | 320 ++++++++++++++++++++++++++++ 2 files changed, 365 insertions(+) create mode 100644 scripts/pkg_in_pipe/release_post.py diff --git a/scripts/pkg_in_pipe/README.md b/scripts/pkg_in_pipe/README.md index 8a69a086..ee141f22 100644 --- a/scripts/pkg_in_pipe/README.md +++ b/scripts/pkg_in_pipe/README.md @@ -30,6 +30,51 @@ The json report can be validated against its schema with: python -m jsonschema -i report.json pkg_in_pipe.schema.json ``` +# Release post generator + +The `release_post.py` script generates the whole XCP-ng release post +from the json report. For each package of a given koji tag, it fetches the descriptions of the +related pull requests from github and prints the `Explain the change to users` section of those +descriptions. + +It needs the `pydantic`, `requests` and `tqdm` python modules. +An optional `--github-token` option (or `GITHUB_TOKEN` environment variable) is used to +avoid the github api rate limits. + +```sh +pkg_in_pipe --json-output report.json +release_post.py --report report.json +``` + +The release post is written on the standard output, as a full post template: the "What changed" +section contains one item per package with the `Explain the change to users` section of the pull +request descriptions, printed verbatim. The post writer can then +reorganize the items into the usual categories. The "Versions" +section lists every package of the tag with its version, showing the previously released +version as well when the report knows it (the `previous_nvr` field, the newest build of the +package in the updates or base tag). The `--version` option overrides the version number of +the post, otherwise it is derived from the tag (e.g. `v8.3-ci` gives 8.3). + +The verbatim version is printed as a single list item: a one line section is written right +after the package name, a multiline section is written entirely on the following lines, +indented under the package name (blank lines are preserved). The HTML comments left in the +template of the pull request descriptions, and the blank lines around them, are removed. +A package without that section still gets an entry, with an empty description, and a +warning is printed on the standard error. For example: + +```markdown +- `xo-lite`: * Update the UiTitle component to use the one from web-core (PR #9869) + +- `amd-microcode`: + Update to 2026-05-19 drop as redistributed by XenServer + Updated CPUs: + BRH-C1 00b00f21: 2025-10-17, rev 0b002161 -> 2025-10-17, rev 0b002162 +``` + +The pull request descriptions are cached (same cache as the report +generator, in `/tmp/pkg_in_pipe.cache`, 24 hours retention). Use `--cache` to use another cache +path and `--re-cache` to refresh the cache. + # Run in docker Before running in docker, the docker image must be built with: diff --git a/scripts/pkg_in_pipe/release_post.py b/scripts/pkg_in_pipe/release_post.py new file mode 100644 index 00000000..97c673dc --- /dev/null +++ b/scripts/pkg_in_pipe/release_post.py @@ -0,0 +1,320 @@ +#!/usr/bin/env python +"""Generate the package update section of the XCP-ng release post. + +Read the json report generated by pkg_in_pipe.py and, for each package of a +given koji tag, print the "Explain the change to users" sections of the pull +requests related to the package builds. +""" + +from __future__ import annotations + +import argparse +import os +import re +import signal +import sys +from collections import defaultdict +from datetime import datetime +from pathlib import Path +from string import Template +from textwrap import dedent +from typing import Literal, cast + +import diskcache # type: ignore[import-untyped] +import requests +from pydantic import BaseModel +from tqdm import tqdm + + +class Warnings(BaseModel): + plane: bool + github: bool + + +class Issue(BaseModel): + sequence_id: int + url: str + milestones: list[str] + + +class PullRequest(BaseModel): + number: int + title: str + url: str + linked: bool + + +class Build(BaseModel): + nvr: str + previous_nvr: str | None = None + package: str + url: str + built_by: str + maintained_by: str | None + issues: list[Issue] + pull_requests: list[PullRequest] + build_linked: bool + + +class TagReport(BaseModel): + tag: str + builds: list[Build] + + +class Report(BaseModel): + generated_at: datetime + generated_info: str | None + warnings: Warnings + error: Literal['koji', 'unknown'] | None + tags: list[TagReport] + + +PR_URL_RE = re.compile(r'^https://github\.com/([^/]+)/([^/]+)/pull/(\d+)$') + +RETENTION_TIME = 24 * 60 * 60 # 24 hours + + +def find_tag_report(report: Report, tag: str) -> TagReport: + for tag_report in report.tags: + if tag_report.tag == tag: + return tag_report + raise SystemExit(f'error: the tag {tag} is not present in the report') + + +def prs_by_package(tag_report: TagReport) -> dict[str, list[PullRequest]]: + prs_by_package: dict[str, list[PullRequest]] = defaultdict(list) + for build in tag_report.builds: + for pr in build.pull_requests: + if all(existing.url != pr.url for existing in prs_by_package[build.package]): + prs_by_package[build.package].append(pr) + return dict(prs_by_package) + + +def fetch_pr_description( + url: str, token: str | None, re_cache: bool, cache: diskcache.Cache +) -> str | None: + cache_key = f'pr-body-1-{url}' + if not re_cache and cache_key in cache: + return cast(str | None, cache[cache_key]) + match = PR_URL_RE.fullmatch(url) + if match is None: + raise RuntimeError(f'not a github pull request url: {url}') + owner, repo, number = cast(tuple[str, str, str], match.groups()) + headers = {'Accept': 'application/vnd.github+json'} + if token: + headers['Authorization'] = f'Bearer {token}' + response = requests.get(f'https://api.github.com/repos/{owner}/{repo}/pulls/{number}', headers=headers, timeout=30) + if response.status_code != 200: + raise RuntimeError(f'got a {response.status_code} response') + body = cast(str | None, response.json().get('body')) + cache.set(cache_key, body, expire=RETENTION_TIME) + return body + + +def fetch_pr_descriptions( + prs: list[PullRequest], token: str | None, re_cache: bool, cache: diskcache.Cache +) -> list[tuple[PullRequest, str]]: + descriptions: list[tuple[PullRequest, str]] = [] + for pr in prs: + try: + description = fetch_pr_description(pr.url, token, re_cache, cache) + except (RuntimeError, requests.RequestException) as e: + print(f'warning: could not fetch the description of {pr.url}: {e}', file=sys.stderr) + continue + if description is None: + print(f'warning: the pull request {pr.url} has no description', file=sys.stderr) + continue + descriptions.append((pr, description)) + return descriptions + + +USER_SECTION_RE = re.compile(r'(?i)^#{1,6}\s*Explain the change to users\s*$') +HEADING_RE = re.compile(r'^#{1,6}(\s|$)') +CODE_FENCE_RE = re.compile(r'^\s*(```+|~~~+)') + + +def extract_user_section(description: str) -> list[str] | None: + """Return the lines of the "Explain the change to users" section of a pull request description. + + The section is returned verbatim, except for the HTML comments (template boilerplate) + and the blank lines around them, which are removed. + """ + in_code = False + section: list[str] | None = None + for line in description.splitlines(): + fence = CODE_FENCE_RE.match(line) + if fence is not None: + in_code = not in_code + if section is not None: + if not in_code and HEADING_RE.match(line): + break + section.append(line) + elif not in_code and USER_SECTION_RE.match(line): + section = [] + if section is None: + return None + section = strip_html_comments(section) + while section and not section[0].strip(): + section.pop(0) + while section and not section[-1].strip(): + section.pop() + return section or None + + +def strip_html_comments(lines: list[str]) -> list[str]: + """Remove the HTML comments (template boilerplate) and the blank lines around them.""" + res: list[str] = [] + in_comment = False + skip_blanks = False + for line in lines: + if in_comment: + if '-->' in line: + in_comment = False + skip_blanks = True + continue + if '' not in line + skip_blanks = True + while res and not res[-1].strip(): + res.pop() + continue + if skip_blanks and not line.strip(): + continue + skip_blanks = False + res.append(line) + while res and not res[-1].strip(): + res.pop() + return res + + +def warn(pbar: tqdm, message: str) -> None: + pbar.clear() + print(f'warning: {message}', file=sys.stderr) + + +def collect_sections(descriptions: list[tuple[PullRequest, str]], pbar: tqdm) -> list[str]: + lines: list[str] = [] + for pr, description in descriptions: + section = extract_user_section(description) + if section is None: + warn(pbar, f'no "Explain the change to users" section in {pr.url}') + continue + lines.extend(section) + return lines + + +def verbatim_lines(package: str, lines: list[str]) -> list[str]: + if len(lines) > 1: + return [f'- `{package}`:'] + [f'\t{line}' if line else '' for line in lines] + if lines: + return [f'- `{package}`: {lines[0]}'] + return [f'- `{package}`:'] + + +def versions_by_package(tag_report: TagReport) -> list[tuple[str, str | None, str]]: + """Return the latest build of each package, as (package, previous_nvr, nvr), sorted by package name. + + The report lists the builds of a tag newest first, so the first occurrence of a package is its latest build. + """ + latest: dict[str, tuple[str | None, str]] = {} + for build in tag_report.builds: + latest.setdefault(build.package, (build.previous_nvr, build.nvr)) + return [(package, previous_nvr, nvr) for package, (previous_nvr, nvr) in sorted(latest.items())] + + +def version_release(package: str, nvr: str) -> str: + """Return the version-release part of a package nvr.""" + return nvr.removeprefix(package + '-') + + +def format_versions(versions: list[tuple[str, str | None, str]]) -> str: + lines = [] + for package, previous_nvr, nvr in versions: + new_version = version_release(package, nvr) + item = f'{version_release(package, previous_nvr)} -> {new_version}' if previous_nvr is not None else new_version + lines.append(f'* `{package}`: {item}') + return '\n'.join(lines) + + +POST_TEMPLATE = dedent('''\ + # New maintenance update candidates for XCP-ng $version LTS + + This batch of updates contains mostly fixes, tools version update, a some improvements. + + ## What changed + + $what_changed + + ## Versions + + $versions + + ## Test on XCP-ng $version + + ```bash + yum clean metadata --enablerepo=xcp-ng-testing,xcp-ng-candidates + yum update --enablerepo=xcp-ng-testing,xcp-ng-candidates + reboot + ``` + + The usual update rules apply: pool coordinator first, etc. + + ## What to test + + As usual, normal use and anything else you want to test. + + ## Test window before official release of the updates + + **X days** + + We would like to thank users who shared feedback since our last call for testing: + ''') + + +def format_post(what_changed: str, versions: str, version: str) -> str: + return Template(POST_TEMPLATE).substitute(what_changed=what_changed, versions=versions, version=version) + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description='Generate the package update section of the XCP-ng release post') + parser.add_argument('--report', help='The json report generated by pkg_in_pipe.py', default='report.json') + parser.add_argument('--tag', help='The koji tag to include in the release post', default='v8.3-ci') + parser.add_argument('--version', help='The XCP-ng version of the release post, e.g. 8.3', default=None) + parser.add_argument('--cache', help='The cache path', default='/tmp/pkg_in_pipe.cache') + parser.add_argument('--re-cache', help='Refresh the cache', action='store_true') + parser.add_argument( + '--github-token', help='The token used to access the Github api', default=os.environ.get('GITHUB_TOKEN') + ) + return parser.parse_args() + + +def main() -> None: + signal.signal(signal.SIGPIPE, signal.SIG_DFL) + args = parse_args() + cache = diskcache.Cache(args.cache) + report = Report.model_validate_json(Path(args.report).read_text()) + tag_report = find_tag_report(report, args.tag) + what_changed: list[str] = [] + with tqdm( + sorted(prs_by_package(tag_report).items()), + desc='packages', + unit='package', + file=sys.stderr, + leave=False, + ) as pbar: + for package, prs in pbar: + descriptions = fetch_pr_descriptions(prs, args.github_token, args.re_cache, cache) + if not descriptions: + warn(pbar, f'no description available for the pull requests of {package}, skipping it') + continue + lines = collect_sections(descriptions, pbar) + if not lines: + warn(pbar, f'no "Explain the change to users" section available for the pull requests of {package}') + what_changed.extend(verbatim_lines(package, lines)) + what_changed.append('') + version = args.version or args.tag[1:].split('-')[0] + print(format_post('\n'.join(what_changed).rstrip(), format_versions(versions_by_package(tag_report)), version)) + + +if __name__ == '__main__': + main() \ No newline at end of file