diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5c871c63..9f244d55 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -12,6 +12,12 @@ "description": "Adversarial hypothesis reviewer — systematically attacks theories and root cause analyses to find weaknesses before they find you", "version": "1.0.0" }, + { + "name": "edge-contribution", + "source": "./plugins/edge-contribution", + "description": "Cross-workstream contribution reporting for OpenShift Edge managers — team allocation heatmap, executive summary, and CSV export — reporting workstreams touched per person and people per workstream, derived from Jira and GitHub activity", + "version": "0.5.0" + }, { "name": "edge-ic", "source": "./plugins/edge-ic", diff --git a/plugins/edge-contribution/.claude-plugin/plugin.json b/plugins/edge-contribution/.claude-plugin/plugin.json new file mode 100644 index 00000000..d98b1d36 --- /dev/null +++ b/plugins/edge-contribution/.claude-plugin/plugin.json @@ -0,0 +1,8 @@ +{ + "name": "edge-contribution", + "description": "Cross-workstream contribution reporting for OpenShift Edge managers — team allocation heatmap, executive summary, and CSV export — reporting workstreams touched per person and people per workstream, derived from Jira and GitHub activity", + "version": "0.5.0", + "author": { "name": "copejon" }, + "homepage": "https://github.com/openshift-eng/edge-tooling", + "license": "Apache-2.0" +} diff --git a/plugins/edge-contribution/.gitignore b/plugins/edge-contribution/.gitignore new file mode 100644 index 00000000..da0f1712 --- /dev/null +++ b/plugins/edge-contribution/.gitignore @@ -0,0 +1,10 @@ +# Python test/build artifacts +.pytest_cache/ +__pycache__/ +*.pyc + +# Generated report artifacts (skills write these into the workdir) +roster.json +member.json +jira_activity.json +github_activity.json diff --git a/plugins/edge-contribution/README.md b/plugins/edge-contribution/README.md new file mode 100644 index 00000000..d0a10a9c --- /dev/null +++ b/plugins/edge-contribution/README.md @@ -0,0 +1,255 @@ +# edge-contribution + +Cross-workstream contribution reporting for the OpenShift Edge team. Three +manager skills for a quarter or date range from the same underlying data: + +- **`heatmap` skill:** a terminal allocation view — a people×workstream + grid shaded by contribution amount (grey→green, log scale) with allocation + signals (over/under/narrow flags), instantly. The output is that grid and its + two legends. For the team-level means and the data-quality block, see + `summary --show-sources`; for those means as raw numbers, see `export` + (`metrics.csv`). +- **`summary` skill:** an executive summary — team activity, per-workstream + volume/contributors, per-member totals and workstreams touched, and data + quality. Reports raw measurements with no judgments. Add `--show-sources` for + collection/attribution + diagnostics. Output in `text` or `markdown` for pasting into Slack, email, or + a Google Doc. +- **`export` skill:** the underlying data as four CSV files for + external analysis, pivot tables, or archival. + +For the individual-contributor view of this same data — one engineer's +workstreams-touched count "N of 6" — see `/edge-ic:workstreams` in the `edge-ic` +plugin. It runs these same collectors over a single member. + +The six workstreams are **SNO, TNA, TNF, LVMS, USHIFT, TOPO**, plus a **SHARED** +column for cross-cutting CI/tooling/docs work. Activity is drawn from Jira +(OCPEDGE/USHIFT/OCPBUGS assignee + QA-contact, OCPSTRAT roles) and GitHub +(authored + reviewed PRs), and each item is attributed to a workstream via a +first-hit-wins chain described below. + +## Install + +```text +/plugin marketplace add openshift-eng/edge-tooling +/plugin install edge-contribution +``` + +## Prerequisites + +- **Jira credentials:** export `JIRA_USERNAME` and `JIRA_API_TOKEN` (the same + credentials the Atlassian MCP uses). The collectors call + `redhat.atlassian.net` REST directly — no MCP at runtime. +- **GitHub CLI:** `gh` installed and authenticated (`gh auth status`), with read + access to `openshift-eng/edge-context` — the roster is fetched via `gh`, so no + local checkout is needed. +- **Python 3.9+** with `requests` available. + +## Usage + +Invoke the skills from Claude Code: + +```text +/edge-contribution:heatmap --quarter 2026Q2 +/edge-contribution:summary --quarter 2026Q2 --format markdown +/edge-contribution:export --quarter 2026Q2 --output-dir /tmp/2026Q2-export +``` + +All three cover the full Eng/QE roster and accept `--include-all`. `heatmap` +also accepts `--ascii` and `--no-color`. `summary` and `export` accept +`--refresh` to re-collect activity even if cached files exist. + +### Export output files + +The `export` skill writes four CSV files: + +| File | Contents | +|------|----------| +| `activities.csv` | One row per activity record — the raw collated data. Columns: `member`, `source` (jira/github), `kind`, `workstream`, `attribution_source`, `repo`, `issue_key`, `url`, `ts`, `unattributed_reason` | +| `matrix.csv` | People × workstream counts. Columns: `member`, `SNO`, `TNA`, `TNF`, `LVMS`, `USHIFT`, `TOPO`, `SHARED`, `TOTAL`, `workstreams_touched`, `over`, `under`, `narrow` | +| `metrics.csv` | Team scalars and per-workstream totals/contributors. Columns: `metric`, `value` | +| `data-quality.csv` | Exclusion and unattributed counts. Columns: `category`, `label`, `count` | + +### Running the pipeline directly + +The skills orchestrate standalone scripts under `bin/`. You can run them by hand: + +```bash +# 1. roster (Eng/QE only) fetched from edge-context via gh +python3 bin/load_context.py --output roster.json + +# 2. collect activity for the window +# Jira collection now happens via MCP tools in skills (see references/pipeline.md) +# For manual runs, you would need to query via MCP and run attribute_jira_activities.py +python3 bin/collect_github.py --members-file roster.json --quarter 2026Q2 \ + --output github_activity.json + +# 3. render a report +python3 bin/report.py --view manager --members-file roster.json \ + --activity jira_activity.json --activity github_activity.json \ + --period 2026Q2 --format text + +python3 bin/report.py --view summary --members-file roster.json \ + --activity jira_activity.json --activity github_activity.json \ + --period 2026Q2 --format markdown + +# the single-member view, consumed by the edge-ic plugin's `workstreams` skill +python3 bin/report.py --view ic --member jdoe@redhat.com \ + --activity jira_activity.json --activity github_activity.json \ + --period 2026Q2 --format text + +# 4. export as CSV +python3 bin/export_csv.py --members-file roster.json \ + --activity jira_activity.json --activity github_activity.json \ + --period 2026Q2 --output-dir /tmp/export-2026Q2 +``` + +## Metrics + +Defined precisely in [`references/metrics.md`](references/metrics.md). In short, +from a binary "did member *m* contribute to workstream *w*?" matrix: + +- **Workstreams touched** (per member) = distinct canonical workstreams the + member contributed to, "of 6" (SHARED is deliberately excluded so the count + stays "N of 6"). +- **Mean people per workstream** = `total_touches / 6` (SHARED excluded), where + `total_touches` is the number of 1s in the matrix. +- **Mean workstreams per person** = `total_touches / active_count`, averaged + over *active* members and shown with the active/total roster count. + +Allocation signals (from the weighted `counts` matrix, not the binary one): + +- **Over-allocated** — members whose total is >2× the team median. +- **Under-allocated** — members whose total is <0.5× the team median and >0. +- **Narrow** — members where a single workstream is >70% of their total. + +Items that cannot be attributed to any workstream are reported as +**unattributed** — counted separately, never silently dropped. For where each +number came from and for data quality, see the `summary` skill. + +## Workstream Attribution + +Activity items are mapped to workstreams via a **first-hit-wins chain**: + +### Jira attribution (assignee/QA-contact/OCPSTRAT roles) + +Assignee and QA-contact activity is searched across `OCPEDGE`, `USHIFT`, and +`OCPBUGS`; OCPSTRAT roles are searched separately because they answer a +different question (SME/assignee on strategic items, cf 10475). + +**Comments are deliberately not collected.** Jira Cloud omits `emailAddress` +from comment authors, so matching a roster member by email could never fire — +the query returned zero records while scanning every issue in the project once +per member. Re-adding it requires matching on `accountId`, not email. + +1. Issue's own `components` field → workstream acronym (e.g., "SNO" → SNO) +2. Parent epic's component (batched lookup via `key in (...)` JQL) +3. Jira project key (e.g., USHIFT-6337 → MicroShift/USHIFT) + +### GitHub attribution (authored/reviewed PRs) + +1. **Excluded repos** — personal-namespace repos (owner not in the org allowlist: + `openshift`, `openshift-eng`, `microshift-io`, `openshift-metal3`, `metal3-io`, + `containers`, `clusterlabs`, `ovn-kubernetes`, `opendatahub-io`, `kubevirt`) + are dropped entirely and do not appear even as unattributed. + + **A Jira key does not override this.** Work counts only when it lands in a + Red Hat product or engineering org. A PR opened against a branch in + someone's own repo is not counted even when its title names a ticket — the + exclusion deliberately runs *before* Jira-key resolution, and + `test_jira_key_does_not_rescue_a_personal_repo_pr` pins that order. + + The allowlist is hand-kept, so a legitimate org missing from it would be + dropped silently — the data-quality block therefore names every dropped + repo, not just a total. That block is available in + `summary --show-sources`. +2. **Jira keys from PR title/body** → resolve each key via the Jira chain above, + emit one activity per distinct workstream. +3. **GitHub repo direct mapping** — repos that unambiguously belong to a single + workstream (e.g., `openshift/lvm-operator` → LVMS, `openshift/microshift` → + USHIFT, `openshift/oc-tnf` → TNF). +4. **SHARED repos** — cross-cutting CI/tooling/docs repos credited to the SHARED + column: `openshift/release`, `openshift-eng/edge-tooling`, + `openshift-eng/edge-context`, `openshift/openshift-docs`, + `openshift/enhancements`. +5. **Unattributed** — if none of the above match, the PR is kept with + `workstream=None` and appears in the data-quality footnote. + +### Activity file format + +Both collectors write the same envelope: + +```json +{ + "activities": [ { "member": "...", "workstream": "...", "..." : "..." } ], + "collector_meta": { + "excluded_personal_repo_prs": 99, + "excluded_personal_repos": { "jeff-roche/roundhouse": 38 } + } +} +``` + +`collector_meta` exists for data-quality counters that no record can carry, +because the thing being counted never produced a record — a PR dropped for +living in a personal namespace is counted but leaves nothing behind. A counter +is either a plain total or a `{label: count}` breakdown; when several activity +files are loaded, totals add and breakdowns merge label by label. `report.py` +also accepts a bare JSON list (the pre-envelope format); those files just report +`collector_meta` counters as 0. + +Unattributed Jira records carry an `unattributed_reason` separating +`no_component_no_parent` (fix the ticket) from `parent_also_empty` (fix the +epic), so the data-quality block says which one to go fix. + +### Standing exclusions + +Two deliberate exclusions (restored with `--include-all` on the `heatmap` skill): + +- **Repo:** `openshift-eng/two-node-toolbox` is **deliberately NOT mapped** to + any workstream. That repo covers BOTH arbiter (TNA) and fencing (TNF) + topologies, so mapping it either way would mis-credit work. This was an + explicit product decision — do not "fix" it by adding a mapping. + +## Architecture + +```text +workstream_map.py internal SNO/TNA/TNF/LVMS/USHIFT/TOPO ↔ Jira component map +load_context.py edge-context roster (fetched via gh) → roster.json (Eng/QE) +Skills + MCP mcp__mcp-atlassian__jira_search → raw_jira_issues.json +attribute_jira_activities.py raw Jira issues → jira_activity.json (assignee/QA/ocpstrat) +collect_github.py gh api graphql (batched) → github_activity.json (pr_authored/pr_reviewed) +report.py activity + roster → contribution matrix → metrics → render +metrics.py matrix → workstreams touched / team means / allocation signals +render.py report → text | markdown +_common.py Shared constants, activity payload, date helpers +``` + +`workstream_map`, `load_context`, `attribute_jira_activities`, `metrics`, `render`, +and `report` are pure and side-effect-free. Jira queries are orchestrated by skills +via MCP tools (`mcp__mcp-atlassian__jira_search`); attribution logic is pure Python. +GitHub I/O is isolated in `collect_github.py` with the `gh` layer injected so tests +stay hermetic. +GitHub queries use batched GraphQL (aliased searches, 6 members per request) to +stay well under the 5000 points/hour budget; body-level `RATE_LIMITED` errors +are detected structurally and retried with bounded exponential backoff. If a +chunk request fails, it splits in half and retries each half. Jira workstream +resolution batches all unique keys extracted from PRs into JQL `key in (...)` +queries. Jira HTTP 429 responses retry with the same backoff policy. Other +authentication and command failures are surfaced immediately. + +## Development + +Built test-first (TDD). Run the suite: + +```bash +python3 -m pytest plugins/edge-contribution/bin/tests +``` + +Optional style/type gate (see `requirements-dev.txt`): + +```bash +black --check bin && ruff check bin && mypy bin +``` + +Every module has happy-path, failure, and edge/boundary tests. See the plan and +`references/` for the design rationale and the workstream/component map. diff --git a/plugins/edge-contribution/bin/_common.py b/plugins/edge-contribution/bin/_common.py new file mode 100644 index 00000000..c50718d4 --- /dev/null +++ b/plugins/edge-contribution/bin/_common.py @@ -0,0 +1,385 @@ +#!/usr/bin/env python3 +"""Shared infrastructure for the Jira/GitHub collectors. + +This module isolates the impure edges of the plugin — environment/auth, HTTP +transport with pagination and retry, and date/quarter parsing — so the domain +modules (workstream_map, load_context, metrics, render) stay pure and trivially +testable. The HTTP transport is injected, which keeps unit tests hermetic: no +live network, no third-party mocking library. +""" + +from __future__ import annotations + +import json +import re +import time +from dataclasses import dataclass, field +from datetime import date +from typing import Callable, Dict, List, Optional, Tuple, Union + +# --- Jira collection constants ---------------------------------------------- + +# Custom field IDs +QA_CONTACT_FIELD = "customfield_10470" +SME_FIELD = "customfield_10475" + +# Projects searched for assignee and QA-contact activity +ACTIVITY_PROJECTS: Tuple[str, ...] = ("OCPEDGE", "USHIFT", "OCPBUGS") + +# Unattributed reasons (different problems require different fixes) +UNATTRIBUTED_NO_PARENT = "no_component_no_parent" +UNATTRIBUTED_PARENT_EMPTY = "parent_also_empty" + +# --- HTTP transport constants ----------------------------------------------- + +_DEFAULT_BASE_URL = "https://redhat.atlassian.net" +_SEARCH_PATH = "/rest/api/3/search/jql" +_DEFAULT_PAGE_SIZE = 100 +_DEFAULT_MAX_RETRIES = 2 +_DEFAULT_COMMAND_MAX_RETRIES = 4 +_DEFAULT_COMMAND_BACKOFF_SECONDS = 5.0 +_DEFAULT_COMMAND_MAX_BACKOFF_SECONDS = 60.0 +_QUARTER_PATTERN = re.compile(r"^(?P\d{4})Q(?P[1-4])$") +_QUARTER_MONTHS = {1: (1, 3), 2: (4, 6), 3: (7, 9), 4: (10, 12)} +_LAST_DAY_OF_MONTH = {1: 31, 3: 31, 6: 30, 9: 30, 10: 31, 12: 31} + + +def activity_payload( + activities: List[dict], **counters: Union[int, Dict[str, int]] +) -> Dict[str, object]: + """Wrap collected activity records with any collector-level counters. + + Some data-quality signals cannot live on a record, because the thing being + counted never produced one — a PR dropped for being in a personal namespace, + for example. ``counters`` carries those out of the collector so ``report.py`` + can show a real number instead of assuming zero. + + A counter is either a plain total or a ``{label: count}`` breakdown. Both + merge across files in ``report._load_activities``: totals add, breakdowns + merge key by key. + """ + return {"activities": activities, "collector_meta": dict(counters)} + + +class CollectorError(Exception): + """A collector could not complete because of an external failure.""" + + +class JiraAuthError(CollectorError): + """Jira credentials are missing or were rejected.""" + + +@dataclass(frozen=True) +class JiraConfig: + """Everything needed to talk to a Jira instance.""" + + base_url: str + username: str + api_token: str + + +@dataclass(frozen=True) +class HttpResponse: + """A minimal HTTP response the transport layer returns. + + Kept deliberately tiny so tests can construct one without ``requests``. + """ + + status_code: int + body: str + headers: Dict[str, str] = field(default_factory=dict) + + def json(self) -> dict: + """Parse the body as JSON, raising ``ValueError`` on malformed input.""" + return json.loads(self.body) + + +# A transport takes (method, path, json_body) and returns an HttpResponse. +Transport = Callable[[str, str, Optional[dict]], HttpResponse] + + +def jira_config_from_env(env: Dict[str, str]) -> JiraConfig: + """Build a ``JiraConfig`` from environment variables. + + Raises ``JiraAuthError`` (before any network call) when the required + credentials are absent. + """ + username = env.get("JIRA_USERNAME") + api_token = env.get("JIRA_API_TOKEN") + if not username or not api_token: + raise JiraAuthError("JIRA_USERNAME and JIRA_API_TOKEN must be set to reach Jira REST") + base_url = env.get("JIRA_BASE_URL", _DEFAULT_BASE_URL).rstrip("/") + return JiraConfig(base_url=base_url, username=username, api_token=api_token) + + +class JiraClient: + """Talks to Jira Cloud REST v3 over an injected transport. + + Pagination uses ``nextPageToken`` (this instance does not use ``startAt``). + Transient 5xx responses are retried; auth failures raise immediately. + """ + + def __init__( + self, + config: JiraConfig, + transport: Transport, + max_retries: int = _DEFAULT_MAX_RETRIES, + sleep: Callable[[float], None] = time.sleep, + ) -> None: + self._config = config + self._transport = transport + self._max_retries = max_retries + self._sleep = sleep + + @property + def base_url(self) -> str: + return self._config.base_url + + def search( + self, jql: str, fields: List[str], page_size: int = _DEFAULT_PAGE_SIZE + ) -> List[dict]: + """Return every issue matching ``jql``, following pagination to the end.""" + issues: List[dict] = [] + next_page_token: Optional[str] = None + while True: + payload = {"jql": jql, "fields": fields, "maxResults": page_size} + if next_page_token: + payload["nextPageToken"] = next_page_token + page = self._request_json("POST", _SEARCH_PATH, payload) + issues.extend(page.get("issues", [])) + next_page_token = page.get("nextPageToken") + if page.get("isLast", not next_page_token) or not next_page_token: + return issues + + def get_issue(self, issue_key: str, fields: List[str]) -> dict: + """Fetch a single issue's requested fields.""" + path = f"/rest/api/3/issue/{issue_key}?fields={','.join(fields)}" + return self._request_json("GET", path, None) + + def _request_json(self, method: str, path: str, body: Optional[dict]) -> dict: + response = self._request_with_retry(method, path, body) + try: + return response.json() + except ValueError as error: + raise CollectorError(f"malformed JSON from Jira {path}: {error}") from error + + def _request_with_retry(self, method: str, path: str, body: Optional[dict]) -> HttpResponse: + attempts = 0 + while True: + response = self._transport(method, path, body) + if response.status_code in (401, 403): + raise JiraAuthError(f"Jira rejected credentials ({response.status_code})") + if response.status_code == 429: + attempts += 1 + if attempts > self._max_retries: + raise CollectorError(f"Jira rate limit after {self._max_retries} retries") + retry_after = next( + ( + value + for key, value in response.headers.items() + if key.lower() == "retry-after" + ), + None, + ) + try: + delay = min(max(float(retry_after), 0.0), 60.0) if retry_after else None + except ValueError: + delay = None + if delay is None: + delay = min(5.0 * (2 ** (attempts - 1)), 60.0) + self._sleep(delay) + continue + if response.status_code >= 500: + attempts += 1 + if attempts > self._max_retries: + raise CollectorError( + f"Jira server error {response.status_code} after " + f"{self._max_retries} retries" + ) + continue + if response.status_code != 200: + raise CollectorError(f"Jira request to {path} failed with {response.status_code}") + return response + + +@dataclass(frozen=True) +class Window: + """An inclusive date range for a reporting period.""" + + start: date + end: date + + def contains(self, day: date) -> bool: + return self.start <= day <= self.end + + def contains_timestamp(self, timestamp: str) -> bool: + """Return whether a Jira ISO timestamp falls within the window.""" + return self.contains(_date_from_timestamp(timestamp)) + + +def parse_date(text: str) -> date: + """Parse an ISO ``YYYY-MM-DD`` date string.""" + return date.fromisoformat(text) + + +def _date_from_timestamp(timestamp: str) -> date: + """Parse the date portion of a Jira timestamp (``2026-05-01T12:00:...``).""" + return date.fromisoformat(timestamp[:10]) + + +def quarter_to_window(quarter: str) -> Window: + """Convert a ``YYYYQn`` label into its inclusive calendar-quarter window. + + Returns the full calendar quarter without date capping. Use ``resolve_window`` + for policy-aware windowing (e.g., capping in-progress quarters at today). + + Raises ``ValueError`` for anything that is not a well-formed quarter. + """ + match = _QUARTER_PATTERN.match(quarter) + if not match: + raise ValueError(f"not a valid quarter label: {quarter!r} (expected e.g. 2026Q2)") + year = int(match.group("year")) + first_month, last_month = _QUARTER_MONTHS[int(match.group("quarter"))] + return Window( + start=date(year, first_month, 1), + end=date(year, last_month, _LAST_DAY_OF_MONTH[last_month]), + ) + + +def resolve_window( + quarter: Optional[str] = None, + from_date: Optional[str] = None, + to_date: Optional[str] = None, + today: Optional[date] = None, +) -> Window: + """Resolve a reporting window from a quarter label or an explicit date pair. + + For quarter-based windows, applies policy: + - Fully past quarters are returned as-is. + - In-progress quarters have the end date capped at today (to avoid future + dates that break GitHub search). + - Future quarters raise ``ValueError`` (meaningless for reporting). + + Raises ``ValueError`` when neither a quarter nor a complete date range is + supplied, or when the quarter has not yet started. + """ + if quarter: + window = quarter_to_window(quarter) + if today is None: + today = date.today() + if window.start > today: + raise ValueError( + f"quarter {quarter} has not started yet (begins {window.start.isoformat()})" + ) + if window.end > today: + # In-progress quarter: cap at today + return Window(window.start, today) + # Fully past quarter: return as-is + return window + if from_date and to_date: + return Window(parse_date(from_date), parse_date(to_date)) + raise ValueError("provide a quarter, or both a start and end date") + + +def build_default_transport(config: JiraConfig) -> Transport: + """Build the production HTTP transport backed by ``requests``. + + ``requests`` is imported lazily so unit tests, which inject a transport, + never require it. + """ + import requests + from requests.auth import HTTPBasicAuth + + auth = HTTPBasicAuth(config.username, config.api_token) + headers = {"Accept": "application/json", "Content-Type": "application/json"} + + def transport(method: str, path: str, body: Optional[dict]) -> HttpResponse: + response = requests.request( + method, + f"{config.base_url}{path}", + json=body, + auth=auth, + headers=headers, + timeout=60, + ) + return HttpResponse(response.status_code, response.text, dict(response.headers)) + + return transport + + +@dataclass(frozen=True) +class CommandResult: + """The outcome of running an external command (e.g. the ``gh`` CLI).""" + + returncode: int + stdout: str + stderr: str + + +# A runner takes an argv list and returns its CommandResult. +CommandRunner = Callable[[List[str]], CommandResult] + + +def is_rate_limit_failure(result: CommandResult) -> bool: + """Return whether a command failed because GitHub rate-limited it.""" + if result.returncode == 0: + return False + message = f"{result.stdout}\n{result.stderr}".lower() + return any( + marker in message + for marker in ( + "rate limit", + "rate_limit", + "abuse detection", + "secondary rate", + ) + ) + + +def _retry_after_seconds(message: str) -> Optional[float]: + """Extract a Retry-After value when the command reports one.""" + match = re.search(r"retry-after\s*[:=]\s*(\d+(?:\.\d+)?)", message, re.IGNORECASE) + return float(match.group(1)) if match else None + + +def run_with_rate_limit_retry( + runner: CommandRunner, + command: List[str], + sleep: Callable[[float], None] = time.sleep, + max_retries: int = _DEFAULT_COMMAND_MAX_RETRIES, + backoff_seconds: float = _DEFAULT_COMMAND_BACKOFF_SECONDS, + max_backoff_seconds: float = _DEFAULT_COMMAND_MAX_BACKOFF_SECONDS, +) -> CommandResult: + """Run a command, retrying rate-limit failures with bounded backoff. + + ``gh`` does not consistently expose GitHub's reset headers, so the default + is exponential backoff (5, 10, 20, 40 seconds), capped at 60 seconds. + When a ``Retry-After`` value is present, it is honored up to the cap. + Other command failures are returned immediately to preserve their original + error handling. + """ + retries = 0 + while True: + result = runner(command) + if not is_rate_limit_failure(result) or retries >= max_retries: + return result + message = f"{result.stdout}\n{result.stderr}" + retry_after = _retry_after_seconds(message) + delay = retry_after if retry_after is not None else backoff_seconds * (2**retries) + sleep(min(delay, max_backoff_seconds)) + retries += 1 + + +def build_default_runner() -> CommandRunner: + """Build the production command runner backed by ``subprocess``. + + ``subprocess`` is imported lazily so unit tests, which inject a runner, + never spawn a real process. + """ + import subprocess + + def runner(command: List[str]) -> CommandResult: + completed = subprocess.run(command, capture_output=True, text=True) + return CommandResult(completed.returncode, completed.stdout, completed.stderr) + + return runner diff --git a/plugins/edge-contribution/bin/attribute_jira_activities.py b/plugins/edge-contribution/bin/attribute_jira_activities.py new file mode 100755 index 00000000..c593de49 --- /dev/null +++ b/plugins/edge-contribution/bin/attribute_jira_activities.py @@ -0,0 +1,222 @@ +#!/usr/bin/env python3 +"""Pure attribution for Jira issues collected via MCP tools. + +Takes raw Jira issues (collected by Claude via mcp__mcp-atlassian__jira_search) +and applies the standard attribution chain: + +1. Issue's own components → workstream +2. Parent epic's component (via pre-resolved map) +3. Jira project key (e.g., USHIFT → MicroShift) + +This script contains NO network calls — all data is pre-fetched. The skill +orchestrates MCP queries and passes results here for processing. + +Input format (raw_jira_issues.json): +{ + "issues": [ + { + "member": "jcope@redhat.com", + "kind": "assignee" | "qa" | "ocpstrat_role", + "fields": { + "key": "OCPEDGE-1234", + "components": [...], + "updated": "2026-08-15T10:30:00Z", + "summary": "...", + "parent": {"key": "OCPEDGE-999"} + } + } + ], + "parent_workstreams": { + "OCPEDGE-999": "SNO" | null + }, + "base_url": "https://redhat.atlassian.net" +} + +Output format (jira_activity.json): +{ + "activities": [ + { + "member": "jcope@redhat.com", + "workstream": "SNO" | null, + "kind": "assignee", + "issue_key": "OCPEDGE-1234", + "url": "https://redhat.atlassian.net/browse/OCPEDGE-1234", + "ts": "2026-08-15T10:30:00Z", + "attribution_source": "component" | "parent" | "project" | null, + "unattributed_reason": "no_component_no_parent" | "parent_also_empty" | null + } + ], + "collector_meta": {} +} +""" + +from __future__ import annotations + +import argparse +import json +from dataclasses import asdict, dataclass +from typing import Dict, List, Optional + +from _common import ( + UNATTRIBUTED_NO_PARENT, + UNATTRIBUTED_PARENT_EMPTY, + activity_payload, +) +from workstream_map import component_to_workstream, project_to_workstream + + +@dataclass(frozen=True) +class Activity: + """One attributed contribution by one member. + + ``workstream`` is ``None`` for items that could not be mapped (unattributed). + ``attribution_source`` is ``None`` exactly when ``workstream`` is ``None``. + Valid sources: "component", "parent", "project". + + ``unattributed_reason`` is set exactly when ``workstream`` is ``None``, and is + one of ``UNATTRIBUTED_NO_PARENT`` / ``UNATTRIBUTED_PARENT_EMPTY``. + """ + + member: str + workstream: Optional[str] + kind: str + issue_key: str + url: str + ts: Optional[str] + attribution_source: Optional[str] = None + unattributed_reason: Optional[str] = None + + +# --- Attribution (pure) ----------------------------------------------------- + + +def _issue_components(issue: dict) -> List[str]: + """Extract component names from Jira issue fields.""" + components = issue.get("fields", {}).get("components") or [] + return [component["name"] for component in components if component.get("name")] + + +def attribute_issue( + issue: dict, parent_workstreams: Dict[str, Optional[str]] +) -> tuple[List[str], Optional[str]]: + """Return (workstream acronyms, attribution_source) for one issue. + + Attribution chain (first hit wins): + 1. Issue's own components (can yield multiple workstreams) -> "component" + 2. Parent epic's component (via resolved map) -> "parent" + 3. Jira project key (e.g., USHIFT -> MicroShift) -> "project" + + Returns ``([], None)`` when nothing matches. + """ + # Step 1: issue's own components + mapped: List[str] = [] + for component_name in _issue_components(issue): + workstream = component_to_workstream(component_name) + if workstream and workstream not in mapped: + mapped.append(workstream) + if mapped: + return (mapped, "component") + + # Step 2: parent epic's component + parent = issue.get("fields", {}).get("parent") + if parent: + parent_key = parent.get("key") + if parent_key: + parent_workstream = parent_workstreams.get(parent_key) + if parent_workstream: + return ([parent_workstream], "parent") + + # Step 3: Jira project key + issue_key = issue.get("fields", {}).get("key") or issue.get("key") + if issue_key: + project_workstream = project_to_workstream(issue_key) + if project_workstream: + return ([project_workstream], "project") + + # Dead end + return ([], None) + + +def unattributed_reason(issue: dict) -> str: + """Return why ``issue`` could not be attributed, for the data-quality report. + + Only meaningful for issues that ``attribute_issue`` returned no workstream for. + An issue that names a parent epic but still ended up unattributed means the + epic was itself component-less — a fixable gap one level up. Note that a parent + Jira refused to return (deleted, or a failed lookup batch) is indistinguishable + from a component-less one here and is counted the same way. + """ + parent = issue.get("fields", {}).get("parent") or {} + if parent.get("key"): + return UNATTRIBUTED_PARENT_EMPTY + return UNATTRIBUTED_NO_PARENT + + +def issue_to_activities( + issue: dict, + member: str, + kind: str, + base_url: str, + parent_workstreams: Dict[str, Optional[str]], +) -> List[Activity]: + """Turn one issue into per-workstream activities (unattributed if unmappable). + + An issue with multiple mappable components fans out to one Activity per + workstream. Unattributed issues produce one Activity with workstream=None. + """ + fields = issue.get("fields", {}) + issue_key = fields.get("key") or issue.get("key", "") + url = f"{base_url}/browse/{issue_key}" + timestamp = fields.get("updated") + + workstreams, attribution_source = attribute_issue(issue, parent_workstreams) + if not workstreams: + return [ + Activity( + member, None, kind, issue_key, url, timestamp, None, unattributed_reason(issue) + ) + ] + return [ + Activity(member, workstream, kind, issue_key, url, timestamp, attribution_source) + for workstream in workstreams + ] + + +# --- Main ------------------------------------------------------------------- + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser( + description="Apply workstream attribution to raw Jira issues collected via MCP." + ) + parser.add_argument("--raw-issues", required=True, help="raw_jira_issues.json from skill") + parser.add_argument("--output", default="jira_activity.json", help="output activities JSON") + args = parser.parse_args(argv) + + # Load raw issues + with open(args.raw_issues, encoding="utf-8") as handle: + raw_data = json.load(handle) + + issues = raw_data.get("issues", []) + parent_workstreams = raw_data.get("parent_workstreams", {}) + base_url = raw_data.get("base_url", "https://redhat.atlassian.net") + + # Apply attribution to each issue + activities: List[Activity] = [] + for issue_record in issues: + member = issue_record["member"] + kind = issue_record["kind"] + issue = issue_record # Issue data is in the record itself under "fields" + + activities.extend(issue_to_activities(issue, member, kind, base_url, parent_workstreams)) + + # Write activities with standard envelope format + payload = activity_payload([asdict(activity) for activity in activities]) + with open(args.output, "w", encoding="utf-8") as handle: + json.dump(payload, handle, indent=2) + + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/edge-contribution/bin/collect_github.py b/plugins/edge-contribution/bin/collect_github.py new file mode 100644 index 00000000..4f1c179e --- /dev/null +++ b/plugins/edge-contribution/bin/collect_github.py @@ -0,0 +1,639 @@ +#!/usr/bin/env python3 +"""Standalone GitHub collector for cross-workstream contribution. + +For a reporting window and a set of roster members, this gathers each member's +pull-request activity via batched GraphQL queries through ``gh api graphql``, +attributing every PR to a workstream: + +* **pr_authored** — PRs the member opened in ``openshift``/``openshift-eng``. +* **pr_reviewed** — PRs the member reviewed in the same orgs. + +Attribution parses referenced Jira keys from the PR title and body, then resolves +every unique key in a single batch prefetch, mapping each to a workstream via its +Jira components. A PR that references no mappable workstream is kept with +``workstream=None`` so it lands in the unattributed bucket rather than being dropped. + +GraphQL queries are chunked to avoid request-size limits; if a chunk fails with an +HTTP error, it is split in half and retried until down to single-slot granularity. +Pagination follows ``pageInfo.hasNextPage`` cursors until all results are fetched. + +PRs in personal-namespace repos are dropped before attribution. Because a dropped +PR leaves no record behind, the drop count is written alongside the records in a +``{"activities": [...], "collector_meta": {...}}`` envelope rather than only being +printed, so the data-quality block downstream can report it. + +Both impure edges — the ``gh`` command runner and the Jira client — are injected, +so this collector's logic is fully unit-testable without a subprocess or network. +""" + +from __future__ import annotations + +import argparse +import json +import re +import time +from dataclasses import asdict, dataclass +from typing import Callable, Dict, Iterable, List, Optional + +from _common import ( + CollectorError, + CommandResult, + CommandRunner, + JiraAuthError, + JiraClient, + Window, + activity_payload, + build_default_runner, + jira_config_from_env, + run_with_rate_limit_retry, + resolve_window, +) +from workstream_map import ( + SHARED_COLUMN, + component_to_workstream, + is_excluded_repo, + is_shared_repo, + project_to_workstream, + repo_to_workstream, +) + + +class GraphqlResponseError(CollectorError): + """A GraphQL response carried errors or omitted a requested search.""" + + +class GraphqlRateLimitExhausted(CollectorError): + """GraphQL rate limit is exhausted; splitting will not help.""" + + +# Re-exported from _common so existing references (and tests) keep resolving +# ``collect_github.CommandResult`` / ``CommandRunner`` / ``build_default_runner``. +__all__ = ["CommandResult", "CommandRunner", "build_default_runner"] + +# A resolver maps a Jira key to (workstream, attribution_source), or (None, None) if unmappable. +WorkstreamResolver = Callable[[str], tuple[Optional[str], Optional[str]]] + +_JIRA_KEY_PATTERN = re.compile( + r"\b(?:OCPEDGE|USHIFT|OCPBUGS|OCPSTRAT|MGMT|ETCD|RHEL|CNV)-\d+\b", + re.IGNORECASE, +) + + +@dataclass(frozen=True) +class SearchSlot: + """One aliased search within a batched GraphQL request.""" + + alias: str + member: str + handle: str + kind: str # "pr_authored" or "pr_reviewed" + cursor: Optional[str] = None + + +@dataclass(frozen=True) +class GithubActivity: + """One attributed pull-request contribution by one member.""" + + member: str + workstream: Optional[str] + kind: str + repo: str + pr_url: str + ts: Optional[str] + source_key: Optional[str] + attribution_source: Optional[str] = None + + +def extract_jira_keys(text: str) -> List[str]: + """Return the distinct known-project Jira keys referenced in ``text``. + + Matching is case-insensitive; keys are normalized to upper case and returned + in first-seen order. + """ + keys: List[str] = [] + for match in _JIRA_KEY_PATTERN.finditer(text or ""): + key = match.group(0).upper() + if key not in keys: + keys.append(key) + return keys + + +# --- Attribution (pure) ----------------------------------------------------- + + +def _pr_repo(pr: dict) -> str: + repository = pr.get("repository") or {} + return repository.get("nameWithOwner") or repository.get("name") or "" + + +def tally_excluded_repos(prs: Iterable[dict]) -> Dict[str, int]: + """Count the dropped PRs per repo, so the report can name what it lost. + + A bare total says "101 PRs vanished" without saying where. That hides the + failure mode this exclusion actually has: the org allowlist is a hand-kept + list, so a legitimate new org reads as a personal namespace and disappears + silently. Naming the repos makes an unexpected one obvious on sight. + """ + tally: Dict[str, int] = {} + for pr in prs: + repo = _pr_repo(pr) + if is_excluded_repo(repo): + tally[repo] = tally.get(repo, 0) + 1 + return tally + + +def pr_to_activities( + pr: dict, member: str, kind: str, resolve_workstream: WorkstreamResolver +) -> List[GithubActivity]: + """Turn one PR into per-workstream activities (unattributed if unmappable). + + Attribution order (first hit wins): + a. Excluded repo → return [] (drop the PR entirely) + b. Jira key resolution → emit one activity per distinct workstream + c. Repo fallback → single activity with source "repo" + d. SHARED repo → single activity with SHARED_COLUMN, source "shared" + e. Unattributed → single activity with workstream None, source None + """ + repo = _pr_repo(pr) + url = pr.get("url", "") + timestamp = pr.get("createdAt") + + # a. Drop excluded repos (personal namespaces) + if is_excluded_repo(repo): + return [] + + # b. Try Jira key resolution + text = f"{pr.get('title', '')}\n{pr.get('body') or ''}" + keys = extract_jira_keys(text) + attributions = _attribute_keys(keys, resolve_workstream) + if attributions: + return [ + GithubActivity(member, workstream, kind, repo, url, timestamp, key, source) + for workstream, key, source in attributions + ] + + # c. Repo fallback + repo_workstream = repo_to_workstream(repo) + if repo_workstream: + return [GithubActivity(member, repo_workstream, kind, repo, url, timestamp, None, "repo")] + + # d. SHARED repo + if is_shared_repo(repo): + return [GithubActivity(member, SHARED_COLUMN, kind, repo, url, timestamp, None, "shared")] + + # e. Unattributed (preserve existing behavior) + source_key = keys[0] if keys else None + return [GithubActivity(member, None, kind, repo, url, timestamp, source_key, None)] + + +def _attribute_keys( + keys: List[str], resolve_workstream: WorkstreamResolver +) -> List[tuple[str, str, str]]: + """Return (workstream, key, source) tuples for attributed keys.""" + attributions: List[tuple[str, str, str]] = [] + seen: set = set() + for key in keys: + workstream, source = resolve_workstream(key) + if workstream and workstream not in seen: + seen.add(workstream) + attributions.append((workstream, key, source)) + return attributions + + +# --- GraphQL batching ------------------------------------------------------- + + +def _window_range(window: Window) -> str: + return f"{window.start.isoformat()}..{window.end.isoformat()}" + + +def build_batch_query(slots: List[SearchSlot], window: Window) -> str: + """Build a GraphQL query with one aliased search per slot.""" + range_str = _window_range(window) + searches = [] + for slot in slots: + if slot.kind == "pr_authored": + query_text = f"author:{slot.handle} type:pr created:{range_str}" + else: # pr_reviewed + query_text = f"reviewed-by:{slot.handle} type:pr updated:{range_str}" + after_clause = f', after: "{slot.cursor}"' if slot.cursor else "" + search = f""" + {slot.alias}: search(query: "{query_text}", type: ISSUE, first: 100{after_clause}) {{ + issueCount + pageInfo {{ hasNextPage endCursor }} + nodes {{ + ... on PullRequest {{ + url + title + body + createdAt + repository {{ nameWithOwner }} + }} + }} + }}""" + searches.append(search) + return "query {" + "".join(searches) + "\n}" + + +def parse_batch_response( + payload: dict, slots: List[SearchSlot], slot_node_counts: Optional[dict] = None +) -> tuple[List[tuple[SearchSlot, dict]], List[SearchSlot]]: + """Parse a GraphQL batch response into PR nodes and continuation slots. + + Validates that: + - The response contains no errors + - Every requested slot appears in data (not missing, not null) + - Node counts are consistent with issueCount when pagination completes + + Args: + slot_node_counts: Optional dict mapping slot.alias to accumulated node count + across all pages, used for completeness checking. + """ + # Check for GraphQL errors + errors = payload.get("errors", []) + if errors: + first_error = errors[0] + message = first_error.get("message", "unknown error") + path = first_error.get("path", []) + path_str = ".".join(str(p) for p in path) if path else "unknown" + raise GraphqlResponseError(f"GraphQL error at {path_str}: {message}") + + # data may be absent or None on certain errors + data = payload.get("data") + if data is None: + raise GraphqlResponseError("GraphQL response missing 'data' field") + + pairs: List[tuple[SearchSlot, dict]] = [] + next_slots: List[SearchSlot] = [] + + for slot in slots: + # Check that the alias is present + if slot.alias not in data: + raise GraphqlResponseError( + f"GraphQL response missing expected alias '{slot.alias}' for {slot.handle}" + ) + + result = data[slot.alias] + # Check for null result (GitHub returns null for failed searches) + if result is None: + raise GraphqlResponseError( + f"GraphQL returned null for alias '{slot.alias}' (handle: {slot.handle})" + ) + + nodes = result.get("nodes", []) + issue_count = result.get("issueCount", 0) + + for node in nodes: + pairs.append((slot, node)) + + page_info = result.get("pageInfo", {}) + has_next_page = page_info.get("hasNextPage", False) + + if has_next_page: + next_slots.append( + SearchSlot( + alias=slot.alias, + member=slot.member, + handle=slot.handle, + kind=slot.kind, + cursor=page_info.get("endCursor"), + ) + ) + else: + # Pagination complete: check completeness if we're tracking counts + if slot_node_counts is not None and issue_count <= 1000: + # GitHub caps search results at 1000, so only check if within that limit + total_nodes = slot_node_counts.get(slot.alias, 0) + len(nodes) + if total_nodes != issue_count: + raise GraphqlResponseError( + f"Incomplete results for {slot.handle}: collected {total_nodes} " + f"nodes but issueCount reports {issue_count}" + ) + + return pairs, next_slots + + +def graphql_rate_limit_adapter(runner: CommandRunner) -> CommandRunner: + """Wrap a runner to detect body-level GraphQL RATE_LIMITED errors. + + GraphQL rate limits arrive as HTTP 200 with errors[].type == "RATE_LIMITED", + invisible to the existing ``is_rate_limit_failure`` check. This adapter + rewrites such a response into a non-zero exit, allowing the retry logic + to handle transient limits while marking quota exhaustion distinctly so + chunk-splitting can avoid making it worse. + """ + + def adapted(command: List[str]) -> CommandResult: + result = runner(command) + if result.returncode == 0: + try: + body = json.loads(result.stdout) + errors = body.get("errors", []) + if any(err.get("type") == "RATE_LIMITED" for err in errors): + # Rewrite to trigger retries, with a marker for no-split behavior + return CommandResult( + returncode=1, + stdout="", + stderr="GraphQL rate_limit exhausted (quota)", + ) + except (ValueError, KeyError): + pass + return result + + return adapted + + +def collect_prs( + runner: CommandRunner, + roster: List[dict], + window: Window, + chunk_size: int = 6, + sleep: Callable[[float], None] = time.sleep, +) -> List[tuple[SearchSlot, dict]]: + """Collect all PRs for the roster via batched GraphQL, following pagination.""" + slots = [] + for idx, member in enumerate(roster): + handle = member["github"] + jira_user = member["jira_username"] + slots.append( + SearchSlot( + alias=f"m{idx}_authored", + member=jira_user, + handle=handle, + kind="pr_authored", + ) + ) + slots.append( + SearchSlot( + alias=f"m{idx}_reviewed", + member=jira_user, + handle=handle, + kind="pr_reviewed", + ) + ) + + adapted_runner = graphql_rate_limit_adapter(runner) + pairs: List[tuple[SearchSlot, dict]] = [] + pending = slots[:] + # Track accumulated node counts per slot alias for completeness checking + slot_node_counts: Dict[str, int] = {} + + while pending: + chunks = [pending[i : i + chunk_size] for i in range(0, len(pending), chunk_size)] + pending = [] + for chunk in chunks: + chunk_pairs, chunk_next = _fetch_chunk( + adapted_runner, chunk, window, slot_node_counts, sleep + ) + # Update counts for each slot in this chunk + for slot, _node in chunk_pairs: + slot_node_counts[slot.alias] = slot_node_counts.get(slot.alias, 0) + 1 + pairs.extend(chunk_pairs) + pending.extend(chunk_next) + + return pairs + + +def _fetch_chunk( + runner: CommandRunner, + chunk: List[SearchSlot], + window: Window, + slot_node_counts: Optional[Dict[str, int]] = None, + sleep: Callable[[float], None] = time.sleep, +) -> tuple[List[tuple[SearchSlot, dict]], List[SearchSlot]]: + """Fetch one chunk, splitting on failure when splitting can help. + + Splits on GraphqlResponseError or transport-level failures (e.g., HTTP 502), + but re-raises rate-limit exhaustion immediately without splitting (since + splitting only issues more requests and makes rate limiting worse). + """ + try: + query = build_batch_query(chunk, window) + result = _run_gh_json(runner, ["gh", "api", "graphql", "-f", f"query={query}"], sleep) + return parse_batch_response(result, chunk, slot_node_counts) + except GraphqlRateLimitExhausted: + # Rate limit exhaustion: splitting makes it worse, re-raise immediately + raise + except (GraphqlResponseError, CollectorError): + # GraphQL errors or transport failures: try splitting + if len(chunk) == 1: + raise + mid = len(chunk) // 2 + left_pairs, left_next = _fetch_chunk(runner, chunk[:mid], window, slot_node_counts, sleep) + right_pairs, right_next = _fetch_chunk(runner, chunk[mid:], window, slot_node_counts, sleep) + return left_pairs + right_pairs, left_next + right_next + + +# --- Orchestration (injected runner + resolver) ----------------------------- + + +def _run_gh_json( + runner: CommandRunner, + command: List[str], + sleep: Callable[[float], None] = time.sleep, +) -> dict: + result = run_with_rate_limit_retry(runner, command, sleep=sleep) + if result.returncode != 0: + # Check for the GraphQL quota marker from the adapter + if "(quota)" in result.stderr: + raise GraphqlRateLimitExhausted("GraphQL rate limit exhausted") + raise CollectorError(f"gh exited {result.returncode}: {result.stderr.strip()}") + try: + return json.loads(result.stdout) + except ValueError as error: + raise CollectorError(f"malformed JSON from gh: {error}") from error + + +def _flatten(activity_lists) -> List[GithubActivity]: + return [activity for activities in activity_lists for activity in activities] + + +# --- Jira workstream resolution (batched) ---------------------------------- + + +def _components_to_workstream(components: List[dict]) -> Optional[str]: + """Map an issue's components to its workstream, if any.""" + for component in components: + workstream = component_to_workstream(component.get("name")) + if workstream: + return workstream + return None + + +def resolve_workstreams( + client: JiraClient, keys: List[str], batch_size: int = 100 +) -> dict[str, tuple[Optional[str], Optional[str]]]: + """Map each Jira key to (workstream, attribution_source). + + Attribution order (first hit wins): + 1. the issue's own components → source "component" + 2. the parent epic's component → source "parent" + 3. project_to_workstream(key) → source "project" + + Keys absent from Jira resolve to (None, None). Authentication errors propagate. + """ + if not keys: + return {} + mapping: dict[str, tuple[Optional[str], Optional[str]]] = {key: (None, None) for key in keys} + + # First pass: fetch issues with their components and parent references + issues_by_key: dict[str, dict] = {} + parent_keys: List[str] = [] + + for i in range(0, len(keys), batch_size): + batch = keys[i : i + batch_size] + jql = f"key in ({','.join(batch)})" + try: + issues = client.search(jql, ["key", "components", "parent"]) + except JiraAuthError: + raise + except CollectorError: + continue + for issue in issues: + key = issue.get("key") + if key: + issues_by_key[key] = issue + # Collect parent keys for batch resolution + parent = issue.get("fields", {}).get("parent") + if parent: + parent_key = parent.get("key") + if parent_key and parent_key not in parent_keys: + parent_keys.append(parent_key) + + # Second pass: batch-resolve parent keys to get their components + parents_by_key: dict[str, dict] = {} + for i in range(0, len(parent_keys), batch_size): + batch = parent_keys[i : i + batch_size] + jql = f"key in ({','.join(batch)})" + try: + parents = client.search(jql, ["key", "components"]) + except JiraAuthError: + raise + except CollectorError: + continue + for parent in parents: + parent_key = parent.get("key") + if parent_key: + parents_by_key[parent_key] = parent + + # Third pass: apply attribution chain for each original key + for key in keys: + issue = issues_by_key.get(key) + if not issue: + # Issue not found in Jira, stays (None, None) + continue + + # 1. Try own components + components = issue.get("fields", {}).get("components") or [] + workstream = _components_to_workstream(components) + if workstream: + mapping[key] = (workstream, "component") + continue + + # 2. Try parent's components + parent = issue.get("fields", {}).get("parent") + if parent: + parent_key = parent.get("key") + parent_issue = parents_by_key.get(parent_key) + if parent_issue: + parent_components = parent_issue.get("fields", {}).get("components") or [] + workstream = _components_to_workstream(parent_components) + if workstream: + mapping[key] = (workstream, "parent") + continue + + # 3. Try project fallback + workstream = project_to_workstream(key) + if workstream: + mapping[key] = (workstream, "project") + continue + + # No attribution found, stays (None, None) + + return mapping + + +def make_prefetched_resolver( + mapping: dict[str, tuple[Optional[str], Optional[str]]] +) -> WorkstreamResolver: + """Return a resolver backed by a precomputed mapping.""" + return lambda key: mapping.get(key, (None, None)) + + +# --- CLI -------------------------------------------------------------------- + + +def main(argv: Optional[List[str]] = None) -> int: + import os + + from _common import JiraClient, build_default_transport + + parser = argparse.ArgumentParser(description="Collect GitHub PR contribution activity.") + parser.add_argument( + "--members-file", + required=True, + help="roster.json produced by load_context.py (needs github + jira_username)", + ) + parser.add_argument("--quarter", help="reporting quarter, e.g. 2026Q2") + parser.add_argument("--from", dest="from_date", help="window start, YYYY-MM-DD") + parser.add_argument("--to", dest="to_date", help="window end, YYYY-MM-DD") + parser.add_argument("--output", default="github_activity.json") + args = parser.parse_args(argv) + + window = resolve_window(args.quarter, args.from_date, args.to_date) + with open(args.members_file, encoding="utf-8") as handle: + roster = json.load(handle) + + runner = build_default_runner() + pr_pairs = collect_prs(runner, roster, window) + + # Extract all unique Jira keys from PRs + all_keys: List[str] = [] + for _, pr in pr_pairs: + text = f"{pr.get('title', '')}\n{pr.get('body') or ''}" + keys = extract_jira_keys(text) + for key in keys: + if key not in all_keys: + all_keys.append(key) + + # Resolve workstreams in batch + config = jira_config_from_env(os.environ) + client = JiraClient(config, build_default_transport(config)) + workstream_map = resolve_workstreams(client, all_keys) + resolve_workstream = make_prefetched_resolver(workstream_map) + + # Attribute every PR + activities: List[GithubActivity] = [] + for slot, pr in pr_pairs: + activities.extend(pr_to_activities(pr, slot.member, slot.kind, resolve_workstream)) + + excluded_repos = tally_excluded_repos(pr for _slot, pr in pr_pairs) + excluded_count = sum(excluded_repos.values()) + + # Excluded PRs produce no activity records at all, so the count would be lost + # if it were only printed. Carry it in the envelope so report.py can show it + # in the data-quality block instead of silently reporting zero. The per-repo + # breakdown rides along so a legitimate org missing from the allowlist shows + # up by name rather than disappearing into the total. + payload = activity_payload( + [asdict(activity) for activity in activities], + excluded_personal_repo_prs=excluded_count, + excluded_personal_repos=excluded_repos, + ) + with open(args.output, "w", encoding="utf-8") as output: + json.dump(payload, output, indent=2) + + # Report summary to stderr + import sys + total_prs = len(pr_pairs) + attributed_prs = total_prs - excluded_count + print( + f"Collected {total_prs} PRs: {attributed_prs} processed, {excluded_count} excluded " + f"(personal repos)", + file=sys.stderr, + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/edge-contribution/bin/export_csv.py b/plugins/edge-contribution/bin/export_csv.py new file mode 100644 index 00000000..07bb7d50 --- /dev/null +++ b/plugins/edge-contribution/bin/export_csv.py @@ -0,0 +1,316 @@ +#!/usr/bin/env python3 +"""Export cross-workstream contribution data as CSV files for external analysis. + +Produces four CSV files from collected activity records: + +* ``activities.csv`` — one row per activity record, with normalized fields +* ``matrix.csv`` — people x workstream counts with allocation flags +* ``metrics.csv`` — team scalars and per-workstream totals +* ``data-quality.csv`` — excluded members/repos and unattributed items + +All builders are pure functions returning CSV strings. The ``main`` function +wires them together with file I/O. +""" + +from __future__ import annotations + +import argparse +import csv +import io +import os +import sys +from typing import List, Optional + +import report +from render import ManagerReport +from workstream_map import display_columns + + +EXPORT_FILES = ("activities.csv", "matrix.csv", "metrics.csv", "data-quality.csv") + + +def build_activities_csv(activities: List[dict]) -> str: + """Build the activities CSV with normalized field names. + + Jira and GitHub records use different field names for the same concepts: + - url: Jira uses ``url``, GitHub uses ``pr_url`` + - issue_key: Jira uses ``issue_key``, GitHub uses ``source_key`` + - repo: only present on GitHub records + + Missing fields write empty strings, never the literal "None". + Rows are sorted by (member, ts) for stable diff output. + """ + buffer = io.StringIO() + writer = csv.writer(buffer, lineterminator="\n") + writer.writerow([ + "member", + "source", + "kind", + "workstream", + "attribution_source", + "repo", + "issue_key", + "url", + "ts", + "unattributed_reason", + ]) + + # Sort activities by member, then timestamp for stable output + sorted_activities = sorted( + activities, + key=lambda a: (a.get("member", ""), a.get("ts", "")), + ) + + for activity in sorted_activities: + kind = activity.get("kind", "") + source = report.classify_source(kind) + + # Normalize field names between Jira and GitHub + url = activity.get("url") or activity.get("pr_url") or "" + issue_key = activity.get("issue_key") or activity.get("source_key") or "" + repo = activity.get("repo", "") + + writer.writerow([ + activity.get("member", ""), + source, + kind, + activity.get("workstream", ""), + activity.get("attribution_source", ""), + repo, + issue_key, + url, + activity.get("ts", ""), + activity.get("unattributed_reason", ""), + ]) + + return buffer.getvalue() + + +def build_matrix_csv(report: ManagerReport) -> str: + """Build the matrix CSV with per-workstream counts and allocation flags. + + Columns: member, , TOTAL, workstreams_touched, over, under, narrow. + The workstream columns come from display_columns() (6 canonical + SHARED), + not metrics.workstreams (which excludes SHARED so workstreams_touched stays + "N of 6"). + """ + buffer = io.StringIO() + writer = csv.writer(buffer, lineterminator="\n") + + workstreams = display_columns() + header = ["member"] + workstreams + ["TOTAL", "workstreams_touched", "over", "under", "narrow"] + writer.writerow(header) + + for member in report.metrics.members: + # Get counts for each workstream + row = [member] + for ws in workstreams: + count = report.matrix.counts.get(member, {}).get(ws, 0) + row.append(count) + + # Add total + total = report.signals.total_by_member.get(member, 0) + row.append(total) + + # Add how many of the six workstreams this member touched + row.append(report.metrics.workstreams_touched_by_member.get(member, 0)) + + # Add flags + row.append("true" if member in report.signals.over_allocated else "false") + row.append("true" if member in report.signals.under_allocated else "false") + row.append("true" if member in report.signals.narrow else "false") + + writer.writerow(row) + + return buffer.getvalue() + + +def build_metrics_csv(report: ManagerReport) -> str: + """Build the metrics CSV with team scalars and per-workstream totals. + + Every row label says what it counts. Team scalars first + (mean_people_per_workstream, mean_workstreams_per_person, active_members, + total_members, team_median_total, grid_max, unattributed), then + per-workstream total and contributors counts. + """ + buffer = io.StringIO() + writer = csv.writer(buffer, lineterminator="\n") + + writer.writerow(["metric", "value"]) + + # Team scalars + writer.writerow( + ["mean_people_per_workstream", round(report.metrics.mean_people_per_workstream, 4)] + ) + writer.writerow( + ["mean_workstreams_per_person", round(report.metrics.mean_workstreams_per_person, 4)] + ) + writer.writerow(["active_members", report.metrics.active_member_count]) + writer.writerow(["total_members", report.metrics.total_member_count]) + writer.writerow(["team_median_total", round(report.signals.team_median_total, 4)]) + writer.writerow(["grid_max", report.signals.grid_max]) + writer.writerow(["unattributed", report.unattributed_count]) + + # Per-workstream metrics. There is deliberately no second per-workstream + # contributor count here: contributors: is the only one, and it also + # covers SHARED. + for workstream, count in report.signals.total_by_workstream.items(): + writer.writerow([f"total:{workstream}", count]) + + for workstream, count in report.signals.contributors_by_workstream.items(): + writer.writerow([f"contributors:{workstream}", count]) + + return buffer.getvalue() + + +def build_data_quality_csv(report: ManagerReport) -> str: + """Build the data quality CSV with all exclusion and unattributed categories. + + Categories: + - excluded_member + - excluded_personal_repo + - unattributed_repo + - unattributed_jira_no_parent + - unattributed_jira_parent_empty + + Unlike the text renderer, this does NOT truncate repo lists — every repo + gets a row since this is the raw export. + """ + buffer = io.StringIO() + writer = csv.writer(buffer, lineterminator="\n") + + writer.writerow(["category", "label", "count"]) + + if not report.data_quality: + return buffer.getvalue() + + dq = report.data_quality + + # Excluded members + for member in dq.excluded_members: + writer.writerow(["excluded_member", member, ""]) + + # Excluded personal repos + for repo, count in sorted(dq.excluded_personal_repos.items(), key=lambda x: x[1], reverse=True): + writer.writerow(["excluded_personal_repo", repo, count]) + + # Unattributed repos + for repo, count in sorted(dq.unattributed_by_repo.items(), key=lambda x: x[1], reverse=True): + writer.writerow(["unattributed_repo", repo, count]) + + # Unattributed Jira tickets + if dq.tickets_no_component_no_parent > 0: + writer.writerow([ + "unattributed_jira_no_parent", + "tickets_no_component_no_parent", + dq.tickets_no_component_no_parent, + ]) + + if dq.tickets_parent_also_empty > 0: + writer.writerow([ + "unattributed_jira_parent_empty", + "tickets_parent_also_empty", + dq.tickets_parent_also_empty, + ]) + + return buffer.getvalue() + + +def main(argv: Optional[List[str]] = None) -> int: + """Export contribution data as four CSV files.""" + parser = argparse.ArgumentParser( + description="Export cross-workstream contribution data as CSV files." + ) + parser.add_argument( + "--activity", + action="append", + required=True, + dest="activity_files", + metavar="PATH", + help="Activity JSON file (repeatable)", + ) + parser.add_argument( + "--members-file", + required=True, + metavar="PATH", + help="Roster JSON file", + ) + parser.add_argument( + "--period", + required=True, + metavar="LABEL", + help="Period label (e.g., 2026Q3)", + ) + parser.add_argument( + "--output-dir", + required=True, + metavar="PATH", + help="Output directory for CSV files", + ) + + args = parser.parse_args(argv) + + # Load activities + activities, collector_meta = report._load_activities(args.activity_files) + + # Load roster and prepare member lists + import json + with open(args.members_file, encoding="utf-8") as handle: + roster_entries = json.load(handle) + + excluded, jira_to_display, display_names = report.prepare_roster( + roster_entries + ) + + # Translate activities to display names + activities = report.translate_activities_to_display_names(activities, jira_to_display) + + # Build manager report + workstreams = display_columns() + manager_report = report.build_manager_report( + activities, display_names, workstreams, args.period, excluded, collector_meta + ) + + # Create output directory + os.makedirs(args.output_dir, exist_ok=True) + + # Build and write each CSV file + files_written = [] + + # activities.csv + activities_csv = build_activities_csv(activities) + activities_path = os.path.join(args.output_dir, "activities.csv") + with open(activities_path, "w", encoding="utf-8") as handle: + handle.write(activities_csv) + files_written.append(activities_path) + print(activities_path) + + # matrix.csv + matrix_csv = build_matrix_csv(manager_report) + matrix_path = os.path.join(args.output_dir, "matrix.csv") + with open(matrix_path, "w", encoding="utf-8") as handle: + handle.write(matrix_csv) + files_written.append(matrix_path) + print(matrix_path) + + # metrics.csv + metrics_csv = build_metrics_csv(manager_report) + metrics_path = os.path.join(args.output_dir, "metrics.csv") + with open(metrics_path, "w", encoding="utf-8") as handle: + handle.write(metrics_csv) + files_written.append(metrics_path) + print(metrics_path) + + # data-quality.csv + data_quality_csv = build_data_quality_csv(manager_report) + data_quality_path = os.path.join(args.output_dir, "data-quality.csv") + with open(data_quality_path, "w", encoding="utf-8") as handle: + handle.write(data_quality_csv) + files_written.append(data_quality_path) + print(data_quality_path) + + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/plugins/edge-contribution/bin/load_context.py b/plugins/edge-contribution/bin/load_context.py new file mode 100644 index 00000000..5160fd61 --- /dev/null +++ b/plugins/edge-contribution/bin/load_context.py @@ -0,0 +1,229 @@ +#!/usr/bin/env python3 +"""Load the OpenShift Edge roster from the edge-context repository. + +Fetches ``people/team-roster.md`` (a GitHub-flavored markdown table) straight +from GitHub via the ``gh`` CLI — no local checkout required — then parses it into +typed ``Member`` records and filters the roster down to engineering and QE roles, +which are the contributors this plugin measures. Workstreams come from +``workstream_map`` — this module does not parse workstream markdown. + +The ``gh`` command runner is injected, so parsing and attribution stay pure and +unit-testable without a subprocess or network access. + +Run standalone to emit the filtered roster as JSON: + + python3 load_context.py [--repo owner/name] [--ref branch] [--output roster.json] +""" + +from __future__ import annotations + +import argparse +import base64 +import binascii +import json +import re +import sys +from dataclasses import asdict, dataclass +from typing import List, Optional + +from _common import CommandRunner, build_default_runner, run_with_rate_limit_retry + +DEFAULT_ROSTER_REPO = "openshift-eng/edge-context" +ROSTER_CONTENT_PATH = "people/team-roster.md" + +_NAME_LINK_PATTERN = re.compile(r"^\[(?P.+?)\]\((?P[^)]+)\)$") +_EXPECTED_COLUMN_COUNT = 4 + + +class ContextParseError(Exception): + """Raised when edge-context data cannot be read or parsed as expected.""" + + +@dataclass(frozen=True) +class Member: + """A roster member and the identities used to attribute their activity.""" + + name: str + github: str + role: str + jira_username: str + + +def is_engineering_or_qe(role: str) -> bool: + """Return whether a role title is a software engineering or QE contributor. + + Managers, product managers, and product security roles are excluded even + though their titles contain the word "Engineer"/"Engineering"; the filter + keys on the full phrases "software engineer" and "software quality engineer". + """ + normalized = role.lower() + return "software engineer" in normalized or "software quality engineer" in normalized + + +def _split_table_row(line: str) -> List[str]: + cells = line.split("|") + if cells and cells[0].strip() == "": + cells = cells[1:] + if cells and cells[-1].strip() == "": + cells = cells[:-1] + return [cell.strip() for cell in cells] + + +def _is_separator_row(cells: List[str]) -> bool: + return bool(cells) and all(set(cell) <= set("-: ") and "-" in cell for cell in cells) + + +def _is_header_row(cells: List[str]) -> bool: + lowered = [cell.lower() for cell in cells] + return "name" in lowered and "github" in lowered and "role" in lowered + + +def _kerberos_from_rover_url(url: str) -> str: + kerberos = url.rstrip("/").rsplit("/", 1)[-1].strip() + if not kerberos or "/" in kerberos or "." in kerberos: + raise ContextParseError(f"cannot derive kerberos id from profile URL: {url!r}") + return kerberos + + +def _member_from_cells(cells: List[str]) -> Member: + if len(cells) != _EXPECTED_COLUMN_COUNT: + raise ContextParseError( + f"roster row has {len(cells)} columns, expected {_EXPECTED_COLUMN_COUNT}: {cells!r}" + ) + name_match = _NAME_LINK_PATTERN.match(cells[0]) + if name_match is None: + raise ContextParseError(f"roster name cell is not a Rover profile link: {cells[0]!r}") + kerberos = _kerberos_from_rover_url(name_match.group("url")) + return Member( + name=name_match.group("name").strip(), + github=cells[1], + role=cells[2], + jira_username=f"{kerberos}@redhat.com", + ) + + +def parse_roster(markdown: str) -> List[Member]: + """Parse every member row from the roster markdown table (unfiltered). + + Raises ``ContextParseError`` if no roster table is present or if a data row + is malformed (wrong column count or a name cell without a Rover link) — bad + rows are surfaced, never silently skipped. + """ + lines = markdown.splitlines() + header_index = _find_header_index(lines) + if header_index is None: + raise ContextParseError("no roster table (Name | GitHub | Role | ...) found in markdown") + + separator = _split_table_row(lines[header_index + 1]) if header_index + 1 < len(lines) else [] + if not _is_separator_row(separator): + raise ContextParseError("roster header is not followed by a table separator row") + + members: List[Member] = [] + for line in lines[header_index + 2 :]: + if not line.strip().startswith("|"): + break + members.append(_member_from_cells(_split_table_row(line))) + if not members: + raise ContextParseError("roster table has a header but no member rows") + return members + + +def _find_header_index(lines: List[str]) -> Optional[int]: + for index, line in enumerate(lines): + if line.strip().startswith("|") and _is_header_row(_split_table_row(line)): + return index + return None + + +def filter_contributors(members: List[Member]) -> List[Member]: + """Keep only engineering and QE members (see ``is_engineering_or_qe``).""" + return [member for member in members if is_engineering_or_qe(member.role)] + + +def _roster_api_command(repo: str, ref: Optional[str]) -> List[str]: + endpoint = f"repos/{repo}/contents/{ROSTER_CONTENT_PATH}" + if ref: + endpoint = f"{endpoint}?ref={ref}" + return ["gh", "api", endpoint] + + +def _parse_contents_json(stdout: str) -> dict: + try: + return json.loads(stdout) + except ValueError as error: + raise ContextParseError(f"malformed JSON from gh api: {error}") from error + + +def _decode_contents(payload: dict) -> str: + encoding = payload.get("encoding") + if encoding != "base64": + raise ContextParseError(f"unexpected roster content encoding: {encoding!r}") + content = payload.get("content") + if not content: + raise ContextParseError("gh api returned no roster file content") + try: + raw = base64.b64decode(content) + except (binascii.Error, ValueError) as error: + raise ContextParseError(f"roster content is not valid base64: {error}") from error + try: + return raw.decode("utf-8") + except UnicodeDecodeError as error: + raise ContextParseError(f"roster content is not valid UTF-8: {error}") from error + + +def fetch_roster_markdown( + runner: CommandRunner, + repo: str = DEFAULT_ROSTER_REPO, + ref: Optional[str] = None, +) -> str: + """Fetch the raw roster markdown from GitHub via ``gh api``. + + Raises ``ContextParseError`` when ``gh`` fails (missing auth, no access, a + 404, or a network error) or when the response is not the expected + base64-encoded contents payload — the failure is always surfaced, never a + silent empty roster. + """ + result = run_with_rate_limit_retry(runner, _roster_api_command(repo, ref)) + if result.returncode != 0: + raise ContextParseError( + f"gh api failed (exit {result.returncode}): {result.stderr.strip()}" + ) + return _decode_contents(_parse_contents_json(result.stdout)) + + +def load_roster_from_github( + runner: CommandRunner, + repo: str = DEFAULT_ROSTER_REPO, + ref: Optional[str] = None, +) -> List[Member]: + """Fetch and parse the roster, returning only Eng/QE contributors.""" + return filter_contributors(parse_roster(fetch_roster_markdown(runner, repo, ref))) + + +def main() -> None: + parser = argparse.ArgumentParser( + description="Load the Eng/QE roster from edge-context via the gh CLI" + ) + parser.add_argument( + "--repo", default=DEFAULT_ROSTER_REPO, help="owner/name of the roster repository" + ) + parser.add_argument("--ref", help="git ref to read (default: the repo's default branch)") + parser.add_argument("--output", help="write roster JSON here (default: stdout)") + args = parser.parse_args() + + try: + members = load_roster_from_github(build_default_runner(), args.repo, args.ref) + except ContextParseError as error: + print(f"error: {error}", file=sys.stderr) + raise SystemExit(1) from error + + payload = json.dumps([asdict(member) for member in members], indent=2) + if args.output: + with open(args.output, "w", encoding="utf-8") as handle: + handle.write(payload + "\n") + else: + print(payload) + + +if __name__ == "__main__": + main() diff --git a/plugins/edge-contribution/bin/metrics.py b/plugins/edge-contribution/bin/metrics.py new file mode 100644 index 00000000..30f6e0d6 --- /dev/null +++ b/plugins/edge-contribution/bin/metrics.py @@ -0,0 +1,267 @@ +#!/usr/bin/env python3 +"""Cross-workstream contribution metrics from a binary contribution matrix. + +Given which workstreams each roster member touched at least once in a period, +this computes the counts defined in ``references/metrics.md``. Every name here +says what it counts, so a reader never has to look up a term: + +* **Workstreams touched** (per member) — how many of the six canonical + workstreams the member contributed to at least once. Reported as "N of 6". +* **Mean people per workstream** — the average number of distinct contributors + a workstream had, over the six canonical workstreams. +* **Mean workstreams per person** — the average number of workstreams an + *active* member touched (members with zero activity are left out, so they do + not deflate the average). + +This module also computes **allocation signals** from the contribution counts +matrix (``counts``), which are relative-to-team indicators of workload distribution: + +* **Over-allocated** — members whose total contribution count is more than twice + the team median among active members. +* **Under-allocated** — members whose total is less than half the team median, but + greater than zero (zero-activity members are excluded from the comparison). +* **Narrow** — members where a single workstream accounts for more than 70% of + their total contributions, indicating siloing. + +The binary ``touched`` field drives the workstreams-touched counts, while +``counts`` drives allocation signals. This asymmetry is intentional: the SHARED +column (for cross-cutting CI/tooling work) contributes to a member's total +workload but is NOT a canonical workstream and does not count toward +workstreams touched. + +All inputs are pure data, so this module has no I/O and is trivially testable. +""" + +from __future__ import annotations + +import statistics +from dataclasses import dataclass, field +from typing import Dict, List, Set + +from workstream_map import SHARED_COLUMN + + +@dataclass(frozen=True) +class ContributionMatrix: + """Who touched which workstreams during the reporting period. + + ``touched`` maps a member to the set of workstream acronyms they contributed + to. Acronyms outside ``workstreams`` are ignored, and a member absent from + ``touched`` is treated as having contributed to nothing. + + ``counts`` maps a member to a dict of workstream -> count. This is the weighted + view used for allocation signals. Members or workstreams absent from ``counts`` + are treated as zero. Columns present in ``counts`` but absent from + ``workstreams`` are ignored when computing allocation signals. + """ + + members: List[str] + workstreams: List[str] + touched: Dict[str, Set[str]] = field(default_factory=dict) + counts: Dict[str, Dict[str, int]] = field(default_factory=dict) + + +@dataclass(frozen=True) +class TeamMetrics: + """The full metric bundle for a team over one reporting period.""" + + workstreams: List[str] + members: List[str] + workstreams_touched_by_member: Dict[str, int] + mean_people_per_workstream: float + mean_workstreams_per_person: float + active_member_count: int + total_member_count: int + + +@dataclass(frozen=True) +class AllocationSignals: + """Relative workload distribution signals for a team over one reporting period. + + These are computed from the weighted ``counts`` matrix and are relative to the + team median. The thresholds (2.0x over, 0.5x under, 70% narrow) are heuristic + indicators of imbalance, not capacity targets. Zero-activity members are excluded + from the median and under/over classifications, as they may be on leave, new, or + misattributed rather than genuinely under-allocated in a comparable sense. + """ + + total_by_member: Dict[str, int] + total_by_workstream: Dict[str, int] + contributors_by_workstream: Dict[str, int] + team_median_total: float + grid_max: int + over_allocated: Set[str] + under_allocated: Set[str] + narrow: Set[str] + + +# Allocation signal thresholds (relative to team median). +# Over-allocated: members with > 2.0x the team median total. +OVER_ALLOCATED_THRESHOLD = 2.0 +# Under-allocated: members with < 0.5x the team median AND total > 0. +UNDER_ALLOCATED_THRESHOLD = 0.5 +# Narrow: members where a single workstream is > 70% of their total. +NARROW_THRESHOLD = 0.70 + + +def _canonical_workstreams(matrix: ContributionMatrix) -> List[str]: + """Return the matrix columns that count as workstreams, i.e. everything but SHARED. + + Callers pass ``display_columns()`` (the six plus SHARED) so that allocation + signals cover cross-cutting work, but the workstreams-touched count must stay + "N of 6". Filtering here rather than trusting the caller keeps that invariant + true no matter which column list is supplied. + """ + return [ws for ws in matrix.workstreams if ws != SHARED_COLUMN] + + +def _touched_workstreams(matrix: ContributionMatrix, member: str) -> Set[str]: + return matrix.touched.get(member, set()) & set(_canonical_workstreams(matrix)) + + +def workstreams_touched(matrix: ContributionMatrix, member: str) -> int: + """Return how many canonical workstreams ``member`` touched. + + Raises ``ValueError`` if ``member`` is not part of the matrix roster. + """ + if member not in matrix.members: + raise ValueError(f"member not in matrix roster: {member!r}") + return len(_touched_workstreams(matrix, member)) + + +def workstreams_touched_by_member(matrix: ContributionMatrix) -> Dict[str, int]: + """Return the workstreams-touched count for every member, in roster order.""" + return {member: len(_touched_workstreams(matrix, member)) for member in matrix.members} + + +def total_touches(matrix: ContributionMatrix) -> int: + """Return how many (member, workstream) pairs saw at least one contribution. + + This is the number of 1s in the binary matrix, i.e. the sum of every + member's workstreams-touched count. It is NOT the number of activity + records — that is the sum of the weighted ``counts`` matrix, which + ``compute_allocation_signals`` reports as ``total_by_member``. + """ + return sum(workstreams_touched_by_member(matrix).values()) + + +def mean_people_per_workstream(matrix: ContributionMatrix) -> float: + """Return the average number of distinct contributors per canonical workstream.""" + canonical = _canonical_workstreams(matrix) + if not canonical: + return 0.0 + return total_touches(matrix) / len(canonical) + + +def active_members(matrix: ContributionMatrix) -> List[str]: + """Return members with at least one contribution, in roster order.""" + return [member for member, count in workstreams_touched_by_member(matrix).items() if count > 0] + + +def mean_workstreams_per_person(matrix: ContributionMatrix) -> float: + """Return the average workstreams-touched count over active members. + + Returns 0.0 when nobody is active. Inactive members are excluded so they + do not deflate the average. + """ + active = active_members(matrix) + if not active: + return 0.0 + return total_touches(matrix) / len(active) + + +def compute_team_metrics(matrix: ContributionMatrix) -> TeamMetrics: + """Compute the full metric bundle for a reporting period.""" + return TeamMetrics( + workstreams=_canonical_workstreams(matrix), + members=list(matrix.members), + workstreams_touched_by_member=workstreams_touched_by_member(matrix), + mean_people_per_workstream=mean_people_per_workstream(matrix), + mean_workstreams_per_person=mean_workstreams_per_person(matrix), + active_member_count=len(active_members(matrix)), + total_member_count=len(matrix.members), + ) + + +def compute_allocation_signals(matrix: ContributionMatrix) -> AllocationSignals: + """Compute relative workload distribution signals from the counts matrix. + + All counts are summed over the columns present in ``matrix.workstreams`` only. + Columns in ``counts`` that are not in ``workstreams`` are ignored. The SHARED + column (if present in ``workstreams``) contributes to member totals, even though + it is not a canonical workstream and does not count toward workstreams touched. + + Zero-activity members (total == 0) are excluded from the team median calculation + and from over/under-allocated sets. This treats them as on leave, new, or + misattributed rather than under-allocated in a comparable sense. + + Returns: + AllocationSignals with all fields populated, even for an empty matrix. + """ + workstream_set = set(matrix.workstreams) + + # Compute total_by_member: sum each member's counts over valid workstreams. + total_by_member: Dict[str, int] = {} + for member in matrix.members: + member_counts = matrix.counts.get(member, {}) + total = sum(count for ws, count in member_counts.items() if ws in workstream_set) + total_by_member[member] = total + + # Compute total_by_workstream: sum down each column over all members. + total_by_workstream: Dict[str, int] = {ws: 0 for ws in matrix.workstreams} + for member in matrix.members: + member_counts = matrix.counts.get(member, {}) + for ws in matrix.workstreams: + total_by_workstream[ws] += member_counts.get(ws, 0) + + # Compute contributors_by_workstream: count members with count > 0 per workstream. + contributors_by_workstream: Dict[str, int] = {ws: 0 for ws in matrix.workstreams} + for member in matrix.members: + member_counts = matrix.counts.get(member, {}) + for ws in matrix.workstreams: + if member_counts.get(ws, 0) > 0: + contributors_by_workstream[ws] += 1 + + # Compute team_median_total over active members only (total > 0). + active_totals = [total for total in total_by_member.values() if total > 0] + team_median_total = statistics.median(active_totals) if active_totals else 0.0 + + # Compute grid_max: largest single cell value in the whole matrix. + grid_max = 0 + for member_counts in matrix.counts.values(): + for ws, count in member_counts.items(): + if ws in workstream_set: + grid_max = max(grid_max, count) + + # Compute over_allocated: members with total > 2.0 * team_median_total. + over_allocated: Set[str] = set() + for member, total in total_by_member.items(): + if total > OVER_ALLOCATED_THRESHOLD * team_median_total: + over_allocated.add(member) + + # Compute under_allocated: members with 0 < total < 0.5 * team_median_total. + under_allocated: Set[str] = set() + for member, total in total_by_member.items(): + if 0 < total < UNDER_ALLOCATED_THRESHOLD * team_median_total: + under_allocated.add(member) + + # Compute narrow: members where a single workstream is > 70% of their total. + narrow: Set[str] = set() + for member, total in total_by_member.items(): + if total > 0: + member_counts = matrix.counts.get(member, {}) + for ws in matrix.workstreams: + if member_counts.get(ws, 0) > NARROW_THRESHOLD * total: + narrow.add(member) + break + + return AllocationSignals( + total_by_member=total_by_member, + total_by_workstream=total_by_workstream, + contributors_by_workstream=contributors_by_workstream, + team_median_total=team_median_total, + grid_max=grid_max, + over_allocated=over_allocated, + under_allocated=under_allocated, + narrow=narrow, + ) diff --git a/plugins/edge-contribution/bin/render.py b/plugins/edge-contribution/bin/render.py new file mode 100644 index 00000000..95932fe9 --- /dev/null +++ b/plugins/edge-contribution/bin/render.py @@ -0,0 +1,1112 @@ +#!/usr/bin/env python3 +"""Render cross-workstream contribution reports. + +Three report shapes are supported: + +* **Manager** — a people x workstream heatmap showing contribution intensity via + a grey-to-green gradient, plus allocation signals (over/under/narrow flags). + Rendered as plain text for the terminal: the grid, scale legend, and flag + legend. The team-level means, data quality notes, and other analytics are in + the Summary view. +* **IC** — one member's workstreams-touched count ("N of M") and a per-workstream + activity breakdown, rendered as plain text. +* **Summary** — an executive summary reporting what was counted, how each item + was attributed, team-level means, per-member totals and workstreams touched, + and data quality notes. Formats: ``text`` (default) and ``markdown`` + (for Slack/docs). + +Rendering is pure: it turns already-computed data structures into a string, with +no I/O, so every format is unit-testable. + +The gradient uses log-spaced bands on a single global maximum so brightness is +comparable across the entire grid. Allocation signals are relative to the team +median and highlight imbalance, not capacity targets. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from math import ceil, log1p +from typing import Dict, List, Optional, Tuple + +from metrics import AllocationSignals, ContributionMatrix, TeamMetrics, compute_allocation_signals + +# Canonical order of activity kinds for IC output columns. +# No "comment" kind: Jira comments are not collected because Jira Cloud +# omits comment author emails and queries could never match a member. +ACTIVITY_KINDS: List[str] = [ + "assignee", + "qa", + "ocpstrat_role", + "pr_authored", + "pr_reviewed", +] + +_VALID_FORMATS = ("text",) +_SUMMARY_FORMATS = ("text", "markdown") + +# Gradient glyphs and colors for contribution intensity heatmap. +# Band 0 = no activity, bands 1..4 = log-spaced buckets of increasing volume. +_BAND_GLYPHS = ("·", "░", "▒", "▓", "█") +_BAND_ASCII = (".", ":", "*", "+", "#") +_BAND_COLORS = (59, 65, 71, 77, 40) # xterm-256: grey #5f5f5f -> green #00d700 + + +_TOP_REPO_LIMIT = 8 +_TWO_NODE_TOOLBOX_NOTE = "(mixed arbiter/fencing; not mapped by design)" + + +def _top_repos( + tally: Dict[str, int], limit: int = _TOP_REPO_LIMIT +) -> Tuple[List[Tuple[str, int]], int, int]: + """Split a repo tally into the biggest ``limit`` entries and a remainder. + + Returns ``(top, remaining_repo_count, remaining_pr_count)``. Every repo list + in the data-quality block truncates the same way, so they share this. + """ + items = sorted(tally.items(), key=lambda pair: pair[1], reverse=True) + rest = items[limit:] + return items[:limit], len(rest), sum(count for _repo, count in rest) + + +def _repo_note(repo: str) -> str: + """Return the trailing explanation for a repo that is unmapped on purpose.""" + if repo == "openshift-eng/two-node-toolbox": + return f" {_TWO_NODE_TOOLBOX_NOTE}" + return "" + + +@dataclass(frozen=True) +class DataQuality: + """Data quality notes for a manager report.""" + + unattributed_by_repo: Dict[str, int] = field(default_factory=dict) + tickets_no_component_no_parent: int = 0 + tickets_parent_also_empty: int = 0 + excluded_personal_repo_prs: int = 0 + # Which repos those excluded PRs lived in. The org allowlist behind the + # exclusion is hand-kept, so naming the repos is how a legitimate org that + # is missing from it becomes visible instead of vanishing into the total. + excluded_personal_repos: Dict[str, int] = field(default_factory=dict) + excluded_members: List[str] = field(default_factory=list) + + +@dataclass(frozen=True) +class WorkstreamLine: + """Per-workstream volume and top contributor for an executive summary.""" + + workstream: str + volume: int + contributors: int + top_contributor: Optional[str] + top_share: int # percent, rounded int + + +@dataclass(frozen=True) +class MemberLine: + """Per-member totals and workstreams-touched count for an executive summary.""" + + member: str + total: int + workstreams_touched: int + top_workstream: Optional[str] + top_share: int # percent, rounded int + + +@dataclass(frozen=True) +class ExecutiveSummary: + """Executive summary of cross-workstream contribution for a reporting period. + + Reports what was counted, how each item was attributed, team-level means, + per-member totals and workstreams touched, and data quality notes. This is + the manager-facing narrative view that complements the at-a-glance heatmap + grid. + """ + + period_label: str + window: Optional[Tuple[str, str]] # ("2026-07-01", "2026-09-30") + metrics: TeamMetrics + signals: AllocationSignals + data_quality: DataQuality + counts_by_source_kind: Dict[str, Dict[str, int]] # "jira"/"github" -> kind -> n + counts_by_attribution: Dict[str, int] # source -> n, "" for None + per_member: List[MemberLine] # sorted total desc + per_workstream: List[WorkstreamLine] # sorted volume desc + total_records: int + jira_projects: Tuple[str, ...] + ocpstrat_project: str + allowed_org_count: int + out_of_window_records: int = 0 + + +@dataclass(frozen=True) +class WorkstreamActivity: + """One workstream's activity for a single member, counted by kind.""" + + workstream: str + counts: Dict[str, int] = field(default_factory=dict) + + def total(self) -> int: + return sum(self.counts.values()) + + +@dataclass(frozen=True) +class ICReport: + """Personal workstream-coverage report for a single member.""" + + period_label: str + member: str + workstreams_touched: int + total_workstreams: int + activity: List[WorkstreamActivity] = field(default_factory=list) + unattributed_count: int = 0 + + +@dataclass(frozen=True) +class ManagerReport: + """Team heatmap report across the full roster.""" + + period_label: str + metrics: TeamMetrics + matrix: ContributionMatrix + unattributed_count: int = 0 + signals: Optional[AllocationSignals] = None + data_quality: Optional[DataQuality] = None + + +def _require_known_format(output_format: str) -> None: + if output_format not in _VALID_FORMATS: + raise ValueError( + f"unknown output format {output_format!r}; expected one of {_VALID_FORMATS}" + ) + + +def _require_known_summary_format(output_format: str) -> None: + if output_format not in _SUMMARY_FORMATS: + raise ValueError( + f"unknown summary format {output_format!r}; expected one of {_SUMMARY_FORMATS}" + ) + + +def shade_band(count: int, grid_max: int) -> int: + """Return a 0..4 shade band index for a contribution count on a log scale. + + Band 0 = no activity (count == 0); bands 1..4 = log-spaced buckets of + increasing volume relative to ``grid_max``. The distribution is log-based + so the full range (1 to grid_max) is split into four perceptually-equal + steps rather than collapsing low counts into a single band. + + Args: + count: Number of contributions in a single cell. + grid_max: Maximum count across the entire grid (for normalization). + + Returns: + Shade band index 0..4 inclusive. Returns 0 when count <= 0 or grid_max <= 0, + and clamps to 4 when count exceeds grid_max. + """ + if count <= 0 or grid_max <= 0: + return 0 + # log1p(count) / log1p(grid_max) ranges from ~0 to 1, scaling by 4 spreads + # it across bands 1..4. Ceil ensures count=1 lands in band 1, not 0. + # Max(1, ...) guards against count < 1 after the log transform, and + # min(..., 4) clamps counts above grid_max to the top band. + return max(1, min(4, ceil(log1p(count) / log1p(grid_max) * 4))) + + +# --- Manager report --------------------------------------------------------- + + +def render_manager( + report: ManagerReport, output_format: str, *, color: bool = True, ascii_only: bool = False +) -> str: + """Render the manager heatmap in text format. + + Args: + report: The manager report to render. + output_format: Must be "text". + color: Whether to include ANSI color codes. + ascii_only: Whether to use strict ASCII glyphs instead of Unicode. + + Returns: + Rendered report as a string. + """ + _require_known_format(output_format) + return _manager_text(report, color=color, ascii_only=ascii_only) + + +def _cell_is_filled(report: ManagerReport, member: str, workstream: str) -> bool: + return workstream in report.matrix.touched.get(member, set()) + + +def _cell_count(report: ManagerReport, member: str, workstream: str) -> int: + """Return the contribution count for a member/workstream cell.""" + return report.matrix.counts.get(member, {}).get(workstream, 0) + + +def _colorize_glyph(glyph: str, band: int, color: bool) -> str: + """Wrap a glyph in ANSI color codes if color is enabled.""" + if not color: + return glyph + color_code = _BAND_COLORS[band] + return f"\x1b[38;5;{color_code}m{glyph}\x1b[0m" + + +def _compute_scale_legend(grid_max: int) -> List[tuple[int, str]]: + """Compute the scale legend by inverting shade_band for each band 1..4. + + Returns a list of (band_index, label) tuples describing the count range for + each non-zero band. The label is "N" for a single-value band or "N-M" for + a range. Band 0 (none) is always "none" and is not returned here. + """ + if grid_max <= 0: + return [] + + # Find the boundary counts for each band by testing shade_band. + # We want the first count that lands in each band, and the last count + # before transitioning to the next band. + legend: List[tuple[int, str]] = [] + for band in range(1, 5): + # Find the first count that produces this band. + first = None + for count in range(1, grid_max + 1): + if shade_band(count, grid_max) == band: + first = count + break + if first is None: + continue + + # Find the last count that produces this band. + last = first + for count in range(first + 1, grid_max + 1): + if shade_band(count, grid_max) == band: + last = count + else: + break + + if first == last: + legend.append((band, str(first))) + else: + legend.append((band, f"{first}-{last}")) + + return legend + + +def _manager_text(report: ManagerReport, *, color: bool = True, ascii_only: bool = False) -> str: + from workstream_map import SHARED_COLUMN, display_columns + + metrics_data = report.metrics + signals = ( + report.signals if report.signals is not None else compute_allocation_signals(report.matrix) + ) + + # Choose glyph set based on ascii_only. + glyphs = _BAND_ASCII if ascii_only else _BAND_GLYPHS + em_dash = "-" if ascii_only else "—" + box_vert = "|" if ascii_only else "│" + flag_over = "^" if ascii_only else "▲" + flag_under = "v" if ascii_only else "▼" + + # Determine column order: six workstreams + SHARED. + display_cols = display_columns() + six_workstreams = [ws for ws in display_cols if ws != SHARED_COLUMN] + shared_present = SHARED_COLUMN in display_cols + + # Sort members by total descending. + members_sorted = sorted( + metrics_data.members, key=lambda m: signals.total_by_member.get(m, 0), reverse=True + ) + + # Compute column widths. + name_width = max((len(m) for m in members_sorted), default=10) + ws_widths = {ws: max(len(ws), 2) for ws in display_cols} + shared_width = ws_widths.get(SHARED_COLUMN, 6) + total_width = max(5, len(str(max(signals.total_by_member.values(), default=0)))) + + # Build title line. + active_count = metrics_data.active_member_count + total_count = metrics_data.total_member_count + lines = [ + f"Cross-Workstream Allocation {em_dash} {report.period_label} " + f"{active_count} of {total_count} members active", + "", + ] + + # Build header row. + header_parts = [f"{'':<{name_width}}"] + for ws in six_workstreams: + header_parts.append(ws.center(ws_widths[ws])) + if shared_present: + header_parts.append(box_vert) + header_parts.append(SHARED_COLUMN.center(shared_width)) + header_parts.append(box_vert) + header_parts.append("TOTAL".center(total_width)) + lines.append(" ".join(header_parts)) + + # Build member rows. + for member in members_sorted: + row_parts = [f"{member:<{name_width}}"] + for ws in six_workstreams: + count = _cell_count(report, member, ws) + band = shade_band(count, signals.grid_max) + glyph = glyphs[band] + # Double the glyph to make it more visible. + cell = (glyph * 2).center(ws_widths[ws]) + row_parts.append(_colorize_glyph(cell, band, color)) + + if shared_present: + shared_count = _cell_count(report, member, SHARED_COLUMN) + shared_band = shade_band(shared_count, signals.grid_max) + shared_glyph = glyphs[shared_band] + shared_cell = (shared_glyph * 2).center(shared_width) + row_parts.append(box_vert) + row_parts.append(_colorize_glyph(shared_cell, shared_band, color)) + + total = signals.total_by_member.get(member, 0) + row_parts.append(box_vert) + row_parts.append(str(total).rjust(total_width)) + + # Add flags. + flags = [] + if member in signals.over_allocated: + flags.append(flag_over) + if member in signals.under_allocated: + flags.append(flag_under) + if member in signals.narrow: + flags.append("narrow") + if flags: + row_parts.append(" " + " ".join(flags)) + + lines.append(" ".join(row_parts)) + + # Build footer separator. + sep_parts = [f"{'─' * name_width if not ascii_only else '-' * name_width}"] + for ws in six_workstreams: + sep_parts.append("─" * ws_widths[ws] if not ascii_only else "-" * ws_widths[ws]) + if shared_present: + sep_parts.append("┼" if not ascii_only else "+") + sep_parts.append("─" * shared_width if not ascii_only else "-" * shared_width) + sep_parts.append("┼" if not ascii_only else "+") + sep_parts.append("─" * total_width if not ascii_only else "-" * total_width) + lines.append(" ".join(sep_parts)) + + # Build TOTAL footer row. + total_parts = [f"{'TOTAL':<{name_width}}"] + for ws in six_workstreams: + ws_total = signals.total_by_workstream.get(ws, 0) + total_parts.append(str(ws_total).center(ws_widths[ws])) + if shared_present: + shared_total = signals.total_by_workstream.get(SHARED_COLUMN, 0) + total_parts.append(box_vert) + total_parts.append(str(shared_total).center(shared_width)) + total_parts.append(box_vert) + lines.append(" ".join(total_parts)) + + # Build PEOPLE footer row. + people_parts = [f"{'PEOPLE':<{name_width}}"] + for ws in six_workstreams: + people_count = signals.contributors_by_workstream.get(ws, 0) + people_parts.append(str(people_count).center(ws_widths[ws])) + if shared_present: + shared_people = signals.contributors_by_workstream.get(SHARED_COLUMN, 0) + people_parts.append(box_vert) + people_parts.append(str(shared_people).center(shared_width)) + people_parts.append(box_vert) + lines.append(" ".join(people_parts)) + + # Build scale legend. + lines.append("") + scale_legend = _compute_scale_legend(signals.grid_max) + legend_parts = [f" scale (log) {glyphs[0]} none"] + for band, label in scale_legend: + legend_parts.append(f"{glyphs[band]} {label}") + lines.append(" ".join(legend_parts)) + + # Build flags legend. + flags_legend = ( + f" flags {flag_over} >2x team median " + f"{flag_under} <0.5x team median " + f"narrow = >70% in one workstream" + ) + lines.append(flags_legend) + + # Text output stops at the flags legend, on purpose. + # + # The team-level means and the data-quality block used to be + # appended here. They restated what the grid's TOTAL and PEOPLE rows + # already show, and pushed the grid off the top of the terminal. The text + # view is the at-a-glance one; for analytics, use --view summary --show-sources. + # Pinned by TestManagerTextIsGridOnly in test_render.py. + return "\n".join(lines) + "\n" + + +# --- IC report -------------------------------------------------------------- + + +def render_ic(report: ICReport, output_format: str) -> str: + """Render a single member's workstream-coverage report in text format. + + Args: + report: The IC report to render. + output_format: Must be "text". + + Returns: + Rendered report as a string. + """ + _require_known_format(output_format) + return _ic_text(report) + + +def _ic_text(report: ICReport) -> str: + lines = [ + f"Personal Workstream Coverage — {report.period_label}", + f"Member: {report.member}", + f"Workstreams touched: {report.workstreams_touched} of {report.total_workstreams}", + "", + ] + if not report.activity: + lines.append("No attributed activity in this period.") + for activity in report.activity: + detail = ", ".join( + f"{kind}={activity.counts[kind]}" + for kind in ACTIVITY_KINDS + if activity.counts.get(kind) + ) + lines.append(f" {activity.workstream}: {detail or 'activity present'}") + if report.unattributed_count: + lines.append("") + lines.append(f"Note: {report.unattributed_count} unattributed items not shown.") + return "\n".join(lines) + "\n" + + +# --- Summary report --------------------------------------------------------- + + +def _compute_attribution_percentages(summary: ExecutiveSummary) -> Dict[str, int]: + """Compute percentages for attribution sources, shared by text and markdown. + + Returns a dict of source -> percentage (as integer 0-100). Percentages are + computed against total_records to avoid disagreement between formats. An + empty summary returns an empty dict without raising ZeroDivisionError. + """ + if summary.total_records == 0: + return {} + percentages = {} + for source, count in summary.counts_by_attribution.items(): + percentages[source] = round(100 * count / summary.total_records) + return percentages + + +def _summary_header(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the title and window lines. + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + + if format == "markdown": + lines.append(f"# OCP-Edge Cross-Workstream Contribution — {summary.period_label}") + lines.append("") + + if summary.window: + from_date, to_date = summary.window + included = summary.metrics.total_member_count + excluded = len(summary.data_quality.excluded_members) + full = included + excluded + lines.append(f"**Window:** {from_date} .. {to_date}") + if excluded > 0: + lines.append(f"**Roster:** {included} of {full} ({excluded} excluded)") + else: + lines.append(f"**Roster:** {included} of {full}") + lines.append("") + else: # text format + lines.append(f"OCP-Edge Cross-Workstream Contribution — {summary.period_label}") + if summary.window: + from_date, to_date = summary.window + included = summary.metrics.total_member_count + excluded = len(summary.data_quality.excluded_members) + full = included + excluded + if excluded > 0: + roster_line = f"Window {from_date} .. {to_date} Roster {included} of {full} ({excluded} excluded)" + else: + roster_line = f"Window {from_date} .. {to_date} Roster {included} of {full}" + lines.append(roster_line) + + return lines + + +def _summary_team(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the TEAM section. + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + from workstream_map import workstream_acronyms + + lines = [] + total_activity = sum(ml.total for ml in summary.per_member) + median = int(summary.signals.team_median_total) + people_per_ws = summary.metrics.mean_people_per_workstream + num_workstreams = len(workstream_acronyms()) + ws_per_person = summary.metrics.mean_workstreams_per_person + active_count = summary.metrics.active_member_count + + if format == "markdown": + lines.append("## Team") + lines.append("") + lines.append(f"- **{total_activity} contributions** — median {median} per person") + lines.append(f"- **{people_per_ws:.1f} people per workstream** (mean of {num_workstreams})") + lines.append(f"- **{ws_per_person:.1f} workstreams per person** ({active_count} active)") + lines.append("") + else: # text format + lines.append("TEAM") + lines.append(f" {total_activity} contributions median {median} per person") + lines.append(f" {people_per_ws:.1f} people per workstream (mean of {num_workstreams})") + lines.append(f" {ws_per_person:.1f} workstreams per person ({active_count} active)") + + return lines + + +def _summary_workstreams(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the WORKSTREAMS table. + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + + if format == "markdown": + lines.append("## Workstreams") + lines.append("") + lines.append("| Workstream | Volume | People | Largest Contributor | Share |") + lines.append("|------------|--------|--------|---------------------|-------|") + + for ws_line in summary.per_workstream: + if ws_line.top_contributor: + lines.append( + f"| {ws_line.workstream} | {ws_line.volume} | {ws_line.contributors} | " + f"{ws_line.top_contributor} | {ws_line.top_share}% |" + ) + else: + lines.append( + f"| {ws_line.workstream} | {ws_line.volume} | {ws_line.contributors} | | |" + ) + lines.append("") + else: # text format + lines.append("") + lines.append("WORKSTREAMS volume people largest contributor") + + for ws_line in summary.per_workstream: + volume_str = f"{ws_line.volume:>6d}" + people_str = f"{ws_line.contributors:>6d}" + + if ws_line.top_contributor: + contrib_str = f"{ws_line.top_contributor:28s} {ws_line.top_share:3d}%" + else: + contrib_str = "" + + lines.append( + f" {ws_line.workstream:30s} {volume_str:>6s} {people_str:>6s} {contrib_str}" + ) + + return lines + + +def _summary_people(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the PEOPLE table. + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + + if format == "markdown": + lines.append("## People") + lines.append("") + lines.append("| Member | Total | Workstreams Touched | Largest Workstream | Share |") + lines.append("|--------|-------|---------------------|-------------------|-------|") + + for member_line in summary.per_member: + touched_str = f"{member_line.workstreams_touched} of 6" + if member_line.top_workstream: + lines.append( + f"| {member_line.member} | {member_line.total} | {touched_str} | " + f"{member_line.top_workstream} | {member_line.top_share}% |" + ) + else: + lines.append(f"| {member_line.member} | {member_line.total} | {touched_str} | | |") + lines.append("") + else: # text format + lines.append("") + lines.append("PEOPLE total touched largest workstream") + + for member_line in summary.per_member: + total_str = f"{member_line.total:>5d}" + touched_str = f"{member_line.workstreams_touched} of 6" + + if member_line.top_workstream: + ws_str = f"{member_line.top_workstream:8s} {member_line.top_share:3d}%" + else: + ws_str = "" + + lines.append(f" {member_line.member:32s} {total_str:>5s} {touched_str:7s} {ws_str}") + + return lines + + +def _summary_data_health(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the DATA HEALTH section. + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + + unattributed_total = summary.counts_by_attribution.get("", 0) + total_records = summary.total_records + if total_records > 0: + pct = round(100 * unattributed_total / total_records) + else: + pct = 0 + + jira_no_comp = summary.data_quality.tickets_no_component_no_parent + jira_parent_empty = summary.data_quality.tickets_parent_also_empty + pr_count = sum(summary.data_quality.unattributed_by_repo.values()) + + if format == "markdown": + lines.append("## Data Health") + lines.append("") + lines.append( + f"**{unattributed_total} of {total_records} records ({pct}%)** could not be placed in a workstream." + ) + lines.append("") + lines.append( + f"- **{jira_no_comp + jira_parent_empty} Jira tickets** — {jira_no_comp} missing a component, {jira_parent_empty} whose epic is also empty" + ) + lines.append(f"- **{pr_count} PRs** — repos not mapped to a workstream") + lines.append("") + else: # text format + lines.append("") + lines.append("DATA HEALTH") + lines.append( + f" {unattributed_total} of {total_records} records ({pct}%) could not be placed in a workstream." + ) + lines.append( + f" {jira_no_comp + jira_parent_empty} Jira tickets {jira_no_comp} missing a component, {jira_parent_empty} whose epic is also empty" + ) + lines.append(f" {pr_count} PRs repos not mapped to a workstream") + + return lines + + +def _summary_what_counted(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the WHAT WAS COUNTED section (--show-sources only). + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + jira_counts = summary.counts_by_source_kind.get("jira", {}) + github_counts = summary.counts_by_source_kind.get("github", {}) + + if format == "markdown": + lines.append("## What Was Counted") + lines.append("") + if jira_counts or github_counts: + lines.append("| Source | Total | Details |") + lines.append("|--------|-------|---------|") + if jira_counts: + jira_total = sum(jira_counts.values()) + jira_detail = " · ".join(f"{kind} {jira_counts[kind]}" for kind in ACTIVITY_KINDS if kind in jira_counts) + lines.append(f"| Jira | {jira_total} | {jira_detail} |") + if github_counts: + github_total = sum(github_counts.values()) + github_detail = " · ".join(f"{kind} {github_counts[kind]}" for kind in ACTIVITY_KINDS if kind in github_counts) + lines.append(f"| GitHub | {github_total} | {github_detail} |") + lines.append("") + if jira_counts: + jira_projects_str = ", ".join(summary.jira_projects) + lines.append(f"**Jira projects:** {jira_projects_str} (assignee + QA contact)") + if summary.ocpstrat_project: + lines.append(f" plus {summary.ocpstrat_project} (SME / assignee roles)") + lines.append("") + lines.append("*Comments are not collected — Jira Cloud hides author emails*") + lines.append("") + if github_counts: + lines.append(f"**GitHub:** any repo whose owner is in the org allowlist ({summary.allowed_org_count} orgs)") + lines.append("") + lines.append("- **authored** = PR created in the window") + lines.append("- **reviewed** = PR updated in the window (GitHub cannot search by review date)") + lines.append("- every record is timestamped with the PR's creation date") + if summary.out_of_window_records > 0: + lines.append(f"- **{summary.out_of_window_records} records** carry a date before the window but are still counted") + lines.append("- → do not filter activities.csv by the ts column") + lines.append("") + else: # text + lines.append("") + lines.append(f"WHAT WAS COUNTED {summary.total_records} records") + if jira_counts: + jira_total = sum(jira_counts.values()) + jira_detail = " · ".join(f"{kind} {jira_counts[kind]}" for kind in ACTIVITY_KINDS if kind in jira_counts) + lines.append(f" Jira {jira_total:4d} {jira_detail}") + jira_projects_str = ", ".join(summary.jira_projects) + lines.append(f" projects {jira_projects_str} (assignee + QA contact)") + if summary.ocpstrat_project: + lines.append(f" plus {summary.ocpstrat_project} (SME / assignee roles)") + lines.append(" comments are not collected — Jira Cloud hides author emails") + if github_counts: + github_total = sum(github_counts.values()) + github_detail = " · ".join(f"{kind} {github_counts[kind]}" for kind in ACTIVITY_KINDS if kind in github_counts) + lines.append(f" GitHub {github_total:5d} {github_detail}") + lines.append(f" any repo whose owner is in the org allowlist ({summary.allowed_org_count} orgs)") + lines.append(" authored = PR created in the window") + lines.append(" reviewed = PR updated in the window (GitHub cannot search by review date)") + lines.append(" every record is timestamped with the PR's creation date") + if summary.out_of_window_records > 0: + lines.append(f" {summary.out_of_window_records} records carry a date before the window but are still counted") + lines.append(" → do not filter activities.csv by the ts column") + + return lines + + +def _summary_how_attributed(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the HOW EACH ITEM WAS ATTRIBUTED section (--show-sources only). + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + percentages = _compute_attribution_percentages(summary) + named_sources = [ + (source, count) for source, count in summary.counts_by_attribution.items() if source + ] + named_sources.sort(key=lambda pair: pair[1], reverse=True) + unattributed_count = summary.counts_by_attribution.get("", 0) + + if format == "markdown": + lines.append("## How Each Item Was Attributed") + lines.append("") + lines.append("**First hit wins**") + lines.append("") + lines.append("| Source | Count | % | Description |") + lines.append("|--------|-------|---|-------------|") + for source, count in named_sources: + pct = percentages.get(source, 0) + desc = _attribution_description(source) + lines.append(f"| {source} | {count} | {pct}% | {desc} |") + if unattributed_count > 0: + pct = percentages.get("", 0) + lines.append( + f"| unattributed | {unattributed_count} | {pct}% | none of the above matched — see EXCLUDED |" + ) + lines.append("") + lines.append( + '*A PR that names a Jira key is resolved through the Jira chain, so it is reported as component / parent / project, not as a separate "jira key" rule.*' + ) + lines.append("") + else: # text + lines.append("") + lines.append("HOW EACH ITEM WAS ATTRIBUTED first hit wins") + for source, count in named_sources: + pct = percentages.get(source, 0) + desc = _attribution_description(source) + if source in ("component", "repo", "project"): + article = "the " + elif source == "shared": + article = "a " + else: + article = "" + lines.append(f" {source:12s} {count:4d} {pct:3d}% {article}{desc}") + if unattributed_count > 0: + pct = percentages.get("", 0) + lines.append( + f" unattributed {unattributed_count:4d} {pct:3d}% none of the above matched — see EXCLUDED" + ) + lines.append("") + lines.append(" A PR that names a Jira key is resolved through the Jira chain, so it is") + lines.append(' reported as component / parent / project, not as a separate "jira key" rule.') + + return lines + + +def _summary_what_excluded(summary: ExecutiveSummary, format: str = "text") -> List[str]: + """Render the WHAT WAS EXCLUDED section (--show-sources only). + + Args: + summary: The executive summary data. + format: Output format - "text" or "markdown". + + Returns: + List of formatted lines. + """ + lines = [] + unattributed_total = summary.counts_by_attribution.get("", 0) + + if format == "markdown": + lines.append("## What Was Excluded") + lines.append("") + + # Excluded members + if summary.data_quality.excluded_members: + count = len(summary.data_quality.excluded_members) + members_str = ", ".join(summary.data_quality.excluded_members) + lines.append(f"**{count} roster members** — {members_str}") + lines.append("") + lines.append("*Managers by role title, not IC contributors. `--include-all` puts them back.*") + lines.append("") + + # Excluded PRs + if summary.data_quality.excluded_personal_repo_prs > 0: + lines.append( + f"**{summary.data_quality.excluded_personal_repo_prs} PRs** — personal-namespace repos, dropped before counting" + ) + lines.append("") + if summary.data_quality.excluded_personal_repos: + top_repos, more_repos, more_prs = _top_repos( + summary.data_quality.excluded_personal_repos + ) + for repo, count in top_repos: + lines.append(f"- {repo}: {count}") + if more_repos: + lines.append(f"- and {more_repos} more repos ({more_prs} PRs)") + lines.append("") + lines.append( + "*The org allowlist is hand-kept — a real org missing from it lands in this list looking like a side project.*" + ) + lines.append("") + + # Unattributed items + if unattributed_total > 0: + lines.append(f"**{unattributed_total} items** — could not be attributed to any workstream") + lines.append("") + + # Unattributed PRs by repo + if summary.data_quality.unattributed_by_repo: + pr_count = sum(summary.data_quality.unattributed_by_repo.values()) + lines.append(f"**{pr_count} PRs by repo:**") + lines.append("") + top_repos, more_repos, more_prs = _top_repos(summary.data_quality.unattributed_by_repo) + for repo, count in top_repos: + note = _repo_note(repo) + if note: + lines.append(f"- {repo}: {count} {note}") + else: + lines.append(f"- {repo}: {count}") + if more_repos: + lines.append(f"- and {more_repos} more repos ({more_prs} PRs)") + lines.append("") + + # Jira tickets with no component and no parent + if summary.data_quality.tickets_no_component_no_parent > 0: + count = summary.data_quality.tickets_no_component_no_parent + lines.append( + f"**{count} Jira tickets** with no component and no parent epic → fix on the ticket" + ) + lines.append("") + + # Jira tickets whose parent epic is also empty + if summary.data_quality.tickets_parent_also_empty > 0: + count = summary.data_quality.tickets_parent_also_empty + lines.append( + f"**{count} Jira tickets** whose parent epic is also empty → fix on the epic" + ) + lines.append("") + else: # text + lines.append("") + lines.append("WHAT WAS EXCLUDED") + + # Excluded members + if summary.data_quality.excluded_members: + count = len(summary.data_quality.excluded_members) + members_str = ", ".join(summary.data_quality.excluded_members) + lines.append(f" {count} roster members {members_str}") + lines.append(" managers by role title, not IC contributors.") + lines.append(" --include-all puts them back.") + + # Excluded PRs + if summary.data_quality.excluded_personal_repo_prs > 0: + lines.append( + f" {summary.data_quality.excluded_personal_repo_prs} PRs personal-namespace repos, dropped before counting" + ) + if summary.data_quality.excluded_personal_repos: + top_repos, more_repos, more_prs = _top_repos( + summary.data_quality.excluded_personal_repos + ) + repo_lines = [] + for repo, count in top_repos: + repo_lines.append(f"{repo} {count}") + lines.append(f" {' · '.join(repo_lines)}") + if more_repos: + lines.append(f" and {more_repos} more repos ({more_prs} PRs)") + lines.append( + " The org allowlist is hand-kept — a real org missing from" + ) + lines.append(" it lands in this list looking like a side project.") + + # Unattributed items + if unattributed_total > 0: + lines.append( + f" {unattributed_total} items could not be attributed to any workstream" + ) + + # Unattributed PRs by repo + if summary.data_quality.unattributed_by_repo: + pr_count = sum(summary.data_quality.unattributed_by_repo.values()) + lines.append(f" {pr_count:3d} PRs, by repo:") + top_repos, more_repos, more_prs = _top_repos(summary.data_quality.unattributed_by_repo) + for repo, count in top_repos: + note = _repo_note(repo) + lines.append(f" {repo} {count}{note}") + if more_repos: + lines.append( + f" and {more_repos} more repos ({more_prs} PRs)" + ) + + # Jira tickets with no component and no parent + if summary.data_quality.tickets_no_component_no_parent > 0: + count = summary.data_quality.tickets_no_component_no_parent + lines.append( + f" {count:3d} Jira tickets with no component and no parent epic" + ) + lines.append(" → fix on the ticket") + + # Jira tickets whose parent epic is also empty + if summary.data_quality.tickets_parent_also_empty > 0: + count = summary.data_quality.tickets_parent_also_empty + lines.append( + f" {count:2d} Jira tickets whose parent epic is also empty" + ) + lines.append(" → fix on the epic") + + return lines + + +def _summary_text(summary: ExecutiveSummary, show_sources: bool = False) -> str: + """Render executive summary in plain text format.""" + lines = [] + + # Header + lines.extend(_summary_header(summary, format="text")) + lines.append("") + + # TEAM section + lines.extend(_summary_team(summary, format="text")) + + # WORKSTREAMS table + lines.extend(_summary_workstreams(summary, format="text")) + + # PEOPLE table + lines.extend(_summary_people(summary, format="text")) + + # DATA HEALTH section + lines.extend(_summary_data_health(summary, format="text")) + + # Source-disclosure sections (only if requested) + if show_sources: + lines.extend(_summary_what_counted(summary, format="text")) + lines.extend(_summary_how_attributed(summary, format="text")) + lines.extend(_summary_what_excluded(summary, format="text")) + + return "\n".join(lines) + "\n" + + +def _attribution_description(source: str) -> str: + """Return the human-readable description for an attribution source.""" + descriptions = { + "component": "issue's own components field", + "repo": "PR's repo maps to exactly one workstream", + "project": "Jira project key (USHIFT → MicroShift)", + "shared": "cross-cutting CI / tooling / docs repo", + "parent": "inherited from the parent epic", + } + return descriptions.get(source, source) + + + + + + + + + + + + + + + + +def _summary_markdown(summary: ExecutiveSummary, show_sources: bool = False) -> str: + """Render executive summary in markdown format with pipe tables.""" + lines = [] + + # Header + lines.extend(_summary_header(summary, format="markdown")) + + # TEAM section + lines.extend(_summary_team(summary, format="markdown")) + + # WORKSTREAMS table + lines.extend(_summary_workstreams(summary, format="markdown")) + + # PEOPLE table + lines.extend(_summary_people(summary, format="markdown")) + + # DATA HEALTH section + lines.extend(_summary_data_health(summary, format="markdown")) + + # Source-disclosure sections (only if requested) + if show_sources: + lines.extend(_summary_what_counted(summary, format="markdown")) + lines.extend(_summary_how_attributed(summary, format="markdown")) + lines.extend(_summary_what_excluded(summary, format="markdown")) + + return "\n".join(lines) + "\n" + + +def render_summary( + summary: ExecutiveSummary, output_format: str, show_sources: bool = False +) -> str: + """Render the executive summary in the requested format. + + Args: + summary: The executive summary to render. + output_format: One of "text" or "markdown". + show_sources: Whether to append the WHAT WAS COUNTED, HOW EACH ITEM WAS + ATTRIBUTED, and WHAT WAS EXCLUDED sections. + + Returns: + Rendered summary as a string. + """ + _require_known_summary_format(output_format) + if output_format == "markdown": + return _summary_markdown(summary, show_sources) + return _summary_text(summary, show_sources) diff --git a/plugins/edge-contribution/bin/report.py b/plugins/edge-contribution/bin/report.py new file mode 100644 index 00000000..6f222f6c --- /dev/null +++ b/plugins/edge-contribution/bin/report.py @@ -0,0 +1,653 @@ +#!/usr/bin/env python3 +"""Aggregate collected activity into the report structures the renderers consume. + +The collectors emit flat activity records (``{member, workstream, kind, ...}``) +wrapped in a ``{"activities": [...], "collector_meta": {...}}`` envelope. This +module turns those records into a binary contribution matrix, computes the team +metrics, and assembles the manager and IC report objects. It also tallies +unattributed items (``workstream is None``) so they can be reported separately +rather than silently dropped. + +``collector_meta`` carries the data-quality counters that no record can express +because the underlying item was dropped before any record existed (personal-repo +PRs). Everything else is derived from the records themselves. + +All functions are pure transforms over already-loaded data — no file or network +I/O — so the aggregation logic is fully unit-testable. The thin file-loading and +rendering wiring lives in ``main``. +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +from typing import Dict, List, Optional, Tuple +from hashlib import sha256 + +import _common +from _common import ACTIVITY_PROJECTS, UNATTRIBUTED_PARENT_EMPTY +from metrics import ContributionMatrix, compute_allocation_signals, compute_team_metrics +from render import ( + ACTIVITY_KINDS, + DataQuality, + ExecutiveSummary, + ICReport, + MemberLine, + ManagerReport, + WorkstreamActivity, + WorkstreamLine, + render_ic, + render_manager, + render_summary, +) +from workstream_map import _ALLOWED_ORGS, display_columns, workstream_acronyms + +# Managers on the Eng/QE roster by role title who are not IC contributors. +# Filtering them from the team median prevents distortion of allocation signals. +# Stored as the hash256 of the member's jira username +EXCLUDED_MEMBERS = ("82e5282cc07f498ff0daf8352e7403e2f16b017dd17c3e589276e406480bb5be" + , "423987e3ab15b31a79a5ebcdecc50f4394886bc47ff70c69385f70adfc4f03a7") + + +def classify_source(kind: str) -> str: + """Classify an activity kind as 'jira', 'github', or 'other'. + + Derives classification from membership in ``ACTIVITY_KINDS`` and the naming + convention: GitHub activity kinds start with ``pr_``, Jira kinds do not. + Unknown kinds (not in ``ACTIVITY_KINDS``) return ``"other"``. + + This avoids duplicating the kind literals into a new list in this module. + """ + if kind.startswith("pr_"): + return "github" + elif kind in ACTIVITY_KINDS: + return "jira" + else: + return "other" + + +def resolve_period_window(period_label: str) -> Optional[Tuple[str, str]]: + """Resolve a period label into a date window tuple. + + - ``YYYYQn`` quarter labels (e.g., ``2026Q3``) are resolved through + ``_common.quarter_to_window`` to get the actual calendar quarter dates. + - ``_`` explicit date ranges (e.g., ``2026-01-15_2026-03-31``) are + split on the underscore and returned as a tuple. + - Any other format returns ``None`` rather than raising, so the header can + omit the window line instead of guessing. + """ + # Try to parse as a quarter label + try: + window = _common.quarter_to_window(period_label) + return (window.start.isoformat(), window.end.isoformat()) + except ValueError: + pass + + # Try to parse as an explicit date range + if "_" in period_label: + parts = period_label.split("_", 1) + if len(parts) == 2: + return (parts[0], parts[1]) + + # Unrecognized format + return None + + +def count_out_of_window_records(activities: List[dict], window: Optional[Tuple[str, str]]) -> int: + """Count records whose timestamp falls outside the resolved window. + + Only meaningful when ``window`` is not None. When window is None, returns 0. + Compares dates, not strings-with-times — a timestamp like "2026-07-01T00:00:00Z" + is inside a window ending 2026-09-30. Records with a missing or unparseable + timestamp must not raise and must not be counted as out-of-window. + + Args: + activities: List of activity records, each potentially having a "ts" field. + window: A tuple of (start_date, end_date) in ISO format (e.g., "2026-07-01"). + + Returns: + Count of records whose timestamp is before window start or after window end. + """ + if window is None: + return 0 + + from datetime import datetime + + start_date, end_date = window + count = 0 + + for activity in activities: + ts = activity.get("ts") + if not ts: + continue + + try: + # Parse the ISO timestamp and extract the date part + dt = datetime.fromisoformat(ts.replace("Z", "+00:00")) + record_date = dt.date().isoformat() + + # Compare dates: record is out-of-window if before start or after end + if record_date < start_date or record_date > end_date: + count += 1 + except (ValueError, AttributeError): + # Malformed timestamp or not a string: don't count, don't raise + continue + + return count + + +def prepare_roster( + roster_entries: List[dict], +) -> Tuple[List[str], Dict[str, str], List[str]]: + """Filter excluded members and build name translation mappings. + + Factors out the roster preparation logic so both the manager and summary + views share one copy. Returns ``(excluded, jira_to_display, display_names)``. + """ + + excluded = [ + entry["jira_username"] + for entry in roster_entries + if sha256(entry["jira_username"].encode()).hexdigest() in EXCLUDED_MEMBERS + ] + roster_entries = [ + entry for entry in roster_entries + if sha256(entry["jira_username"].encode()).hexdigest() not in EXCLUDED_MEMBERS + ] + + # Build jira_username -> display_name mapping + jira_to_display = { + entry["jira_username"]: entry.get("name", entry["jira_username"]) + for entry in roster_entries + } + + # Extract display names for the matrix + display_names = [jira_to_display[entry["jira_username"]] for entry in roster_entries] + + return excluded, jira_to_display, display_names + + +def translate_activities_to_display_names( + activities: List[dict], jira_to_display: Dict[str, str] +) -> List[dict]: + """Translate activity member fields from jira_username to display name. + + This is used for the manager view to show real names instead of jira usernames. + Activities are shallow-copied so the originals are not mutated. + """ + result = [] + for activity in activities: + translated = dict(activity) + member = activity.get("member") + if member: + translated["member"] = jira_to_display.get(member, member) + result.append(translated) + return result + + +def build_contribution_matrix( + activities: List[dict], members: List[str], workstreams: List[str] +) -> ContributionMatrix: + """Build the binary matrix: a member touches a workstream with ≥1 activity. + + Also populates the counts matrix (member -> workstream -> count) for allocation + signals. The workstreams list should be display_columns() (7 columns including + SHARED) for the manager view, so allocation signals include cross-cutting work. + The workstreams-touched count internally filters out SHARED to stay "N of 6". + """ + workstream_set = set(workstreams) + member_set = set(members) + touched: Dict[str, set] = {member: set() for member in members} + counts: Dict[str, Dict[str, int]] = {member: {} for member in members} + for activity in activities: + member = activity.get("member") + workstream = activity.get("workstream") + if member in member_set and workstream in workstream_set: + touched[member].add(workstream) + if workstream not in counts[member]: + counts[member][workstream] = 0 + counts[member][workstream] += 1 + return ContributionMatrix( + members=list(members), workstreams=list(workstreams), touched=touched, counts=counts + ) + + +def count_unattributed(activities: List[dict], member: Optional[str] = None) -> int: + """Count activities that could not be mapped to a workstream. + + When ``member`` is given, count only that member's unattributed items. + """ + return sum( + 1 + for activity in activities + if activity.get("workstream") is None + and (member is None or activity.get("member") == member) + ) + + +def build_workstream_activities( + activities: List[dict], member: str, workstreams: List[str] +) -> List[WorkstreamActivity]: + """Summarize one member's activity per touched workstream, counted by kind.""" + workstream_set = set(workstreams) + counts_by_workstream: Dict[str, Dict[str, int]] = {} + for activity in activities: + if activity.get("member") != member: + continue + workstream = activity.get("workstream") + if workstream not in workstream_set: + continue + kind = activity.get("kind", "") + kind_counts = counts_by_workstream.setdefault(workstream, {}) + kind_counts[kind] = kind_counts.get(kind, 0) + 1 + return [ + WorkstreamActivity(workstream, counts_by_workstream[workstream]) + for workstream in workstreams + if workstream in counts_by_workstream + ] + + +def build_data_quality( + activities: List[dict], + excluded_members: List[str], + collector_meta: Optional[Dict[str, int]] = None, +) -> DataQuality: + """Tally data quality signals from activity records. + + Counts unattributed items (workstream is None) by category: + - GitHub PRs: tallied per repo + - Jira tickets: split by ``unattributed_reason`` into "no component and no + parent epic" versus "the parent epic is component-less too" + + A Jira record written before ``unattributed_reason`` existed has no reason + field; those fall into the no-component-no-parent bucket, which is where they + were already being counted. + + ``collector_meta`` carries counters the records cannot express because the + underlying item was dropped: ``excluded_personal_repo_prs`` (the total) and + ``excluded_personal_repos`` (the same PRs broken down by repo). + """ + meta = collector_meta or {} + unattributed_by_repo: Dict[str, int] = {} + tickets_no_component_no_parent = 0 + tickets_parent_also_empty = 0 + + for activity in activities: + if activity.get("workstream") is not None: + continue # Only count unattributed items + + repo = activity.get("repo") + if repo: + # GitHub PR with no workstream attribution + unattributed_by_repo[repo] = unattributed_by_repo.get(repo, 0) + 1 + elif activity.get("unattributed_reason") == UNATTRIBUTED_PARENT_EMPTY: + tickets_parent_also_empty += 1 + else: + # Jira ticket with no component and nothing above it to fall back to + tickets_no_component_no_parent += 1 + + return DataQuality( + unattributed_by_repo=unattributed_by_repo, + tickets_no_component_no_parent=tickets_no_component_no_parent, + tickets_parent_also_empty=tickets_parent_also_empty, + excluded_personal_repo_prs=meta.get("excluded_personal_repo_prs", 0), + excluded_personal_repos=dict(meta.get("excluded_personal_repos") or {}), + excluded_members=list(excluded_members), + ) + + +def build_workstream_lines( + matrix: ContributionMatrix, workstreams: List[str] +) -> List[WorkstreamLine]: + """Build workstream lines sorted by volume descending. + + Args: + matrix: The contribution matrix with counts. + workstreams: List of workstreams to include. + + Returns: + List of WorkstreamLine objects sorted by volume descending. + """ + lines = [] + for ws in workstreams: + volume = sum(matrix.counts.get(member, {}).get(ws, 0) for member in matrix.members) + contributors = sum( + 1 for member in matrix.members if ws in matrix.touched.get(member, set()) + ) + + # Find top contributor and their share + if volume > 0: + member_counts = [ + (member, matrix.counts.get(member, {}).get(ws, 0)) for member in matrix.members + ] + top_member, top_count = max(member_counts, key=lambda x: x[1]) + if top_count > 0: + top_share = round(100 * top_count / volume) + top_contributor = top_member + else: + top_share = 0 + top_contributor = None + else: + top_share = 0 + top_contributor = None + + lines.append( + WorkstreamLine( + workstream=ws, + volume=volume, + contributors=contributors, + top_contributor=top_contributor, + top_share=top_share, + ) + ) + + # Sort by volume descending + lines.sort(key=lambda line: line.volume, reverse=True) + return lines + + +def build_manager_report( + activities: List[dict], + members: List[str], + workstreams: List[str], + period_label: str, + excluded_members: Optional[List[str]] = None, + collector_meta: Optional[Dict[str, int]] = None, +) -> ManagerReport: + """Assemble the manager heatmap + scores report.""" + matrix = build_contribution_matrix(activities, members, workstreams) + signals = compute_allocation_signals(matrix) + data_quality = build_data_quality(activities, excluded_members or [], collector_meta) + return ManagerReport( + period_label=period_label, + metrics=compute_team_metrics(matrix), + matrix=matrix, + unattributed_count=count_unattributed(activities), + signals=signals, + data_quality=data_quality, + ) + + +def build_executive_summary( + activities: List[dict], + members: List[str], + workstreams: List[str], + period_label: str, + excluded_members: Optional[List[str]] = None, + collector_meta: Optional[Dict[str, int]] = None, +) -> ExecutiveSummary: + """Assemble the executive summary report. + + Reuses ``build_manager_report`` for the matrix, metrics, signals, and data + quality rather than recomputing them. Pulls static config from the actual + constants so the summary describes real behavior, not a stale copy. + """ + # Reuse the manager report to get matrix, metrics, signals, and data_quality + manager_report = build_manager_report( + activities, members, workstreams, period_label, excluded_members, collector_meta + ) + + # Count by source and kind + counts_by_source_kind: Dict[str, Dict[str, int]] = {"jira": {}, "github": {}} + for activity in activities: + kind = activity.get("kind", "") + source = classify_source(kind) + if source in ("jira", "github"): + if kind not in counts_by_source_kind[source]: + counts_by_source_kind[source][kind] = 0 + counts_by_source_kind[source][kind] += 1 + + # Count by attribution source, using empty string for None + counts_by_attribution: Dict[str, int] = {} + for activity in activities: + attr_source = activity.get("attribution_source") + key = "" if attr_source is None else attr_source + counts_by_attribution[key] = counts_by_attribution.get(key, 0) + 1 + + # Build per-member lines with top workstream and share + per_member = [] + for member in members: + total = sum(manager_report.matrix.counts.get(member, {}).values()) + touched = manager_report.metrics.workstreams_touched_by_member.get(member, 0) + + # Find top workstream and share + member_counts = manager_report.matrix.counts.get(member, {}) + if member_counts and total > 0: + top_ws, top_count = max(member_counts.items(), key=lambda x: x[1]) + top_share = round(100 * top_count / total) + top_workstream = top_ws + else: + top_workstream = None + top_share = 0 + + per_member.append( + MemberLine( + member=member, + total=total, + workstreams_touched=touched, + top_workstream=top_workstream, + top_share=top_share, + ) + ) + + # Sort by total descending + per_member.sort(key=lambda ml: ml.total, reverse=True) + + # Build per-workstream lines + per_workstream = build_workstream_lines(manager_report.matrix, workstreams) + + # Resolve the period window + window = resolve_period_window(period_label) + + # Count records whose timestamp falls outside the window + out_of_window_count = count_out_of_window_records(activities, window) + + return ExecutiveSummary( + period_label=period_label, + window=window, + metrics=manager_report.metrics, + signals=manager_report.signals, + data_quality=manager_report.data_quality, + counts_by_source_kind=counts_by_source_kind, + counts_by_attribution=counts_by_attribution, + per_member=per_member, + per_workstream=per_workstream, + total_records=len(activities), + jira_projects=ACTIVITY_PROJECTS, + ocpstrat_project="OCPSTRAT", + allowed_org_count=len(_ALLOWED_ORGS), + out_of_window_records=out_of_window_count, + ) + + +def build_ic_report( + activities: List[dict], member: str, workstreams: List[str], period_label: str +) -> ICReport: + """Assemble the single-member workstream-coverage report.""" + matrix = build_contribution_matrix(activities, [member], workstreams) + touched = len(matrix.touched.get(member, set())) + return ICReport( + period_label=period_label, + member=member, + workstreams_touched=touched, + total_workstreams=len(workstreams), + activity=build_workstream_activities(activities, member, workstreams), + unattributed_count=count_unattributed(activities, member), + ) + + +# --- CLI -------------------------------------------------------------------- + + +def should_use_color( + fmt: str, isatty: bool, env_no_color: Optional[str], no_color_flag: bool +) -> bool: + """Determine whether to use ANSI color codes. + + Color is enabled only when ALL of the following are true: + - format is "text" + - NO_COLOR environment variable is unset or empty + - stdout is a TTY + - --no-color flag is not set + + The --no-color flag always wins (forces color off). + """ + if no_color_flag: + return False + if fmt != "text": + return False + if env_no_color: + return False + if not isatty: + return False + return True + + +def parse_activity_file(payload) -> tuple[List[dict], Dict[str, int]]: + """Split one loaded activity file into its records and its collector counters. + + Collectors write ``{"activities": [...], "collector_meta": {...}}``. A bare + list is the pre-envelope format and still loads, just with no counters. + """ + if isinstance(payload, list): + return payload, {} + return payload.get("activities", []), payload.get("collector_meta", {}) or {} + + +def _merge_counter(accumulated, value): + """Combine one collector counter across two activity files. + + A counter is either a plain total or a ``{label: count}`` breakdown such as + ``excluded_personal_repos``. Totals add; breakdowns merge label by label. + Treating a breakdown like a total would raise, and skipping it would throw + away the repo names the data-quality block needs. + """ + if isinstance(value, dict): + merged = dict(accumulated or {}) + for label, count in value.items(): + merged[label] = merged.get(label, 0) + count + return merged + return (accumulated or 0) + value + + +def _load_activities(paths: List[str]) -> tuple[List[dict], Dict[str, object]]: + """Load every activity file, concatenating records and merging counters.""" + activities: List[dict] = [] + collector_meta: Dict[str, object] = {} + for path in paths: + with open(path, encoding="utf-8") as handle: + records, meta = parse_activity_file(json.load(handle)) + activities.extend(records) + for key, value in meta.items(): + collector_meta[key] = _merge_counter(collector_meta.get(key), value) + return activities, collector_meta + + +def main(argv: Optional[List[str]] = None) -> int: + parser = argparse.ArgumentParser(description="Render a contribution report.") + parser.add_argument("--view", required=True, choices=("manager", "ic", "summary")) + parser.add_argument( + "--activity", + action="append", + required=True, + help="an activity JSON file (repeat for jira + github)", + ) + parser.add_argument("--members-file", help="roster.json (required for the manager view)") + parser.add_argument("--member", help="Jira username (required for the ic view)") + parser.add_argument("--period", required=True, help="reporting period label, e.g. 2026Q2") + parser.add_argument("--format", default="text", choices=("text", "markdown")) + parser.add_argument("--include-all", action="store_true", help="include excluded members") + parser.add_argument("--ascii", action="store_true", help="use ASCII-only glyphs") + parser.add_argument("--no-color", action="store_true", help="disable color output") + parser.add_argument( + "--show-sources", + action="store_true", + help="add the sections describing what was counted, how each item was " + "attributed, and what was excluded (summary view only)", + ) + args = parser.parse_args(argv) + + # Guard: markdown format only works with summary view + if args.format == "markdown" and args.view in ("manager", "ic"): + parser.error("--format markdown is only supported with --view summary") + + activities, collector_meta = _load_activities(args.activity) + + use_color = should_use_color( + args.format, + sys.stdout.isatty(), + os.environ.get("NO_COLOR"), + args.no_color, + ) + + if args.view == "manager": + if not args.members_file: + parser.error("--members-file is required for --view manager") + with open(args.members_file, encoding="utf-8") as handle: + roster_entries = json.load(handle) + + excluded, jira_to_display, display_names = prepare_roster(roster_entries) + + # Translate activities to use display names + translated_activities = translate_activities_to_display_names(activities, jira_to_display) + + # Use display_columns() to include SHARED for allocation signals + workstreams = display_columns() + + rendered = render_manager( + build_manager_report( + translated_activities, + display_names, + workstreams, + args.period, + excluded, + collector_meta, + ), + args.format, + color=use_color, + ascii_only=args.ascii, + ) + elif args.view == "summary": + if not args.members_file: + parser.error("--members-file is required for --view summary") + with open(args.members_file, encoding="utf-8") as handle: + roster_entries = json.load(handle) + + excluded, jira_to_display, display_names = prepare_roster(roster_entries) + + # Translate activities to use display names + translated_activities = translate_activities_to_display_names(activities, jira_to_display) + + # Use display_columns() to include SHARED for allocation signals + workstreams = display_columns() + + rendered = render_summary( + build_executive_summary( + translated_activities, + display_names, + workstreams, + args.period, + excluded, + collector_meta, + ), + args.format, + show_sources=args.show_sources, + ) + else: + if not args.member: + parser.error("--member is required for --view ic") + # IC view uses the 6 canonical workstreams, not display_columns() + workstreams = workstream_acronyms() + rendered = render_ic( + build_ic_report(activities, args.member, workstreams, args.period), + args.format, + ) + + print(rendered) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/plugins/edge-contribution/bin/tests/__init__.py b/plugins/edge-contribution/bin/tests/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/plugins/edge-contribution/bin/tests/test_collect_github.py b/plugins/edge-contribution/bin/tests/test_collect_github.py new file mode 100644 index 00000000..6a1c63e2 --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_collect_github.py @@ -0,0 +1,803 @@ +"""Tests for collect_github.py — GitHub PR collection via batched GraphQL queries. + +The command runner and the Jira workstream resolver are both injected, so tests +are hermetic (no ``gh`` subprocess, no Jira network). Coverage spans: +- Batched GraphQL query construction with aliased searches +- Pagination following hasNextPage/endCursor +- GraphQL response validation (errors[], null/missing aliases, issueCount completeness) +- Chunk splitting on retriable failures (GraphQL errors, transport failures) +- Rate-limit detection and non-splitting behavior +- Happy-path attribution of authored/reviewed PRs +- Boundary/anti-cheat cases: key in body only, multiple keys deduped across + workstreams, unmapped/absent keys → unattributed (never dropped) +""" + +import json +import os +import sys +import unittest +from datetime import date + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import _common # noqa: E402 +import collect_github # noqa: E402 + +WINDOW = _common.Window(date(2026, 4, 1), date(2026, 6, 30)) + + +def _pr( + title, body="", repo="openshift/example", url="https://gh/pr/1", created="2026-05-01T00:00:00Z" +): + return { + "title": title, + "body": body, + "repository": {"nameWithOwner": repo}, + "url": url, + "createdAt": created, + } + + +def _graphql_response(alias, prs, has_next_page=False, end_cursor=None): + """Wrap PRs in a GraphQL batch response structure.""" + return { + "data": { + alias: { + "issueCount": len(prs), + "pageInfo": { + "hasNextPage": has_next_page, + "endCursor": end_cursor, + }, + "nodes": prs, + } + } + } + + +class FakeRunner: + def __init__(self, result): + self._result = result + self.commands = [] + + def __call__(self, command): + self.commands.append(command) + return self._result + + +def _ok(stdout): + return collect_github.CommandResult(returncode=0, stdout=stdout, stderr="") + + +def _resolver(mapping): + """Build a resolver that returns (workstream, source) tuples.""" + return lambda key: mapping.get(key, (None, None)) + + +class TestExtractJiraKeys(unittest.TestCase): + def test_finds_known_project_keys(self): + keys = collect_github.extract_jira_keys("Fixes OCPEDGE-123 and USHIFT-7") + assert keys == ["OCPEDGE-123", "USHIFT-7"] + + def test_dedupes_and_uppercases(self): + assert collect_github.extract_jira_keys("ocpbugs-1 OCPBUGS-1") == ["OCPBUGS-1"] + + def test_ignores_unknown_project_prefixes(self): + assert collect_github.extract_jira_keys("FOO-1 BAR-22") == [] + + def test_empty_text_yields_no_keys(self): + assert collect_github.extract_jira_keys("") == [] + + +class TestBatchQuery(unittest.TestCase): + def test_builds_one_alias_per_slot(self): + slots = [ + collect_github.SearchSlot("a0", "user1", "h1", "pr_authored"), + collect_github.SearchSlot("r0", "user1", "h1", "pr_reviewed"), + ] + query = collect_github.build_batch_query(slots, WINDOW) + assert "a0:" in query + assert "r0:" in query + + def test_authored_uses_author_and_created(self): + slot = collect_github.SearchSlot("a0", "user", "handle", "pr_authored") + query = collect_github.build_batch_query([slot], WINDOW) + assert "author:handle" in query + assert "created:2026-04-01..2026-06-30" in query + + def test_reviewed_uses_reviewed_by_and_updated(self): + slot = collect_github.SearchSlot("r0", "user", "handle", "pr_reviewed") + query = collect_github.build_batch_query([slot], WINDOW) + assert "reviewed-by:handle" in query + assert "updated:2026-04-01..2026-06-30" in query + + def test_cursored_slot_includes_after(self): + slot = collect_github.SearchSlot("a0", "user", "h", "pr_authored", cursor="abc123") + query = collect_github.build_batch_query([slot], WINDOW) + assert 'after: "abc123"' in query + + def test_non_cursored_slot_omits_after(self): + slot = collect_github.SearchSlot("a0", "user", "h", "pr_authored") + query = collect_github.build_batch_query([slot], WINDOW) + assert "after:" not in query + + +class TestAttributionEdgeCases(unittest.TestCase): + def test_key_in_body_only_is_found(self): + pr = _pr("no key here", "relates to USHIFT-9") + activities = collect_github.pr_to_activities( + pr, "u@redhat.com", "pr_authored", _resolver({"USHIFT-9": ("USHIFT", "component")}) + ) + assert [a.workstream for a in activities] == ["USHIFT"] + assert activities[0].attribution_source == "component" + + def test_multiple_keys_attributed_across_workstreams_deduped(self): + pr = _pr("OCPEDGE-1 OCPEDGE-1", "also OCPBUGS-5") + activities = collect_github.pr_to_activities( + pr, + "u@redhat.com", + "pr_authored", + _resolver({"OCPEDGE-1": ("SNO", "component"), "OCPBUGS-5": ("TNF", "component")}), + ) + assert {a.workstream for a in activities} == {"SNO", "TNF"} + + def test_key_resolving_to_none_is_unattributed_not_dropped(self): + pr = _pr("OCPEDGE-2 planning", "") + activities = collect_github.pr_to_activities( + pr, "u@redhat.com", "pr_authored", _resolver({"OCPEDGE-2": (None, None)}) + ) + assert len(activities) == 1 + assert activities[0].workstream is None + assert activities[0].source_key == "OCPEDGE-2" + assert activities[0].attribution_source is None + + def test_pr_with_no_key_is_unattributed(self): + pr = _pr("cleanup", "no ticket") + activities = collect_github.pr_to_activities( + pr, "u@redhat.com", "pr_authored", _resolver({}) + ) + assert len(activities) == 1 + assert activities[0].workstream is None + assert activities[0].source_key is None + assert activities[0].attribution_source is None + + +class TestFailureInputs(unittest.TestCase): + def test_gh_nonzero_exit_raises(self): + runner = FakeRunner(collect_github.CommandResult(1, "", "gh: not authenticated")) + with self.assertRaises(_common.CollectorError): + collect_github._run_gh_json(runner, ["gh", "api", "graphql"]) + + def test_gh_malformed_json_raises(self): + runner = FakeRunner(_ok("{not json")) + with self.assertRaises(_common.CollectorError): + collect_github._run_gh_json(runner, ["gh", "api", "graphql"]) + + def test_rate_limit_is_retried_with_exponential_backoff(self): + rate_limited = collect_github.CommandResult( + 1, "", "HTTP 403: You have exceeded a secondary rate limit" + ) + runner = _SequenceRunner([rate_limited, rate_limited, _ok("[]")]) + delays = [] + activities = collect_github._run_gh_json(runner, ["gh", "search", "prs"], delays.append) + assert activities == [] + assert runner.commands == [["gh", "search", "prs"]] * 3 + assert delays == [5.0, 10.0] + + def test_rate_limit_retries_are_bounded(self): + rate_limited = collect_github.CommandResult(1, "", "API rate limit exceeded") + runner = _SequenceRunner([rate_limited] * 6) + delays = [] + with self.assertRaises(_common.CollectorError): + collect_github._run_gh_json(runner, ["gh", "search", "prs"], delays.append) + assert len(runner.commands) == 5 + assert delays == [5.0, 10.0, 20.0, 40.0] + + +class _FakeSearchClient: + def __init__(self, issues_by_jql, errors=None): + self._issues = issues_by_jql + self._errors = errors or {} + self.search_calls = [] + + def search(self, jql, fields): + self.search_calls.append(jql) + if jql in self._errors: + raise self._errors[jql] + return self._issues.get(jql, []) + + +def _issue_with_components(key, components): + return {"key": key, "fields": {"components": [{"name": name} for name in components]}} + + +class TestJiraWorkstreamResolver(unittest.TestCase): + def test_resolves_component_to_workstream(self): + client = _FakeSearchClient( + {"key in (OCPEDGE-1)": [_issue_with_components("OCPEDGE-1", ["SNO"])]} + ) + mapping = collect_github.resolve_workstreams(client, ["OCPEDGE-1"]) + assert mapping["OCPEDGE-1"] == ("SNO", "component") + + def test_planning_component_resolves_to_none(self): + client = _FakeSearchClient( + {"key in (OCPEDGE-2)": [_issue_with_components("OCPEDGE-2", ["Planning"])]} + ) + mapping = collect_github.resolve_workstreams(client, ["OCPEDGE-2"]) + assert mapping["OCPEDGE-2"] == (None, None) + + def test_missing_issue_resolves_to_none(self): + client = _FakeSearchClient({"key in (OCPBUGS-9)": []}) + mapping = collect_github.resolve_workstreams(client, ["OCPBUGS-9"]) + assert mapping["OCPBUGS-9"] == (None, None) + + def test_auth_error_is_not_swallowed(self): + client = _FakeSearchClient({}, errors={"key in (OCPEDGE-1)": _common.JiraAuthError("401")}) + with self.assertRaises(_common.JiraAuthError): + collect_github.resolve_workstreams(client, ["OCPEDGE-1"]) + + def test_batches_large_key_lists(self): + keys = [f"KEY-{i}" for i in range(250)] + issues = {f"KEY-{i}": _issue_with_components(f"KEY-{i}", ["SNO"]) for i in range(250)} + client = _FakeSearchClient( + { + f"key in ({','.join(keys[:100])})": [issues[k] for k in keys[:100]], + f"key in ({','.join(keys[100:200])})": [issues[k] for k in keys[100:200]], + f"key in ({','.join(keys[200:])})": [issues[k] for k in keys[200:]], + } + ) + mapping = collect_github.resolve_workstreams(client, keys) + assert len(client.search_calls) >= 3 # Initial batch + possible parent batches + assert all(mapping[k] == ("SNO", "component") for k in keys) + + def test_prefetched_resolver_returns_from_mapping(self): + mapping = {"OCPEDGE-1": ("SNO", "component"), "OCPEDGE-2": (None, None)} + resolve = collect_github.make_prefetched_resolver(mapping) + assert resolve("OCPEDGE-1") == ("SNO", "component") + assert resolve("OCPEDGE-2") == (None, None) + assert resolve("UNKNOWN") == (None, None) + + +def _pr_json(title, body): + import json + + return json.dumps(_pr(title, body)) + + +class _SequenceRunner: + def __init__(self, results): + self._results = list(results) + self.commands = [] + + def __call__(self, command): + self.commands.append(command) + return self._results.pop(0) + + +class TestBatching(unittest.TestCase): + def test_roster_generates_authored_and_reviewed_slots(self): + roster = [{"jira_username": "u1", "github": "h1"}, {"jira_username": "u2", "github": "h2"}] + response = _graphql_response("m0_authored", []) + response["data"]["m0_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + response["data"]["m1_authored"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + response["data"]["m1_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = FakeRunner(_ok(json.dumps(response))) + pairs = collect_github.collect_prs(runner, roster, WINDOW) + # Should have made one request (4 slots fits in default chunk size) + assert len(runner.commands) == 1 + # All aliases present + query = runner.commands[0][-1] + assert "m0_authored:" in query + assert "m0_reviewed:" in query + assert "m1_authored:" in query + assert "m1_reviewed:" in query + + def test_has_next_page_produces_cursored_follow_up(self): + pr1 = _pr("First", url="https://gh/pr/1") + pr2 = _pr("Second", url="https://gh/pr/2") + # issueCount should be the total (2) across both pages + page1 = { + "data": { + "m0_authored": { + "issueCount": 2, # Total count + "pageInfo": {"hasNextPage": True, "endCursor": "cursor1"}, + "nodes": [pr1], + }, + "m0_reviewed": { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + }, + } + } + page2 = { + "data": { + "m0_authored": { + "issueCount": 2, # Same total count + "pageInfo": {"hasNextPage": False}, + "nodes": [pr2], + } + } + } + runner = _SequenceRunner([_ok(json.dumps(page1)), _ok(json.dumps(page2))]) + roster = [{"jira_username": "user", "github": "handle"}] + pairs = collect_github.collect_prs(runner, roster, WINDOW) + # Two requests: initial + pagination + assert len(runner.commands) == 2 + # Second request has cursor + assert 'after: "cursor1"' in runner.commands[1][-1] + # Both PRs returned + assert len(pairs) == 2 + urls = {pair[1]["url"] for pair in pairs} + assert urls == {"https://gh/pr/1", "https://gh/pr/2"} + + def test_failing_chunk_splits_and_retries(self): + # Simulate a large chunk that 502s, then succeeds when split + roster = [{"jira_username": f"u{i}", "github": f"h{i}"} for i in range(6)] + fail_result = collect_github.CommandResult(22, "", "HTTP 502") + success = _graphql_response("m0_authored", []) + for i in range(6): + success["data"][f"m{i}_authored"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + success["data"][f"m{i}_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = _SequenceRunner([fail_result, _ok(json.dumps(success)), _ok(json.dumps(success))]) + collect_github.collect_prs(runner, roster, WINDOW, chunk_size=12) + # First request (all 12 slots) fails, then splits into 2 chunks of 6 + assert len(runner.commands) == 3 + + def test_single_slot_failure_re_raises(self): + roster = [{"jira_username": "user", "github": "handle"}] + runner = FakeRunner(collect_github.CommandResult(1, "", "persistent error")) + with self.assertRaises(_common.CollectorError): + collect_github.collect_prs(runner, roster, WINDOW) + + +class TestGraphqlRateLimit(unittest.TestCase): + def test_rate_limited_error_in_body_is_retried(self): + rate_limited_body = json.dumps({"errors": [{"type": "RATE_LIMITED", "message": "..."}]}) + rate_limited_result = _ok(rate_limited_body) + success = _ok(json.dumps(_graphql_response("m0_authored", []))) + runner = _SequenceRunner([rate_limited_result, success]) + delays = [] + adapted = collect_github.graphql_rate_limit_adapter(runner) + collect_github._run_gh_json( + adapted, ["gh", "api", "graphql", "-f", "query=..."], delays.append + ) + # Should have retried + assert len(runner.commands) == 2 + assert len(delays) == 1 + + def test_pr_body_containing_rate_limit_text_is_not_treated_as_error(self): + pr = _pr("Normal PR", "This PR fixes a rate limit bug") + response = _graphql_response("m0_authored", [pr]) + runner = FakeRunner(_ok(json.dumps(response))) + adapted = collect_github.graphql_rate_limit_adapter(runner) + result = adapted(["gh", "api", "graphql", "-f", "query=..."]) + # Should not be rewritten as an error + assert result.returncode == 0 + + def test_rate_limit_exhaustion_does_not_split(self): + # GraphQL rate limit should be retried but not split + rate_limited_body = json.dumps( + {"errors": [{"type": "RATE_LIMITED", "message": "exhausted"}]} + ) + rate_limited_result = _ok(rate_limited_body) + runner = FakeRunner(rate_limited_result) + adapted = collect_github.graphql_rate_limit_adapter(runner) + roster = [{"jira_username": "u1", "github": "h1"}, {"jira_username": "u2", "github": "h2"}] + delays = [] + with self.assertRaises(collect_github.GraphqlRateLimitExhausted): + collect_github.collect_prs(adapted, roster, WINDOW, chunk_size=4, sleep=delays.append) + # Retries should happen (max 4), but no splitting + # 1 initial + 4 retries = 5 total, all with the same 4 slots + assert len(runner.commands) == 5 + assert delays == [5.0, 10.0, 20.0, 40.0] + # All commands should be identical (no splitting into smaller chunks) + first_query = runner.commands[0][-1] + assert all(cmd[-1] == first_query for cmd in runner.commands) + + +class TestGraphqlResponseValidation(unittest.TestCase): + def test_errors_array_raises_graphql_response_error(self): + payload = { + "errors": [{"message": "Something went wrong", "path": ["search", "m0_authored"]}], + "data": None, + } + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + with self.assertRaises(collect_github.GraphqlResponseError) as ctx: + collect_github.parse_batch_response(payload, [slot]) + # Error message should include GitHub's message + assert "Something went wrong" in str(ctx.exception) + + def test_null_alias_raises_graphql_response_error_not_attribute_error(self): + # Bug 2: null alias used to crash with AttributeError + payload = {"data": {"m0_authored": None}} + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + with self.assertRaises(collect_github.GraphqlResponseError) as ctx: + collect_github.parse_batch_response(payload, [slot]) + assert "null" in str(ctx.exception).lower() + assert "handle" in str(ctx.exception) + + def test_missing_alias_raises_graphql_response_error(self): + payload = {"data": {}} + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + with self.assertRaises(collect_github.GraphqlResponseError) as ctx: + collect_github.parse_batch_response(payload, [slot]) + assert "missing" in str(ctx.exception).lower() + assert "m0_authored" in str(ctx.exception) + + def test_missing_data_field_raises(self): + payload = {"errors": []} + with self.assertRaises(collect_github.GraphqlResponseError) as ctx: + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + collect_github.parse_batch_response(payload, [slot]) + assert "missing 'data'" in str(ctx.exception) + + def test_incomplete_issue_count_raises(self): + # When pagination completes but node count doesn't match issueCount + payload = { + "data": { + "m0_authored": { + "issueCount": 10, + "pageInfo": {"hasNextPage": False}, + "nodes": [_pr("PR1"), _pr("PR2")], # Only 2 nodes, but issueCount=10 + } + } + } + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + # No prior pages, so total is just 2 + with self.assertRaises(collect_github.GraphqlResponseError) as ctx: + collect_github.parse_batch_response(payload, [slot], slot_node_counts={}) + assert "Incomplete" in str(ctx.exception) + assert "collected 2" in str(ctx.exception) + assert "issueCount reports 10" in str(ctx.exception) + + def test_issue_count_above_1000_skips_check(self): + # GitHub caps results at 1000, so we skip completeness check for > 1000 + payload = { + "data": { + "m0_authored": { + "issueCount": 1500, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + } + } + slot = collect_github.SearchSlot("m0_authored", "user", "handle", "pr_authored") + # Should not raise despite mismatch + pairs, next_slots = collect_github.parse_batch_response( + payload, [slot], slot_node_counts={} + ) + assert pairs == [] + assert next_slots == [] + + def test_multi_slot_chunk_splits_on_graphql_error(self): + # A multi-slot chunk with an error should split and succeed + fail_payload = { + "errors": [{"message": "timeout", "path": []}], + "data": None, + } + success_1 = _graphql_response("m0_authored", []) + success_1["data"]["m0_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + success_2 = _graphql_response("m1_authored", []) + success_2["data"]["m1_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = _SequenceRunner( + [ + _ok(json.dumps(fail_payload)), + _ok(json.dumps(success_1)), + _ok(json.dumps(success_2)), + ] + ) + roster = [ + {"jira_username": "u0", "github": "h0"}, + {"jira_username": "u1", "github": "h1"}, + ] + # Should split after failure and succeed + pairs = collect_github.collect_prs(runner, roster, WINDOW, chunk_size=4) + assert len(runner.commands) == 3 # Initial fail + 2 splits + + +class TestAuthoredCollectionHappyPath(unittest.TestCase): + def test_authored_pr_flows_through_collect_prs_to_activity(self): + # End-to-end: roster → collect_prs → activities with correct member/kind/workstream + pr = _pr("OCPEDGE-123: Add feature", "description", repo="openshift/example") + response = _graphql_response("m0_authored", [pr]) + response["data"]["m0_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = FakeRunner(_ok(json.dumps(response))) + roster = [{"jira_username": "alice@redhat.com", "github": "alice-gh"}] + + pairs = collect_github.collect_prs(runner, roster, WINDOW) + assert len(pairs) == 1 + slot, pr_node = pairs[0] + assert slot.kind == "pr_authored" + assert slot.member == "alice@redhat.com" # jira_username, not github handle + assert slot.handle == "alice-gh" + assert pr_node["title"] == "OCPEDGE-123: Add feature" + + # Convert to activities + resolve = _resolver({"OCPEDGE-123": ("SNO", "component")}) + activities = collect_github.pr_to_activities(pr_node, slot.member, slot.kind, resolve) + assert len(activities) == 1 + activity = activities[0] + assert activity.kind == "pr_authored" + assert activity.member == "alice@redhat.com" + assert activity.workstream == "SNO" + assert activity.repo == "openshift/example" + assert activity.source_key == "OCPEDGE-123" + assert activity.attribution_source == "component" + + +class TestReviewedCollection(unittest.TestCase): + def test_reviewed_pr_flows_through_collect_prs_to_activity(self): + pr = _pr("USHIFT-456: Fix bug", "body text", repo="openshift-eng/tooling") + response = _graphql_response("m0_reviewed", [pr]) + response["data"]["m0_authored"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = FakeRunner(_ok(json.dumps(response))) + roster = [{"jira_username": "bob@redhat.com", "github": "bob-gh"}] + + pairs = collect_github.collect_prs(runner, roster, WINDOW) + assert len(pairs) == 1 + slot, pr_node = pairs[0] + assert slot.kind == "pr_reviewed" + assert slot.member == "bob@redhat.com" + assert pr_node["title"] == "USHIFT-456: Fix bug" + + resolve = _resolver({"USHIFT-456": ("USHIFT", "component")}) + activities = collect_github.pr_to_activities(pr_node, slot.member, slot.kind, resolve) + assert len(activities) == 1 + activity = activities[0] + assert activity.kind == "pr_reviewed" + assert activity.member == "bob@redhat.com" + assert activity.workstream == "USHIFT" + assert activity.attribution_source == "component" + + +class TestUnattributedPathEndToEnd(unittest.TestCase): + def test_unmappable_key_flows_through_as_unattributed(self): + # Verify that a PR with an unmappable key carries through collect_prs + # and becomes an unattributed activity (not dropped) + # Use OCPEDGE prefix (recognized) but map it to None (unmapped) + pr = _pr("OCPEDGE-999: roadmap", "planning work") + response = _graphql_response("m0_authored", [pr]) + response["data"]["m0_reviewed"] = { + "issueCount": 0, + "pageInfo": {"hasNextPage": False}, + "nodes": [], + } + runner = FakeRunner(_ok(json.dumps(response))) + roster = [{"jira_username": "charlie@redhat.com", "github": "charlie-gh"}] + + pairs = collect_github.collect_prs(runner, roster, WINDOW) + slot, pr_node = pairs[0] + + # Resolver maps OCPEDGE-999 to None (unmapped workstream) + resolve = _resolver({"OCPEDGE-999": (None, None)}) + activities = collect_github.pr_to_activities(pr_node, slot.member, slot.kind, resolve) + assert len(activities) == 1 + assert activities[0].workstream is None + assert activities[0].source_key == "OCPEDGE-999" + assert activities[0].member == "charlie@redhat.com" + assert activities[0].attribution_source is None + + +class TestWave2Attribution(unittest.TestCase): + """Wave 2 attribution features: repo fallback, parent hop, SHARED, exclusions.""" + + def test_jira_attribution_wins_over_repo(self): + # PR in openshift/lvm-operator citing a TNF-component ticket attributes to TNF + pr = _pr("OCPEDGE-100: Add test", "TNF work", repo="openshift/lvm-operator") + resolve = _resolver({"OCPEDGE-100": ("TNF", "component")}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 1 + assert activities[0].workstream == "TNF" + assert activities[0].attribution_source == "component" + + def test_repo_fallback_when_no_jira_keys(self): + # Keyless PR in openshift/lvm-operator → LVMS via repo fallback + pr = _pr("cleanup", "no ticket", repo="openshift/lvm-operator") + resolve = _resolver({}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 1 + assert activities[0].workstream == "LVMS" + assert activities[0].attribution_source == "repo" + assert activities[0].source_key is None + + def test_keyless_pr_in_microshift_repo_attributes_to_ushift(self): + # Keyless PR in openshift/microshift → USHIFT via repo fallback + pr = _pr("fix bug", "no key", repo="openshift/microshift") + resolve = _resolver({}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 1 + assert activities[0].workstream == "USHIFT" + assert activities[0].attribution_source == "repo" + + def test_ushift_project_key_with_no_component_uses_project_fallback(self): + # USHIFT-nnn key with no component → USHIFT via project fallback + # This happens in resolve_workstreams, so we need to test that + client = _FakeSearchClient( + { + "key in (USHIFT-6337)": [ + {"key": "USHIFT-6337", "fields": {"components": [], "parent": None}} + ] + } + ) + mapping = collect_github.resolve_workstreams(client, ["USHIFT-6337"]) + assert mapping["USHIFT-6337"] == ("USHIFT", "project") + + def test_parent_hop_when_issue_has_no_component(self): + # PR citing a component-less ticket whose parent has "Two Node Fencing" → TNF + client = _FakeSearchClient( + { + "key in (OCPEDGE-200)": [ + { + "key": "OCPEDGE-200", + "fields": { + "components": [], + "parent": {"key": "OCPEDGE-100"}, + }, + } + ], + "key in (OCPEDGE-100)": [ + _issue_with_components("OCPEDGE-100", ["Two Node Fencing"]) + ], + } + ) + mapping = collect_github.resolve_workstreams(client, ["OCPEDGE-200"]) + assert mapping["OCPEDGE-200"] == ("TNF", "parent") + + def test_shared_repo_attributes_to_shared_column(self): + # Keyless PR in openshift/release → SHARED + pr = _pr("Update CI config", "no key", repo="openshift/release") + resolve = _resolver({}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 1 + assert activities[0].workstream == "SHARED" + assert activities[0].attribution_source == "shared" + + def test_excluded_repo_returns_empty(self): + # PR in personal namespace → dropped + pr = _pr("Personal project", "no key", repo="jeff-roche/roundhouse") + resolve = _resolver({}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert activities == [] + + def test_jira_key_does_not_rescue_a_personal_repo_pr(self): + # Product rule: work only counts when it lands in a Red Hat product or + # engineering org. A Jira key in the title does not promote a PR opened + # against someone's own repo. Pins the order of the attribution chain — + # exclusion must run before the Jira-key lookup, not after. + pr = _pr( + "OCPBUGS-104449: Add ODF documentation for TNF two-node clusters", + "OCPBUGS-104449", + repo="pablofontanilla/tnf-preset", + ) + resolve = _resolver({"OCPBUGS-104449": ("TNF", "component")}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert activities == [] + + def test_jira_key_still_attributes_inside_an_org(self): + # The mirror of the rule above: the same PR in an org repo does count, + # so the exclusion is not silently swallowing everything. + pr = _pr( + "OCPBUGS-104449: Add ODF documentation for TNF two-node clusters", + "OCPBUGS-104449", + repo="openshift/oc-tnf", + ) + resolve = _resolver({"OCPBUGS-104449": ("TNF", "component")}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert [activity.workstream for activity in activities] == ["TNF"] + + def test_two_node_toolbox_keyless_pr_is_unattributed(self): + # openshift-eng/two-node-toolbox is intentionally NOT in the repo map + # Keyless PR → workstream None (regression pin) + pr = _pr("Update docs", "no key", repo="openshift-eng/two-node-toolbox") + resolve = _resolver({}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 1 + assert activities[0].workstream is None + assert activities[0].attribution_source is None + + def test_multiple_keys_to_different_workstreams_emits_multiple_activities(self): + # Already covered, but explicitly verify it still works + pr = _pr("OCPEDGE-1 and OCPBUGS-2", "multi-workstream") + resolve = _resolver({"OCPEDGE-1": ("SNO", "component"), "OCPBUGS-2": ("TNF", "component")}) + activities = collect_github.pr_to_activities(pr, "u@redhat.com", "pr_authored", resolve) + assert len(activities) == 2 + assert {a.workstream for a in activities} == {"SNO", "TNF"} + + def test_json_output_contains_attribution_source(self): + # Verify the dataclass serialization includes attribution_source + import json + from dataclasses import asdict + + activity = collect_github.GithubActivity( + member="u@redhat.com", + workstream="SNO", + kind="pr_authored", + repo="openshift/example", + pr_url="https://gh/pr/1", + ts="2026-05-01T00:00:00Z", + source_key="OCPEDGE-1", + attribution_source="component", + ) + serialized = json.loads(json.dumps(asdict(activity))) + assert serialized["attribution_source"] == "component" + + +class TestResolveWorkstreamsEmptyInput(unittest.TestCase): + def test_empty_input_makes_zero_queries(self): + client = _FakeSearchClient({}) + mapping = collect_github.resolve_workstreams(client, []) + assert mapping == {} + assert len(client.search_calls) == 0 + + +class TestTallyExcludedRepos(unittest.TestCase): + """A bare total hides which org went missing; the tally names names.""" + + def test_counts_dropped_prs_per_repo(self): + prs = [ + _pr("a", repo="jeff-roche/roundhouse"), + _pr("b", repo="jeff-roche/roundhouse"), + _pr("c", repo="jaypoulz/edge-tooling"), + ] + assert collect_github.tally_excluded_repos(prs) == { + "jeff-roche/roundhouse": 2, + "jaypoulz/edge-tooling": 1, + } + + def test_org_repos_are_not_tallied(self): + prs = [_pr("a", repo="openshift/lvm-operator"), _pr("b", repo="kubevirt/hco")] + assert collect_github.tally_excluded_repos(prs) == {} + + def test_no_prs_yields_empty_tally(self): + assert collect_github.tally_excluded_repos([]) == {} + + def test_total_matches_the_sum_of_the_tally(self): + # The existing excluded_personal_repo_prs counter must stay consistent + # with the per-repo breakdown, or the two numbers will disagree in the + # data-quality block. + prs = [ + _pr("a", repo="jeff-roche/cankan"), + _pr("b", repo="sshnaidm/spenda"), + _pr("c", repo="openshift/origin"), + ] + assert sum(collect_github.tally_excluded_repos(prs).values()) == 2 + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_common.py b/plugins/edge-contribution/bin/tests/test_common.py new file mode 100644 index 00000000..caa6bffd --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_common.py @@ -0,0 +1,226 @@ +"""Tests for _common.py — Jira/GitHub config, clients, and date helpers. + +Covers happy-path config and pagination, failure inputs (missing env, 401, +persistent 5xx, malformed JSON), rate-limit backoff, and boundary cases (empty +result set, retry that eventually succeeds, quarter/date parsing). +""" + +import json +import os +import sys +import unittest +from datetime import date + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import _common # noqa: E402 + + +class _FakeTransport: + """Records calls and replays a queue of canned responses.""" + + def __init__(self, responses): + self._responses = list(responses) + self.calls = [] + + def __call__(self, method, path, body): + self.calls.append((method, path, body)) + response = self._responses.pop(0) + if callable(response): + return response(method, path, body) + return response + + +def _client(responses, max_retries=2, sleep=lambda _delay: None): + config = _common.JiraConfig(base_url="https://example.test", username="u", api_token="t") + return _common.JiraClient( + config, _FakeTransport(responses), max_retries=max_retries, sleep=sleep + ) + +class TestJiraClientHappyPath(unittest.TestCase): + def test_search_walks_two_pages_then_stops(self): + page_one = _common.HttpResponse(200, '{"issues": [{"key": "A-1"}], "nextPageToken": "t2"}') + page_two = _common.HttpResponse(200, '{"issues": [{"key": "A-2"}], "isLast": true}') + client = _client([page_one, page_two]) + issues = client.search("project = A", ["key"]) + assert [issue["key"] for issue in issues] == ["A-1", "A-2"] + + def test_empty_result_returns_empty_list(self): + client = _client([_common.HttpResponse(200, '{"issues": [], "isLast": true}')]) + assert client.search("project = A", ["key"]) == [] + + def test_get_issue_returns_parsed_body(self): + client = _client([_common.HttpResponse(200, '{"key": "A-1", "fields": {}}')]) + assert client.get_issue("A-1", ["components"])["key"] == "A-1" + + +class TestJiraClientFailureInputs(unittest.TestCase): + def test_401_raises_auth_error(self): + client = _client([_common.HttpResponse(401, "nope")]) + with self.assertRaises(_common.JiraAuthError): + client.search("project = A", ["key"]) + + def test_persistent_5xx_raises_after_retries(self): + responses = [_common.HttpResponse(503, "busy") for _ in range(5)] + client = _client(responses, max_retries=2) + with self.assertRaises(_common.CollectorError): + client.search("project = A", ["key"]) + + def test_malformed_json_raises_collector_error(self): + client = _client([_common.HttpResponse(200, "{not json")]) + with self.assertRaises(_common.CollectorError): + client.search("project = A", ["key"]) + + def test_rate_limit_retries_with_backoff(self): + delays = [] + responses = [ + _common.HttpResponse(429, "busy"), + _common.HttpResponse(200, '{"issues": [], "isLast": true}'), + ] + client = _client(responses, sleep=delays.append) + assert client.search("project = A", ["key"]) == [] + assert delays == [5.0] + + def test_rate_limit_honors_retry_after_header(self): + delays = [] + responses = [ + _common.HttpResponse(429, "busy", {"Retry-After": "17"}), + _common.HttpResponse(200, '{"issues": [], "isLast": true}'), + ] + client = _client(responses, sleep=delays.append) + assert client.search("project = A", ["key"]) == [] + assert delays == [17.0] + + +class TestJiraClientEdgeCases(unittest.TestCase): + def test_retry_then_success(self): + responses = [ + _common.HttpResponse(503, "busy"), + _common.HttpResponse(200, '{"issues": [{"key": "A-1"}], "isLast": true}'), + ] + client = _client(responses, max_retries=2) + assert [issue["key"] for issue in client.search("project = A", ["key"])] == ["A-1"] + + +class TestDateHelpers(unittest.TestCase): + def test_parse_date(self): + assert _common.parse_date("2026-04-01") == date(2026, 4, 1) + + def test_quarter_to_window_q2(self): + window = _common.quarter_to_window("2026Q2") + assert window.start == date(2026, 4, 1) + assert window.end == date(2026, 6, 30) + + def test_quarter_to_window_q4(self): + window = _common.quarter_to_window("2026Q4") + assert window.start == date(2026, 10, 1) + assert window.end == date(2026, 12, 31) + + def test_invalid_quarter_raises(self): + with self.assertRaises(ValueError): + _common.quarter_to_window("2026Q9") + + def test_window_contains_jira_timestamp(self): + window = _common.Window(date(2026, 4, 1), date(2026, 6, 30)) + assert window.contains_timestamp("2026-05-01T12:00:00.000+0000") is True + assert window.contains_timestamp("2026-03-31T12:00:00.000+0000") is False + + +class TestCommandRateLimitRetry(unittest.TestCase): + def test_retries_only_rate_limit_failures(self): + results = [ + _common.CommandResult(1, "", "HTTP 403: secondary rate limit"), + _common.CommandResult(0, "[]", ""), + ] + calls = [] + delays = [] + + def runner(command): + calls.append(command) + return results.pop(0) + + result = _common.run_with_rate_limit_retry(runner, ["gh", "api"], delays.append) + assert result.returncode == 0 + assert len(calls) == 2 + assert delays == [5.0] + + def test_honors_retry_after_and_caps_delay(self): + result = _common.CommandResult(1, "", "rate limit; Retry-After: 120") + delays = [] + + def runner(command): + return result + + final = _common.run_with_rate_limit_retry( + runner, ["gh", "api"], delays.append, max_retries=1 + ) + assert final is result + assert delays == [60.0] + + +class TestResolveWindow(unittest.TestCase): + def test_quarter_takes_precedence(self): + window = _common.resolve_window(quarter="2026Q2") + assert window.start == date(2026, 4, 1) + + def test_explicit_date_pair(self): + window = _common.resolve_window(from_date="2026-01-15", to_date="2026-02-20") + assert window.start == date(2026, 1, 15) + assert window.end == date(2026, 2, 20) + + def test_missing_everything_raises(self): + with self.assertRaises(ValueError): + _common.resolve_window() + + def test_partial_date_range_raises(self): + with self.assertRaises(ValueError): + _common.resolve_window(from_date="2026-01-15") + + def test_past_quarter_returned_as_is(self): + # Inject today as 2026-09-28 (current date per plan) + # 2026Q1 is fully in the past + today = date(2026, 9, 28) + window = _common.resolve_window(quarter="2026Q1", today=today) + assert window.start == date(2026, 1, 1) + assert window.end == date(2026, 3, 31) + + def test_in_progress_quarter_capped_at_today(self): + # Inject today as 2026-09-28 + # 2026Q3 is in progress (July 1 - Sept 30) + today = date(2026, 9, 28) + window = _common.resolve_window(quarter="2026Q3", today=today) + assert window.start == date(2026, 7, 1) + assert window.end == date(2026, 9, 28) # Capped at today + + def test_future_quarter_raises(self): + # Inject today as 2026-09-28 + # 2026Q4 starts Oct 1, which is in the future + today = date(2026, 9, 28) + with self.assertRaises(ValueError) as ctx: + _common.resolve_window(quarter="2026Q4", today=today) + assert "has not started yet" in str(ctx.exception) + + +class TestActivityPayload(unittest.TestCase): + def test_records_are_wrapped_with_empty_meta_by_default(self): + assert _common.activity_payload([{"member": "a"}]) == { + "activities": [{"member": "a"}], + "collector_meta": {}, + } + + def test_counters_land_in_collector_meta(self): + payload = _common.activity_payload([], excluded_personal_repo_prs=101) + assert payload["collector_meta"] == {"excluded_personal_repo_prs": 101} + + def test_a_zero_counter_is_still_emitted(self): + # Reporting "0 excluded" is a real finding; omitting the key would make it + # indistinguishable from a collector that never counted at all. + payload = _common.activity_payload([], excluded_personal_repo_prs=0) + assert payload["collector_meta"] == {"excluded_personal_repo_prs": 0} + + def test_payload_is_json_serializable(self): + json.dumps(_common.activity_payload([{"member": "a"}], excluded_personal_repo_prs=1)) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_export_csv.py b/plugins/edge-contribution/bin/tests/test_export_csv.py new file mode 100644 index 00000000..190dc35b --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_export_csv.py @@ -0,0 +1,501 @@ +"""Tests for export_csv.py — CSV export builders. + +Covers header correctness, field normalization between Jira and GitHub records, +source derivation from kind, flag columns in matrix.csv, empty-data handling, +and the EXPORT_FILES contract pinning what the skill document relies on. +""" + +import os +import sys +import unittest +from io import StringIO + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import export_csv # noqa: E402 +from metrics import AllocationSignals, ContributionMatrix, TeamMetrics # noqa: E402 +from render import DataQuality, ManagerReport # noqa: E402 + + +class TestBuildActivitiesCSV(unittest.TestCase): + def test_emits_documented_header(self): + csv_output = export_csv.build_activities_csv([]) + lines = csv_output.strip().split("\n") + assert len(lines) == 1 + assert ( + lines[0] + == "member,source,kind,workstream,attribution_source,repo,issue_key,url,ts,unattributed_reason" + ) + + def test_jira_record_fills_issue_key_and_url_leaves_repo_empty(self): + activities = [ + { + "member": "alice", + "kind": "assignee", + "workstream": "SNO", + "attribution_source": "component", + "issue_key": "OCPEDGE-1234", + "url": "https://issues.redhat.com/browse/OCPEDGE-1234", + "ts": "2026-07-01T10:00:00Z", + } + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + assert len(lines) == 2 + # Repo field should be empty for Jira records + assert "alice,jira,assignee,SNO,component,,OCPEDGE-1234," in lines[1] + + def test_github_record_fills_repo_and_uses_pr_url_for_url(self): + activities = [ + { + "member": "bob", + "kind": "pr_authored", + "workstream": "LVMS", + "attribution_source": "repo", + "repo": "openshift/lvm-operator", + "pr_url": "https://github.com/openshift/lvm-operator/pull/123", + "source_key": "OCPEDGE-5678", + "ts": "2026-07-02T11:00:00Z", + } + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + assert len(lines) == 2 + # Should use pr_url for url field, and include repo + assert "bob,github,pr_authored,LVMS,repo,openshift/lvm-operator,OCPEDGE-5678," in lines[1] + assert "https://github.com/openshift/lvm-operator/pull/123" in lines[1] + + def test_github_record_without_source_key_leaves_issue_key_empty(self): + activities = [ + { + "member": "charlie", + "kind": "pr_reviewed", + "workstream": "USHIFT", + "attribution_source": "repo", + "repo": "openshift/microshift", + "pr_url": "https://github.com/openshift/microshift/pull/456", + "ts": "2026-07-03T12:00:00Z", + } + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + assert len(lines) == 2 + # issue_key field should be empty when source_key is absent + # Check that there are no literal "None" strings + assert "None" not in lines[1] + # The issue_key field (7th column) should be empty + assert "openshift/microshift,," in lines[1] + + def test_source_column_derived_from_kind(self): + activities = [ + {"member": "alice", "kind": "assignee", "workstream": "SNO"}, + {"member": "bob", "kind": "pr_authored", "workstream": "LVMS"}, + {"member": "charlie", "kind": "qa", "workstream": "TNA"}, + {"member": "dave", "kind": "pr_reviewed", "workstream": "TNF"}, + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + assert len(lines) == 5 # header + 4 records + # Check source column (2nd column after member) + assert ",jira," in lines[1] # assignee + assert ",github," in lines[2] # pr_authored + assert ",jira," in lines[3] # qa + assert ",github," in lines[4] # pr_reviewed + + def test_missing_fields_write_empty_not_none(self): + activities = [ + { + "member": "alice", + "kind": "assignee", + # Most fields missing + } + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + # No literal "None" should appear + assert "None" not in csv_output + + def test_sorted_by_member_then_ts(self): + activities = [ + {"member": "bob", "kind": "assignee", "ts": "2026-07-02"}, + {"member": "alice", "kind": "assignee", "ts": "2026-07-03"}, + {"member": "alice", "kind": "assignee", "ts": "2026-07-01"}, + ] + csv_output = export_csv.build_activities_csv(activities) + lines = csv_output.strip().split("\n") + assert len(lines) == 4 + # alice (sorted first alphabetically), with earlier ts first + assert lines[1].startswith("alice,") + assert "2026-07-01" in lines[1] + assert lines[2].startswith("alice,") + assert "2026-07-03" in lines[2] + # bob comes after alice + assert lines[3].startswith("bob,") + + +class TestBuildMatrixCSV(unittest.TestCase): + def test_emits_documented_header_with_workstreams(self): + matrix = ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + touched={"alice": set()}, + counts={"alice": {}}, + ) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + members=["alice"], + workstreams_touched_by_member={"alice": 0}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=1, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={"alice": 0}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + ) + csv_output = export_csv.build_matrix_csv(report) + lines = csv_output.strip().split("\n") + assert len(lines) == 2 # header + 1 member + # Header should have workstreams + TOTAL, workstreams_touched, flags + assert lines[0] == ( + "member,SNO,TNA,TNF,LVMS,USHIFT,TOPO,SHARED,TOTAL," + "workstreams_touched,over,under,narrow" + ) + + def test_includes_flag_and_workstreams_touched_columns(self): + matrix = ContributionMatrix( + members=["alice", "bob"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + touched={"alice": {"SNO", "TNA"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 100, "TNA": 100}, "bob": {"SNO": 10}}, + ) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + members=["alice", "bob"], + workstreams_touched_by_member={"alice": 2, "bob": 1}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=2, + total_member_count=2, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={"alice": 200, "bob": 10}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=50.0, + grid_max=100, + over_allocated={"alice"}, + under_allocated={"bob"}, + narrow=set(), + ), + ) + csv_output = export_csv.build_matrix_csv(report) + lines = csv_output.strip().split("\n") + assert len(lines) == 3 # header + 2 members + # alice row: over=true, under=false, narrow=false, workstreams_touched=2, total=200 + alice_row = [line for line in lines if line.startswith("alice,")][0] + assert ",200,2,true,false,false" in alice_row + # bob row: over=false, under=true, narrow=false, workstreams_touched=1, total=10 + bob_row = [line for line in lines if line.startswith("bob,")][0] + assert ",10,1,false,true,false" in bob_row + + def test_member_name_with_comma_is_csv_quoted(self): + """Ported from test_render.py — verifies CSV quoting for names with commas.""" + matrix = ContributionMatrix( + members=["Cope, Jon"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + touched={"Cope, Jon": {"SNO"}}, + counts={"Cope, Jon": {"SNO": 1}}, + ) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + members=["Cope, Jon"], + workstreams_touched_by_member={"Cope, Jon": 1}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=1, + total_member_count=1, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={"Cope, Jon": 1}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=1, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + ) + csv_output = export_csv.build_matrix_csv(report) + # Parse the CSV to verify the name is correctly quoted and round-trips + import csv + import io + + lines = csv_output.strip().split("\n") + rows = list(csv.reader(io.StringIO(csv_output))) + # First data row should have the member name + assert rows[1][0] == "Cope, Jon" + # The SNO column (index 1) should have the count + assert rows[1][1] == "1" + + def test_csv_contains_raw_counts_not_binary(self): + """Ported from test_render.py — verifies raw counts appear in CSV, not 0/1.""" + matrix = ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 42}}, + ) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + members=["alice"], + workstreams_touched_by_member={"alice": 1}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=1, + total_member_count=1, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={"alice": 42}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=42, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + ) + csv_output = export_csv.build_matrix_csv(report) + # Parse and verify the SNO column (index 1) contains the raw count 42 + import csv + import io + + rows = list(csv.reader(io.StringIO(csv_output))) + assert rows[1][1] == "42", "SNO column should contain raw count 42, not binary 0/1" + + +class TestBuildMetricsCSV(unittest.TestCase): + def test_emits_documented_header(self): + matrix = ContributionMatrix(members=[], workstreams=[], touched={}, counts={}) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=[], + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + ) + csv_output = export_csv.build_metrics_csv(report) + lines = csv_output.strip().split("\n") + assert lines[0] == "metric,value" + + @staticmethod + def _two_workstream_report(): + matrix = ContributionMatrix( + members=["alice", "bob"], + workstreams=["SNO", "TNA"], + touched={"alice": {"SNO"}, "bob": {"SNO", "TNA"}}, + counts={"alice": {"SNO": 50}, "bob": {"SNO": 50, "TNA": 50}}, + ) + return ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=["SNO", "TNA"], + members=["alice", "bob"], + workstreams_touched_by_member={"alice": 1, "bob": 2}, + mean_people_per_workstream=1.5, + mean_workstreams_per_person=1.5, + active_member_count=2, + total_member_count=2, + ), + matrix=matrix, + unattributed_count=5, + signals=AllocationSignals( + total_by_member={"alice": 50, "bob": 100}, + total_by_workstream={"SNO": 100, "TNA": 50}, + contributors_by_workstream={"SNO": 2, "TNA": 1}, + team_median_total=75.0, + grid_max=100, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + ) + + def test_includes_team_scalars_and_per_workstream_metrics(self): + report = self._two_workstream_report() + csv_output = export_csv.build_metrics_csv(report) + lines = csv_output.strip().split("\n") + + # Check team scalars are present, each under a label that says what it counts + assert "mean_people_per_workstream,1.5" in csv_output + assert "mean_workstreams_per_person,1.5" in csv_output + assert "active_members,2" in csv_output + assert "total_members,2" in csv_output + assert "team_median_total,75.0" in csv_output + assert "grid_max,100" in csv_output + assert "unattributed,5" in csv_output + + # Check per-workstream metrics + assert "total:SNO,100" in csv_output + assert "total:TNA,50" in csv_output + assert "contributors:SNO,2" in csv_output + assert "contributors:TNA,1" in csv_output + + def test_no_opaque_metric_labels_remain(self): + """The words 'flexibility' and 'opportunity' must never reach a user. + + metrics.csv was the only place they surfaced. `flexibility:` also + duplicated `contributors:` row for row, so it was removed outright + rather than renamed. + """ + csv_output = export_csv.build_metrics_csv(self._two_workstream_report()) + + assert "flexibility" not in csv_output + assert "opportunity" not in csv_output + + labels = [line.split(",")[0] for line in csv_output.strip().split("\n")[1:]] + assert len(labels) == len(set(labels)), f"duplicate metric labels: {labels}" + + +class TestBuildDataQualityCSV(unittest.TestCase): + def test_emits_documented_header(self): + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=[], + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + matrix=ContributionMatrix(members=[], workstreams=[], touched={}, counts={}), + data_quality=DataQuality(), + ) + csv_output = export_csv.build_data_quality_csv(report) + lines = csv_output.strip().split("\n") + assert lines[0] == "category,label,count" + + def test_does_not_truncate_repo_lists(self): + # Create more than 8 repos (the _TOP_REPO_LIMIT in render.py) + many_repos = {f"org/repo{i}": i + 1 for i in range(15)} + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=[], + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + matrix=ContributionMatrix(members=[], workstreams=[], touched={}, counts={}), + data_quality=DataQuality(excluded_personal_repos=many_repos), + ) + csv_output = export_csv.build_data_quality_csv(report) + lines = [ + line + for line in csv_output.strip().split("\n") + if line.startswith("excluded_personal_repo,") + ] + # All 15 repos should be present, not truncated to 8 + assert len(lines) == 15 + + +class TestExportFiles(unittest.TestCase): + def test_export_files_constant_matches_actual_output(self): + # The EXPORT_FILES constant pins the contract that the skill doc relies on + assert export_csv.EXPORT_FILES == ( + "activities.csv", + "matrix.csv", + "metrics.csv", + "data-quality.csv", + ) + + +class TestEmptyActivities(unittest.TestCase): + def test_empty_activities_produces_header_only_files(self): + # All builders should handle empty input gracefully + csv_output = export_csv.build_activities_csv([]) + assert len(csv_output.strip().split("\n")) == 1 # header only + + matrix = ContributionMatrix(members=[], workstreams=[], touched={}, counts={}) + report = ManagerReport( + period_label="2026Q3", + metrics=TeamMetrics( + workstreams=[], + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + matrix=matrix, + signals=AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=DataQuality(), + ) + + csv_output = export_csv.build_matrix_csv(report) + assert len(csv_output.strip().split("\n")) == 1 # header only + + csv_output = export_csv.build_metrics_csv(report) + # Metrics will have header + 7 scalar rows even with empty data + assert "metric,value" in csv_output + + csv_output = export_csv.build_data_quality_csv(report) + # Data quality might have header + jira category rows even with no repos + assert "category,label,count" in csv_output + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_load_context.py b/plugins/edge-contribution/bin/tests/test_load_context.py new file mode 100644 index 00000000..da6073ad --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_load_context.py @@ -0,0 +1,185 @@ +"""Tests for load_context.py — the edge-context team roster. + +The roster markdown is fetched straight from GitHub via the ``gh`` CLI (no local +checkout); an injected command runner keeps these tests hermetic (no subprocess, +no network). Coverage spans happy-path parsing, failure inputs (malformed rows, +missing Rover link, no table, gh failure, malformed JSON, wrong encoding, invalid +UTF-8), and the boundary/anti-cheat cases around the Eng+QE role filter, whose +titles are a substring trap ("Manager, Engineering" and "Product Security +Engineer" contain "Engineer" but must be excluded). +""" + +import base64 +import dataclasses +import json +import os +import sys +import unittest + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import _common # noqa: E402 +import load_context # noqa: E402 + +WELL_FORMED_ROSTER = """# Team Roster + +## OpenShift Edge + +| Name | GitHub | Role | Location | +|------|--------|------|----------| +| [Foo Bar](https://rover.redhat.com/people/profile/foobar) | foobar | Senior Software Engineer | Texas | +| [Fuzz Buzz](https://rover.redhat.com/people/profile/fuzzbuzz) | fuzzbuzz | Senior Software Quality Engineer | North Carolina | +""" + +INCLUDED_ROLES = [ + "Software Engineer", + "Associate Software Engineer", + "Senior Software Engineer", + "Principal Software Engineer", + "Senior Software Quality Engineer", + "Principal Software Quality Engineer", + "Associate Software Quality Engineer", +] +EXCLUDED_ROLES = [ + "Senior Manager, Engineering", + "Manager, Engineering", + "Associate Manager, Engineering", + "Principal Product Security Engineer", + "Principal Product Manager - Technical", + "Project Manager - Technical", +] + +EXPECTED_ROSTER_ENDPOINT = "repos/openshift-eng/edge-context/contents/people/team-roster.md" + + +class FakeRunner: + """A command runner that records commands and returns a canned result.""" + + def __init__(self, result): + self.result = result + self.commands = [] + + def __call__(self, command): + self.commands.append(command) + return self.result + + +def _ok(stdout): + return _common.CommandResult(returncode=0, stdout=stdout, stderr="") + + +def _contents_response(markdown, encoding="base64", encoder=base64.b64encode): + content = encoder(markdown.encode("utf-8")).decode("ascii") + return _ok(json.dumps({"content": content, "encoding": encoding})) +class TestMemberDataclass(unittest.TestCase): + def test_member_has_only_required_fields(self): + """Member dataclass should only have name, github, role, jira_username.""" + member = load_context.Member( + name="Test User", + github="testuser", + role="Software Engineer", + jira_username="testuser@redhat.com" + ) + fields = {f.name for f in dataclasses.fields(member)} + assert fields == {"name", "github", "role", "jira_username"} + # Explicitly verify removed fields don't exist + assert not hasattr(member, "location") + assert not hasattr(member, "kerberos") + + +class TestParseRosterFailureInputs(unittest.TestCase): + def test_row_with_wrong_column_count_raises(self): + markdown = WELL_FORMED_ROSTER + "| [X](https://rover.redhat.com/people/profile/x) | gh |\n" + with self.assertRaises(load_context.ContextParseError): + load_context.parse_roster(markdown) + + def test_markdown_without_a_roster_table_raises(self): + with self.assertRaises(load_context.ContextParseError): + load_context.parse_roster("# Team Roster\n\nNo table here.\n") + + +class TestFetchRosterMarkdownHappyPath(unittest.TestCase): + def test_decodes_base64_content_to_markdown(self): + runner = FakeRunner(_contents_response(WELL_FORMED_ROSTER)) + markdown = load_context.fetch_roster_markdown(runner) + assert markdown == WELL_FORMED_ROSTER + + def test_calls_gh_api_for_the_default_roster_path(self): + runner = FakeRunner(_contents_response(WELL_FORMED_ROSTER)) + load_context.fetch_roster_markdown(runner) + assert runner.commands[0] == ["gh", "api", EXPECTED_ROSTER_ENDPOINT] + + def test_newline_wrapped_base64_still_decodes(self): + # GitHub's contents API wraps base64 content in newlines. + runner = FakeRunner(_contents_response(WELL_FORMED_ROSTER, encoder=base64.encodebytes)) + markdown = load_context.fetch_roster_markdown(runner) + assert markdown == WELL_FORMED_ROSTER + + +class TestFetchRosterMarkdownCommandPlumbing(unittest.TestCase): + def test_ref_is_added_as_a_query_parameter(self): + runner = FakeRunner(_contents_response(WELL_FORMED_ROSTER)) + load_context.fetch_roster_markdown(runner, ref="release-4.20") + assert runner.commands[0][-1] == f"{EXPECTED_ROSTER_ENDPOINT}?ref=release-4.20" + + def test_custom_repo_is_used_in_the_endpoint(self): + runner = FakeRunner(_contents_response(WELL_FORMED_ROSTER)) + load_context.fetch_roster_markdown(runner, repo="acme/config") + assert runner.commands[0][-1] == "repos/acme/config/contents/people/team-roster.md" + + +class TestFetchRosterMarkdownFailureInputs(unittest.TestCase): + def test_nonzero_exit_raises_context_parse_error(self): + runner = FakeRunner(_common.CommandResult(1, "", "gh: not authenticated")) + with self.assertRaises(load_context.ContextParseError): + load_context.fetch_roster_markdown(runner) + + def test_malformed_json_raises_context_parse_error(self): + runner = FakeRunner(_ok("this is not json")) + with self.assertRaises(load_context.ContextParseError): + load_context.fetch_roster_markdown(runner) + + def test_unexpected_encoding_raises_context_parse_error(self): + runner = FakeRunner(_ok(json.dumps({"content": "abc", "encoding": "none"}))) + with self.assertRaises(load_context.ContextParseError): + load_context.fetch_roster_markdown(runner) + + def test_content_that_is_not_utf8_raises_context_parse_error(self): + invalid_utf8 = base64.b64encode(b"\xff\xfe\xfa").decode("ascii") + runner = FakeRunner(_ok(json.dumps({"content": invalid_utf8, "encoding": "base64"}))) + with self.assertRaises(load_context.ContextParseError): + load_context.fetch_roster_markdown(runner) + + def test_empty_content_raises_rather_than_returning_empty(self): + runner = FakeRunner(_ok(json.dumps({"content": "", "encoding": "base64"}))) + with self.assertRaises(load_context.ContextParseError): + load_context.fetch_roster_markdown(runner) + + +class TestLoadRosterFromGithub(unittest.TestCase): + def test_propagates_a_fetch_failure(self): + runner = FakeRunner(_common.CommandResult(1, "", "gh: not authenticated")) + with self.assertRaises(load_context.ContextParseError): + load_context.load_roster_from_github(runner) + + +class TestRoleFilterEdgeCases(unittest.TestCase): + def test_engineering_and_qe_titles_are_included(self): + for role in INCLUDED_ROLES: + assert load_context.is_engineering_or_qe(role) is True, role + + def test_manager_pm_and_security_titles_are_excluded(self): + for role in EXCLUDED_ROLES: + assert load_context.is_engineering_or_qe(role) is False, role + + def test_filter_is_case_insensitive(self): + assert load_context.is_engineering_or_qe("SENIOR SOFTWARE ENGINEER") is True + + def test_whitespace_padding_in_cells_is_tolerated(self): + padded = WELL_FORMED_ROSTER.replace("| copejon |", "| copejon |") + members = load_context.parse_roster(padded) + assert members[0].github == "foobar" + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_metrics.py b/plugins/edge-contribution/bin/tests/test_metrics.py new file mode 100644 index 00000000..174c7ade --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_metrics.py @@ -0,0 +1,466 @@ +"""Tests for metrics.py — workstreams touched, and the team means, from a binary matrix. + +Covers a hand-computed happy-path matrix, failure inputs (empty roster, unknown +member), and boundary/anti-cheat cases: all-zero matrix, zero-contributor +workstreams still present in the per-workstream vector, a member touching all +six, and the single-active-member identity between the team mean and that +member's own count. +""" + +import os +import sys +import unittest + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import metrics # noqa: E402 + +SIX = ["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO"] + + +def _matrix(members, touched): + return metrics.ContributionMatrix(members=members, workstreams=list(SIX), touched=touched) + + +class TestMetricsHappyPath(unittest.TestCase): + def setUp(self): + # alice touches 3 workstreams, bob 1, carol none. + self.matrix = _matrix( + members=["alice", "bob", "carol"], + touched={"alice": {"SNO", "TNA", "TNF"}, "bob": {"SNO"}, "carol": set()}, + ) + + def test_workstreams_touched_per_member(self): + assert metrics.workstreams_touched(self.matrix, "alice") == 3 + assert metrics.workstreams_touched(self.matrix, "bob") == 1 + assert metrics.workstreams_touched(self.matrix, "carol") == 0 + + def test_total_touches_is_sum_of_ones(self): + assert metrics.total_touches(self.matrix) == 4 + + def test_mean_people_per_workstream_is_over_all_six_workstreams(self): + assert metrics.mean_people_per_workstream(self.matrix) == 4 / 6 + + def test_mean_workstreams_per_person_is_over_active_members(self): + # 4 ones across 2 active members (carol is inactive). + assert metrics.mean_workstreams_per_person(self.matrix) == 2.0 + + def test_active_members_excludes_zero_contribution_members(self): + assert metrics.active_members(self.matrix) == ["alice", "bob"] + + def test_compute_team_metrics_bundles_everything(self): + result = metrics.compute_team_metrics(self.matrix) + assert result.active_member_count == 2 + assert result.total_member_count == 3 + assert result.mean_people_per_workstream == 4 / 6 + assert result.mean_workstreams_per_person == 2.0 + assert result.workstreams_touched_by_member == {"alice": 3, "bob": 1, "carol": 0} + + +class TestMetricsFailureInputs(unittest.TestCase): + def test_empty_roster_does_not_divide_by_zero(self): + empty = _matrix(members=[], touched={}) + result = metrics.compute_team_metrics(empty) + assert result.mean_workstreams_per_person == 0.0 + assert result.mean_people_per_workstream == 0.0 + assert result.active_member_count == 0 + + def test_workstreams_touched_for_unknown_member_raises(self): + matrix = _matrix(members=["alice"], touched={"alice": {"SNO"}}) + with self.assertRaises(ValueError): + metrics.workstreams_touched(matrix, "nobody") + + +class TestMetricsEdgeCases(unittest.TestCase): + def test_all_zero_matrix_zeros_both_team_means(self): + matrix = _matrix(members=["alice", "bob"], touched={"alice": set(), "bob": set()}) + result = metrics.compute_team_metrics(matrix) + assert result.mean_workstreams_per_person == 0.0 + assert result.mean_people_per_workstream == 0.0 + assert set(result.workstreams_touched_by_member.values()) == {0} + + def test_zero_contributor_workstream_is_present_in_vector(self): + # Ported from the deleted flexibility_vector test: the per-workstream + # contributor counts must list every canonical workstream in order, + # including those nobody touched, so a thin workstream shows as 0 + # rather than vanishing from the report. Counts are supplied because + # contributors_by_workstream reads the weighted matrix, not `touched` — + # without them every column would read 0 and the test would pass for + # the wrong reason. + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=list(SIX), + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 1}}, + ) + vector = metrics.compute_allocation_signals(matrix).contributors_by_workstream + assert vector["SNO"] == 1 + assert vector["TOPO"] == 0 + assert list(vector.keys()) == SIX + + def test_member_active_in_all_six_touches_six(self): + matrix = _matrix(members=["alice"], touched={"alice": set(SIX)}) + assert metrics.workstreams_touched(matrix, "alice") == 6 + + def test_single_active_member_team_mean_equals_their_own_count(self): + matrix = _matrix(members=["alice"], touched={"alice": {"SNO", "TNA"}}) + assert metrics.mean_workstreams_per_person(matrix) == metrics.workstreams_touched( + matrix, "alice" + ) + + def test_touched_entries_outside_canonical_workstreams_are_ignored(self): + matrix = _matrix(members=["alice"], touched={"alice": {"SNO", "BOGUS"}}) + assert metrics.workstreams_touched(matrix, "alice") == 1 + + +class TestAllocationSignalsBackwardCompatibility(unittest.TestCase): + """Verify that ContributionMatrix built without counts field still works.""" + + def test_matrix_without_counts_defaults_to_empty_dict(self): + # Construct a matrix with only members, workstreams, and touched. + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], workstreams=list(SIX), touched={"alice": {"SNO"}} + ) + # The counts field should default to an empty dict. + assert matrix.counts == {} + + def test_all_existing_functions_work_without_counts(self): + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=list(SIX), touched={"alice": {"SNO", "TNA"}} + ) + # Every count derived from `touched` should still work. + assert metrics.workstreams_touched(matrix, "alice") == 2 + assert metrics.mean_people_per_workstream(matrix) > 0 + assert metrics.mean_workstreams_per_person(matrix) == 2 + result = metrics.compute_team_metrics(matrix) + assert result.active_member_count == 1 + + def test_allocation_signals_with_empty_counts(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], workstreams=list(SIX), touched={"alice": {"SNO"}} + ) + signals = metrics.compute_allocation_signals(matrix) + # All totals should be zero when counts is empty. + assert signals.total_by_member == {"alice": 0, "bob": 0} + assert all(v == 0 for v in signals.total_by_workstream.values()) + assert signals.team_median_total == 0.0 + assert signals.grid_max == 0 + + +class TestAllocationSignalsSums(unittest.TestCase): + """Test total_by_member and total_by_workstream summing logic.""" + + def test_total_by_member_sums_across_workstreams(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=["SNO", "TNA", "TNF"], + counts={"alice": {"SNO": 10, "TNA": 5, "TNF": 3}, "bob": {"SNO": 2}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_member == {"alice": 18, "bob": 2} + + def test_total_by_workstream_sums_down_column(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO", "TNA"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 5}, "carol": {"TNA": 3}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_workstream == {"SNO": 15, "TNA": 3} + + def test_column_absent_from_counts_appears_as_zero(self): + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO", "TNA", "TNF"], counts={"alice": {"SNO": 10}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_workstream == {"SNO": 10, "TNA": 0, "TNF": 0} + + def test_column_in_counts_but_not_in_workstreams_is_ignored(self): + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA"], + counts={"alice": {"SNO": 10, "BOGUS": 999}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # BOGUS should not contribute to alice's total. + assert signals.total_by_member == {"alice": 10} + assert "BOGUS" not in signals.total_by_workstream + + +class TestAllocationSignalsContributors(unittest.TestCase): + """Test contributors_by_workstream counts members, not contributions.""" + + def test_contributors_by_workstream_counts_members(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO", "TNA"], + counts={"alice": {"SNO": 100}, "bob": {"SNO": 1}, "carol": {"TNA": 50}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.contributors_by_workstream == {"SNO": 2, "TNA": 1} + + def test_zero_contributors_workstream_present(self): + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO", "TNA"], counts={"alice": {"SNO": 10}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.contributors_by_workstream == {"SNO": 1, "TNA": 0} + + +class TestAllocationSignalsMedian(unittest.TestCase): + """Test team_median_total ignores zero-activity members.""" + + def test_team_median_total_ignores_zero_activity_members(self): + # alice: 10, bob: 20, carol: 0. Median of [10, 20] is 15.0. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 20}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.team_median_total == 15.0 + + def test_team_median_total_correct_for_even_active_count(self): + # alice: 10, bob: 30. Median is 20.0. + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], workstreams=["SNO"], counts={"alice": {"SNO": 10}, "bob": {"SNO": 30}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.team_median_total == 20.0 + + def test_team_median_total_correct_for_odd_active_count(self): + # alice: 10, bob: 20, carol: 30. Median is 20. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 20}, "carol": {"SNO": 30}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.team_median_total == 20.0 + + def test_team_median_total_zero_when_nobody_active(self): + matrix = metrics.ContributionMatrix(members=["alice", "bob"], workstreams=["SNO"], counts={}) + signals = metrics.compute_allocation_signals(matrix) + assert signals.team_median_total == 0.0 + + +class TestAllocationSignalsGridMax(unittest.TestCase): + """Test grid_max on normal, all-zero, and empty matrices.""" + + def test_grid_max_on_normal_grid(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=["SNO", "TNA"], + counts={"alice": {"SNO": 10, "TNA": 25}, "bob": {"SNO": 15}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.grid_max == 25 + + def test_grid_max_on_all_zero_grid(self): + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO"], counts={"alice": {"SNO": 0}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.grid_max == 0 + + def test_grid_max_on_empty_matrix(self): + matrix = metrics.ContributionMatrix(members=[], workstreams=[], counts={}) + signals = metrics.compute_allocation_signals(matrix) + assert signals.grid_max == 0 + + +class TestAllocationSignalsOverUnderAllocated(unittest.TestCase): + """Test over_allocated and under_allocated boundary behavior.""" + + def test_over_allocated_boundary(self): + # Median is 10. 2.0 * 10 = 20. alice at exactly 20 should NOT be over. + # bob at 21 should be over. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO"], + counts={"alice": {"SNO": 20}, "bob": {"SNO": 21}, "carol": {"SNO": 10}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median of [10, 20, 21] is 20. 2.0 * 20 = 40. + # alice: 20 <= 40, bob: 21 <= 40, carol: 10 <= 40. None over. + # Let me recalculate: median([10, 20, 21]) = 20. + # Over threshold is > 40. So no one is over. + # Actually, let me construct a clearer example. + # Let's make median = 10 by having [5, 10, 15]. Median is 10. + # Over threshold is > 20. alice at 20 should not be over, bob at 21 should be. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol", "dave"], + workstreams=["SNO"], + counts={"alice": {"SNO": 20}, "bob": {"SNO": 21}, "carol": {"SNO": 5}, "dave": {"SNO": 15}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median of [5, 15, 20, 21] is (15 + 20) / 2 = 17.5. 2.0 * 17.5 = 35. + # alice: 20 <= 35, bob: 21 <= 35, carol: 5 <= 35, dave: 15 <= 35. None over. + # Let me construct another example: [10, 10, 10, 30]. Median is (10+10)/2 = 10. + # 2.0 * 10 = 20. dave at 30 should be over. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol", "dave"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 10}, "carol": {"SNO": 10}, "dave": {"SNO": 30}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median of [10, 10, 10, 30] is (10+10)/2 = 10. 2.0 * 10 = 20. + # dave: 30 > 20, so dave is over. + assert "dave" in signals.over_allocated + # Now test exactly at the boundary: [10, 10, 10, 20]. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol", "dave"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 10}, "carol": {"SNO": 10}, "dave": {"SNO": 20}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median is 10. 2.0 * 10 = 20. dave at exactly 20 is NOT over (> not >=). + assert "dave" not in signals.over_allocated + + def test_under_allocated_boundary(self): + # Median is 10. 0.5 * 10 = 5. alice at exactly 5 should NOT be under. + # bob at 4 should be under. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol"], + workstreams=["SNO"], + counts={"alice": {"SNO": 5}, "bob": {"SNO": 4}, "carol": {"SNO": 10}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median of [4, 5, 10] is 5. 0.5 * 5 = 2.5. + # alice: 5 >= 2.5, bob: 4 >= 2.5. None under. + # Let me construct a clearer example: [10, 10, 10, 4]. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol", "dave"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 10}, "carol": {"SNO": 10}, "dave": {"SNO": 4}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median of [4, 10, 10, 10] is (10+10)/2 = 10. 0.5 * 10 = 5. + # dave: 4 < 5, so dave is under. + assert "dave" in signals.under_allocated + # Now test exactly at the boundary: [10, 10, 10, 5]. + matrix = metrics.ContributionMatrix( + members=["alice", "bob", "carol", "dave"], + workstreams=["SNO"], + counts={"alice": {"SNO": 10}, "bob": {"SNO": 10}, "carol": {"SNO": 10}, "dave": {"SNO": 5}}, + ) + signals = metrics.compute_allocation_signals(matrix) + # Median is 10. 0.5 * 10 = 5. dave at exactly 5 is NOT under (< not <=). + assert "dave" not in signals.under_allocated + + def test_zero_activity_member_not_in_over_or_under_allocated(self): + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], workstreams=["SNO"], counts={"alice": {"SNO": 10}} + ) + signals = metrics.compute_allocation_signals(matrix) + # bob has total 0, should not be in either set. + assert "bob" not in signals.over_allocated + assert "bob" not in signals.under_allocated + + +class TestAllocationSignalsNarrow(unittest.TestCase): + """Test narrow: members where a single workstream is > 70% of their total.""" + + def test_member_at_exactly_70_percent_is_not_narrow(self): + # alice: SNO: 7, TNA: 3. SNO is exactly 70%. + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO", "TNA"], counts={"alice": {"SNO": 7, "TNA": 3}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert "alice" not in signals.narrow + + def test_member_at_71_percent_is_narrow(self): + # alice: SNO: 71, TNA: 29. SNO is 71%. + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO", "TNA"], counts={"alice": {"SNO": 71, "TNA": 29}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert "alice" in signals.narrow + + def test_member_with_single_workstream_is_narrow(self): + # alice only in SNO (100%). + matrix = metrics.ContributionMatrix( + members=["alice"], workstreams=["SNO", "TNA"], counts={"alice": {"SNO": 10}} + ) + signals = metrics.compute_allocation_signals(matrix) + assert "alice" in signals.narrow + + def test_zero_total_member_is_not_narrow(self): + matrix = metrics.ContributionMatrix(members=["alice"], workstreams=["SNO"], counts={}) + signals = metrics.compute_allocation_signals(matrix) + assert "alice" not in signals.narrow + + +class TestAllocationSignalsSharedAsymmetry(unittest.TestCase): + """Test SHARED asymmetry: contributes to totals but not to workstreams touched.""" + + def test_shared_contributes_to_total_by_member(self): + # alice: SNO: 10, SHARED: 5. Total should be 15. + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA", "SHARED"], + counts={"alice": {"SNO": 10, "SHARED": 5}}, + ) + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_member == {"alice": 15} + + def test_shared_does_not_affect_workstreams_touched(self): + # alice touches SNO and SHARED. Workstreams touched should be 1 (only SNO). + # But wait, the test is about the asymmetry. SHARED is in workstreams here, + # so it WILL be counted if touched is {"alice": {"SNO", "SHARED"}}. + # The asymmetry is that SHARED is NOT in the canonical six workstreams, + # but it can appear in the workstreams list for display purposes. + # Let me construct this correctly. + # The canonical six are SNO, TNA, TNF, LVMS, USHIFT, TOPO. + # SHARED is NOT in that list. So if matrix.workstreams is the canonical six, + # SHARED won't count as a workstream touched. If display_columns includes + # SHARED, allocation signals include it but the touched count does not. + # Let me set up the test to show that: + # matrix.workstreams = canonical six only. alice touches SNO via touched. + # alice has SNO: 10, SHARED: 5 in counts. But SHARED is not in workstreams, + # so it should be ignored. + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10, "SHARED": 5}}, + ) + # Workstreams touched should be 1 (only SNO). + assert metrics.workstreams_touched(matrix, "alice") == 1 + # But total_by_member should only count SNO (10), not SHARED (5), + # because SHARED is not in workstreams. + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_member == {"alice": 10} + + # Now test the opposite: SHARED is in workstreams. + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO", "SHARED"], + touched={"alice": {"SNO"}}, # SHARED not in touched + counts={"alice": {"SNO": 10, "SHARED": 5}}, + ) + # Workstreams touched should still be 1 (only SNO in touched). + assert metrics.workstreams_touched(matrix, "alice") == 1 + # But total_by_member should now include SHARED (15). + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_member == {"alice": 15} + + +class TestAllocationSignalsEmptyMatrix(unittest.TestCase): + """Test empty matrix returns well-formed AllocationSignals.""" + + def test_empty_matrix_no_exception(self): + matrix = metrics.ContributionMatrix(members=[], workstreams=[], counts={}) + signals = metrics.compute_allocation_signals(matrix) + assert signals.total_by_member == {} + assert signals.total_by_workstream == {} + assert signals.contributors_by_workstream == {} + assert signals.team_median_total == 0.0 + assert signals.grid_max == 0 + assert signals.over_allocated == set() + assert signals.under_allocated == set() + assert signals.narrow == set() + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_render.py b/plugins/edge-contribution/bin/tests/test_render.py new file mode 100644 index 00000000..edc91929 --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_render.py @@ -0,0 +1,1447 @@ +"""Tests for render.py — text and markdown renderers for all report types. + +Covers happy-path structure, failure inputs (unknown format, empty roster), and +boundary cases: single-column matrix, all-zero matrix, gradient rendering, and +summary source-disclosure sections. +""" + +import csv +import io +import os +import sys +import unittest + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import metrics # noqa: E402 +import render # noqa: E402 + +SIX = ["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO"] + + +def _manager_report( + members, touched, workstreams=None, counts=None, unattributed=0, period="2026Q2" +): + matrix = metrics.ContributionMatrix( + members=members, workstreams=list(workstreams or SIX), touched=touched, counts=counts or {} + ) + return render.ManagerReport( + period_label=period, + metrics=metrics.compute_team_metrics(matrix), + matrix=matrix, + unattributed_count=unattributed, + ) + + +class TestManagerHappyPath(unittest.TestCase): + def setUp(self): + self.report = _manager_report( + members=["alice", "bob", "carol"], + touched={"alice": {"SNO", "TNA"}, "bob": {"SNO"}, "carol": set()}, + ) + + def test_text_grid_shows_gradient_glyphs(self): + # With counts, we should see gradient glyphs instead of checkmarks. + report = _manager_report( + members=["alice", "bob"], + touched={"alice": {"SNO", "TNA"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 10, "TNA": 2}, "bob": {"SNO": 1}}, + ) + text = render.render_manager(report, "text", color=False) + # Should contain gradient glyphs, not checkmarks. + assert "✓" not in text + # Should contain scale legend. + assert "scale" in text.lower() + + +class TestManagerFailureInputs(unittest.TestCase): + def test_unknown_format_raises(self): + report = _manager_report(members=["alice"], touched={"alice": {"SNO"}}) + with self.assertRaises(ValueError): + render.render_manager(report, "pdf") + + def test_empty_roster_renders_without_crashing(self): + report = _manager_report(members=[], touched={}) + for output_format in ("text",): + rendered = render.render_manager(report, output_format) + assert isinstance(rendered, str) + assert rendered != "" + + +class TestManagerEdgeCases(unittest.TestCase): + def test_single_workstream_column(self): + # Replaced: no longer testing CSV column selection; just verify a one-workstream report renders. + report = _manager_report(members=["alice"], touched={"alice": {"SNO"}}, workstreams=["SNO"]) + text = render.render_manager(report, "text", color=False) + assert "alice" in text + assert "SNO" in text + + def test_all_zero_matrix_has_no_filled_cells_in_text(self): + report = _manager_report(members=["alice", "bob"], touched={"alice": set(), "bob": set()}) + text = render.render_manager(report, "text", color=False) + # With gradient, should only see the "none" glyph (·). + assert "·" in text or "." in text + + +class TestManagerTextIsGridOnly(unittest.TestCase): + """Text output is the grid plus its two legends, and stops there. + + A manager reads the text view at a glance in a terminal. Everything after + the flags legend — the team-level means, the unattributed note, the + data-quality block — repeated what the grid's TOTAL/PEOPLE rows already + show, or buried it under a wall of names. All of it is still computed, and + available via --view summary --show-sources; only the manager text view is + trimmed. + """ + + def _loaded_report(self): + """A report carrying every optional block, so omission is a real choice.""" + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + dq = render.DataQuality( + unattributed_by_repo={"openshift/foo": 5}, + tickets_no_component_no_parent=3, + tickets_parent_also_empty=2, + excluded_personal_repo_prs=9, + excluded_personal_repos={"someone/side-project": 9}, + excluded_members=["foobar@redhat.com"], + ) + return render.ManagerReport( + period_label="2026Q2", + metrics=metrics.compute_team_metrics(matrix), + matrix=matrix, + unattributed_count=8, + data_quality=dq, + ) + + def test_text_omits_the_team_level_means(self): + text = render.render_manager(self._loaded_report(), "text", color=False) + assert "people per workstream" not in text + assert "workstreams per person" not in text + + def test_text_omits_the_data_quality_block(self): + text = render.render_manager(self._loaded_report(), "text", color=False) + assert "Data quality" not in text + assert "openshift/foo" not in text + assert "someone/side-project" not in text + + def test_text_omits_the_unattributed_note(self): + text = render.render_manager(self._loaded_report(), "text", color=False) + assert "unattributed" not in text.lower() + + def test_text_last_line_is_the_flags_legend(self): + # Pins the cut point. Anything appended after the legend fails here. + text = render.render_manager(self._loaded_report(), "text", color=False) + assert text.strip().splitlines()[-1].lstrip().startswith("flags") + + +class TestICReport(unittest.TestCase): + def _ic_report(self, workstreams_touched=2, activity=None, period="2026Q2"): + activity = ( + activity + if activity is not None + else [ + render.WorkstreamActivity("SNO", {"assignee": 3, "pr_authored": 2}), + render.WorkstreamActivity("TNA", {"pr_reviewed": 1}), + ] + ) + return render.ICReport( + period_label=period, + member="foobar@redhat.com", + workstreams_touched=workstreams_touched, + total_workstreams=6, + activity=activity, + ) + + def test_text_shows_workstreams_touched_of_total(self): + text = render.render_ic(self._ic_report(), "text") + assert "2 of 6" in text or "2/6" in text + assert "SNO" in text + assert "TNA" in text + + def test_unknown_format_raises(self): + with self.assertRaises(ValueError): + render.render_ic(self._ic_report(), "pdf") + + def test_zero_workstreams_touched_still_renders(self): + text = render.render_ic(self._ic_report(workstreams_touched=0, activity=[]), "text") + assert "0 of 6" in text or "0/6" in text + + +class TestShadeBand(unittest.TestCase): + def test_count_zero_returns_zero(self): + assert render.shade_band(0, 100) == 0 + + def test_count_one_with_grid_max_one_returns_four(self): + # When count equals grid_max, it should be maximum intensity (band 4). + assert render.shade_band(1, 1) == 4 + + def test_count_equals_grid_max_returns_four(self): + assert render.shade_band(69, 69) == 4 + + def test_grid_max_zero_returns_zero(self): + assert render.shade_band(10, 0) == 0 + + def test_negative_count_returns_zero(self): + assert render.shade_band(-5, 100) == 0 + + def test_count_above_grid_max_clamps_to_four(self): + assert render.shade_band(100, 50) == 4 + + def test_log_spacing_spreads_values(self): + # With grid_max 69, different counts should land in different bands. + grid_max = 69 + band_1 = render.shade_band(1, grid_max) + band_3 = render.shade_band(3, grid_max) + band_10 = render.shade_band(10, grid_max) + band_60 = render.shade_band(60, grid_max) + + # These should not all be the same band (linear would collapse them). + all_bands = {band_1, band_3, band_10, band_60} + assert len(all_bands) > 1, "Log spacing should spread counts across multiple bands" + + def test_band_values_are_in_range(self): + # All band values should be 0..4 inclusive. + for count in range(0, 100): + for grid_max in [1, 10, 50, 100]: + band = render.shade_band(count, grid_max) + assert ( + 0 <= band <= 4 + ), f"Band {band} out of range for count={count}, grid_max={grid_max}" + + +class TestGradientHeatmap(unittest.TestCase): + def test_color_true_contains_ansi_codes(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + text = render.render_manager(report, "text", color=True) + assert "\x1b[38;5;" in text, "Color output should contain ANSI color codes" + + def test_color_false_has_no_ansi_codes(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + text = render.render_manager(report, "text", color=False) + assert "\x1b" not in text, "No-color output should not contain any escape sequences" + + def test_ascii_only_true_is_pure_ascii(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + text = render.render_manager(report, "text", ascii_only=True, color=False) + # Should be encodable as ASCII without error. + text.encode("ascii") + + def test_ascii_only_false_uses_unicode(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + text = render.render_manager(report, "text", ascii_only=False, color=False) + # Should contain Unicode box drawing or other non-ASCII characters. + # The em-dash or box drawing characters would fail ASCII encoding. + with self.assertRaises(UnicodeEncodeError): + text.encode("ascii") + + def test_high_count_differs_from_low_count_glyph(self): + report = _manager_report( + members=["alice", "bob"], + touched={"alice": {"SNO"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 1}, "bob": {"SNO": 60}}, + ) + text = render.render_manager(report, "text", color=False) + # With different counts, the glyphs should differ. + # We can't assert exact glyphs without parsing, but we can check that + # multiple gradient glyphs appear. + glyph_set = set(["░", "▒", "▓", "█"]) + found_glyphs = sum(1 for g in glyph_set if g in text) + assert found_glyphs >= 2, "Different counts should produce different glyphs" + + def test_empty_roster_renders_without_crash(self): + report = _manager_report(members=[], touched={}, counts={}) + for output_format in ("text",): + rendered = render.render_manager(report, output_format, color=False) + assert isinstance(rendered, str) + assert rendered != "" + + def test_single_member_renders(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 5}}, + ) + text = render.render_manager(report, "text", color=False) + assert "alice" in text + + def test_all_zero_matrix_renders(self): + report = _manager_report( + members=["alice", "bob"], + touched={"alice": set(), "bob": set()}, + counts={}, + ) + text = render.render_manager(report, "text", color=False) + assert "alice" in text + assert "bob" in text + + def test_matrix_with_empty_counts_dict_renders(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={}, + ) + text = render.render_manager(report, "text", color=False) + assert "alice" in text + + def test_signals_none_still_renders(self): + # When signals is None, it should be computed internally. + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + report = render.ManagerReport( + period_label="2026Q2", + metrics=metrics.compute_team_metrics(matrix), + matrix=matrix, + signals=None, # Explicitly None. + ) + text = render.render_manager(report, "text", color=False) + assert "alice" in text + + def test_data_quality_none_omits_section(self): + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + text = render.render_manager(report, "text", color=False) + assert "Data quality" not in text + + def _report_with_excluded_repos(self, repos, total): + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + dq = render.DataQuality(excluded_personal_repo_prs=total, excluded_personal_repos=repos) + return render.ManagerReport( + period_label="2026Q2", + metrics=metrics.compute_team_metrics(matrix), + matrix=matrix, + unattributed_count=0, + data_quality=dq, + ) + + def _summary_with_excluded_repos(self, repos, total): + """Build an ExecutiveSummary with excluded repos for testing --show-sources.""" + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + team_metrics = metrics.compute_team_metrics(matrix) + dq = render.DataQuality(excluded_personal_repo_prs=total, excluded_personal_repos=repos) + return render.ExecutiveSummary( + period_label="2026Q2", + window=("2026-04-01", "2026-06-30"), + metrics=team_metrics, + signals=metrics.compute_allocation_signals(matrix), + data_quality=dq, + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE",), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + out_of_window_records=0, + ) + + # The excluded-repo listing now lives in the summary --show-sources view. + + def test_excluded_repos_are_named_in_html(self): + summary = self._summary_with_excluded_repos( + {"jeff-roche/roundhouse": 38, "jaypoulz/edge-tooling": 5}, 43 + ) + text = render.render_summary(summary, "text", show_sources=True) + assert "jeff-roche/roundhouse" in text + assert "jaypoulz/edge-tooling" in text + + def test_excluded_repos_listed_biggest_first(self): + summary = self._summary_with_excluded_repos( + {"small/one": 1, "big/two": 50, "mid/three": 10}, 61 + ) + text = render.render_summary(summary, "text", show_sources=True) + assert text.index("big/two") < text.index("mid/three") < text.index("small/one") + + def test_long_excluded_repo_list_is_truncated(self): + repos = {f"person{i}/repo": 1 for i in range(12)} + summary = self._summary_with_excluded_repos(repos, 12) + text = render.render_summary(summary, "text", show_sources=True) + assert "and 4 more repos" in text + + def test_no_excluded_repos_adds_no_listing(self): + # The total alone must still render; an empty breakdown is what a + # pre-breakdown activity file produces. + summary = self._summary_with_excluded_repos({}, 101) + text = render.render_summary(summary, "text", show_sources=True) + assert "101 PRs" in text + assert "personal-namespace repos" in text + assert "more repos" not in text + + def test_shared_column_appears_but_separated(self): + from workstream_map import SHARED_COLUMN + + # SHARED should appear in the output but be visually separated. + report = _manager_report( + members=["alice"], + touched={"alice": {"SNO", SHARED_COLUMN}}, + counts={"alice": {"SNO": 5, SHARED_COLUMN: 3}}, + ) + text = render.render_manager(report, "text", color=False, ascii_only=True) + assert SHARED_COLUMN in text + assert "|" in text # ASCII box vert separator. + + def test_rows_ordered_by_total_descending(self): + report = _manager_report( + members=["alice", "bob", "carol"], + touched={"alice": {"SNO"}, "bob": {"SNO", "TNA"}, "carol": {"SNO"}}, + counts={"alice": {"SNO": 5}, "bob": {"SNO": 10, "TNA": 20}, "carol": {"SNO": 1}}, + ) + text = render.render_manager(report, "text", color=False) + lines = text.splitlines() + # Find the member rows (they should be after the header). + member_rows = [ + line for line in lines if "alice" in line or "bob" in line or "carol" in line + ] + # Bob (30 total) should come first, then alice (5), then carol (1). + assert member_rows[0].startswith("bob") or "bob" in member_rows[0] + assert member_rows[1].startswith("alice") or "alice" in member_rows[1] + assert member_rows[2].startswith("carol") or "carol" in member_rows[2] + + +class TestRenderSummary(unittest.TestCase): + def test_render_summary_rejects_html_format(self): + # Build a minimal ExecutiveSummary. + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE", "USHIFT", "OCPBUGS"), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + with self.assertRaises(ValueError): + render.render_summary(summary, "html") + + def test_render_summary_rejects_csv_format(self): + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE", "USHIFT", "OCPBUGS"), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + with self.assertRaises(ValueError): + render.render_summary(summary, "csv") + + def test_render_summary_accepts_text_format(self): + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE", "USHIFT", "OCPBUGS"), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + result = render.render_summary(summary, "text") + assert isinstance(result, str) + + def test_render_summary_accepts_markdown_format(self): + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE", "USHIFT", "OCPBUGS"), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + result = render.render_summary(summary, "markdown") + assert isinstance(result, str) + + def _make_loaded_summary(self): + """A summary with all sections populated, matching the plan's example.""" + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=SIX, + touched={"alice": {"SNO", "TNA"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 100, "TNA": 50}, "bob": {"SNO": 30}}, + ) + team_metrics = metrics.compute_team_metrics(matrix) + signals = metrics.AllocationSignals( + total_by_member={"alice": 150, "bob": 30}, + total_by_workstream={ + "SNO": 130, + "TNA": 50, + "TNF": 0, + "LVMS": 0, + "USHIFT": 0, + "TOPO": 0, + }, + contributors_by_workstream={ + "SNO": 2, + "TNA": 1, + "TNF": 0, + "LVMS": 0, + "USHIFT": 0, + "TOPO": 0, + }, + team_median_total=65.0, + grid_max=100, + over_allocated={"alice"}, + under_allocated=set(), + narrow=set(), + ) + dq = render.DataQuality( + unattributed_by_repo={ + "openshift-eng/two-node-toolbox": 23, + "openshift/some-repo": 5, + }, + tickets_no_component_no_parent=104, + tickets_parent_also_empty=33, + excluded_personal_repo_prs=99, + excluded_personal_repos={"foobar/fuzzbuzz": 38, "user/repo": 10}, + excluded_members=["foobar@redhat.com", "fuzzbuzz@redhat.com"], + ) + return render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=team_metrics, + signals=signals, + data_quality=dq, + counts_by_source_kind={ + "jira": {"assignee": 737, "qa": 94, "ocpstrat_role": 45}, + "github": {"pr_authored": 555, "pr_reviewed": 729}, + }, + counts_by_attribution={ + "component": 847, + "repo": 337, + "project": 333, + "shared": 277, + "parent": 117, + "": 249, + }, + per_member=[ + render.MemberLine( + member="alice", total=150, workstreams_touched=2, top_workstream="SNO", top_share=67 + ), + render.MemberLine( + member="bob", total=30, workstreams_touched=1, top_workstream="SNO", top_share=100 + ), + ], + per_workstream=[ + render.WorkstreamLine( + workstream="SNO", + volume=130, + contributors=2, + top_contributor="alice", + top_share=77, + ), + render.WorkstreamLine( + workstream="TNA", + volume=50, + contributors=1, + top_contributor="alice", + top_share=100, + ), + render.WorkstreamLine( + workstream="TNF", volume=0, contributors=0, top_contributor=None, top_share=0 + ), + render.WorkstreamLine( + workstream="LVMS", volume=0, contributors=0, top_contributor=None, top_share=0 + ), + render.WorkstreamLine( + workstream="USHIFT", volume=0, contributors=0, top_contributor=None, top_share=0 + ), + render.WorkstreamLine( + workstream="TOPO", volume=0, contributors=0, top_contributor=None, top_share=0 + ), + ], + total_records=2160, + jira_projects=("OCPEDGE", "USHIFT", "OCPBUGS"), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + + def test_text_output_contains_total_records(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + assert "2160" in text + + def test_text_output_contains_attribution_tally(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + assert "component" in text.lower() + assert "847" in text + assert "repo" in text.lower() + assert "337" in text + + def test_text_output_contains_team_scores(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text") + # TEAM section carries the count, under a label that says what it counts + assert "contributions" in text.lower() + assert "people per workstream" in text.lower() + + def test_text_output_contains_per_member_table(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text") + assert "alice" in text + assert "bob" in text + assert "150" in text + assert "30" in text + + def test_text_output_contains_all_exclusion_categories(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + # Check for excluded members + assert "foobar@redhat.com" in text or "2 roster members" in text + # Check for excluded PRs + assert "99" in text # excluded_personal_repo_prs + # Check for unattributed items + assert "249" in text # unattributed count + + def test_text_output_preserves_two_node_toolbox_note(self): + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + assert "two-node-toolbox" in text + assert "not mapped by design" in text + + def test_markdown_output_contains_total_records(self): + summary = self._make_loaded_summary() + markdown = render.render_summary(summary, "markdown", show_sources=True) + assert "2160" in markdown + + def test_markdown_output_contains_attribution_tally(self): + summary = self._make_loaded_summary() + markdown = render.render_summary(summary, "markdown", show_sources=True) + assert "component" in markdown.lower() + assert "847" in markdown + + def test_markdown_output_contains_team_scores(self): + summary = self._make_loaded_summary() + markdown = render.render_summary(summary, "markdown") + # TEAM section contains contributions count + assert "contributions" in markdown.lower() + assert "people per workstream" in markdown.lower() + + def test_markdown_output_contains_per_member_table(self): + summary = self._make_loaded_summary() + markdown = render.render_summary(summary, "markdown") + assert "alice" in markdown + assert "bob" in markdown + + def test_markdown_output_has_pipe_tables(self): + summary = self._make_loaded_summary() + markdown = render.render_summary(summary, "markdown") + # Markdown tables use pipes + assert "|" in markdown + + def test_percentages_handle_empty_summary_without_error(self): + # An empty summary should not cause ZeroDivisionError. + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=None, + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=[], + workstreams_touched_by_member={}, + mean_people_per_workstream=0.0, + mean_workstreams_per_person=0.0, + active_member_count=0, + total_member_count=0, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=0.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + # Should not raise + text = render.render_summary(summary, "text") + assert isinstance(text, str) + markdown = render.render_summary(summary, "markdown") + assert isinstance(markdown, str) + + def test_attribution_descriptions_are_grammatical(self): + # Test exact descriptions to prevent regressions like "the inherited from". + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + # Check that descriptions are grammatical (no "the X" where X starts with a vowel or article) + assert "the issue's own components field" in text + assert "the PR's repo maps to exactly one workstream" in text + assert "the Jira project key" in text + # These must NOT have "the" prefix + assert "a cross-cutting CI" in text or "cross-cutting CI" in text + assert "the cross-cutting CI" not in text, "shared description must not have 'the' prefix" + assert "inherited from the parent epic" in text + assert "the inherited from" not in text, "parent description must not have 'the' prefix" + + def test_workstream_order_is_canonical_not_alphabetical(self): + # WORKSTREAMS table should be sorted by volume descending. + # Since SNO has 130 and TNA has 50, they should appear in that order. + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text") + lines = text.splitlines() + sno_found = False + tna_found = False + for i, line in enumerate(lines): + if "SNO" in line and sno_found == False: + sno_line = i + sno_found = True + if "TNA" in line and tna_found == False: + tna_line = i + tna_found = True + if sno_found and tna_found: + assert sno_line < tna_line, "SNO (130 volume) should appear before TNA (50 volume)" + + def test_activity_kind_order_is_canonical_not_alphabetical(self): + # Jira and GitHub detail lines must use ACTIVITY_KINDS order, not alphabetical. + summary = self._make_loaded_summary() + text = render.render_summary(summary, "text", show_sources=True) + # Find the Jira line with kinds + for line in text.splitlines(): + if "assignee" in line and "qa" in line and "ocpstrat_role" in line: + # Should be assignee · qa · ocpstrat_role (ACTIVITY_KINDS order) + # NOT assignee · ocpstrat_role · qa (alphabetical) + assignee_pos = line.index("assignee") + qa_pos = line.index("qa") + ocpstrat_pos = line.index("ocpstrat_role") + assert ( + assignee_pos < qa_pos < ocpstrat_pos + ), "Jira kinds must follow ACTIVITY_KINDS order" + break + + def test_no_trailing_whitespace_in_rendered_output(self): + # No line in either format may end in whitespace. + # Test with a member who has no flags to catch trailing spaces. + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=SIX, + touched={"alice": {"SNO"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 100}, "bob": {"SNO": 50}}, + ) + team_metrics = metrics.compute_team_metrics(matrix) + signals = metrics.compute_allocation_signals(matrix) + + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=team_metrics, + signals=signals, + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[ + render.MemberLine( + member="alice", total=100, workstreams_touched=1, top_workstream="SNO", top_share=100 + ), + render.MemberLine( + member="bob", total=50, workstreams_touched=1, top_workstream="SNO", top_share=100 + ), + ], + per_workstream=[], + total_records=0, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + for output_format in ("text", "markdown"): + result = render.render_summary(summary, output_format) + for line in result.splitlines(): + assert not line.endswith(" ") and not line.endswith( + "\t" + ), f"Line ends in whitespace ({output_format}): {line!r}" + + def test_roster_count_arithmetic_is_correct(self): + # metrics.total_member_count is post-exclusion, not pre-exclusion. + # With 16 active and 2 excluded, output must be "16 of 18 (2 excluded)". + matrix = metrics.ContributionMatrix( + members=["m" + str(i) for i in range(16)], + workstreams=SIX, + touched={}, + counts={}, + ) + team_metrics = metrics.compute_team_metrics(matrix) + signals = metrics.compute_allocation_signals(matrix) + + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=team_metrics, + signals=signals, + data_quality=render.DataQuality( + excluded_members=["foobar@redhat.com", "fuzzbuzz@redhat.com"], + ), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text") + assert ( + "Roster 16 of 18 (2 excluded)" in text + ), "Roster count must show included of full, not double-subtract excluded" + markdown = render.render_summary(summary, "markdown") + assert "16 of 18 (2 excluded)" in markdown + + def test_roster_count_omits_excluded_clause_when_zero(self): + # When there are no excluded members, the "(N excluded)" clause should be omitted. + matrix = metrics.ContributionMatrix( + members=["m" + str(i) for i in range(16)], + workstreams=SIX, + touched={}, + counts={}, + ) + team_metrics = metrics.compute_team_metrics(matrix) + signals = metrics.compute_allocation_signals(matrix) + + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=team_metrics, + signals=signals, + data_quality=render.DataQuality( + excluded_members=[], + ), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text") + assert "Roster 16 of 16" in text, "Roster count with no exclusions" + assert "excluded)" not in text.lower(), "Should not show '(0 excluded)'" + markdown = render.render_summary(summary, "markdown") + assert "16 of 16" in markdown + assert "excluded)" not in markdown.lower() + + def test_github_caveat_appears_in_text_format(self): + # When out_of_window_records > 0, the GitHub caveat must appear in text output. + summary = self._make_loaded_summary() + # Replace with a summary that has out_of_window_records + summary_with_out_of_window = render.ExecutiveSummary( + period_label=summary.period_label, + window=summary.window, + metrics=summary.metrics, + signals=summary.signals, + data_quality=summary.data_quality, + counts_by_source_kind=summary.counts_by_source_kind, + counts_by_attribution=summary.counts_by_attribution, + per_member=summary.per_member, + per_workstream=summary.per_workstream, + total_records=summary.total_records, + jira_projects=summary.jira_projects, + ocpstrat_project=summary.ocpstrat_project, + allowed_org_count=summary.allowed_org_count, + out_of_window_records=145, + ) + text = render.render_summary(summary_with_out_of_window, "text", show_sources=True) + # Check for key phrases in the caveat + assert "authored" in text + assert "created in the window" in text or "created" in text.lower() + assert "reviewed" in text + assert "updated" in text.lower() + assert "GitHub cannot search by review date" in text or "review date" in text.lower() + assert "creation date" in text.lower() + assert "145" in text + assert "still counted" in text or "counted" in text.lower() + + def test_github_caveat_appears_in_markdown_format(self): + # When out_of_window_records > 0, the GitHub caveat must appear in markdown output. + summary = self._make_loaded_summary() + summary_with_out_of_window = render.ExecutiveSummary( + period_label=summary.period_label, + window=summary.window, + metrics=summary.metrics, + signals=summary.signals, + data_quality=summary.data_quality, + counts_by_source_kind=summary.counts_by_source_kind, + counts_by_attribution=summary.counts_by_attribution, + per_member=summary.per_member, + per_workstream=summary.per_workstream, + total_records=summary.total_records, + jira_projects=summary.jira_projects, + ocpstrat_project=summary.ocpstrat_project, + allowed_org_count=summary.allowed_org_count, + out_of_window_records=145, + ) + markdown = render.render_summary(summary_with_out_of_window, "markdown", show_sources=True) + # Check for key phrases in the caveat + assert "authored" in markdown + assert "created" in markdown.lower() + assert "reviewed" in markdown + assert "updated" in markdown.lower() + assert "review date" in markdown.lower() + assert "creation date" in markdown.lower() + assert "145" in markdown + + def test_github_caveat_omits_count_when_zero(self): + # When out_of_window_records is 0, the count sentence must be omitted, + # but the method explanation must still appear. + summary = self._make_loaded_summary() + summary_with_zero = render.ExecutiveSummary( + period_label=summary.period_label, + window=summary.window, + metrics=summary.metrics, + signals=summary.signals, + data_quality=summary.data_quality, + counts_by_source_kind=summary.counts_by_source_kind, + counts_by_attribution=summary.counts_by_attribution, + per_member=summary.per_member, + per_workstream=summary.per_workstream, + total_records=summary.total_records, + jira_projects=summary.jira_projects, + ocpstrat_project=summary.ocpstrat_project, + allowed_org_count=summary.allowed_org_count, + out_of_window_records=0, + ) + text = render.render_summary(summary_with_zero, "text", show_sources=True) + # Method explanation should still appear + assert "authored" in text + assert "reviewed" in text + # But the specific count/warning sentences should not appear + assert "carry a date before the window" not in text + assert "do not filter activities.csv" not in text + + def test_github_caveat_includes_filter_warning_when_nonzero(self): + # When out_of_window_records > 0, must warn against filtering activities.csv by ts. + summary = self._make_loaded_summary() + summary_with_out_of_window = render.ExecutiveSummary( + period_label=summary.period_label, + window=summary.window, + metrics=summary.metrics, + signals=summary.signals, + data_quality=summary.data_quality, + counts_by_source_kind=summary.counts_by_source_kind, + counts_by_attribution=summary.counts_by_attribution, + per_member=summary.per_member, + per_workstream=summary.per_workstream, + total_records=summary.total_records, + jira_projects=summary.jira_projects, + ocpstrat_project=summary.ocpstrat_project, + allowed_org_count=summary.allowed_org_count, + out_of_window_records=145, + ) + text = render.render_summary(summary_with_out_of_window, "text", show_sources=True) + assert "ts" in text.lower() or "timestamp" in text.lower() + assert "filter" in text.lower() or "filtering" in text.lower() + + +class TestSummaryRefactored(unittest.TestCase): + """Tests for the refactored summary view with the --show-sources flag.""" + + def test_default_summary_does_not_contain_source_sections(self): + # Default output should NOT contain WHAT WAS COUNTED, HOW EACH ITEM WAS ATTRIBUTED, or WHAT WAS EXCLUDED + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=render.DataQuality(), + counts_by_source_kind={"jira": {"assignee": 10}, "github": {}}, + counts_by_attribution={"component": 10}, + per_member=[ + render.MemberLine( + member="alice", total=10, workstreams_touched=1, top_workstream="SNO", top_share=100 + ) + ], + per_workstream=[ + render.WorkstreamLine( + workstream="SNO", + volume=10, + contributors=1, + top_contributor="alice", + top_share=100, + ) + ], + total_records=10, + jira_projects=("OCPEDGE",), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + text = render.render_summary(summary, "text", show_sources=False) + assert "WHAT WAS COUNTED" not in text + assert "HOW EACH ITEM WAS ATTRIBUTED" not in text + assert "WHAT WAS EXCLUDED" not in text + + def test_default_summary_does_not_contain_flag_strings(self): + # Default output should NOT contain "over", "under", or "narrow" as flags + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=SIX, + touched={"alice": {"SNO"}, "bob": {"TNA"}}, + counts={"alice": {"SNO": 200}, "bob": {"TNA": 10}}, + ) + signals = metrics.compute_allocation_signals(matrix) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=signals, + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[ + render.MemberLine( + member="alice", total=200, workstreams_touched=1, top_workstream="SNO", top_share=100 + ), + render.MemberLine( + member="bob", total=10, workstreams_touched=1, top_workstream="TNA", top_share=100 + ), + ], + per_workstream=[], + total_records=0, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text", show_sources=False) + # Check that flag tokens don't appear in the member section + # Look for lines that would have flags and verify they don't contain the tokens + lines = text.splitlines() + for line in lines: + if "alice" in line.lower() or "bob" in line.lower(): + # These lines should not contain flag tokens as standalone words + # Match whole words to avoid false positives like "cover" or "underline" + import re + + assert not re.search(r"\bover\b", line), f"Found 'over' flag in: {line}" + assert not re.search(r"\bunder\b", line), f"Found 'under' flag in: {line}" + assert not re.search(r"\bnarrow\b", line), f"Found 'narrow' flag in: {line}" + + def test_show_sources_true_adds_all_sections(self): + # With show_sources=True, output should contain all three source sections + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + dq = render.DataQuality( + unattributed_by_repo={"openshift-eng/two-node-toolbox": 5}, + tickets_no_component_no_parent=3, + excluded_members=["foobar@redhat.com"], + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=dq, + counts_by_source_kind={"jira": {"assignee": 10}, "github": {}}, + counts_by_attribution={"component": 5, "": 5}, # 5 unattributed + per_member=[ + render.MemberLine( + member="alice", total=10, workstreams_touched=1, top_workstream="SNO", top_share=100 + ) + ], + per_workstream=[], + total_records=10, + jira_projects=("OCPEDGE",), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + text = render.render_summary(summary, "text", show_sources=True) + assert "WHAT WAS COUNTED" in text + assert "HOW EACH ITEM WAS ATTRIBUTED" in text + assert "WHAT WAS EXCLUDED" in text + assert "foobar@redhat.com" in text # excluded member + assert "two-node-toolbox" in text + assert "not mapped by design" in text + + def test_workstream_line_dataclass_exists(self): + # Test that WorkstreamLine exists with correct fields + line = render.WorkstreamLine( + workstream="SNO", volume=100, contributors=5, top_contributor="alice", top_share=50 + ) + assert line.workstream == "SNO" + assert line.volume == 100 + assert line.contributors == 5 + assert line.top_contributor == "alice" + assert line.top_share == 50 + + def test_workstream_line_with_zero_volume(self): + # Zero-volume workstream should have top_contributor=None, top_share=0 + line = render.WorkstreamLine( + workstream="TOPO", volume=0, contributors=0, top_contributor=None, top_share=0 + ) + assert line.volume == 0 + assert line.top_contributor is None + assert line.top_share == 0 + + def test_member_line_has_top_workstream_and_share(self): + # New MemberLine should have top_workstream and top_share, no flags + line = render.MemberLine( + member="alice", total=150, workstreams_touched=3, top_workstream="SNO", top_share=60 + ) + assert line.member == "alice" + assert line.total == 150 + assert line.workstreams_touched == 3 + assert line.top_workstream == "SNO" + assert line.top_share == 60 + # Should not have flags attribute + assert not hasattr(line, "flags") + + def test_default_summary_contains_team_section(self): + # Default output should contain TEAM section + matrix = metrics.ContributionMatrix( + members=["alice", "bob"], + workstreams=SIX, + touched={"alice": {"SNO", "TNA"}, "bob": {"SNO"}}, + counts={"alice": {"SNO": 50, "TNA": 30}, "bob": {"SNO": 20}}, + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[ + render.MemberLine( + member="alice", total=80, workstreams_touched=2, top_workstream="SNO", top_share=62 + ), + render.MemberLine( + member="bob", total=20, workstreams_touched=1, top_workstream="SNO", top_share=100 + ), + ], + per_workstream=[ + render.WorkstreamLine( + workstream="SNO", + volume=70, + contributors=2, + top_contributor="alice", + top_share=71, + ), + render.WorkstreamLine( + workstream="TNA", + volume=30, + contributors=1, + top_contributor="alice", + top_share=100, + ), + ], + total_records=100, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text", show_sources=False) + assert "TEAM" in text + assert "contributions" in text.lower() + assert "median" in text.lower() + + def test_default_summary_contains_workstreams_section(self): + # Default output should contain WORKSTREAMS section + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[ + render.MemberLine( + member="alice", total=10, workstreams_touched=1, top_workstream="SNO", top_share=100 + ) + ], + per_workstream=[ + render.WorkstreamLine( + workstream="SNO", + volume=10, + contributors=1, + top_contributor="alice", + top_share=100, + ) + ], + total_records=10, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text", show_sources=False) + assert "WORKSTREAMS" in text + assert "volume" in text.lower() + assert "people" in text.lower() or "contributors" in text.lower() + + def test_default_summary_contains_people_section(self): + # Default output should contain PEOPLE section + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=render.DataQuality(), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[ + render.MemberLine( + member="alice", total=10, workstreams_touched=1, top_workstream="SNO", top_share=100 + ) + ], + per_workstream=[], + total_records=10, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text", show_sources=False) + assert "PEOPLE" in text + assert "alice" in text + + def test_default_summary_contains_data_health_section(self): + # Default output should contain DATA HEALTH section + matrix = metrics.ContributionMatrix( + members=["alice"], + workstreams=SIX, + touched={"alice": {"SNO"}}, + counts={"alice": {"SNO": 10}}, + ) + summary = render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.compute_team_metrics(matrix), + signals=metrics.compute_allocation_signals(matrix), + data_quality=render.DataQuality( + unattributed_by_repo={"org/repo": 5}, tickets_no_component_no_parent=3 + ), + counts_by_source_kind={}, + counts_by_attribution={"": 8}, + per_member=[ + render.MemberLine( + member="alice", total=10, workstreams_touched=1, top_workstream="SNO", top_share=100 + ) + ], + per_workstream=[], + total_records=18, + jira_projects=(), + ocpstrat_project="", + allowed_org_count=0, + ) + text = render.render_summary(summary, "text", show_sources=False) + assert "DATA HEALTH" in text + assert "could not be placed" in text.lower() or "unattributed" in text.lower() + + +class TestUnifiedSummaryRendering(unittest.TestCase): + """Test the unified rendering functions with format parameter.""" + + def _minimal_summary(self): + """Create a minimal ExecutiveSummary for testing.""" + return render.ExecutiveSummary( + period_label="2026Q3", + window=("2026-07-01", "2026-09-30"), + metrics=metrics.TeamMetrics( + workstreams=SIX, + members=["alice", "bob"], + workstreams_touched_by_member={}, + mean_people_per_workstream=2.5, + mean_workstreams_per_person=3.0, + active_member_count=2, + total_member_count=2, + ), + signals=metrics.AllocationSignals( + total_by_member={}, + total_by_workstream={}, + contributors_by_workstream={}, + team_median_total=10.0, + grid_max=0, + over_allocated=set(), + under_allocated=set(), + narrow=set(), + ), + data_quality=render.DataQuality(excluded_members=[]), + counts_by_source_kind={}, + counts_by_attribution={}, + per_member=[], + per_workstream=[], + total_records=0, + jira_projects=("OCPEDGE",), + ocpstrat_project="OCPSTRAT", + allowed_org_count=10, + ) + + def test_summary_header_text_format(self): + """_summary_header should produce text format when format='text'.""" + summary = self._minimal_summary() + lines = render._summary_header(summary, format="text") + assert isinstance(lines, list) + assert "OCP-Edge Cross-Workstream Contribution" in lines[0] + assert not lines[0].startswith("#") # No markdown heading + + def test_summary_header_markdown_format(self): + """_summary_header should produce markdown format when format='markdown'.""" + summary = self._minimal_summary() + lines = render._summary_header(summary, format="markdown") + assert isinstance(lines, list) + assert lines[0].startswith("# OCP-Edge") # Markdown H1 heading + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_report.py b/plugins/edge-contribution/bin/tests/test_report.py new file mode 100644 index 00000000..14d404c0 --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_report.py @@ -0,0 +1,893 @@ +"""Tests for report.py — aggregate collected activity into report structures. + +Covers happy-path matrix/report construction, failure inputs (activity for an +unknown member or an unknown workstream is ignored, never counted), and +boundary/anti-cheat cases: unattributed items counted but kept out of the +matrix, duplicate contributions counted once toward workstreams touched but summed +per kind for the IC breakdown, and a member with no activity touching 0. +""" + +import json +import os +import sys +import tempfile +import unittest + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import _common # noqa: E402 +import report # noqa: E402 + +SIX = ["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO"] +SEVEN = SIX + ["SHARED"] + + +def _activity(member, workstream, kind="assignee"): + return {"member": member, "workstream": workstream, "kind": kind} + + +class TestBuildContributionMatrix(unittest.TestCase): + def test_touched_sets_reflect_activity(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "TNA"), + _activity("bob", "SNO"), + ] + matrix = report.build_contribution_matrix(activities, ["alice", "bob"], SIX) + assert matrix.touched["alice"] == {"SNO", "TNA"} + assert matrix.touched["bob"] == {"SNO"} + + def test_activity_for_unknown_member_is_ignored(self): + activities = [_activity("stranger", "SNO")] + matrix = report.build_contribution_matrix(activities, ["alice"], SIX) + assert matrix.touched["alice"] == set() + + def test_activity_for_unknown_workstream_is_ignored(self): + activities = [_activity("alice", "BOGUS")] + matrix = report.build_contribution_matrix(activities, ["alice"], SIX) + assert matrix.touched["alice"] == set() + + def test_unattributed_activity_is_not_in_matrix(self): + activities = [_activity("alice", None)] + matrix = report.build_contribution_matrix(activities, ["alice"], SIX) + assert matrix.touched["alice"] == set() + + def test_counts_populated_correctly(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "TNA"), + _activity("bob", "SNO"), + ] + matrix = report.build_contribution_matrix(activities, ["alice", "bob"], SIX) + assert matrix.counts["alice"]["SNO"] == 2 + assert matrix.counts["alice"]["TNA"] == 1 + assert matrix.counts["bob"]["SNO"] == 1 + + def test_seven_columns_with_shared_still_counts_n_of_6(self): + # Matrix built with display_columns() (7 columns: 6 + SHARED). + # Workstreams touched should still count only the 6 canonical ones. + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SHARED"), + ] + matrix = report.build_contribution_matrix(activities, ["alice"], SEVEN) + # Matrix has 7 workstreams including SHARED + assert len(matrix.workstreams) == 7 + assert "SHARED" in matrix.workstreams + # Alice touched both SNO and SHARED + assert matrix.touched["alice"] == {"SNO", "SHARED"} + # But workstreams touched should be 1 (SHARED doesn't count) + from metrics import workstreams_touched + + assert workstreams_touched(matrix, "alice") == 1 + + +class TestCountUnattributed(unittest.TestCase): + def test_counts_only_none_workstream(self): + activities = [_activity("alice", None), _activity("alice", "SNO"), _activity("bob", None)] + assert report.count_unattributed(activities) == 2 + + def test_scoped_to_member(self): + activities = [_activity("alice", None), _activity("bob", None)] + assert report.count_unattributed(activities, member="alice") == 1 + + +class TestBuildManagerReport(unittest.TestCase): + def test_report_bundles_metrics_and_unattributed(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "TNA"), + _activity("bob", "SNO"), + _activity("bob", None), + ] + result = report.build_manager_report(activities, ["alice", "bob"], SIX, "2026Q2") + assert result.metrics.workstreams_touched_by_member == {"alice": 2, "bob": 1} + assert result.signals.contributors_by_workstream["SNO"] == 2 + assert result.unattributed_count == 1 + assert result.period_label == "2026Q2" + + +class TestTranslateActivitiesToDisplayNames(unittest.TestCase): + def test_translates_member_field(self): + activities = [ + _activity("alice@redhat.com", "SNO"), + _activity("bob@redhat.com", "TNA"), + ] + mapping = {"alice@redhat.com": "Alice Doe", "bob@redhat.com": "Bob Smith"} + translated = report.translate_activities_to_display_names(activities, mapping) + assert translated[0]["member"] == "Alice Doe" + assert translated[1]["member"] == "Bob Smith" + assert translated[0]["workstream"] == "SNO" + assert translated[1]["workstream"] == "TNA" + + def test_fallback_to_jira_username_when_missing(self): + activities = [_activity("charlie@redhat.com", "SNO")] + mapping = {"alice@redhat.com": "Alice Doe"} + translated = report.translate_activities_to_display_names(activities, mapping) + assert translated[0]["member"] == "charlie@redhat.com" + + def test_does_not_mutate_original_activities(self): + activities = [_activity("alice@redhat.com", "SNO")] + mapping = {"alice@redhat.com": "Alice Doe"} + report.translate_activities_to_display_names(activities, mapping) + # Original should be unchanged + assert activities[0]["member"] == "alice@redhat.com" + + +class TestBuildDataQuality(unittest.TestCase): + def test_tallies_unattributed_by_repo(self): + activities = [ + {"member": "alice", "workstream": None, "repo": "org/repo-a"}, + {"member": "bob", "workstream": None, "repo": "org/repo-a"}, + {"member": "charlie", "workstream": None, "repo": "org/repo-b"}, + ] + dq = report.build_data_quality(activities, []) + assert dq.unattributed_by_repo == {"org/repo-a": 2, "org/repo-b": 1} + + def test_ignores_attributed_records(self): + activities = [ + {"member": "alice", "workstream": "SNO", "repo": "org/repo-a"}, + {"member": "bob", "workstream": None, "repo": "org/repo-b"}, + ] + dq = report.build_data_quality(activities, []) + assert dq.unattributed_by_repo == {"org/repo-b": 1} + + def test_tallies_jira_tickets_no_component_no_parent(self): + activities = [ + {"member": "alice", "workstream": None}, # Jira ticket, no repo key + {"member": "bob", "workstream": None}, + ] + dq = report.build_data_quality(activities, []) + assert dq.tickets_no_component_no_parent == 2 + + def test_mixed_jira_and_github(self): + activities = [ + {"member": "alice", "workstream": None, "repo": "org/repo-a"}, # GitHub + {"member": "bob", "workstream": None}, # Jira + {"member": "charlie", "workstream": "SNO"}, # Attributed, ignored + ] + dq = report.build_data_quality(activities, []) + assert dq.unattributed_by_repo == {"org/repo-a": 1} + assert dq.tickets_no_component_no_parent == 1 + + def test_excluded_members_captured(self): + dq = report.build_data_quality([], ["alice", "bob"]) + assert dq.excluded_members == ["alice", "bob"] + + def test_undistinguishable_fields_are_zero(self): + # tickets_parent_also_empty and excluded_personal_repo_prs cannot be derived + dq = report.build_data_quality([], []) + assert dq.tickets_parent_also_empty == 0 + assert dq.excluded_personal_repo_prs == 0 + + +class TestShouldUseColor(unittest.TestCase): + def test_all_conditions_true_enables_color(self): + assert report.should_use_color("text", True, None, False) is True + + def test_no_color_flag_disables(self): + assert report.should_use_color("text", True, None, True) is False + + def test_non_text_format_disables(self): + assert report.should_use_color("markdown", True, None, False) is False + + def test_env_no_color_disables(self): + assert report.should_use_color("text", True, "1", False) is False + assert report.should_use_color("text", True, "true", False) is False + + def test_not_a_tty_disables(self): + assert report.should_use_color("text", False, None, False) is False + + def test_no_color_flag_overrides_all(self): + # Even if all other conditions are true, --no-color wins + assert report.should_use_color("text", True, None, True) is False + + +class TestExcludedMembers(unittest.TestCase): + def test_excluded_members_filtered_from_manager_report(self): + # Simulate EXCLUDED_MEMBERS being filtered + activities = [ + _activity("alice", "SNO"), + _activity("fuzzbuzz@redhat.com", "TNA"), # This is in EXCLUDED_MEMBERS + _activity("bob", "LVMS"), + ] + # Build report with fuzzbuzz excluded + members_without_excluded = ["alice", "bob"] + excluded = ["fuzzbuzz@redhat.com"] + report_obj = report.build_manager_report( + activities, members_without_excluded, SIX, "2026Q2", excluded + ) + # fuzzbuzz should not be in the matrix + assert "fuzzbuzz@redhat.com" not in report_obj.matrix.members + assert "alice" in report_obj.matrix.members + assert "bob" in report_obj.matrix.members + # But fuzzbuzz should be in data_quality.excluded_members + assert report_obj.data_quality.excluded_members == ["fuzzbuzz@redhat.com"] + + +class TestParseActivityFile(unittest.TestCase): + def test_envelope_yields_records_and_meta(self): + payload = { + "activities": [_activity("alice", "SNO")], + "collector_meta": {"excluded_personal_repo_prs": 7}, + } + records, meta = report.parse_activity_file(payload) + assert records == [_activity("alice", "SNO")] + assert meta == {"excluded_personal_repo_prs": 7} + + def test_bare_list_is_accepted_as_the_pre_envelope_format(self): + records, meta = report.parse_activity_file([_activity("alice", "SNO")]) + assert records == [_activity("alice", "SNO")] + assert meta == {} + + def test_envelope_without_meta_yields_empty_meta(self): + records, meta = report.parse_activity_file({"activities": []}) + assert records == [] + assert meta == {} + + def test_null_meta_yields_empty_meta_not_none(self): + _records, meta = report.parse_activity_file({"activities": [], "collector_meta": None}) + assert meta == {} + + +class TestLoadActivities(unittest.TestCase): + def _write(self, tmpdir, name, payload): + path = os.path.join(tmpdir, name) + with open(path, "w", encoding="utf-8") as handle: + json.dump(payload, handle) + return path + + def test_counters_sum_across_files_and_records_concatenate(self): + with tempfile.TemporaryDirectory() as tmpdir: + jira = self._write( + tmpdir, + "jira.json", + {"activities": [_activity("alice", "SNO")], "collector_meta": {}}, + ) + github = self._write( + tmpdir, + "github.json", + { + "activities": [_activity("bob", "LVMS")], + "collector_meta": {"excluded_personal_repo_prs": 5}, + }, + ) + activities, meta = report._load_activities([jira, github]) + assert len(activities) == 2 + assert meta == {"excluded_personal_repo_prs": 5} + + def test_round_trips_the_envelope_the_collectors_actually_write(self): + # Pins the two ends together: if _common.activity_payload changes shape, + # this fails here rather than silently reporting zero excluded PRs. + written = _common.activity_payload( + [_activity("alice", "SNO")], excluded_personal_repo_prs=101 + ) + records, meta = report.parse_activity_file(json.loads(json.dumps(written))) + assert records == [_activity("alice", "SNO")] + assert meta["excluded_personal_repo_prs"] == 101 + + def test_breakdown_counters_merge_key_by_key(self): + # excluded_personal_repos is a {repo: count} map, not a total. Summing it + # like an int would raise; ignoring it would lose the repo names. + with tempfile.TemporaryDirectory() as tmpdir: + first = self._write( + tmpdir, + "a.json", + { + "activities": [], + "collector_meta": {"excluded_personal_repos": {"a/one": 2, "b/two": 1}}, + }, + ) + second = self._write( + tmpdir, + "b.json", + { + "activities": [], + "collector_meta": {"excluded_personal_repos": {"a/one": 3, "c/three": 5}}, + }, + ) + _activities, meta = report._load_activities([first, second]) + assert meta["excluded_personal_repos"] == {"a/one": 5, "b/two": 1, "c/three": 5} + + def test_totals_and_breakdowns_merge_side_by_side(self): + with tempfile.TemporaryDirectory() as tmpdir: + path = self._write( + tmpdir, + "gh.json", + { + "activities": [], + "collector_meta": { + "excluded_personal_repo_prs": 3, + "excluded_personal_repos": {"a/one": 3}, + }, + }, + ) + _activities, meta = report._load_activities([path, path]) + assert meta["excluded_personal_repo_prs"] == 6 + assert meta["excluded_personal_repos"] == {"a/one": 6} + + def test_legacy_bare_list_file_loads_alongside_an_envelope(self): + with tempfile.TemporaryDirectory() as tmpdir: + legacy = self._write(tmpdir, "legacy.json", [_activity("alice", "SNO")]) + modern = self._write( + tmpdir, + "modern.json", + { + "activities": [_activity("bob", "LVMS")], + "collector_meta": {"excluded_personal_repo_prs": 3}, + }, + ) + activities, meta = report._load_activities([legacy, modern]) + assert len(activities) == 2 + assert meta == {"excluded_personal_repo_prs": 3} + + +class TestBuildDataQualityCounters(unittest.TestCase): + """The two counters that used to be hardcoded to 0.""" + + def test_parent_also_empty_is_separated_from_no_parent(self): + activities = [ + {"member": "alice", "workstream": None, "kind": "assignee"}, + { + "member": "alice", + "workstream": None, + "kind": "assignee", + "unattributed_reason": "parent_also_empty", + }, + { + "member": "bob", + "workstream": None, + "kind": "qa", + "unattributed_reason": "no_component_no_parent", + }, + ] + quality = report.build_data_quality(activities, []) + assert quality.tickets_parent_also_empty == 1 + assert quality.tickets_no_component_no_parent == 2 + + def test_attributed_tickets_are_not_counted(self): + activities = [ + { + "member": "alice", + "workstream": "SNO", + "kind": "assignee", + "unattributed_reason": "parent_also_empty", + } + ] + quality = report.build_data_quality(activities, []) + assert quality.tickets_parent_also_empty == 0 + assert quality.tickets_no_component_no_parent == 0 + + def test_github_prs_are_bucketed_by_repo_not_by_ticket_reason(self): + activities = [ + {"member": "alice", "workstream": None, "kind": "pr_authored", "repo": "o/x"}, + {"member": "bob", "workstream": None, "kind": "pr_authored", "repo": "o/x"}, + ] + quality = report.build_data_quality(activities, []) + assert quality.unattributed_by_repo == {"o/x": 2} + assert quality.tickets_no_component_no_parent == 0 + assert quality.tickets_parent_also_empty == 0 + + def test_excluded_personal_repo_prs_comes_from_collector_meta(self): + quality = report.build_data_quality([], [], {"excluded_personal_repo_prs": 101}) + assert quality.excluded_personal_repo_prs == 101 + + def test_missing_collector_meta_leaves_the_counter_at_zero(self): + assert report.build_data_quality([], []).excluded_personal_repo_prs == 0 + + def test_excluded_repos_breakdown_reaches_data_quality(self): + quality = report.build_data_quality( + [], + [], + { + "excluded_personal_repo_prs": 5, + "excluded_personal_repos": {"jeff-roche/roundhouse": 5}, + }, + ) + assert quality.excluded_personal_repos == {"jeff-roche/roundhouse": 5} + + def test_missing_breakdown_leaves_an_empty_map_not_none(self): + assert report.build_data_quality([], []).excluded_personal_repos == {} + + def test_manager_report_threads_collector_meta_through(self): + report_obj = report.build_manager_report( + [_activity("alice", "SNO")], + ["alice"], + SIX, + "2026Q2", + [], + {"excluded_personal_repo_prs": 12}, + ) + assert report_obj.data_quality.excluded_personal_repo_prs == 12 + + +class TestBuildICReport(unittest.TestCase): + def test_workstreams_touched_and_per_kind_breakdown(self): + activities = [ + _activity("alice", "SNO", "assignee"), + _activity("alice", "SNO", "pr_authored"), + _activity("alice", "TNA", "comment"), + _activity("bob", "SNO", "assignee"), + ] + result = report.build_ic_report(activities, "alice", SIX, "2026Q2") + assert result.workstreams_touched == 2 + assert result.total_workstreams == 6 + by_workstream = {activity.workstream: activity.counts for activity in result.activity} + assert by_workstream["SNO"] == {"assignee": 1, "pr_authored": 1} + assert by_workstream["TNA"] == {"comment": 1} + + def test_duplicate_contributions_count_once_but_sum_per_kind(self): + activities = [ + _activity("alice", "SNO", "comment"), + _activity("alice", "SNO", "comment"), + ] + result = report.build_ic_report(activities, "alice", SIX, "2026Q2") + assert result.workstreams_touched == 1 + assert result.activity[0].counts == {"comment": 2} + + def test_member_with_no_activity_touches_zero_workstreams(self): + result = report.build_ic_report([], "alice", SIX, "2026Q2") + assert result.workstreams_touched == 0 + assert result.activity == [] + + def test_workstreams_appear_in_canonical_order(self): + activities = [_activity("alice", "TOPO"), _activity("alice", "SNO")] + result = report.build_ic_report(activities, "alice", SIX, "2026Q2") + assert [activity.workstream for activity in result.activity] == ["SNO", "TOPO"] + + +class TestClassifySource(unittest.TestCase): + def test_jira_kinds_classified_as_jira(self): + assert report.classify_source("assignee") == "jira" + assert report.classify_source("qa") == "jira" + assert report.classify_source("ocpstrat_role") == "jira" + + def test_github_kinds_classified_as_github(self): + assert report.classify_source("pr_authored") == "github" + assert report.classify_source("pr_reviewed") == "github" + + def test_unknown_kind_classified_as_other(self): + assert report.classify_source("unknown_kind") == "other" + assert report.classify_source("comment") == "other" + + +class TestResolvePeriodWindow(unittest.TestCase): + def test_quarter_label_resolves_to_real_dates(self): + window = report.resolve_period_window("2026Q3") + assert window == ("2026-07-01", "2026-09-30") + + def test_explicit_date_range_splits_on_underscore(self): + window = report.resolve_period_window("2026-01-15_2026-03-31") + assert window == ("2026-01-15", "2026-03-31") + + def test_invalid_label_returns_none(self): + assert report.resolve_period_window("invalid") is None + assert report.resolve_period_window("2026") is None + assert report.resolve_period_window("") is None + +class TestCountOutOfWindowRecords(unittest.TestCase): + def test_counts_records_before_window(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-06-30T23:59:59Z"}, + {**_activity("bob", "TNA"), "ts": "2026-07-01T00:00:00Z"}, + ] + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records(activities, window) + assert count == 1 + + def test_counts_records_after_window(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-09-30T23:59:59Z"}, + {**_activity("bob", "TNA"), "ts": "2026-10-01T00:00:00Z"}, + ] + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records(activities, window) + assert count == 1 + + def test_boundary_dates_are_inside(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-07-01T00:00:00Z"}, + {**_activity("bob", "TNA"), "ts": "2026-09-30T23:59:59Z"}, + ] + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records(activities, window) + assert count == 0 + + def test_missing_ts_does_not_raise(self): + activities = [ + _activity("alice", "SNO"), # No ts field + ] + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records(activities, window) + assert count == 0 + + def test_malformed_ts_does_not_raise(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "not-a-date"}, + {**_activity("bob", "TNA"), "ts": "2026-07-15T00:00:00Z"}, + ] + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records(activities, window) + # Malformed should not count as out-of-window + assert count == 0 + + def test_none_window_returns_zero(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-06-15T00:00:00Z"}, + ] + count = report.count_out_of_window_records(activities, None) + assert count == 0 + + def test_empty_activities_returns_zero(self): + window = ("2026-07-01", "2026-09-30") + count = report.count_out_of_window_records([], window) + assert count == 0 + + +class TestBuildWorkstreamLines(unittest.TestCase): + """Tests for build_workstream_lines function.""" + + def test_returns_one_line_per_workstream_sorted_by_volume(self): + # Build a matrix with known counts + matrix = report.build_contribution_matrix( + [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "TNA"), + _activity("bob", "SNO"), + _activity("bob", "LVMS"), + _activity("bob", "LVMS"), + _activity("bob", "LVMS"), + ], + ["alice", "bob"], + SEVEN, + ) + workstreams = SEVEN + lines = report.build_workstream_lines(matrix, workstreams) + # Should be sorted by volume descending + # SNO and LVMS both have volume 3, so either can be first (ties break deterministically) + assert len(lines) == 7 # All 7 workstreams + assert lines[0].volume == 3 + assert lines[1].volume == 3 + assert lines[0].workstream in ("SNO", "LVMS") + assert lines[1].workstream in ("SNO", "LVMS") + assert lines[0].workstream != lines[1].workstream # Both appear but different + # TNA should be third with volume 1 + assert lines[2].workstream == "TNA" + assert lines[2].volume == 1 + + def test_top_contributor_and_share_correct(self): + # alice has 100 of 150 in SNO = 67% + matrix = report.build_contribution_matrix( + [_activity("alice", "SNO")] * 100 + [_activity("bob", "SNO")] * 50, + ["alice", "bob"], + SIX, + ) + lines = report.build_workstream_lines(matrix, SIX) + sno = next(line for line in lines if line.workstream == "SNO") + assert sno.top_contributor == "alice" + assert sno.top_share == 67 # rounded int percent + + def test_zero_volume_workstream_no_zero_division(self): + # A workstream with zero activity should have top_contributor=None, top_share=0 + matrix = report.build_contribution_matrix( + [_activity("alice", "SNO")], + ["alice"], + SIX, + ) + lines = report.build_workstream_lines(matrix, SIX) + # TNA should have volume 0 + tna = next((line for line in lines if line.workstream == "TNA"), None) + if tna: + assert tna.volume == 0 + assert tna.top_contributor is None + assert tna.top_share == 0 + + +class TestBuildExecutiveSummaryRefactored(unittest.TestCase): + """Tests for updated build_executive_summary with per_workstream and member top_workstream.""" + + def test_populates_per_workstream(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("bob", "SNO"), + _activity("bob", "TNA"), + ] + summary = report.build_executive_summary( + activities, ["alice", "bob"], SEVEN, "2026Q3", [], {} + ) + assert hasattr(summary, "per_workstream") + assert len(summary.per_workstream) > 0 + # SNO should be first (volume 3) + assert summary.per_workstream[0].workstream == "SNO" + assert summary.per_workstream[0].volume == 3 + + def test_member_line_has_top_workstream_and_share(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "TNA"), + ] + summary = report.build_executive_summary(activities, ["alice"], SEVEN, "2026Q3", [], {}) + alice = summary.per_member[0] + assert alice.top_workstream == "SNO" + assert alice.top_share == 75 # 3 of 4 + + def test_empty_activities_produces_summary_without_error(self): + # Should not raise ZeroDivisionError + summary = report.build_executive_summary([], [], SEVEN, "2026Q3", [], {}) + # With no members, per_workstream should have all workstreams with volume=0 + assert len(summary.per_workstream) == 7 # SEVEN includes SHARED + assert all(ws.volume == 0 for ws in summary.per_workstream) + assert summary.per_member == [] + + +class TestBuildExecutiveSummary(unittest.TestCase): + def test_source_kind_split(self): + activities = [ + _activity("alice", "SNO", "assignee"), + _activity("alice", "SNO", "assignee"), + _activity("alice", "TNA", "qa"), + _activity("bob", "SNO", "pr_authored"), + _activity("bob", "LVMS", "pr_reviewed"), + ] + summary = report.build_executive_summary( + activities, ["alice", "bob"], SEVEN, "2026Q3", [], {} + ) + + assert summary.counts_by_source_kind["jira"]["assignee"] == 2 + assert summary.counts_by_source_kind["jira"]["qa"] == 1 + assert summary.counts_by_source_kind["github"]["pr_authored"] == 1 + assert summary.counts_by_source_kind["github"]["pr_reviewed"] == 1 + + def test_attribution_tally_including_none(self): + activities = [ + {**_activity("alice", "SNO"), "attribution_source": "component"}, + {**_activity("alice", "TNA"), "attribution_source": "repo"}, + {**_activity("bob", None), "attribution_source": None}, + {**_activity("bob", None), "attribution_source": None}, + ] + summary = report.build_executive_summary( + activities, ["alice", "bob"], SEVEN, "2026Q3", [], {} + ) + + assert summary.counts_by_attribution["component"] == 1 + assert summary.counts_by_attribution["repo"] == 1 + assert summary.counts_by_attribution[""] == 2 # None uses empty string as key + + def test_per_member_sorted_by_total_desc(self): + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "TNA"), + _activity("bob", "LVMS"), + _activity("charlie", "SNO"), + _activity("charlie", "TNA"), + _activity("charlie", "TNF"), + _activity("charlie", "LVMS"), + ] + summary = report.build_executive_summary( + activities, ["alice", "bob", "charlie"], SEVEN, "2026Q3", [], {} + ) + + assert len(summary.per_member) == 3 + assert summary.per_member[0].member == "charlie" + assert summary.per_member[0].total == 4 + assert summary.per_member[1].member == "alice" + assert summary.per_member[1].total == 3 + assert summary.per_member[2].member == "bob" + assert summary.per_member[2].total == 1 + + def test_per_member_has_top_workstream_and_share(self): + # Create activities with clear top workstreams + activities = [ + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("alice", "SNO"), + _activity("bob", "LVMS"), + ] + summary = report.build_executive_summary( + activities, ["alice", "bob"], SEVEN, "2026Q3", [], {} + ) + + # Check that top_workstream and top_share are set correctly + alice_line = next(ml for ml in summary.per_member if ml.member == "alice") + bob_line = next(ml for ml in summary.per_member if ml.member == "bob") + + # Alice has 3 in SNO, bob has 1 in LVMS + assert alice_line.top_workstream == "SNO" + assert alice_line.top_share == 100 # 3 of 3 = 100% + assert bob_line.top_workstream == "LVMS" + assert bob_line.top_share == 100 # 1 of 1 = 100% + + def test_empty_activities_produces_zeros_not_error(self): + # This should not raise ZeroDivisionError + summary = report.build_executive_summary([], [], SEVEN, "2026Q3", [], {}) + + assert summary.total_records == 0 + assert summary.per_member == [] + assert summary.counts_by_source_kind == {"jira": {}, "github": {}} + assert summary.counts_by_attribution == {} + + def test_out_of_window_records_populated_when_window_resolved(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-06-30T23:59:59Z"}, + {**_activity("bob", "TNA"), "ts": "2026-07-15T00:00:00Z"}, + {**_activity("charlie", "LVMS"), "ts": "2026-10-01T00:00:00Z"}, + ] + summary = report.build_executive_summary( + activities, ["alice", "bob", "charlie"], SEVEN, "2026Q3", [], {} + ) + # 2026Q3 is 2026-07-01 to 2026-09-30, so alice and charlie are out of window + assert summary.out_of_window_records == 2 + + def test_out_of_window_records_zero_when_all_in_window(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-07-01T00:00:00Z"}, + {**_activity("bob", "TNA"), "ts": "2026-09-30T23:59:59Z"}, + ] + summary = report.build_executive_summary( + activities, ["alice", "bob"], SEVEN, "2026Q3", [], {} + ) + assert summary.out_of_window_records == 0 + + def test_out_of_window_records_zero_when_window_not_resolved(self): + activities = [ + {**_activity("alice", "SNO"), "ts": "2026-06-30T23:59:59Z"}, + ] + # Use a period that doesn't resolve to a window (no underscore, not a quarter) + summary = report.build_executive_summary( + activities, ["alice"], SEVEN, "custom-period", [], {} + ) + assert summary.out_of_window_records == 0 + + +class TestFormatRestrictions(unittest.TestCase): + """Test that invalid format choices are rejected by argparse.""" + + def test_manager_view_rejects_html_format(self): + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f: + json.dump([{"jira_username": "alice", "full_name": "Alice"}], f) + roster = f.name + try: + with self.assertRaises(SystemExit) as cm: + report.main( + [ + "--view", + "manager", + "--members-file", + roster, + "--activity", + roster, # dummy, won't be reached + "--period", + "2026Q3", + "--format", + "html", + ] + ) + self.assertEqual(cm.exception.code, 2) + finally: + os.unlink(roster) + + def test_manager_view_rejects_csv_format(self): + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f: + json.dump([{"jira_username": "alice", "full_name": "Alice"}], f) + roster = f.name + try: + with self.assertRaises(SystemExit) as cm: + report.main( + [ + "--view", + "manager", + "--members-file", + roster, + "--activity", + roster, # dummy + "--period", + "2026Q3", + "--format", + "csv", + ] + ) + self.assertEqual(cm.exception.code, 2) + finally: + os.unlink(roster) + + def test_ic_view_rejects_html_format(self): + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f: + json.dump([], f) + activity = f.name + try: + with self.assertRaises(SystemExit) as cm: + report.main( + [ + "--view", + "ic", + "--member", + "alice", + "--activity", + activity, + "--period", + "2026Q3", + "--format", + "html", + ] + ) + self.assertEqual(cm.exception.code, 2) + finally: + os.unlink(activity) + + def test_summary_view_rejects_csv_format(self): + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f: + json.dump([{"jira_username": "alice", "full_name": "Alice"}], f) + roster = f.name + try: + with self.assertRaises(SystemExit) as cm: + report.main( + [ + "--view", + "summary", + "--members-file", + roster, + "--activity", + roster, # dummy + "--period", + "2026Q3", + "--format", + "csv", + ] + ) + self.assertEqual(cm.exception.code, 2) + finally: + os.unlink(roster) + + def test_output_flag_is_rejected(self): + with tempfile.NamedTemporaryFile(mode="w", suffix=".json", delete=False) as f: + json.dump([{"jira_username": "alice", "full_name": "Alice"}], f) + roster = f.name + try: + with self.assertRaises(SystemExit) as cm: + report.main( + [ + "--view", + "manager", + "--members-file", + roster, + "--activity", + roster, + "--period", + "2026Q3", + "--output", + "/tmp/x", + ] + ) + self.assertEqual(cm.exception.code, 2) + finally: + os.unlink(roster) + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/tests/test_workstream_map.py b/plugins/edge-contribution/bin/tests/test_workstream_map.py new file mode 100644 index 00000000..c7667931 --- /dev/null +++ b/plugins/edge-contribution/bin/tests/test_workstream_map.py @@ -0,0 +1,263 @@ +"""Tests for workstream_map.py — the internal workstream -> Jira component map. + +Covers happy-path lookups, failure inputs (unknown/None/empty), and the +boundary/anti-cheat cases: case-insensitivity, whitespace, multi-component +aliases, and the explicitly-excluded ``Planning`` component. +""" + +import os +import sys +import unittest + +sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..")) + +import workstream_map # noqa: E402 + +# The canonical map, restated here independently of the module so the test is a +# real oracle rather than a mirror of the implementation. +EXPECTED_ALIAS_TO_ACRONYM = { + "SNO": "SNO", + "Two Node with Arbiter": "TNA", + "TNF": "TNF", + "Two Node Fencing": "TNF", + "Logical Volume Manager Storage": "LVMS", + "MicroShift": "USHIFT", + "Topology Transitions": "TOPO", + "Mutable Topology": "TOPO", +} +EXPECTED_ACRONYMS_IN_ORDER = ["SNO", "TNA", "TNF", "LVMS", "USHIFT", "TOPO"] + + +class TestWorkstreamsHappyPath(unittest.TestCase): + def test_component_to_workstream_maps_exact_names(self): + assert workstream_map.component_to_workstream("SNO") == "SNO" + assert workstream_map.component_to_workstream("Two Node Fencing") == "TNF" + assert workstream_map.component_to_workstream("MicroShift") == "USHIFT" + + def test_workstream_acronyms_returns_six_in_canonical_order(self): + assert workstream_map.workstream_acronyms() == EXPECTED_ACRONYMS_IN_ORDER + + def test_workstreams_returns_six_entries_with_name_and_components(self): + streams = workstream_map.workstreams() + assert len(streams) == 6 + for stream in streams: + assert stream.acronym + assert stream.name + assert len(stream.components) >= 1 + + +class TestWorkstreamsFailureInputs(unittest.TestCase): + def test_unknown_component_returns_none(self): + assert workstream_map.component_to_workstream("Foobar") is None + + def test_none_component_returns_none_without_raising(self): + assert workstream_map.component_to_workstream(None) is None + + def test_empty_string_component_returns_none(self): + assert workstream_map.component_to_workstream("") is None + + def test_whitespace_only_component_returns_none(self): + assert workstream_map.component_to_workstream(" ") is None + + +class TestWorkstreamsEdgeCases(unittest.TestCase): + def test_lookup_is_case_insensitive(self): + assert workstream_map.component_to_workstream("sno") == "SNO" + assert workstream_map.component_to_workstream("microSHIFT") == "USHIFT" + + def test_lookup_trims_surrounding_whitespace(self): + assert workstream_map.component_to_workstream(" TNF ") == "TNF" + + def test_tnf_aliases_both_map_to_tnf(self): + assert workstream_map.component_to_workstream("TNF") == "TNF" + assert workstream_map.component_to_workstream("Two Node Fencing") == "TNF" + + def test_topology_aliases_both_map_to_topo(self): + assert workstream_map.component_to_workstream("Topology Transitions") == "TOPO" + assert workstream_map.component_to_workstream("Mutable Topology") == "TOPO" + + def test_planning_component_is_excluded(self): + assert workstream_map.component_to_workstream("Planning") is None + + def test_every_known_alias_maps_as_expected(self): + for alias, acronym in EXPECTED_ALIAS_TO_ACRONYM.items(): + assert workstream_map.component_to_workstream(alias) == acronym, alias + + +class TestDisplayColumns(unittest.TestCase): + def test_returns_six_workstreams_plus_shared(self): + cols = workstream_map.display_columns() + assert len(cols) == 7 + + def test_ends_with_shared_column(self): + cols = workstream_map.display_columns() + assert cols[-1] == workstream_map.SHARED_COLUMN + + def test_shared_not_in_workstream_acronyms(self): + """SHARED is a display column but does not count toward workstreams touched.""" + acronyms = workstream_map.workstream_acronyms() + assert len(acronyms) == 6 + assert workstream_map.SHARED_COLUMN not in acronyms + + def test_mutating_returned_list_does_not_affect_module_state(self): + cols1 = workstream_map.display_columns() + cols1.append("MUTATED") + cols2 = workstream_map.display_columns() + assert len(cols2) == 7 + assert "MUTATED" not in cols2 + + +class TestProjectToWorkstream(unittest.TestCase): + def test_ushift_project_maps_to_ushift(self): + assert workstream_map.project_to_workstream("USHIFT-6337") == "USHIFT" + + def test_lowercase_issue_key_works(self): + assert workstream_map.project_to_workstream("ushift-1") == "USHIFT" + + def test_unknown_project_returns_none(self): + assert workstream_map.project_to_workstream("OCPEDGE-123") is None + + def test_none_returns_none(self): + assert workstream_map.project_to_workstream(None) is None + + def test_empty_string_returns_none(self): + assert workstream_map.project_to_workstream("") is None + + def test_malformed_key_no_dash_returns_none(self): + assert workstream_map.project_to_workstream("NODASH") is None + + def test_only_dash_returns_none(self): + assert workstream_map.project_to_workstream("-5") is None + + def test_whitespace_tolerant(self): + assert workstream_map.project_to_workstream(" USHIFT-42 ") == "USHIFT" + + +class TestRepoToWorkstream(unittest.TestCase): + def test_lvm_operator_maps_to_lvms(self): + assert workstream_map.repo_to_workstream("openshift/lvm-operator") == "LVMS" + + def test_topolvm_maps_to_lvms(self): + assert workstream_map.repo_to_workstream("openshift/topolvm") == "LVMS" + + def test_openshift_microshift_maps_to_ushift(self): + assert workstream_map.repo_to_workstream("openshift/microshift") == "USHIFT" + + def test_microshift_io_microshift_maps_to_ushift(self): + assert workstream_map.repo_to_workstream("microshift-io/microshift") == "USHIFT" + + def test_oc_tnf_maps_to_tnf(self): + assert workstream_map.repo_to_workstream("openshift/oc-tnf") == "TNF" + + def test_case_insensitive(self): + assert workstream_map.repo_to_workstream("OpenShift/LVM-Operator") == "LVMS" + + def test_unknown_repo_returns_none(self): + assert workstream_map.repo_to_workstream("openshift/unknown") is None + + def test_none_returns_none(self): + assert workstream_map.repo_to_workstream(None) is None + + def test_empty_string_returns_none(self): + assert workstream_map.repo_to_workstream("") is None + + def test_two_node_toolbox_is_deliberately_unmapped(self): + """Regression pin: two-node-toolbox covers both TNA and TNF, so it must not be mapped.""" + assert workstream_map.repo_to_workstream("openshift-eng/two-node-toolbox") is None + + +class TestIsSharedRepo(unittest.TestCase): + def test_openshift_release_is_shared(self): + assert workstream_map.is_shared_repo("openshift/release") is True + + def test_edge_tooling_is_shared(self): + assert workstream_map.is_shared_repo("openshift-eng/edge-tooling") is True + + def test_edge_context_is_shared(self): + assert workstream_map.is_shared_repo("openshift-eng/edge-context") is True + + def test_openshift_docs_is_shared(self): + assert workstream_map.is_shared_repo("openshift/openshift-docs") is True + + def test_enhancements_is_shared(self): + assert workstream_map.is_shared_repo("openshift/enhancements") is True + + def test_microshift_not_shared(self): + assert workstream_map.is_shared_repo("openshift/microshift") is False + + def test_none_returns_false(self): + assert workstream_map.is_shared_repo(None) is False + + def test_empty_string_returns_false(self): + assert workstream_map.is_shared_repo("") is False + + +class TestIsExcludedRepo(unittest.TestCase): + def test_personal_namespaces_are_excluded(self): + assert workstream_map.is_excluded_repo("jeff-roche/roundhouse") is True + assert workstream_map.is_excluded_repo("jaypoulz/edge-tooling") is True + assert workstream_map.is_excluded_repo("eggfoobar/two-node-toolbox") is True + + def test_openshift_org_not_excluded(self): + assert workstream_map.is_excluded_repo("openshift/foo") is False + + def test_openshift_eng_org_not_excluded(self): + assert workstream_map.is_excluded_repo("openshift-eng/bar") is False + + def test_microshift_io_org_not_excluded(self): + assert workstream_map.is_excluded_repo("microshift-io/baz") is False + + def test_openshift_metal3_org_not_excluded(self): + assert workstream_map.is_excluded_repo("openshift-metal3/qux") is False + + def test_metal3_io_org_not_excluded(self): + assert workstream_map.is_excluded_repo("metal3-io/quux") is False + + def test_containers_org_not_excluded(self): + assert workstream_map.is_excluded_repo("containers/podman") is False + + def test_clusterlabs_org_not_excluded(self): + assert workstream_map.is_excluded_repo("clusterlabs/pacemaker") is False + + def test_ovn_kubernetes_org_not_excluded(self): + assert workstream_map.is_excluded_repo("ovn-kubernetes/ovn") is False + + def test_opendatahub_io_org_not_excluded(self): + assert workstream_map.is_excluded_repo("opendatahub-io/opendatahub") is False + + def test_kubevirt_org_not_excluded(self): + # Real org, not a personal namespace. Team members carry OCPBUGS work + # into hyperconverged-cluster-operator; dropping it loses real PRs. + assert workstream_map.is_excluded_repo("kubevirt/hyperconverged-cluster-operator") is False + + def test_none_returns_false(self): + assert workstream_map.is_excluded_repo(None) is False + + def test_empty_string_returns_false(self): + assert workstream_map.is_excluded_repo("") is False + + def test_malformed_repo_no_slash_returns_false(self): + assert workstream_map.is_excluded_repo("noslash") is False + + +class TestIndexSanity(unittest.TestCase): + """Sanity checks on the index constants to catch configuration errors.""" + + def test_no_overlap_between_repo_index_and_shared_repos(self): + """A repo cannot be both workstream-specific and shared.""" + repo_index_repos = {repo.lower() for repo in workstream_map._REPO_INDEX.keys()} + shared_repos = {repo.lower() for repo in workstream_map._SHARED_REPOS} + overlap = repo_index_repos & shared_repos + assert len(overlap) == 0, f"Repos in both _REPO_INDEX and _SHARED_REPOS: {overlap}" + + def test_all_indexed_repos_are_allowed_orgs(self): + """Every repo in _REPO_INDEX and _SHARED_REPOS must be from an allowed org.""" + all_repos = set(workstream_map._REPO_INDEX.keys()) | workstream_map._SHARED_REPOS + for repo in all_repos: + owner = repo.split("/")[0].lower() + assert owner in workstream_map._ALLOWED_ORGS, \ + f"Repo {repo} has owner {owner} not in _ALLOWED_ORGS" + + +if __name__ == "__main__": + unittest.main() diff --git a/plugins/edge-contribution/bin/workstream_map.py b/plugins/edge-contribution/bin/workstream_map.py new file mode 100644 index 00000000..eefe867f --- /dev/null +++ b/plugins/edge-contribution/bin/workstream_map.py @@ -0,0 +1,232 @@ +#!/usr/bin/env python3 +"""Canonical OpenShift Edge workstream -> Jira component map. + +This is the single in-repo source of truth for the six workstreams and the Jira +OCPEDGE components that belong to each. It is intentionally a code constant so the +plugin works without waiting on an edge-context documentation PR. + +The map is list-valued because live OCPEDGE components are not 1:1 with +workstreams: TNF owns both ``TNF`` and ``Two Node Fencing``; TOPO owns both +``Topology Transitions`` and ``Mutable Topology``. The ``Planning`` component is +not a workstream and is intentionally absent, so it resolves to ``None``. + +This module also handles attribution recovery for work that lacks Jira components: +- Jira project key -> workstream (e.g., the USHIFT project is MicroShift work) +- GitHub repo -> workstream (e.g., openshift/lvm-operator is LVMS work) +- Shared repos (cross-cutting tooling/docs) that belong to no single workstream +- Personal-namespace repo exclusion (side projects not representing team work) +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import Dict, FrozenSet, List, Optional, Tuple + + +@dataclass(frozen=True) +class Workstream: + """A long-lived stream of Edge engineering work and its Jira components.""" + + acronym: str + name: str + components: Tuple[str, ...] + + +# Order here defines the canonical column order used by metrics and renderers. +_WORKSTREAMS: Tuple[Workstream, ...] = ( + Workstream("SNO", "Single Node", ("SNO",)), + Workstream("TNA", "Two Node with Arbiter", ("Two Node with Arbiter",)), + Workstream("TNF", "Two Node Fencing", ("TNF", "Two Node Fencing")), + Workstream("LVMS", "Logical Volume Manager Storage", ("Logical Volume Manager Storage",)), + Workstream("USHIFT", "MicroShift", ("MicroShift",)), + Workstream("TOPO", "Topology Transitions", ("Topology Transitions", "Mutable Topology")), +) + +# Jira components that exist in OCPEDGE but are deliberately not workstreams. +# Kept for documentation; they resolve to None like any other unmapped component. +NON_WORKSTREAM_COMPONENTS: Tuple[str, ...] = ("Planning",) + +# The SHARED pseudo-column for cross-cutting work that belongs to no single workstream. +SHARED_COLUMN: str = "SHARED" + +# Jira project key -> workstream acronym. Seed with projects whose tickets are +# unambiguously attributable even without component metadata. +# The USHIFT Jira project's display name is literally "MicroShift" (verified +# against redhat.atlassian.net; e.g., USHIFT-6337, USHIFT-6685), so a ticket in +# that project is MicroShift work by definition even with no component. +_PROJECT_INDEX: Dict[str, str] = { + "USHIFT": "USHIFT", +} + +# GitHub repo (owner/name) -> workstream acronym. Only repos that unambiguously +# belong to a single workstream are mapped here. +# IMPORTANT: openshift-eng/two-node-toolbox is DELIBERATELY ABSENT. That repo +# covers BOTH arbiter (TNA) and fencing (TNF) topologies, so mapping it either +# way would mis-credit work. This was an explicit product decision, not an +# oversight — do not add it. +_REPO_INDEX: Dict[str, str] = { + "openshift/lvm-operator": "LVMS", # LVMS operator core + "openshift/topolvm": "LVMS", # LVMS storage provisioner + "openshift/microshift": "USHIFT", # MicroShift upstream + "microshift-io/microshift": "USHIFT", # MicroShift community fork + "openshift/oc-tnf": "TNF", # Two-Node Fencing test suite +} + +# Repos that are cross-cutting (CI/tooling/docs) and belong to no single workstream. +_SHARED_REPOS: FrozenSet[str] = frozenset([ + "openshift/release", + "openshift-eng/edge-tooling", + "openshift-eng/edge-context", + "openshift/openshift-docs", + "openshift/enhancements", +]) + +# GitHub org allowlist for identifying team repos. Repos whose owner is NOT in +# this set are considered personal namespaces (side projects) and are excluded +# from attribution. This is an allowlist rather than a denylist of people because +# new personal forks appear constantly, while new orgs are rare. +_ALLOWED_ORGS: FrozenSet[str] = frozenset([ + "openshift", + "openshift-eng", + "microshift-io", + "openshift-metal3", + "metal3-io", + "containers", + "clusterlabs", + "ovn-kubernetes", + "opendatahub-io", + "kubevirt", +]) + + +def _normalize(component_name: str) -> str: + return component_name.strip().lower() + + +_COMPONENT_INDEX: Dict[str, str] = { + _normalize(component): stream.acronym + for stream in _WORKSTREAMS + for component in stream.components +} + + +def workstreams() -> Tuple[Workstream, ...]: + """Return the six workstreams in canonical order.""" + return _WORKSTREAMS + + +def workstream_acronyms() -> List[str]: + """Return the six workstream acronyms in canonical order.""" + return [stream.acronym for stream in _WORKSTREAMS] + + +def component_to_workstream(component_name: Optional[str]) -> Optional[str]: + """Map a Jira component name to a workstream acronym. + + Returns ``None`` for a genuine miss: unknown components, non-workstream + components such as ``Planning``, and empty/``None`` input. Lookup is + case-insensitive and ignores surrounding whitespace. + """ + if component_name is None: + return None + key = _normalize(component_name) + if not key: + return None + return _COMPONENT_INDEX.get(key) + + +def display_columns() -> List[str]: + """Return the six workstream acronyms plus SHARED, in canonical order. + + This is the full column list for the heatmap display. The SHARED column + captures cross-cutting work (CI/tooling/docs) that belongs to no single + workstream. Returns a NEW list each call so callers cannot mutate module state. + """ + return workstream_acronyms() + [SHARED_COLUMN] + + +def project_to_workstream(issue_key: Optional[str]) -> Optional[str]: + """Map a Jira issue key to a workstream acronym via its project prefix. + + Extracts the project key from a full issue identifier (e.g., "USHIFT-6337" + -> "USHIFT") and looks it up in the project index. Returns ``None`` for + ``None``, empty string, malformed keys (no "-"), or unknown projects. + Lookup is case-insensitive and whitespace-tolerant. + + Args: + issue_key: Full Jira issue key like "USHIFT-6337" or "OCPEDGE-123" + + Returns: + Workstream acronym if the project is mapped, else ``None`` + """ + if issue_key is None: + return None + normalized = issue_key.strip().upper() + if not normalized or "-" not in normalized: + return None + project_key = normalized.split("-", 1)[0] + return _PROJECT_INDEX.get(project_key) + + +def repo_to_workstream(repo: Optional[str]) -> Optional[str]: + """Map a GitHub repo (owner/name) to a workstream acronym. + + Only repos that unambiguously belong to a single workstream are mapped. + Returns ``None`` for ``None``, empty string, or unknown repos. Lookup is + case-insensitive and whitespace-tolerant. + + Args: + repo: GitHub repository in "owner/name" format + + Returns: + Workstream acronym if the repo is mapped, else ``None`` + """ + if repo is None: + return None + normalized = repo.strip().lower() + if not normalized: + return None + return _REPO_INDEX.get(normalized) + + +def is_shared_repo(repo: Optional[str]) -> bool: + """Check if a GitHub repo is cross-cutting (CI/tooling/docs). + + Shared repos belong to no single workstream and are credited to the SHARED + column instead. Lookup is case-insensitive. + + Args: + repo: GitHub repository in "owner/name" format + + Returns: + ``True`` if the repo is shared, ``False`` otherwise + """ + if repo is None: + return False + normalized = repo.strip().lower() + if not normalized: + return False + return normalized in _SHARED_REPOS + + +def is_excluded_repo(repo: Optional[str]) -> bool: + """Check if a GitHub repo is a personal-namespace side project. + + Repos whose owner is NOT in the org allowlist are considered personal + namespaces and are excluded from attribution. Returns ``False`` for + ``None``, empty string, or malformed input (no "/") to avoid excluding + something we cannot parse. Org comparison is case-insensitive. + + Args: + repo: GitHub repository in "owner/name" format + + Returns: + ``True`` if the repo should be excluded, ``False`` otherwise + """ + if repo is None: + return False + normalized = repo.strip().lower() + if not normalized or "/" not in normalized: + return False + owner = normalized.split("/", 1)[0] + return owner not in _ALLOWED_ORGS diff --git a/plugins/edge-contribution/pyproject.toml b/plugins/edge-contribution/pyproject.toml new file mode 100644 index 00000000..b169f5d0 --- /dev/null +++ b/plugins/edge-contribution/pyproject.toml @@ -0,0 +1,26 @@ +# Tooling config for the edge-contribution plugin's Python code (bin/). +# The plugin ships no installable package; this file only configures the +# quality gate (pytest + black + ruff + mypy). Run the tools from this dir. + +[tool.pytest.ini_options] +testpaths = ["bin/tests"] +python_files = ["test_*.py"] + +[tool.black] +line-length = 100 +target-version = ["py39"] + +[tool.ruff] +line-length = 100 +target-version = "py39" + +[tool.ruff.lint] +# pycodestyle, pyflakes, isort, pep8-naming, flake8-bugbear, comprehensions, pyupgrade. +select = ["E", "F", "I", "N", "B", "C4", "UP"] + +[tool.mypy] +python_version = "3.9" +warn_unused_ignores = true +disallow_untyped_defs = true +warn_return_any = true +no_implicit_optional = true diff --git a/plugins/edge-contribution/references/edge-context.md b/plugins/edge-contribution/references/edge-context.md new file mode 100644 index 00000000..38e8d2c3 --- /dev/null +++ b/plugins/edge-contribution/references/edge-context.md @@ -0,0 +1,72 @@ +# edge-context inputs + +The plugin reads its **roster** from the `openshift-eng/edge-context` repo. The +**workstream→Jira-component map is internal** to the plugin +(`bin/workstream_map.py`), so the plugin is not blocked on any edge-context PR. + +## Fetching via `gh` (no local checkout) + +`load_context.py` reads the roster file's contents straight from GitHub through +the authenticated `gh` CLI — there is no checkout to keep in sync: + +```bash +gh api repos/openshift-eng/edge-context/contents/people/team-roster.md +``` + +The response is the GitHub contents payload (`{"content": "", +"encoding": "base64", ...}`); the script base64-decodes `content` to the raw +markdown. `--repo owner/name` overrides the repository and `--ref ` +pins a git ref (default: the repository's default branch). Any `gh` failure — +missing auth, no read access, a 404, or a network error — surfaces as a +`ContextParseError`; the roster is never silently empty. + +The command runner is injected, so parsing is unit-tested against canned +contents payloads with no subprocess or network (see +`bin/tests/test_load_context.py`). + +## Roster: `people/team-roster.md` + +A GitHub-flavored markdown pipe table. `load_context.py` parses these columns: + +| Column | Used for | +|----------|-----------------------------------------------------------------| +| Name | `name`, and the Rover profile link that yields `kerberos` | +| GitHub | `github` handle (used by the GitHub PR collector) | +| Role | the Eng/QE filter (see below) | +| Location | ignored | + +Derived identity: + +- `kerberos` comes from the Rover profile URL in the Name cell + (`…/people/profile/`). +- `jira_username = @redhat.com`. + +### Eng/QE filter + +Only engineers and quality engineers are included. A member is kept when their +role (lower-cased) contains **"software engineer"** or **"software quality +engineer"**. This deliberately includes titles like "Associate Software +Engineer" and "Senior Software Quality Engineer", and excludes +"Manager, Engineering", "Project Manager - Technical", and +"Principal Product Security Engineer" (they contain "Engineer"/"Engineering" but +not the qualifying phrase). See `bin/tests/test_load_context.py`. + +## Workstreams (internal, not parsed from edge-context) + +The six workstreams and their Jira component aliases are the canonical constant in +`bin/workstream_map.py`: + +| Workstream | Jira component(s) | +|------------|--------------------------------------------| +| SNO | `SNO` | +| TNA | `Two Node with Arbiter` | +| TNF | `TNF`, `Two Node Fencing` | +| LVMS | `Logical Volume Manager Storage` | +| USHIFT | `MicroShift` | +| TOPO | `Topology Transitions`, `Mutable Topology` | +| _(excluded)_ | `Planning` — cutline epics, not a workstream | + +Keeping this map in code (rather than parsing `about/workstreams.md`) is a +deliberate decision so development isn't gated on doc-repo review. Mirroring the +map into edge-context docs is a non-blocking follow-up; if the two ever diverge, +`bin/workstream_map.py` is the source of truth at runtime. diff --git a/plugins/edge-contribution/references/metrics.md b/plugins/edge-contribution/references/metrics.md new file mode 100644 index 00000000..ab2efb25 --- /dev/null +++ b/plugins/edge-contribution/references/metrics.md @@ -0,0 +1,79 @@ +# Contribution metrics + +Every metric name below says what it counts. Nothing here is a coined term you +have to look up. + +The counts derive from a single **binary contribution matrix** `M`, where +`M[member][workstream] = 1` iff the member has at least one activity item +attributed to that workstream during the reporting window, and `0` otherwise. + +There are six canonical workstreams (see `bin/workstream_map.py`): **SNO, TNA, +TNF, LVMS, USHIFT, TOPO**. Workstreams touched is always reported "of 6". + +An activity item is attributed to a workstream by mapping its Jira component(s) +through `workstream_map.component_to_workstream`. Items that map to nothing (an +unknown component, or the `Planning` cutline component) are **unattributed**: +they are counted and reported separately, but never placed in the matrix. + +## Workstreams touched (per member) + +The number of distinct workstreams a member touched: + +```text +workstreams_touched(m) = sum over w of M[m][w] +``` + +Reported as "N of 6". + +## Total touches (team) + +The number of 1s in the matrix — equivalently, the sum of every member's +workstreams-touched count: + +```text +total_touches = sum over m, w of M[m][w] +``` + +This is **not** the number of activity records. That is the sum of the weighted +`counts` matrix, reported as `total_by_member` in the allocation signals. + +## Mean people per workstream (team scalar) + +The average number of distinct contributors a canonical workstream had: + +```text +mean_people_per_workstream = total_touches / 6 +``` + +A high value means the team is broadly cross-trained; a low value flags +workstreams with thin coverage. + +The per-workstream contributor count itself lives in the allocation signals as +`contributors_by_workstream`, which is computed from the weighted `counts` +matrix and also covers SHARED. It is exported as `contributors:` in +`metrics.csv`. There is deliberately only one per-workstream contributor count; +an earlier second name for it was removed. + +## Mean workstreams per person (team scalar) + +The average workstreams-touched count over **active** members (those with at +least one contribution), reported alongside the active/total roster count: + +```text +mean_workstreams_per_person = total_touches / active_count (0.0 when active_count == 0) +``` + +Averaging over active members answers "when someone contributes, across how many +workstreams do they spread?" without inactive members deflating the number. + +## Edge-case contract + +- **Empty roster / no workstreams** → both team scalars are `0.0` (no division by + zero). +- **All-zero matrix** → `active_count == 0` and both team scalars are `0.0`. +- **Single active member** → `mean_workstreams_per_person == + workstreams_touched(that member)`. +- **Zero-contributor workstream** → still present in + `contributors_by_workstream` as `0`, in canonical order. + +These are enforced by `bin/tests/test_metrics.py`. diff --git a/plugins/edge-contribution/references/pipeline.md b/plugins/edge-contribution/references/pipeline.md new file mode 100644 index 00000000..64976a66 --- /dev/null +++ b/plugins/edge-contribution/references/pipeline.md @@ -0,0 +1,198 @@ +# Manager Skill Pipeline — Shared Setup + +All three manager skills (`heatmap`, `summary`, `export`) run the same four-step +setup: period resolution, workdir, roster load, and activity collection. Those +steps live here so the skills cannot drift apart. + +`PLUGIN_ROOT` is `${CLAUDE_PLUGIN_ROOT}` (this plugin's directory). + +## 1. Resolve the period + +`PERIOD` = the `--quarter` value or `_` (used for labels and the +workdir name). `` = `--quarter $QUARTER` or `--from $FROM --to $TO`, +passed to both collectors. + +If the skill accepts `--refresh`, read that flag too — you'll need it in step 4. + +## 2. Set up the workdir + +```bash +WORKDIR="/tmp/edge-contribution-$PERIOD" +mkdir -p "$WORKDIR" +``` + +## 3. Load the full Eng/QE roster + +Fetched from GitHub via `gh`, no checkout needed: + +```bash +python3 "$PLUGIN_ROOT/bin/load_context.py" --output "$WORKDIR/roster.json" +``` + +## 4. Collect activity (Jira + GitHub) for the whole roster + +**Before you run the collectors**, check if both activity files already exist and +the user did NOT pass `--refresh`: + +```bash +if [[ -f "$WORKDIR/jira_activity.json" && -f "$WORKDIR/github_activity.json" && -z "$REFRESH" ]]; then + # Files exist and no --refresh, reuse them + JIRA_AGE=$(( $(date +%s) - $(stat -f %m "$WORKDIR/jira_activity.json" 2>/dev/null || stat -c %Y "$WORKDIR/jira_activity.json") )) + GITHUB_AGE=$(( $(date +%s) - $(stat -f %m "$WORKDIR/github_activity.json" 2>/dev/null || stat -c %Y "$WORKDIR/github_activity.json") )) + echo "Reusing cached activity files (Jira: ${JIRA_AGE}s old, GitHub: ${GITHUB_AGE}s old)" +fi +``` + +If files exist and `--refresh` was not set, **skip to step 5** (the next skill step). + +Otherwise, collect both sources: + +### 4a. Collect Jira activity via MCP + +Jira activity is collected through the `mcp__mcp-atlassian__jira_search` MCP tool, +then attributed locally via Python. This replaces direct REST API calls with +Claude-native MCP tools. + +**Extract roster members:** + +```bash +MEMBERS=$(python3 -c " +import json +roster = json.load(open('$WORKDIR/roster.json')) +print(' '.join([m['jira_username'] for m in roster])) +") +``` + +**Resolve period dates** from ``: + +```bash +if [[ -n "$QUARTER" ]]; then + PERIOD_START=$(python3 -c "from _common import resolve_window; w = resolve_window('$QUARTER', None, None); print(w.start.isoformat())") + PERIOD_END=$(python3 -c "from _common import resolve_window; w = resolve_window('$QUARTER', None, None); print(w.end.isoformat())") +else + PERIOD_START="$FROM" + PERIOD_END="$TO" +fi +``` + +**Query Jira for each member** (3 queries per member): + +For each member in `$MEMBERS`, execute these three MCP queries **in parallel**: + +1. **Assignee activity:** + - JQL: `project in (OCPEDGE, USHIFT, OCPBUGS) AND assignee = "{member}" AND updated >= "{PERIOD_START}" AND updated <= "{PERIOD_END}"` + - Fields: `key,components,updated,summary,parent` + - Tag each result with: `member={member}`, `kind=assignee` + +2. **QA Contact activity (cf[10470]):** + - JQL: `project in (OCPEDGE, USHIFT, OCPBUGS) AND cf[10470] = "{member}" AND updated >= "{PERIOD_START}" AND updated <= "{PERIOD_END}"` + - Fields: `key,components,updated,summary,parent` + - Tag each result with: `member={member}`, `kind=qa` + +3. **OCPSTRAT SME/Assignee activity (cf[10475]):** + - JQL: `project = OCPSTRAT AND (cf[10475] = "{member}" OR assignee = "{member}") AND updated >= "{PERIOD_START}" AND updated <= "{PERIOD_END}"` + - Fields: `key,components,updated,summary,parent` + - Tag each result with: `member={member}`, `kind=ocpstrat_role` + +**Pagination:** Each MCP query may return a `next_page_token`. If present, repeat +the query with `page_token` set to continue fetching. Loop until no token is +returned. + +**Accumulate all results** into a list of issue records: + +```json +[ + { + "member": "jcope@redhat.com", + "kind": "assignee", + "fields": { + "key": "OCPEDGE-1234", + "components": [...], + "updated": "2026-08-15T10:30:00Z", + "summary": "...", + "parent": {"key": "OCPEDGE-999"} + } + }, + ... +] +``` + +**Extract unique parent keys:** + +```bash +PARENT_KEYS=$(python3 -c " +import json +issues = json.load(open('$WORKDIR/raw_issues_temp.json')) +parent_keys = list(set([ + issue['fields']['parent']['key'] + for issue in issues + if issue.get('fields', {}).get('parent', {}).get('key') +])) +print(','.join(parent_keys)) +") +``` + +**Resolve parent components** (batched, 100 keys per query): + +Split `$PARENT_KEYS` into batches of 100. For each batch: + +- JQL: `key in (OCPEDGE-1234,OCPEDGE-1235,...)` +- Fields: `key,components` + +For each returned parent issue, map its key to the **first component** that +resolves to a workstream (via `component_to_workstream()` from `workstream_map.py`). +If no component maps, store `null` for that parent key. + +Build a map: `{parent_key: workstream_or_null}` + +**Write intermediate JSON:** + +```bash +python3 -c " +import json +data = { + 'issues': json.load(open('$WORKDIR/raw_issues_temp.json')), + 'parent_workstreams': json.load(open('$WORKDIR/parent_map_temp.json')), + 'base_url': 'https://redhat.atlassian.net' +} +json.dump(data, open('$WORKDIR/raw_jira_issues.json', 'w'), indent=2) +" +``` + +**Apply attribution:** + +```bash +python3 "$PLUGIN_ROOT/bin/attribute_jira_activities.py" \ + --raw-issues "$WORKDIR/raw_jira_issues.json" \ + --output "$WORKDIR/jira_activity.json" +``` + +This produces `jira_activity.json` in the same format as the legacy REST collector. + +**Error handling:** + +- **Auth error (401/403):** Display auth error, suggest checking `JIRA_USERNAME` and `JIRA_API_TOKEN`, stop. +- **Transient error (429/500/503):** Retry once, then skip that member with a warning. +- **Member with no results:** Valid — all 3 queries can return empty. Continue. + +### 4b. Collect GitHub activity via Python script + +GitHub activity is collected via the existing Python script (uses GitHub GraphQL API): + +```bash +python3 "$PLUGIN_ROOT/bin/collect_github.py" --members-file "$WORKDIR/roster.json" \ + --output "$WORKDIR/github_activity.json" +``` + +**Note:** GitHub collection is unchanged and continues to use the Python REST client. + +--- + +Collection takes minutes, and a manager will typically run all three skills over +the same period back to back. Reusing the cache saves ~5 minutes per skill after +the first. + +## What comes next + +Each skill runs its own report or export command here. See the individual skill +docs for step 5. diff --git a/plugins/edge-contribution/requirements-dev.txt b/plugins/edge-contribution/requirements-dev.txt new file mode 100644 index 00000000..31e49b5d --- /dev/null +++ b/plugins/edge-contribution/requirements-dev.txt @@ -0,0 +1,12 @@ +# Development + CI gate for the edge-contribution plugin. +# Runtime needs only the Python stdlib plus `requests` (already present system-wide) +# and the `gh` CLI. These are the tools that gate a merge: +# python -m pytest bin/tests +# black --check bin +# ruff check bin +# mypy bin +pytest>=7 +black>=24 +ruff>=0.4 +mypy>=1.8 +requests>=2.25 diff --git a/plugins/edge-contribution/skills/export/SKILL.md b/plugins/edge-contribution/skills/export/SKILL.md new file mode 100644 index 00000000..ab58cbe8 --- /dev/null +++ b/plugins/edge-contribution/skills/export/SKILL.md @@ -0,0 +1,95 @@ +--- +name: export +description: Export OCP-Edge cross-workstream contribution data for a quarter or date range as four CSV files — raw activities, member×workstream matrix, team metrics, and data quality counts. +allowed-tools: + - Bash + - Read + - Write + - AskUserQuestion + - mcp__mcp-atlassian__jira_search +user-invocable: true +--- + +# Manager Data Export + +Export the OCP-Edge Eng/QE team's cross-workstream contribution data as four CSV +files for external analysis, pivot tables, or archival. The export covers the +same data the `heatmap` and `summary` skills render, but in a structured tabular +format for tools like Excel, R, or Python. + +## Output files + +All four files are written to `--output-dir` (defaults to +`/tmp/edge-contribution-$PERIOD/export-$PERIOD`): + +| File | Contents | +|------|----------| +| `activities.csv` | One row per activity record — the raw collated data. Columns: `member`, `source` (jira/github), `kind`, `workstream`, `attribution_source`, `repo`, `issue_key`, `url`, `ts`, `unattributed_reason` | +| `matrix.csv` | People × workstream counts. Columns: `member`, `SNO`, `TNA`, `TNF`, `LVMS`, `USHIFT`, `TOPO`, `SHARED`, `TOTAL`, `workstreams_touched`, `over`, `under`, `narrow` | +| `metrics.csv` | Team scalars and per-workstream totals/contributors. Columns: `metric`, `value` | +| `data-quality.csv` | Exclusion and unattributed counts. Columns: `category`, `label`, `count` | + +## Prerequisites + +Same as `heatmap` — see the plugin README for Jira credentials, `gh` auth, and +Python requirements. + +## Arguments + +- `--quarter ` — e.g. `2026Q2`. Or use an explicit range: +- `--from --to `. +- `--output-dir ` — where to write the four CSV files. Defaults to + `/tmp/edge-contribution-$PERIOD/export-$PERIOD`. +- `--refresh` — re-collect activity even if cached files exist. + +If neither `--quarter` nor a `--from/--to` pair is given, ask the user which +period to report on with AskUserQuestion. + +## Steps + +1. **Follow the shared pipeline** (steps 1-4 in + [`references/pipeline.md`](../../references/pipeline.md)). Parse the + `--refresh` flag if the user supplied it. + +2. **Set the output directory:** + + ```bash + OUTPUT_DIR="${OUTPUT_DIR:-$WORKDIR/export-$PERIOD}" + ``` + +3. **Export the data:** + + ```bash + python3 "$PLUGIN_ROOT/bin/export_csv.py" \ + --members-file "$WORKDIR/roster.json" \ + --activity "$WORKDIR/jira_activity.json" \ + --activity "$WORKDIR/github_activity.json" \ + --period "$PERIOD" --output-dir "$OUTPUT_DIR" + ``` + +4. **Present the result — the four file paths and what's in them, nothing else.** + + List the four paths (`activities.csv`, `matrix.csv`, `metrics.csv`, + `data-quality.csv`) with a one-line description per file, exactly as + documented in the "Output files" table above. + + **Do not add anything.** No commentary on the numbers, no reading off totals, + no notes about which period was used. The user asked for the data; give them + the paths. Answer follow-up questions if asked; volunteer nothing. + +## See also + +- **`heatmap`** — the grid alone, instantly. +- **`summary`** — the executive summary with team-level means, and (with + `--show-sources`) where every number came from. + +## Where the data comes from + +**Do NOT filter `activities.csv` by the `ts` column to isolate a quarter.** GitHub +PR review collection uses `updated:` rather than `created:`, +because the GitHub search API has no review-date qualifier. Every GitHub record +is timestamped with the PR's creation date (authored and reviewed alike), so a +review done in-window of an older PR carries a pre-window timestamp. In live +2026Q3, 145 of 1284 GitHub records (11%) have a `ts` outside the quarter. They +are real in-window contributions and ARE counted. Filtering by `ts` drops those +rows incorrectly. diff --git a/plugins/edge-contribution/skills/heatmap/SKILL.md b/plugins/edge-contribution/skills/heatmap/SKILL.md new file mode 100644 index 00000000..d6d083fa --- /dev/null +++ b/plugins/edge-contribution/skills/heatmap/SKILL.md @@ -0,0 +1,113 @@ +--- +name: heatmap +description: Show the whole OCP-Edge Eng/QE team's cross-workstream contribution as a terminal allocation view — a people×workstream grid shaded by contribution amount (grey→green, log scale) with allocation signals (over/under/narrow flags) — for a quarter or date range. For team-level means and data quality, see summary --show-sources; for those means as raw numbers, see export. +allowed-tools: + - Bash + - Read + - Write + - AskUserQuestion + - mcp__mcp-atlassian__jira_search +user-invocable: true +--- + +# Manager Team Heatmap + +Report how the OCP-Edge Eng/QE team spread across the six workstreams (SNO, TNA, +TNF, LVMS, USHIFT, TOPO) plus cross-cutting work (SHARED) in a reporting period. +Output is a terminal **allocation view** — a people×workstream grid where shade +(grey→green, log scale) encodes contribution amount. + +The output is the grid and its two legends. For the team-level means and the +data-quality block, use `/edge-contribution:summary --show-sources`, which +reports them as *people per workstream* and *workstreams per person*. The same +two numbers are exported as `mean_people_per_workstream` and +`mean_workstreams_per_person` by `/edge-contribution:export` (`metrics.csv`). + +## Prerequisites + +- `JIRA_USERNAME` and `JIRA_API_TOKEN` exported (same credentials the Atlassian + MCP uses). The collectors hit `redhat.atlassian.net` REST directly. +- `gh` authenticated (`gh auth status`) with read access to + `openshift-eng/edge-context` (the roster is fetched via `gh`) and for PR + collection. +- Python 3.9+ with `requests` available. + +## Arguments + +- `--quarter ` — e.g. `2026Q2`. Or use an explicit range: +- `--from --to `. +- `--ascii` — use ASCII-only glyphs (`. : * + #`) instead of Unicode block + chars; flag glyphs become `^` and `v`. +- `--no-color` — disable ANSI color codes (text format only). + +If neither `--quarter` nor a `--from/--to` pair is given, ask the user which +period to report on with AskUserQuestion. + +## Steps + +1. **Follow the shared pipeline** (steps 1-4 in + [`references/pipeline.md`](../../references/pipeline.md)). The heatmap skill + does not use `--refresh`, so you can skip parsing that flag. + +2. **Render the allocation view.** + + `PLUGIN_ROOT` is `${CLAUDE_PLUGIN_ROOT}` (this plugin's directory). + + Render to stdout with `--no-color`. The grid goes to stdout; the `--output` + flag no longer exists. + + ```bash + python3 "$PLUGIN_ROOT/bin/report.py" --view manager \ + --members-file "$WORKDIR/roster.json" \ + --activity "$WORKDIR/jira_activity.json" \ + --activity "$WORKDIR/github_activity.json" \ + --period "$PERIOD" --format text --no-color + ``` + + Pass any `--ascii` flags the user supplied. + +3. **Present the result — the rendered output and nothing else.** + + Paste `report.py`'s stdout verbatim in a fenced code block. That is the grid, + the scale legend, and the flag legend — `_manager_text` stops there by + design. + + **Do not add anything.** No "what to look at" section, no reading off the + ▲/▼/narrow flags, no scores, no unattributed count, no data-quality notes, + no commentary on who is over-allocated or which workstreams look thin, no + note about which period or data was used. The grid says all of it, and a + prose retelling invites the model to editorialize about named people. The + user reads the grid. Answer follow-up questions if asked; volunteer nothing. + +## See also + +- **`summary`** — the executive summary with team-level means, per-member + totals, data quality/exclusions, and (with `--show-sources`) where every + number came from. +- **`export`** — the underlying data as four CSV files for external analysis, + including `mean_people_per_workstream` and `mean_workstreams_per_person` in + `metrics.csv`. + +## Notes + +- Contribution counts are shaded on a log scale (grey→green: `· ░ ▒ ▓ █` or + ASCII `. : * + #`) so the full range (1 to grid_max) is perceptually balanced. +- Allocation signals (▲/▼/narrow) are relative to the team median, not external + capacity targets. Members with zero activity are excluded from the median. +- The **SHARED** column (cross-cutting CI/tooling/docs work: `openshift/release`, + `openshift-eng/edge-tooling`, `openshift-eng/edge-context`, + `openshift/openshift-docs`, `openshift/enhancements`) contributes to a member's + total workload but is deliberately excluded from the workstreams-touched + count — that count stays "N of 6". +- Jira activity is collected from `OCPEDGE`, `USHIFT`, and `OCPBUGS` (assignee + and QA contact), plus `OCPSTRAT` roles. Comments are **not** collected — Jira + Cloud hides comment author emails, so the match could never fire and the + query scanned every issue in the project once per member for zero records. +- The canonical six workstreams and their Jira component/GitHub repo mappings + live in `bin/workstream_map.py`. Attribution uses a first-hit-wins chain: + - Jira: own component → parent epic's component → project key + - GitHub: Jira keys from PR title/body → repo → SHARED repos +- For data quality and exclusion details, see the `summary` skill. +- Activity files are `{"activities": [...], "collector_meta": {...}}`. A bare + JSON list is the older format and still loads, but its `collector_meta` + counters read as 0 — re-collect if you need them. diff --git a/plugins/edge-contribution/skills/summary/SKILL.md b/plugins/edge-contribution/skills/summary/SKILL.md new file mode 100644 index 00000000..fb26096b --- /dev/null +++ b/plugins/edge-contribution/skills/summary/SKILL.md @@ -0,0 +1,121 @@ +--- +name: summary +description: Executive summary of OCP-Edge cross-workstream contribution for a quarter or date range — team activity, per-workstream volume, per-member totals, and data quality — in text or markdown format. Reports raw measurements with no judgments. +allowed-tools: + - Bash + - Read + - Write + - AskUserQuestion + - mcp__mcp-atlassian__jira_search +user-invocable: true +--- + +# Manager Executive Summary + +An executive summary of how the OCP-Edge Eng/QE team contributed across the six +workstreams (SNO, TNA, TNF, LVMS, USHIFT, TOPO) plus cross-cutting work (SHARED) +in a reporting period. The summary reports raw measurements with no judgments — +no "over", "under", "narrow" flags, no risk labels. The manager applies their own +judgment. The default output contains four sections: + +- **TEAM** — total contributions, median per person, people per workstream (mean), + workstreams per person (active members only). +- **WORKSTREAMS** — volume, contributor count, and largest contributor with their + percentage share, sorted by volume descending. +- **PEOPLE** — total contributions, workstreams touched (N of 6), and largest + workstream with percentage share, sorted by total descending. +- **DATA HEALTH** — unattributed record count and breakdown by category (Jira + tickets vs PRs, unmapped repos vs missing components). + +Add `--show-sources` to include three additional sections describing where the +numbers came from: what was counted (sources, projects, collection method), how +each item was attributed (first-hit-wins chain with counts), and what was +excluded (managers, personal repos, full unattributed breakdowns). + +Output formats are `text` (default, a plain-text report) and `markdown` (tables +and headings for pasting into Slack, email, or a Google Doc). + +## Prerequisites + +Same as `heatmap` — see the plugin README for Jira credentials, `gh` auth, and +Python requirements. + +## Arguments + +- `--quarter ` — e.g. `2026Q2`. Or use an explicit range: +- `--from --to `. +- `--format ` — defaults to `text`. +- `--show-sources` — append the collection and attribution detail (what was + counted, how each item was attributed, full exclusion lists). +- `--include-all` — include excluded members (managers on the roster by role + title who are not IC contributors). +- `--refresh` — re-collect activity even if cached files exist. + +If neither `--quarter` nor a `--from/--to` pair is given, ask the user which +period to report on with AskUserQuestion. + +## Steps + +1. **Follow the shared pipeline** (steps 1-4 in + [`references/pipeline.md`](../../references/pipeline.md)). Parse the + `--refresh` flag if the user supplied it. + +2. **Render the executive summary:** + + ```bash + FORMAT="${FORMAT:-text}" # default to text if not set + SHOW_SOURCES_FLAG="" + if [[ "$SHOW_SOURCES" == "true" ]]; then + SHOW_SOURCES_FLAG="--show-sources" + fi + python3 "$PLUGIN_ROOT/bin/report.py" --view summary \ + --members-file "$WORKDIR/roster.json" \ + --activity "$WORKDIR/jira_activity.json" \ + --activity "$WORKDIR/github_activity.json" \ + --period "$PERIOD" --format "$FORMAT" $SHOW_SOURCES_FLAG + ``` + + Pass `--include-all` if the user supplied it. + +3. **Present the result — the rendered output and nothing else.** + + Paste `report.py`'s stdout verbatim in a fenced code block (for `text`) or as + markdown (for `markdown`). + + **Do not add anything.** No "what to look at" section, no reading a share + percentage back as a risk, no commentary on which workstreams look thin or + who is carrying too much, no note about which period or data was used. The + summary reports raw measurements — it deliberately carries no flags, labels, + or thresholds, and adding that judgement in prose puts back exactly what the + output was built to leave out. The user reads the summary. Answer follow-up + questions if asked; volunteer nothing. + +## See also + +- **`heatmap`** — the grid alone, instantly. +- **`export`** — the underlying data as four CSV files. + +## Source-disclosure sections + +The `--show-sources` flag appends three sections after DATA HEALTH: + +- **WHAT WAS COUNTED** — record totals by source (Jira/GitHub) and kind + (assignee/qa/pr_authored/pr_reviewed), with collection-method notes (which + projects, which orgs, the GitHub review-date caveat). +- **HOW EACH ITEM WAS ATTRIBUTED** — the first-hit-wins chain (component → repo + → project → shared → parent), with a count and percentage for each step and + the unattributed total. +- **WHAT WAS EXCLUDED** — what was deliberately left out (managers, + personal-namespace repos, unattributed items), with counts, repo names, and + specific examples. The full unattributed breakdown appears here, including the + `openshift-eng/two-node-toolbox` note ("mixed arbiter/fencing; not mapped by + design"). + +GitHub PR review collection uses `updated:` rather than +`created:`, because the GitHub search API has no review-date qualifier. +Every GitHub record is timestamped with the PR's creation date (authored and +reviewed alike), so a review done in-window of an older PR carries a pre-window +timestamp. In live 2026Q3, 145 of 1284 GitHub records (11%) have a `ts` outside +the quarter. They are real in-window contributions and ARE counted by the +summary and the grid. The WHAT WAS COUNTED section reports this count when +non-zero.