diff --git a/.github/scripts/filter_codeql_sarif.py b/.github/scripts/filter_codeql_sarif.py new file mode 100644 index 0000000..dc57365 --- /dev/null +++ b/.github/scripts/filter_codeql_sarif.py @@ -0,0 +1,42 @@ +"""Apply CodeQL source suppressions before uploading SARIF to GitHub.""" + +import argparse +import json +from pathlib import Path + + +# Removes accepted source suppressions while retaining all other results and analysis metadata +def filter_source_suppressions(report): + if report.get("version") != "2.1.0" or not isinstance(report.get("runs"), list) or not report["runs"]: + raise ValueError("Expected a SARIF 2.1.0 report with at least one run") + removed = 0 + for run in report["runs"]: + if run["tool"]["driver"]["name"] not in ("CodeQL", "CodeQL command-line toolchain") or "results" not in run: + continue + retained = [] + for result in run["results"]: + suppressed = any(suppression.get("kind") == "inSource" and suppression.get("status", "accepted") == "accepted" for suppression in (result.get("suppressions") or [])) + if suppressed: + removed += 1 + else: + retained.append(result) + run["results"] = retained + return removed + + +# Writes a filtered copy of one CodeQL report and reports how many source suppressions were applied +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("source", type=Path) + parser.add_argument("destination", type=Path) + args = parser.parse_args() + if args.source.resolve() == args.destination.resolve(): + parser.error("Source and destination must differ to preserve the original report") + report = json.loads(args.source.read_text(encoding="utf-8")) + removed = filter_source_suppressions(report) + args.destination.write_text(json.dumps(report) + "\n", encoding="utf-8") + print(f"Applied {removed} CodeQL source suppressions") + + +if __name__ == "__main__": + main() diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 51ff9db..6a8dcc7 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -37,10 +37,21 @@ jobs: build-mode: none # The extended suite adds security queries that the default suite leaves out queries: security-extended - # The suppression query turns the in-code codeql[...] comments into suppressed alerts + # Mark results covered by source suppression comments for the upload filter packs: codeql/python-queries:AlertSuppression.ql - name: Perform CodeQL analysis uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 with: category: "/language:python" + output: sarif-results + upload: failure-only + + - name: Apply source suppressions + run: python .github/scripts/filter_codeql_sarif.py sarif-results/python.sarif sarif-results/python-filtered.sarif + + - name: Upload CodeQL results + uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0 + with: + sarif_file: sarif-results/python-filtered.sarif + category: "/language:python" diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index e5db7ab..3342c4f 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -1,6 +1,7 @@ # Optional local hooks that catch what CI would reject, before a commit is written. # Enable them once with: # pip install pre-commit +# pip install -e '.[lint]' # pre-commit install # Check the whole repository at any time with: # pre-commit run --all-files @@ -24,10 +25,13 @@ repos: # where it needs no Go toolchain on a contributor's machine - id: detect-private-key - # The same linter and version CI runs - - repo: https://github.com/astral-sh/ruff-pre-commit - rev: v0.16.7 + # Runs the ruff already installed from the [lint] extra rather than a second copy pinned here. + # One pin in pyproject.toml is what keeps the hook and CI on the same version + - repo: local hooks: - id: ruff-check + name: ruff check + entry: ruff check args: [--no-fix] + language: system files: '^(github_monitor\.py|tests/.*\.py)$' diff --git a/CITATION.cff b/CITATION.cff index 49c5c3e..aedbc5d 100644 --- a/CITATION.cff +++ b/CITATION.cff @@ -10,8 +10,8 @@ authors: given-names: Michal alias: misiektoja email: misiektoja-github@rm-rf.ninja -version: "2.7" -date-released: 2026-09-18 +version: "2.8" +date-released: 2026-09-22 license: GPL-3.0-or-later repository-code: "https://github.com/misiektoja/github_monitor" keywords: diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 71bc818..c0d0fef 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -18,10 +18,11 @@ cd github_monitor pip install -e '.[test]' ``` -Optional local hooks catch what CI would reject before a commit is written: +Optional local hooks catch what CI would reject before a commit is written. The lint hook calls the Ruff installed by the `lint` extra rather than a copy of its own, so it always matches the version CI runs: ```sh pip install pre-commit +pip install -e '.[lint]' pre-commit install ``` @@ -41,6 +42,8 @@ mkdocs build --strict The default suite is offline. It never contacts GitHub and network calls are replaced with local test doubles. See [tests/README.md](tests/README.md) for what each test file covers. +CodeQL runs the extended security queries. For a verified false positive, put a `codeql[rule-id]` comment immediately above the reported line and explain why it is safe. The workflow filters results with accepted source suppressions before upload. Other findings remain reportable. + CI runs the same three checks on every push and pull request, across Python 3.10 through 3.14. The linter is pinned in the `lint` extra so a new ruff release cannot fail a build on a rule that did not exist when the change was written; the pre-commit hook pins the same version. A change to the monitoring loop, authentication or GitHub data handling is not verified by the offline suite alone. Exercise it against a real account and say so in the pull request, without usernames or credentials. diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 280287a..ef394be 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -2,6 +2,25 @@ This is a high-level summary of the most important changes. +# Changes in 2.8 (22 Sep 2026) + +Version **2.8** caps how much of a large push a single notification reports and gives every monitoring failure unified subject and body across email and webhook, followed by a **recovery alert** when monitoring resumes. Network failures now link to a new **Connection Problems** page section. Alert delivery messages stay within the correct check report and alert channels that still use placeholder configuration values are shown as not configured. + +**Features and improvements**: + +- **NEW:** **Push event commit limits** - A push with hundreds of commits no longer spends one API request per commit and no longer fills the notification. **`PUSH_COMMITS_LIMIT`** (default **10**, also **`--push-commits-limit`**) sets how many commits of one push are reported with their date, author URL, statistics and changed files. **`PUSH_COMMITS_ORDER`** picks which end of the push keeps them, **`newest`** by default. The remaining commits are replaced by one line naming how many were left out, or listed one line each with their SHA, author and first message line by setting **`PUSH_COMMITS_OVERFLOW = 'summary'`**. **`PUSH_FILES_LIMIT`** (default **20**, also **`--push-files-limit`**) caps the changed files listed per commit. Set either limit to **0** to report everything in full. +- **IMPROVE:** **Every profile change is sent as HTML** - The alerts about a changed **blog URL**, an **updated account**, a **profile visibility** switch and a **block** now carry an HTML part like the other alerts +- **IMPROVE:** **Failure alerts share one shape** - Every monitoring failure email and webhook uses the subject **`GitHub Monitor error: (user: )`** and lists the fix, the guide link, how many checks failed in a row, since when and when the next retry happens. A **recovery alert** follows on the channels that received the failure alert once monitoring resumes. `-e` / `--no-error-notify` and `--no-webhook-error-notify` switch both off + +**Bug fixes**: + +- **BUGFIX:** **An unset profile value reads the same in both email parts** - A bio, location, name, company, email or blog URL that was not set printed **`None`** in the plain text part while the HTML part left the line empty. Both parts now say the same thing +- **BUGFIX:** **Network failures point at the right page** - A timed-out or unreachable GitHub request ended with a **`Guide:`** now link to the new **Connection Problems** section, which explains the automatic retries and what to check if the failure continues. +- **BUGFIX:** **A failed visibility check no longer ends the run** - The profile visibility lookup only handled GitHub API errors, so a timed-out or blocked connection ended the monitoring run with a traceback. The lookup now survives any failure and reports that it could not answer +- **BUGFIX:** **Reported changes name the window they were observed in** - The **`Check interval:`** line under a change always showed the configured polling interval and a date range built from it, so a change found after a failed check was reported over a window the tool had not been watching. It now measures from the previous successful check, which is the configured interval while checks run on schedule +- **BUGFIX:** **Alert deliveries stay inside their report** - The hourly **`Monitoring degraded`** reminder closed its report before the error alert was sent, so **`Sending email notification to ...`** and its webhook equivalent landed under the separator and started a second, headless block. The reminder now closes below its delivery lines, keeping one check's report in one block +- **BUGFIX:** **Unset alert channels are reported as unset** - The verbose startup summary read the values the sample configuration ships as a real destination, so a run that had never been given a mail server printed **`Email transport: your_smtp_server_ssl:587`**, a recipient of **`your_receiver_email`** and a webhook provider of **`Discord`**. Those rows now read **`Not configured`** and the channel rollup above them reads **`Off (not configured)`** rather than naming alert types nothing could deliver + # Changes in 2.7 (18 Sep 2026) Version **2.7** adds **guided setup**, a read-only **Doctor preflight check** and **private SMTP password entry**. **Coloured output**, startup summaries and verbose/debug modes make monitoring easier to follow. It improves **contribution and repository-closure alerts**, preserves history during failed checks and protects configuration and credentials. Documentation is searchable and release downloads can be verified. diff --git a/docs/configuration.md b/docs/configuration.md index c9313f3..6f993cf 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -59,6 +59,31 @@ By default all events are monitored, but if you want to limit it, then remove th EVENTS_TO_MONITOR=['PushEvent', 'PullRequestEvent', 'IssuesEvent', 'ForkEvent', 'ReleaseEvent', 'DiscussionEvent'] ``` + +## Push Event Commits + +A push can carry hundreds of commits. Reporting every one in full costs an extra GitHub API request each and produces a notification nobody reads, so only `PUSH_COMMITS_LIMIT` commits of a push are reported with their date, author URL, statistics and changed files. The rest are replaced by one line naming how many were left out. + +```ini +PUSH_COMMITS_LIMIT = 10 +PUSH_COMMITS_ORDER = 'newest' +PUSH_COMMITS_OVERFLOW = 'count' +PUSH_FILES_LIMIT = 20 +``` + +| Option | Values | Effect | +| --- | --- | --- | +| `PUSH_COMMITS_LIMIT` | integer, `0` for no limit | Commits of one push reported in full, also `--push-commits-limit` | +| `PUSH_COMMITS_ORDER` | `'newest'`, `'oldest'` | Which end of the push keeps the detailed commits | +| `PUSH_COMMITS_OVERFLOW` | `'count'`, `'summary'` | Whether the remaining commits become a single count or get one line each | +| `PUSH_FILES_LIMIT` | integer, `0` for no limit | Changed files listed per commit, also `--push-files-limit` | + +With the built-in values a 300-commit push spends about 12 requests instead of about 300. It reports the 10 newest commits in full and replaces the other 290 with `Commits 1-290 not reported in full`. The compare URL in the same report links the complete diff. + +The limits in effect appear as `Push commit details` and `Push changed files` in the startup summary, which `--verbose` and `--debug` print on screen and every run writes to the log file. + +Set `PUSH_COMMITS_OVERFLOW = 'summary'` to list those commits one line each instead, with their SHA, author and first message line. Those lines are built from data the tool already fetched, so they cost no extra requests, but a large push then produces a long notification. Set `PUSH_COMMITS_LIMIT = 0` to report every commit of every push in full. + ## Repositories to Monitor @@ -122,6 +147,8 @@ python3 -c "import pytz; print('\n'.join(pytz.all_timezones))" Email notifications need SMTP server details for the sending account. Add them to `github_monitor.conf` or use the setup wizard. Setup checks the login without sending an email. To replace only the password, run `github_monitor --set-smtp-password`. Password entry is hidden and preserves spaces. +Every alert is sent as both HTML and plain text in one message. Mail clients that render HTML show the account, the changed value and the check interval in bold, with repositories, commits, issues and profile addresses linked. Clients that do not fall back to the plain text, which is unchanged. + Send one test message to verify the settings: ```sh diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index e12477a..2d40d92 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -40,9 +40,29 @@ Every failure is reported in the same three-part shape: what went wrong, a `To f | Webhook alerts never arrive | Provider mismatch or a stale destination | [Webhook Settings](configuration.md#webhook-settings) then run `github_monitor --send-test-webhook` | | `github_monitor` is not found after installation | The shell has not picked up the new command | [Installation and Command Problems](#installation-and-command-problems) | | Escape sequences such as `[36m` printed as text or no colour at all | The terminal cannot display ANSI colour or colour was switched off | [Terminal Colours Look Wrong](#terminal-colours-look-wrong) | +| `GitHub did not answer in time`, `GitHub could not be reached` or `GitHub is temporarily unavailable` | A network problem between this machine and GitHub or a GitHub outage | [Connection Problems](#connection-problems) | +| `This process ran out of file descriptors` | The operating system limit on open files was reached | [Too Many Open Files](#too-many-open-files) | A continuing outage produces a `* Monitoring degraded` reminder once an hour, even when the [liveness reminder](usage.md#liveness-reminder) is switched off. `* Monitoring recovered` marks recovery. Use `--verbose` to see the first failed check. + +## Connection Problems + +`GitHub did not answer in time` and `GitHub could not be reached` mean a check got no answer from GitHub. `GitHub is temporarily unavailable` means GitHub answered with a server error. The report names the interval after which the check is retried, so a short outage needs no action. While the network is down only the first request prints its retry attempts, because the requests behind it stop retrying once it has shown that GitHub cannot be reached. A failure that lasts produces the hourly `Monitoring degraded` reminder and `Monitoring recovered` when it clears. + +If the failure continues, check the internet connection, DNS and any firewall or proxy between this machine and GitHub. A certificate error points at TLS interception on the network, see [TLS Verification](configuration.md#tls-verification). A server error that lasts is a GitHub outage, so wait for it to end. + +To confirm that GitHub is reachable from this machine, run: + +```sh +github_monitor --doctor +``` + + +## Too Many Open Files + +`This process ran out of file descriptors` means the operating system limit on open files was reached. It is a local limit and not a GitHub problem. Raise it with `ulimit -n 4096` in the shell that starts the tool or set `LimitNOFILE=` in the systemd unit, then restart the tool. + ## Terminal Colours Look Wrong diff --git a/docs/usage.md b/docs/usage.md index 5374208..c0b3431 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -97,10 +97,12 @@ Monitoring mode prints the settings that are actually in effect before the first Optional features appear once you switch them on. -Use `--verbose` or `--debug` for the full startup summary, including output paths, notification settings, secret sources and runtime information. +Use `--verbose` or `--debug` for the full startup summary, including output paths, notification settings, push event limits, secret sources and runtime information. Use `--truncate N` or `TRUNCATE_CHARS` to limit screen line width. Set it to `999` to detect the terminal width automatically. Truncation does not change log files and is ignored when logging is disabled with `-d`. +Use `--push-commits-limit N` and `--push-files-limit N` to control how much of a large push is reported in full. See [Push Event Commits](configuration.md#push-event-commits). + The tool clears the terminal when monitoring starts. Set `CLEAR_SCREEN` to `False` to keep whatever is already on the screen. The screen is never cleared when output is redirected to a file or a pipe, in debug mode or for a command that prints a result and exits, such as `--doctor`, `--help` and the test senders. @@ -219,16 +221,18 @@ github_monitor github_username -m -y The `-y` flag only works if tracking of daily contributions is enabled (`-m`). -To disable sending an email on errors (enabled by default): +To disable sending an email on errors and the recovery alert that follows (both enabled by default): - set `ERROR_NOTIFICATION` to `False` -- or use the `-e` flag +- or use the `-e` / `--no-error-notify` flag ```sh github_monitor github_username -e ``` -Email and webhook error alerts are sent after **5 minutes** of a continuing failure. Problems that need your action, such as a rejected token, alert immediately. Each kind of failure alerts once per channel. Failed deliveries are retried after 5 minutes, with increasing waits up to an hour. Alerts can fire again after monitoring recovers. +Email and webhook error alerts are sent after **5 minutes** of a continuing failure. Problems that need your action, such as a rejected token, alert immediately. Each kind of failure alerts once per channel. Failed deliveries are retried after 5 minutes, with increasing waits up to an hour. A new outage after a recovery alerts again. A channel that could not receive the failure alert while the outage lasted is told about the failure and its recovery together, so a blocked channel is not left without any word of an outage. + +A failure alert carries the subject `GitHub Monitor error: (user: )` and lists the fix, the guide link, how many checks failed in a row, since when and when the next retry is. When the failure clears, a matching `GitHub Monitor recovered: ...` alert goes to the channels the failure alert reached. `-e` / `--no-error-notify` switches both off for email and `--no-webhook-error-notify` switches both off for webhooks. Missing optional details, such as a push event's file list, are reported on screen without triggering an error alert. When several checks fail, the alert shows the action to take and the number of other failures. @@ -254,7 +258,9 @@ Webhook event controls mirror the email categories but work independently: | Repository changes | `WEBHOOK_REPO_NOTIFICATION` | `--webhook-repo-changes` | | Repository update date changes | `WEBHOOK_REPO_UPDATE_DATE_NOTIFICATION` | `--webhook-repo-update-date` | | Daily contribution changes | `WEBHOOK_CONTRIB_NOTIFICATION` | `--webhook-daily-contribs` | -| Monitoring errors | `WEBHOOK_ERROR_NOTIFICATION` | Enable with `--webhook-errors` or disable with `--no-webhook-error-notify` | +| Monitoring errors and recoveries | `WEBHOOK_ERROR_NOTIFICATION` | Enable with `--webhook-errors` or disable with `--no-webhook-error-notify` | + +A monitoring error webhook carries the same title and fields as the error email, without the timestamp the webhook service shows itself, and the matching recovery alert follows on the same channel. `--no-webhook-error-notify` switches both off. Use `--webhook` or `--no-webhook` to turn all configured webhook alerts on or off for one run. A category override also enables the master webhook switch. For example: diff --git a/github_monitor.py b/github_monitor.py index 951a1e4..41a7c98 100755 --- a/github_monitor.py +++ b/github_monitor.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """ Author: Michal Szymanski -v2.7 +v2.8 OSINT tool implementing real-time tracking of GitHub users activities including profile and repositories changes: https://github.com/misiektoja/github_monitor/ @@ -18,7 +18,7 @@ wcwidth (optional, measures wide characters correctly when TRUNCATE_CHARS is set) """ -VERSION = "2.7" +VERSION = "2.8" # --------------------------- # CONFIGURATION SECTION START @@ -95,7 +95,7 @@ # Can also be enabled via the -y flag CONTRIB_NOTIFICATION = False -# Whether to send an email on errors +# Whether to send an email on errors and the recovery alert that follows once the failure clears # Can also be disabled via the -e flag ERROR_NOTIFICATION = True @@ -151,7 +151,7 @@ # Can also be enabled via the --webhook-daily-contribs flag WEBHOOK_CONTRIB_NOTIFICATION = False -# Whether to send a webhook notification on monitoring errors +# Whether to send a webhook notification on monitoring errors and the recovery alert that follows once the failure clears # Can also be enabled via --webhook-errors or disabled via --no-webhook-error-notify WEBHOOK_ERROR_NOTIFICATION = True @@ -238,6 +238,27 @@ # any events older than the most recent EVENTS_NUMBER will be missed EVENTS_NUMBER = 30 # 1 page +# Maximum number of commits from a single push event reported in full, with date, author URL, stats and +# changed files. Every detailed commit costs one extra API request, so a large push would otherwise spend +# hundreds of requests and produce a notification nobody reads +# Set to 0 to report every commit in full +# Can also be set using the --push-commits-limit flag +PUSH_COMMITS_LIMIT = 10 + +# Which end of an oversized push keeps the detailed commits +# 'newest' keeps the head of the push, 'oldest' keeps the start of the pushed range +PUSH_COMMITS_ORDER = 'newest' + +# How the commits beyond PUSH_COMMITS_LIMIT are reported +# 'count' replaces them with one line stating how many were left out +# 'summary' lists each one on a single line with its SHA, author and first message line, at no request cost +PUSH_COMMITS_OVERFLOW = 'count' + +# Maximum number of changed files listed per commit, with the remainder replaced by a count +# Set to 0 to list every changed file +# Can also be set using the --push-files-limit flag +PUSH_FILES_LIMIT = 20 + # If True, track user's repository changes (changed stargazers, watchers, forks, issues, PRs, discussions, description, update date etc.) # Can also be enabled using the -j flag TRACK_REPOS_CHANGES = False @@ -457,6 +478,10 @@ LOCAL_TIMEZONE_STATE = "config" EVENTS_TO_MONITOR = [] EVENTS_NUMBER = 0 +PUSH_COMMITS_LIMIT = 0 +PUSH_COMMITS_ORDER = "" +PUSH_COMMITS_OVERFLOW = "" +PUSH_FILES_LIMIT = 0 TRACK_REPOS_CHANGES = False VERIFY_REPOSITORY_CLOSURES = True REPOS_TO_MONITOR = [] @@ -497,6 +522,10 @@ # Failures from the current check, retained until all enabled monitoring paths have finished MONITOR_CHECK_FAILURES: dict = {} +# True once a call has proved the network unreachable, so the calls behind it fail fast rather than each paying +# the full retry schedule for a connection that is already known to be down +NET_OUTAGE_CONFIRMED = False + VERBOSE_MODE = False DEBUG_MODE = False DELIVERY_CONFIRMATIONS = True @@ -532,6 +561,8 @@ SMTP_GUIDE_URL = f"{DOCS_BASE_URL}/configuration/#smtp-settings" WEBHOOK_GUIDE_URL = f"{DOCS_BASE_URL}/configuration/#webhook-settings" DIAGNOSTICS_GUIDE_URL = f"{DOCS_BASE_URL}/troubleshooting/#verbose-and-debug-output" +CONNECTION_GUIDE_URL = f"{DOCS_BASE_URL}/troubleshooting/#connection-problems" +DESCRIPTOR_LIMIT_GUIDE_URL = f"{DOCS_BASE_URL}/troubleshooting/#too-many-open-files" TLS_GUIDE_URL = f"{DOCS_BASE_URL}/configuration/#tls-verification" SUPPORT_GUIDE_URL = f"{DOCS_BASE_URL}/about/#support" DOCTOR_GUIDE_URL = f"{DOCS_BASE_URL}/troubleshooting/#doctor-preflight" @@ -554,6 +585,10 @@ ERROR_ALERT_RETRY_SECONDS = 300 # 5 minutes ERROR_ALERT_RETRY_MAX_SECONDS = 3600 # 1 hour +# When the run last read the data a change is compared against, which a failing check leaves further back than +# the configured interval +LAST_CHECK_TS = 0 + stdout_bck = None csvfieldnames = ['Date', 'Type', 'Name', 'Old', 'New'] @@ -769,9 +804,10 @@ def bootstrap_doctor_dependency_report(module_finder=None, stream=None): MIN_REDACTABLE_SECRET_LENGTH = 12 -# Tracks the error alert per channel: what was delivered, and how long a channel that failed waits before the next attempt +# Tracks the error alert per channel: what was delivered, how long a channel that failed waits before the next +# attempt, and the failure and outage start the recovery alert names once the outage clears class ErrorAlertState: - # Starts with nothing delivered and no channel on hold + # Starts with nothing delivered, no channel on hold and no failure remembered def __init__(self) -> None: self.email_sent = False self.webhook_sent = False @@ -779,15 +815,30 @@ def __init__(self) -> None: self.webhook_failures = 0 self.email_retry_at = 0 self.webhook_retry_at = 0 + self.advice = None + self.since = 0 - # Forgets the delivered alert and any hold, so the next failure earns each channel a new one + # Forgets the delivered alert, any hold and the remembered failure, so the next outage earns each channel a new alert def reset(self) -> None: self.__init__() + # Remembers the latest failure and when the outage began, so the recovery alert can say what cleared and how long it took + def remember(self, advice, since: int) -> None: + self.advice = advice + self.since = since + + # Tells whether a channel delivered the failure alert of this outage and is still switched on, so it is owed the recovery alert + def delivered(self, channel: str, enabled) -> bool: + return bool(enabled) and getattr(self, f"{channel}_sent") + # Tells whether a channel still owes the alert and its wait after a failed attempt, if any, has passed def pending(self, channel: str, enabled, now: int) -> bool: return bool(enabled) and not getattr(self, f"{channel}_sent") and now >= getattr(self, f"{channel}_retry_at") + # Tells whether a channel was owed the failure alert but never received it, so the recovery can tell it the whole story + def missed(self, channel: str, enabled) -> bool: + return bool(enabled) and not getattr(self, f"{channel}_sent") and getattr(self, f"{channel}_failures") > 0 + # Records one attempt, holding a channel that failed for a growing wait so a broken server is not dialled on every check def record(self, channel: str, attempted: bool, delivered: bool, now: int) -> None: if not attempted: @@ -991,10 +1042,14 @@ def sanitize_terminal_text(message): _USER_TAG_RE = re.compile(r"((?:GitHub|for|by|of|fetch)[\t ]+user:?|\buser[:=])([\t ]+|(?<==))([A-Za-z0-9](?:[A-Za-z0-9-]{0,38}))(?![A-Za-z0-9:-])") _QUOTED_CONTEXT_RE = re.compile(r"\b(repo|user)\s+$", re.IGNORECASE) _DURATION_RE = re.compile(r"~?\b[0-9]{1,20}[ \t]{1,20}(?:seconds?|minutes?|hours?|days?|weeks?|months?|years?)\b", re.IGNORECASE) -_LONG_DATE_RE = re.compile(r"\b(?:\w{3}\s+)?\d{1,2}\s+\w{3}(?:\s+\d{2,4})?[\s,]*\d{2}:\d{2}(:\d{2})?(\s*[AP]M)?\b", re.IGNORECASE) +# The weekday in front of a date, taken from the abbreviations the running locale prints. A date is separated +# from its weekday by one space, so the wide gap of a padded listing column cannot pull the word before it, +# such as the last word of a line, into the date +_WEEKDAY_ABBR_PATTERN = "|".join(re.escape(day_abbr) for day_abbr in calendar.day_abbr) +_LONG_DATE_RE = re.compile(r"\b(?:(?:" + _WEEKDAY_ABBR_PATTERN + r")[\t ])?\d{1,2}\s+\w{3}(?:\s+\d{2,4})?[\s,]*\d{2}:\d{2}(:\d{2})?(\s*[AP]M)?\b", re.IGNORECASE) _TIME_ONLY_RE = re.compile(r"(?") +# Turns a bare URL inside already escaped HTML text into a link, so an alert that prints an address is clickable +def html_autolink_urls(content): + return re.sub(r"(?\"']+[^\s<>\"'.,;:!?)\]])", r'\1', str(content)) + + # Returns the advice a cancelled secret entry reports, worded the same way by every one-shot secret command def secret_entry_cancelled_advice(subject, flag, guide_url): return make_recovery_advice("secret.entry", f"{subject[:1].upper()}{subject[1:]} setup was cancelled and the dotenv file was not changed", recovery_fix_with_guide(f"Run {flag} again when you have the value ready", guide_url), False) @@ -3349,10 +3421,12 @@ def print_liveness_banner(message): # Reminds about a lasting failure once an hour, so a broken run still says it is alive without repeating itself -def print_outage_liveness(target, advice, since, failures=0): +def print_outage_liveness(target, advice, since, failures=0, close=True): count = f", {failures} failed {'check' if failures == 1 else 'checks'}" if failures else "" print(f"* Monitoring degraded for {target}. {advice.summary} since {get_date_from_ts(since)}{count}") - print_cur_ts("Liveness check, timestamp:\t") + # A caller with an alert still to deliver closes the report itself, so the delivery lines stay inside it + if close: + print_cur_ts("Liveness check, timestamp:\t") # Notes that a reported outage now fails differently, in one line rather than a second full report @@ -3361,9 +3435,82 @@ def print_outage_change(target, advice): # Reports that a failure cleared, since a throttled failure no longer stops printing when it is over -def print_outage_recovery(target, lasted): +def print_outage_recovery(target, lasted, close=True): print(f"* Monitoring recovered for {target} after {display_time(max(1, lasted))}") - print_cur_ts("Timestamp:\t\t\t") + # A caller with a recovery alert to deliver closes the report itself, so the delivery lines stay inside it + if close: + print_cur_ts("Timestamp:\t\t\t") + + +# Builds the subject every failure alert shares, so an inbox fed by several monitors sorts them by tool +def recovery_alert_subject(advice, target): + return f"GitHub Monitor error: {advice.summary} (user: {target})" + + +# Returns the text groups a failure alert lists under its summary, shared by the plain and HTML bodies +def recovery_alert_sections(advice, retry_seconds, failed_checks=0, failing_since=0): + retry_lines = [] + # A first failure has no run to count, so the count and its start appear once a second check has failed + if failed_checks > 1: + retry_lines.append(f"Failed checks in a row: {failed_checks}") + retry_lines.append(f"Failing since: {get_date_from_ts(failing_since)}") + retry_lines.append(f"Next retry in: {display_time(retry_seconds)}") + sections = [f"To fix: {advice.fix}", "\n".join(retry_lines)] + # A detail that only repeats the summary spends a line saying nothing + if DEBUG_MODE and advice.detail and advice.detail != advice.summary: + sections.append(f"Technical detail: {advice.detail}") + return sections + + +# Builds the plain text of a failure alert, without the timestamp when a webhook service shows its own +def recovery_alert_body(advice, retry_seconds, failed_checks=0, failing_since=0, timestamp=True): + body = "\n\n".join([advice.summary, *recovery_alert_sections(advice, retry_seconds, failed_checks, failing_since)]) + return body + get_cur_ts("\n\nTimestamp: ") if timestamp else body + + +# Bolds the values a reader scans a failure alert for: how often it has failed and since when +def html_bold_outage_fields(content): + for label in ("Failed checks in a row: ", "Failing since: "): + content = re.sub(f"({re.escape(label)})([^<]+)", r"\1\2", content, count=1) + return content + + +# Builds the HTML email body of a failure alert with the same parts as the plain text and the summary in bold +def recovery_alert_body_html(advice, retry_seconds, failed_checks=0, failing_since=0, timestamp=True): + parts = [f"{html_text(advice.summary)}", *(html_autolink_urls(html_text(section)) for section in recovery_alert_sections(advice, retry_seconds, failed_checks, failing_since))] + if timestamp: + parts.append(get_cur_ts("Timestamp: ")) + return html_bold_outage_fields(f"{'

'.join(parts)}") + + +# Builds the subject of the alert that answers a delivered failure alert once the outage clears +def outage_recovered_alert_subject(target, lasted): + return f"GitHub Monitor recovered: monitoring {target} resumed after {display_time(lasted)}" + + +# Builds the plain text of the recovery alert, naming the failure it closes +def outage_recovered_alert_body(advice, target, lasted, timestamp=True): + body = f"Monitoring recovered for {target} after {display_time(lasted)}.\n\nThe failure was: {advice.summary}" + return body + get_cur_ts("\n\nTimestamp: ") if timestamp else body + + +# Builds the HTML email body of the recovery alert +def outage_recovered_alert_body_html(advice, target, lasted, timestamp=True): + body = f"Monitoring recovered for {html.escape(target)} after {html.escape(display_time(lasted))}.

The failure was: {html_text(advice.summary)}" + return f"{body}{get_cur_ts('

Timestamp: ') if timestamp else ''}" + + +# Tells a channel that never received the failure alert about the whole outage, since a bare recovery would close +# a failure it was never told about +def outage_missed_alert_body(advice, target, lasted, timestamp=True): + body = f"Monitoring failed for {target} at {get_date_from_ts(int(time.time()) - lasted)} and recovered after {display_time(lasted)}.\n\nThe failure was: {advice.summary}\n\nThe failure alert could not be delivered here while the failure lasted." + return body + get_cur_ts("\n\nTimestamp: ") if timestamp else body + + +# Builds the HTML email body of the combined failure and recovery alert +def outage_missed_alert_body_html(advice, target, lasted, timestamp=True): + body = f"Monitoring failed for {html.escape(target)} at {html.escape(get_date_from_ts(int(time.time()) - lasted))} and recovered after {html.escape(display_time(lasted))}.

The failure was: {html_text(advice.summary)}

The failure alert could not be delivered here while the failure lasted." + return f"{body}{get_cur_ts('

Timestamp: ') if timestamp else ''}" # Yields the exception and each cause or context up to max_depth, to walk an exception chain @@ -3376,6 +3523,24 @@ def iter_exc_chain(error, max_depth=8): current = getattr(current, "__cause__", None) or getattr(current, "__context__", None) +# Names the transport failure behind an exception chain, since a timeout raised with no message leaves the text rules nothing to read +def network_failure_code(error): + timed_out = False + unreachable = False + for current in iter_exc_chain(error): + name = type(current).__name__ + # A TLS failure has its own advice, so a chain that names one is left to the rules that recognize it + if "SSL" in name or "Certificate" in name: + return "" + if isinstance(current, TimeoutError) or "Timeout" in name: + timed_out = True + elif isinstance(current, ConnectionError) or name in ("gaierror", "herror") or any(term in name for term in ("Connect", "ProxyError", "NameResolution", "Unreachable")): + unreachable = True + if timed_out: + return "network.timeout" + return "network.unavailable" if unreachable else "" + + # Reports whether this process hit the local file descriptor limit rather than a remote failure def is_too_many_open_files(error): for current in iter_exc_chain(error): @@ -3408,32 +3573,33 @@ def classify_recovery_error(error, context="runtime", detail="", install_context debug_command = render_command(["--debug"], install_context=install_context) # Checked ahead of every context, since a local descriptor limit is not a failure of whatever call hit it if error is not None and is_too_many_open_files(error): - return make_recovery_advice("resource.exhausted", "This process ran out of file descriptors, which is a local limit and not a GitHub problem", recovery_fix_with_guide("Raise the file descriptor limit, for example with 'ulimit -n 4096', or set LimitNOFILE= if you run under systemd, then restart the tool", DIAGNOSTICS_GUIDE_URL), False, detail) + return make_recovery_advice("resource.exhausted", "This process ran out of file descriptors, which is a local limit and not a GitHub problem", recovery_fix_with_guide("Raise the file descriptor limit, for example with 'ulimit -n 4096', or set LimitNOFILE= if you run under systemd, then restart the tool", DESCRIPTOR_LIMIT_GUIDE_URL), False, detail) if selected_context == "connectivity": # Classified from the error, because a failed endpoint check has one answer whatever the exception was timed_out = isinstance(error, (req.Timeout, TimeoutError, socket.timeout)) summary = "The connectivity endpoint did not answer in time" if timed_out else "The connectivity endpoint could not be reached" # No guide, because no page covers this check and the doctor report already ends with the troubleshooting link return make_recovery_advice("network.timeout" if timed_out else "network.unavailable", summary, CONNECTIVITY_ENDPOINT_FIX, True, detail) - if isinstance(error, (req.Timeout, TimeoutError, socket.timeout)): - return make_recovery_advice("network.timeout", "The network request timed out", recovery_fix_with_guide("Check connectivity and increase the configured timeout before trying again", DIAGNOSTICS_GUIDE_URL), True, detail) - if isinstance(error, (req.ConnectionError, socket.gaierror)): - return make_recovery_advice("network.unavailable", "The configured service could not be reached", recovery_fix_with_guide("Check the network and configured service URL then try again", DIAGNOSTICS_GUIDE_URL), True, detail) - if isinstance(error, req.RequestException): - return make_recovery_advice("network.unavailable", "The configured service request failed", recovery_fix_with_guide("Check the network and configured service URL then try again", DIAGNOSTICS_GUIDE_URL), True, detail) + # The chain is read alongside the type, since a transport error can arrive wrapped and with an empty message + transport_code = network_failure_code(error) + if isinstance(error, (req.Timeout, TimeoutError, socket.timeout)) or transport_code == "network.timeout": + return make_recovery_advice("network.timeout", "GitHub did not answer in time", recovery_fix_with_guide(TRANSIENT_NETWORK_FIX, CONNECTION_GUIDE_URL), True, detail) + if isinstance(error, (req.ConnectionError, socket.gaierror, req.RequestException)) or transport_code == "network.unavailable": + return make_recovery_advice("network.unavailable", "GitHub could not be reached", recovery_fix_with_guide(TRANSIENT_NETWORK_FIX, CONNECTION_GUIDE_URL), True, detail) if isinstance(error, BadCredentialsException): return make_recovery_advice("auth.github_token_invalid", "GitHub rejected the configured token", recovery_fix_with_guide(f"Create or review the token then run: {token_command}", AUTH_GUIDE_URL), False, detail) if isinstance(error, RateLimitExceededException): - return make_recovery_advice("github.rate_limited", "GitHub API rate limiting paused the request", recovery_fix_with_guide("Wait for the reported reset time before trying again", DIAGNOSTICS_GUIDE_URL), True, detail) + return make_recovery_advice("github.rate_limited", "GitHub API rate limiting paused the request", recovery_fix_with_guide("Wait for the reported reset time before trying again", INTERVALS_GUIDE_URL), True, detail) if isinstance(error, UnknownObjectException): code = "target.not_found" if selected_context == "target" else "github.not_found" - return make_recovery_advice(code, "GitHub could not find the requested resource", recovery_fix_with_guide("Check the target name and token access then try again", DIAGNOSTICS_GUIDE_URL), False, detail) + return make_recovery_advice(code, "GitHub could not find the requested resource", recovery_fix_with_guide("Check the target name and token access then try again", QUICK_START_GUIDE_URL), False, detail) if isinstance(error, GithubException): status = getattr(error, "status", None) if status == 403: return make_recovery_advice("github.forbidden", "GitHub refused access to the requested resource", recovery_fix_with_guide("Check token permissions and resource visibility", AUTH_GUIDE_URL), False, detail) - retryable = status is None or (isinstance(status, int) and status >= 500) - return make_recovery_advice("github.api_error", "GitHub returned an API error", recovery_fix_with_guide(f"Try again or run {debug_command} for sanitized technical detail", DIAGNOSTICS_GUIDE_URL), retryable, detail) + if isinstance(status, int) and status >= 500: + return make_recovery_advice("github.unavailable", "GitHub is temporarily unavailable", recovery_fix_with_guide("Usually nothing to do, the tool retries on its own. If it continues, wait for GitHub to recover", CONNECTION_GUIDE_URL), True, detail) + return make_recovery_advice("github.api_error", "GitHub returned an API error", recovery_fix_with_guide(f"Try again or run {debug_command} for sanitized technical detail", DIAGNOSTICS_GUIDE_URL), status is None, detail) if isinstance(error, smtplib.SMTPAuthenticationError): return make_recovery_advice("smtp.authentication", "The SMTP server rejected the configured credentials", recovery_fix_with_guide("Check SMTP_USER and replace SMTP_PASSWORD before sending another test", SMTP_GUIDE_URL), False, detail) if isinstance(error, PermissionError): @@ -3614,34 +3780,72 @@ def mask_email_address(address): return f"{masked}@{domain}" +# Returns whether a mail server is set rather than left empty or still holding the placeholder the sample configuration ships +def smtp_server_configured(): + return secret_is_set(SMTP_HOST) and bool(SMTP_PORT) + + +# Returns whether an email alert has both a server to send through and an address to reach +def email_channel_configured(): + return smtp_server_configured() and secret_is_set(RECEIVER_EMAIL) + + +# Returns whether a webhook alert has a destination to post to +def webhook_channel_configured(): + return bool(normalized_webhook_provider()) and secret_is_set(WEBHOOK_URL) + + +# Rolls one channel's enabled alerts into the state its summary row reports, which is off while the channel has no destination +def _startup_notification_state(categories, configured): + if not categories: + return "Off" + return "On (" + ", ".join(categories) + ")" if configured else "Off (not configured)" + + # Names the mail server this run would use, leaving out the account that signs in to it def startup_email_transport(): - if not SMTP_HOST or not SMTP_PORT: + if not smtp_server_configured(): return "Not configured" return f"{SMTP_HOST}:{SMTP_PORT} ({'STARTTLS' if SMTP_SSL else 'TLS off'})" # Names the configured webhook service and whether the channel is switched on, which are two separate settings def startup_webhook_provider(): - if not normalized_webhook_provider() or not str(WEBHOOK_URL or "").strip(): + if not webhook_channel_configured(): return "Not configured" return f"{webhook_provider_display_name()} ({'enabled' if WEBHOOK_ENABLED else 'disabled'})" +# Describes how much of a push the run reports in full +def startup_push_commit_details(): + if DO_NOT_MONITOR_GITHUB_EVENTS: + return "Inactive" + if not PUSH_COMMITS_LIMIT: + return "Every commit in full" + return f"{PUSH_COMMITS_LIMIT} {PUSH_COMMITS_ORDER} per push, rest by {PUSH_COMMITS_OVERFLOW}" + + +# Describes how many changed files each detailed commit lists +def startup_push_changed_files(): + if DO_NOT_MONITOR_GITHUB_EVENTS: + return "Inactive" + return f"{PUSH_FILES_LIMIT} per commit" if PUSH_FILES_LIMIT else "Every changed file" + + # Builds concise and complete startup rows without exposing private values def build_startup_summary(target, config_path, env_path, output_path): install_context = detect_install_context() email_categories = _startup_email_notification_categories() webhook_categories = _startup_webhook_notification_categories() - email_state = "On (" + ", ".join(email_categories) + ")" if email_categories else "Off" - webhook_state = "On (" + ", ".join(webhook_categories) + ")" if webhook_categories else "Off" + email_state = _startup_notification_state(email_categories, email_channel_configured()) + webhook_state = _startup_notification_state(webhook_categories, webhook_channel_configured()) from_dotenv, from_environment, from_config, from_command_line = startup_secret_buckets() return [ StartupSummaryRow("Target", str(target), concise=True), StartupSummaryRow("Polling interval", display_time(GITHUB_CHECK_INTERVAL), concise=True), StartupSummaryRow("Notifications (email)", email_state, concise=True), StartupSummaryRow("Email transport", startup_email_transport()), - StartupSummaryRow("Email recipient", mask_email_address(RECEIVER_EMAIL) if RECEIVER_EMAIL else "Not configured"), + StartupSummaryRow("Email recipient", mask_email_address(RECEIVER_EMAIL) if secret_is_set(RECEIVER_EMAIL) else "Not configured"), StartupSummaryRow("Notifications (webhook)", webhook_state, concise=True), StartupSummaryRow("Webhook provider", startup_webhook_provider()), StartupSummaryRow("Delivery confirmations", str(DELIVERY_CONFIRMATIONS)), @@ -3655,6 +3859,8 @@ def build_startup_summary(target, config_path, env_path, output_path): StartupSummaryRow("Closure verification budget", f"{REPOSITORY_CLOSURE_REQUEST_BUDGET} requests/check (shared)" if TRACK_REPOS_CHANGES and VERIFY_REPOSITORY_CLOSURES else "Inactive"), StartupSummaryRow("Track contribution changes", str(TRACK_CONTRIB_CHANGES)), StartupSummaryRow("Monitor GitHub events", str(not DO_NOT_MONITOR_GITHUB_EVENTS)), + StartupSummaryRow("Push commit details", startup_push_commit_details()), + StartupSummaryRow("Push changed files", startup_push_changed_files()), StartupSummaryRow("Owned repositories only", str(not GET_ALL_REPOS)), StartupSummaryRow("Liveness output", display_time(LIVENESS_CHECK_INTERVAL) if LIVENESS_CHECK_INTERVAL else "Disabled", concise=bool(LIVENESS_CHECK_INTERVAL)), StartupSummaryRow("CSV output", str(CSV_FILE) if CSV_FILE else "Disabled", concise=bool(CSV_FILE)), @@ -3680,11 +3886,23 @@ def build_startup_summary(target, config_path, env_path, output_path): # Rows that detail the channel named right above them, indented so the block reads as one setting with its details STARTUP_SUMMARY_NESTED_LABELS = ("Email transport", "Email recipient", "Email images", "Webhook provider", "ntfy images") +# The column every summary value starts in, which also lets the colouriser recognize a summary row +STARTUP_SUMMARY_VALUE_COLUMN = 32 + +# Matches a summary row by that padded label column, since no log line puts a value there +_STARTUP_SUMMARY_ROW_RE = re.compile(r"^\*(?: {1,3})[^:\s][^:]*: {2,}(?=\S)") + + +# Returns whether a line is a startup summary row rather than ordinary output +def is_startup_summary_row(line): + match = _STARTUP_SUMMARY_ROW_RE.match(line) + return bool(match) and match.end() == STARTUP_SUMMARY_VALUE_COLUMN + # Formats one startup summary row with aligned plain ASCII columns def format_startup_summary_row(row): indent = " " if row.label in STARTUP_SUMMARY_NESTED_LABELS else "" - prefix = f"* {indent}{(row.label + ':'):<{30 - len(indent)}}" + prefix = f"* {indent}{(row.label + ':'):<{STARTUP_SUMMARY_VALUE_COLUMN - 2 - len(indent)}}" if row.label in ("Notifications (email)", "Notifications (webhook)"): return textwrap.fill(row.value, width=100, initial_indent=prefix, subsequent_indent=" " * len(prefix), break_long_words=False, break_on_hyphens=False) + "\n" return f"{prefix}{row.value}\n" @@ -4135,8 +4353,9 @@ def send_webhook(title: str, description: str, notification_type: str = "event", return 1 -# Sends one alert through the independently enabled email and webhook channels -def send_notification_channels(notification_type: str, subject: str, body: str, body_html: str = "", email_enabled: bool = False, webhook_enabled: Optional[bool] = None) -> tuple[bool, bool]: +# Sends one alert through the independently enabled email and webhook channels, with its own webhook text when the +# email body carries a part such as the timestamp that the webhook service already shows +def send_notification_channels(notification_type: str, subject: str, body: str, body_html: str = "", email_enabled: bool = False, webhook_enabled: Optional[bool] = None, webhook_body: Optional[str] = None, webhook_body_html: Optional[str] = None) -> tuple[bool, bool]: email_attempted = bool(email_enabled) webhook_attempted = webhook_event_enabled(notification_type) if webhook_enabled is None else bool(webhook_enabled) email_delivered = False @@ -4146,7 +4365,9 @@ def send_notification_channels(notification_type: str, subject: str, body: str, email_delivered = send_email(subject, body, body_html, SMTP_SSL) == 0 if webhook_attempted: print(f"Sending webhook notification via {webhook_provider_display_name()}") - webhook_delivered = send_webhook(subject, body, notification_type, force=True, discord_description=html_body_to_discord_markdown(body_html)) == 0 + webhook_description = body if webhook_body is None else webhook_body + webhook_markdown = html_body_to_discord_markdown(body_html if webhook_body_html is None else webhook_body_html) + webhook_delivered = send_webhook(subject, webhook_description, notification_type, force=True, discord_description=webhook_markdown) == 0 # Delivery, not the attempt, so a channel that failed is retried while one that succeeded is not resent return email_delivered, webhook_delivered @@ -4385,6 +4606,25 @@ def get_range_of_dates_from_tss(ts1, ts2, between_sep=" - ", short=False): return str(out_str) +# Returns how long the window a change was observed in lasted and when it ended, falling back to the configured +# interval until the run has a previous check to measure from +def observed_window(): + ended = int(time.time()) + return max(1, ended - LAST_CHECK_TS if LAST_CHECK_TS else GITHUB_CHECK_INTERVAL), ended + + +# Returns the window a change was observed in, as a duration followed by the dates it spans +def check_window_text(): + lasted, ended = observed_window() + return f"{display_time(lasted)} ({get_range_of_dates_from_tss(ended - lasted, ended, short=True)})" + + +# Returns the same window with the duration emphasized, for an HTML notification body +def check_window_html(): + lasted, ended = observed_window() + return f"{html.escape(display_time(lasted))} ({html.escape(get_range_of_dates_from_tss(ended - lasted, ended, short=True))})" + + # Checks if the timezone name is correct def is_valid_timezone(tz_name): return tz_name in pytz.all_timezones @@ -4672,18 +4912,26 @@ def github_object_name(value): # Callers wrap a lambda and invoke the result immediately, so a lambda that reads a loop variable is # evaluated inside the same iteration. Those call sites carry a noqa marker for the loop-binding rule # Retries a GitHub operation with current settings and either returns its fallback or raises the final failure -def gh_call(fn: Callable[..., Any], retries=None, backoff=None, default: Any = None, *, raise_on_failure=False) -> Callable[..., Any]: +def gh_call(fn: Callable[..., Any], retries=None, backoff=None, default: Any = None, *, operation: str = "", raise_on_failure=False) -> Callable[..., Any]: retries = NET_MAX_RETRIES if retries is None else retries backoff = NET_BASE_BACKOFF_SEC if backoff is None else backoff + # Callers wrap a lambda, whose __name__ says nothing, so the label they pass is what a reader sees + label = operation or fn.__name__ # Keeps the original exception available to callers that must distinguish an unavailable feed from an empty one def wrapped(*args: Any, **kwargs: Any) -> Any: + global NET_OUTAGE_CONFIRMED last_error = None - for i in range(1, retries + 1): + outage_proved = False + # Retrying every later call of a check learns nothing once one has proved the network unreachable, so the + # schedule collapses to a single attempt until a call gets through again + attempts = 1 if NET_OUTAGE_CONFIRMED else retries + for i in range(1, attempts + 1): try: - debug_print("PyGithub retry wrapper", operation=fn.__name__, attempt=f"{i}/{retries}") + debug_print("PyGithub retry wrapper", operation=label, attempt=f"{i}/{attempts}") result = fn(*args, **kwargs) - debug_print("PyGithub retry wrapper", operation=fn.__name__, outcome="OK", attempt=f"{i}/{retries}") + NET_OUTAGE_CONFIRMED = False + debug_print("PyGithub retry wrapper", operation=label, outcome="OK", attempt=f"{i}/{attempts}") return result except RateLimitExceededException as e: last_error = e @@ -4710,27 +4958,38 @@ def wrapped(*args: Any, **kwargs: Any) -> Any: else: sleep_for = int(backoff * i) - retryable = i < retries - debug_print("PyGithub retry wrapper", operation=fn.__name__, outcome="failed", error=f"{type(e).__name__}: {e}", retryable=retryable, attempt=f"{i}/{retries}") + retryable = i < attempts + debug_print("PyGithub retry wrapper", operation=label, outcome="failed", error=f"{type(e).__name__}: {e}", retryable=retryable, attempt=f"{i}/{attempts}") if retryable: - print(f"* {fn.__name__} rate limited, sleeping {sleep_for}s (retry {i}/{retries})") - debug_monitor_wait_timing(f"GitHub rate limit before attempt {i + 1}/{retries}", sleep_for) + print(f"* {label} rate limited by GitHub, sleeping {sleep_for}s (retry {i}/{attempts})") + debug_monitor_wait_timing(f"GitHub rate limit before attempt {i + 1}/{attempts}", sleep_for) time.sleep(sleep_for) continue except NET_ERRORS as e: last_error = e - retryable = i < retries + advice = classify_recovery_error(e) + # A missing resource, a refused token or a local descriptor limit answers the same way every + # time, so the schedule ends rather than spending attempts proving it + permanent = not advice.retryable + retryable = i < attempts and not permanent delay = backoff * i - debug_print("PyGithub retry wrapper", operation=fn.__name__, outcome="failed", error=f"{type(e).__name__}: {e}", retryable=retryable, attempt=f"{i}/{retries}") + debug_print("PyGithub retry wrapper", operation=label, outcome="failed", error=f"{type(e).__name__}: {e}", retryable=retryable, permanent=permanent, attempt=f"{i}/{attempts}") + if permanent: + break + # Only a transport failure says the network itself is down, which is what the breaker reads + outage_proved = bool(network_failure_code(e)) if retryable: - print(f"* {fn.__name__} error: {sanitize_error_text(e)} (retry {i}/{retries})") - debug_monitor_wait_timing(f"GitHub request retry attempt {i + 1}/{retries}", delay) + print(f"* {label} failed: {advice.summary} (retry {i}/{attempts})") + debug_monitor_wait_timing(f"GitHub request retry attempt {i + 1}/{attempts}", delay) time.sleep(delay) + # Set before the raise, so a caller that handles its own failure still shortens the rest of the check + if outage_proved: + NET_OUTAGE_CONFIRMED = True if raise_on_failure and last_error is not None: raise last_error - verbose_degraded_feature(f"GitHub operation {fn.__name__}", "its dependent alerts", last_error) - debug_print("PyGithub retry wrapper", operation=fn.__name__, outcome="default", after=f"{retries} attempts") + verbose_degraded_feature(label, "its dependent alerts", last_error) + debug_print("PyGithub retry wrapper", operation=label, outcome="default", after=f"{attempts} attempts") return default return wrapped @@ -5447,6 +5706,141 @@ def safe_truncate_text(text, max_length=MAX_EVENT_BODY_LENGTH): return result +@dataclass(frozen=True) +class PushCommit: + sha: str | None + message: str + author: str | None + # The listing entry a push comparison already returned, used only when the detail request fails + known: object | None = None + + +# Builds a push commit from an event payload entry, which carries no date, stats or file list +def push_commit_from_payload(entry): return PushCommit(entry.get("sha"), entry.get("message") or "", (entry.get("author") or {}).get("name")) + + +# Builds a push commit from a comparison entry, whose message and author cost no extra request +def push_commit_from_compare(entry): + git_commit = getattr(entry, "commit", None) + author = getattr(git_commit, "author", None) + return PushCommit(getattr(entry, "sha", None) or getattr(entry, "id", None), getattr(git_commit, "message", "") or "", getattr(author, "name", None), entry) + + +# The index range of the commits a push reports in full, given the configured limit and the end it keeps +def push_detail_range(total, limit=None, order=None): + limit = PUSH_COMMITS_LIMIT if limit is None else limit + if not isinstance(limit, int) or isinstance(limit, bool) or limit <= 0 or limit >= total: + return 0, total + if str(PUSH_COMMITS_ORDER if order is None else order).strip().casefold() == "oldest": + return 0, limit + return total - limit, total + + +# Prints the changed files of one commit, capped so a single large commit cannot fill the notification +def print_push_changed_files(files, limit=None): + limit = PUSH_FILES_LIMIT if limit is None else limit + shown = files[:limit] if isinstance(limit, int) and not isinstance(limit, bool) and limit > 0 else files + st = "" + for changed in shown: + st += print_v(f" • '{changed.filename}' - {changed.status} (+{changed.additions} / -{changed.deletions})") + remaining = len(files) - len(shown) + if remaining > 0: + st += print_v(f" • ... and {remaining} more {'file' if remaining == 1 else 'files'}") + return st + + +# Prints the commits a push left undetailed, either as one line each or as a single count +def print_push_skipped_commits(commits, first_number, overflow=None): + if not commits: + return "" + last_number = first_number + len(commits) - 1 + span = f"Commit {first_number}" if first_number == last_number else f"Commits {first_number}-{last_number}" + if str(PUSH_COMMITS_OVERFLOW if overflow is None else overflow).strip().casefold() == "count": + return print_v(f"\n{span} not reported in full ({len(commits)} {'commit' if len(commits) == 1 else 'commits'}, PUSH_COMMITS_LIMIT is {PUSH_COMMITS_LIMIT})") + st = print_v(f"\n{span} (summary only):") + for commit in commits: + first_line = (commit.message or "").split("\n", 1)[0].strip() or "(no commit message)" + author = f" - {commit.author}" if commit.author else "" + st += print_v(f" • {(commit.sha or 'unknown')[:12]}{author} - '{first_line}'") + return st + + +# Prints the full report for one commit of a push, including the request its stats and file list need +def print_push_commit(repo, number, total, commit): + st = print_v(f"\n=== Commit {number}/{total} ===") + st += print_v("." * HORIZONTAL_LINE1) + + message = commit.message or "" + is_multiline = "\n" in message + if message: + first_line = message.split("\n", 1)[0] + st += print_v(f" - Commit message:\t\t'{first_line}...'" if is_multiline else f" - Commit message:\t\t'{message}'") + + commit_details = None + if repo and commit.sha: + debug_github_operation("event commit lookup", commit.sha) + commit_details = gh_call(lambda: repo.get_commit(commit.sha), operation="Commit details")() + + # The comparison entry already holds the date and links, so a failed detail request still reports them + described = commit_details or commit.known + commit_date = getattr(getattr(getattr(described, "commit", None), "author", None), "date", None) + if commit_date: + st += print_v(f" - Commit date:\t\t\t{get_date_from_ts(commit_date)}") + + if commit.sha: + st += print_v(f" - Commit SHA:\t\t\t{commit.sha}") + st += print_v(f" - Commit author:\t\t{commit.author or 'N/A'}") + + author_url = getattr(getattr(described, "author", None), "html_url", None) + if author_url: + st += print_v(f" - Commit author URL:\t\t{author_url}") + + html_url = getattr(described, "html_url", None) + if html_url: + st += print_v(f" - Commit URL:\t\t\t{html_url}") + st += print_v(f" - Commit raw patch URL:\t{html_url}.patch") + + if commit_details: + stats = getattr(commit_details, "stats", None) + additions = stats.additions if stats else 0 + deletions = stats.deletions if stats else 0 + stats_total = stats.total if stats else 0 + st += print_v(f"\n - Additions/Deletions:\t\t+{additions} / -{deletions} ({stats_total})") + + try: + files = list(commit_details.files) + except Exception as exc: + verbose_degraded_feature("Commit file list", "complete push event details", exc, enrichment=True) + files = None + st += print_v(f" - Files changed:\t\t{len(files) if files is not None else 'N/A'}") + if files: + st += print_v(" - Changed files list:") + st += print_push_changed_files(files) + + if is_multiline: + st += print_v("\n - Commit full message:") + st += print_v(f"\n'{message}'") + + st += print_v("." * HORIZONTAL_LINE1) + return st + + +# Prints the commits of a push, detailing at most PUSH_COMMITS_LIMIT of them and reporting the rest cheaply +def print_push_commits(repo, commits): + total = len(commits) + if not total: + return "" + start, end = push_detail_range(total) + st = "" + if end - start < total: + st = print_v(f"Detailed commits:\t\t{end - start} of {total} ({'oldest' if start == 0 else 'newest'})") + st += print_push_skipped_commits(commits[:start], 1) + for number, commit in enumerate(commits[start:end], start=start + 1): + st += print_push_commit(repo, number, total, commit) + st += print_push_skipped_commits(commits[end:], end + 1) + return st + + # Prints details about passed GitHub event def github_print_event(event, g, time_passed=False, ts: datetime | None = None): @@ -5479,7 +5873,7 @@ def github_print_event(event, g, time_passed=False, ts: datetime | None = None): # For ForkEvent, prefer the source repo if available if event.type == "ForkEvent" and repo is not None: try: - parent = gh_call(lambda: getattr(repo, "parent", None))() + parent = gh_call(lambda: getattr(repo, "parent", None), operation="Fork source repository metadata")() if parent: repo = parent except Exception as exc: @@ -5530,152 +5924,30 @@ def github_print_event(event, g, time_passed=False, ts: datetime | None = None): # Prefer commits from payload when present (older API behavior) if event.payload.get("commits"): - commits = event.payload["commits"] - commits_total = len(commits) - st += print_v(f"\nNumber of commits:\t\t{commits_total}") - for commit_count, commit in enumerate(commits, start=1): - st += print_v(f"\n=== Commit {commit_count}/{commits_total} ===") - st += print_v("." * HORIZONTAL_LINE1) - - commit_message = commit['message'] - is_multiline = '\n' in commit_message - if is_multiline: - first_line = commit_message.split('\n', 1)[0] - st += print_v(f" - Commit message:\t\t'{first_line}...'") - else: - st += print_v(f" - Commit message:\t\t'{commit_message}'") - - commit_details = None - if repo: - debug_github_operation("event commit lookup", commit["sha"]) - commit_details = gh_call(lambda: repo.get_commit(commit["sha"]))() # noqa: B023 - - if commit_details: - commit_date = commit_details.commit.author.date - st += print_v(f" - Commit date:\t\t\t{get_date_from_ts(commit_date)}") - - st += print_v(f" - Commit SHA:\t\t\t{commit['sha']}") - st += print_v(f" - Commit author:\t\t{commit['author']['name']}") - - if commit_details and commit_details.author: - st += print_v(f" - Commit author URL:\t\t{commit_details.author.html_url}") - - if commit_details: - st += print_v(f" - Commit URL:\t\t\t{commit_details.html_url}") - st += print_v(f" - Commit raw patch URL:\t{commit_details.html_url}.patch") - - stats = getattr(commit_details, "stats", None) - additions = stats.additions if stats else 0 - deletions = stats.deletions if stats else 0 - stats_total = stats.total if stats else 0 - st += print_v(f"\n - Additions/Deletions:\t\t+{additions} / -{deletions} ({stats_total})") - - if commit_details: - try: - file_count = sum(1 for _ in commit_details.files) - except Exception as exc: - verbose_degraded_feature("Commit file list", "complete push event details", exc, enrichment=True) - file_count = "N/A" - st += print_v(f" - Files changed:\t\t{file_count}") - if file_count: - st += print_v(f" - Changed files list:") - for f in commit_details.files: - st += print_v(f" • '{f.filename}' - {f.status} (+{f.additions} / -{f.deletions})") - - if is_multiline: - st += print_v(f"\n - Commit full message:") - st += print_v(f"\n'{commit_message}'") - else: - pass - st += print_v("." * HORIZONTAL_LINE1) + commits = [push_commit_from_payload(entry) for entry in event.payload["commits"]] + st += print_v(f"\nNumber of commits:\t\t{len(commits)}") + st += print_push_commits(repo, commits) # Fallback for new Events API where PushEvent no longer includes commit summaries elif event.type == "PushEvent" and repo: before_sha = event.payload.get("before") head_sha = event.payload.get("head") or event.payload.get("after") - # Debug when payload has no commits - # st += print_v("\n[debug] PushEvent payload has no 'commits' array; using compare API") - # st += print_v(f"[debug] before:\t\t\t{before_sha}") - # st += print_v(f"[debug] head/after:\t\t{head_sha}") - # if size_hint is not None: - # st += print_v(f"[debug] size (hint):\t\t{size_hint}") - if before_sha and head_sha and before_sha != head_sha: try: - compare = gh_call(lambda: repo.compare(before_sha, head_sha))() + compare = gh_call(lambda: repo.compare(before_sha, head_sha), operation="Push comparison")() except Exception as e: verbose_degraded_feature("Push comparison", "complete push event details", e, enrichment=True) compare = None st += print_v(f"* Error using compare({before_sha[:12]}...{head_sha[:12]}): {sanitize_error_text(e)}") if compare: - commits = list(compare.commits) - commits_total = len(commits) + commits = [push_commit_from_compare(entry) for entry in compare.commits] short_repo = getattr(repo, "full_name", repo_name) compare_url = f"{github_web_base()}/{short_repo}/compare/{before_sha[:12]}...{head_sha[:12]}" - st += print_v(f"\nNumber of commits:\t\t{commits_total}") + st += print_v(f"\nNumber of commits:\t\t{len(commits)}") st += print_v(f"Compare URL:\t\t\t{compare_url}") - - for commit_count, c in enumerate(commits, start=1): - st += print_v(f"\n=== Commit {commit_count}/{commits_total} ===") - st += print_v("." * HORIZONTAL_LINE1) - - commit_sha = getattr(c, "sha", None) or getattr(c, "id", None) - if repo and commit_sha: - debug_github_operation("event commit lookup", commit_sha) - commit_details = gh_call(lambda: repo.get_commit(commit_sha))() if (repo and commit_sha) else None # noqa: B023 - - commit_message = commit_details.commit.message if commit_details and commit_details.commit else "" - is_multiline = '\n' in commit_message if commit_message else False - if commit_message: - if is_multiline: - first_line = commit_message.split('\n', 1)[0] - st += print_v(f" - Commit message:\t\t'{first_line}...'") - else: - st += print_v(f" - Commit message:\t\t'{commit_message}'") - - if commit_details: - commit_date = commit_details.commit.author.date - st += print_v(f" - Commit date:\t\t\t{get_date_from_ts(commit_date)}") - - if commit_sha: - st += print_v(f" - Commit SHA:\t\t\t{commit_sha}") - - author_name = None - if commit_details and commit_details.commit and commit_details.commit.author: - author_name = commit_details.commit.author.name - st += print_v(f" - Commit author:\t\t{author_name or 'N/A'}") - - if commit_details and commit_details.author: - st += print_v(f" - Commit author URL:\t\t{commit_details.author.html_url}") - - if commit_details: - st += print_v(f" - Commit URL:\t\t\t{commit_details.html_url}") - st += print_v(f" - Commit raw patch URL:\t{commit_details.html_url}.patch") - - stats = getattr(commit_details, "stats", None) - additions = stats.additions if stats else 0 - deletions = stats.deletions if stats else 0 - stats_total = stats.total if stats else 0 - st += print_v(f"\n - Additions/Deletions:\t\t+{additions} / -{deletions} ({stats_total})") - - try: - file_count = sum(1 for _ in commit_details.files) - except Exception as exc: - verbose_degraded_feature("Commit file list", "complete push event details", exc, enrichment=True) - file_count = "N/A" - st += print_v(f" - Files changed:\t\t{file_count}") - if file_count and file_count != "N/A": - st += print_v(" - Changed files list:") - for f in commit_details.files: - st += print_v(f" • '{f.filename}' - {f.status} (+{f.additions} / -{f.deletions})") - - if is_multiline and commit_message: - st += print_v(f"\n - Commit full message:") - st += print_v(f"\n'{commit_message}'") - - st += print_v("." * HORIZONTAL_LINE1) + st += print_push_commits(repo, commits) else: st += print_v("\nNo compare range available (forced push, tag push, or identical before/after)") @@ -6197,32 +6469,32 @@ def handle_profile_change(label, count_old, count_new, list_old, raw_list, user, m_subject = f"GitHub user {user} {label.lower()} list changed" m_body = (f"{label} list changed {label_context} user {user}\n" f"{removed_mbody}{removed_list_str}{added_mbody}{added_list_str}\n" - f"Check interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + f"Check interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"{label} list changed {label_context} user {html.escape(user)}
" f"{removed_mbody_html if removed_items else ''}{removed_list_str_html if removed_items else ''}" f"{added_mbody_html if added_items else ''}{added_list_str_html if added_items else ''}
" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) else: m_subject = f"GitHub user {user} {label.lower()} number has changed! ({diff_str}, {old_count} -> {new_count})" m_body = (f"{label} number changed {label_context} user {user} from {old_count} to {new_count} ({diff_str})\n" f"{removed_mbody}{removed_list_str}{added_mbody}{added_list_str}\n" - f"Check interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + f"Check interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"{label} number changed {label_context} user {html.escape(user)} from {old_count} to {new_count} ({html.escape(diff_str)})
" f"{removed_mbody_html if removed_items else ''}{removed_list_str_html if removed_items else ''}" f"{added_mbody_html if added_items else ''}{added_list_str_html if added_items else ''}
" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") return list_new, new_count @@ -6246,17 +6518,17 @@ def check_repo_list_changes(count_old, count_new, list_old, list_new, label, rep m_subject = f"GitHub user {user} number of {label.lower()} for repo '{repo_name}' has changed! ({diff_str}, {count_old} -> {count_new})" m_body = (f"* Repo '{repo_name}': number of {label.lower()} changed from {count_old} to {count_new} ({diff_str})\n" f"* Repo URL: {repo_url}\n\n" - f"Check interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + f"Check interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"* Repo '{html.escape(repo_name)}': number of {html.escape(label.lower())} changed from {count_old} to {count_new} ({html.escape(diff_str)})
" f"* Repo URL: {html.escape(repo_url)}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("repo", m_subject, m_body, m_body_html, REPO_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") return @@ -6379,33 +6651,33 @@ def check_repo_list_changes(count_old, count_new, list_old, list_new, label, rep m_subject = f"GitHub user {user} {label.lower()} list changed for repo '{repo_name}'!" m_body = (f"* Repo '{repo_name}': {label.lower()} list changed\n" f"* Repo URL: {repo_url}\n{removed_mbody}{removed_list_str}{added_mbody}{added_list_str}\n" - f"Check interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + f"Check interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"* Repo '{html.escape(repo_name)}': {html.escape(label.lower())} list changed
" f"* Repo URL: {html.escape(repo_url)}
" f"{removed_mbody_html}{removed_list_str_html}" f"{added_mbody_html}{added_list_str_html}
" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) else: m_subject = f"GitHub user {user} number of {label.lower()} for repo '{repo_name}' has changed! ({diff_str}, {old_count} -> {new_count})" m_body = (f"* Repo '{repo_name}': number of {label.lower()} changed from {old_count} to {new_count} ({diff_str})\n" f"* Repo URL: {repo_url}\n{removed_mbody}{removed_list_str}{added_mbody}{added_list_str}\n" - f"Check interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + f"Check interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"* Repo '{html.escape(repo_name)}': number of {html.escape(label.lower())} changed from {old_count} to {new_count} ({html.escape(diff_str)})
" f"* Repo URL: {html.escape(repo_url)}
" f"{removed_mbody_html}{removed_list_str_html}" f"{added_mbody_html}{added_list_str_html}
" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("repo", m_subject, m_body, m_body_html, REPO_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") @@ -6667,7 +6939,7 @@ def load_startup_secrets(env_file=None, configured_settings=None, report_errors= except Exception as exc: env_path = DOTENV_FILE if DOTENV_FILE else None verbose_degraded_feature("Dotenv loading", "dotenv-based private settings", exc) - advice = make_recovery_advice("file.unreadable", "The dotenv file could not be read", recovery_fix_with_guide("Check DOTENV_FILE and its permissions or disable it with --env-file none", CONFIG_GUIDE_URL), False, f"{type(exc).__name__}: {exc}") + advice = make_recovery_advice("file.unreadable", "The dotenv file could not be read", recovery_fix_with_guide("Check DOTENV_FILE and its permissions or disable it with --env-file none", SECRETS_GUIDE_URL), False, f"{type(exc).__name__}: {exc}") if errors_out is not None: errors_out.append(advice.summary + f": {advice.detail}") if report_errors: @@ -7100,7 +7372,7 @@ def is_blocked_by(user): user_endpoint = f"{GITHUB_API_URL}/user" debug_http_request("GET", user_endpoint, "authenticated viewer lookup for block detection", 15, headers=headers, token=GITHUB_TOKEN) - response = req.get(user_endpoint, headers=headers, timeout=15, verify=VERIFY_SSL) + response = gh_call(lambda: req.get(user_endpoint, headers=headers, timeout=15, verify=VERIFY_SSL), operation="Block status", raise_on_failure=True)() debug_http_response("GET", user_endpoint, "authenticated viewer lookup for block detection", response.status_code) if response.status_code != 200: verbose_degraded_feature("Block status", "block and unblock alerts") @@ -7119,7 +7391,7 @@ def is_blocked_by(user): """ payload = {"query": query, "variables": {"login": user}} debug_http_request("POST", graphql_endpoint, "target block relationship lookup", 15, headers=headers, token=GITHUB_TOKEN) - response_graphql = req.post(graphql_endpoint, json=payload, headers=headers, timeout=15, verify=VERIFY_SSL) + response_graphql = gh_call(lambda: req.post(graphql_endpoint, json=payload, headers=headers, timeout=15, verify=VERIFY_SSL), operation="Block status", raise_on_failure=True)() debug_http_response("POST", graphql_endpoint, "target block relationship lookup", response_graphql.status_code) if response_graphql.status_code == 404: @@ -7160,7 +7432,7 @@ def get_starred_count(user): """ payload = {"query": query, "variables": {"login": user}} debug_http_request("POST", graphql_endpoint, "starred repository count", 15, headers=headers, token=GITHUB_TOKEN) - response = req.post(graphql_endpoint, json=payload, headers=headers, timeout=15, verify=VERIFY_SSL) + response = gh_call(lambda: req.post(graphql_endpoint, json=payload, headers=headers, timeout=15, verify=VERIFY_SSL), operation="Starred repository count", raise_on_failure=True)() debug_http_response("POST", graphql_endpoint, "starred repository count", response.status_code) if not response.ok: @@ -7181,7 +7453,7 @@ def has_private_banner(user): try: url = f"{GITHUB_HTML_URL.rstrip('/')}/{user}" debug_http_request("GET", url, "public profile visibility page", 15) - r = req.get(url, timeout=15, verify=VERIFY_SSL) + r = gh_call(lambda: req.get(url, timeout=15, verify=VERIFY_SSL), operation="Profile visibility", raise_on_failure=True)() debug_http_response("GET", url, "public profile visibility page", r.status_code) return r.ok and "activity is private" in r.text.lower() except Exception as exc: @@ -7189,8 +7461,8 @@ def has_private_banner(user): return False -# Returns True if the user's GitHub profile is public -def is_profile_public(g: Github, user, new_account_days=30): +# Returns whether the user's GitHub profile is public, or None when the lookup could not answer +def is_profile_public(g: Github, user, new_account_days=30) -> Optional[bool]: if has_private_banner(user): return False @@ -7208,16 +7480,19 @@ def is_profile_public(g: Github, user, new_account_days=30): try: debug_print("PyGithub", operation="recent public event probe", endpoint=diagnostic_endpoint(GITHUB_API_URL), timeout=f"{PYGITHUB_TIMEOUT_SECONDS}s", token=mask_secret(GITHUB_TOKEN), target=user) - events_iter = iter(u.get_events()) - next(events_iter) + gh_call(lambda: next(iter(u.get_events())), operation="Public profile detection", raise_on_failure=True)() return True except StopIteration as exc: debug_swallowed_exception("Recent public event probe returned no events", exc) - except GithubException as exc: + except NET_ERRORS as exc: + # A probe that never reached GitHub cannot say the profile is private, so the caller is told nothing verbose_degraded_feature("Public profile detection", "profile visibility alerts", exc) + return None - except GithubException as exc: + # Broad on purpose, since a best-effort visibility probe must never end the monitoring run + except Exception as exc: verbose_degraded_feature("Public profile detection", "profile visibility alerts", exc) + return None return False @@ -7279,7 +7554,7 @@ def get_daily_contributions(username: str, start: Optional[dt.date] = None, end: variables = {"login": username, "from": start_iso, "to": end_iso} debug_http_request("POST", url, "daily contribution calendar", 30, headers=headers, token=token) - r = requests.post(url, json={"query": query, "variables": variables}, headers=headers, timeout=30, verify=VERIFY_SSL) + r = gh_call(lambda: requests.post(url, json={"query": query, "variables": variables}, headers=headers, timeout=30, verify=VERIFY_SSL), operation="Daily contribution count", raise_on_failure=True)() # noqa: B023 debug_http_response("POST", url, "daily contribution calendar", r.status_code) r.raise_for_status() data = r.json() @@ -7394,11 +7669,9 @@ def report_monitor_failure(user, advice, error_alert, monitor_recovery_tracker, elif outage_outcome == "changed": print_outage_change(user, advice) elif outage_outcome == "reminder": - print_outage_liveness(user, advice, outage.since, outage.failures) + print_outage_liveness(user, advice, outage.since, outage.failures, close=False) - m_subject = f"{advice.summary} (GitHub user: {user})" - m_body = f"{advice.summary}\n\nTo fix: {advice.fix}\n\nGitHub Monitor will retry in {display_time(GITHUB_CHECK_INTERVAL)}.{get_cur_ts(nl_ch + nl_ch + 'Timestamp: ')}" - m_body_html = f"{html_text(advice.summary)}

To fix: {html_text(advice.fix)}

GitHub Monitor will retry in {html.escape(display_time(GITHUB_CHECK_INTERVAL))}.{get_cur_ts('

Timestamp: ')}" + error_alert.remember(advice, outage.since) # Attempted on every failing check rather than only on the report, so a channel that failed is tried again # A failure the tool can retry away is alerted once the outage has lasted ERROR_ALERT_AFTER_SECONDS, one it cannot at once alert_due = not advice.retryable or int(time.time()) - outage.since >= ERROR_ALERT_AFTER_SECONDS @@ -7406,19 +7679,56 @@ def report_monitor_failure(user, advice, error_alert, monitor_recovery_tracker, error_email_pending = alert_due and error_alert.pending("email", ERROR_NOTIFICATION, now) error_webhook_pending = alert_due and error_alert.pending("webhook", webhook_event_enabled("error"), now) if error_email_pending or error_webhook_pending: - email_delivered, webhook_delivered = send_notification_channels("error", m_subject, m_body, m_body_html, error_email_pending, error_webhook_pending) + m_subject = recovery_alert_subject(advice, user) + # Built once without the timestamp, which the webhook service shows itself and only the email carries + webhook_body = recovery_alert_body(advice, GITHUB_CHECK_INTERVAL, outage.failures, outage.since, timestamp=False) + webhook_body_html = recovery_alert_body_html(advice, GITHUB_CHECK_INTERVAL, outage.failures, outage.since, timestamp=False) + m_body = webhook_body + get_cur_ts(nl_ch + nl_ch + "Timestamp: ") + m_body_html = recovery_alert_body_html(advice, GITHUB_CHECK_INTERVAL, outage.failures, outage.since) + email_delivered, webhook_delivered = send_notification_channels("error", m_subject, m_body, m_body_html, error_email_pending, error_webhook_pending, webhook_body=webhook_body, webhook_body_html=webhook_body_html) error_alert.record("email", error_email_pending, email_delivered, now) error_alert.record("webhook", error_webhook_pending, webhook_delivered, now) delivery_reported = True # A retry can reach the screen on a check the outage reporter keeps quiet, and a delivery line - # with nothing under it reads as a run that stopped there - if outage_outcome in ("full", "changed") or delivery_reported: + # with nothing under it reads as a run that stopped there. The reminder closes last so the lines it + # carries stay inside the report rather than landing under the separator that ended it + if outage_outcome == "reminder": + print_cur_ts("Liveness check, timestamp:\t") + elif outage_outcome in ("full", "changed") or delivery_reported: + print_cur_ts("Timestamp:\t\t\t") + + +# Reports a successful check after an outage and answers each delivered failure alert with a recovery alert on the same channel +def report_monitor_recovery(user, error_alert, outage): + lasted = outage.recovered() + if lasted is not None: + lasted = max(1, lasted) + advice = error_alert.advice + # Gated on the channels the failure alert reached and on their switches, so a channel that never heard of the outage stays quiet + email_owed = advice is not None and error_alert.delivered("email", ERROR_NOTIFICATION) + webhook_owed = advice is not None and error_alert.delivered("webhook", webhook_event_enabled("error")) + # A channel whose failure alert never got through hears about the outage and its end together, rather than + # nothing at all, which is what a channel blocked for the length of the outage would otherwise receive + email_missed = advice is not None and error_alert.missed("email", ERROR_NOTIFICATION) + webhook_missed = advice is not None and error_alert.missed("webhook", webhook_event_enabled("error")) + print_outage_recovery(user, lasted, close=False) + if email_owed or webhook_owed or email_missed or webhook_missed: + m_subject = outage_recovered_alert_subject(user, lasted) + email_text, email_html = (outage_missed_alert_body, outage_missed_alert_body_html) if email_missed else (outage_recovered_alert_body, outage_recovered_alert_body_html) + webhook_text, webhook_html = (outage_missed_alert_body, outage_missed_alert_body_html) if webhook_missed else (outage_recovered_alert_body, outage_recovered_alert_body_html) + webhook_body = webhook_text(advice, user, lasted, timestamp=False) + webhook_body_html = webhook_html(advice, user, lasted, timestamp=False) + m_body = email_text(advice, user, lasted, timestamp=False) + get_cur_ts(nl_ch + nl_ch + "Timestamp: ") + m_body_html = email_html(advice, user, lasted) + send_notification_channels("error", m_subject, m_body, m_body_html, email_owed or email_missed, webhook_owed or webhook_missed, webhook_body=webhook_body, webhook_body_html=webhook_body_html) print_cur_ts("Timestamp:\t\t\t") + error_alert.reset() # Monitors activity of the specified GitHub user def github_monitor_user(user, csv_file_name): + global LAST_CHECK_TS, NET_OUTAGE_CONFIRMED mark_monitoring_started() @@ -7658,6 +7968,9 @@ def github_monitor_user(user, csv_file_name): verbose_notice(f"Initial snapshot completed for {user}") # The snapshot names its features differently from the checks, so its outages are not carried into the loop reset_degraded_features() + # The initial snapshot is what the first check compares against, so the window a change is reported in starts here + LAST_CHECK_TS = int(time.time()) + debug_monitor_wait_timing("initial monitoring interval", GITHUB_CHECK_INTERVAL) time.sleep(GITHUB_CHECK_INTERVAL) alive_since = int(time.time()) @@ -7673,6 +7986,8 @@ def github_monitor_user(user, csv_file_name): check_number += 1 MONITOR_CHECK_FAILURES.clear() DEGRADED_FEATURES_SEEN.clear() + # Each check decides for itself whether the network is reachable, so the breaker never outlives one + NET_OUTAGE_CONFIRMED = False reports_before_check = REPORTS_PRINTED check_started_at = debug_monitor_check_start(check_number, user) @@ -7686,11 +8001,11 @@ def github_monitor_user(user, csv_file_name): user_myself_url = g_user_myself.html_url auth_refresh_version = GITHUB_AUTH_REFRESH_VERSION print("* GitHub API client recreated after token reload") - debug_github_operation("monitored user profile refresh", user) + debug_github_operation("monitored account lookup", user) g_user = g.get_user(user) except (GithubException, Exception) as e: - verbose_degraded_feature("Monitored user refresh", "all profile, repository and event alerts", e) + verbose_degraded_feature("Monitored account lookup", "all profile, repository and event alerts", e) advice = classify_recovery_error(e, "target") report_monitor_failure(user, advice, error_alert, monitor_recovery_tracker, outage) @@ -7702,8 +8017,8 @@ def github_monitor_user(user, csv_file_name): # Changed followings try: debug_github_operation("followings refresh", user) - followings_raw = gh_call(lambda: list(g_user.get_following()), raise_on_failure=True)() # noqa: B023 - followings_count = gh_call(lambda: g_user.following)() # noqa: B023 + followings_raw = gh_call(lambda: list(g_user.get_following()), operation="Followings", raise_on_failure=True)() # noqa: B023 + followings_count = gh_call(lambda: g_user.following, operation="Following count")() # noqa: B023 except NET_ERRORS as e: verbose_degraded_feature("Followings", "following change alerts", e) print_degraded_error("Followings could not be refreshed", e) @@ -7717,8 +8032,8 @@ def github_monitor_user(user, csv_file_name): # Changed followers try: debug_github_operation("followers refresh", user) - followers_raw = gh_call(lambda: list(g_user.get_followers()), raise_on_failure=True)() # noqa: B023 - followers_count = gh_call(lambda: g_user.followers)() # noqa: B023 + followers_raw = gh_call(lambda: list(g_user.get_followers()), operation="Followers", raise_on_failure=True)() # noqa: B023 + followers_count = gh_call(lambda: g_user.followers, operation="Follower count")() # noqa: B023 except NET_ERRORS as e: verbose_degraded_feature("Followers", "follower change alerts", e) print_degraded_error("Followers could not be refreshed", e) @@ -7733,15 +8048,15 @@ def github_monitor_user(user, csv_file_name): try: if GET_ALL_REPOS: debug_github_operation("all repository refresh", user) - repos_raw = gh_call(lambda: list(g_user.get_repos()), raise_on_failure=True)() # noqa: B023 - repos_count = gh_call(lambda: g_user.public_repos)() # noqa: B023 + repos_raw = gh_call(lambda: list(g_user.get_repos()), operation="Public repository list", raise_on_failure=True)() # noqa: B023 + repos_count = gh_call(lambda: g_user.public_repos, operation="Public repository count")() # noqa: B023 else: debug_github_operation("owned repository refresh", user) - repos_raw = gh_call(lambda: [repo for repo in g_user.get_repos(type='owner') if not repo.fork and repo.owner.login == user_login], raise_on_failure=True)() # noqa: B023 + repos_raw = gh_call(lambda: [repo for repo in g_user.get_repos(type='owner') if not repo.fork and repo.owner.login == user_login], operation="Public repository list", raise_on_failure=True)() # noqa: B023 repos_count = len(repos_raw) except NET_ERRORS as e: - verbose_degraded_feature("Repositories", "repository change alerts", e) - print_degraded_error("Repositories could not be refreshed", e) + verbose_degraded_feature("Public repository list", "repository list change alerts", e) + print_degraded_error("The public repository list could not be refreshed", e) print_cur_ts("Timestamp:\t\t\t") repos_raw = None repos_count = None @@ -7752,7 +8067,7 @@ def github_monitor_user(user, csv_file_name): # Changed starred repositories try: debug_github_operation("starred repository refresh", user) - starred_list = gh_call(lambda: list(g_user.get_starred()), raise_on_failure=True)() # noqa: B023 + starred_list = gh_call(lambda: list(g_user.get_starred()), operation="Starred repositories", raise_on_failure=True)() # noqa: B023 starred_count = len(starred_list) except NET_ERRORS as e: verbose_degraded_feature("Starred repositories", "starred repository change alerts", e) @@ -7778,21 +8093,21 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} daily contributions changed from {contrib_old} to {contrib_curr}!" - m_body = (f"GitHub user {user} daily contributions changed on {get_short_date_from_ts(contrib_state['day'], show_hour=False)} from {contrib_old} to {contrib_curr}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}") + m_body = (f"GitHub user {user} daily contributions changed on {get_short_date_from_ts(contrib_state['day'], show_hour=False)} from {contrib_old} to {contrib_curr}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}") m_body_html = ( f"" f"GitHub user {html.escape(user)} daily contributions changed on {html.escape(get_short_date_from_ts(contrib_state['day'], show_hour=False))} from {contrib_old} to {contrib_curr}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("contrib", m_subject, m_body, m_body_html, CONTRIB_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed bio - bio = gh_call(lambda: g_user.bio, default=profile_field_unavailable)() # noqa: B023 + bio = gh_call(lambda: g_user.bio, default=profile_field_unavailable, operation="Profile bio")() # noqa: B023 report_unavailable_profile_field("bio", bio, profile_field_unavailable) if has_nullable_profile_field_changed(bio, bio_old, profile_field_unavailable): print(f"* Bio has changed for user {user} !\n") @@ -7806,26 +8121,26 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} bio has changed!" - m_body = f"GitHub user {user} bio has changed\n\nOld bio:\n\n{bio_old}\n\nNew bio:\n\n{bio}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" - bio_old_html = markdown_to_html(bio_old, convert_line_breaks=True) if bio_old else "" - bio_html = markdown_to_html(bio, convert_line_breaks=True) if bio else "" + m_body = f"GitHub user {user} bio has changed\n\nOld bio:\n\n{bio_old}\n\nNew bio:\n\n{bio}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" + bio_old_html = markdown_to_html(bio_old, convert_line_breaks=True) if bio_old else html.escape(str(bio_old)) + bio_html = markdown_to_html(bio, convert_line_breaks=True) if bio else html.escape(str(bio)) m_body_html = ( f"" f"GitHub user {html.escape(user)} bio has changed

" f"Old bio:

{bio_old_html}

" f"New bio:

{bio_html}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) bio_old = bio - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed location - location = gh_call(lambda: g_user.location, default=profile_field_unavailable)() # noqa: B023 + location = gh_call(lambda: g_user.location, default=profile_field_unavailable, operation="Profile location")() # noqa: B023 report_unavailable_profile_field("location", location, profile_field_unavailable) if has_nullable_profile_field_changed(location, location_old, profile_field_unavailable): print(f"* Location has changed for user {user} !\n") @@ -7839,24 +8154,24 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} location has changed!" - m_body = f"GitHub user {user} location has changed\n\nOld location: {location_old}\n\nNew location: {location}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} location has changed\n\nOld location: {location_old}\n\nNew location: {location}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" m_body_html = ( f"" f"GitHub user {html.escape(user)} location has changed

" - f"Old location: {html.escape(location_old or '')}

" - f"New location: {html.escape(location or '')}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Old location: {html.escape(str(location_old))}

" + f"New location: {html.escape(str(location))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) location_old = location - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed user name - user_name = gh_call(lambda: g_user.name, default=profile_field_unavailable)() # noqa: B023 + user_name = gh_call(lambda: g_user.name, default=profile_field_unavailable, operation="Profile name")() # noqa: B023 report_unavailable_profile_field("name", user_name, profile_field_unavailable) if has_nullable_profile_field_changed(user_name, user_name_old, profile_field_unavailable): print(f"* User name has changed for user {user} !\n") @@ -7870,24 +8185,24 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} name has changed!" - m_body = f"GitHub user {user} name has changed\n\nOld user name: {user_name_old}\n\nNew user name: {user_name}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} name has changed\n\nOld user name: {user_name_old}\n\nNew user name: {user_name}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" m_body_html = ( f"" f"GitHub user {html.escape(user)} name has changed

" - f"Old user name: {html.escape(user_name_old or '')}

" - f"New user name: {html.escape(user_name or '')}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Old user name: {html.escape(str(user_name_old))}

" + f"New user name: {html.escape(str(user_name))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) user_name_old = user_name - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed company - company = gh_call(lambda: g_user.company, default=profile_field_unavailable)() # noqa: B023 + company = gh_call(lambda: g_user.company, default=profile_field_unavailable, operation="Profile company")() # noqa: B023 report_unavailable_profile_field("company", company, profile_field_unavailable) if has_nullable_profile_field_changed(company, company_old, profile_field_unavailable): print(f"* User company has changed for user {user} !\n") @@ -7901,24 +8216,24 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} company has changed!" - m_body = f"GitHub user {user} company has changed\n\nOld company: {company_old}\n\nNew company: {company}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} company has changed\n\nOld company: {company_old}\n\nNew company: {company}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" m_body_html = ( f"" f"GitHub user {html.escape(user)} company has changed

" - f"Old company: {html.escape(company_old or '')}

" - f"New company: {html.escape(company or '')}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Old company: {html.escape(str(company_old))}

" + f"New company: {html.escape(str(company))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) company_old = company - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed email - email = gh_call(lambda: g_user.email, default=profile_field_unavailable)() # noqa: B023 + email = gh_call(lambda: g_user.email, default=profile_field_unavailable, operation="Profile email")() # noqa: B023 report_unavailable_profile_field("email", email, profile_field_unavailable) if has_nullable_profile_field_changed(email, email_old, profile_field_unavailable): print(f"* User email has changed for user {user} !\n") @@ -7932,24 +8247,24 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} email has changed!" - m_body = f"GitHub user {user} email has changed\n\nOld email: {email_old}\n\nNew email: {email}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} email has changed\n\nOld email: {email_old}\n\nNew email: {email}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" m_body_html = ( f"" f"GitHub user {html.escape(user)} email has changed

" - f"Old email: {html.escape(email_old or '')}

" - f"New email: {html.escape(email or '')}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Old email: {html.escape(str(email_old))}

" + f"New email: {html.escape(str(email))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) email_old = email - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed blog URL - blog = gh_call(lambda: g_user.blog, default=profile_field_unavailable)() # noqa: B023 + blog = gh_call(lambda: g_user.blog, default=profile_field_unavailable, operation="Profile blog")() # noqa: B023 report_unavailable_profile_field("blog URL", blog, profile_field_unavailable) if has_nullable_profile_field_changed(blog, blog_old, profile_field_unavailable): print(f"* User blog URL has changed for user {user} !\n") @@ -7963,16 +8278,24 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} blog URL has changed!" - m_body = f"GitHub user {user} blog URL has changed\n\nOld blog URL: {blog_old}\n\nNew blog URL: {blog}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} blog URL has changed\n\nOld blog URL: {blog_old}\n\nNew blog URL: {blog}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body_html = ( + f"" + f"GitHub user {html.escape(user)} blog URL has changed

" + f"Old blog URL: {html_autolink_urls(html.escape(str(blog_old)))}

" + f"New blog URL: {html_autolink_urls(html.escape(str(blog)))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" + f"" + ) - send_notification_channels("profile", m_subject, m_body, "", PROFILE_NOTIFICATION) + send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) blog_old = blog - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Changed account update date - account_updated_date = gh_call(lambda: g_user.updated_at)() # noqa: B023 + account_updated_date = gh_call(lambda: g_user.updated_at, operation="Profile update date")() # noqa: B023 if account_updated_date is not None and account_updated_date != account_updated_date_old: print(f"* User account has been updated for user {user} ! (after {calculate_timespan(account_updated_date, account_updated_date_old, show_seconds=False, granularity=2)})\n") print(f"Old account update date:\t{get_date_from_ts(account_updated_date_old)}\n") @@ -7985,17 +8308,29 @@ def github_monitor_user(user, csv_file_name): print_csv_write_error(e) m_subject = f"GitHub user {user} account has been updated! (after {calculate_timespan(account_updated_date, account_updated_date_old, show_seconds=False, granularity=2)})" - m_body = f"GitHub user {user} account has been updated (after {calculate_timespan(account_updated_date, account_updated_date_old, show_seconds=False, granularity=2)})\n\nOld account update date: {get_date_from_ts(account_updated_date_old)}\n\nNew account update date: {get_date_from_ts(account_updated_date)}\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} account has been updated (after {calculate_timespan(account_updated_date, account_updated_date_old, show_seconds=False, granularity=2)})\n\nOld account update date: {get_date_from_ts(account_updated_date_old)}\n\nNew account update date: {get_date_from_ts(account_updated_date)}\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" + updated_timespan = calculate_timespan(account_updated_date, account_updated_date_old, show_seconds=False, granularity=2) + m_body_html = ( + f"" + f"GitHub user {html.escape(user)} account has been updated (after {html.escape(updated_timespan)})

" + f"Old account update date: {html.escape(get_date_from_ts(account_updated_date_old))}

" + f"New account update date: {html.escape(get_date_from_ts(account_updated_date))}

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" + f"" + ) - send_notification_channels("profile", m_subject, m_body, "", PROFILE_NOTIFICATION) + send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) account_updated_date_old = account_updated_date - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Profile visibility changed public = is_profile_public(g, user) - if public != public_old: + # A lookup that could not answer neither alerts nor becomes the baseline the next check compares against + if public is not None and public_old is None: + public_old = public + if public is not None and public != public_old: def _get_profile_status(public): return "public" if public else "private" @@ -8009,12 +8344,18 @@ def _get_profile_status(public): print_csv_write_error(e) m_subject = f"GitHub user {user} has changed profile visibility to '{_get_profile_status(public)}' !" - m_body = f"GitHub user {user} has changed profile visibility to '{_get_profile_status(public)}' !\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} has changed profile visibility to '{_get_profile_status(public)}' !\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body_html = ( + f"" + f"GitHub user {html.escape(user)} has changed profile visibility to '{html.escape(_get_profile_status(public))}' !

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" + f"" + ) - send_notification_channels("profile", m_subject, m_body, "", PROFILE_NOTIFICATION) + send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) public_old = public - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Blocked status changed @@ -8037,12 +8378,18 @@ def _get_blocked_status(blocked, public): print_csv_write_error(e) m_subject = f"GitHub user {user} has {'blocked' if blocked else 'unblocked'} you!" - m_body = f"GitHub user {user} has {'blocked' if blocked else 'unblocked'} you!\n\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} has {'blocked' if blocked else 'unblocked'} you!\n\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body_html = ( + f"" + f"GitHub user {html.escape(user)} has {'blocked' if blocked else 'unblocked'} you!

" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" + f"" + ) - send_notification_channels("profile", m_subject, m_body, "", PROFILE_NOTIFICATION) + send_notification_channels("profile", m_subject, m_body, m_body_html, PROFILE_NOTIFICATION) blocked_old = blocked - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") list_of_repos = [] @@ -8052,10 +8399,10 @@ def _get_blocked_status(blocked, public): try: if GET_ALL_REPOS: - repos_list = gh_call(lambda: list(g_user.get_repos()), raise_on_failure=True)() # noqa: B023 + repos_list = gh_call(lambda: list(g_user.get_repos()), operation="Public repository list", raise_on_failure=True)() # noqa: B023 else: debug_github_operation("owned repository detail refresh", user) - repos_list = gh_call(lambda: [repo for repo in g_user.get_repos(type='owner') if not repo.fork and repo.owner.login == user_login], raise_on_failure=True)() # noqa: B023 + repos_list = gh_call(lambda: [repo for repo in g_user.get_repos(type='owner') if not repo.fork and repo.owner.login == user_login], operation="Public repository list", raise_on_failure=True)() # noqa: B023 except NET_ERRORS as e: repos_list = None verbose_degraded_feature("Repository detail feed", "repository detail alerts", e) @@ -8142,7 +8489,7 @@ def _get_blocked_status(blocked, public): except Exception as e: print_csv_write_error(e) m_subject = f"GitHub user {user} repo '{r_name}' update date has changed ! (after {calculate_timespan(r_update, r_update_old, show_seconds=False, granularity=2)})" - m_body = f"{r_message}\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"{r_message}\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" timespan_str = calculate_timespan(r_update, r_update_old, show_seconds=False, granularity=2) m_body_html = ( f"" @@ -8150,11 +8497,11 @@ def _get_blocked_status(blocked, public): f"* Repo URL: {html.escape(r_url)}

" f"Old repo update date: {html.escape(get_date_from_ts(r_update_old))}

" f"New repo update date: {html.escape(get_date_from_ts(r_update))}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("repo_update", m_subject, m_body, m_body_html, REPO_UPDATE_DATE_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") # Number of stars for repo changed @@ -8190,7 +8537,7 @@ def _get_blocked_status(blocked, public): except Exception as e: print_csv_write_error(e) m_subject = f"GitHub user {user} repo '{r_name}' description has changed !" - m_body = f"{r_message}\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"{r_message}\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" r_descr_old_html = markdown_to_html(r_descr_old, convert_line_breaks=True) if r_descr_old else "" r_descr_html = markdown_to_html(r_descr, convert_line_breaks=True) if r_descr else "" m_body_html = ( @@ -8200,11 +8547,11 @@ def _get_blocked_status(blocked, public): f"to:

" f"'{r_descr_html}'

" f"* Repo URL: {html.escape(r_url)}

" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("repo", m_subject, m_body, m_body_html, REPO_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") list_of_repos_old = list_of_repos @@ -8213,7 +8560,7 @@ def _get_blocked_status(blocked, public): if not DO_NOT_MONITOR_GITHUB_EVENTS: debug_github_operation("recent event refresh", user) try: - events = gh_call(lambda: list(islice(g_user.get_events(), EVENTS_NUMBER)), raise_on_failure=True)() # noqa: B023 + events = gh_call(lambda: list(islice(g_user.get_events(), EVENTS_NUMBER)), operation="Recent events", raise_on_failure=True)() # noqa: B023 except NET_ERRORS as e: events = None verbose_degraded_feature("Recent events", "new event alerts", e) @@ -8275,7 +8622,7 @@ def _get_blocked_status(blocked, public): print_csv_write_error(e) m_subject = f"GitHub user {user} has new {event.type} (repo: {repo_name})" - m_body = f"GitHub user {user} has new {event.type} event\n\n{event_text}\nCheck interval: {display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)}){get_cur_ts(nl_ch + 'Timestamp: ')}" + m_body = f"GitHub user {user} has new {event.type} event\n\n{event_text}\nCheck interval: {check_window_text()}{get_cur_ts(nl_ch + 'Timestamp: ')}" event_payload = None try: if hasattr(event, 'payload'): @@ -8287,13 +8634,13 @@ def _get_blocked_status(blocked, public): f"" f"GitHub user {html.escape(user)} has new {html.escape(event.type)} event

" f"{event_text_html}
" - f"Check interval: {html.escape(display_time(GITHUB_CHECK_INTERVAL))} ({html.escape(get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True))}){get_cur_ts('
Timestamp: ')}" + f"Check interval: {check_window_html()}{get_cur_ts('
Timestamp: ')}" f"" ) send_notification_channels("event", m_subject, m_body, m_body_html, EVENT_NOTIFICATION) - print(f"Check interval:\t\t\t{display_time(GITHUB_CHECK_INTERVAL)} ({get_range_of_dates_from_tss(int(time.time()) - GITHUB_CHECK_INTERVAL, int(time.time()), short=True)})") + print(f"Check interval:\t\t\t{check_window_text()}") print_cur_ts("Timestamp:\t\t\t") last_event_id_old = last_event_id @@ -8303,7 +8650,7 @@ def _get_blocked_status(blocked, public): verbose_degraded_feature("Recent events", "new event alerts") if MONITOR_CHECK_FAILURES: - failures = [(feature, classify_recovery_error(error) if error is not None else make_recovery_advice("github.api_error", "The monitoring check did not return usable data", recovery_fix_with_guide("Check connectivity and resource access, then let the next check retry", DIAGNOSTICS_GUIDE_URL), True)) for feature, error in MONITOR_CHECK_FAILURES.items()] + failures = [(feature, classify_recovery_error(error) if error is not None else make_recovery_advice("github.api_error", "The monitoring check did not return usable data", recovery_fix_with_guide("Check connectivity and resource access, then let the next check retry", CONNECTION_GUIDE_URL), True)) for feature, error in MONITOR_CHECK_FAILURES.items()] feature, advice = next(((feature, advice) for feature, advice in failures if not advice.retryable), failures[0]) # One failure carries the fix, but an alert that hides the rest understates the outage. The # count rather than the names keeps the text stable while a per-repository failure set changes @@ -8312,11 +8659,8 @@ def _get_blocked_status(blocked, public): advice = make_recovery_advice(advice.code, summary, advice.fix, advice.retryable, advice.detail) report_monitor_failure(user, advice, error_alert, monitor_recovery_tracker, outage) else: - error_alert.reset() + report_monitor_recovery(user, error_alert, outage) monitor_recovery_tracker.reset() - outage_lasted = outage.recovered() - if outage_lasted is not None: - print_outage_recovery(user, outage_lasted) report_recovered_features() close_pending_notice_block() @@ -8328,6 +8672,9 @@ def _get_blocked_status(blocked, public): print_liveness_banner(f"Monitoring healthy for {user}. No tracked change since the last check") alive_since = int(time.time()) + # Only a check that got this far advanced the baselines, so a failing check leaves the window where it was + LAST_CHECK_TS = int(time.time()) + debug_monitor_check_timing(check_number, user, check_started_at, GITHUB_CHECK_INTERVAL, outcome="degraded" if MONITOR_CHECK_FAILURES else "OK") debug_monitor_wait_timing("normal monitoring interval", GITHUB_CHECK_INTERVAL) time.sleep(GITHUB_CHECK_INTERVAL) @@ -8379,9 +8726,13 @@ def apply_webhook_cli_overrides(args: argparse.Namespace, parser: argparse.Argum # Applies monitoring, output and email command-line overrides to effective settings def apply_monitoring_cli_overrides(args: argparse.Namespace, parser: argparse.ArgumentParser, strict=True) -> None: - global CSV_FILE, DISABLE_LOGGING, PROFILE_NOTIFICATION, EVENT_NOTIFICATION, REPO_NOTIFICATION, REPO_UPDATE_DATE_NOTIFICATION, ERROR_NOTIFICATION, GITHUB_CHECK_INTERVAL, LIVENESS_REMINDER_SECONDS, DO_NOT_MONITOR_GITHUB_EVENTS, TRACK_REPOS_CHANGES, REPOS_TO_MONITOR, GET_ALL_REPOS, CONTRIB_NOTIFICATION, TRACK_CONTRIB_CHANGES, WEBHOOK_REPO_NOTIFICATION, WEBHOOK_REPO_UPDATE_DATE_NOTIFICATION, WEBHOOK_CONTRIB_NOTIFICATION, WEBHOOK_EVENT_NOTIFICATION + global CSV_FILE, DISABLE_LOGGING, PROFILE_NOTIFICATION, EVENT_NOTIFICATION, REPO_NOTIFICATION, REPO_UPDATE_DATE_NOTIFICATION, ERROR_NOTIFICATION, GITHUB_CHECK_INTERVAL, LIVENESS_REMINDER_SECONDS, DO_NOT_MONITOR_GITHUB_EVENTS, TRACK_REPOS_CHANGES, REPOS_TO_MONITOR, GET_ALL_REPOS, CONTRIB_NOTIFICATION, TRACK_CONTRIB_CHANGES, WEBHOOK_REPO_NOTIFICATION, WEBHOOK_REPO_UPDATE_DATE_NOTIFICATION, WEBHOOK_CONTRIB_NOTIFICATION, WEBHOOK_EVENT_NOTIFICATION, PUSH_COMMITS_LIMIT, PUSH_FILES_LIMIT if args.check_interval is not None: GITHUB_CHECK_INTERVAL = args.check_interval + if getattr(args, "push_commits_limit", None) is not None: + PUSH_COMMITS_LIMIT = args.push_commits_limit + if getattr(args, "push_files_limit", None) is not None: + PUSH_FILES_LIMIT = args.push_files_limit if args.csv_file is not None: CSV_FILE = os.path.expanduser(args.csv_file) elif CSV_FILE: @@ -8612,6 +8963,8 @@ def runtime_configuration_errors(): positive_numbers = (("CHECK_INTERNET_TIMEOUT", CHECK_INTERNET_TIMEOUT),) nonnegative_numbers = (("NET_BASE_BACKOFF_SEC", NET_BASE_BACKOFF_SEC),) positive_integers = (("GITHUB_CHECK_INTERVAL", GITHUB_CHECK_INTERVAL), ("EVENTS_NUMBER", EVENTS_NUMBER), ("NET_MAX_RETRIES", NET_MAX_RETRIES)) + nonnegative_integers = (("PUSH_COMMITS_LIMIT", PUSH_COMMITS_LIMIT), ("PUSH_FILES_LIMIT", PUSH_FILES_LIMIT)) + choices = (("PUSH_COMMITS_ORDER", PUSH_COMMITS_ORDER, ("newest", "oldest")), ("PUSH_COMMITS_OVERFLOW", PUSH_COMMITS_OVERFLOW, ("summary", "count"))) for name, value in positive_numbers: if not finite_number(value) or value <= 0: errors.append(f"{name} must be a number greater than zero, not {value!r}") @@ -8621,6 +8974,12 @@ def runtime_configuration_errors(): for name, value in positive_integers: if not isinstance(value, int) or isinstance(value, bool) or value <= 0: errors.append(f"{name} must be an integer greater than zero, not {value!r}") + for name, value in nonnegative_integers: + if not isinstance(value, int) or isinstance(value, bool) or value < 0: + errors.append(f"{name} must be an integer zero or greater, not {value!r}") + for name, value, allowed in choices: + if not isinstance(value, str) or value.strip().casefold() not in allowed: + errors.append(f"{name} must be {' or '.join(repr(choice) for choice in allowed)}, not {value!r}") if not isinstance(SMTP_PORT, int) or isinstance(SMTP_PORT, bool) or not 1 <= SMTP_PORT <= 65535: errors.append(f"SMTP_PORT must be an integer from 1 through 65535, not {SMTP_PORT!r}") return errors @@ -8882,7 +9241,7 @@ def doctor_check_monitoring(report, contribution_checker=None): doctor_probe_feed(operation, factory) report.add("Monitoring", "PASS", label) except Exception as exc: - advice = make_recovery_advice("github.api_error", label.replace(" is accessible", " is unavailable"), recovery_fix_with_guide("Check target visibility, token access and GitHub API availability", DIAGNOSTICS_GUIDE_URL), False) + advice = make_recovery_advice("github.api_error", label.replace(" is accessible", " is unavailable"), recovery_fix_with_guide("Check target visibility, token access and GitHub API availability", QUICK_START_GUIDE_URL), False) report.add("Monitoring", "FAIL", advice.summary, f"{type(exc).__name__}: {sanitize_error_text(exc)}", advice) if DO_NOT_MONITOR_GITHUB_EVENTS: report.add("Monitoring", "PASS", "GitHub event monitoring is disabled", "No event feed check was needed") @@ -8900,7 +9259,7 @@ def doctor_check_monitoring(report, contribution_checker=None): checker(report.target_name, today_local(), report.github_token) report.add("Monitoring", "PASS", "Daily contribution feed is accessible") except Exception as exc: - advice = make_recovery_advice("github.api_error", "Daily contribution feed is unavailable", recovery_fix_with_guide("Check token access, timezone and GitHub GraphQL availability", DIAGNOSTICS_GUIDE_URL), False) + advice = make_recovery_advice("github.api_error", "Daily contribution feed is unavailable", recovery_fix_with_guide("Check token access, timezone and GitHub GraphQL availability", QUICK_START_GUIDE_URL), False) report.add("Monitoring", "FAIL", advice.summary, f"{type(exc).__name__}: {sanitize_error_text(exc)}", advice) elif TRACK_CONTRIB_CHANGES: report.add("Monitoring", "SKIP", "Daily contribution feed was not checked", "The target profile was not fetched, so no lookup was attempted") @@ -10787,7 +11146,7 @@ def main(): dest="notify_errors", action="store_false", default=None, - help="Disable email on errors" + help="Disable email on errors and the recovery alert that follows" ) notify.add_argument( "--send-test-email", @@ -10866,14 +11225,14 @@ def main(): dest="webhook_errors", action="store_true", default=None, - help="Send webhook alerts when monitoring has a problem" + help="Send webhook alerts when monitoring has a problem and the recovery alert that follows" ) webhook_error_toggle.add_argument( "--no-webhook-error-notify", dest="webhook_errors", action="store_false", default=None, - help="Disable webhook alerts when monitoring has a problem" + help="Disable webhook alerts when monitoring has a problem and the recovery alert that follows" ) webhook_notify.add_argument( "--send-test-webhook", @@ -10981,6 +11340,20 @@ def main(): type=int, help="Max characters per screen line (not log), use 999 to auto-detect terminal width, ignored if -d is set" ) + opts.add_argument( + "--push-commits-limit", + dest="push_commits_limit", + metavar="N", + type=int, + help="Max commits of one push event reported in full, use 0 for no limit, the rest are summarized" + ) + opts.add_argument( + "--push-files-limit", + dest="push_files_limit", + metavar="N", + type=int, + help="Max changed files listed per commit, use 0 for no limit" + ) opts.add_argument( "-m", "--track-contribs-changes", dest="track_contribs_changes", diff --git a/grc/conf.monitor_logs b/grc/conf.monitor_logs index cef474c..a2ffa5f 100644 --- a/grc/conf.monitor_logs +++ b/grc/conf.monitor_logs @@ -12,7 +12,7 @@ # repository green event, post, trophy, achievement, rank bright_green # branch, reel bright_magenta date, date_range magenta # timestamp_value cyan info cyan link blue underline -# warning, signal, status_change yellow error red +# warning, signal, status_change, status_away yellow error red # email bright_cyan webhook bright_blue # boolean_true, count_up, duration, status_active green # boolean_false, count_down, status_inactive, status_offline red @@ -25,6 +25,10 @@ # paint a drop red, which needs both values at once, so both are green here. They also know # whether a monitored target is a handle or an opaque identifier, so here a target is read as a # handle unless it has the shape of a Spotify URI ID or a Steam64 ID. +# +# The padded listing tables have no rule here. Their columns are positional, so a rule wide enough +# to read one tool's table reads across the column boundaries of another's. The tools colour those +# rows at the point they print them instead. - @@ -142,7 +146,7 @@ regexp=^[\t ]*(?:[-*][\t ]+)?(?:Last track duration|Match duration|Match finishe colours=unchanged, green count=stop - -regexp=^[\t ]*(?:[-*][\t ]+)?(?:Current game|Title name|User is currently in-game|In-game|Game|Commit message|IP Address|Proxy IP Address|Proxy URL|Proxy Certificate|Proxy for Webhooks):[\t ]+(\S.*?)(?:[\t ]*\((PS[A-Z0-9_]*|MOBILE_APP)\))?[\t ]*$ +regexp=^[\t ]*(?:[-*][\t ]+)?(?:Current game|Title name|User is currently in-game|In-game|Game|Commit message|IP Address|Proxy IP Address|Proxy URL|Proxy Certificate|Proxy for Webhooks):[\t ]+(\S.*?)(?:[\t ]*\((PS[A-Z0-9_]*|MOBILE_APP|Xbox One Series X/S|Android Phone/Tablet|Xbox One X/S|iPhone/iPad|Xbox 360|Xbox One|Windows|Android|XONEX|X360|XONE|XSX)\))?[\t ]*$ colours=unchanged, bright_yellow, blue count=stop - @@ -179,6 +183,24 @@ count=stop regexp=^(-[\t ]+)([^:(\n]+?)([\t ]+\()([^()]+)(\)[\t ]*)$ colours=unchanged, unchanged, bright_cyan underline, unchanged, bright_yellow, unchanged count=stop +- +# A friends list row names one player and the presence Xbox reports for them. The state word is mixed case +# here, unlike the capitalised keyword an event line prints, so these rules do not reach a presence report +regexp=^([^\s:*-][^:\n]*?)([\t ]{2,})(Online)([\t ]+\()([^()\n]+)(\))[\t ]*$ +colours=unchanged, bright_cyan underline, unchanged, green, unchanged, bright_yellow, unchanged +count=stop +- +regexp=^([^\s:*-][^:\n]*?)([\t ]{2,})(Online)[\t ]*$ +colours=unchanged, bright_cyan underline, unchanged, green +count=stop +- +regexp=^([^\s:*-][^:\n]*?)([\t ]{2,})(Away)[\t ]*$ +colours=unchanged, bright_cyan underline, unchanged, yellow +count=stop +- +regexp=^([^\s:*-][^:\n]*?)([\t ]{2,})(Offline)[\t ]*$ +colours=unchanged, bright_cyan underline, unchanged, red +count=stop - @@ -287,21 +309,35 @@ colours=unchanged, unchanged, bright_yellow, unchanged regexp=\b(?:changed status|changed game)\b colours=yellow - -# A status named in prose rather than in a Status: row -regexp=(?i)changed status from[\t ]+(online|available|active)\b +# A status named in prose rather than in a Status: row, read through the same table that row uses. The +# value between "from" and "to" may be two words, so the middle is not matched as a single word +regexp=(?i)changed status from[\t ]+(online|available|active|playing|private mode)\b colours=unchanged, green - -regexp=(?i)changed status from[\t ]+(offline|inactive|standby|away)\b +regexp=(?i)changed status from[\t ]+(away)\b +colours=unchanged, yellow +- +regexp=(?i)changed status from[\t ]+(snooze)\b +colours=unchanged, magenta +- +regexp=(?i)changed status from[\t ]+(offline|invisible|inactive|standby)\b colours=unchanged, red - -regexp=(?i)changed status from[\t ]+\w+[\t ]+to[\t ]+(online|available|active)\b +regexp=(?i)changed status from[\t ]+[\w ]+?[\t ]+to[\t ]+(online|available|active|playing|private mode)\b colours=unchanged, green - -regexp=(?i)changed status from[\t ]+\w+[\t ]+to[\t ]+(offline|inactive|standby|away)\b +regexp=(?i)changed status from[\t ]+[\w ]+?[\t ]+to[\t ]+(away)\b +colours=unchanged, yellow +- +regexp=(?i)changed status from[\t ]+[\w ]+?[\t ]+to[\t ]+(snooze)\b +colours=unchanged, magenta +- +regexp=(?i)changed status from[\t ]+[\w ]+?[\t ]+to[\t ]+(offline|invisible|inactive|standby)\b colours=unchanged, red - -# The console a game was launched on -regexp=\((PS[A-Z0-9_]*|MOBILE_APP)\) +# The console a game was launched on. Only the names the tools print are listed, so a bracketed word in +# ordinary prose is not read as a platform +regexp=\((PS[A-Z0-9_]*|MOBILE_APP|Xbox One Series X/S|Android Phone/Tablet|Xbox One X/S|iPhone/iPad|Xbox 360|Xbox One|Windows|Android|XONEX|X360|XONE|XSX)\) colours=unchanged, blue - @@ -397,10 +433,12 @@ colours=blue underline regexp=(?=7.0", "pyyaml>=6.0", "wcwidth>=0.2.7"] # Pinned so a new ruff release cannot fail CI on a rule that did not exist when the change was written -lint = ["ruff==0.16.7"] +lint = ["ruff==0.16.8"] [project.urls] Homepage = "https://github.com/misiektoja/github_monitor" diff --git a/tests/README.md b/tests/README.md index ec2a8f2..858801b 100644 --- a/tests/README.md +++ b/tests/README.md @@ -37,6 +37,7 @@ and again before anything is published to PyPI. | File | Area under test | | --- | --- | +| `test_codeql_workflow.py` | Source suppression filtering, retained security findings, invalid reports and CodeQL upload ordering | | `test_repository_closure.py` | Verified issue, PR and discussion closures, retained snapshots, shared request limits, fair rotation and transport settings | | `test_notification_receipts.py` | SMTP acceptance despite cleanup failures, receipt controls and unchanged notification content | | `test_configuration_notification_boundaries.py` | Invalid output settings, CLI precedence and strict webhook fields with legacy JSON support | @@ -62,8 +63,11 @@ and again before anything is published to PyPI. | `test_github_token_setup.py` | Hidden token entry, targeted dotenv updates and refusal to save an invalid token | | `test_help_screen.py` | The `--help` screen: the shared argument group names, the task-grouped examples and the startup banner | | `test_install_method_commands.py` | PyPI and downloaded-script detection with portable plus exact POSIX and Windows commands | +| `test_email_html.py` | HTML notification bodies: escaping, the Discord markdown form and the plain-text match | | `test_monitoring_loop.py` | The primary monitoring loop driven through outages: the error alert on both channels, once per failure category, retried per channel and re-armed after a recovery | +| `test_missed_alert_recovery.py` | The recovery alert sent to a channel that never received the failure alert | | `test_profile_fields.py` | Addition, removal and failure handling for nullable profile fields | +| `test_push_commit_limits.py` | Push commit detail limits: the kept range, summarized and counted overflow, the changed-file cap, rejected settings and the startup summary rows | | `test_recovery_errors.py` | Closed recovery codes, classification, retryability and layered secret redaction that leaves ordinary output intact | | `test_repository_contracts.py` | Governance documents, issue templates, action pinning, release gating, the CI contract and the documentation site: its pinned page set, one title per page, no section on two pages, resolving links and the runtime guide URLs | | `test_repository_metadata.py` | Governance files, citation, funding, line endings, the declared editor style, the pinned linter and release integrity | diff --git a/tests/test_codeql_workflow.py b/tests/test_codeql_workflow.py new file mode 100644 index 0000000..763813e --- /dev/null +++ b/tests/test_codeql_workflow.py @@ -0,0 +1,102 @@ +"""CodeQL upload filtering keeps unsuppressed security findings reportable.""" + +import copy +import json +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + + +PROJECT_ROOT = Path(__file__).resolve().parents[1] +FILTER_SCRIPT = PROJECT_ROOT / ".github" / "scripts" / "filter_codeql_sarif.py" + + +@pytest.mark.parametrize("tool_name", ["CodeQL", "CodeQL command-line toolchain"]) +# Keeps unsuppressed findings and metadata in reports from both CodeQL tool names +def test_filter_keeps_unsuppressed_results_and_metadata(tmp_path, tool_name): + suppressions = [ + None, + [], + [{"kind": "inSource"}], + [{"kind": "inSource", "status": "accepted"}], + [{"kind": "inSource", "status": "rejected"}], + [{"kind": "inSource", "status": "underReview"}], + [{"kind": "external", "status": "accepted"}], + [{"status": "accepted"}], + [{"kind": "inSource", "status": None}], + ] + results = [{ + "ruleId": "py/clear-text-logging-sensitive-data", + "message": {"text": str(index)}, + "suppressions": value, + "partialFingerprints": {"primaryLocationLineHash": str(index)}, + } for index, value in enumerate(suppressions)] + results.append({"ruleId": "py/clear-text-logging-sensitive-data", "message": {"text": "No suppression property"}}) + report = {"version": "2.1.0", "runs": [ + {"tool": {"driver": {"name": tool_name}}, "automationDetails": {"id": "/language:python/"}, "results": results}, + {"tool": {"driver": {"name": "CodeQL"}}, "results": [results[2]]}, + {"tool": {"driver": {"name": "CodeQL"}}}, + {"tool": {"driver": {"name": "Other scanner"}}, "results": [results[2]]}, + ]} + expected = copy.deepcopy(report) + expected["runs"][0]["results"] = [result for index, result in enumerate(results) if index not in (2, 3)] + expected["runs"][1]["results"] = [] + source = tmp_path / "input.sarif" + destination = tmp_path / "output.sarif" + original = json.dumps(report) + source.write_text(original, encoding="utf-8") + + completed = subprocess.run([sys.executable, str(FILTER_SCRIPT), str(source), str(destination)], check=True, capture_output=True, text=True) + + assert json.loads(destination.read_text(encoding="utf-8")) == expected + assert source.read_text(encoding="utf-8") == original + assert completed.stdout == "Applied 3 CodeQL source suppressions\n" + + +@pytest.mark.parametrize("content", ["not JSON", "{}", '{"version": "2.1.0", "runs": []}', '{"version": "2.1.0", "runs": [{"tool": {"driver": {"name": "CodeQL"}}, "results": null}]}']) +# Rejects invalid reports before creating an uploadable file +def test_filter_fails_before_upload_on_invalid_report(tmp_path, content): + source = tmp_path / "input.sarif" + destination = tmp_path / "output.sarif" + source.write_text(content, encoding="utf-8") + + completed = subprocess.run([sys.executable, str(FILTER_SCRIPT), str(source), str(destination)], capture_output=True, text=True) + + assert completed.returncode != 0 + assert not destination.exists() + + +# Prevents accidental replacement of the original analysis evidence +def test_filter_preserves_original_report_when_paths_match(tmp_path): + source = tmp_path / "input.sarif" + original = '{"version": "2.1.0", "runs": [{"tool": {"driver": {"name": "CodeQL"}}}]}' + source.write_text(original, encoding="utf-8") + + completed = subprocess.run([sys.executable, str(FILTER_SCRIPT), str(source), str(source)], capture_output=True, text=True) + + assert completed.returncode != 0 + assert source.read_text(encoding="utf-8") == original + + +# Uploads only the filtered report after successful analysis and filtering +def test_workflow_filters_before_upload(): + workflow = yaml.safe_load((PROJECT_ROOT / ".github" / "workflows" / "codeql.yml").read_text(encoding="utf-8")) + steps = workflow["jobs"]["analyze"]["steps"] + initialize = next(step for step in steps if step.get("uses", "").startswith("github/codeql-action/init@")) + analyze = next(step for step in steps if step.get("uses", "").startswith("github/codeql-action/analyze@")) + apply = next(step for step in steps if "filter_codeql_sarif.py" in step.get("run", "")) + upload = next(step for step in steps if step.get("uses", "").startswith("github/codeql-action/upload-sarif@")) + + assert initialize["with"]["queries"] == "security-extended" + assert initialize["with"]["packs"] == "codeql/python-queries:AlertSuppression.ql" + assert analyze["with"]["upload"] == "failure-only" + assert apply["run"] == "python .github/scripts/filter_codeql_sarif.py sarif-results/python.sarif sarif-results/python-filtered.sarif" + assert analyze["with"]["output"] == "sarif-results" + assert upload["with"]["sarif_file"] == "sarif-results/python-filtered.sarif" + assert upload["with"]["category"] == analyze["with"]["category"] + assert steps.index(analyze) < steps.index(apply) < steps.index(upload) + assert "if" not in upload + assert not any(step.get("continue-on-error") for step in (analyze, apply, upload)) diff --git a/tests/test_diagnostic_modes.py b/tests/test_diagnostic_modes.py index c49ea62..c83ee10 100644 --- a/tests/test_diagnostic_modes.py +++ b/tests/test_diagnostic_modes.py @@ -105,8 +105,8 @@ def test_degraded_feature_lines_are_closed_once_by_the_check(gm_module, monkeypa gm_module.close_pending_notice_block() lines = [line for line in capsys.readouterr().out.splitlines() if line.strip()] - assert lines[0] == "* Block status is unavailable, so block and unblock alerts cannot fire" - assert lines[1] == "* Starred repository count is unavailable, so starred repository change alerts cannot fire" + assert lines[0] == "* Block status: unavailable, so block and unblock alerts cannot fire" + assert lines[1] == "* Starred repository count: unavailable, so starred repository change alerts cannot fire" assert lines[2].startswith("Timestamp:") assert set(lines[3]) == {"\u2500"} @@ -136,7 +136,7 @@ def test_a_lasting_degraded_feature_is_reported_once(gm_module, monkeypatch, cap gm_module.report_recovered_features() output = capsys.readouterr().out - assert output.count("Block status is unavailable, so block and unblock alerts cannot fire") == 1 + assert output.count("Block status: unavailable, so block and unblock alerts cannot fire") == 1 assert "is available again" not in output @@ -153,7 +153,7 @@ def test_a_recovered_feature_is_reported_once(gm_module, monkeypatch, capsys): gm_module.report_recovered_features() output = capsys.readouterr().out - assert output.count("Block status is available again, so block and unblock alerts can fire again") == 1 + assert output.count("Block status: available again, so block and unblock alerts can fire again") == 1 # Verifies a verbose notice stays silent while verbose mode is off, so the trailer cannot leak into a quiet run @@ -211,7 +211,7 @@ def test_swallowed_exception_reports_degraded_feature(gm_module, monkeypatch, ca assert gm_module.is_blocked_by("octocat") is None output = capsys.readouterr().out - assert "Block status is unavailable, so block and unblock alerts cannot fire" in output + assert "Block status: unavailable, so block and unblock alerts cannot fire" in output assert "Block status: outcome=degraded, error=ConnectionError" in output assert token not in output diff --git a/tests/test_email_html.py b/tests/test_email_html.py new file mode 100644 index 0000000..68082a7 --- /dev/null +++ b/tests/test_email_html.py @@ -0,0 +1,212 @@ +"""Tests for the HTML notification body, its Discord markdown form and its match with the plain text.""" + +import datetime +import difflib +import html as html_module +import json +import os +import re +from itertools import count +from pathlib import Path + +import pytest +import requests + +from test_monitoring_loop import OUTAGE, FakeGithub, FakeUser, LoopStopped + +USER = "watched" + + +# Reduces one HTML body back to the text it represents, independently of the module's own converter +def html_to_text(body_html): + text = re.sub(r"(?is)", "", str(body_html or "")) + text = re.sub(r"(?is)", "\n", text) + text = re.sub(r"(?s)<[^>]+>", "", text) + return html_module.unescape(text) + + +# Replaces every anchor with its destination, which is what a plain body prints on a bare URL line +def links_as_destinations(line): + return re.sub(r'(?is)]*?href="([^"]*)"[^>]*>.*?', r"\1", line) + + +# Reports whether one HTML line says what its plain counterpart says, whether it links a name or a bare URL +def line_agrees(plain_line, html_line): + return plain_line in (html_to_text(html_line), html_to_text(links_as_destinations(html_line))) + + +# Returns what differs between the plain body and the HTML body, empty when their lines and blank lines match +def structural_diff(body, body_html): + plain_lines = body.split("\n") + html_lines = re.sub(r"(?is)", "", str(body_html or "")).split("
") + if len(plain_lines) == len(html_lines) and all(line_agrees(*pair) for pair in zip(plain_lines, html_lines, strict=True)): + return "" + return "\n".join(difflib.unified_diff(plain_lines, [html_to_text(line) for line in html_lines], fromfile="plain", tofile="html-reduced", lineterm="")) + + +# Builds the account the loop sees after the change, carrying every profile field a check compares +def changed_user(): + user = FakeUser(USER) + user.name = "Renamed Person" + user.location = "Warsaw" + user.bio = "Writes monitors" + user.company = "Example Inc" + user.email = "watched@example.test" + user.blog = "https://example.test/blog" + user.updated_at = datetime.datetime(2026, 2, 1, tzinfo=datetime.timezone.utc) + return user + + +# Runs the real loop against the scripted lookups and returns every alert it handed to the delivery helper +def alerts_from_loop(gm_module, monkeypatch, tmp_path, lookups, stop_after, visibility=None, blocked=None): + captured = [] + sleeps = [] + now = [1_800_000_000.0] + + def record(notification_type, subject, body, body_html="", email_enabled=False, webhook_enabled=None, webhook_body=None, webhook_body_html=None): + captured.append({"type": notification_type, "subject": subject, "body": body, "body_html": body_html, "webhook_body": body if webhook_body is None else webhook_body, "discord": gm_module.html_body_to_discord_markdown(body_html if webhook_body_html is None else webhook_body_html)}) + return bool(email_enabled), bool(webhook_enabled) + + def stopping_sleep(seconds): + sleeps.append(seconds) + now[0] += seconds + if len(sleeps) >= stop_after: + raise LoopStopped + + # Answers the block-status check without reaching GitHub, which is not what these tests drive + def transport(session, request, **kwargs): + response = requests.Response() + response.request = request + response.url = request.url + response.status_code = 200 + response._content = json.dumps({"login": "viewer"} if request.url.endswith("/user") else {"data": {"user": {"viewerCanFollow": True}}}).encode() + return response + + monkeypatch.chdir(tmp_path) + monkeypatch.setattr(gm_module.time, "sleep", stopping_sleep) + monkeypatch.setattr(gm_module.time, "time", lambda: now[0]) + monkeypatch.setattr(gm_module, "GITHUB_CHECK_INTERVAL", 300) + monkeypatch.setattr(gm_module, "LIVENESS_REMINDER_SECONDS", 30000) + monkeypatch.setattr(gm_module, "PROFILE_NOTIFICATION", True) + monkeypatch.setattr(gm_module, "ERROR_NOTIFICATION", True) + monkeypatch.setattr(gm_module, "WEBHOOK_ENABLED", True) + monkeypatch.setattr(gm_module, "WEBHOOK_ERROR_NOTIFICATION", True) + monkeypatch.setattr(gm_module, "TRACK_REPOS_CHANGES", False) + monkeypatch.setattr(gm_module, "TRACK_CONTRIB_CHANGES", False) + monkeypatch.setattr(gm_module, "GET_ALL_REPOS", False) + monkeypatch.setattr(gm_module, "DEBUG_MODE", False) + monkeypatch.setattr(gm_module, "VERBOSE_MODE", False) + monkeypatch.setattr(requests.Session, "send", transport) + monkeypatch.setattr(gm_module, "is_profile_public", visibility or (lambda *arguments, **keywords: True)) + if blocked is not None: + monkeypatch.setattr(gm_module, "is_blocked_by", blocked) + monkeypatch.setattr(gm_module, "send_notification_channels", record) + monkeypatch.setattr(gm_module, "create_github_client", lambda *arguments, **keywords: FakeGithub(lookups)) + with pytest.raises(LoopStopped): + gm_module.github_monitor_user(USER, "") + return captured + + +@pytest.fixture +# Every alert a timeline covering the profile changes, a visibility flip, a block and an outage produces +def timeline_alerts(gm_module, monkeypatch, tmp_path, capsys): + visibility_checks = count(1) + blocked_checks = count(1) + alerts = alerts_from_loop( + gm_module, monkeypatch, tmp_path, + [changed_user(), changed_user(), changed_user(), OUTAGE], + 5, + visibility=lambda *arguments, **keywords: next(visibility_checks) < 4, + blocked=lambda *arguments, **keywords: next(blocked_checks) >= 3, + ) + capsys.readouterr() + return alerts + + +# Verifies a value taken from GitHub is escaped before it reaches the HTML body +def test_untrusted_text_is_escaped(gm_module): + assert gm_module.html_text("") == "<script>alert(1)</script>" + assert gm_module.html_text("line\nbreak") == "line
break" + + +# Verifies a bare URL in an alert becomes a link while one already inside an attribute is left alone +def test_bare_urls_are_linked_once(gm_module): + assert gm_module.html_autolink_urls("Guide: https://example.test/a") == 'Guide: https://example.test/a' + assert gm_module.html_autolink_urls('x') == 'x' + + +# Verifies the Discord body carries the email's emphasis and links instead of raw markup +def test_discord_markdown_mirrors_the_html_body(gm_module): + body_html = f'GitHub user {USER} bio has changed

Guide: docs' + + assert gm_module.html_body_to_discord_markdown(body_html) == f"GitHub user **{USER}** bio has changed\n\nGuide: [docs](https://example.test/a)" + + +# Verifies the failure alert bolds its summary and the two values that say how bad the outage is +def test_the_failure_alert_bolds_its_summary_and_outage_fields(gm_module): + advice = gm_module.make_recovery_advice("github.unavailable", "GitHub is unreachable", gm_module.recovery_fix_with_guide("Retry later", "https://example.test/guide"), True) + + rendered = gm_module.recovery_alert_body_html(advice, 60, failed_checks=2, failing_since=1800000000) + + assert rendered.startswith("GitHub is unreachable

") + assert 'Guide: ' in rendered + assert "Failed checks in a row: 2" in rendered + assert "Failing since: " in rendered + # The retry delay is configured rather than observed, so it carries no emphasis + assert "Next retry in: 1 minute" in rendered + assert rendered.endswith("") + + +# Verifies the timeline reaches both alert types, so the structural check is not silently narrow +def test_the_timeline_covers_the_profile_and_failure_alerts(timeline_alerts): + assert {alert["type"] for alert in timeline_alerts} == {"profile", "error"} + + +# Verifies the timeline reaches the profile fields that had no HTML body of their own +def test_the_timeline_covers_the_fields_without_their_own_html_body(timeline_alerts): + subjects = " | ".join(alert["subject"] for alert in timeline_alerts) + + assert "blog URL has changed" in subjects + assert "account has been updated" in subjects + assert "changed profile visibility" in subjects + assert "blocked you" in subjects + + +# Verifies every alert carries an HTML body next to its plain one +def test_every_alert_has_an_html_body(timeline_alerts): + assert [alert["subject"] for alert in timeline_alerts if not alert["body_html"]] == [] + + +# Verifies each HTML body reduces back to its plain body, so no line break was added or lost +def test_html_bodies_match_the_plain_text(timeline_alerts): + mismatches = [f"{alert['type']}: {alert['subject']}\n{structural_diff(alert['body'], alert['body_html'])}" for alert in timeline_alerts if structural_diff(alert["body"], alert["body_html"])] + + assert mismatches == [] + + +# Verifies every HTML body is one complete document, so no fragment reaches a mail client unwrapped +def test_html_bodies_are_complete_documents(timeline_alerts): + for alert in timeline_alerts: + assert alert["body_html"].startswith("") + assert alert["body_html"].endswith("") + + +# Verifies the Discord body keeps the wording the ntfy body carries once its markers are removed +def test_discord_bodies_keep_the_plain_wording(timeline_alerts): + for alert in timeline_alerts: + stripped = re.sub(r"\[([^\]]*)\]\((?:[^)]*)\)", r"\1", alert["discord"]).replace("**", "").replace("*", "") + + assert stripped == alert["webhook_body"].strip() + + +# Verifies the watched account is the bold subject of every alert that names it +def test_alerts_bold_the_account_they_name(timeline_alerts): + for alert in timeline_alerts: + if alert["type"] == "profile": + assert f"{USER}" in alert["body_html"] + + +# Writes the captured alerts as JSON when PREVIEW_ALERTS_JSON names a destination, so a preview tool can render them +@pytest.mark.skipif(not os.environ.get("PREVIEW_ALERTS_JSON"), reason="set PREVIEW_ALERTS_JSON to dump the alerts") +def test_dump_the_alerts_for_a_preview(timeline_alerts): + Path(os.environ["PREVIEW_ALERTS_JSON"]).write_text(json.dumps(timeline_alerts, indent=2), encoding="utf-8") diff --git a/tests/test_event_configuration.py b/tests/test_event_configuration.py index 1750e25..1fd1ffa 100644 --- a/tests/test_event_configuration.py +++ b/tests/test_event_configuration.py @@ -26,3 +26,15 @@ def test_gh_call_returns_configured_default(gm_module): result = gm_module.gh_call(failure, retries=2, backoff=0, default=marker)() assert result is marker assert failure.call_count == 2 + + +# Confirms a retry line names the work and the classified failure rather than the wrapped lambda and a raw library message +def test_gh_call_retry_line_names_the_operation(gm_module, capsys): + failure = Mock(side_effect=requests.exceptions.ConnectTimeout("HTTPSConnectionPool(host='api.github.com', port=443): Read timed out.")) + + gm_module.gh_call(failure, retries=2, backoff=0, operation="Public repository list")() + + output = capsys.readouterr().out + assert "* Public repository list failed: GitHub did not answer in time (retry 1/2)" in output + assert "" not in output + assert "HTTPSConnectionPool" not in output diff --git a/tests/test_github_feed_retries.py b/tests/test_github_feed_retries.py index 4cdc533..dc56e69 100644 --- a/tests/test_github_feed_retries.py +++ b/tests/test_github_feed_retries.py @@ -1,5 +1,7 @@ """Exercise retry exhaustion through real PyGithub clients and paginated responses.""" +import ast +import inspect import json from urllib.parse import urlsplit @@ -200,6 +202,155 @@ def send(session, request, **kwargs): monkeypatch.setattr(requests.Session, "send", send) with pytest.raises(MonitorComplete): gm.github_monitor_user("watched", "") - assert deliveries == [2, 5] + # The failed feed alerts on check 2, the complete check on 4 answers it with a recovery alert and check 5 fails again + assert deliveries == [2, 4, 5] output = capsys.readouterr().out assert "GitHub rejected the configured token" in output + + +# Every monitoring request the tool makes, whichever transport it uses +RETRYING_FEEDS = { + "starred count": lambda gm: gm.get_starred_count("watched"), + "profile visibility": lambda gm: gm.has_private_banner("watched"), + "block status": lambda gm: gm.is_blocked_by("watched"), + "daily contributions": lambda gm: gm.get_daily_contributions("watched", token="synthetic-token"), +} + + +# Installs a transport that refuses every request and counts the attempts it was given +def refusing_transport(monkeypatch, calls, error=None): + def send(session, request, **kwargs): + calls.append(request.url) + raise error or requests.ConnectionError("Connection refused") + + monkeypatch.setattr(requests.sessions.Session, "send", send) + + +@pytest.mark.parametrize("feed", sorted(RETRYING_FEEDS)) +# One blip must not kill one feed while another rides it out, so every feed shares the same retry schedule +def test_every_monitoring_feed_retries_a_transport_failure(gm_module, monkeypatch, restored_globals, feed): + gm = gm_module + monkeypatch.setattr(gm, "GITHUB_TOKEN", "synthetic-token") + monkeypatch.setattr(gm, "NET_MAX_RETRIES", 3) + monkeypatch.setattr(gm, "NET_BASE_BACKOFF_SEC", 0) + calls = [] + refusing_transport(monkeypatch, calls) + + try: + RETRYING_FEEDS[feed](gm) + except Exception: + # Some feeds report the failure to their caller and some swallow it, but both retried first + pass + + assert len(calls) == 3 + + +# A check that already proved the network unreachable learns nothing from making every later feed wait for it +def test_a_confirmed_outage_stops_the_later_calls_from_retrying(gm_module, monkeypatch, restored_globals): + gm = gm_module + monkeypatch.setattr(gm, "GITHUB_TOKEN", "synthetic-token") + monkeypatch.setattr(gm, "NET_MAX_RETRIES", 4) + monkeypatch.setattr(gm, "NET_BASE_BACKOFF_SEC", 0) + calls = [] + refusing_transport(monkeypatch, calls) + + gm.get_starred_count("watched") + first = len(calls) + gm.has_private_banner("watched") + + assert first == 4 + assert len(calls) - first == 1 + assert gm.NET_OUTAGE_CONFIRMED is True + + +# The breaker must not outlast the outage, so a call that gets through arms the full schedule again +def test_a_call_that_gets_through_rearms_the_retry_schedule(gm_module, monkeypatch, restored_globals): + gm = gm_module + monkeypatch.setattr(gm, "GITHUB_TOKEN", "synthetic-token") + monkeypatch.setattr(gm, "NET_MAX_RETRIES", 3) + monkeypatch.setattr(gm, "NET_BASE_BACKOFF_SEC", 0) + monkeypatch.setattr(gm, "NET_OUTAGE_CONFIRMED", True) + calls = [] + + def send(session, request, **kwargs): + calls.append(request.url) + if len(calls) == 1: + return github_response(request, {"data": {"user": {"starredRepositories": {"totalCount": 7}}}}) + raise requests.ConnectionError("Connection refused") + + monkeypatch.setattr(requests.sessions.Session, "send", send) + + assert gm.get_starred_count("watched") == 7 + assert gm.NET_OUTAGE_CONFIRMED is False + gm.has_private_banner("watched") + assert len(calls) - 1 == 3 + + +# A failure that answers the same way every time is not worth a schedule, and a local limit gets worse for one +@pytest.mark.parametrize("error", [None, "descriptors"]) +def test_a_permanent_failure_is_not_retried(gm_module, monkeypatch, restored_globals, error): + gm = gm_module + monkeypatch.setattr(gm, "GITHUB_TOKEN", "synthetic-token") + monkeypatch.setattr(gm, "NET_MAX_RETRIES", 5) + monkeypatch.setattr(gm, "NET_BASE_BACKOFF_SEC", 0) + calls = [] + + if error == "descriptors": + refusing_transport(monkeypatch, calls, requests.ConnectionError("Connection failed")) + monkeypatch.setattr(gm, "is_too_many_open_files", lambda failure: True) + else: + def send(session, request, **kwargs): + calls.append(request.url) + return github_response(request, {"message": "Not Found"}, 404) + + monkeypatch.setattr(requests.sessions.Session, "send", send) + + client = gm.Github(auth=gm.Auth.Token("synthetic-token")) + with pytest.raises(gm.NET_ERRORS): + gm.gh_call(lambda: client.get_user("watched").name, operation="Profile name", raise_on_failure=True)() + + assert len(calls) == 1 + + +# A retry line and the degraded notice that follows it report one fetch, so a reader must not see two names +@pytest.mark.parametrize("feed,name", [("starred count", "Starred repository count"), ("profile visibility", "Profile visibility"), ("block status", "Block status")]) +def test_a_retry_line_and_its_degraded_notice_name_one_feed(gm_module, monkeypatch, capsys, restored_globals, feed, name): + gm = gm_module + monkeypatch.setattr(gm, "GITHUB_TOKEN", "synthetic-token") + monkeypatch.setattr(gm, "VERBOSE_MODE", True) + monkeypatch.setattr(gm, "NET_MAX_RETRIES", 2) + monkeypatch.setattr(gm, "NET_BASE_BACKOFF_SEC", 0) + refusing_transport(monkeypatch, []) + + RETRYING_FEEDS[feed](gm) + + output = capsys.readouterr().out + assert f"* {name} failed:" in output + assert f"{name}: unavailable" in output + + +# Collects the labels a retry line prints beside the feature names a degraded notice reports +def retry_labels_and_feature_names(gm): + tree = ast.parse(inspect.getsource(gm)) + raising, features = set(), set() + for node in ast.walk(tree): + if not isinstance(node, ast.Call): + continue + name = node.func.id if isinstance(node.func, ast.Name) else getattr(node.func, "attr", "") + keywords = {keyword.arg: keyword.value for keyword in node.keywords} + if name == "gh_call": + operation = keywords.get("operation") + raises = keywords.get("raise_on_failure") + if isinstance(operation, ast.Constant) and isinstance(raises, ast.Constant) and raises.value: + raising.add(operation.value) + if name == "verbose_degraded_feature" and node.args and isinstance(node.args[0], ast.Constant): + features.add(node.args[0].value) + return raising, features + + +# A fetch whose caller reports it under another name reads as two separate failures a few lines apart +def test_every_retry_label_matches_the_feature_its_caller_reports(gm_module): + raising, features = retry_labels_and_feature_names(gm_module) + + assert raising, "the sweep stopped finding raising calls, update its matching" + assert not raising - features, "a retry line names a fetch its degraded notice calls something else: " + ", ".join(sorted(raising - features)) diff --git a/tests/test_missed_alert_recovery.py b/tests/test_missed_alert_recovery.py new file mode 100644 index 0000000..06f0194 --- /dev/null +++ b/tests/test_missed_alert_recovery.py @@ -0,0 +1,45 @@ +#!/usr/bin/env python3 +"""Covers the recovery alert sent to a channel that never received the failure alert.""" + +import github_monitor as monitor + + +# Records the alert a dispatcher hands the channels instead of delivering it +class RecordingChannels: + def __init__(self): + self.calls = [] + + def __call__(self, notification_type, subject, body, body_html="", email_enabled=False, webhook_enabled=None, **keywords): + self.calls.append({"subject": subject, "body": body, "body_html": body_html, "webhook_body": keywords.get("webhook_body", ""), "webhook_description": keywords.get("webhook_description", ""), "email": email_enabled, "webhook": webhook_enabled}) + return bool(email_enabled), bool(webhook_enabled) + + +# Verifies a channel whose failure alert never landed is told about the outage and its end together, since a +# channel blocked for the length of the outage would otherwise hear nothing at all +def test_a_channel_that_missed_the_failure_alert_is_told_about_the_whole_outage(monkeypatch): + channels = RecordingChannels() + monkeypatch.setattr(monitor, "send_notification_channels", channels) + monkeypatch.setattr(monitor, "ERROR_NOTIFICATION", True) + monkeypatch.setattr(monitor, "webhook_event_enabled", lambda notification_type: True) + monkeypatch.setattr(monitor, "LOCAL_TIMEZONE", "UTC") + state = monitor.ErrorAlertState() + state.email_sent = True + state.webhook_failures = 1 + state.advice = monitor.RecoveryAdvice("service.unavailable", "The service is temporarily unavailable", "Wait for the service to recover", True) + state.since = int(monitor.time.time()) - 600 + outage = monitor.OutageReporter() + outage.code = "service.unavailable" + outage.reported = True + outage.since = int(monitor.time.time()) - 600 + monkeypatch.setattr(monitor, "print_outage_recovery", lambda *arguments, **keywords: None) + monkeypatch.setattr(monitor, "print_cur_ts", lambda *arguments, **keywords: None) + + monitor.report_monitor_recovery("watched-user", state, outage) + + assert len(channels.calls) == 1 + call = channels.calls[0] + assert (call["email"], call["webhook"]) == (True, True) + assert call["body"].startswith("Monitoring recovered for watched-user after 10 minutes.") + assert call["webhook_body"].startswith("Monitoring failed for watched-user at ") + assert "The failure was: The service is temporarily unavailable" in call["webhook_body"] + assert call["webhook_body"].endswith("The failure alert could not be delivered here while the failure lasted.") diff --git a/tests/test_monitoring_loop.py b/tests/test_monitoring_loop.py index 96d7783..73c7291 100644 --- a/tests/test_monitoring_loop.py +++ b/tests/test_monitoring_loop.py @@ -3,6 +3,7 @@ import datetime import json from itertools import count +from typing import Optional import pytest import requests @@ -35,7 +36,11 @@ def __init__(self, login): self.login = login self.name = "Watched Person" self.html_url = f"https://github.com/{login}" - self.location = self.bio = self.company = self.email = self.blog = None + self.location: Optional[str] = None + self.bio: Optional[str] = None + self.company: Optional[str] = None + self.email: Optional[str] = None + self.blog: Optional[str] = None self.created_at = datetime.datetime(2020, 1, 1, tzinfo=datetime.timezone.utc) self.updated_at = datetime.datetime(2026, 1, 1, tzinfo=datetime.timezone.utc) self.followers = self.following = self.public_repos = 0 @@ -139,9 +144,9 @@ def test_any_failure_alerts_both_channels_once(gm_module, monkeypatch, tmp_path) errors = error_alerts_for(gm_module, monkeypatch, tmp_path, [OUTAGE], [(True, True)], 6) assert [(call["email"], call["webhook"]) for call in errors] == [(True, True)] - assert errors[0]["subject"] == "An unexpected error stopped the requested action (GitHub user: watched)" + assert errors[0]["subject"] == "GitHub Monitor error: An unexpected error stopped the requested action (user: watched)" assert "To fix:" in errors[0]["body"] - assert f"retry in {gm_module.display_time(300)}" in errors[0]["body"] + assert f"Next retry in: {gm_module.display_time(300)}" in errors[0]["body"] # The guide link sits under the fix in the HTML body too, since HTML renders the newline the fix carries as a space @@ -150,7 +155,7 @@ def test_the_guide_link_keeps_its_own_line_in_the_html_body(gm_module, monkeypat parts = errors[0]["body_html"].split("
") fix_index = next(index for index, part in enumerate(parts) if part.startswith("To fix: ")) - assert parts[fix_index + 1].startswith("Guide: https://") + assert parts[fix_index + 1].startswith('Guide:
") + assert built_in["PUSH_COMMITS_LIMIT"] == 10 + assert built_in["PUSH_COMMITS_ORDER"] == "newest" + assert built_in["PUSH_COMMITS_OVERFLOW"] == "count" + assert built_in["PUSH_FILES_LIMIT"] == 20 + + +# Confirms the detailed range keeps the head of the push by default +def test_detail_range_keeps_the_newest_commits(gm_module): + assert gm_module.push_detail_range(300, limit=10, order="newest") == (290, 300) + + +# Confirms the detailed range can keep the start of the pushed range instead +def test_detail_range_keeps_the_oldest_commits(gm_module): + assert gm_module.push_detail_range(300, limit=10, order="oldest") == (0, 10) + + +# Confirms a push at or below the limit is reported in full +def test_detail_range_covers_a_small_push(gm_module): + assert gm_module.push_detail_range(4, limit=10, order="newest") == (0, 4) + + +# Confirms a zero limit turns the cap off +def test_detail_range_without_a_limit_covers_everything(gm_module): + assert gm_module.push_detail_range(300, limit=0, order="newest") == (0, 300) + + +# Confirms an oversized push spends one request per detailed commit and none on the rest +def test_large_push_only_requests_details_for_the_kept_commits(gm_module, capsys): + repo = FakeRepo() + _, _, _, text = gm_module.github_print_event(_push_event(25), FakeGithub(repo)) + capsys.readouterr() + assert len(repo.requested) == 10 + assert "Number of commits:\t\t25" in text + assert "Detailed commits:\t\t10 of 25 (newest)" in text + assert "=== Commit 16/25 ===" in text + assert "=== Commit 15/25 ===" not in text + + +# Confirms the summary mode gives the commits left undetailed one line each +def test_skipped_commits_can_be_summarized(gm_module, monkeypatch, capsys): + monkeypatch.setattr(gm_module, "PUSH_COMMITS_OVERFLOW", "summary") + _, _, _, text = gm_module.github_print_event(_push_event(25), FakeGithub(FakeRepo())) + capsys.readouterr() + assert "Commits 1-15 (summary only):" in text + assert f"• {1:040x}"[:14] in text + assert "- 'Commit 1'" in text + + +# Confirms the built-in overflow mode replaces the skipped commits with a single line +def test_skipped_commits_are_reported_as_a_count(gm_module, capsys): + _, _, _, text = gm_module.github_print_event(_push_event(25), FakeGithub(FakeRepo())) + capsys.readouterr() + assert "Commits 1-15 not reported in full (15 commits, PUSH_COMMITS_LIMIT is 10)" in text + assert "(summary only)" not in text + + +# Confirms the oldest order details the start of the pushed range +def test_oldest_order_details_the_start_of_the_push(gm_module, monkeypatch, capsys): + monkeypatch.setattr(gm_module, "PUSH_COMMITS_ORDER", "oldest") + _, _, _, text = gm_module.github_print_event(_push_event(25), FakeGithub(FakeRepo())) + capsys.readouterr() + assert "Detailed commits:\t\t10 of 25 (oldest)" in text + assert "=== Commit 10/25 ===" in text + assert "Commits 11-25 not reported in full" in text + + +# Confirms a push within the limit reports every commit in full and adds no limit notice +def test_small_push_is_reported_in_full(gm_module, capsys): + repo = FakeRepo() + _, _, _, text = gm_module.github_print_event(_push_event(3), FakeGithub(repo)) + capsys.readouterr() + assert len(repo.requested) == 3 + assert "Detailed commits:" not in text + assert "summary only" not in text + + +# Confirms one commit cannot fill the report with changed-file lines +def test_changed_file_list_is_capped(gm_module, monkeypatch, capsys): + monkeypatch.setattr(gm_module, "PUSH_FILES_LIMIT", 3) + _, _, _, text = gm_module.github_print_event(_push_event(1), FakeGithub(FakeRepo(files_per_commit=10))) + capsys.readouterr() + assert "Files changed:\t\t10" in text + assert "file2.py" in text + assert "file4.py" not in text + assert "... and 7 more files" in text + + +# Confirms every changed file is listed when the file limit is off +def test_changed_file_list_can_be_uncapped(gm_module, monkeypatch, capsys): + monkeypatch.setattr(gm_module, "PUSH_FILES_LIMIT", 0) + _, _, _, text = gm_module.github_print_event(_push_event(1), FakeGithub(FakeRepo(files_per_commit=10))) + capsys.readouterr() + assert "file9.py" in text + assert "more files" not in text + + +# Confirms unusable push limits are reported rather than silently ignored +@pytest.mark.parametrize(("name", "value", "expected"), [("PUSH_COMMITS_LIMIT", -1, "PUSH_COMMITS_LIMIT must be an integer zero or greater"), ("PUSH_FILES_LIMIT", "many", "PUSH_FILES_LIMIT must be an integer zero or greater"), ("PUSH_COMMITS_ORDER", "middle", "PUSH_COMMITS_ORDER must be 'newest' or 'oldest'"), ("PUSH_COMMITS_OVERFLOW", "", "PUSH_COMMITS_OVERFLOW must be 'summary' or 'count'")]) +def test_invalid_push_settings_are_reported(gm_module, monkeypatch, name, value, expected): + monkeypatch.setattr(gm_module, name, value) + assert any(error.startswith(expected) for error in gm_module.runtime_configuration_errors()) + + +# Builds the startup rows describing the push limits +def _push_rows(gm_module): + return [row for row in gm_module.build_startup_summary("octocat", "github_monitor.conf", None, "out.log") if row.label.startswith("Push ")] + + +# Confirms the startup summary names every push limit in effect +def test_startup_summary_reports_the_push_limits(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "DO_NOT_MONITOR_GITHUB_EVENTS", False) + assert [(row.label, row.value) for row in _push_rows(gm_module)] == [("Push commit details", "10 newest per push, rest by count"), ("Push changed files", "20 per commit")] + + +# Confirms the push limits reach the screen in verbose and debug mode and stay out of the concise summary +def test_push_limit_rows_are_part_of_the_full_summary(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "DO_NOT_MONITOR_GITHUB_EVENTS", False) + rows = _push_rows(gm_module) + full, concise = io.StringIO(), io.StringIO() + gm_module.emit_startup_summary(rows, show_full=True, stream=full) + gm_module.emit_startup_summary(rows, show_full=False, stream=concise) + assert "Push commit details" in full.getvalue() and "Push changed files" in full.getvalue() + assert concise.getvalue().strip() == "" + + +# Confirms an uncapped run says so rather than printing a limit nothing applies +def test_startup_summary_reports_uncapped_push_reporting(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "DO_NOT_MONITOR_GITHUB_EVENTS", False) + monkeypatch.setattr(gm_module, "PUSH_COMMITS_LIMIT", 0) + monkeypatch.setattr(gm_module, "PUSH_FILES_LIMIT", 0) + assert [row.value for row in _push_rows(gm_module)] == ["Every commit in full", "Every changed file"] + + +# Confirms the push limits are reported as inactive when no event is monitored +def test_startup_summary_reports_push_limits_as_inactive(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "DO_NOT_MONITOR_GITHUB_EVENTS", True) + assert [row.value for row in _push_rows(gm_module)] == ["Inactive", "Inactive"] diff --git a/tests/test_recovery_errors.py b/tests/test_recovery_errors.py index 84ff553..dae37a4 100644 --- a/tests/test_recovery_errors.py +++ b/tests/test_recovery_errors.py @@ -397,8 +397,8 @@ def test_a_failed_refresh_names_the_list_and_carries_a_fix(gm_module, capsys): gm_module.print_degraded_error("Followers could not be refreshed", req.ConnectionError("no route to host")) printed = capsys.readouterr().out - assert "* Error: Followers could not be refreshed: The configured service could not be reached" in printed - assert "To fix: Check the network and configured service URL then try again" in printed + assert "* Error: Followers could not be refreshed: GitHub could not be reached" in printed + assert "To fix: Usually nothing to do, the tool retries on its own." in printed assert "Guide: " in printed @@ -720,3 +720,40 @@ def test_the_loop_tracks_the_error_alert_through_the_state(gm_module): assert source.count('error_alert.pending("email"') == source.count('error_alert.record("email"') >= 1 assert source.count('error_alert.pending("webhook"') == source.count('error_alert.record("webhook"') >= 1 assert not re.search(r"^\s*error_(email|webhook)_sent = ", source, re.MULTILINE) + + +# The verbose and debug section explains the output flags, so a network failure sent there finds nothing about its cause +@pytest.mark.parametrize("error", [req.Timeout("timed out"), req.ConnectionError("connection refused")]) +def test_network_failures_link_to_the_connection_guide(gm_module, error): + advice = gm_module.classify_recovery_error(error) + + assert advice.retryable is True + assert f"\nGuide: {gm_module.CONNECTION_GUIDE_URL}" in advice.fix + assert advice.fix.startswith("Usually nothing to do, the tool retries on its own.") + assert "--doctor" not in advice.fix + assert "--debug" not in advice.fix + + +# Verifies the failure and recovery alert subjects name the tool, so an inbox fed by several monitors sorts them apart +def test_the_alert_subjects_name_the_tool(gm_module): + advice = gm_module.make_recovery_advice("network.timeout", "GitHub did not answer in time", "do the thing", True) + + assert gm_module.recovery_alert_subject(advice, "watched") == "GitHub Monitor error: GitHub did not answer in time (user: watched)" + assert gm_module.outage_recovered_alert_subject("watched", 514) == "GitHub Monitor recovered: monitoring watched resumed after 8 minutes, 34 seconds" + + +# Verifies the failure alert body counts the run only once a check has failed again and carries the cause only in debug +def test_the_failure_alert_body_lists_the_retry_and_hides_the_cause(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "DEBUG_MODE", False) + advice = gm_module.make_recovery_advice("network.timeout", "GitHub did not answer in time", "Usually nothing to do", True, "the socket gave up") + + first = gm_module.recovery_alert_body(advice, 300) + later = gm_module.recovery_alert_body(advice, 300, 4, 1800000000) + + assert first.startswith("GitHub did not answer in time\n\nTo fix: Usually nothing to do\n\nNext retry in: 5 minutes") + assert "Failed checks in a row" not in first + assert "Failed checks in a row: 4" in later and "Failing since: " in later + assert "Technical detail" not in later + assert "Timestamp: " not in gm_module.recovery_alert_body(advice, 300, timestamp=False) + monkeypatch.setattr(gm_module, "DEBUG_MODE", True) + assert "Technical detail: the socket gave up" in gm_module.recovery_alert_body(advice, 300) diff --git a/tests/test_recovery_safety.py b/tests/test_recovery_safety.py index b71fcaf..1d696fe 100644 --- a/tests/test_recovery_safety.py +++ b/tests/test_recovery_safety.py @@ -242,3 +242,53 @@ def test_interpolated_secret_follows_startup_and_reload_precedence(monitor, monk assert monitor.NTFY_ACCESS_TOKEN == "synthetic-export-value" monitor.reload_secrets_signal_handler(signal.SIGHUP, None) assert monitor.NTFY_ACCESS_TOKEN == "synthetic-export-value" + + +# Verifies the outage reminder can hold its trailer, so an alert delivered by the same check is printed inside +# the report rather than under the separator that ended it +def test_the_outage_reminder_can_hold_its_trailer(monitor, monkeypatch, capsys): + monkeypatch.setattr(monitor, "LOCAL_TIMEZONE", "UTC") + advice = monitor.make_recovery_advice("network.unavailable", "The network is unreachable", "Check connectivity then retry", True) + + monitor.print_outage_liveness("watched-target", advice, 1_000_000, 2, close=False) + held = capsys.readouterr().out + monitor.print_outage_liveness("watched-target", advice, 1_000_000, 2) + closed = capsys.readouterr().out + + assert "* Monitoring degraded for watched-target. The network is unreachable since " in held + assert "Liveness check, timestamp:" not in held, "a reminder that closes itself leaves the alert outside the report" + assert "Liveness check, timestamp:" in closed + + +# Verifies every failing path defers the reminder trailer, since the alert it may deliver prints after the reminder +def test_every_failing_path_defers_the_reminder_trailer(monitor): + source = Path(monitor.__file__).read_text(encoding="utf-8") + calls = [line.strip() for line in source.splitlines() if "print_outage_liveness(" in line and not line.lstrip().startswith("def ")] + + assert calls, "the outage reminder is never reported" + assert all("close=False" in call for call in calls), calls + # One trailer inside the helper and at least one in every path that defers it + assert source.count('print_cur_ts("Liveness check, timestamp:\\t")') >= len(calls) + 1 + + +# Verifies a reported change names the window the run observed, which a failing check leaves further back than +# the configured interval +def test_a_reported_change_names_the_window_the_run_observed(monitor, monkeypatch): + monkeypatch.setattr(monitor, "LOCAL_TIMEZONE", "UTC") + monkeypatch.setattr(monitor, "GITHUB_CHECK_INTERVAL", 3600) + monkeypatch.setattr(monitor, "LAST_CHECK_TS", int(monitor.time.time()) - 60) + + assert monitor.observed_window()[0] == 60 + assert monitor.check_window_text().startswith("1 minute (") + assert monitor.check_window_html().startswith("1 minute (") + + monkeypatch.setattr(monitor, "LAST_CHECK_TS", 0) + assert monitor.observed_window()[0] == 3600, "the configured interval is all a run knows before its first check" + + +# Verifies no report still builds its window from the configured interval, which a failing check makes wrong +def test_no_report_builds_its_window_from_the_configured_interval(monitor): + source = Path(monitor.__file__).read_text(encoding="utf-8") + + assert "int(time.time()) - GITHUB_CHECK_INTERVAL" not in source + assert source.count("check_window_text()") + source.count("check_window_html()") >= 48 diff --git a/tests/test_repository_metadata.py b/tests/test_repository_metadata.py index 53bf782..946cd80 100644 --- a/tests/test_repository_metadata.py +++ b/tests/test_repository_metadata.py @@ -145,17 +145,27 @@ def test_support_document_routes_every_request_type(): assert concept in support -# Confirms the optional local hooks run the same linter version CI installs, or a clean commit still fails CI -def test_local_hooks_match_the_pinned_linter(): - pinned = re.search(r'lint = \["ruff==([^"]+)"\]', read_asset("pyproject.toml")) - assert pinned is not None +# Verifies the optional local hooks run the ruff the pinned extra installs, since a second pin here would drift +# apart from it every time one side is bumped and a locally clean commit would still fail CI +def test_local_hooks_run_the_pinned_linter(): + assert re.search(r'lint = \["ruff==([^"]+)"\]', read_asset("pyproject.toml")) is not None - hooks = read_yaml_asset(".pre-commit-config.yaml")["repos"] - ruff_hook = next(entry for entry in hooks if "ruff-pre-commit" in entry["repo"]) - assert ruff_hook["rev"] == f"v{pinned.group(1)}" + repos = read_yaml_asset(".pre-commit-config.yaml")["repos"] + assert not any("ruff" in entry["repo"] for entry in repos), "ruff must not be pinned a second time in the hook configuration" + + ruff_hook = next(hook for entry in repos if entry["repo"] == "local" for hook in entry["hooks"] if hook["id"] == "ruff-check") + assert ruff_hook["language"] == "system" + assert ruff_hook["entry"].split() == ["ruff", "check"] lint_steps = read_yaml_asset(".github/workflows/tests.yml")["jobs"]["lint"]["steps"] - assert any("ruff check" in step.get("run", "") for step in lint_steps) + assert any("[lint]" in step.get("run", "") for step in lint_steps) + lint_command = next(step["run"] for step in lint_steps if "ruff check" in step.get("run", "")) + + # Both sides must also reach the same files, or the hook stays quiet about code CI rejects + covered = re.compile(ruff_hook["files"]) + assert covered.match("github_monitor.py") + assert covered.match("tests/test_repository_metadata.py") + assert "github_monitor.py tests" in lint_command # Confirms published archives stay verifiable, since an unsigned download cannot be told apart from a tampered one diff --git a/tests/test_startup_summary_channels.py b/tests/test_startup_summary_channels.py index 45f8a11..6134f1f 100644 --- a/tests/test_startup_summary_channels.py +++ b/tests/test_startup_summary_channels.py @@ -161,3 +161,22 @@ def test_the_webhook_send_line_names_the_provider(monkeypatch, capsys, provider, monitor.send_notification_channels("error", "subject", "body", webhook_enabled=True) assert f"Sending webhook notification via {expected}" in capsys.readouterr().out + + +# Verifies a configuration still holding the shipped sample values reports no channel, rather than naming a server and a recipient no alert can reach +@pytest.mark.parametrize("label,setting,placeholder", [ + ("Email transport", "SMTP_HOST", "your_smtp_server_ssl"), + ("Email recipient", "RECEIVER_EMAIL", "your_receiver_email"), + ("Webhook provider", "WEBHOOK_URL", "your_webhook_url"), +]) +def test_a_placeholder_destination_is_reported_as_unconfigured(monkeypatch, label, setting, placeholder): + monkeypatch.setattr(monitor, setting, placeholder) + + assert summary_values()[label] == "Not configured" + + +# Verifies a channel with its alert types on but no destination is not reported as live, since the rollup is the only line the short view prints +def test_a_channel_without_a_destination_is_reported_as_off(): + assert monitor._startup_notification_state(["errors"], True) == "On (errors)" + assert monitor._startup_notification_state(["errors"], False) == "Off (not configured)" + assert monitor._startup_notification_state([], False) == "Off" diff --git a/tests/test_terminal_color.py b/tests/test_terminal_color.py index 688d256..34b68ff 100644 --- a/tests/test_terminal_color.py +++ b/tests/test_terminal_color.py @@ -575,7 +575,7 @@ def test_target_username_uses_the_username_colour_everywhere(colored, line): # Verifies prose following the word "user" is not mistaken for a login -@pytest.mark.parametrize("line", ["- Stargazer/watcher user lists:\tFetched for 16/16 repositories", "* Error: The user details could not be read: boom", "Old user name:\t\t\tOcto Cat", "* Monitored user refresh failed"]) +@pytest.mark.parametrize("line", ["- Stargazer/watcher user lists:\tFetched for 16/16 repositories", "* Error: The user details could not be read: boom", "Old user name:\t\t\tOcto Cat", "* Monitored account lookup failed"]) def test_prose_after_the_word_user_is_not_coloured(colored, line): result = monitor._colorize_line(line) assert monitor.ANSI_ESCAPE_RE.sub("", result) == line @@ -873,3 +873,44 @@ def test_the_early_output_config_carries_the_help_theme(monkeypatch, tmp_path): monitor.apply_early_output_config() assert monitor.COLOR_THEME == {"help_heading": "bright_red"} + + +# Verifies a settings row is never painted as a log line, since a label or a value can read like an error keyword +@pytest.mark.parametrize("label,value", [("Error retry timer", "3 minutes"), ("Polling interval", "5 minutes, longer after a failure")]) +def test_a_summary_row_is_not_painted_by_a_log_keyword(monkeypatch, label, value): + monkeypatch.setattr(monitor, "COLOR_ENABLED", True) + monkeypatch.setattr(monitor, "_COLOR_STYLES", {name: f"<{name}>" for name in monitor.DEFAULT_COLOR_THEME}) + line = monitor.format_startup_summary_row(monitor.StartupSummaryRow(label, value)).rstrip("\n") + + # The value highlights still apply, so only the whole-row block styles have to be absent + coloured = monitor._colorize_line(line) + + assert "" not in coloured and "" not in coloured + + +# Verifies an ordinary error line still carries the block colour the summary rows opt out of +def test_an_error_line_is_still_painted(monkeypatch): + monkeypatch.setattr(monitor, "COLOR_ENABLED", True) + monkeypatch.setattr(monitor, "_COLOR_STYLES", {"error": ""}) + + assert "" in monitor._colorize_line("* Error: the request failed") + + +# Verifies the row shape the colouriser matches is the one the summary emitter prints, so the two cannot drift +def test_every_summary_row_is_recognised_by_its_value_column(): + for row in (monitor.StartupSummaryRow("Target", "someone"), monitor.StartupSummaryRow("Email transport", "Not configured")): + line = monitor.format_startup_summary_row(row).rstrip("\n") + assert monitor.is_startup_summary_row(line) + assert line.index(row.value.split(" ")[0]) == monitor.STARTUP_SUMMARY_VALUE_COLUMN + + assert not monitor.is_startup_summary_row("* Error: something failed") + assert not monitor.is_startup_summary_row("* Warning: a timeout was hit") + + +# Verifies a date does not reach back over a padded gap and read the word in front of it as a weekday +def test_a_wide_gap_before_a_date_is_not_read_as_a_weekday(): + padded = monitor._LONG_DATE_RE.search("A padded column end 07 Feb 26, 00:05:42") + weekday = monitor._LONG_DATE_RE.search("Sun 06 Apr 2025, 21:21:46") + + assert padded is not None and padded.group(0) == "07 Feb 26, 00:05:42" + assert weekday is not None and weekday.group(0) == "Sun 06 Apr 2025, 21:21:46" diff --git a/tests/test_webhook_notifications.py b/tests/test_webhook_notifications.py index 4df0aea..a3dc5c8 100644 --- a/tests/test_webhook_notifications.py +++ b/tests/test_webhook_notifications.py @@ -35,6 +35,15 @@ def json(self): return self.payload +# Gives both channels a destination, since the rollup rows report a channel with none as off whatever its alert types are +def configure_channel_destinations(gm_module, monkeypatch): + monkeypatch.setattr(gm_module, "SMTP_HOST", "smtp.example.com") + monkeypatch.setattr(gm_module, "SMTP_PORT", 587) + monkeypatch.setattr(gm_module, "RECEIVER_EMAIL", "michal.k@example.com") + monkeypatch.setattr(gm_module, "WEBHOOK_PROVIDER", "discord") + monkeypatch.setattr(gm_module, "WEBHOOK_URL", "https://discord.com/api/webhooks/123/private-token") + + # Enables one valid test webhook without affecting email settings def configure_webhook(gm_module, monkeypatch, provider="discord"): monkeypatch.setattr(gm_module, "WEBHOOK_ENABLED", True) @@ -55,6 +64,7 @@ def test_startup_notification_summaries_use_compact_rollups(gm_module, monkeypat webhook_settings = {"WEBHOOK_ENABLED": True, "WEBHOOK_PROFILE_NOTIFICATION": True, "WEBHOOK_EVENT_NOTIFICATION": True, "WEBHOOK_REPO_NOTIFICATION": True, "WEBHOOK_REPO_UPDATE_DATE_NOTIFICATION": True, "WEBHOOK_CONTRIB_NOTIFICATION": True, "WEBHOOK_ERROR_NOTIFICATION": True} for setting, value in {**email_settings, **webhook_settings}.items(): monkeypatch.setattr(gm_module, setting, value) + configure_channel_destinations(gm_module, monkeypatch) rows = {row.label: row.value for row in gm_module.build_startup_summary("octocat", None, None, None)} assert rows["Notifications (email)"] == "On (profile, events, repositories, repository updates, contributions, errors)" assert rows["Notifications (webhook)"] == "On (profile, events, repositories, repository updates, contributions, errors)" @@ -62,6 +72,7 @@ def test_startup_notification_summaries_use_compact_rollups(gm_module, monkeypat # Verifies webhook categories remain off while the master switch is disabled def test_startup_webhook_summary_respects_master_switch(gm_module, monkeypatch): + configure_channel_destinations(gm_module, monkeypatch) monkeypatch.setattr(gm_module, "WEBHOOK_ENABLED", False) monkeypatch.setattr(gm_module, "WEBHOOK_PROFILE_NOTIFICATION", True) monkeypatch.setattr(gm_module, "WEBHOOK_ERROR_NOTIFICATION", True)