Skip to content

Nightly Tests

Nightly Tests #2

Workflow file for this run

name: Nightly Tests
# Least-privilege token (OSSF Scorecard: Token-Permissions). packages: read is
# needed by the called performance_report.yml jobs, whose postgres service
# container pulls ghcr.io/topoteretes/pgvector:pg17 with the GITHUB_TOKEN.
# Nothing in this workflow publishes packages, so no write scope.
permissions:
contents: read
packages: read
on:
# No `schedule:` here on purpose. A cron only ever fires against the default
# branch's copy of this file, so a plain schedule could only ever test main's
# stale copy of the job graph. `.github/workflows/nightly_scheduler.yml`
# carries the cron on main and dispatches this workflow at the ref we want:
# main daily at 18:00 UTC (peak US hours, so the cloud perf arms measure the
# platform under real load), dev weekly on Sunday.
# Deliberately NOT on pull_request. This workflow used to run on any PR that
# touched the nightly/perf workflow files, to validate the edit before merge.
# The cost of that was the whole nightly gating the PR: ~13 jobs including
# Ollama, llama-cpp, four cloud/rust perf arms and a 100k-node benchmark, any
# one of which reddens the PR for something the PR did not cause. It also
# meant a nightly-only regression on dev surfaced as a failing check on an
# unrelated PR. Validate workflow edits with `workflow_dispatch` on the branch
# instead (Actions -> Nightly Tests -> Run workflow -> pick the branch).
workflow_dispatch:
inputs:
cadence:
description: 'Why this run happened: daily (dev), weekly (main), manual.'
required: false
default: manual
type: string
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
env:
RUNTIME__LOG_LEVEL: ERROR
ENV: 'dev'
jobs:
# ══ Performance: each caller runs BOTH file_based and postgres backends ════
perf-small-llm:
name: Performance — 50 Small Documents (real LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: llm
label: 50_small_documents
runs: '3'
num_memories: '50'
memories_key: nightly_ci_artifacts/performance_test_artifacts/memories.json
secrets: inherit
perf-small-mock:
name: Performance — 50 Small Documents (mock LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: mock_llm
label: 50_small_documents
runs: '3'
num_memories: '50'
memories_key: nightly_ci_artifacts/performance_test_artifacts/memories.json
mock_memories_key: nightly_ci_artifacts/performance_test_artifacts/mock_memories.json
secrets: inherit
perf-wap-llm:
name: Performance — War and Peace (real LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: llm
label: war_and_peace
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/war_and_peace.json
secrets: inherit
perf-wap-mock:
name: Performance — War and Peace (mock LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: mock_llm
label: war_and_peace
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/war_and_peace.json
mock_memories_key: nightly_ci_artifacts/performance_test_artifacts/mock_war_and_peace.json
secrets: inherit
# Technical datasheets: 164 real PDFs (~1.35M chars). Complements the
# existing corpora on document *count* -- War and Peace is one long document,
# the 50-doc set is short synthetic ones. This is many medium-sized real-world
# documents, which is the shape most user datasets actually have.
#
# No Rust arms yet: those read fixtures from topoteretes/cognee-rs
# (scripts/perf/fixtures/<label>/), so they land with that repo's PR.
perf-datasheets-llm:
name: Performance — Datasheets (real LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: llm
label: datasheets
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/datasheets.json
secrets: inherit
perf-datasheets-mock:
name: Performance — Datasheets (mock LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: mock_llm
label: datasheets
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/datasheets.json
mock_memories_key: nightly_ci_artifacts/performance_test_artifacts/mock_datasheets.json
secrets: inherit
# 27x-multiplied capture of the same book (~100k graph nodes/edges per run):
# the production-scale stress benchmark. Mock-only — the corpus IS a replay
# artifact; a "real LLM" variant would re-extract 27 identical copies.
perf-wap-large-mock:
name: Performance — Medium graph — Artificial 100k nodes/edges (mock LLM)
uses: ./.github/workflows/performance_report.yml
with:
mode: mock_llm
label: war_and_peace_large
runs: '3'
# postgres only. The file_based arm has never produced a report: ladybug's
# add_edges cannot land the first 2,000-edge chunk of a 100,689-node graph
# inside the subprocess worker's 300s per-call deadline, so cognify times
# out, rolls back, and three runs walk into GitHub's 6h job ceiling — red
# nightly plus a wasted 6h runner, every single night. Re-enable by
# dropping `backends` once the adapter scales (COG-6185 / CLO-594).
backends: postgres
memories_key: nightly_ci_artifacts/performance_test_artifacts/war_and_peace.json
mock_memories_key: nightly_ci_artifacts/performance_test_artifacts/mock_war_and_peace_large.json
secrets: inherit
# ══ Performance: Cognee Cloud tenant. No mock variants — the LLM runs ══════
# server-side on the tenant, so there is nothing to mock client-side.
perf-small-cloud:
name: Performance — 50 Small Documents (cloud)
uses: ./.github/workflows/performance_report_cloud.yml
with:
label: 50_small_documents
runs: '3'
num_memories: '50'
memories_key: nightly_ci_artifacts/performance_test_artifacts/memories.json
secrets: inherit
perf-wap-cloud:
name: Performance — War and Peace (cloud)
uses: ./.github/workflows/performance_report_cloud.yml
with:
label: war_and_peace
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/war_and_peace.json
secrets: inherit
perf-datasheets-cloud:
name: Performance — Datasheets (cloud)
uses: ./.github/workflows/performance_report_cloud.yml
with:
label: datasheets
runs: '3'
memories_key: nightly_ci_artifacts/performance_test_artifacts/datasheets.json
secrets: inherit
# ══ Performance (Rust SDK): builds latest cognee-rs and drives the SAME ═════
# ══ orchestrator via cognee-cli bench (file_based, mock LLM, offline). ══════
perf-rust-llm:
name: Performance — Rust SDK — 50 Small Documents (real LLM)
uses: ./.github/workflows/performance_report_rust.yml
with:
runs: '3'
mode: llm
label: 50_small_documents
memories: scripts/perf/fixtures/memories.json
secrets: inherit
perf-rust:
name: Performance — Rust SDK — 50 Small Documents (mock LLM)
uses: ./.github/workflows/performance_report_rust.yml
with:
runs: '3'
mode: mock_llm
label: 50_small_documents
memories: scripts/perf/fixtures/memories.json
cassette: scripts/perf/fixtures/cassette.json
secrets: inherit
perf-rust-wap-llm:
name: Performance — Rust SDK — War and Peace (real LLM)
uses: ./.github/workflows/performance_report_rust.yml
with:
runs: '3'
mode: llm
label: war_and_peace
memories: scripts/perf/fixtures/war_and_peace/memories.json
secrets: inherit
perf-rust-wap:
name: Performance — Rust SDK — War and Peace (mock LLM)
uses: ./.github/workflows/performance_report_rust.yml
with:
runs: '3'
mode: mock_llm
label: war_and_peace
memories: scripts/perf/fixtures/war_and_peace/memories.json
cassette: scripts/perf/fixtures/war_and_peace/cassette.json
secrets: inherit
notify:
name: Test Completion Status
needs: [
perf-small-llm,
perf-small-mock,
perf-wap-llm,
perf-wap-mock,
perf-datasheets-llm,
perf-datasheets-mock,
perf-wap-large-mock,
perf-small-cloud,
perf-wap-cloud,
perf-datasheets-cloud,
perf-rust,
perf-rust-wap,
perf-rust-llm,
perf-rust-wap-llm,
]
runs-on: ubuntu-latest
if: ${{ !cancelled() }}
steps:
# This job runs repo scripts (the Slack block renderer, the MotherDuck
# ETL), so it needs a working tree.
- name: Checkout repository
uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6.1.0
- name: Determine status
id: status
env:
# Via env, not inline ${{ }}: `cadence` is a free-text dispatch input
# and inline interpolation would splice it into this shell script.
CADENCE: ${{ inputs.cadence }}
run: |
if [[ "${{ needs.perf-small-llm.result }}" == "success" &&
"${{ needs.perf-small-mock.result }}" == "success" &&
"${{ needs.perf-wap-llm.result }}" == "success" &&
"${{ needs.perf-wap-mock.result }}" == "success" &&
"${{ needs.perf-datasheets-llm.result }}" == "success" &&
"${{ needs.perf-datasheets-mock.result }}" == "success" &&
"${{ needs.perf-wap-large-mock.result }}" == "success" &&
"${{ needs.perf-small-cloud.result }}" == "success" &&
"${{ needs.perf-wap-cloud.result }}" == "success" &&
"${{ needs.perf-datasheets-cloud.result }}" == "success" &&
"${{ needs.perf-rust.result }}" == "success" &&
"${{ needs.perf-rust-wap.result }}" == "success" &&
"${{ needs.perf-rust-llm.result }}" == "success" &&
"${{ needs.perf-rust-wap-llm.result }}" == "success" ]]; then
echo "passed=true" >> "$GITHUB_OUTPUT"
echo "emoji=✅" >> "$GITHUB_OUTPUT"
echo "summary=All nightly test suites completed successfully!" >> "$GITHUB_OUTPUT"
else
echo "passed=false" >> "$GITHUB_OUTPUT"
echo "emoji=❌" >> "$GITHUB_OUTPUT"
echo "summary=One or more nightly test suites failed." >> "$GITHUB_OUTPUT"
fi
echo "ran_at=$(date -u '+%Y-%m-%d %H:%M UTC')" >> "$GITHUB_OUTPUT"
# Two nightlies now exist (daily dev, weekly main). Everything below
# that reaches a human or the warehouse has to say which one this is.
echo "cadence=${CADENCE:-manual}" >> "$GITHUB_OUTPUT"
- name: Presign performance report links
id: presign
env:
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_S3_DEV_USER_KEY_ID }}
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_S3_DEV_USER_SECRET_KEY }}
AWS_DEFAULT_REGION: eu-west-1
BUCKET: github-runner-cognee-tests
# "<slack-output-name> <s3-key>" per line. The key already encodes the
# backend (file_based/ or postgres/), so presigning is fully generic.
REPORT_KEYS: |
url_file_small_llm ${{ needs.perf-small-llm.outputs.file_based_html_key }}
url_file_small_mock ${{ needs.perf-small-mock.outputs.file_based_html_key }}
url_file_wap_llm ${{ needs.perf-wap-llm.outputs.file_based_html_key }}
url_file_wap_mock ${{ needs.perf-wap-mock.outputs.file_based_html_key }}
url_pg_small_llm ${{ needs.perf-small-llm.outputs.postgres_html_key }}
url_pg_small_mock ${{ needs.perf-small-mock.outputs.postgres_html_key }}
url_pg_wap_llm ${{ needs.perf-wap-llm.outputs.postgres_html_key }}
url_pg_wap_mock ${{ needs.perf-wap-mock.outputs.postgres_html_key }}
url_file_datasheets_llm ${{ needs.perf-datasheets-llm.outputs.file_based_html_key }}
url_file_datasheets_mock ${{ needs.perf-datasheets-mock.outputs.file_based_html_key }}
url_pg_datasheets_llm ${{ needs.perf-datasheets-llm.outputs.postgres_html_key }}
url_pg_datasheets_mock ${{ needs.perf-datasheets-mock.outputs.postgres_html_key }}
url_pg_wap_large_mock ${{ needs.perf-wap-large-mock.outputs.postgres_html_key }}
url_cloud_small ${{ needs.perf-small-cloud.outputs.cloud_html_key }}
url_cloud_wap ${{ needs.perf-wap-cloud.outputs.cloud_html_key }}
url_cloud_datasheets ${{ needs.perf-datasheets-cloud.outputs.cloud_html_key }}
url_rust ${{ needs.perf-rust.outputs.html_key }}
url_rust_wap ${{ needs.perf-rust-wap.outputs.html_key }}
url_rust_llm ${{ needs.perf-rust-llm.outputs.html_key }}
url_rust_wap_llm ${{ needs.perf-rust-wap-llm.outputs.html_key }}
run: |
set -euo pipefail
# 7 days = 604800s is the maximum lifetime for an IAM-user presigned URL.
# Skip empty keys (a failed perf job produces no report).
while read -r name key; do
[ -n "${key:-}" ] || continue
echo "$name=$(aws s3 presign "s3://$BUCKET/$key" --expires-in 604800)" >> "$GITHUB_OUTPUT"
done <<< "$REPORT_KEYS"
- name: Build Slack blocks
id: blocks
env:
STATUS_EMOJI: ${{ steps.status.outputs.emoji }}
STATUS_SUMMARY: ${{ steps.status.outputs.summary }}
STATUS_RAN_AT: ${{ steps.status.outputs.ran_at }}
STATUS_BRANCH: ${{ github.ref_name }}
STATUS_CADENCE: ${{ steps.status.outputs.cadence }}
STATUS_SHA: ${{ github.sha }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
# One arm per line: emoji|title|result|metrics-json|report-url.
# Adding a benchmark arm is a line here plus its REPORT_KEYS entry.
ARMS: |
📊|File Based — 50 Small Documents — Real LLM|${{ needs.perf-small-llm.result }}|${{ needs.perf-small-llm.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_small_llm }}
📊|File Based — 50 Small Documents — Mock LLM|${{ needs.perf-small-mock.result }}|${{ needs.perf-small-mock.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_small_mock }}
📚|File Based — War and Peace — Real LLM|${{ needs.perf-wap-llm.result }}|${{ needs.perf-wap-llm.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_wap_llm }}
📚|File Based — War and Peace — Mock LLM|${{ needs.perf-wap-mock.result }}|${{ needs.perf-wap-mock.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_wap_mock }}
🧪|File Based — Datasheets — Real LLM|${{ needs.perf-datasheets-llm.result }}|${{ needs.perf-datasheets-llm.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_datasheets_llm }}
🧪|File Based — Datasheets — Mock LLM|${{ needs.perf-datasheets-mock.result }}|${{ needs.perf-datasheets-mock.outputs.file_based_metrics }}|${{ steps.presign.outputs.url_file_datasheets_mock }}
📊|Full Postgres — 50 Small Documents — Real LLM|${{ needs.perf-small-llm.result }}|${{ needs.perf-small-llm.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_small_llm }}
📊|Full Postgres — 50 Small Documents — Mock LLM|${{ needs.perf-small-mock.result }}|${{ needs.perf-small-mock.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_small_mock }}
📚|Full Postgres — War and Peace — Real LLM|${{ needs.perf-wap-llm.result }}|${{ needs.perf-wap-llm.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_wap_llm }}
📚|Full Postgres — War and Peace — Mock LLM|${{ needs.perf-wap-mock.result }}|${{ needs.perf-wap-mock.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_wap_mock }}
📚|Full Postgres — Medium graph - Artificial 100k nodes/edges — Mock LLM|${{ needs.perf-wap-large-mock.result }}|${{ needs.perf-wap-large-mock.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_wap_large_mock }}
🧪|Full Postgres — Datasheets — Real LLM|${{ needs.perf-datasheets-llm.result }}|${{ needs.perf-datasheets-llm.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_datasheets_llm }}
🧪|Full Postgres — Datasheets — Mock LLM|${{ needs.perf-datasheets-mock.result }}|${{ needs.perf-datasheets-mock.outputs.postgres_metrics }}|${{ steps.presign.outputs.url_pg_datasheets_mock }}
📊|Cognee Cloud — 50 Small Documents|${{ needs.perf-small-cloud.result }}|${{ needs.perf-small-cloud.outputs.cloud_metrics }}|${{ steps.presign.outputs.url_cloud_small }}
📚|Cognee Cloud — War and Peace|${{ needs.perf-wap-cloud.result }}|${{ needs.perf-wap-cloud.outputs.cloud_metrics }}|${{ steps.presign.outputs.url_cloud_wap }}
🧪|Cognee Cloud — Datasheets|${{ needs.perf-datasheets-cloud.result }}|${{ needs.perf-datasheets-cloud.outputs.cloud_metrics }}|${{ steps.presign.outputs.url_cloud_datasheets }}
📊|Rust SDK (file_based) — 50 Small Documents — Real LLM|${{ needs.perf-rust-llm.result }}|${{ needs.perf-rust-llm.outputs.metrics }}|${{ steps.presign.outputs.url_rust_llm }}
📊|Rust SDK (file_based) — 50 Small Documents — Mock LLM|${{ needs.perf-rust.result }}|${{ needs.perf-rust.outputs.metrics }}|${{ steps.presign.outputs.url_rust }}
📚|Rust SDK (file_based) — War and Peace — Real LLM|${{ needs.perf-rust-wap-llm.result }}|${{ needs.perf-rust-wap-llm.outputs.metrics }}|${{ steps.presign.outputs.url_rust_wap_llm }}
📚|Rust SDK (file_based) — War and Peace — Mock LLM|${{ needs.perf-rust-wap.result }}|${{ needs.perf-rust-wap.outputs.metrics }}|${{ steps.presign.outputs.url_rust_wap }}
run: python .github/scripts/build_nightly_slack_blocks.py
- name: Post status to Slack
# Posts on scheduled and manual (workflow_dispatch) runs. There is no
# longer a pull_request trigger, so a workflow edit surfaces its reports
# by being dispatched on the branch rather than by opening a PR.
uses: slackapi/slack-github-action@b0fa283ad8fea605de13dc3f449259339835fc52 # v2.1.0
with:
method: chat.postMessage
token: ${{ secrets.SLACK_NIGHTLY_BOT_TOKEN }}
payload: |
channel: ${{ secrets.SLACK_NIGHTLY_CHANNEL_ID }}
text: "${{ steps.status.outputs.emoji }} Nightly Tests [${{ github.ref_name }}] (${{ steps.status.outputs.ran_at }}): ${{ steps.status.outputs.summary }}"
blocks: ${{ steps.blocks.outputs.blocks }}
- name: Load performance reports into MotherDuck
# Reads the report JSON the perf jobs already uploaded to S3 and loads it
# into ci_analytics.nightly (SDK-37). Deliberately placed BEFORE the
# failure gate and marked continue-on-error: a warehouse hiccup must not
# turn a green nightly red, and the reports from a failed suite are the
# interesting rows, so they still need to land.
# The pull_request trigger is gone, so this guard is now a belt-and-braces
# no-op. Kept deliberately: if anyone reinstates a PR trigger, a
# validation run must not write to the shared warehouse.
if: ${{ github.event_name != 'pull_request' }}
continue-on-error: true
env:
# duckdb is PINNED: an unpinned install pulled a release MotherDuck
# rejects and broke the sibling analytics ETL (COG-5953).
DUCKDB_VERSION: '1.4.5'
motherduck_token: ${{ secrets.MOTHERDUCK_TOKEN }}
MD_TARGET: ci_analytics.nightly
# The ETL does CREATE OR REPLACE VIEW from the script on the ref that
# ran it. With two refs running it, view definitions would be
# last-writer-wins ACROSS refs, flip-flopping every week where the
# two ETLs differ. Only the DAILY run defines views (main, so the
# warehouse always reflects released ETL code -- a dev-side ETL
# change takes effect at the release merge); the weekly dev run
# loads rows only.
PERF_REFRESH_VIEWS: ${{ steps.status.outputs.cadence == 'weekly' && 'false' || 'true' }}
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_S3_DEV_USER_KEY_ID }}
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_S3_DEV_USER_SECRET_KEY }}
AWS_DEFAULT_REGION: eu-west-1
PERF_BUCKET: github-runner-cognee-tests
run: |
set -euo pipefail
if [ -z "${motherduck_token:-}" ]; then
echo "MOTHERDUCK_TOKEN is not set — skipping warehouse load."
exit 0
fi
python -m pip install --quiet "duckdb==${DUCKDB_VERSION}"
python .github/scripts/motherduck_nightly_etl.py
- name: Fail if any suite failed
if: ${{ steps.status.outputs.passed != 'true' }}
run: |
echo "One or more nightly test suites failed."
exit 1