Snapshot 20427
Normalized text
Scripts and page chrome removed; this is what change detection compares.
{
"components": [
{
"created_at": "2013-08-26T04:27:06.711Z",
"description": null,
"group_id": null,
"id": "r86fqtcmnn65",
"name": "Web",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-10T05:07:21.220Z"
},
{
"created_at": "2015-05-11T01:35:06.981Z",
"description": null,
"group_id": null,
"id": "1mpkbbd1kh0h",
"name": "Agent API",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-21T23:12:04.144Z"
},
{
"components": [
"r21z46hqn47c",
"fx3sc4nr2w8v"
],
"created_at": "2026-05-11T02:54:29.661Z",
"description": null,
"group_id": null,
"id": "0zsswmznk66n",
"name": "REST API",
"page_id": "ltljpr68dygn",
"position": 3,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.557Z"
},
{
"created_at": "2015-02-28T01:22:55.169Z",
"description": null,
"group_id": null,
"id": "3n7fzwsysx86",
"name": "Job Queue",
"page_id": "ltljpr68dygn",
"position": 5,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-10T05:08:44.545Z"
},
{
"components": [
"m3fr6snn2496",
"rj879wvx0lyd",
"r0jrhpsydz4w",
"mk54vxw70994"
],
"created_at": "2015-05-11T01:32:50.895Z",
"description": null,
"group_id": null,
"id": "tplm771kdwfs",
"name": "Notifications",
"page_id": "ltljpr68dygn",
"position": 6,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.596Z"
},
{
"components": [
"9pwnnl0dp0j5",
"87hpgxv274l0",
"sd4q7mk8n853",
"fb1llxplp5xq"
],
"created_at": "2025-02-06T06:29:12.958Z",
"description": null,
"group_id": null,
"id": "56ph4s45xl9s",
"name": "Hosted Agents",
"page_id": "ltljpr68dygn",
"position": 8,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.623Z"
},
{
"components": [
"75z9h6yq0vrj",
"12zzrz9bykcj",
"jdvdr3k7kdbq"
],
"created_at": "2025-10-30T02:26:14.118Z",
"description": null,
"group_id": null,
"id": "dr5xs6jv5m8v",
"name": "Package Registries",
"page_id": "ltljpr68dygn",
"position": 9,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.638Z"
},
{
"components": [
"sw9wpknm8p8g",
"dmqj9zfy07kf",
"msyttmm8jpns"
],
"created_at": "2022-09-06T05:23:51.907Z",
"description": null,
"group_id": null,
"id": "lnkc32srq0df",
"name": "Test Engine",
"page_id": "ltljpr68dygn",
"position": 10,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.653Z"
},
{
"created_at": "2019-05-27T11:22:24.081Z",
"description": null,
"group_id": null,
"id": "pcwww1g6plvh",
"name": "SCM Integrations",
"page_id": "ltljpr68dygn",
"position": 11,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.666Z"
},
{
"components": [
"z9xdz6bnf069",
"7hzjm77m5lhd",
"r0qvv92dtt1x",
"92bvq28m13sk",
"nmmdp7mwb1nw",
"nf6qypynz4gc"
],
"created_at": "2015-05-11T01:19:03.082Z",
"description": "Third party SCM providers which may affect your builds",
"group_id": null,
"id": "wj7nh2k1l54h",
"name": "SCM Providers",
"page_id": "ltljpr68dygn",
"position": 12,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.700Z"
},
{
"components": [
"m20p1dn7lr5h",
"5m66h4qss3g9",
"sq0ls35dpmvy",
"3bwdr960jnym"
],
"created_at": "2015-05-11T01:17:54.959Z",
"description": "Third party services we depend upon",
"group_id": null,
"id": "hd242hws8942",
"name": "Third Party Services",
"page_id": "ltljpr68dygn",
"position": 13,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-05-11T02:54:44.728Z"
},
{
"created_at": "2023-05-30T05:57:03.881Z",
"description": "https://buildkite.com/docs",
"group_id": null,
"id": "8wzdjbrhcyjk",
"name": "Docs",
"page_id": "ltljpr68dygn",
"position": 14,
"showcase": false,
"start_date": "2023-05-30T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-05-11T02:54:44.739Z"
}
],
"incidents": [
{
"components": [
{
"created_at": "2015-05-11T01:32:50.904Z",
"description": null,
"group_id": "tplm771kdwfs",
"id": "m3fr6snn2496",
"name": "GitHub Commit Status Notifications",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-08-25T23:38:18.738Z"
},
{
"created_at": "2015-05-11T01:33:31.806Z",
"description": null,
"group_id": "tplm771kdwfs",
"id": "rj879wvx0lyd",
"name": "Email Notifications",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-01T16:52:23.716Z"
},
{
"created_at": "2015-05-11T01:33:51.204Z",
"description": null,
"group_id": "tplm771kdwfs",
"id": "r0jrhpsydz4w",
"name": "Slack Notifications",
"page_id": "ltljpr68dygn",
"position": 3,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-07-28T20:57:34.488Z"
},
{
"created_at": "2018-05-30T21:53:19.571Z",
"description": null,
"group_id": "tplm771kdwfs",
"id": "mk54vxw70994",
"name": "Webhook Notifications",
"page_id": "ltljpr68dygn",
"position": 7,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-08-25T23:38:18.843Z"
}
],
"created_at": "2026-09-30T01:03:48.760Z",
"id": "s9twc1cmh5cv",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "m3fr6snn2496",
"name": "Notifications - GitHub Commit Status Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "r0jrhpsydz4w",
"name": "Notifications - Slack Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "mk54vxw70994",
"name": "Notifications - Webhook Notifications",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "Outbound notification latency remains at expected levels. Some customers may have received duplicate notifications before the underlying error was resolved.",
"created_at": "2026-09-30T02:32:48.408Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-30T02:32:48.408Z",
"id": "t7wcq7xykydd",
"incident_id": "s9twc1cmh5cv",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-30T02:32:48.408Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "m3fr6snn2496",
"name": "Notifications - GitHub Commit Status Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "r0jrhpsydz4w",
"name": "Notifications - Slack Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "mk54vxw70994",
"name": "Notifications - Webhook Notifications",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "Outbound notification queues have now fully caught up for all impacted customers.",
"created_at": "2026-09-30T01:43:28.711Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-30T01:43:28.711Z",
"id": "36gzxnzfz0wz",
"incident_id": "s9twc1cmh5cv",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-30T01:43:28.711Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "m3fr6snn2496",
"name": "Notifications - GitHub Commit Status Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "r0jrhpsydz4w",
"name": "Notifications - Slack Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "mk54vxw70994",
"name": "Notifications - Webhook Notifications",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "We've resolved the underlying error, and outbound notifications queues are now in the process of catching up for impacted customers.",
"created_at": "2026-09-30T01:19:45.077Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-30T01:19:45.077Z",
"id": "wk7mbhc9xsg7",
"incident_id": "s9twc1cmh5cv",
"status": "identified",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-30T01:19:45.077Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "m3fr6snn2496",
"name": "Notifications - GitHub Commit Status Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "r0jrhpsydz4w",
"name": "Notifications - Slack Notifications",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "mk54vxw70994",
"name": "Notifications - Webhook Notifications",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "We are investigating delays to build and job notifications for a subset of customers.",
"created_at": "2026-09-30T01:03:48.819Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-30T01:03:48.819Z",
"id": "dg2h4mw2qr2l",
"incident_id": "s9twc1cmh5cv",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-30T01:03:48.819Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-30T01:43:28.711Z",
"name": "Delayed notifications",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-30T02:32:48.408Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/2cwcn74l5qcy",
"started_at": "2026-09-30T01:03:48.747Z",
"status": "resolved",
"updated_at": "2026-09-30T02:32:48.434Z"
},
{
"components": [],
"created_at": "2026-09-22T20:07:55.810Z",
"id": "t1h2d1f4y6qx",
"impact": "major",
"impact_override": "major",
"incident_updates": [
{
"affected_components": null,
"body": "Services are fully recovered. Webhook ingestion was affected between 19:09 UTC - 19:32 UTC.",
"created_at": "2026-09-22T20:50:02.597Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-22T20:50:02.597Z",
"id": "2nvt2z03109f",
"incident_id": "t1h2d1f4y6qx",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-22T20:50:02.597Z",
"wants_twitter_update": false
},
{
"affected_components": null,
"body": "Our engineers have addressed the issue and we're monitoring recovery.",
"created_at": "2026-09-22T20:26:22.873Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-22T20:26:22.873Z",
"id": "bb9vr99zb63w",
"incident_id": "t1h2d1f4y6qx",
"status": "identified",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-22T20:26:22.873Z",
"wants_twitter_update": false
},
{
"affected_components": null,
"body": "We're seeing an issue affecting webhook ingestion and execution of GraphQL queries and are investigating the cause.",
"created_at": "2026-09-22T20:07:55.844Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-22T20:07:55.844Z",
"id": "vn7y3y8zs8wm",
"incident_id": "t1h2d1f4y6qx",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-22T20:07:55.844Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": null,
"name": "Issues with webhook ingestion, GraphQL queries",
"page_id": "ltljpr68dygn",
"postmortem_body": "## Service Impact\n\nOn September 22, 2026, between 19:09 and 19:32 UTC, customers experienced failures ingesting webhooks and executing GraphQL queries. Some web application requests also failed or timed out.\n\nDuring this period, 2.0% of requests to our webhook ingestion endpoints and 20.3% of requests to our external GraphQL endpoint returned HTTP 500 errors.\n\nWebhook ingestion errors ceased by 19:30 UTC, and external GraphQL errors returned to background levels by 19:32 UTC.\n\n## Incident Summary\n\nWhile investigating a separate API performance issue, we made a manual configuration change intended to rollback a routing change for one database proxy. The command instead changed the routing configuration for another database proxy, disrupting access to a database shard used internally by Buildkite.\n\nSome customer-facing operations search across all database shards to locate their shard, where routing information isn't available in the request. Those requests round robin through available shards, and could encounter the unavailable shard and wait for a database connection to time out. This extended the impact beyond our internal workloads.\n\nWe reverted the incorrect configuration and verified that the database proxy had healthy endpoints and load-balancer targets.\n\nTwo contributing factors allowed the change to cause an outage: the command was not independently checked before execution, and our infrastructure accepted a Service configuration that selected the wrong database proxy workload. The cross-shard lookup behaviour then allowed the disruption to affect requests for other customers.\n\nThis configuration-change outage was separate from the API performance issue we were originally investigating.\n\n## Changes we're making\n\n* Safer operational procedures: We have updated our runbook and introduced an explicit second person review requirement for manual production commands.\n* Automated safeguards: We’ve added validation to prevent database proxy routing changes from selecting the wrong workload.\n* Improved failure isolation: We’ve been working on an ongoing initiative to make database shard discovery more resilient, eliminating the remaining cases where an unavailable shard may impact requests for other customers.",
"postmortem_body_last_updated_at": "2026-09-28T04:24:38.750Z",
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": true,
"postmortem_published_at": "2026-09-28T04:24:38.750Z",
"reminder_intervals": null,
"resolved_at": "2026-09-22T20:50:02.597Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/b9fymqh5z2nh",
"started_at": "2026-09-22T20:07:55.804Z",
"status": "resolved",
"updated_at": "2026-09-28T04:24:38.763Z"
},
{
"components": [],
"created_at": "2026-09-21T23:47:23.128Z",
"id": "vr7dsq0p8pq0",
"impact": "none",
"impact_override": "none",
"incident_updates": [
{
"affected_components": null,
"body": "We did not experience any customer impact during this AWS us-east-1 outage.",
"created_at": "2026-09-22T00:32:50.053Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-22T00:32:50.053Z",
"id": "mg5lblw11198",
"incident_id": "vr7dsq0p8pq0",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-22T00:32:50.053Z",
"wants_twitter_update": false
},
{
"affected_components": null,
"body": "We're aware of a major AWS outage in us-east-1, we are taking action to prevent impact and monitoring the situation.",
"created_at": "2026-09-21T23:47:23.167Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-21T23:47:23.167Z",
"id": "xc73ctdqf003",
"incident_id": "vr7dsq0p8pq0",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-21T23:47:23.167Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-21T23:47:23.167Z",
"name": "AWS us-east-1 outage",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-22T00:32:50.053Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/46lpdf6x0fky",
"started_at": "2026-09-21T23:47:23.121Z",
"status": "resolved",
"updated_at": "2026-09-22T00:32:50.071Z"
},
{
"components": [
{
"created_at": "2015-05-11T01:35:06.981Z",
"description": null,
"group_id": null,
"id": "1mpkbbd1kh0h",
"name": "Agent API",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-21T23:12:04.144Z"
}
],
"created_at": "2026-09-21T22:35:59.057Z",
"id": "r2w4j1wyj0ss",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "This incident has been resolved.",
"created_at": "2026-09-21T23:46:06.379Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-21T23:46:06.379Z",
"id": "b7ttl707nvd2",
"incident_id": "r2w4j1wyj0ss",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-21T23:46:06.379Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "operational",
"old_status": "degraded_performance"
}
],
"body": "We observed a slight increase in latency for build creation isolated to a few customers, this has now passed and we are seeing signs of improvement.",
"created_at": "2026-09-21T23:12:04.170Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-21T23:12:04.170Z",
"id": "fw3gq4ztmh07",
"incident_id": "r2w4j1wyj0ss",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-21T23:12:04.170Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "We're seeing increased latency and error rates for a subset of our customers. We're currently investigating and will provide status updates as they become available.",
"created_at": "2026-09-21T22:35:59.153Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-21T22:35:59.153Z",
"id": "3sgnjt74l5v1",
"incident_id": "r2w4j1wyj0ss",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-21T22:35:59.153Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-21T23:12:04.170Z",
"name": "Increased latency and error rates",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-21T23:46:06.379Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/8y70fpt2ztbt",
"started_at": "2026-09-21T22:35:59.048Z",
"status": "resolved",
"updated_at": "2026-09-21T23:46:06.392Z"
},
{
"components": [
{
"created_at": "2013-08-26T04:27:06.655Z",
"description": null,
"group_id": "0zsswmznk66n",
"id": "r21z46hqn47c",
"name": "REST API",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-10T05:08:01.169Z"
},
{
"created_at": "2013-08-26T04:27:06.711Z",
"description": null,
"group_id": null,
"id": "r86fqtcmnn65",
"name": "Web",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-10T05:07:21.220Z"
},
{
"created_at": "2015-05-11T01:35:06.981Z",
"description": null,
"group_id": null,
"id": "1mpkbbd1kh0h",
"name": "Agent API",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-21T23:12:04.144Z"
},
{
"created_at": "2015-02-28T01:22:55.169Z",
"description": null,
"group_id": null,
"id": "3n7fzwsysx86",
"name": "Job Queue",
"page_id": "ltljpr68dygn",
"position": 5,
"showcase": true,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-10T05:08:44.545Z"
}
],
"created_at": "2026-09-09T19:07:24.948Z",
"id": "7nc7xd7zkxjx",
"impact": "major",
"impact_override": "major",
"incident_updates": [
{
"affected_components": [
{
"code": "r86fqtcmnn65",
"name": "Web",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "3n7fzwsysx86",
"name": "Job Queue",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "r21z46hqn47c",
"name": "REST API - REST API",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "We have seen full recovery for customers since 20:28 UTC. \nWe experienced an autoscaling feedback loop which increased the number of connections to our redis cluster above its ability to respond. This had widespread impact for all of our customers with Web UI, Agent API, REST API and job queue impact between 18:43-19:21 UTC, and again between 20:02-20:28 UTC. A full post incident review will be available later this week.",
"created_at": "2026-09-10T01:54:23.117Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-10T01:54:23.000Z",
"id": "fd62mqdfjq1p",
"incident_id": "7nc7xd7zkxjx",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-10T01:55:53.986Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "r86fqtcmnn65",
"name": "Web",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "3n7fzwsysx86",
"name": "Job Queue",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "r21z46hqn47c",
"name": "REST API - REST API",
"new_status": "operational",
"old_status": "degraded_performance"
}
],
"body": "We are seeing improvements across the affected services, and are seeing services return to normal functionality. We are continuing to monitor and are determining the root cause.",
"created_at": "2026-09-09T20:53:15.478Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-09T20:53:15.478Z",
"id": "p24y3vptm6zn",
"incident_id": "7nc7xd7zkxjx",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-09T20:53:15.478Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "r86fqtcmnn65",
"name": "Web",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "3n7fzwsysx86",
"name": "Job Queue",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "r21z46hqn47c",
"name": "REST API - REST API",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "We are continuing to investigate elevated error rates across multiple services, and are working to determine the cause.",
"created_at": "2026-09-09T20:28:55.957Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-09T20:28:55.957Z",
"id": "s3j52qg0c8q3",
"incident_id": "7nc7xd7zkxjx",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-09T20:28:55.957Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "1mpkbbd1kh0h",
"name": "Agent API",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "We are continuing to investigate this issue. We are seeing impact on the Agent API which will affect job scheduling, artifact uploads, and an increase in 5xx responses from the Agent API endpoints.",
"created_at": "2026-09-09T19:41:13.281Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-09T19:41:13.281Z",
"id": "kndc6yylqmg2",
"incident_id": "7nc7xd7zkxjx",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-09T19:41:13.281Z",
"wants_twitter_update": true
},
{
"affected_components": null,
"body": "We've spotted that something has gone wrong. We're currently investigating the issue, and will provide an update soon.",
"created_at": "2026-09-09T19:07:24.987Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-09T19:07:24.987Z",
"id": "851n5jgjs09t",
"incident_id": "7nc7xd7zkxjx",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-09T19:07:24.987Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-09T20:53:15.478Z",
"name": "Buildkite service disruption",
"page_id": "ltljpr68dygn",
"postmortem_body": "## Service Impact\n\n_All times UTC unless stated otherwise._\n\nCustomers experienced elevated errors and latency across the Buildkite web interface, REST API, Agent API, and job queue during two periods: 18:43:00–19:21:00 and 20:02:00–20:28:00 on September 9, 2026.\n\nDuring these periods, customers encountered failed API requests, delayed job dispatch, and errors in agent operations including authentication, job acceptance, and artifact uploads. Some jobs impacted by the incident did not recover automatically. These jobs had to be manually cancelled and retried to complete successfully.\n\nBetween 18:43:00–19:21:00: Agent API had an error rate up to 11.2% with no latency impact. REST API had a 9.7% error rate with no latency impact. Web request had a 2.6% error rate with no latency impact. Job queues were delayed by up to 3 minutes.\n\nBetween 20:02:00–20:28:00: Agent API had an error rate up to 24.7% with no latency impact. REST API had a 4.9% error rate with no latency impact. Web requests had a 0.8% error rate with no latency impact. Job queues were delayed by up to 10 minutes. During this window, notifications were also delayed up to 1 minute.\n\nCore services fully recovered by 20:28:00.\n\n## Incident Summary\n\n### Background\n\nSeveral Buildkite services use a shared Redis cluster for coordination, caching, and agent-facing operations. This is one of the last pieces of non-shard aligned shared infrastructure, and is on our roadmap to address. AWS enforces a connection limit of 65,000 connections per node on this Redis cluster.\n\nAs part of our ongoing migration from Amazon ECS to Amazon EKS, we reduced the number of Ruby threads in each Puma and Sidekiq process to their framework defaults. In ECS, we historically configured high thread counts, which caused resource contention and increased latency without a corresponding increase in throughput.\n\nWe’d intentionally timed this work as part of the EKS migration because the migration already required us to redesign how these services are sized and scaled: EKS can respond directly to request queues and worker utilization, allowing us to run more, smaller application containers while preserving total capacity. This gives customers faster, more predictable API responses and background-job processing, particularly during periods of high load.\n\n### What happened\n\nWe began testing Puma Agent services on EKS in July and migrated traffic incrementally from late July, completing the migration earlier this week. Running more, smaller processes increased total Redis connection demand—a trade-off we had modeled, but not adequately accounted for before completing the migration.\n\nBefore the incident, our shared services were already using much of the available connection capacity on the impacted Redis cluster, around 60-80% throughout a 24 hour window.\n\nAt 18:43:10 we started seeing elevated HTTP request errors. At this time, our application reported a burst of Redis connection timeouts, all targeting a single replica in the cluster. Clients with active Redis connections then attempted to reconnect. Failed connections were retried, causing a surge in connection attempts that exhausted that same Redis host’s connection-tracking allowance. This caused the host to reject new connections from the Agent API’s application pods, which led to increased request latency for our Puma Agent service.\n\nAt 18:43:30, the increased request latency from this network failure caused our application platform to rapidly start more Puma Agent containers to respond to the increase in request latency. Within a minute, we’d doubled our running pod count. Each additional container opened its own pool of Redis connections.\n\nAt 18:44:00, several Redis nodes hit their limit of 65,000 connections, and latency of requests slowed further triggering more Puma Agent autoscaling.\n\nOnce the Redis connection limits were reached, applications could no longer reliably connect to Redis. This caused failures across agent authentication, job assignment, artifact uploads, and other API operations.\n\nThe failures then created a feedback loop: slower requests caused our shard aligned Puma Agent services to autoscale, and those additional containers opened more Redis connections, placing further pressure on Redis.\n\nBy 18:50:00, our system stabilized at max, running about five times the number of pods we’d started with. This stable state allowed the autoscaling to start bringing pod counts back down.\n\nBy 19:10:00 the error rate had returned to 0 and and services were no longer degraded.\n\nBy 19:20:00 our pod counts had autoscaled down to more reasonable but still high numbers, and the Redis cluster was functioning again. At this point customers saw recovery.\n\nAt 20:02:00, a routine application deployment temporarily introduced additional containers. Since the Redis cluster was already under strain, this quickly increased request latency which added autoscaling on top of the regular deployment surge numbers. This amplified the connection growth again and caused the second period of degraded service, lasting until 20:28:00.\n\nThe influx of new pods resulted in a very large number of Envoy Gateway configuration updates, which in turn caused some network throttling of the underlying node. This caused packet drops and delays updating our Cilium Operator, which controls configuration for the cluster’s overlay network. This caused resulted some Buildkite Jobs to remain in a stuck state for some customers, requiring manual retries.\n\n### How we responded\n\nWhen we saw that the Redis cluster had reached its capacity to handle incoming connections, we immediately began adding more nodes to the cluster, increasing incoming connection capacity by 33%.\n\nWhen we saw the impact of deployments, we immediately paused further application deployments while we investigated the connection growth.\n\nWe increased the Puma Agent utilization autoscaling threshold, reducing the rate at which workloads scale in response to brief load increases. We also increased our scale up stabilization window to prevent future runaway autoscaling of our workloads.\n\nWe reduced the configured maximum surge of our workloads during application deployments, to reduce pod churn impact on connections.\n\nSince Redis connections and latency remained stable, we then resumed deployments while continuing to monitor the cluster.\n\n## Changes we're making\n\n**Reduce autoscaling sensitivity.**\n\n* We increased the utilization threshold that triggers the Puma Agent autoscaling, so short-lived latency increases do not cause rapid increases in application and Redis connection demand.\n* We increased our scale up stabilization window so that we only add more pods in response to scaling triggers in incremental bursts. This will throttle scaling up based on downstream latency and runaway scaling.\n\n**Reduce deployment-related load.**\n\n* We reduced the maximum deployment surge for Puma and Sidekiq workloads, limiting the additional capacity and downstream connections introduced during deployments.\n\n**Increase immediate Redis capacity by 33%.**\n\n* We added more nodes to the Redis cluster to handle more connections, given the shape of our connection load has shifted with our different thread counts in EKS\n\n**Reduce connection pressure from high-traffic workloads.**\n\n* We moved the two largest Puma Agent sharded deployments back to ECS while we evaluate how the threading changes impact Redis connections and our autoscaling sensitivity.\n\n**Bound aggregate Redis connection demand.**\n\n* We are reviewing Redis connection-pool sizing, application thread counts, and workload scaling limits so that application growth cannot exceed the available Redis connection capacity.",
"postmortem_body_last_updated_at": "2026-09-11T06:31:09.038Z",
"postmortem_ignored": false,
"postmortem_notified_subscribers": true,
"postmortem_notified_twitter": true,
"postmortem_published_at": "2026-09-11T06:31:09.038Z",
"reminder_intervals": null,
"resolved_at": "2026-09-10T01:54:23.000Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/s27r4b25q77q",
"started_at": "2026-09-09T19:07:24.941Z",
"status": "resolved",
"updated_at": "2026-09-11T06:31:09.053Z"
},
{
"components": [
{
"created_at": "2024-10-23T08:15:58.991Z",
"description": "Buildkite's hosted compute in the Pipelines product\r\n\r\nhttps://buildkite.com/docs/pipelines/hosted-agents/overview",
"group_id": "56ph4s45xl9s",
"id": "9pwnnl0dp0j5",
"name": "Hosted Agents",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": "2024-10-23T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.951Z"
},
{
"created_at": "2025-02-06T06:30:06.953Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "87hpgxv274l0",
"name": "MacOS",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.972Z"
},
{
"created_at": "2025-02-06T06:29:12.991Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "sd4q7mk8n853",
"name": "Linux (ARM64)",
"page_id": "ltljpr68dygn",
"position": 3,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.990Z"
},
{
"created_at": "2025-02-06T06:29:50.002Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "fb1llxplp5xq",
"name": "Linux (AMD64)",
"page_id": "ltljpr68dygn",
"position": 4,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:42.009Z"
}
],
"created_at": "2026-09-04T01:42:41.358Z",
"id": "0qmndj65cf0r",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "operational",
"old_status": "operational"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "Jobs running on Hosted Agents are now being dispatched promptly.",
"created_at": "2026-09-04T02:39:31.048Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-04T02:39:31.048Z",
"id": "rbmqqyz1vlxy",
"incident_id": "0qmndj65cf0r",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-04T02:39:31.048Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "operational",
"old_status": "degraded_performance"
}
],
"body": "We have deployed mitigations and have observed recovery across Hosted Agents. We will continue monitoring.",
"created_at": "2026-09-04T02:23:42.041Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-04T02:23:42.041Z",
"id": "f0ncfqfbw8pk",
"incident_id": "0qmndj65cf0r",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-04T02:23:42.041Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "We're seeing increased latency and error rates within our Hosted Agents. We're currently investigating and will provide status updates as they become available.",
"created_at": "2026-09-04T01:42:41.476Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-04T01:42:41.476Z",
"id": "jtqptzkp88jf",
"incident_id": "0qmndj65cf0r",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-04T01:42:41.476Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-04T02:23:42.041Z",
"name": "Hosted Agents - Increased latency and error rates",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-04T02:39:31.048Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/53b786pmb5r9",
"started_at": "2026-09-04T01:42:41.346Z",
"status": "resolved",
"updated_at": "2026-09-04T02:39:31.065Z"
},
{
"components": [
{
"created_at": "2024-10-23T08:15:58.991Z",
"description": "Buildkite's hosted compute in the Pipelines product\r\n\r\nhttps://buildkite.com/docs/pipelines/hosted-agents/overview",
"group_id": "56ph4s45xl9s",
"id": "9pwnnl0dp0j5",
"name": "Hosted Agents",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": "2024-10-23T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.951Z"
}
],
"created_at": "2026-09-03T06:00:38.629Z",
"id": "v59rky62f2h4",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "Jobs running on Hosted Agents are now being dispatched promptly.",
"created_at": "2026-09-03T07:06:38.030Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-03T07:06:38.030Z",
"id": "fw38q10q5xjc",
"incident_id": "v59rky62f2h4",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-03T07:06:38.030Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "operational",
"old_status": "partial_outage"
}
],
"body": "We're seeing recovery in hosted job scheduling rates. Your jobs should run, but there may be some delays as we work through any backlogs.",
"created_at": "2026-09-03T06:44:09.302Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-03T06:44:09.302Z",
"id": "07hs5nz77xvj",
"incident_id": "v59rky62f2h4",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-03T06:44:09.302Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "partial_outage",
"old_status": "operational"
}
],
"body": "Some job dispatches to hosted agents are timing out, leading to jobs starting late. We're currently investigating and will update again soon.",
"created_at": "2026-09-03T06:09:32.713Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-03T06:09:32.713Z",
"id": "3smck91nvb5f",
"incident_id": "v59rky62f2h4",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-03T06:09:32.713Z",
"wants_twitter_update": false
},
{
"affected_components": null,
"body": "We're seeing increased latency and error rates for a subset of our customers. We're currently investigating and will provide status updates as they become available.",
"created_at": "2026-09-03T06:00:38.699Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-03T06:00:38.699Z",
"id": "y0y7gqts6k47",
"incident_id": "v59rky62f2h4",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-03T06:00:38.699Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-03T06:44:09.302Z",
"name": "Scheduling delay for hosted agents",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-03T07:06:38.030Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/ft4f1fbwjbgx",
"started_at": "2026-09-03T06:00:38.620Z",
"status": "resolved",
"updated_at": "2026-09-03T07:06:38.046Z"
},
{
"components": [
{
"created_at": "2024-10-23T08:15:58.991Z",
"description": "Buildkite's hosted compute in the Pipelines product\r\n\r\nhttps://buildkite.com/docs/pipelines/hosted-agents/overview",
"group_id": "56ph4s45xl9s",
"id": "9pwnnl0dp0j5",
"name": "Hosted Agents",
"page_id": "ltljpr68dygn",
"position": 1,
"showcase": true,
"start_date": "2024-10-23T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.951Z"
},
{
"created_at": "2025-02-06T06:30:06.953Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "87hpgxv274l0",
"name": "MacOS",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.972Z"
},
{
"created_at": "2025-02-06T06:29:12.991Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "sd4q7mk8n853",
"name": "Linux (ARM64)",
"page_id": "ltljpr68dygn",
"position": 3,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:41.990Z"
},
{
"created_at": "2025-02-06T06:29:50.002Z",
"description": null,
"group_id": "56ph4s45xl9s",
"id": "fb1llxplp5xq",
"name": "Linux (AMD64)",
"page_id": "ltljpr68dygn",
"position": 4,
"showcase": true,
"start_date": "2025-02-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-09-04T02:23:42.009Z"
}
],
"created_at": "2026-09-02T14:45:58.064Z",
"id": "zf4v5577x0gf",
"impact": "major",
"impact_override": "major",
"incident_updates": [
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "operational",
"old_status": "degraded_performance"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "operational",
"old_status": "degraded_performance"
}
],
"body": "Job dispatch for Hosted Agents is no longer experiencing delays.",
"created_at": "2026-09-02T15:51:19.363Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-02T15:51:19.363Z",
"id": "0b1h6wd2lscy",
"incident_id": "zf4v5577x0gf",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-02T15:51:19.363Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
}
],
"body": "We are continuing to process through the backlog of delayed jobs and monitoring.",
"created_at": "2026-09-02T15:43:03.983Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-02T15:43:03.983Z",
"id": "6ynjbcp47lgd",
"incident_id": "zf4v5577x0gf",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-02T15:43:03.983Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
}
],
"body": "We're beginning to see recovery with delayed job dispatch for Hosted Agents. We are currently working through the backlog of delayed jobs.",
"created_at": "2026-09-02T15:29:28.858Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-02T15:29:28.858Z",
"id": "vstqg9h47q18",
"incident_id": "zf4v5577x0gf",
"status": "identified",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-02T15:29:28.858Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "degraded_performance",
"old_status": "degraded_performance"
}
],
"body": "Our engineers are continuing to investigate an issue with delayed job dispatch for Hosted Agents. Job dispatch for self-hosted agents remains unaffected.",
"created_at": "2026-09-02T15:16:31.260Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-02T15:16:31.260Z",
"id": "zp67cqt360v0",
"incident_id": "zf4v5577x0gf",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-02T15:16:31.260Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "9pwnnl0dp0j5",
"name": "Hosted Agents - Hosted Agents",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "87hpgxv274l0",
"name": "Hosted Agents - MacOS",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "sd4q7mk8n853",
"name": "Hosted Agents - Linux (ARM64)",
"new_status": "degraded_performance",
"old_status": "operational"
},
{
"code": "fb1llxplp5xq",
"name": "Hosted Agents - Linux (AMD64)",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "We're investigating an issue with job dispatch to our Hosted Agents.",
"created_at": "2026-09-02T14:45:58.242Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-02T14:45:58.242Z",
"id": "lcwgcq0wwl6j",
"incident_id": "zf4v5577x0gf",
"status": "investigating",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-02T14:45:58.242Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-02T15:43:03.983Z",
"name": "Delays in Hosted Agents job dispatch",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-02T15:51:19.363Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/k3rcz9j9xkyq",
"started_at": "2026-09-02T14:45:58.047Z",
"status": "resolved",
"updated_at": "2026-09-02T23:27:53.534Z"
},
{
"components": [
{
"created_at": "2015-05-11T01:33:31.806Z",
"description": null,
"group_id": "tplm771kdwfs",
"id": "rj879wvx0lyd",
"name": "Email Notifications",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": false,
"start_date": null,
"status": "operational",
"updated_at": "2026-09-01T16:52:23.716Z"
}
],
"created_at": "2026-09-01T15:32:42.774Z",
"id": "9hd5d7qxhtq4",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "operational",
"old_status": "partial_outage"
}
],
"body": "Our outbound email queue has finished processing all previously failed deliveries. Email notifications are processing successfully.",
"created_at": "2026-09-01T16:52:23.749Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-01T16:52:23.749Z",
"id": "jrmw905f12qg",
"incident_id": "9hd5d7qxhtq4",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-01T16:52:23.749Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "partial_outage",
"old_status": "major_outage"
}
],
"body": "We are seeing successful email deliveries and are monitoring processing of our outbound email queue.",
"created_at": "2026-09-01T16:38:04.623Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-01T16:38:04.623Z",
"id": "m7xpc13ryr44",
"incident_id": "9hd5d7qxhtq4",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-01T16:38:04.623Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "major_outage",
"old_status": "major_outage"
}
],
"body": "Our on-call engineers are working on failing over to an alternative upstream email provider to restore outbound email notifications. No other Buildkite services have been impacted by this disruption.",
"created_at": "2026-09-01T16:19:01.836Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-01T16:19:01.836Z",
"id": "ptjy26wrh1w1",
"incident_id": "9hd5d7qxhtq4",
"status": "identified",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-01T16:19:01.836Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "rj879wvx0lyd",
"name": "Notifications - Email Notifications",
"new_status": "major_outage",
"old_status": "operational"
}
],
"body": "We are experiencing delivery issues with email notifications due to an issue with an upstream provider.",
"created_at": "2026-09-01T15:32:42.847Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-09-01T15:32:42.847Z",
"id": "lltgnjc05vq7",
"incident_id": "9hd5d7qxhtq4",
"status": "identified",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-09-01T15:32:42.847Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-09-01T16:38:04.623Z",
"name": "Delivery issues with email notifications",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-09-01T16:52:23.749Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/3wm6j0rl4b1r",
"started_at": "2026-09-01T15:32:42.765Z",
"status": "resolved",
"updated_at": "2026-09-01T16:52:23.765Z"
},
{
"components": [
{
"created_at": "2022-09-06T05:24:19.824Z",
"description": "Ingestion queue processing for Test Analytics",
"group_id": "lnkc32srq0df",
"id": "dmqj9zfy07kf",
"name": "Ingestion",
"page_id": "ltljpr68dygn",
"position": 2,
"showcase": true,
"start_date": "2022-09-06T00:00:00.000Z",
"status": "operational",
"updated_at": "2026-08-27T11:47:23.200Z"
}
],
"created_at": "2026-08-27T10:56:08.903Z",
"id": "rg1tyqgzbg3w",
"impact": "minor",
"impact_override": "minor",
"incident_updates": [
{
"affected_components": [
{
"code": "dmqj9zfy07kf",
"name": "Test Engine - Ingestion",
"new_status": "operational",
"old_status": "degraded_performance"
}
],
"body": "Processing of uploaded test execution data has caught up and is back to normal.",
"created_at": "2026-08-27T11:47:23.235Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-08-27T11:47:23.235Z",
"id": "0lbdjs0h530c",
"incident_id": "rg1tyqgzbg3w",
"status": "resolved",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-08-27T11:47:23.235Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "dmqj9zfy07kf",
"name": "Test Engine - Ingestion",
"new_status": "degraded_performance",
"old_status": "operational"
}
],
"body": "Catch-up processing continues, most of the backlog is processed, ingestion latency is decreasing.",
"created_at": "2026-08-27T11:23:26.640Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-08-27T11:23:26.640Z",
"id": "0wgmsxf3kps4",
"incident_id": "rg1tyqgzbg3w",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-08-27T11:23:26.640Z",
"wants_twitter_update": false
},
{
"affected_components": [
{
"code": "dmqj9zfy07kf",
"name": "Test Engine - Ingestion",
"new_status": "operational",
"old_status": "operational"
}
],
"body": "Internal processing of uploaded test execution data has been delayed, and is currently catching up. No data has been dropped, and all data should be processed over the next hour.",
"created_at": "2026-08-27T10:56:08.976Z",
"custom_tweet": null,
"deliver_notifications": true,
"display_at": "2026-08-27T10:56:08.976Z",
"id": "q7lq68ysjdnp",
"incident_id": "rg1tyqgzbg3w",
"status": "monitoring",
"tweet_id": null,
"twitter_updated_at": null,
"updated_at": "2026-08-27T10:56:08.976Z",
"wants_twitter_update": false
}
],
"metadata": {},
"monitoring_at": "2026-08-27T10:56:08.976Z",
"name": "Test Engine ingestion delayed, catching up",
"page_id": "ltljpr68dygn",
"postmortem_body": null,
"postmortem_body_last_updated_at": null,
"postmortem_ignored": false,
"postmortem_notified_subscribers": false,
"postmortem_notified_twitter": false,
"postmortem_published_at": null,
"reminder_intervals": null,
"resolved_at": "2026-08-27T11:47:23.235Z",
"scheduled_auto_completed": false,
"scheduled_auto_in_progress": false,
"scheduled_for": null,
"scheduled_remind_prior": false,
"scheduled_reminded_at": null,
"scheduled_until": null,
"shortlink": "https://stspg.io/5ld12yyzc5q3",
"started_at": "2026-08-27T10:56:08.887Z",
"status": "resolved",
"updated_at": "2026-08-27T11:47:23.254Z"
}
],
"page": {
"id": "ltljpr68dygn",
"name": "Buildkite",
"url": "https://www.buildkitestatus.com"
},
"status": {
"description": "All Systems Operational",
"indicator": "none"
}
}