{
"vendor": "InfluxData",
"slug": "influxdata",
"platform": "statuspage",
"status_url": "https://status.influxdata.com",
"last_checked": "2026-09-16T12:28:20Z",
"last_state": "degraded",
"history_backfilled": true,
"first_watched": "2026-09-04T07:06:16Z",
"incidents": [
{
"body": "The issue has been identified and a fix is being implemented.",
"first_seen": "2026-09-16T12:28:20Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"started_at": "2026-09-16T10:08:27.330Z",
"state": "identified",
"title": "Query performance degradation in Azure Westeurope.",
"updated_at": "2026-09-16T11:26:47.071Z",
"url": "https://stspg.io/p7rlkkfdswnm"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-12T12:30:15Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-09-11T19:26:26.397Z",
"resolved_inferred": false,
"started_at": "2026-09-11T16:32:25.112Z",
"state": "resolved",
"title": "SSO Login Service Availability (other login methods not affected)",
"updated_at": "2026-09-11T19:26:26.417Z",
"url": "https://stspg.io/7dt3n0grpnx2"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-11T12:29:42Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-09-11T09:46:51.096Z",
"resolved_inferred": false,
"started_at": "2026-09-11T08:45:49.624Z",
"state": "resolved",
"title": "Query performance degradation in Azure Westeurope",
"updated_at": "2026-09-11T09:46:51.113Z",
"url": "https://stspg.io/p3644c1wj0zk"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-07T12:26:33Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-09-07T16:03:18.940Z",
"resolved_inferred": false,
"started_at": "2026-09-07T10:11:46.600Z",
"state": "resolved",
"title": "Query performance degradation in Azure West Europe",
"updated_at": "2026-09-07T16:03:18.954Z",
"url": "https://stspg.io/ctrr38xqs81c"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-08-25T15:41:54.986Z",
"resolved_inferred": false,
"started_at": "2026-08-25T13:49:50.103Z",
"state": "resolved",
"title": "Query performance degradation in Azure Westeurope",
"updated_at": "2026-08-25T15:41:55.005Z",
"url": "https://stspg.io/rrn627w0gk7s"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-05-27T16:58:45.280Z",
"resolved_inferred": false,
"started_at": "2026-05-27T15:51:52.865Z",
"state": "resolved",
"title": "Increased Error rates in Azure eastus",
"updated_at": "2026-05-27T16:58:45.295Z",
"url": "https://stspg.io/yzpp1z4brvmg"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-03-17T22:41:04.909Z",
"resolved_inferred": false,
"started_at": "2026-03-17T19:18:18.200Z",
"state": "resolved",
"title": "Degraded query performance in AWS eu-central-1 Serverless",
"updated_at": "2026-03-17T22:41:04.925Z",
"url": "https://stspg.io/3f1bp28mtcww"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-03-10T20:24:40.504Z",
"resolved_inferred": false,
"started_at": "2026-03-10T17:26:08.315Z",
"state": "resolved",
"title": "Intermittent Read/Write errors in Azure prod01-us-east-1",
"updated_at": "2026-03-10T20:24:40.523Z",
"url": "https://stspg.io/zpvfz0jprvkd"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-03-06T04:34:22.679Z",
"resolved_inferred": false,
"started_at": "2026-03-06T03:05:20.656Z",
"state": "resolved",
"title": "Elevated Ingest Failures: Cloud Serverless AWS, EU-Central",
"updated_at": "2026-03-06T04:34:22.693Z",
"url": "https://stspg.io/1dkl20dg6156"
},
{
"body": "Cloud Dedicated Management API and Admin UI access issue has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-02-19T00:10:42.882Z",
"resolved_inferred": false,
"started_at": "2026-02-18T23:23:23.905Z",
"state": "resolved",
"title": "Outage of Cloud Dedicated Management API and Admin UI",
"updated_at": "2026-02-19T00:10:42.905Z",
"url": "https://stspg.io/f4pwyg61xljj"
},
{
"body": "Cloud Dedicated Management API and Admin UI access issue has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-02-17T17:11:41.893Z",
"resolved_inferred": false,
"started_at": "2026-02-16T16:48:47.149Z",
"state": "resolved",
"title": "Outage of Cloud Dedicated Management API and Admin UI",
"updated_at": "2026-02-17T17:11:41.913Z",
"url": "https://stspg.io/93lxwbqj412c"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-02-12T16:31:41.777Z",
"resolved_inferred": false,
"started_at": "2026-02-12T11:26:55.161Z",
"state": "resolved",
"title": "Query errors in Azure West Europe",
"updated_at": "2026-02-12T16:31:41.790Z",
"url": "https://stspg.io/fkyjr2hkpgcm"
},
{
"body": "## Summary\n\nA software change was introduced that was intended to improve data compaction for very large partitions. The change had a bug that, under certain circumstances, caused some data to be dropped during the compaction process. When compaction completed, the new compacted file \\(missing a portion of the source data\\) replaced the original files. This made some data unavailable until it was restored from backup.\n\n\u200c\n\nBecause the underlying data remained intact in backups, no permanent data loss occurred. After identifying the issue, we rolled back the software change and initiated a full data restoration from backups. While the restoration process is reliable, restoring data at cluster scale while maintaining normal system operations required an extended recovery period. During the restoration process, some customers experienced elevated TTBR and degraded query performance while restoration was underway.\u00a0\n\n## Cause\n\nData in InfluxDB Cloud 2 is organized into 64 partitions. Each partition is regularly compacted to improve query performance since uncompacted data is slower to query.\u00a0\u00a0\n\n\u200c\n\nIn January, we identified that one of the partitions in prod101-us-east-1 \\(partition 47\\) was no longer compacting successfully. As a result, partially compacted files accumulated in this partition and began impacting performance. To address this, our engineering team designed and implemented an optimization to run multiple compaction jobs to run in parallel. Each job targeted a subset of the data in the partition \\(based on the org id prefix\\), enabling heavily loaded partitions, such as partition 47, to be compacted more effectively.  \n\nThis code change was reviewed by two additional engineers as part of our standard release process. However, the optimization introduced a bug that, in certain circumstances, caused the compactor to fail when reading some of the source TSM data files. This compactor did not correctly handle this failure and instead silently skipped the TSM files, compacted the rest of the data, and then replaced the original files with the compacted data. This issue was not detected during code review or internal testing, and only surfaced after the change was applied to prod101-us-east-1.\u00a0\n\n## Recovery\n\nOnce the issue was identified, we immediately rolled back the compaction change to prevent any further impact. We then began restoring the affected data.\u00a0\n\n\u200c\n\nCloud 2 maintains two separate backup sources. Each cluster is protected by weekly TSM snapshots, and all incoming writes are recorded as snapshots in Kafka logs. Together, these provide a complete record of the data written to the cluster and allow us to fully restore the affected data.\n\n\u200c\n\nBecause the issue only impacted partitions that were compacted while the buggy code was running, not all data in the cluster was affected. However, as a precaution, the recovery process prioritized restoring the most recent data first as it is typically the most critical data for customer operations. Recent data is reconstructed from Kafka logs, which contain a complete record of writes into the cluster. Historical data is then restored from TSM snapshots.\u00a0\u00a0\n\n\u200c\n\nThe cluster remained online throughout the recovery process because customers rely on it for real-time workloads. As a result, restoration had to be performed incrementally so recovery activity would not disrupt ongoing ingest and query operations.\u00a0\n\n\u200c\n\nAlthough this approach takes longer than restoring the cluster offline from a full backup, it allows the system to remain available while the recovery progresses. During portions of the recovery, restoration activity resulted in elevated TTBR and slower query performance.\n\n\u200c\n\n## Timeline\n\nFeb 9 - Deployed compaction change to prod101-us-east-1.\n\nFeb 9 - First ticket raised indicating potential data availability issues.\u00a0 Investigation began.\n\nFeb 10 - Additional tickets raised. Root cause identified.\n\nFeb 10 - Compaction change rolled back.\n\nFeb 10 - Data restoration process initiated, starting with the most recent data.\n\nFeb 12 - Recent data \\(Feb 2-10\\) fully restored for a test set of impacted customers. Recent data restoration began for all customers.\u00a0\n\nFeb 17 - Modified restoration process to improve efficiency for historical data recovery.\n\nFeb 23 - Recent data \\(Feb 2-9\\) fully restored for all customers.\n\nFeb 23 - Historical data restoration resumed for all customer data from Feb 2 and earlier\n\nFeb 25 - Historical data is fully restored for initial test set of customers.\n\nMar 8 - Historical data fully restored for all paying customers.\n\n\u200c\n\n## Future mitigation\n\n\u200c\n\nWe recognize the disruption this incident caused and sincerely apologize to customers who were impacted. During portions of the recovery process, some customers experienced degraded query performance while restoration was underway. Reliable access to your data is critical, and we take incidents like this very seriously.\n\n\u200c\n\nFollowing this incident, we are implementing several improvements to reduce the risk of similar issues in the future.\u00a0\n\n\u200c\n\nWe are expanding our testing coverage for changes that affect critical storage components such as the compactor, including additional test scenarios that better reflect production-scale workloads.\u00a0\n\n\u200c\n\nWe are also enhancing safeguards around the compaction process to ensure failures are detected and handled safely.\n\n\u200c\n\nIn addition, we are reviewing our restoration workflows and operational procedures to reduce the potential for elevated TTBR or query performance degradation during large-scale restoration.\n\n\u200c\n\nWe remain committed to strengthening the resilience and operational safety of the platform.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "none",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-03-10T01:19:52.582Z",
"resolved_inferred": false,
"started_at": "2026-02-10T17:37:31.884Z",
"state": "postmortem",
"title": "Restoration of historical data in AWS, US-East-1",
"updated_at": "2026-03-10T01:22:05.861Z",
"url": "https://stspg.io/vbrkgjcggffn"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-26T22:57:25.877Z",
"resolved_inferred": false,
"started_at": "2026-01-26T22:45:57.141Z",
"state": "resolved",
"title": "Intermittent UI login errors on AWS us-east-1",
"updated_at": "2026-01-26T22:57:25.890Z",
"url": "https://stspg.io/vcxzy0jn4pck"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-23T07:42:52.540Z",
"resolved_inferred": false,
"started_at": "2026-01-23T03:19:53.999Z",
"state": "resolved",
"title": "Query failures in US-EAST-1",
"updated_at": "2026-01-23T07:42:52.564Z",
"url": "https://stspg.io/fj8371hb74b2"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-21T16:48:01.654Z",
"resolved_inferred": false,
"started_at": "2026-01-21T16:08:44.548Z",
"state": "resolved",
"title": "Cloud2 UI in the AWS EU-CENTRAL Region Experiencing Intermittent Authentication Issues (read/write tokens are not affected)",
"updated_at": "2026-01-21T16:48:01.679Z",
"url": "https://stspg.io/mbhwjk0xtx91"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-20T16:41:19.050Z",
"resolved_inferred": false,
"started_at": "2026-01-20T12:31:12.627Z",
"state": "resolved",
"title": "Increased query failures in AWS eu-central-1",
"updated_at": "2026-01-20T16:41:19.066Z",
"url": "https://stspg.io/xl226v3p2zmg"
},
{
"body": "**Summary**\n\nOn January 15, 2026, user login for Cloud 2 was failing intermittently in the eu-central AWS\n\ncluster. The impact was that for some users, they could not login via the web UI. For others,\n\nthey were able to login but then could not see their resources \\(buckets, dashboards etc.\\) in the\n\nweb UI. This was an intermittent issue, not affecting all users, and in some cases, recoverable\n\nby retrying. It was also isolated to the web UI and did not impact API-based writes and queries.\n\n**Cause of the Incident**\n\nThe incident was caused by the clock on some of the nodes being slightly behind the master\n\nclock. When a user logs in, a JWT session token is signed with a nbf \\(not-before\\) timestamp\n\nbased on the signing server's clock. When the request hits a gateway pod, the token is checked\n\nand one of the checks is to confirm that the token time is valid. If the gateway pod is one whose\n\nnode clock is behind the signing node's clock, the JWT is rejected because from that node's\n\nperspective, the token's nbf time is in the future.\n\nThe reason why this problem was intermittent was that a load balancer distributed the\n\nauthentication requests across pods, and not all the pods had clock-skew. The same token\n\nworked fine on some of the nodes \\(where the time was correct\\) and not on a few nodes \\(where\n\nthe time was incorrect\\).\n\n**Recovery**\n\nWe identified the nodes whose clock time had drifted, and drained the affected nodes.\n\nWhen the service was restarted on new nodes with the correct time, the problem was\n\nresolved.\n\n**Timeline**\n\n* 2026-01-15 10:12 UTC - Customers reported errors when navigating to dashboards and\n\ntasks in the UI in the eu-central AWS cluster, and engineering began to investigate the\n\nissue.\n\n* 2026-01-15 14:34 UTC - Additional customers reported the same errors.\n* 2026-01-15 16:28 UTC - The issue was identified as an intermittent user login issue.\n* 2026-01-15 21:12 UTC - Further investigation revealed clock skew in some nodes on the\n\ncluster.\n\n* 2026-01-15 21:41 UTC - The nodes that were showing signs of clock skew were\n\ndrained.\n\n* 2026-01-15 21:45 UTC - Confirmed that authentications were successful.\n\n* 2026-01-16 16:11 UTC - An additional node was showing signs of clock skew, so the\n\nnode was drained and the incident was closed.\n\n**Future mitigation**\n\nWe are still researching why the clocks drifted on these nodes, as these nodes use\n\nthe AWS-provided time synchronization service, which should have kept the\n\nclocks in sync.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-16T16:21:25.660Z",
"resolved_inferred": false,
"started_at": "2026-01-15T17:18:55.000Z",
"state": "postmortem",
"title": "Intermittent Authentication Issues for AWS EU-CENTRAL (read/write token authentication is not affected)",
"updated_at": "2026-01-16T23:08:56.936Z",
"url": "https://stspg.io/j9pr53c9h4bl"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-15T19:29:58.712Z",
"resolved_inferred": false,
"started_at": "2026-01-15T17:15:15.000Z",
"state": "resolved",
"title": "Disrupted query performance in AWS eu-central-1",
"updated_at": "2026-01-15T19:29:58.729Z",
"url": "https://stspg.io/vz9v69fsxswk"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-12T12:10:02.906Z",
"resolved_inferred": false,
"started_at": "2026-01-12T11:32:18.012Z",
"state": "resolved",
"title": "Query errors in AWS eu-central-1",
"updated_at": "2026-01-12T12:10:02.925Z",
"url": "https://stspg.io/d48pkz26wmyj"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-13T15:55:39.967Z",
"resolved_inferred": false,
"started_at": "2026-01-09T10:41:22.441Z",
"state": "resolved",
"title": "Outage in AWS eu-central-1 - visibility to tasks",
"updated_at": "2026-01-13T15:55:39.984Z",
"url": "https://stspg.io/6k5t1bxlj1yp"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-08T18:18:44.247Z",
"resolved_inferred": false,
"started_at": "2026-01-08T13:24:32.896Z",
"state": "resolved",
"title": "High query error rates in AWS EU-CENTRAL-1 cluster",
"updated_at": "2026-01-08T18:18:44.264Z",
"url": "https://stspg.io/09p9r0bvpjqj"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2026-01-07T05:16:44.753Z",
"resolved_inferred": false,
"started_at": "2026-01-06T21:27:27.928Z",
"state": "resolved",
"title": "Increase in query times in the US-EAST-1 and EU-CENTRAL-1 clusters",
"updated_at": "2026-01-07T05:16:44.773Z",
"url": "https://stspg.io/65nc9x3p5ymk"
},
{
"body": "Query error rates have returned to normal levels",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-12-09T22:11:03.794Z",
"resolved_inferred": false,
"started_at": "2025-12-09T16:19:11.476Z",
"state": "resolved",
"title": "Degraded Query Performance For Multiple Cloud Dedicated Clusters",
"updated_at": "2025-12-09T22:11:03.814Z",
"url": "https://stspg.io/8k34w6hhktvb"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-12-09T11:37:03.668Z",
"resolved_inferred": false,
"started_at": "2025-12-09T09:49:41.382Z",
"state": "resolved",
"title": "Query errors in AWS eu-central-1 (Serverless)",
"updated_at": "2025-12-09T11:37:03.685Z",
"url": "https://stspg.io/w524qf6d0979"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-20T21:39:17.066Z",
"resolved_inferred": false,
"started_at": "2025-11-20T21:03:31.844Z",
"state": "resolved",
"title": "InfluxDB Cloud Serverless UI isn\u2019t showing measurements, and the issue is also affecting API queries.",
"updated_at": "2025-11-20T21:39:17.097Z",
"url": "https://stspg.io/mn2vzkq36g64"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-20T14:07:43.944Z",
"resolved_inferred": false,
"started_at": "2025-11-20T08:52:19.082Z",
"state": "resolved",
"title": "Google OAuth Logins issues for Dedicated and Serverless customers",
"updated_at": "2025-11-20T14:07:43.962Z",
"url": "https://stspg.io/zgdnprb90b6k"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-19T01:47:22.482Z",
"resolved_inferred": false,
"started_at": "2025-11-18T19:48:25.748Z",
"state": "resolved",
"title": "Degraded query performance in GCP us-central1",
"updated_at": "2025-11-19T01:47:22.500Z",
"url": "https://stspg.io/1nd18xst2b2v"
},
{
"body": "RCA for Cloud 2 prod01-us-east-1 outage on Nov 11, 2025\n\n# Summary\n\nThe incident began during a planned migration of storage pods to a newer, more performant node class in Azure. The root cause was insufficient capacity in the new Azure node pool. When additional capacity could not be obtained, the team had to spend significant time rebalancing workloads across existing node pools to find a configuration capable of sustaining the write and query load.\n\n\u200c\n\nClear communication was difficult during the incident because multiple mitigation paths were being explored in parallel, and we did not have confidence in any single fix until late in the recovery. A contributing factor to the duration of the outage was the amount of time Azure requires to apply node pool adjustments, which slowed our ability to test and iterate on potential solutions. Once sufficient capacity was available and the storage pods were redistributed appropriately, the cluster recovered.\u00a0\n\n\u200c\n\nBackground\n\nAs part of an ongoing initiative to improve performance across Cloud 2, we are upgrading clusters to newer virtual machine types that were not available when the platform originally launched. These newer node classes provide higher throughput and allow us to right-size workloads more efficiently. This upgrade process has been completed successfully and without downtime in our AWS and GCP environments, as well as in our Azure staging cluster.\n\n\u200c\n\nAs with all planned maintenance, an on-call engineering team was monitoring the migration and prepared to intervene if unexpected behavior occurred.\n\n\u200c\n\nIncident Details\u00a0\n\nCloud 2 clusters consist of 128 storage pods \\(64 primary and 64 secondary\\). Normal operation requires all 64 pods to be available \\(either primary or secondary\\). Our standard migration sequence moves secondary pods first, monitors stability, and then transitions primary pods once the new node pool has demonstrated adequate capacity.\n\n\u200c\n\nDuring the migration, secondary pods were successfully moved to the new Azure node pool and appeared stable under their normal load. However, when we initiated the transition of primary pods, the increased traffic on the secondaries exposed that the new pool did not have enough capacity to sustain the full workload. As soon as the capacity issue became apparent, engineers began working through multiple mitigation paths in parallel, including alternative node-pool configurations and fallback options. While the primaries were active, they were absorbing a significant share of read and write traffic, masking the fact that the secondaries were running close to their limits. When we attempted to scale the new node pool to provide the required capacity, Azure reported that no additional nodes of that type were available in the region. This prevented in-place scaling and removed the most direct mitigation path. To restore service, the team had to evaluate multiple fallback configurations using other node pools still available to us.\n\n\u200c\n\nEach adjustment required Azure to perform a full node-pool update cycle, which includes provisioning a temporary pool, shifting workloads, and then applying changes to the original pool. These operations take a significant amount of time in Azure, which slowed our ability to test and validate alternative configurations quickly.\n\n\u200c\n\nThe combination of these factors \\(an under-capacity new node pool, limited regional availability for scaling, and long Azure node-pool update times\\) extended the overall duration of the outage. Once additional capacity using the prior node class was provisioned and storage pods were redistributed across the updated pool layout, the cluster stabilized and normal performance resumed.\n\n# Communication\n\nCommunication during the incident did not meet expectations. Although we issued updates on our status page \\([status.influxdata.com](http://status.influxdata.com)\\), they lacked clarity because the team was evaluating several mitigation paths concurrently and did not have confidence that any single approach would resolve the issue.\n\n\u00a0\n\nWe were also slow to communicate full recovery. Additionally, an internal miscommunication led to inaccurate information being shared in our community Slack, suggesting the issue was not being addressed due to the U.S. holiday. This was incorrect; engineers were actively working throughout the incident. We recognize the impact this had on customer trust and have identified this as a process issue.\n\n# Future Mitigations\n\n1. Strengthened Capacity and Quota Planning\n\nWe will incorporate earlier and stricter quota and capacity validation into the migration process. Future node-pool changes will require confirming that available Azure quota is at least double the projected capacity needed for a safe migration.\n\n\u200c\n\n2\\. Revised Migration Procedure\n\nWe will update our migration approach to move smaller slices of primaries and secondaries together. This will surface capacity issues earlier and prevent scenarios where a node pool appears healthy until the final migration step.\n\n\u200c\n\n3\\. Improved Internal Communication During Incidents\n\nWe are refining our internal communication channels to ensure all customer-facing employees receive timely, authoritative information during outages. We are also reviewing expectations for status-page updates to ensure clarity and consistency.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-12T04:17:02.534Z",
"resolved_inferred": false,
"started_at": "2025-11-11T02:46:44.576Z",
"state": "postmortem",
"title": "Degraded read performance in Azure US-East 1",
"updated_at": "2025-11-14T18:40:40.390Z",
"url": "https://stspg.io/kbfvmsfd67gx"
},
{
"body": "RCA for Cloud 2 prod01-us-central-1 query outage on Nov 10, 2025\n\n# Summary\n\nThe SRE team has been working on a long-term project to review and rebalance the storage workloads, to provide better separation of workloads within nodepools, which will improve performance, reliability and ease of maintenance. \n\nThe Cloud 2 service is designed with a set of primary and secondary storage pods. During normal deployments, whenever a primary pod is restarted, the secondary pod takes over, so that the service is not interrupted. The change that was made in this cluster was a change to the primary and the secondary pods. When the configuration change was made in prod01-us-central-1, it caused some of the primary storage pods and their corresponding secondary pods to be restarted at the same time. While the pods were unavailable, queries failed. Once the pods restarted, the query error rate went back to normal levels and the cluster recovered. \n\n# Timeline \n\n**Time \\(UTC\\)   What happened** \n\n**8:45pm**          The engineer-on-call was paged due to a high rate of query errors in the cluster. Upon investigation, the primary and secondary pods for the same storage slice are unavailable. \n\n**8:50pm**          Engineer-on-call identified that the problem was due to a misconfiguration that caused the primary and secondary pods to be restarted at the same time. \n\n**9:00pm**          The system will recover by itself when the pods restart with the new configuration, so no intervention is required.\n\n**9:20pm**          Queriers start recovering, and error rate is returning to pre-incident level. Some lingering errors in nginx. \n\n**10:00pm**        Query backlog is fully cleared. The cluster has fully recovered.\n\n# Future Mitigations \n\n**1. Reviewing procedures relating to storage pod reconfigurations** \n\nGoing forward we will ensure that configuration changes for primary pods and secondary pods are never applied at the same time.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-10T23:52:48.794Z",
"resolved_inferred": false,
"started_at": "2025-11-10T21:04:34.000Z",
"state": "postmortem",
"title": "Degraded query performance in GCP us-central1",
"updated_at": "2025-11-25T23:55:00.497Z",
"url": "https://stspg.io/j31d7rqd8h8k"
},
{
"body": "The incident has been resolved, and operations are back to normal",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-11-05T21:54:27.532Z",
"resolved_inferred": false,
"started_at": "2025-11-05T18:17:10.240Z",
"state": "resolved",
"title": "Degraded writes in AWS us-east-1 Serverless Cluster",
"updated_at": "2025-11-10T22:37:19.593Z",
"url": "https://stspg.io/4sqvwdkl118z"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-28T16:06:19.158Z",
"resolved_inferred": false,
"started_at": "2025-10-28T09:49:25.714Z",
"state": "resolved",
"title": "504 gateway errors in us-east-1",
"updated_at": "2025-10-28T16:06:19.174Z",
"url": "https://stspg.io/g208mtppxbpd"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "critical",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-28T16:05:54.739Z",
"resolved_inferred": false,
"started_at": "2025-10-28T09:23:36.987Z",
"state": "resolved",
"title": "Degraded Query Performance and High Error Rates \u2013 AWS EU-Central Serverless",
"updated_at": "2025-10-28T16:05:54.752Z",
"url": "https://stspg.io/5m2h149kmfhf"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-28T04:57:57.959Z",
"resolved_inferred": false,
"started_at": "2025-10-28T02:04:14.570Z",
"state": "resolved",
"title": "Degraded Query Performance and High Error Rates \u2013 AWS EU-Central Serverless",
"updated_at": "2025-10-28T04:57:57.977Z",
"url": "https://stspg.io/tdnv754s1t9g"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-21T00:15:19.266Z",
"resolved_inferred": false,
"started_at": "2025-10-20T18:45:58.343Z",
"state": "resolved",
"title": "Intermittent API access in AWS us-east-1",
"updated_at": "2025-10-21T00:15:19.283Z",
"url": "https://stspg.io/gyhhyyrk1vxw"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-20T15:27:53.374Z",
"resolved_inferred": false,
"started_at": "2025-10-20T10:44:22.833Z",
"state": "resolved",
"title": "Intermittent issues with API access in AWS us-east-1",
"updated_at": "2025-10-20T15:27:53.396Z",
"url": "https://stspg.io/s084b1h60cxh"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-17T18:56:47.765Z",
"resolved_inferred": false,
"started_at": "2025-10-17T17:20:32.139Z",
"state": "resolved",
"title": "Increased query latency on AWS us-east-1",
"updated_at": "2025-10-17T18:56:47.780Z",
"url": "https://stspg.io/r4f1c5yz22wj"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-15T22:05:07.130Z",
"resolved_inferred": false,
"started_at": "2025-10-15T18:53:56.000Z",
"state": "resolved",
"title": "Degraded SQL Query and write Performance \u2013 AWS EU-Central-1 Serverless",
"updated_at": "2025-10-15T22:05:07.144Z",
"url": "https://stspg.io/ddyw3ch71j51"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "none",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-10-03T17:15:28.398Z",
"resolved_inferred": false,
"started_at": "2025-10-03T15:11:07.468Z",
"state": "resolved",
"title": "AWS eu-central-1: Degraded database query performance",
"updated_at": "2025-10-03T17:15:28.414Z",
"url": "https://stspg.io/dlqfxpb80945"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "minor",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-09-24T19:21:12.263Z",
"resolved_inferred": false,
"started_at": "2025-09-24T16:00:35.146Z",
"state": "resolved",
"title": "504 responses to queries in us-east-1",
"updated_at": "2025-09-24T19:21:12.287Z",
"url": "https://stspg.io/vl9zc403nlxm"
},
{
"body": "This incident has been resolved.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-09-18T01:02:53.337Z",
"resolved_inferred": false,
"started_at": "2025-09-17T23:00:26.954Z",
"state": "resolved",
"title": "The login and signup pages for Cloud2 accounts are currently down.",
"updated_at": "2025-09-18T01:02:53.355Z",
"url": "https://stspg.io/gqcxm6h35h4m"
},
{
"body": "A fix has been implemented and service to the Quartz Signup page should now be restored.",
"first_seen": "2026-09-04T07:06:16Z",
"impact": "major",
"last_seen": "2026-09-16T12:28:20Z",
"resolved_at": "2025-09-04T22:33:28.189Z",
"resolved_inferred": false,
"started_at": "2025-09-04T22:30:18.657Z",
"state": "resolved",
"title": "Quartz Signup Issues",
"updated_at": "2025-09-04T22:33:28.204Z",
"url": "https://stspg.io/mns3rwdvvmkm"
}
]
}