|
2026-07-29
ยง
|
| 10:21 |
<cwilliams@cumin1003> |
dbctl commit (dc=all): 'Depooling db1186 (T431660)', diff saved to https://phabricator.wikimedia.org/P95493 and previous config saved to /var/cache/conftool/dbconfig/20260729-102111-cwilliams.json |
[production] |
| 10:21 |
<cwilliams@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 1 day, 0:00:00 on db1186.eqiad.wmnet with reason: Maintenance |
[production] |
| 10:14 |
<ayounsi@cumin1003> |
END (PASS) - Cookbook sre.dns.admin (exit_code=0) DNS admin: pool magru [reason: router upgrade, T431750] |
[production] |
| 10:14 |
<ayounsi@cumin1003> |
START - Cookbook sre.dns.admin DNS admin: pool magru [reason: router upgrade, T431750] |
[production] |
| 09:54 |
<godog> |
test dumps-nfs failover - T432325 |
[tools] |
| 09:53 |
<XioNoX> |
reboot cr2-magru - T431750 |
[production] |
| 09:52 |
<XioNoX> |
drain cr2-magru - T431750 |
[production] |
| 09:48 |
<root@cumin1003> |
START - Cookbook sre.mysql.pool pool db1221: Maintenance |
[production] |
| 09:47 |
<godog> |
enable 'dumps_use_nfs_lb: true' for nfs worker prefix - T432325 |
[tools] |
| 09:45 |
<dcaro@cloudcumin1001> |
END (PASS) - Cookbook wmcs.toolforge.component.deploy (exit_code=0) for component maintain-kubeusers (T433184) |
[tools] |
| 09:45 |
<elukey@cumin1003> |
END (PASS) - Cookbook sre.hosts.reboot-single (exit_code=0) for host zookeeper-test1002.eqiad.wmnet |
[production] |
| 09:45 |
<dcaro@cloudcumin1001> |
START - Cookbook wmcs.toolforge.component.deploy for component maintain-kubeusers (T433184) |
[tools] |
| 09:45 |
<root@cumin1003> |
START - Cookbook sre.mysql.pool pool db1188: Maintenance |
[production] |
| 09:45 |
<root@cumin1003> |
START - Cookbook sre.mysql.pool pool db1175: Maintenance |
[production] |
| 09:44 |
<btullis@dns1004> |
END - running authdns-update |
[production] |
| 09:42 |
<btullis@dns1004> |
START - running authdns-update |
[production] |
| 09:42 |
<cwilliams@cumin1003> |
dbctl commit (dc=all): 'Depooling db1221 (T431660)', diff saved to https://phabricator.wikimedia.org/P95483 and previous config saved to /var/cache/conftool/dbconfig/20260729-094200-cwilliams.json |
[production] |
| 09:41 |
<cwilliams@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2 days, 0:00:00 on 7 hosts with reason: Maintenance |
[production] |
| 09:41 |
<cwilliams@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 1 day, 0:00:00 on db1221.eqiad.wmnet with reason: Maintenance |
[production] |
| 09:41 |
<elukey@cumin1003> |
START - Cookbook sre.hosts.reboot-single for host zookeeper-test1002.eqiad.wmnet |
[production] |
| 09:41 |
<root@cumin1003> |
END (PASS) - Cookbook sre.mysql.pool (exit_code=0) pool db1199: Maintenance |
[production] |
| 09:40 |
<dcaro@cloudcumin1001> |
END (PASS) - Cookbook wmcs.toolforge.component.deploy (exit_code=0) for component maintain-kubeusers (T433184) |
[tools] |
| 09:40 |
<dcaro@cloudcumin1001> |
START - Cookbook wmcs.toolforge.component.deploy for component maintain-kubeusers (T433184) |
[tools] |
| 09:39 |
<cwilliams@cumin1003> |
dbctl commit (dc=all): 'Depooling db1188 (T431660)', diff saved to https://phabricator.wikimedia.org/P95481 and previous config saved to /var/cache/conftool/dbconfig/20260729-093917-cwilliams.json |
[production] |
| 09:39 |
<cwilliams@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 1 day, 0:00:00 on db1188.eqiad.wmnet with reason: Maintenance |
[production] |
| 09:38 |
<root@cumin1003> |
END (PASS) - Cookbook sre.mysql.pool (exit_code=0) pool db1182: Maintenance |
[production] |
| 09:38 |
<cwilliams@cumin1003> |
dbctl commit (dc=all): 'Depooling db1175 (T431660)', diff saved to https://phabricator.wikimedia.org/P95479 and previous config saved to /var/cache/conftool/dbconfig/20260729-093842-cwilliams.json |
[production] |
| 09:38 |
<cwilliams@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 1 day, 0:00:00 on db1175.eqiad.wmnet with reason: Maintenance |
[production] |
| 09:38 |
<root@cumin1003> |
END (PASS) - Cookbook sre.mysql.pool (exit_code=0) pool db1166: Maintenance |
[production] |
| 09:35 |
<root@cumin1003> |
END (PASS) - Cookbook sre.mysql.pool (exit_code=0) pool db1169: Maintenance |
[production] |
| 09:33 |
<marostegui@cumin1003> |
conftool action : set/pooled=yes; selector: name=clouddb1033.eqiad.wmnet,service=s8 |
[production] |
| 09:33 |
<marostegui@cumin1003> |
conftool action : set/pooled=yes; selector: name=clouddb1033.eqiad.wmnet,service=s5 |
[production] |
| 09:33 |
<marostegui@cumin1003> |
conftool action : set/weight=100; selector: name=clouddb1033.eqiad.wmnet,service=s8 |
[production] |
| 09:33 |
<marostegui@cumin1003> |
conftool action : set/weight=100; selector: name=clouddb1033.eqiad.wmnet,service=s5 |
[production] |
| 09:21 |
<bwojtowicz@deploy1003> |
helmfile [ml-serve-eqiad] Ran 'sync' command on namespace 'llm' for release 'main' . |
[production] |
| 09:21 |
<XioNoX> |
reboot cr1-magru - T431750 |
[production] |
| 09:17 |
<XioNoX> |
drain cr1-magru - T431750 |
[production] |
| 09:15 |
<elukey@cumin1003> |
END (PASS) - Cookbook sre.hosts.reboot-single (exit_code=0) for host idm1001.wikimedia.org |
[production] |
| 09:15 |
<btullis@deploy1003> |
helmfile [dse-k8s-eqiad] DONE helmfile.d/dse-k8s-services/spark-history: apply |
[production] |
| 09:14 |
<btullis@deploy1003> |
helmfile [dse-k8s-eqiad] START helmfile.d/dse-k8s-services/spark-history: apply |
[production] |
| 09:13 |
<btullis@deploy1003> |
helmfile [dse-k8s-eqiad] DONE helmfile.d/dse-k8s-services/airflow-analytics-test: apply |
[production] |
| 09:13 |
<btullis@deploy1003> |
helmfile [dse-k8s-eqiad] START helmfile.d/dse-k8s-services/airflow-analytics-test: apply |
[production] |
| 09:11 |
<ayounsi@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2:00:00 on cr2-magru,cr2-magru IPv6,cr2-magru.mgmt with reason: router upgrade |
[production] |
| 09:11 |
<elukey@cumin1003> |
START - Cookbook sre.hosts.reboot-single for host idm1001.wikimedia.org |
[production] |
| 09:11 |
<elukey@cumin1003> |
END (PASS) - Cookbook sre.hosts.reboot-single (exit_code=0) for host idp1005.wikimedia.org |
[production] |
| 09:07 |
<elukey@cumin1003> |
START - Cookbook sre.hosts.reboot-single for host idp1005.wikimedia.org |
[production] |
| 09:04 |
<elukey@cumin1003> |
END (PASS) - Cookbook sre.hosts.reboot-single (exit_code=0) for host idp2005.wikimedia.org |
[production] |
| 09:00 |
<elukey@cumin1003> |
START - Cookbook sre.hosts.reboot-single for host idp2005.wikimedia.org |
[production] |
| 09:00 |
<ayounsi@cumin1003> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 1:00:00 on cr1-magru,cr1-magru IPv6,cr1-magru.mgmt with reason: router upgrade |
[production] |
| 09:00 |
<marostegui> |
Dropping renamed tables T426341 |
[production] |