|
2026-09-21
ยง
|
| 18:32 |
<cdobbins@cumin1004> |
conftool action : set/pooled=yes; selector: name=ncredir5003.* |
[production] |
| 18:30 |
<Emperor> |
repool eqiad sessionstore T437915 |
[production] |
| 18:30 |
<mvernon@cumin1004> |
START - Cookbook sre.discovery.service-route pool sessionstore in eqiad: sessionstore1005 repaired |
[production] |
| 18:27 |
<mvernon@cumin1004> |
END (PASS) - Cookbook sre.discovery.service-route (exit_code=0) check sessionstore: maintenance |
[production] |
| 18:27 |
<mvernon@cumin1004> |
START - Cookbook sre.discovery.service-route check sessionstore: maintenance |
[production] |
| 18:25 |
<ebernhardson@deploy1003> |
helmfile [dse-k8s-eqiad] DONE helmfile.d/dse-k8s-services/opensearch-semantic-search-test: apply |
[production] |
| 18:25 |
<ebernhardson@deploy1003> |
helmfile [dse-k8s-eqiad] START helmfile.d/dse-k8s-services/opensearch-semantic-search-test: apply |
[production] |
| 18:24 |
<cdobbins@cumin1004> |
END (PASS) - Cookbook sre.hosts.reimage (exit_code=0) for host ncredir5003.eqsin.wmnet with OS trixie |
[production] |
| 17:54 |
<cdobbins@cumin1004> |
END (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2:00:00 on ncredir5003.eqsin.wmnet with reason: host reimage |
[production] |
| 17:50 |
<cdobbins@cumin1004> |
START - Cookbook sre.hosts.downtime for 2:00:00 on ncredir5003.eqsin.wmnet with reason: host reimage |
[production] |
| 17:41 |
<andrewbogott> |
systemctl restart mariadb@s3.service mariadb@x3.service on clouddb1022; running close to 100% RAM again T438200 |
[admin] |
| 17:40 |
<jclark@cumin1004> |
END (PASS) - Cookbook sre.hosts.reimage (exit_code=0) for host sessionstore1005.eqiad.wmnet with OS bookworm |
[production] |
| 17:30 |
<jclark@cumin1004> |
END (FAIL) - Cookbook sre.hosts.provision (exit_code=99) for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 17:29 |
<jclark@cumin1004> |
END (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2:00:00 on sessionstore1005.eqiad.wmnet with reason: host reimage |
[production] |
| 17:26 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.downtime for 2:00:00 on sessionstore1005.eqiad.wmnet with reason: host reimage |
[production] |
| 17:12 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.provision for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 17:00 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.reimage for host sessionstore1005.eqiad.wmnet with OS bookworm |
[production] |
| 16:56 |
<ebernhardson@deploy1003> |
helmfile [dse-k8s-eqiad] DONE helmfile.d/dse-k8s-services/opensearch-semantic-search-test: apply |
[production] |
| 16:56 |
<ebernhardson@deploy1003> |
helmfile [dse-k8s-eqiad] START helmfile.d/dse-k8s-services/opensearch-semantic-search-test: apply |
[production] |
| 16:54 |
<cdobbins@cumin1004> |
START - Cookbook sre.hosts.reimage for host ncredir5003.eqsin.wmnet with OS trixie |
[production] |
| 16:46 |
<cdobbins@cumin1004> |
conftool action : set/pooled=yes; selector: name=ncredir6002.* |
[production] |
| 16:44 |
<jclark@cumin1004> |
END (PASS) - Cookbook sre.hosts.provision (exit_code=0) for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:44 |
<tappof@cumin1004> |
DONE (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 5 days, 0:00:00 on kafka-logging1003.eqiad.wmnet with reason: migrating to kafka-logging1006 |
[production] |
| 16:36 |
<cdobbins@cumin1004> |
END (PASS) - Cookbook sre.hosts.reimage (exit_code=0) for host ncredir6002.drmrs.wmnet with OS trixie |
[production] |
| 16:32 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.provision for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:27 |
<jclark@cumin1004> |
END (FAIL) - Cookbook sre.hosts.provision (exit_code=99) for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:27 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.provision for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:23 |
<jclark@cumin1004> |
END (FAIL) - Cookbook sre.hosts.provision (exit_code=99) for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:22 |
<jclark@cumin1004> |
START - Cookbook sre.hosts.provision for host sessionstore1005.mgmt.eqiad.wmnet with chassis set policy GRACEFUL_RESTART and with Dell SCP reboot policy GRACEFUL |
[production] |
| 16:15 |
<cmooney@cumin1004> |
END (PASS) - Cookbook sre.dns.netbox (exit_code=0) |
[production] |
| 16:15 |
<cmooney@cumin1004> |
END (PASS) - Cookbook sre.puppet.sync-netbox-hiera (exit_code=0) generate netbox hiera data: "Triggered by cookbooks.sre.dns.netbox: add entries for new eqiad links - cmooney@cumin1004" |
[production] |
| 16:15 |
<cmooney@cumin1004> |
START - Cookbook sre.puppet.sync-netbox-hiera generate netbox hiera data: "Triggered by cookbooks.sre.dns.netbox: add entries for new eqiad links - cmooney@cumin1004" |
[production] |
| 16:13 |
<cdobbins@cumin1004> |
END (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2:00:00 on ncredir6002.drmrs.wmnet with reason: host reimage |
[production] |
| 16:10 |
<cmooney@cumin1004> |
START - Cookbook sre.dns.netbox |
[production] |
| 16:09 |
<cdobbins@cumin1004> |
START - Cookbook sre.hosts.downtime for 2:00:00 on ncredir6002.drmrs.wmnet with reason: host reimage |
[production] |
| 16:01 |
<cklimas@deploy1003> |
helmfile [codfw] DONE helmfile.d/services/wikifeeds: apply |
[production] |
| 16:00 |
<cklimas@deploy1003> |
helmfile [codfw] START helmfile.d/services/wikifeeds: apply |
[production] |
| 16:00 |
<cklimas@deploy1003> |
helmfile [eqiad] DONE helmfile.d/services/wikifeeds: apply |
[production] |
| 16:00 |
<cklimas@deploy1003> |
helmfile [eqiad] START helmfile.d/services/wikifeeds: apply |
[production] |
| 16:00 |
<elukey@cumin1004> |
END (PASS) - Cookbook sre.hosts.reimage (exit_code=0) for host registry2004.codfw.wmnet with OS trixie |
[production] |
| 15:55 |
<cklimas@deploy1003> |
helmfile [staging] DONE helmfile.d/services/wikifeeds: apply |
[production] |
| 15:54 |
<cklimas@deploy1003> |
helmfile [staging] START helmfile.d/services/wikifeeds: apply |
[production] |
| 15:54 |
<hashar> |
gerrit: on mediawiki/extensions/QuickSurveys deleted branch `master-backup` which was pointing at 1ad54717c059bcd49093d902eab2c098b4efe91a (which is in `master`) |
[releng] |
| 15:49 |
<fceratto@deploy1003> |
helmfile [aux-k8s-eqiad] 'sync' command on namespace 'zarcillo' for release 'main' . |
[production] |
| 15:44 |
<jdlrobson@deploy1003> |
Finished scap sync-world: Backport for [[gerrit:1343579|Fixes: '.action_context' should be string (T437122)]] (duration: 12m 40s) |
[production] |
| 15:42 |
<elukey@cumin1004> |
END (PASS) - Cookbook sre.hosts.downtime (exit_code=0) for 2:00:00 on registry2004.codfw.wmnet with reason: host reimage |
[production] |
| 15:41 |
<dcaro@cloudcumin1001> |
END (PASS) - Cookbook wmcs.vps.create_project (exit_code=0) for trove-only project bier-db in eqiad1 (T438363) |
[bier-db] |
| 15:39 |
<jdlrobson@deploy1003> |
jdlrobson: Continuing with deployment |
[production] |
| 15:39 |
<cdobbins@cumin1004> |
START - Cookbook sre.hosts.reimage for host ncredir6002.drmrs.wmnet with OS trixie |
[production] |
| 15:38 |
<elukey@cumin1004> |
START - Cookbook sre.hosts.downtime for 2:00:00 on registry2004.codfw.wmnet with reason: host reimage |
[production] |