diff --git a/3.5.sh b/3.5.sh index 0c2fe34..83a8065 100755 --- a/3.5.sh +++ b/3.5.sh @@ -1,53 +1,47 @@ #!/bin/bash # OpenStack 计算节点(CentOS 7.9 / Rocky)基础环境配置检测 -# 参考 3.4.sh:按手册章节汇总,异常提示对应步骤。 # 用法:在 compute 上以 root 执行 bash 3.5.sh。 # 可用 CHECK_CONTROLLER_IP 覆盖默认控制节点 IP(192.168.30.129)。 # 仅检查;不修改配置、启停服务、执行重置脚本或清理/重建 YUM 缓存。 -# 退出码:0=全部通过,1=有失败,2=无失败但存在待人工核查项。 -# VMware 设置、远端状态及命令执行历史不能由本机完整验证,正常时仍可能返回 2。 +# 退出码:0=本次检查项全部通过,1=有失败。 +# 仅检查本机可验证的当前状态,不评价 VMware 设置、远端操作或命令历史。 export LANG=en_US.UTF-8 export LC_ALL=en_US.UTF-8 set -o pipefail +# 学生可见输出:每组一行,失败时列出原因;仅统计实际检查结果。 RED='\033[0;31m' GREEN='\033[0;32m' -YELLOW='\033[0;33m' CYAN='\033[0;36m' NC='\033[0m' PASS_CNT=0 FAIL_CNT=0 -SKIP_CNT=0 SECTION_TITLE='' SECTION_PASS=0 SECTION_FAIL=0 -SECTION_SKIP=0 SECTION_DETAILS=() -# 正常检查仅计数;每章输出一行汇总,异常才展开具体步骤。 pass() { PASS_CNT=$((PASS_CNT+1)); } -detail() { +fail() { + FAIL_CNT=$((FAIL_CNT+1)) if [ -n "$SECTION_TITLE" ]; then SECTION_DETAILS+=("$1") else - printf '%s\n' "$1" + printf '%b[✘ 失败] %s%b\n' "$RED" "$1" "$NC" fi } -fail() { FAIL_CNT=$((FAIL_CNT+1)); detail " [✘ 失败] $1"; } -skip() { SKIP_CNT=$((SKIP_CNT+1)); detail " [− 跳过] $1"; } section_summary() { [ -n "$SECTION_TITLE" ] || return 0 - local passed=$((PASS_CNT-SECTION_PASS)) failed=$((FAIL_CNT-SECTION_FAIL)) skipped=$((SKIP_CNT-SECTION_SKIP)) + local passed=$((PASS_CNT-SECTION_PASS)) failed=$((FAIL_CNT-SECTION_FAIL)) message + # 没有实际检查结果的分组不显示,也不记为通过。 if [ "$failed" -gt 0 ]; then - printf '%b[✘] %s:通过 %s,失败 %s,跳过 %s%b\n' "$RED" "$SECTION_TITLE" "$passed" "$failed" "$skipped" "$NC" - elif [ "$skipped" -gt 0 ]; then - printf '%b[−] %s:通过 %s,跳过 %s%b\n' "$YELLOW" "$SECTION_TITLE" "$passed" "$skipped" "$NC" - else - printf '%b[✔] %s:%s 项通过%b\n' "$GREEN" "$SECTION_TITLE" "$passed" "$NC" - fi - if [ "${#SECTION_DETAILS[@]}" -gt 0 ]; then - printf '%s\n' "${SECTION_DETAILS[@]}" + printf '%b[✘ 失败] %s:通过 %s,失败 %s%b\n' "$RED" "$SECTION_TITLE" "$passed" "$failed" "$NC" + for message in "${SECTION_DETAILS[@]}"; do + printf '%b · %s%b\n' "$RED" "$message" "$NC" + done + elif [ "$passed" -gt 0 ]; then + printf '%b[✔ 通过] %s(%s 项)%b\n' "$GREEN" "$SECTION_TITLE" "$passed" "$NC" fi } info() { @@ -55,13 +49,24 @@ info() { SECTION_TITLE=$1 SECTION_PASS=$PASS_CNT SECTION_FAIL=$FAIL_CNT - SECTION_SKIP=$SKIP_CNT SECTION_DETAILS=() } +finish() { + section_summary + printf '\n检查完毕:通过 %b%s%b 项,失败 %b%s%b 项。\n' \ + "$GREEN" "$PASS_CNT" "$NC" "$RED" "$FAIL_CNT" "$NC" + if [ "$FAIL_CNT" -gt 0 ]; then + printf '%b请按失败提示修正后重试。%b\n' "$RED" "$NC" + exit 1 + fi + printf '%b本次检查项全部通过。%b\n' "$GREEN" "$NC" + exit 0 +} + +printf '%bOpenStack 计算节点基础环境自测%b\n' "$CYAN" "$NC" -printf '%bOpenStack 计算节点基础环境检测%b\n' "$CYAN" "$NC" if [ "$EUID" -ne 0 ]; then - fail '[准备工作] 请在计算节点 compute 上使用 root 权限执行此脚本。' + fail '请在计算节点 compute 上使用 root 权限执行此脚本。' exit 1 fi @@ -72,13 +77,11 @@ REPO_DIR=/etc/yum.repos.d info '准备工作:检查工具与虚拟化' for cmd in awk sort grep tr hostname ip getent ping rpm systemctl getenforce chronyc timeout curl yum; do if ! command -v "$cmd" >/dev/null 2>&1; then - fail "缺少命令 ${cmd},请先补齐 CentOS 7 实验环境及 chrony 等检查工具。" + fail "缺少命令 ${cmd},请安装对应软件包" fi done if [ "$FAIL_CNT" -gt 0 ]; then - section_summary - printf '%b检查工具不齐,暂不能可靠检查;请补齐后重新运行。%b\n' "$RED" "$NC" - exit 1 + finish fi valid_ipv4() { @@ -90,34 +93,29 @@ valid_ipv4() { } if ! valid_ipv4 "$CONTROLLER_IP"; then fail 'CHECK_CONTROLLER_IP 必须为有效的 IPv4 地址。' - section_summary - exit 1 + finish fi if grep -Eq '(^|[[:space:]])(vmx|svm)([[:space:]]|$)' /proc/cpuinfo 2>/dev/null; then pass 'compute 可见 CPU 硬件虚拟化扩展' else - fail '[准备工作第 2、3 步] compute 未检测到 vmx/svm,请检查克隆后的虚拟机是否启用嵌套虚拟化。' + fail '未检测到 vmx/svm,请启用嵌套虚拟化' fi -skip '[准备工作第 1~4 步] 请在 VMware 中核对 VMnet1/VMnet8、controller 虚拟化开关、克隆来源和两台虚拟机的电源状态;本机无法完整确认。' -info '一、重置控制节点环境' -skip '[第 1 步] 无法从 compute 证明 controller 已执行 env3-4-local.sh;请在 controller 核查重置结果,并运行 3.4.sh 验证服务。' - -info '二、基础安全设置' +info '基础安全设置' state=$(systemctl is-active firewalld 2>/dev/null) case "$state" in inactive) pass 'firewalld 已停止';; - *) fail "[第 1 步] firewalld 状态为 ${state:-查询失败},应为 inactive;请检查 systemctl status firewalld。";; + *) fail "firewalld 状态为 ${state:-查询失败},应为 inactive";; esac state=$(systemctl is-enabled firewalld 2>/dev/null) case "$state" in disabled|masked) pass 'firewalld 已禁止开机启动';; - *) fail "[第 1 步] firewalld 自启状态为 ${state:-查询失败},应为 disabled 或 masked。";; + *) fail "firewalld 自启状态为 ${state:-查询失败},应为 disabled 或 masked。";; esac case "$(getenforce 2>/dev/null)" in Disabled|Permissive) pass 'SELinux 当前未强制执行';; - *) fail '[第 2 步] SELinux 仍为 Enforcing 或状态查询失败,请检查 setenforce 0 的执行结果。';; + *) fail 'SELinux 仍为 Enforcing 或状态查询失败';; esac # 读取最后一个有效赋值,忽略注释;不 source 系统配置。 selinux=$(awk ' @@ -131,19 +129,19 @@ selinux=$(awk ' if [ "$selinux" = disabled ]; then pass 'SELinux 永久配置为 disabled' else - fail '[第 3 步] /etc/selinux/config 的有效 SELINUX 值应为 disabled。' + fail '/etc/selinux/config 的有效 SELINUX 值应为 disabled。' fi -info '三、配置主机名解析' +info '配置主机名解析' if [ "$(hostname 2>/dev/null)" = compute ]; then pass '当前主机名为 compute' else - fail '[计算节点第 1 步] 当前主机名应为 compute,请检查 hostnamectl set-hostname compute。' + fail '当前主机名应为 compute,请检查 hostnamectl set-hostname compute。' fi if [ "$(tr -d '[:space:]' 2>/dev/null < /etc/hostname)" = compute ]; then pass '持久主机名为 compute' else - fail '[计算节点第 1 步] /etc/hostname 应保存 compute,避免重启后主机名恢复。' + fail '/etc/hostname 应保存 compute,避免重启后主机名恢复。' fi LOCAL_IPS=$(ip -o -4 addr show scope global 2>/dev/null | awk '{split($4,a,"/"); print a[1]}') hosts_addresses() { @@ -153,13 +151,13 @@ COMPUTE_IP=$(hosts_addresses compute) if [ "$(hosts_addresses controller)" = "$CONTROLLER_IP" ]; then pass '/etc/hosts 中 controller 映射正确且无冲突' else - fail "[两节点 hosts 配置] 本机 /etc/hosts 应将 controller 唯一映射到 ${CONTROLLER_IP},不能保留旧地址或回环地址。" + fail "/etc/hosts 中 controller 应唯一映射到 ${CONTROLLER_IP}" fi if valid_ipv4 "$COMPUTE_IP" && [ "$COMPUTE_IP" != "$CONTROLLER_IP" ] && \ printf '%s\n' "$LOCAL_IPS" | grep -Fxq "$COMPUTE_IP"; then pass '/etc/hosts 中 compute 映射到本机 IPv4 地址' else - fail '[计算节点 hosts 配置] compute 应唯一映射到本机实际管理 IPv4 地址,不能使用 XXX、controller 地址或回环地址。' + fail '/etc/hosts 中 compute 应唯一映射到本机管理 IPv4 地址' fi for name in controller compute; do expected=$CONTROLLER_IP @@ -168,21 +166,20 @@ for name in controller compute; do if [ -n "$expected" ] && [ "$resolved" = "$expected" ]; then pass "$name 的系统 IPv4 解析与 hosts 一致" else - fail "[连通性验证] $name 的实际系统解析与 hosts 不一致或解析失败,请检查 /etc/hosts 和 /etc/nsswitch.conf。" + fail "$name 解析异常,请检查 /etc/hosts 和 /etc/nsswitch.conf" fi if timeout 8 ping -4 -c 2 -W 2 "$name" >/dev/null 2>&1; then pass "可通过主机名 ping 通 $name" else - fail "[连通性验证] 无法 ping 通 ${name},请检查地址、网卡、路由及 ICMP 是否被过滤。" + fail "ping ${name} 失败,请检查地址和网络" fi done -skip '[控制节点 hosts 配置及反向连通性] 请在 controller 核对 compute 的实际 IP,并执行 ping compute;本机 ping 成功不能证明反向解析正确。' -info '四、配置 NTP 时间同步' +info '配置 NTP 时间同步' if rpm -q chrony >/dev/null 2>&1; then pass 'chrony 已安装' else - fail '[第 1 步] chrony 软件包未安装。' + fail 'chrony 软件包未安装。' fi if awk ' { sub(/[#!;].*/, "") } @@ -193,7 +190,7 @@ if awk ' ' /etc/chrony.conf 2>/dev/null; then pass 'chrony 配置包含唯一的 server controller iburst' else - fail '[第 2 步] /etc/chrony.conf 应有唯一的有效 server controller iburst 配置。' + fail '/etc/chrony.conf 应有唯一的有效 server controller iburst 配置。' fi if awk ' { sub(/[#!;].*/, "") } @@ -202,35 +199,35 @@ if awk ' ' /etc/chrony.conf 2>/dev/null; then pass 'chrony 主配置中没有其他启用的时间源' else - fail '[第 2 步] 请注释默认外网 server/pool 及其他时间源,仅保留 controller。' + fail '请注释默认外网 server/pool 及其他时间源,仅保留 controller。' fi -if systemctl is-active --quiet chronyd; then +if systemctl is-active --quiet chronyd 2>/dev/null; then pass 'chronyd 正在运行' else - fail '[第 3 步] chronyd 未运行。' + fail 'chronyd 未运行。' fi if [ "$(systemctl is-enabled chronyd 2>/dev/null)" = enabled ]; then pass 'chronyd 已持久设置开机自启' else - fail '[第 3 步] chronyd 未设置持久的开机自启。' + fail 'chronyd 未设置持久的开机自启。' fi # 使用 -n 比较实际 IP,兼容显示为 controller、FQDN 或 IP 的情况。 if sources=$(timeout 10 chronyc -n sources 2>/dev/null); then if printf '%s\n' "$sources" | awk -v ip="$CONTROLLER_IP" '$1=="^*" && $2==ip { found=1 } END { exit !found }'; then pass 'chronyd 当前已选中 controller 为同步源' else - fail "[第 4 步] 未看到 ^* ${CONTROLLER_IP};请运行 chronyc -n sources,检查 controller 的 NTP 服务、UDP 123 及网络;刚启动可稍后重试。" + fail "尚未同步到 ${CONTROLLER_IP},请检查 chronyc -n sources" fi if printf '%s\n' "$sources" | awk -v ip="$CONTROLLER_IP" '$1 ~ /^[\^=]/ && $2!=ip { bad=1 } END { exit bad }'; then pass '运行中的 chronyd 未加载其他服务器或对等时间源' else - fail '[第 2、3 步] chronyd 仍加载其他时间源,请检查 include/sourcedir 配置及修改后是否重启服务。' + fail 'chronyd 仍加载其他时间源,请检查配置并重启服务' fi else - fail '[第 4 步] chronyc 查询失败或超时,请检查 chronyd。' + fail 'chronyc 查询失败或超时,请检查 chronyd。' fi -info '五、配置软件源与 YUM 缓存' +info '配置软件源与 YUM 缓存' # 按指定文件、节读取键值,避免注释及其他节同名键误判。 repo_value() { awk -v section="$2" -v key="$3" ' @@ -275,14 +272,14 @@ if [ "${#REPO_FILES[@]}" -gt 0 ]; then if [ -z "$extra_repos" ]; then pass '未遗留其他启用的软件源' else - fail "[软件源配置] 仍有额外启用的仓库:$(printf '%s' "$extra_repos" | tr '\n' ' ');请按手册备份并替换旧源。" + fail "仍有额外启用的仓库:$(printf '%s' "$extra_repos" | tr '\n' ' ')" fi fi while IFS='|' read -r file id path; do url="http://192.168.192.205:3080/$path/" config_ok=1 if [ ! -r "$REPO_DIR/$file" ]; then - fail "[软件源配置] 缺少或无法读取 $REPO_DIR/${file}。" + fail "缺少或无法读取 $REPO_DIR/${file}。" config_ok=0 else for key in baseurl enabled gpgcheck; do @@ -292,13 +289,13 @@ while IFS='|' read -r file id path; do if [ "$value" = "$expected" ]; then pass "$id 的 $key 正确" else - fail "[软件源配置] $file 的 [$id] $key 应为 ${expected}。" + fail "$file 的 [$id] $key 应为 ${expected}。" config_ok=0 fi done for key in mirrorlist metalink; do if [ -n "$(repo_value "$REPO_DIR/$file" "$id" "$key")" ]; then - fail "[软件源配置] [$id] 存在额外的 ${key},请按手册仅使用本地 baseurl。" + fail "[$id] 存在额外的 ${key},请按手册仅使用本地 baseurl。" config_ok=0 fi done @@ -307,7 +304,7 @@ while IFS='|' read -r file id path; do if printf '%s\n' "$excludes" | awk '{for(i=1;i<=NF;i++){if($i=="sip")s=1;if($i=="PyQt4")p=1}} END{exit !(s&&p)}'; then pass 'Rocky 源已排除 sip 和 PyQt4' else - fail '[软件源配置] [local-openstack-rocky] 应设置 exclude=sip,PyQt4。' + fail '[local-openstack-rocky] 应设置 exclude=sip,PyQt4。' config_ok=0 fi fi @@ -324,7 +321,7 @@ while IFS='|' read -r file id path; do if [ "$count" = 1 ]; then pass "$id 定义唯一" else - fail "[软件源配置] $id 在 .repo 文件中出现 $count 次,应只定义一次。" + fail "$id 在 .repo 文件中出现 $count 次,应只定义一次。" config_ok=0 fi # GET 元数据而非仅访问目录,避免把空目录/404 当成仓库可用。 @@ -333,7 +330,7 @@ while IFS='|' read -r file id path; do printf '%s\n' "$metadata" | grep -Fq ''; then pass "$id 的仓库索引可访问" else - fail "[镜像连通性] 无法获取 $id 的有效 repomd.xml,请检查 VMnet8、镜像服务器和路径 ${url}。" + fail "$id 仓库索引访问失败,请检查 ${url}" fi if [ "$config_ok" -eq 1 ]; then # -C 使用已有缓存;禁用插件,不执行 clean/makecache,也不下载软件包。 @@ -341,10 +338,8 @@ while IFS='|' read -r file id path; do --setopt="$id.skip_if_unavailable=0" list available >/dev/null 2>&1; then pass "$id 的现有 YUM 缓存可读取" else - fail "[清理并重建 YUM 缓存] $id 的缓存查询失败或超时;配置正确且网络连通后,请执行 yum clean all 和 yum makecache。" + fail "$id 缓存查询失败,请检查网络并重建 YUM 缓存" fi - else - skip "[YUM 缓存] $id 配置未通过,修复后再验证缓存。" fi done <<'REPOS' CentOS-Base.repo|local-base|vault-base @@ -354,16 +349,4 @@ CentOS-QEMU-EV.repo|local-qemu-ev|vault-centos-qemu-ev CentOS-OpenStack-rocky.repo|local-openstack-rocky|vault-centos-openstack-rocky REPOS -section_summary -printf '\n合计:通过 %b%s%b,失败 %b%s%b,跳过 %b%s%b\n' \ - "$GREEN" "$PASS_CNT" "$NC" "$RED" "$FAIL_CNT" "$NC" "$YELLOW" "$SKIP_CNT" "$NC" -printf '说明:检查当前配置及缓存可用性,不能证明曾执行过重置、克隆或 yum clean all/makecache 命令。\n' -if [ "$FAIL_CNT" -gt 0 ]; then - printf '%b请按异常项的手册步骤修正后重试。%b\n' "$RED" "$NC" - exit 1 -elif [ "$SKIP_CNT" -gt 0 ]; then - printf '%b本机已检查项通过;仍有待人工核查内容,请按跳过提示确认。%b\n' "$YELLOW" "$NC" - exit 2 -else - exit 0 -fi +finish