高可用基础
高可用(High Availability,HA)确保服务在单点故障时仍可持续运行。本文介绍三种常用的高可用方案:keepalived(VIP 漂移)、HAProxy(负载均衡)和 Pacemaker/Corosync(集群资源管理)。
keepalived(虚拟 IP 漂移)
keepalived 基于 VRRP 协议实现虚拟 IP 地址的自动漂移,当主节点故障时,备用节点自动接管 VIP。
安装
# 在所有节点安装
sudo apt update
sudo apt install keepalived -y架构示例
节点 1 (MASTER): 192.168.1.11
节点 2 (BACKUP): 192.168.1.12
虚拟 IP (VIP): 192.168.1.100主节点配置
sudo tee /etc/keepalived/keepalived.conf << 'EOF'
global_defs {
router_id LVS_NODE1
script_user root
enable_script_security
}
# 健康检查脚本
vrrp_script check_nginx {
script "/usr/bin/curl -s -o /dev/null -w '%{http_code}' http://127.0.0.1 | grep -q 200"
interval 5
weight -20
fall 3
rise 2
}
vrrp_instance VI_1 {
state MASTER
interface eth0
virtual_router_id 51
priority 100
advert_int 1
authentication {
auth_type PASS
auth_pass MyS3cretP@ss
}
virtual_ipaddress {
192.168.1.100/24 dev eth0
}
track_script {
check_nginx
}
# 通知脚本(可选)
notify_master "/etc/keepalived/notify.sh master"
notify_backup "/etc/keepalived/notify.sh backup"
notify_fault "/etc/keepalived/notify.sh fault"
}
EOF备用节点配置
sudo tee /etc/keepalived/keepalived.conf << 'EOF'
global_defs {
router_id LVS_NODE2
script_user root
enable_script_security
}
vrrp_script check_nginx {
script "/usr/bin/curl -s -o /dev/null -w '%{http_code}' http://127.0.0.1 | grep -q 200"
interval 5
weight -20
fall 3
rise 2
}
vrrp_instance VI_1 {
state BACKUP
interface eth0
virtual_router_id 51
priority 90
advert_int 1
authentication {
auth_type PASS
auth_pass MyS3cretP@ss
}
virtual_ipaddress {
192.168.1.100/24 dev eth0
}
track_script {
check_nginx
}
}
EOF通知脚本
sudo tee /etc/keepalived/notify.sh << 'SCRIPT'
#!/bin/bash
STATE=$1
DATETIME=$(date '+%Y-%m-%d %H:%M:%S')
echo "${DATETIME} - 状态变更为: ${STATE}" >> /var/log/keepalived-notify.log
case $STATE in
master)
echo "${DATETIME} - 当前节点成为 MASTER" >> /var/log/keepalived-notify.log
# 可在此发送告警通知
;;
backup)
echo "${DATETIME} - 当前节点成为 BACKUP" >> /var/log/keepalived-notify.log
;;
fault)
echo "${DATETIME} - 当前节点进入 FAULT 状态" >> /var/log/keepalived-notify.log
;;
esac
SCRIPT
sudo chmod +x /etc/keepalived/notify.sh启动服务
# 在所有节点启动
sudo systemctl enable --now keepalived
# 查看状态
sudo systemctl status keepalived
# 查看 VIP
ip addr show eth0 | grep 192.168.1.100
# 查看日志
sudo journalctl -u keepalived -f
# 测试故障转移
# 在主节点停止 keepalived
sudo systemctl stop keepalived
# 观察备用节点是否接管 VIPHAProxy(负载均衡)
HAProxy 是高性能的 TCP/HTTP 负载均衡器,支持多种调度算法和健康检查。
安装
sudo apt update
sudo apt install haproxy -y
# 查看版本
haproxy -v基本配置
sudo tee /etc/haproxy/haproxy.cfg << 'EOF'
global
log /dev/log local0
log /dev/log local1 notice
chroot /var/lib/haproxy
stats socket /run/haproxy/admin.sock mode 660 level admin
stats timeout 30s
user haproxy
group haproxy
daemon
# 性能调优
maxconn 50000
tune.ssl.default-dh-param 2048
defaults
log global
mode http
option httplog
option dontlognull
option forwardfor
option http-server-close
timeout connect 5000
timeout client 50000
timeout server 50000
errorfile 400 /etc/haproxy/errors/400.http
errorfile 403 /etc/haproxy/errors/403.http
errorfile 408 /etc/haproxy/errors/408.http
errorfile 500 /etc/haproxy/errors/500.http
errorfile 502 /etc/haproxy/errors/502.http
errorfile 503 /etc/haproxy/errors/503.http
errorfile 504 /etc/haproxy/errors/504.http
# 统计页面
listen stats
bind *:8404
stats enable
stats uri /stats
stats refresh 10s
stats admin if LOCALHOST
stats auth admin:haproxy_pass
# HTTP 前端
frontend http_front
bind *:80
# HTTPS 重定向
# redirect scheme https code 301 if !{ ssl_fc }
default_backend http_back
# HTTPS 前端
frontend https_front
bind *:443 ssl crt /etc/haproxy/certs/example.com.pem
http-request set-header X-Forwarded-Proto https
# 基于域名路由
acl host_app1 hdr(host) -i app1.example.com
acl host_app2 hdr(host) -i app2.example.com
use_backend app1_back if host_app1
use_backend app2_back if host_app2
default_backend http_back
# 后端服务器组
backend http_back
balance roundrobin
option httpchk GET /health
http-check expect status 200
server web1 192.168.1.21:80 check inter 5s fall 3 rise 2
server web2 192.168.1.22:80 check inter 5s fall 3 rise 2
server web3 192.168.1.23:80 check inter 5s fall 3 rise 2 backup
backend app1_back
balance leastconn
option httpchk GET /health
server app1a 192.168.1.31:8080 check
server app1b 192.168.1.32:8080 check
backend app2_back
balance source
server app2a 192.168.1.41:8080 check
server app2b 192.168.1.42:8080 check
EOFTCP 负载均衡(数据库等)
# 在 haproxy.cfg 中添加
cat << 'EOF' | sudo tee -a /etc/haproxy/haproxy.cfg
# MySQL 负载均衡
listen mysql_cluster
bind *:3306
mode tcp
option mysql-check user haproxy
balance roundrobin
server mysql1 192.168.1.51:3306 check inter 5s
server mysql2 192.168.1.52:3306 check inter 5s backup
# Redis Sentinel
listen redis
bind *:6379
mode tcp
option tcp-check
balance first
server redis1 192.168.1.61:6379 check inter 3s
server redis2 192.168.1.62:6379 check inter 3s
EOF负载均衡算法
| 算法 | 说明 | 适用场景 |
|---|---|---|
roundrobin | 轮询 | 默认,服务器性能相近 |
leastconn | 最少连接 | 长连接、处理时间不均 |
source | 源 IP 哈希 | 需要会话保持 |
uri | URI 哈希 | 缓存服务器 |
first | 填满第一个再用第二个 | 节约资源 |
启动与验证
# 检查配置
sudo haproxy -c -f /etc/haproxy/haproxy.cfg
# 启动服务
sudo systemctl enable --now haproxy
# 查看状态页面
curl http://localhost:8404/stats
# 防火墙
sudo ufw allow 80/tcp
sudo ufw allow 443/tcp
sudo ufw allow 8404/tcpkeepalived + HAProxy 组合
最常见的高可用方案是 keepalived + HAProxy:两台 HAProxy 互为主备,通过 keepalived 管理 VIP。
# keepalived 配置中检查 HAProxy 状态
sudo tee /etc/keepalived/keepalived.conf << 'EOF'
global_defs {
router_id HAPROXY_NODE1
script_user root
enable_script_security
}
vrrp_script check_haproxy {
script "/usr/bin/killall -0 haproxy"
interval 2
weight -30
fall 3
rise 2
}
vrrp_instance VI_1 {
state MASTER
interface eth0
virtual_router_id 51
priority 100
advert_int 1
authentication {
auth_type PASS
auth_pass HAProxy_HA
}
virtual_ipaddress {
192.168.1.100/24
}
track_script {
check_haproxy
}
}
EOFPacemaker / Corosync(集群资源管理)
Pacemaker 和 Corosync 提供企业级的集群资源管理,支持复杂的资源依赖和故障策略。
安装
# 在所有节点安装
sudo apt update
sudo apt install pacemaker corosync pcs resource-agents -y
# 设置 hacluster 用户密码(所有节点相同)
sudo passwd hacluster
# 启动 pcs 管理服务
sudo systemctl enable --now pcsd配置集群
# 在一个节点上执行认证(所有节点)
sudo pcs host auth node1 node2 -u hacluster -p yourpassword
# 创建集群
sudo pcs cluster setup ha-cluster node1 node2
# 启动集群
sudo pcs cluster start --all
sudo pcs cluster enable --all
# 查看集群状态
sudo pcs cluster status
sudo pcs status
# 对于两节点集群,禁用 quorum 策略
sudo pcs property set no-quorum-policy=ignore
# 禁用 STONITH(测试环境,生产环境应配置 fencing 设备)
sudo pcs property set stonith-enabled=false配置集群资源
# 添加虚拟 IP 资源
sudo pcs resource create cluster_vip ocf:heartbeat:IPaddr2 \
ip=192.168.1.100 \
cidr_netmask=24 \
nic=eth0 \
op monitor interval=10s
# 添加 Nginx 资源
sudo pcs resource create web_server ocf:heartbeat:nginx \
configfile=/etc/nginx/nginx.conf \
op monitor interval=10s timeout=30s \
op start timeout=60s \
op stop timeout=60s
# 确保 VIP 和 Nginx 在同一节点
sudo pcs constraint colocation add web_server with cluster_vip INFINITY
# 确保 VIP 先于 Nginx 启动
sudo pcs constraint order cluster_vip then web_server
# 创建资源组(更简单的方式)
sudo pcs resource group add web_group cluster_vip web_server
# 查看资源状态
sudo pcs resource status
sudo pcs constraint show资源管理
# 手动迁移资源
sudo pcs resource move web_group node2
# 清除迁移约束(迁移后必须执行)
sudo pcs resource clear web_group
# 启用/禁用资源
sudo pcs resource disable web_server
sudo pcs resource enable web_server
# 清除资源错误状态
sudo pcs resource cleanup web_server
# 维护模式
sudo pcs node standby node1 # 将节点置为待机
sudo pcs node unstandby node1 # 恢复节点
sudo pcs property set maintenance-mode=true # 全局维护模式
sudo pcs property set maintenance-mode=false集群监控
# 实时监控集群状态
sudo crm_mon -1
# 查看集群配置
sudo pcs config show
# 查看集群日志
sudo journalctl -u pacemaker -f
sudo journalctl -u corosync -f
# Corosync 成员状态
sudo corosync-cmapctl | grep members方案对比
| 特性 | keepalived | HAProxy | Pacemaker/Corosync |
|---|---|---|---|
| 主要功能 | VIP 漂移 | 负载均衡 | 集群资源管理 |
| 配置复杂度 | 简单 | 中等 | 较复杂 |
| 资源类型 | 仅 VIP | HTTP/TCP 均衡 | 任意(VIP、服务、文件系统等) |
| 健康检查 | 基于脚本 | 丰富的协议检查 | 基于资源代理 |
| 适用规模 | 2 节点 | 大规模 | 2-32 节点 |
| 典型用途 | 主备切换 | 流量分发 | 企业级 HA |
选择建议
- 简单主备:keepalived 即可满足
- Web 负载均衡:HAProxy(可配合 keepalived 实现 HA)
- 复杂集群:Pacemaker/Corosync(多种资源、复杂依赖关系)
- 最常见组合:keepalived + HAProxy,简单高效
Last updated on