构建一个高效的CI/CD流水线是现代软件交付的核心。本文将分享经过实战验证的最佳实践,帮助你优化从代码提交到生产部署的每一个环节。
一、流水线设计原则#
1.1 FAST原则#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
| # F - Fast(快速)
# 每个阶段都应该尽可能快速完成
# ❌ 慢速流水线
stages:
- build # 10分钟
- test # 20分钟
- deploy # 5分钟
# 总计: 35分钟
# ✅ 快速流水线
stages:
- lint # 30秒 - 快速失败
- unit # 2分钟 - 核心测试
- build # 3分钟 - 仅在通过后构建
- e2e # 5分钟 - 并行运行
- deploy # 1分钟 - 自动化部署
# 总计: 11分钟(理想情况)
# A - Automated(自动化)
# 减少人工干预,提高一致性
# ❌ 手动触发多阶段
manual-build:
stage: build
when: manual
script:
- ./build.sh
manual-test:
stage: test
when: manual
script:
- ./test.sh
# ✅ 全自动化
auto-build:
stage: build
script:
- ./build.sh
auto-test:
stage: test
needs: [auto-build] # 自动依赖
script:
- ./test.sh
# S - Secure(安全)
# 在每个环节嵌入安全检查
security-scan:
stage: test
parallel:
matrix:
scan:
- dependency-check
- sast-scan
- container-scan
script:
- npm run $scan
# T - Traceable(可追溯)
# 完整的审计日志和版本追踪
traceable-deploy:
stage: deploy
script:
- |
echo "部署信息:" > deploy-info.txt
echo "Commit: ${CI_COMMIT_SHA}" >> deploy-info.txt
echo "Branch: ${CI_COMMIT_REF_NAME}" >> deploy-info.txt
echo "Author: ${GITLAB_USER_NAME}" >> deploy-info.txt
echo "Pipeline: ${CI_PIPELINE_ID}" >> deploy-info.txt
- ./deploy.sh
artifacts:
paths:
- deploy-info.txt
expire_in: 90 days
|
1.2 阶段划分策略#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
| # 经典的5阶段流水线
stages:
.pre: # 准备阶段
- 环境检查
- 依赖下载
- 配置验证
.quality: # 质量阶段
- 代码规范检查
- 静态分析
- 安全扫描
.test: # 测试阶段
- 单元测试
- 集成测试
- 端到端测试
.build: # 构建阶段
- 编译打包
- Docker镜像
- 版本标记
.deploy: # 部署阶段
- 环境部署
- 健康检查
- 回滚准备
# 实战示例
lint:
stage: .pre
script:
- npm run lint
- npm run type-check
rules:
- if: $CI_PIPELINE_SOURCE == "merge_request_event"
unit-test:
stage: .test
coverage: '/All files[^|]*\|[^|]*\s+([\d\.]+)/'
script:
- npm run test:unit -- --coverage
artifacts:
reports:
coverage_report:
coverage_format: cobertura
path: coverage/cobertura-coverage.xml
expire_in: 7 days
integration-test:
stage: .test
services:
- postgres:15
- redis:alpine
variables:
POSTGRES_DB: test_db
REDIS_URL: redis://redis:6379
script:
- npm run test:integration
artifacts:
when: always
reports:
junit: test-results/integration.xml
build:
stage: .build
dependencies:
- unit-test
- integration-test
script:
- npm run build
- docker build -t $CI_REGISTRY_IMAGE:$CI_COMMIT_SHA .
- docker push $CI_REGISTRY_IMAGE:$CI_COMMIT_SHA
only:
- main
- /^release\/.*$/
deploy-staging:
stage: .deploy
environment:
name: staging
url: https://staging.example.com
script:
- kubectl set image deployment/app app=$CI_REGISTRY_IMAGE:$CI_COMMIT_SHA
- kubectl rollout status deployment/app
needs:
- build
|
二、性能优化#
2.1 缓存策略#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
| # 多层缓存架构
cache-config:
stage: .pre
cache:
# 策略1: 依赖缓存
- key:
files:
- package-lock.json
paths:
- node_modules/
policy: pull-push
# 策略2: 构建缓存
- key: ${CI_COMMIT_REF_SLUG}-build
paths:
- .next/cache/
- dist/.vite/
policy: pull-push
# 策略3: Docker层缓存
- key: ${CI_COMMIT_REF_SLUG}-docker
paths:
- .docker/cache
policy: pull-push
# 实战:npm缓存优化
# ❌ 低效缓存
cache:
paths:
- node_modules/
# 每次都重新下载
# ✅ 智能缓存
cache:
key:
files:
- package-lock.json
paths:
- node_modules/
policy: pull-push
# 仅当package-lock.json变化时重建
# Docker构建缓存
build-image:
stage: .build
script:
- |
# 使用BuildKit缓存
DOCKER_BUILDKIT=1 docker build \
--cache-from $CI_REGISTRY_IMAGE:latest \
--cache-to $CI_REGISTRY_IMAGE:cache \
--build-arg BUILDKIT_INLINE_CACHE=1 \
-t $CI_REGISTRY_IMAGE:$CI_COMMIT_SHA \
.
|
2.2 并行执行#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
| # 策略1: 矩阵并行
test-matrix:
stage: .test
parallel:
matrix:
NODE_VERSION: [18, 20, 21]
TEST_SUITE: [unit, integration, e2e]
image: node:${NODE_VERSION}
script:
- npm run test:${TEST_SUITE}
# 策略2: 文件级并行
# 将测试文件分组并行运行
parallel-tests:
stage: .test
parallel: 5 # 分成5组
script:
- |
TOTAL_TESTS=$(find test -name "*.test.js" | wc -l)
TESTS_PER_JOB=$((TOTAL_TESTS / PARALLEL_PROCESS_COUNT))
START=$((PARALLEL_PROCESS_NUMBER * TESTS_PER_JOB))
END=$((START + TESTS_PER_JOB))
TESTS=$(find test -name "*.test.js" | sed -n "${START},${END}p")
npm test -- $TESTS
# 策略3: 阶段并行
# 独立任务完全并行
lint:
stage: quality
script: npm run lint
type-check:
stage: quality
script: npm run type-check
security-scan:
stage: quality
script: npm run security
# 以上三个任务并行执行
# 策略4: 依赖图并行
# Next.js等现代框架支持
test:
stage: .test
script:
- npm run test -- --shard=${CI_NODE_INDEX}/${CI_NODE_TOTAL}
parallel: 4
|
2.3 增量构建#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
| # 检测变化文件
detect-changes:
stage: .pre
script:
- |
# 获取变化的文件
CHANGED_FILES=$(git diff --name-only $CI_COMMIT_BEFORE $CI_COMMIT_SHA)
# 检测变化的服务
echo "$CHANGED_FILES" | grep -q "service-a/" && echo "service-a" > changed_services.txt
echo "$CHANGED_FILES" | grep -q "service-b/" && echo "service-b" >> changed_services.txt
# 保存为artifacts
artifacts:
paths:
- changed_services.txt
# 条件构建
build-service:
stage: .build
script:
- |
if grep -q "my-service" changed_services.txt; then
cd services/my-service
npm run build
else
echo "服务无变化,跳过构建"
exit 0
fi
# 使用GitLab的变化检测
conditional-job:
stage: .build
script:
- npm run build
rules:
- changes:
- src/**/* # 仅src变化时运行
- package.json
when: on_success
|
三、测试策略#
3.1 测试金字塔实施#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
| # 1. 底层:单元测试(快速、大量)
unit-test:
stage: test
image: node:20
script:
- npm run test:unit -- --coverage --maxWorkers=4
coverage: '/All files[^|]*\|[^|]*\s+([\d\.]+)/'
artifacts:
reports:
coverage_report:
coverage_format: cobertura
path: coverage/cobertura-coverage.xml
junit: test-results/unit.xml
rules:
- if: '$CI_PIPELINE_SOURCE == "merge_request_event"'
- if: '$CI_COMMIT_BRANCH == "main"'
# 2. 中层:集成测试(中等速度)
integration-test:
stage: test
services:
- name: postgres:15-alpine
alias: db
variables:
POSTGRES_DB: test
POSTGRES_USER: tester
POSTGRES_PASSWORD: secret
- name: redis:alpine
alias: redis
variables:
DATABASE_URL: postgresql://tester:secret@db:5432/test
REDIS_URL: redis://redis:6379
script:
- npm run test:integration
artifacts:
reports:
junit: test-results/integration.xml
# 3. 顶层:E2E测试(慢速、少量)
e2e-test:
stage: test
image: mcr.microsoft.com/playwright:v1.40.0
script:
- npm ci
- npx playwright install --with-deps
- npm run test:e2e
artifacts:
when: always
paths:
- playwright-report/
- test-results/
expire_in: 30 days
parallel: 3 # 分3组并行
rules:
- if: '$CI_COMMIT_BRANCH == "main"'
- if: '$CI_PIPELINE_SOURCE == "schedule"'
|
3.2 测试数据管理#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
| # 测试数据库初始化
test-db-init:
stage: .pre
services:
- postgres:15
script:
- |
# 导入测试数据
psql $DATABASE_URL < schema/test.sql
psql $DATABASE_URL < fixtures/test-data.sql
# 或使用种子数据
npm run db:seed:test
artifacts:
reports:
# 生成测试数据报告
junit: test-db-report.xml
# Mock服务
mock-external-apis:
stage: .pre
script:
- |
# 启动Mock服务器
docker run -d -p 8080:8080 mockserver/mockserver
# 配置Mock期望
curl -X PUT http://localhost:8080/expectations \
-d @test/mocks/api-expectations.json
cache:
paths:
- test/mocks/
|
四、部署策略#
4.1 多环境部署#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
| # 开发环境:自动部署
deploy-dev:
stage: deploy
environment:
name: development
url: https://dev.example.com
on_stop: stop-dev # 定义停止作业
script:
- helm upgrade --install myapp ./charts/myapp \
--namespace dev \
--set image.tag=$CI_COMMIT_SHA \
--set env=development \
--wait
only:
- develop
# 停止开发环境
stop-dev:
stage: deploy
environment:
name: development
action: stop
script:
- helm uninstall myapp --namespace dev
when: manual
only:
- develop
# 预发布环境:手动触发
deploy-staging:
stage: deploy
environment:
name: staging
url: https://staging.example.com
script:
- |
# 蓝绿部署
helm upgrade --install myapp-blue ./charts/myapp \
--namespace staging \
--set image.tag=$CI_COMMIT_SHA \
--set color=blue
# 健康检查
./scripts/health-check.sh https://staging.example.com
when: manual
only:
- main
# 生产环境:审批流程
deploy-prod:
stage: deploy
environment:
name: production
url: https://example.com
script:
- |
# 金丝雀发布
helm upgrade --install myapp ./charts/myapp \
--namespace production \
--set image.tag=$CI_COMMIT_SHA \
--set canary.enabled=true \
--set canary.traffic=10
# 监控金丝雀
sleep 300 # 等待5分钟
# 检查错误率
ERROR_RATE=$(./scripts/get-error-rate.sh)
if [ $ERROR_RATE -lt 1 ]; then
# 增加流量
helm upgrade myapp ./charts/myapp \
--set canary.traffic=100
else
# 回滚
helm rollback myapp
exit 1
fi
when: manual
only:
- tags
|
4.2 回滚机制#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
| # 自动回滚
deploy-with-rollback:
stage: deploy
script:
- |
# 保存当前版本
CURRENT_VERSION=$(kubectl get deployment app -o jsonpath='{.spec.template.spec.containers[0].image}')
echo "CURRENT=$CURRENT_VERSION" > version.txt
# 部署新版本
kubectl set image deployment/app app=$NEW_IMAGE
# 等待滚动更新
if ! kubectl rollout status deployment/app --timeout=5m; then
echo "部署失败,执行回滚"
kubectl rollout undo deployment/app
exit 1
fi
# 健康检查
if ! ./scripts/health-check.sh; then
echo "健康检查失败,执行回滚"
kubectl rollout undo deployment/app
exit 1
fi
artifacts:
paths:
- version.txt
# 手动回滚作业
manual-rollback:
stage: deploy
needs: [deploy-with-rollback]
script:
- |
# 从artifacts获取版本
CURRENT_VERSION=$(cat version.txt | grep CURRENT | cut -d= -f2)
# 回滚到上一个版本
kubectl rollout undo deployment/app
kubectl rollout status deployment/app
# 验证回滚
if ./scripts/health-check.sh; then
echo "回滚成功"
else
echo "回滚失败,需要人工介入"
exit 1
fi
when: manual
|
五、监控和告警#
5.1 Pipeline指标#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
| # 收集构建指标
metrics:
stage: .post
script:
- |
# 构建时长
DURATION=$((SECONDS - CI_PIPELINE_STARTED_AT))
# 测试覆盖率
COVERAGE=$(cat coverage/coverage-summary.json | jq '.total.lines.pct')
# 发送到监控系统
curl -X POST $METRICS_ENDPOINT \
-H "Content-Type: application/json" \
-d "{
\"pipeline_id\": \"$CI_PIPELINE_ID\",
\"project\": \"$CI_PROJECT_NAME\",
\"branch\": \"$CI_COMMIT_REF_NAME\",
\"duration\": $DURATION,
\"coverage\": $COVERAGE,
\"status\": \"$CI_JOB_STATUS\"
}"
# 构建报告
build-report:
stage: .post
script:
- |
# 生成HTML报告
cat > build-report.html <<EOF
<!DOCTYPE html>
<html>
<head><title>构建报告 #${CI_PIPELINE_ID}</title></head>
<body>
<h1>构建详情</h1>
<ul>
<li>提交: ${CI_COMMIT_SHA}</li>
<li>作者: ${GITLAB_USER_NAME}</li>
<li>分支: ${CI_COMMIT_REF_NAME}</li>
<li>状态: ${CI_PIPELINE_SOURCE}</li>
</ul>
<h2>作业</h2>
<table>
<tr><th>作业</th><th>状态</th><th>时长</th></tr>
$(gitlab-ci-local job-list --format html)
</table>
</body>
</html>
EOF
artifacts:
paths:
- build-report.html
expire_in: 30 days
|
5.2 告警配置#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
| # Slack告警
notify-slack:
stage: .post
script:
- |
COLOR=$([ "$CI_PIPELINE_SOURCE" == "schedule" ] && echo "#00FF00" || echo "#FF0000")
curl -X POST $SLACK_WEBHOOK \
-H 'Content-Type: application/json' \
-d "{
\"attachments\": [{
\"color\": \"$COLOR\",
\"title\": \"Pipeline ${CI_PIPELINE_ID}\",
\"fields\": [
{\"title\": \"项目\", \"value\": \"${CI_PROJECT_NAME}\"},
{\"title\": \"分支\", \"value\": \"${CI_COMMIT_REF_NAME}\"},
{\"title\": \"提交\", \"value\": \"${CI_COMMIT_SHA}\"},
{\"title\": \"作者\", \"value\": \"${GITLAB_USER_NAME}\"},
{\"title\": \"状态\", \"value\": \"${CI_PIPELINE_STATUS}\"}
]
}]
}"
when: always
# 邮件告警(失败时)
notify-email:
stage: .post
script:
- |
mail -s "Pipeline失败: ${CI_PROJECT_NAME}" \
-a build-report.html \
team@example.com <<< "Pipeline ${CI_PIPELINE_ID} 失败,请查看附件"
when: failure
only:
- main
|
六、安全实践#
6.1 Secret管理#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
| # ✅ 正确的Secret使用
secure-deploy:
stage: deploy
variables:
# 使用GitLab CI/CD变量
API_KEY: ${API_KEY}
DB_PASSWORD: ${DB_PASSWORD}
# Secret不会出现在日志中
script:
- |
# 在子shell中使用
(export DB_PASSWORD && ./deploy.sh)
# 或使用secret文件
echo ${DB_PASSWORD} > db_secret.txt
chmod 600 db_secret.txt
./deploy.sh --db-file db_secret.txt
rm db_secret.txt
# ❌ 错误:Secret泄露
leaky-secret:
stage: deploy
script:
- echo "Password: ${DB_PASSWORD}" # 会出现在日志中
- export SECRET=${SECRET} # set -x会显示
- ./config.sh ${API_KEY} # 进程列表可见
# Secret轮换
rotate-secrets:
stage: .pre
script:
- |
# 定期轮换Secret
if [ "$CI_PIPELINE_SOURCE" == "schedule" ]; then
# 生成新的API密钥
NEW_KEY=$(openssl rand -hex 32)
# 更新外部服务
curl -X POST $API_ENDPOINT \
-H "Authorization: Bearer ${ADMIN_TOKEN}" \
-d "{\"new_key\":\"$NEW_KEY\"}"
# 更新GitLab变量
curl --request PUT --header "PRIVATE-TOKEN: ${GITLAB_TOKEN}" \
"$CI_API_V4_URL/projects/$CI_PROJECT_ID/variables/API_KEY" \
--form "value=$NEW_KEY"
fi
only:
- schedules
|
6.2 安全扫描集成#
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
| # 依赖漏洞扫描
dependency-scan:
stage: quality
image: node:20
script:
- npm audit --audit-level=moderate
- npm audit --json > audit-report.json
artifacts:
reports:
sast: audit-report.json
allow_failure: true
# SAST静态分析
sast:
stage: quality
include:
- template: Security/SAST.gitlab-ci.yml
# 容器镜像扫描
container-scan:
stage: quality
image: aquasec/trivy:latest
script:
- trivy image --exit-code 1 --no-progress $CI_REGISTRY_IMAGE:$CI_COMMIT_SHA
allow_failure: true
# 代码质量检查
code-quality:
stage: quality
image: sonarsource/sonar-scanner-cli
script:
- sonar-scanner \
-Dsonar.projectKey=${CI_PROJECT_NAME} \
-Dsonar.sources=src \
-Dsonar.host.url=${SONAR_HOST} \
-Dsonar.login=${SONAR_TOKEN}
allow_failure: true
|
构建优秀的CI/CD流水线需要关注:
- 性能优化:缓存、并行、增量构建
- 质量保证:多层次测试、自动化检查
- 部署策略:多环境、金丝雀、自动回滚
- 监控告警:实时反馈、快速定位
- 安全第一:Secret管理、漏洞扫描
记住:CI/CD是持续改进的过程,不断优化才能发挥最大价值。