Trong môi trường ảo hóa, việc sử dụng các ảnh đĩa QCOW2 đã được nén là một giải pháp phổ biến để tối ưu hóa không gian lưu trữ. Tuy nhiên, thực tế vận hành cho thấy các máy ảo (VM) khởi động từ ảnh nén thường có thời gian boot chậm hơn đáng kể. Bài viết này sẽ đi sâu vào phân tích mã nguồn QEMU để làm rõ nguyên nhân kỹ thuật gây ra hiện tượng trên.
Bối cảnh kỹ thuật
Lệnh thường dùng để tạo ảnh QCOW2 nén là:
qemu-img convert -p -c -O qcow2 source_disk.qcow2 compressed_disk.qcow2
Để hiểu rõ cơ chế tác động, chúng ta cần xem xét luồng xử lý I/O của backend driver qcow2 trong mã nguồn QEMU (phiên bản 9.2.0).
Cơ chế xử lý đọc dữ liệu (Read Path)
Tại file block/qcow2.c, driver block bdrv_qcow2 định nghĩa các hàm callback chịu trách nhiệm đọc và ghi dữ liệu. Cụ thể, hàm qcow2_co_preadv_part được đăng ký để xử lý các yêu cầu đọc.
Đoạn mã dưới đây mô phỏng cấu trúc đăng ký driver, tập trung vào các thao tác I/O chính:
// block/qcow2.c
BlockDriver bdrv_qcow2 = {
.format_name = "qcow2",
.instance_size = sizeof(BDRVQcow2State),
// Các hàm callback quản lý vòng đời
.bdrv_open = qcow2_open,
.bdrv_close = qcow2_close,
// Đăng ký hàm xử lý đọc
.bdrv_co_preadv_part = qcow2_co_preadv_part,
.bdrv_co_pwritev_part = qcow2_co_pwritev_part,
// Hỗ trợ ghi nén
.bdrv_co_pwritev_compressed_part = qcow2_co_pwritev_compressed_part,
// ...
};
Khi Guest OS thực hiện thao tác đọc, hàm qcow2_co_preadv_part sẽ được gọi. Hàm này có nhiệm vụ xác định vị trí thực tế của dữ liệu trên file và xử lý theo từng loại cluster (cụm dữ liệu).
static int coroutine_fn
qcow2_co_preadv_part(BlockDriverState *bs, int64_t offset, int64_t bytes,
QEMUIOVector *qiov, size_t qiov_offset,
BdrvRequestFlags flags)
{
BDRVQcow2State *s = bs->opaque;
int ret = 0;
unsigned int chunk_bytes;
uint64_t host_offset = 0;
QCow2SubclusterType sc_type;
while (bytes > 0) {
chunk_bytes = MIN(bytes, INT_MAX);
// Xác định vị trí và loại của cluster hiện tại
qemu_co_mutex_lock(&s->lock);
ret = qcow2_get_host_offset(bs, offset, &chunk_bytes, &host_offset, &sc_type);
qemu_co_mutex_unlock(&s->lock);
if (ret < 0) {
break;
}
// Xử lý theo loại cluster
switch (sc_type) {
case QCOW2_SUBCLUSTER_ZERO_PLAIN:
case QCOW2_SUBCLUSTER_UNALLOCATED_PLAIN:
// Trả về dữ liệu zero nếu chưa cấp phát
qemu_iovec_memset(qiov, qiov_offset, 0, chunk_bytes);
break;
default:
// Giao việc xử lý cho task cụ thể
ret = qcow2_schedule_read_task(bs, sc_type, host_offset,
offset, chunk_bytes, qiov, qiov_offset);
if (ret < 0) {
goto fail;
}
break;
}
bytes -= chunk_bytes;
offset += chunk_bytes;
qiov_offset += chunk_bytes;
}
fail:
return ret;
}
Logic phân loại và xử lý nén
Tiếp theo, hàm xử lý task đọc (qcow2_process_read_task - tên giả định dựa trên logic gốc qcow2_co_preadv_task) sẽ kiểm tra loại subcluster. Nếu dữ liệu ở trạng thái nén (QCOW2_SUBCLUSTER_COMPRESSED), một luồng xử lý riêng biệt sẽ được kích hoạt.
static int coroutine_fn
qcow2_process_read_task(BlockDriverState *bs, QCow2SubclusterType type,
uint64_t host_off, uint64_t guest_off, uint64_t len,
QEMUIOVector *qiov, size_t qiov_off)
{
switch (type) {
case QCOW2_SUBCLUSTER_COMPRESSED:
// Gọi hàm giải nén chuyên dụng
return qcow2_handle_compressed_read(bs, host_off, guest_off,
len, qiov, qiov_off);
case QCOW2_SUBCLUSTER_NORMAL:
// Đọc dữ liệu thường (có thể mã hóa)
if (bs->encrypted) {
return qcow2_read_encrypted(bs, host_off, guest_off, len, qiov, qiov_off);
}
return bdrv_co_preadv_part(bs->data_file, host_off, len, qiov, qiov_off, 0);
default:
// Xử lý các trường hợp khác (zero, unallocated)
return 0;
}
}
Chi phí giải nén dữ liệu
Hàm qcow2_handle_compressed_read chính là nguyên nhân chính gây ra độ trễ. Thay vì đọc trực tiếp dữ liệu từ đĩa vào bộ nhớ Guest, quy trình phải thực hiện các bước trung gian tốn kém tài nguyên CPU.
static int coroutine_fn
qcow2_handle_compressed_read(BlockDriverState *bs, uint64_t host_off,
uint64_t guest_off, uint64_t len,
QEMUIOVector *qiov, size_t qiov_off)
{
BDRVQcow2State *s = bs->opaque;
int ret;
uint8_t *compressed_buf;
uint8_t *decompressed_buf;
int compressed_size;
int offset_in_cluster = offset_into_cluster(s, guest_off);
// Tính toán kích thước và phân bổ buffer
qcow2_calc_compressed_size(s, host_off, &compressed_size);
compressed_buf = g_try_malloc(compressed_size);
decompressed_buf = qemu_blockalign(bs, s->cluster_size);
// 1. Đọc dữ liệu nén từ thiết bị lưu trữ
ret = bdrv_co_pread(bs->file, host_off, compressed_size, compressed_buf, 0);
if (ret < 0) {
goto cleanup;
}
// 2. Giải nén dữ liệu (Tiêu tốn CPU)
ret = qcow2_decompress_data(bs, decompressed_buf, s->cluster_size,
compressed_buf, compressed_size);
if (ret < 0) {
goto cleanup;
}
// 3. Sao chép dữ liệu đã giải nén vào buffer của Guest
qemu_iovec_from_buf(qiov, qiov_off, decompressed_buf + offset_in_cluster, len);
cleanup:
qemu_vfree(decompressed_buf);
g_free(compressed_buf);
return ret;
}
Cuối cùng, hàm qcow2_decompress_data sẽ lựa chọn thuật toán giải nén phù hợp (mặc định là Zlib hoặc Zstd nếu được biên dịch).
ssize_t coroutine_fn
qcow2_decompress_data(BlockDriverState *bs, void *dst, size_t dst_size,
const void *src, size_t src_size)
{
BDRVQcow2State *s = bs->opaque;
Qcow2CompressFunc decompress_func;
switch (s->compression_type) {
case QCOW2_COMPRESSION_TYPE_ZLIB:
decompress_func = qcow2_zlib_decompress;
break;
#ifdef CONFIG_ZSTD
case QCOW2_COMPRESSION_TYPE_ZSTD:
decompress_func = qcow2_zstd_decompress;
break;
#endif
default:
return -ENOTSUP;
}
return decompress_func(dst, dst_size, src, src_size);
}
Kết luận
Qua việc phân tích luồng xử lý I/O, có thể thấy rằng khi sử dụng ảnh QCOW2 được nén, mỗi thao tác đọc dữ liệu của Guest VM đều kích hoạt một quy trình giải nén trên Host. Việc này buộc CPU của Host phải thực hiện tính toán giải nén trước khi trả dữ liệu về cho Guest. Chính chi phí tính toán này (CPU overhead) và độ trễ của thuật toán giải nén là nguyên nhân khiến hiệu năng đọc bị giảm sút, biểu hiện rõ nhất là thời gian khởi động máy ảo chậm hơn.