2 * Block driver for the QCOW version 2 format
4 * Copyright (c) 2004-2006 Fabrice Bellard
6 * Permission is hereby granted, free of charge, to any person obtaining a copy
7 * of this software and associated documentation files (the "Software"), to deal
8 * in the Software without restriction, including without limitation the rights
9 * to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
10 * copies of the Software, and to permit persons to whom the Software is
11 * furnished to do so, subject to the following conditions:
13 * The above copyright notice and this permission notice shall be included in
14 * all copies or substantial portions of the Software.
16 * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
17 * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
18 * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL
19 * THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
20 * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
21 * OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
29 #include "block/coroutine.h"
32 //#define DEBUG_ALLOC2
35 #define QCOW_MAGIC (('Q' << 24) | ('F' << 16) | ('I' << 8) | 0xfb)
37 #define QCOW_CRYPT_NONE 0
38 #define QCOW_CRYPT_AES 1
40 #define QCOW_MAX_CRYPT_CLUSTERS 32
42 /* indicate that the refcount of the referenced cluster is exactly one. */
43 #define QCOW_OFLAG_COPIED (1ULL << 63)
44 /* indicate that the cluster is compressed (they never have the copied flag) */
45 #define QCOW_OFLAG_COMPRESSED (1ULL << 62)
46 /* The cluster reads as all zeros */
47 #define QCOW_OFLAG_ZERO (1ULL << 0)
49 #define REFCOUNT_SHIFT 1 /* refcount size is 2 bytes */
51 #define MIN_CLUSTER_BITS 9
52 #define MAX_CLUSTER_BITS 21
54 #define L2_CACHE_SIZE 16
56 /* Must be at least 4 to cover all cases of refcount table growth */
57 #define REFCOUNT_CACHE_SIZE 4
59 #define DEFAULT_CLUSTER_SIZE 65536
62 #define QCOW2_OPT_LAZY_REFCOUNTS "lazy-refcounts"
63 #define QCOW2_OPT_DISCARD_REQUEST "pass-discard-request"
64 #define QCOW2_OPT_DISCARD_SNAPSHOT "pass-discard-snapshot"
65 #define QCOW2_OPT_DISCARD_OTHER "pass-discard-other"
67 typedef struct QCowHeader
{
70 uint64_t backing_file_offset
;
71 uint32_t backing_file_size
;
72 uint32_t cluster_bits
;
73 uint64_t size
; /* in bytes */
74 uint32_t crypt_method
;
75 uint32_t l1_size
; /* XXX: save number of clusters instead ? */
76 uint64_t l1_table_offset
;
77 uint64_t refcount_table_offset
;
78 uint32_t refcount_table_clusters
;
79 uint32_t nb_snapshots
;
80 uint64_t snapshots_offset
;
82 /* The following fields are only valid for version >= 3 */
83 uint64_t incompatible_features
;
84 uint64_t compatible_features
;
85 uint64_t autoclear_features
;
87 uint32_t refcount_order
;
88 uint32_t header_length
;
89 } QEMU_PACKED QCowHeader
;
91 typedef struct QCowSnapshot
{
92 uint64_t l1_table_offset
;
97 uint64_t vm_state_size
;
100 uint64_t vm_clock_nsec
;
104 typedef struct Qcow2Cache Qcow2Cache
;
106 typedef struct Qcow2UnknownHeaderExtension
{
109 QLIST_ENTRY(Qcow2UnknownHeaderExtension
) next
;
111 } Qcow2UnknownHeaderExtension
;
114 QCOW2_FEAT_TYPE_INCOMPATIBLE
= 0,
115 QCOW2_FEAT_TYPE_COMPATIBLE
= 1,
116 QCOW2_FEAT_TYPE_AUTOCLEAR
= 2,
119 /* Incompatible feature bits */
121 QCOW2_INCOMPAT_DIRTY_BITNR
= 0,
122 QCOW2_INCOMPAT_CORRUPT_BITNR
= 1,
123 QCOW2_INCOMPAT_DIRTY
= 1 << QCOW2_INCOMPAT_DIRTY_BITNR
,
124 QCOW2_INCOMPAT_CORRUPT
= 1 << QCOW2_INCOMPAT_CORRUPT_BITNR
,
126 QCOW2_INCOMPAT_MASK
= QCOW2_INCOMPAT_DIRTY
127 | QCOW2_INCOMPAT_CORRUPT
,
130 /* Compatible feature bits */
132 QCOW2_COMPAT_LAZY_REFCOUNTS_BITNR
= 0,
133 QCOW2_COMPAT_LAZY_REFCOUNTS
= 1 << QCOW2_COMPAT_LAZY_REFCOUNTS_BITNR
,
135 QCOW2_COMPAT_FEAT_MASK
= QCOW2_COMPAT_LAZY_REFCOUNTS
,
138 enum qcow2_discard_type
{
139 QCOW2_DISCARD_NEVER
= 0,
140 QCOW2_DISCARD_ALWAYS
,
141 QCOW2_DISCARD_REQUEST
,
142 QCOW2_DISCARD_SNAPSHOT
,
147 typedef struct Qcow2Feature
{
151 } QEMU_PACKED Qcow2Feature
;
153 typedef struct Qcow2DiscardRegion
{
154 BlockDriverState
*bs
;
157 QTAILQ_ENTRY(Qcow2DiscardRegion
) next
;
158 } Qcow2DiscardRegion
;
160 typedef struct BDRVQcowState
{
167 int l1_vm_state_index
;
170 uint64_t cluster_offset_mask
;
171 uint64_t l1_table_offset
;
174 Qcow2Cache
* l2_table_cache
;
175 Qcow2Cache
* refcount_block_cache
;
177 uint8_t *cluster_cache
;
178 uint8_t *cluster_data
;
179 uint64_t cluster_cache_offset
;
180 QLIST_HEAD(QCowClusterAlloc
, QCowL2Meta
) cluster_allocs
;
182 uint64_t *refcount_table
;
183 uint64_t refcount_table_offset
;
184 uint32_t refcount_table_size
;
185 int64_t free_cluster_index
;
186 int64_t free_byte_offset
;
190 uint32_t crypt_method
; /* current crypt method, 0 if no key yet */
191 uint32_t crypt_method_header
;
192 AES_KEY aes_encrypt_key
;
193 AES_KEY aes_decrypt_key
;
194 uint64_t snapshots_offset
;
197 QCowSnapshot
*snapshots
;
201 bool use_lazy_refcounts
;
204 bool discard_passthrough
[QCOW2_DISCARD_MAX
];
206 uint64_t incompatible_features
;
207 uint64_t compatible_features
;
208 uint64_t autoclear_features
;
210 size_t unknown_header_fields_size
;
211 void* unknown_header_fields
;
212 QLIST_HEAD(, Qcow2UnknownHeaderExtension
) unknown_header_ext
;
213 QTAILQ_HEAD (, Qcow2DiscardRegion
) discards
;
217 /* XXX: use std qcow open function ? */
218 typedef struct QCowCreateState
{
221 uint16_t *refcount_block
;
222 uint64_t *refcount_table
;
223 int64_t l1_table_offset
;
224 int64_t refcount_table_offset
;
225 int64_t refcount_block_offset
;
230 typedef struct Qcow2COWRegion
{
232 * Offset of the COW region in bytes from the start of the first cluster
233 * touched by the request.
237 /** Number of sectors to copy */
242 * Describes an in-flight (part of a) write request that writes to clusters
243 * that are not referenced in their L2 table yet.
245 typedef struct QCowL2Meta
247 /** Guest offset of the first newly allocated cluster */
250 /** Host offset of the first newly allocated cluster */
251 uint64_t alloc_offset
;
254 * Number of sectors from the start of the first allocated cluster to
255 * the end of the (possibly shortened) request
259 /** Number of newly allocated clusters */
263 * Requests that overlap with this allocation and wait to be restarted
264 * when the allocating request has completed.
266 CoQueue dependent_requests
;
269 * The COW Region between the start of the first allocated cluster and the
270 * area the guest actually writes to.
272 Qcow2COWRegion cow_start
;
275 * The COW Region between the area the guest actually writes to and the
276 * end of the last allocated cluster.
278 Qcow2COWRegion cow_end
;
280 /** Pointer to next L2Meta of the same write request */
281 struct QCowL2Meta
*next
;
283 QLIST_ENTRY(QCowL2Meta
) next_in_flight
;
287 QCOW2_CLUSTER_UNALLOCATED
,
288 QCOW2_CLUSTER_NORMAL
,
289 QCOW2_CLUSTER_COMPRESSED
,
293 typedef enum QCow2MetadataOverlap
{
294 QCOW2_OL_MAIN_HEADER_BITNR
= 0,
295 QCOW2_OL_ACTIVE_L1_BITNR
= 1,
296 QCOW2_OL_ACTIVE_L2_BITNR
= 2,
297 QCOW2_OL_REFCOUNT_TABLE_BITNR
= 3,
298 QCOW2_OL_REFCOUNT_BLOCK_BITNR
= 4,
299 QCOW2_OL_SNAPSHOT_TABLE_BITNR
= 5,
300 QCOW2_OL_INACTIVE_L1_BITNR
= 6,
301 QCOW2_OL_INACTIVE_L2_BITNR
= 7,
303 QCOW2_OL_MAX_BITNR
= 8,
306 QCOW2_OL_MAIN_HEADER
= (1 << QCOW2_OL_MAIN_HEADER_BITNR
),
307 QCOW2_OL_ACTIVE_L1
= (1 << QCOW2_OL_ACTIVE_L1_BITNR
),
308 QCOW2_OL_ACTIVE_L2
= (1 << QCOW2_OL_ACTIVE_L2_BITNR
),
309 QCOW2_OL_REFCOUNT_TABLE
= (1 << QCOW2_OL_REFCOUNT_TABLE_BITNR
),
310 QCOW2_OL_REFCOUNT_BLOCK
= (1 << QCOW2_OL_REFCOUNT_BLOCK_BITNR
),
311 QCOW2_OL_SNAPSHOT_TABLE
= (1 << QCOW2_OL_SNAPSHOT_TABLE_BITNR
),
312 QCOW2_OL_INACTIVE_L1
= (1 << QCOW2_OL_INACTIVE_L1_BITNR
),
313 /* NOTE: Checking overlaps with inactive L2 tables will result in bdrv
315 QCOW2_OL_INACTIVE_L2
= (1 << QCOW2_OL_INACTIVE_L2_BITNR
),
316 } QCow2MetadataOverlap
;
318 /* Perform all overlap checks which don't require disk access */
319 #define QCOW2_OL_CACHED \
320 (QCOW2_OL_MAIN_HEADER | QCOW2_OL_ACTIVE_L1 | QCOW2_OL_ACTIVE_L2 | \
321 QCOW2_OL_REFCOUNT_TABLE | QCOW2_OL_REFCOUNT_BLOCK | \
322 QCOW2_OL_SNAPSHOT_TABLE | QCOW2_OL_INACTIVE_L1)
324 /* The default checks to perform */
325 #define QCOW2_OL_DEFAULT QCOW2_OL_CACHED
327 #define L1E_OFFSET_MASK 0x00ffffffffffff00ULL
328 #define L2E_OFFSET_MASK 0x00ffffffffffff00ULL
329 #define L2E_COMPRESSED_OFFSET_SIZE_MASK 0x3fffffffffffffffULL
331 #define REFT_OFFSET_MASK 0xffffffffffffff00ULL
333 static inline int64_t start_of_cluster(BDRVQcowState
*s
, int64_t offset
)
335 return offset
& ~(s
->cluster_size
- 1);
338 static inline int64_t offset_into_cluster(BDRVQcowState
*s
, int64_t offset
)
340 return offset
& (s
->cluster_size
- 1);
343 static inline int size_to_clusters(BDRVQcowState
*s
, int64_t size
)
345 return (size
+ (s
->cluster_size
- 1)) >> s
->cluster_bits
;
348 static inline int64_t size_to_l1(BDRVQcowState
*s
, int64_t size
)
350 int shift
= s
->cluster_bits
+ s
->l2_bits
;
351 return (size
+ (1ULL << shift
) - 1) >> shift
;
354 static inline int offset_to_l2_index(BDRVQcowState
*s
, int64_t offset
)
356 return (offset
>> s
->cluster_bits
) & (s
->l2_size
- 1);
359 static inline int64_t align_offset(int64_t offset
, int n
)
361 offset
= (offset
+ n
- 1) & ~(n
- 1);
365 static inline int64_t qcow2_vm_state_offset(BDRVQcowState
*s
)
367 return (int64_t)s
->l1_vm_state_index
<< (s
->cluster_bits
+ s
->l2_bits
);
370 static inline int qcow2_get_cluster_type(uint64_t l2_entry
)
372 if (l2_entry
& QCOW_OFLAG_COMPRESSED
) {
373 return QCOW2_CLUSTER_COMPRESSED
;
374 } else if (l2_entry
& QCOW_OFLAG_ZERO
) {
375 return QCOW2_CLUSTER_ZERO
;
376 } else if (!(l2_entry
& L2E_OFFSET_MASK
)) {
377 return QCOW2_CLUSTER_UNALLOCATED
;
379 return QCOW2_CLUSTER_NORMAL
;
383 /* Check whether refcounts are eager or lazy */
384 static inline bool qcow2_need_accurate_refcounts(BDRVQcowState
*s
)
386 return !(s
->incompatible_features
& QCOW2_INCOMPAT_DIRTY
);
389 static inline uint64_t l2meta_cow_start(QCowL2Meta
*m
)
391 return m
->offset
+ m
->cow_start
.offset
;
394 static inline uint64_t l2meta_cow_end(QCowL2Meta
*m
)
396 return m
->offset
+ m
->cow_end
.offset
397 + (m
->cow_end
.nb_sectors
<< BDRV_SECTOR_BITS
);
400 // FIXME Need qcow2_ prefix to global functions
402 /* qcow2.c functions */
403 int qcow2_backing_read1(BlockDriverState
*bs
, QEMUIOVector
*qiov
,
404 int64_t sector_num
, int nb_sectors
);
406 int qcow2_mark_dirty(BlockDriverState
*bs
);
407 int qcow2_mark_corrupt(BlockDriverState
*bs
);
408 int qcow2_mark_consistent(BlockDriverState
*bs
);
409 int qcow2_update_header(BlockDriverState
*bs
);
411 /* qcow2-refcount.c functions */
412 int qcow2_refcount_init(BlockDriverState
*bs
);
413 void qcow2_refcount_close(BlockDriverState
*bs
);
415 int qcow2_update_cluster_refcount(BlockDriverState
*bs
, int64_t cluster_index
,
416 int addend
, enum qcow2_discard_type type
);
418 int64_t qcow2_alloc_clusters(BlockDriverState
*bs
, int64_t size
);
419 int qcow2_alloc_clusters_at(BlockDriverState
*bs
, uint64_t offset
,
421 int64_t qcow2_alloc_bytes(BlockDriverState
*bs
, int size
);
422 void qcow2_free_clusters(BlockDriverState
*bs
,
423 int64_t offset
, int64_t size
,
424 enum qcow2_discard_type type
);
425 void qcow2_free_any_clusters(BlockDriverState
*bs
, uint64_t l2_entry
,
426 int nb_clusters
, enum qcow2_discard_type type
);
428 int qcow2_update_snapshot_refcount(BlockDriverState
*bs
,
429 int64_t l1_table_offset
, int l1_size
, int addend
);
431 int qcow2_check_refcounts(BlockDriverState
*bs
, BdrvCheckResult
*res
,
434 void qcow2_process_discards(BlockDriverState
*bs
, int ret
);
436 int qcow2_check_metadata_overlap(BlockDriverState
*bs
, int chk
, int64_t offset
,
438 int qcow2_pre_write_overlap_check(BlockDriverState
*bs
, int chk
, int64_t offset
,
441 /* qcow2-cluster.c functions */
442 int qcow2_grow_l1_table(BlockDriverState
*bs
, uint64_t min_size
,
444 int qcow2_write_l1_entry(BlockDriverState
*bs
, int l1_index
);
445 void qcow2_l2_cache_reset(BlockDriverState
*bs
);
446 int qcow2_decompress_cluster(BlockDriverState
*bs
, uint64_t cluster_offset
);
447 void qcow2_encrypt_sectors(BDRVQcowState
*s
, int64_t sector_num
,
448 uint8_t *out_buf
, const uint8_t *in_buf
,
449 int nb_sectors
, int enc
,
452 int qcow2_get_cluster_offset(BlockDriverState
*bs
, uint64_t offset
,
453 int *num
, uint64_t *cluster_offset
);
454 int qcow2_alloc_cluster_offset(BlockDriverState
*bs
, uint64_t offset
,
455 int n_start
, int n_end
, int *num
, uint64_t *host_offset
, QCowL2Meta
**m
);
456 uint64_t qcow2_alloc_compressed_cluster_offset(BlockDriverState
*bs
,
458 int compressed_size
);
460 int qcow2_alloc_cluster_link_l2(BlockDriverState
*bs
, QCowL2Meta
*m
);
461 int qcow2_discard_clusters(BlockDriverState
*bs
, uint64_t offset
,
462 int nb_sectors
, enum qcow2_discard_type type
);
463 int qcow2_zero_clusters(BlockDriverState
*bs
, uint64_t offset
, int nb_sectors
);
465 int qcow2_expand_zero_clusters(BlockDriverState
*bs
);
467 /* qcow2-snapshot.c functions */
468 int qcow2_snapshot_create(BlockDriverState
*bs
, QEMUSnapshotInfo
*sn_info
);
469 int qcow2_snapshot_goto(BlockDriverState
*bs
, const char *snapshot_id
);
470 int qcow2_snapshot_delete(BlockDriverState
*bs
,
471 const char *snapshot_id
,
474 int qcow2_snapshot_list(BlockDriverState
*bs
, QEMUSnapshotInfo
**psn_tab
);
475 int qcow2_snapshot_load_tmp(BlockDriverState
*bs
, const char *snapshot_name
);
477 void qcow2_free_snapshots(BlockDriverState
*bs
);
478 int qcow2_read_snapshots(BlockDriverState
*bs
);
480 /* qcow2-cache.c functions */
481 Qcow2Cache
*qcow2_cache_create(BlockDriverState
*bs
, int num_tables
);
482 int qcow2_cache_destroy(BlockDriverState
* bs
, Qcow2Cache
*c
);
484 void qcow2_cache_entry_mark_dirty(Qcow2Cache
*c
, void *table
);
485 int qcow2_cache_flush(BlockDriverState
*bs
, Qcow2Cache
*c
);
486 int qcow2_cache_set_dependency(BlockDriverState
*bs
, Qcow2Cache
*c
,
487 Qcow2Cache
*dependency
);
488 void qcow2_cache_depends_on_flush(Qcow2Cache
*c
);
490 int qcow2_cache_empty(BlockDriverState
*bs
, Qcow2Cache
*c
);
492 int qcow2_cache_get(BlockDriverState
*bs
, Qcow2Cache
*c
, uint64_t offset
,
494 int qcow2_cache_get_empty(BlockDriverState
*bs
, Qcow2Cache
*c
, uint64_t offset
,
496 int qcow2_cache_put(BlockDriverState
*bs
, Qcow2Cache
*c
, void **table
);