/* This is used to stop/restart our threads. *CannotuseSIGTERMnorSIGKILL,sincethese *aresentoutbyinitonrunlevelchanges *IchooseSIGHUPfornow.
*/ #define DRBD_SIGKILL SIGHUP
/* for sending/receiving the bitmap,
* possibly in some encoding scheme */ struct bm_xfer_ctx { /* "const" *storestotalbitsandlongwords *ofthebitmap,sowedon'tneedto
* call the accessor functions over and again. */ unsignedlong bm_bits; unsignedlong bm_words; /* during xfer, current position within the bitmap */ unsignedlong bit_offset; unsignedlong word_offset;
staticinlineenum drbd_thread_state get_t_state(struct drbd_thread *thi)
{ /* THINK testing the t_state seems to be uncritical in all cases *(butthread_{start,stop}),sowecanreadit*without*thelock.
* --lge */
smp_rmb(); return thi->t_state;
}
struct drbd_work { struct list_head list; int (*cb)(struct drbd_work *, int cancel);
};
/* if local IO is not allowed, will be NULL. *iflocalIO_is_allowed,holdsthelocallysubmittedbioclone, *or,afterlocalIOcompletion,theERR_PTR(error).
* see drbd_request_endio(). */ struct bio *private_bio;
struct drbd_interval i;
/* epoch: used to check on "completion" whether this req was in *thecurrentepoch,andwethereforehavetocloseit, *causingap_barrierpackettobesend,startinganewepoch. * *Thiscorrespondsto"barrier"instructp_barrier[_ack], *andto"barrier_nr"instructdrbd_epoch(andvarious *comments/functionparameters/localvariablenames).
*/ unsignedint epoch;
struct list_head tl_requests; /* ring list in the transfer log */ struct bio *master_bio; /* master bio pointer */
/* for generic IO accounting */ unsignedlong start_jif;
/* for DRBD internal statistics */
/* Minimal set of time stamps to determine if we wait for activity log *transactions,localdiskorpeer.32bit"jiffies"aregoodenough, *wedon'texpectaDRBDrequesttobestalledforseveralmonth.
*/
/* before actual request processing */ unsignedlong in_actlog_jif;
/* local disk */ unsignedlong pre_submit_jif;
/* per connection */ unsignedlong pre_send_jif; unsignedlong acked_jif; unsignedlong net_done_jif;
/* Possibly even more detail to track each phase: *master_completion_jif *howlongdidittaketocompletethemasterbio *(applicationvisiblelatency) *allocated_jif *howlongthemasterbiowasblockeduntilwefinallyallocated *atrackingstruct *in_actlog_jif *howlongdidwewaitforactivitylogtransactions * *net_queued_jif *whendidwefinallyqueueitforsending *pre_send_jif *whendidwestartsendingit *post_send_jif *howlongdidweblockinthenetworkstacktryingtosendit *acked_jif *whendidwereceive(orfake,inprotocolA)aremoteACK *net_done_jif *whendidwereceivefinalacknowledgement(P_BARRIER_ACK), *ordecide,e.g.onconnectionloss,thatwedonolongerexpect *anythingfromthispeerforthisrequest. * *pre_submit_jif *post_sub_jif *whendidwestartsubmitingtothelowerleveldevice, *andhowlongdidweblockinthatsubmitfunction *local_completion_jif *howlongdidittakethelowerleveldevicetocompletethisrequest
*/
/* once it hits 0, we may complete the master_bio */
atomic_t completion_ref; /* once it hits 0, we may destroy this drbd_request object */ struct kref kref;
unsigned rq_state; /* see comments above _req_mod() */
};
struct drbd_epoch { struct drbd_connection *connection; struct list_head list; unsignedint barrier_nr;
atomic_t epoch_size; /* increased on every request added. */
atomic_t active; /* increased on every req. added, and dec on every finished. */ unsignedlong flags;
};
/* drbd_epoch flag bits */ enum {
DE_HAVE_BARRIER_NUMBER,
};
enum epoch_event {
EV_PUT,
EV_GOT_BARRIER_NR,
EV_BECAME_LAST,
EV_CLEANUP = 32, /* used as flag */
};
struct digest_info { int digest_size; void *digest;
};
struct drbd_peer_request { struct drbd_work w; struct drbd_peer_device *peer_device; struct drbd_epoch *epoch; /* for writes */ struct page *pages;
blk_opf_t opf;
atomic_t pending_bios; struct drbd_interval i; /* see comments on ee flag bits below */ unsignedlong flags; unsignedlong submit_jif; union {
u64 block_id; struct digest_info *digest;
};
};
/* Equivalent to bio_op and req_op. */ #define peer_req_op(peer_req) \
((peer_req)->opf & REQ_OP_MASK)
/* ee flag bits. *Whilecorrespondingbiosareinflight,theonlymodificationwillbe *set_bitWAS_ERROR,whichhastobeatomic. *Ifnobiosareinflightyet,orallhavebeencompleted, *non-atomicmodificationtoee->flagsisok.
*/ enum {
__EE_CALL_AL_COMPLETE_IO,
__EE_MAY_SET_IN_SYNC,
/* is this a TRIM aka REQ_OP_DISCARD? */
__EE_TRIM, /* explicit zero-out requested, or *ourlowerlevelcannothandletrim,
* and we want to fall back to zeroout instead */
__EE_ZEROOUT,
/* In case a barrier failed,
* we need to resubmit without the barrier flag. */
__EE_RESUBMITTED,
/* we may have several bios per peer request. *ifanyofthosefail,wesetthisflagatomically
* from the endio callback */
__EE_WAS_ERROR,
/* This ee has a pointer to a digest instead of a block id */
__EE_HAS_DIGEST,
/* Conflicting local requests need to be restarted after this request */
__EE_RESTART_REQUESTS,
/* The peer wants a write ACK for this (wire proto C) */
__EE_SEND_WRITE_ACK,
/* Is set when net_conf had two_primaries set while creating this peer_req */
__EE_IN_INTERVAL_TREE,
/* for debugfs: */ /* has this been submitted, or does it still wait for something else? */
__EE_SUBMITTED,
/* this is/was a write request */
__EE_WRITE,
/* hand back using mempool_free(e, drbd_buffer_page_pool) */
__EE_RELEASE_TO_MEMPOOL,
/* this is/was a write same request */
__EE_WRITE_SAME,
/* this originates from application on peer
* (not some resync or verify or other DRBD internal request) */
__EE_APPLICATION,
/* flag bits per device */ enum {
UNPLUG_REMOTE, /* sending a "UnplugRemote" could help */
MD_DIRTY, /* current uuids and flags not yet on disk */
USE_DEGR_WFC_T, /* degr-wfc-timeout instead of wfc-timeout. */
CL_ST_CHG_SUCCESS,
CL_ST_CHG_FAIL,
CRASHED_PRIMARY, /* This node was a crashed primary. *Getsclearedwhenthestate.conn
* goes into C_CONNECTED state. */
CONSIDER_RESYNC,
MD_NO_FUA, /* Users wants us to not use FUA/FLUSH on meta data dev */
BITMAP_IO, /* suspend application io;
once no more io in flight, start bitmap io */
BITMAP_IO_QUEUED, /* Started bitmap IO */
WAS_IO_ERROR, /* Local disk failed, returned IO error */
WAS_READ_ERROR, /* Local disk READ failed (set additionally to the above) */
FORCE_DETACH, /* Force-detach from local disk, aborting any pending local IO */
RESYNC_AFTER_NEG, /* Resync after online grow after the attach&negotiate finished. */
RESIZE_PENDING, /* Size change detected locally, waiting for the response from
* the peer, if it changed there as well. */
NEW_CUR_UUID, /* Create new current UUID when thawing IO */
AL_SUSPENDED, /* Activity logging is currently suspended. */
AHEAD_TO_SYNC_SOURCE, /* Ahead -> SyncSource queued */
B_RS_H_DONE, /* Before resync handler done (already executed) */
DISCARD_MY_DATA, /* discard_my_data flag per volume */
READ_BALANCE_RR,
FLUSH_PENDING, /* if set, device->flush_jif is when we submitted that flush
* from drbd_flush_after_epoch() */
/* cleared only after backing device related structures have been destroyed. */
GOING_DISKLESS, /* Disk is being detached, because of io-error, or admin request. */
/* to be used in drbd_device_post_work() */
GO_DISKLESS, /* tell worker to schedule cleanup before detach */
DESTROY_DISK, /* tell worker to close backing devices and destroy related structures. */
MD_SYNC, /* tell worker to call drbd_md_sync() */
RS_START, /* tell worker to start resync/OV */
RS_PROGRESS, /* tell worker that resync made significant progress */
RS_DONE, /* tell worker that resync is done */
};
struct drbd_bitmap; /* opaque for drbd_device */
/* definition of bits in bm_flags to be used in drbd_bm_lock
* and drbd_bitmap_io and friends. */ enum bm_flag { /* currently locked for bulk operation */
BM_LOCKED_MASK = 0xf,
/* in detail, that is: */
BM_DONT_CLEAR = 0x1,
BM_DONT_SET = 0x2,
BM_DONT_TEST = 0x4,
/* so we can mark it locked for bulk operation,
* and still allow all non-bulk operations */
BM_IS_LOCKED = 0x8,
/* testing bits, as well as setting new bits allowed, but clearing bits *wouldbeunexpected.Usedduringbitmapreceive.Settingnewbits
* requires sending of "out-of-sync" information, though. */
BM_LOCKED_SET_ALLOWED = BM_DONT_CLEAR | BM_IS_LOCKED,
/* for drbd_bm_write_copy_pages, everything is allowed,
* only concurrent bulk operations are locked out. */
BM_LOCKED_CHANGE_ALLOWED = BM_IS_LOCKED,
};
struct drbd_work_queue { struct list_head q;
spinlock_t q_lock; /* to protect the list. */
wait_queue_head_t q_wait;
};
struct drbd_socket { struct mutex mutex; struct socket *socket; /* this way we get our
* send/receive buffers off the stack */ void *sbuf; void *rbuf;
};
struct fifo_buffer { unsignedint head_index; unsignedint size; int total; /* sum of all values */ int values[] __counted_by(size);
}; externstruct fifo_buffer *fifo_alloc(unsignedint fifo_size);
/* flag bits per connection */ enum {
NET_CONGESTED, /* The data socket is congested */
RESOLVE_CONFLICTS, /* Set on one node, cleared on the peer! */
SEND_PING,
GOT_PING_ACK, /* set when we receive a ping_ack packet, ping_wait gets woken */
CONN_WD_ST_CHG_REQ, /* A cluster wide state change on the connection is active */
CONN_WD_ST_CHG_OKAY,
CONN_WD_ST_CHG_FAIL,
CONN_DRY_RUN, /* Expect disconnect after resync handshake. */
CREATE_BARRIER, /* next P_DATA is preceded by a P_BARRIER */
STATE_SENT, /* Do not change state/UUIDs while this is set */
CALLBACK_PENDING, /* Whether we have a call_usermodehelper(, UMH_WAIT_PROC) *pending,fromdrbdworkercontext.
*/
DISCONNECT_SENT,
DEVICE_WORK_PENDING, /* tell worker that some device has pending work */
};
unsigned susp:1; /* IO suspended by user */ unsigned susp_nod:1; /* IO suspended because no data */ unsigned susp_fen:1; /* IO suspended because fence peer handler runs */
struct drbd_connection { struct list_head connections; struct drbd_resource *resource; #ifdef CONFIG_DEBUG_FS struct dentry *debugfs_conn; struct dentry *debugfs_conn_callback_history; struct dentry *debugfs_conn_oldest_requests; #endif struct kref kref; struct idr peer_devices; /* volume number to peer device mapping */ enum drbd_conns cstate; /* Only C_STANDALONE to C_WF_REPORT_PARAMS */ struct mutex cstate_mutex; /* Protects graceful disconnects */ unsignedint connect_cnt; /* Inc each time a connection is established */
unsignedlong flags; struct net_conf *net_conf; /* content protected by rcu */
wait_queue_head_t ping_wait; /* Woken upon reception of a ping, and a state change */
struct sockaddr_storage my_addr; int my_addr_len; struct sockaddr_storage peer_addr; int peer_addr_len;
struct drbd_socket data; /* data/barrier/cstate/parameter packets */ struct drbd_socket meta; /* ping/ack (metadata) packets */ int agreed_pro_version; /* actually used protocol version */
u32 agreed_features; unsignedlong last_received; /* in jiffies, either socket */ unsignedint ko_count;
struct list_head transfer_log; /* all requests not yet fully processed */
struct crypto_shash *cram_hmac_tfm; struct crypto_shash *integrity_tfm; /* checksums we compute, updates protected by connection->data->mutex */ struct crypto_shash *peer_integrity_tfm; /* checksums we verify, only accessed from receiver thread */ struct crypto_shash *csums_tfm; struct crypto_shash *verify_tfm; void *int_dig_in; void *int_dig_vv;
/* receiver side */ struct drbd_epoch *current_epoch;
spinlock_t epoch_lock; unsignedint epochs;
atomic_t current_tle_nr; /* transfer log epoch number */ unsigned current_tle_writes; /* writes seen within this tl epoch */
unsignedlong last_reconnect_jif; /* empty member on older kernels without blk_start_plug() */ struct blk_plug receiver_plug; struct drbd_thread receiver; struct drbd_thread worker; struct drbd_thread ack_receiver; struct workqueue_struct *ack_sender;
/* whether this sender thread
* has processed a single write yet. */ bool seen_any_write_yet;
/* Which barrier number to send with the next P_BARRIER */ int current_epoch_nr;
/* how many write requests have been sent *withreq->epoch==current_epoch_nr.
* If none, no P_BARRIER will be sent. */ unsigned current_epoch_writes;
} send;
};
/* Used after attach while negotiating new disk state. */ union drbd_state new_state_tmp;
union drbd_dev_state state;
wait_queue_head_t misc_wait;
wait_queue_head_t state_wait; /* upon each state change. */ unsignedint send_cnt; unsignedint recv_cnt; unsignedint read_cnt; unsignedint writ_cnt; unsignedint al_writ_cnt; unsignedint bm_writ_cnt;
atomic_t ap_bio_cnt; /* Requests we need to complete */
atomic_t ap_actlog_cnt; /* Requests waiting for activity log */
atomic_t ap_pending_cnt; /* AP data packets on the wire, ack expected */
atomic_t rs_pending_cnt; /* RS request/data packets on the wire */
atomic_t unacked_cnt; /* Need to send replies for */
atomic_t local_cnt; /* Waiting for local completion */
atomic_t suspend_cnt;
/* Interval tree of pending local requests */ struct rb_root read_requests; struct rb_root write_requests;
/* for statistics and timeouts */ /* [0] read, [1] write */ struct list_head pending_master_completion[2]; struct list_head pending_completion[2];
/* use checksums for *this* resync */ bool use_csums; /* blocks to resync in this run [unit BM_BLOCK_SIZE] */ unsignedlong rs_total; /* number of resync blocks that failed in this run */ unsignedlong rs_failed; /* Syncer's start time [unit jiffies] */ unsignedlong rs_start; /* cumulated time in PausedSyncX state [unit jiffies] */ unsignedlong rs_paused; /* skipped because csum was equal [unit BM_BLOCK_SIZE] */ unsignedlong rs_same_csum; #define DRBD_SYNC_MARKS 8 #define DRBD_SYNC_MARK_STEP (3*HZ) /* block not up-to-date at mark [unit BM_BLOCK_SIZE] */ unsignedlong rs_mark_left[DRBD_SYNC_MARKS]; /* marks's time [unit jiffies] */ unsignedlong rs_mark_time[DRBD_SYNC_MARKS]; /* current index into rs_mark_{left,time} */ int rs_last_mark; unsignedlong rs_last_bcast; /* [unit jiffies] */
/* where does the admin want us to start? (sector) */
sector_t ov_start_sector;
sector_t ov_stop_sector; /* where are we now? (sector) */
sector_t ov_position; /* Start sector of out of sync range (to merge printk reporting). */
sector_t ov_last_oos_start; /* size of out-of-sync range in sectors. */
sector_t ov_last_oos_size; unsignedlong ov_left; /* in bits */
struct drbd_bitmap *bitmap; unsignedlong bm_resync_fo; /* bit offset for drbd_bm_find_next */
/* Used to track operations of resync... */ struct lru_cache *resync; /* Number of locked elements in resync LRU */ unsignedint resync_locked; /* resync extent number waiting for application requests */ unsignedint resync_wenr;
int open_cnt;
u64 *p_uuid;
struct list_head active_ee; /* IO in progress (P_DATA gets written to disk) */ struct list_head sync_ee; /* IO in progress (P_RS_DATA_REPLY gets written to disk) */ struct list_head done_ee; /* need to send P_WRITE_ACK */ struct list_head read_ee; /* [RS]P_DATA_REQUEST being read */
struct list_head resync_reads;
atomic_t pp_in_use; /* allocated from page pool */
atomic_t pp_in_use_by_net; /* sendpage()d, still referenced by tcp */
wait_queue_head_t ee_wait; struct drbd_md_io md_io;
spinlock_t al_lock;
wait_queue_head_t al_wait; struct lru_cache *act_log; /* activity log */ unsignedint al_tr_number; int al_tr_cycle;
wait_queue_head_t seq_wait;
atomic_t packet_seq; unsignedint peer_seq;
spinlock_t peer_seq_lock; unsignedlong comm_bm_set; /* communicated number of set bits. */ struct bm_io_work bm_io_work;
u64 ed_uuid; /* UUID of the exposed data */ struct mutex own_state_mutex; struct mutex *state_mutex; /* either own_state_mutex or first_peer_device(device)->connection->cstate_mutex */ char congestion_reason; /* Why we where congested... */
atomic_t rs_sect_in; /* for incoming resync data rate, SyncTarget */
atomic_t rs_sect_ev; /* for submitted resync data rate, both */ int rs_last_sect_ev; /* counter to compare with */ int rs_last_events; /* counter of read or write "events" (unit sectors)
* on the lower level device when we last looked. */ int c_sync_rate; /* current resync rate after syncer throttle magic */ struct fifo_buffer *rs_plan_s; /* correction values of resync planer (RCU, connection->conn_update) */ int rs_in_flight; /* resync sectors in flight (to proxy, in proxy and from proxy) */
atomic_t ap_in_flight; /* App sectors in flight (waiting for ack) */ unsignedint peer_max_bio_size; unsignedint local_max_bio_size;
/* any requests that would block in drbd_make_request()
* are deferred to this single-threaded work queue */ struct submit_worker submit;
};
/* Our old fixed size meta data layout *allowsuptoabout3.8TB,soifyouwantmore,
* you need to use the "flexible" meta data format. */ #define MD_128MB_SECT (128LLU << 11) /* 128 MB, unit sectors */ #define MD_4kB_SECT 8 #define MD_32kB_SECT 64
/* One activity log extent represents 4M of storage */ #define AL_EXTENT_SHIFT 22 #define AL_EXTENT_SIZE (1<<AL_EXTENT_SHIFT)
/* We could make these currently hardcoded constants configurable *variablesatcreate-mdtime(orevenre-configurableatruntime?). *WhichwillrequiresomemorechangestotheDRBD"superblock" *andattachcode. * *updatespertransaction: *Thismanychangestotheactivesetcanbeloggedwithonetransaction. *Thisnumberisarbitrary. *contextpertransaction: *Thismanycontextextentnumbersareloggedwitheachtransaction. *Thisnumberisresultingfromthetransactionblocksize(4k),thelayout *ofthetransactionheader,andthenumberofupdatespertransaction. *Seedrbd_actlog.c:structal_transaction_on_disk
* */ #define AL_UPDATES_PER_TRANSACTION 64// arbitrary #define AL_CONTEXT_PER_TRANSACTION 919// (4096 - 36 - 6*64)/4
/* resync bitmap */ /* 16MB sized 'bitmap extent' to track syncer usage */ struct bm_extent { int rs_left; /* number of bits set (out of sync) in this extent. */ int rs_failed; /* number of failed resync requests in this extent. */ unsignedlong flags; struct lc_element lce;
};
#define BME_NO_WRITES 0/* bm_extent.flags: no more requests on this one! */ #define BME_LOCKED 1/* bm_extent.flags: syncer active on this one. */ #define BME_PRIORITY 2/* finish resync IO on this extent ASAP! App IO waiting! */
/* We do bitmap IO in units of 4k blocks.
* We also still have a hardcoded 4k per bit relation. */ #define BM_BLOCK_SHIFT 12/* 4k per bit */ #define BM_BLOCK_SIZE (1<<BM_BLOCK_SHIFT) /* mostly arbitrarily set the represented size of one bitmap extent, *akaresyncextent,to16MiB(whichisalso512Byteworthofbitmap
* at 4k per bit resolution) */ #define BM_EXT_SHIFT 24/* 16 MiB per resync extent */ #define BM_EXT_SIZE (1<<BM_EXT_SHIFT)
#if (BM_EXT_SHIFT != 24) || (BM_BLOCK_SHIFT != 12) #error"HAVE YOU FIXED drbdmeta AS WELL??" #endif
/* thus many _storage_ sectors are described by one bit */ #define BM_SECT_TO_BIT(x) ((x)>>(BM_BLOCK_SHIFT-9)) #define BM_BIT_TO_SECT(x) ((sector_t)(x)<<(BM_BLOCK_SHIFT-9)) #define BM_SECT_PER_BIT BM_BIT_TO_SECT(1)
/* bit to represented kilo byte conversion */ #define Bit2KB(bits) ((bits)<<(BM_BLOCK_SHIFT-10))
/* in which _bitmap_ extent (resp. sector) the bit for a certain
* _storage_ sector is located in */ #define BM_SECT_TO_EXT(x) ((x)>>(BM_EXT_SHIFT-9)) #define BM_BIT_TO_EXT(x) ((x) >> (BM_EXT_SHIFT - BM_BLOCK_SHIFT))
/* first storage sector a bitmap extent corresponds to */ #define BM_EXT_TO_SECT(x) ((sector_t)(x) << (BM_EXT_SHIFT-9)) /* how much _storage_ sectors we have per bitmap extent */ #define BM_SECT_PER_EXT BM_EXT_TO_SECT(1) /* how many bits are covered by one bitmap extent (resync extent) */ #define BM_BITS_PER_EXT (1UL << (BM_EXT_SHIFT - BM_BLOCK_SHIFT))
/* in one sector of the bitmap, we have this many activity_log extents. */ #define AL_EXT_PER_BM_SECT (1 << (BM_EXT_SHIFT - AL_EXTENT_SHIFT))
/* the extent in "PER_EXTENT" below is an activity log extent *weneedthatmany(longwords/bytes)tostorethebitmap *ofoneAL_EXTENT_SIZEchunkofstorage. *wecanstorethebitmapforthatmanyAL_EXTENTSwithin *onesectorofthe_on_disk_bitmap: *bit0bit37bit38bit(512*8)-1 *...|........|........|..// ..|........| *sect.0`296`304^(512*8*8)-1 * #defineBM_WORDS_PER_EXT((AL_EXT_SIZE/BM_BLOCK_SIZE)/BITS_PER_LONG) #defineBM_BYTES_PER_EXT((AL_EXT_SIZE/BM_BLOCK_SIZE)/8)// 128 #defineBM_EXT_PER_SECT(512/BM_BYTES_PER_EXTENT)// 4
*/
#define DRBD_MAX_SECTORS_32 (0xffffffffLU) /* we have a certain meta data variant that has a fixed on-disk size of 128 *MiB,ofwhich4kareour"superblock",and32karethefixedsizeactivity *log,leavingthismanysectorsforthebitmap.
*/
#define DRBD_MAX_SECTORS_FIXED_BM \
((MD_128MB_SECT - MD_32kB_SECT - MD_4kB_SECT) * (1LL<<(BM_EXT_SHIFT-9))) #define DRBD_MAX_SECTORS DRBD_MAX_SECTORS_FIXED_BM /* 16 TB in units of sectors */ #if BITS_PER_LONG == 32 /* adjust by one page worth of bitmap, *sowewon'twraparoundindrbd_bm_find_next_bit.
* you should use 64bit OS for that much storage, anyways. */ #define DRBD_MAX_SECTORS_FLEX BM_BIT_TO_SECT(0xffff7fff) #else /* we allow up to 1 PiB now on 64bit architecture with "flexible" meta data */ #define DRBD_MAX_SECTORS_FLEX (1UL << 51) /* corresponds to (1UL << 38) bits right now. */ #endif
/* Estimate max bio size as 256 * PAGE_SIZE, *sofortypicalPAGE_SIZEof4k,thatis(1<<20)Byte. *Sincewemayliveinamixed-platformcluster, *welimitustoaplatformagnosticconstantherefornow. *AfollowupcommitmayallowevenbiggerBIOsizes,
* once we thought that through. */ #define DRBD_MAX_BIO_SIZE (1U << 20) #if DRBD_MAX_BIO_SIZE > (BIO_MAX_VECS << PAGE_SHIFT) #error Architecture not supported: DRBD_MAX_BIO_SIZE > BIO_MAX_SIZE #endif #define DRBD_MAX_BIO_SIZE_SAFE (1U << 12) /* Works always = 4k */
#define DRBD_MAX_SIZE_H80_PACKET (1U << 15) /* Header 80 only allows packets up to 32KiB data */ #define DRBD_MAX_BIO_SIZE_P95 (1U << 17) /* Protocol 95 to 99 allows bios up to 128KiB */
/* For now, don't allow more than half of what we can "activate" in one *activitylogtransactiontobediscardedinonego.Wemayneedtorework
* drbd_al_begin_io() to allow for even larger discard ranges */ #define DRBD_MAX_BATCH_BIO_SIZE (AL_UPDATES_PER_TRANSACTION/2*AL_EXTENT_SIZE) #define DRBD_MAX_BBIO_SECTORS (DRBD_MAX_BATCH_BIO_SIZE >> 9)
externint drbd_bm_init(struct drbd_device *device); externint drbd_bm_resize(struct drbd_device *device, sector_t sectors, int set_new_bits); externvoid drbd_bm_cleanup(struct drbd_device *device); externvoid drbd_bm_set_all(struct drbd_device *device); externvoid drbd_bm_clear_all(struct drbd_device *device); /* set/clear/test only a few bits at a time */ externint drbd_bm_set_bits( struct drbd_device *device, unsignedlong s, unsignedlong e); externint drbd_bm_clear_bits( struct drbd_device *device, unsignedlong s, unsignedlong e); externint drbd_bm_count_bits( struct drbd_device *device, constunsignedlong s, constunsignedlong e); /* bm_set_bits variant for use while holding drbd_bm_lock,
* may process the whole bitmap in one go */ externvoid _drbd_bm_set_bits(struct drbd_device *device, constunsignedlong s, constunsignedlong e); externint drbd_bm_test_bit(struct drbd_device *device, unsignedlong bitnr); externint drbd_bm_e_weight(struct drbd_device *device, unsignedlong enr); externint drbd_bm_read(struct drbd_device *device, struct drbd_peer_device *peer_device) __must_hold(local); externvoid drbd_bm_mark_for_writeout(struct drbd_device *device, int page_nr); externint drbd_bm_write(struct drbd_device *device, struct drbd_peer_device *peer_device) __must_hold(local); externvoid drbd_bm_reset_al_hints(struct drbd_device *device) __must_hold(local); externint drbd_bm_write_hinted(struct drbd_device *device) __must_hold(local); externint drbd_bm_write_lazy(struct drbd_device *device, unsigned upper_idx) __must_hold(local); externint drbd_bm_write_all(struct drbd_device *device, struct drbd_peer_device *peer_device) __must_hold(local); externint drbd_bm_write_copy_pages(struct drbd_device *device, struct drbd_peer_device *peer_device) __must_hold(local); extern size_t drbd_bm_words(struct drbd_device *device); externunsignedlong drbd_bm_bits(struct drbd_device *device); extern sector_t drbd_bm_capacity(struct drbd_device *device);
#define DRBD_END_OF_BITMAP (~(unsignedlong)0) externunsignedlong drbd_bm_find_next(struct drbd_device *device, unsignedlong bm_fo); /* bm_find_next variants for use while you hold drbd_bm_lock() */ externunsignedlong _drbd_bm_find_next(struct drbd_device *device, unsignedlong bm_fo); externunsignedlong _drbd_bm_find_next_zero(struct drbd_device *device, unsignedlong bm_fo); externunsignedlong _drbd_bm_total_weight(struct drbd_device *device); externunsignedlong drbd_bm_total_weight(struct drbd_device *device); /* for receive_bitmap */ externvoid drbd_bm_merge_lel(struct drbd_device *device, size_t offset,
size_t number, unsignedlong *buffer); /* for _drbd_send_bitmap */ externvoid drbd_bm_get_lel(struct drbd_device *device, size_t offset,
size_t number, unsignedlong *buffer);
/* We also need a standard (emergency-reserve backed) page pool *formetadataIO(activitylog,bitmap). *Wecankeepitglobal,aslongasitisusedas"Npagesatatime". *128shouldbeplenty,currentlyweprobablycangetawaywithasfewas1.
*/ #define DRBD_MIN_POOL_PAGES 128 extern mempool_t drbd_md_io_page_pool; extern mempool_t drbd_buffer_page_pool;
/* We also need to make sure we get a bio
* when we need it for housekeeping purposes */ externstruct bio_set drbd_md_io_bio_set;
/* And a bio_set for cloning */ externstruct bio_set drbd_io_bio_set;
/* Returns the number of 512 byte sectors of the device */ staticinline sector_t drbd_get_capacity(struct block_device *bdev)
{ return bdev ? bdev_nr_sectors(bdev) : 0;
}
switch (bdev->md.meta_dev_idx) { case DRBD_MD_INDEX_INTERNAL: case DRBD_MD_INDEX_FLEX_INT:
s = drbd_get_capacity(bdev->backing_bdev)
? min_t(sector_t, DRBD_MAX_SECTORS_FLEX,
drbd_md_first_sector(bdev))
: 0; break; case DRBD_MD_INDEX_FLEX_EXT:
s = min_t(sector_t, DRBD_MAX_SECTORS_FLEX,
drbd_get_capacity(bdev->backing_bdev)); /* clip at maximum size the meta device can support */
s = min_t(sector_t, s,
BM_EXT_TO_SECT(bdev->md.md_size_sect
- bdev->md.bm_offset)); break; default:
s = min_t(sector_t, DRBD_MAX_SECTORS,
drbd_get_capacity(bdev->backing_bdev));
} return s;
}
if (meta_dev_idx == DRBD_MD_INDEX_FLEX_EXT) return0;
/* Since drbd08, internal meta data is always "flexible".
* position: last 4k aligned block of 4k size */ if (meta_dev_idx == DRBD_MD_INDEX_INTERNAL ||
meta_dev_idx == DRBD_MD_INDEX_FLEX_INT) return (drbd_get_capacity(bdev->backing_bdev) & ~7ULL) - 8;
/* external, some index; this is the old fixed size layout */ return MD_128MB_SECT * bdev->md.meta_dev_idx;
}
/* To get the ack_receiver out of the blocking network stack, *soitcanchangeitssk_rcvtimeofromidle-toping-timeout, *andsendaping,weneedtosendasignal.
* Which signal we send is irrelevant. */ staticinlinevoid wake_ack_receiver(struct drbd_connection *connection)
{ struct task_struct *task = connection->ack_receiver.task; if (task && get_t_state(&connection->ack_receiver) == RUNNING)
send_sig(SIGXCPU, task, 1);
}
if (ap_pending_cnt == 0)
wake_up(&device->misc_wait); return ap_pending_cnt;
}
/* counts how many resync-related answers we still expect from the peer *increasedecrease *C_SYNC_TARGETsendsP_RS_DATA_REQUEST(andexpectsP_RS_DATA_REPLY) *C_SYNC_SOURCEsendsP_RS_DATA_REPLY(andexpectsP_WRITE_ACKwithID_SYNCER) *(orP_NEG_ACKwithID_SYNCER)
*/ staticinlinevoid inc_rs_pending(struct drbd_peer_device *peer_device)
{
atomic_inc(&peer_device->device->rs_pending_cnt);
}
/* counts how many answers we still need to send to the peer. *increasedon *receive_DataunlessprotocolA; *weneedtosendaP_RECV_ACK(protoB) *orP_WRITE_ACK(protoC) *receive_RSDataReply(recv_resync_read)weneedtosendaP_WRITE_ACK *receive_DataRequest(receive_RSDataRequest)weneedtosendbackP_DATA *receive_Barrier_*weneedtosendaP_BARRIER_ACK
*/ staticinlinevoid inc_unacked(struct drbd_device *device)
{
atomic_inc(&device->unacked_cnt);
}
staticinlinevoid put_ldev(struct drbd_device *device)
{ enum drbd_disk_state disk_state = device->state.disk; /* We must check the state *before* the atomic_dec becomes visible, *orwehaveatheoreticalracewheresomeonehittingzero, *whilestatestillD_FAILED,willthenseeD_DISKLESSinthe
* condition below and calling into destroy, where he must not, yet. */ int i = atomic_dec_return(&device->local_cnt);
/* This may be called from some endio handler,
* so we must not sleep here. */
__release(local);
D_ASSERT(device, i >= 0); if (i == 0) { if (disk_state == D_DISKLESS) /* even internal references gone, safe to destroy */
drbd_device_post_work(device, DESTROY_DISK); if (disk_state == D_FAILED) /* all application IO references gone. */ if (!test_and_set_bit(GOING_DISKLESS, &device->flags))
drbd_device_post_work(device, GO_DISKLESS);
wake_up(&device->misc_wait);
}
}
/* this throttles on-the-fly application requests *accordingtomax_bufferssettings;
* maybe re-implement using semaphores? */ staticinlineint drbd_get_max_buffers(struct drbd_device *device)
{ struct net_conf *nc; int mxb;
rcu_read_lock();
nc = rcu_dereference(first_peer_device(device)->connection->net_conf);
mxb = nc ? nc->max_buffers : 1000000; /* arbitrary limit on open requests */
rcu_read_unlock();
return mxb;
}
staticinlineint drbd_state_is_stable(struct drbd_device *device)
{ union drbd_dev_state s = device->state;
/* DO NOT add a default clause, we want the compiler to warn us
* for any newly introduced state we may have forgotten to add here */
switch ((enum drbd_conns)s.conn) { /* new io only accepted when there is no connection, ... */ case C_STANDALONE: case C_WF_CONNECTION: /* ... or there is a well established connection. */ case C_CONNECTED: case C_SYNC_SOURCE: case C_SYNC_TARGET: case C_VERIFY_S: case C_VERIFY_T: case C_PAUSED_SYNC_S: case C_PAUSED_SYNC_T: case C_AHEAD: case C_BEHIND: /* transitional states, IO allowed */ case C_DISCONNECTING: case C_UNCONNECTED: case C_TIMEOUT: case C_BROKEN_PIPE: case C_NETWORK_FAILURE: case C_PROTOCOL_ERROR: case C_TEAR_DOWN: case C_WF_REPORT_PARAMS: case C_STARTING_SYNC_S: case C_STARTING_SYNC_T: break;
/* Allow IO in BM exchange states with new protocols */ case C_WF_BITMAP_S: if (first_peer_device(device)->connection->agreed_pro_version < 96) return0; break;
/* no new io accepted in these states */ case C_WF_BITMAP_T: case C_WF_SYNC_UUID: case C_MASK: /* not "stable" */ return0;
}
switch ((enum drbd_disk_state)s.disk) { case D_DISKLESS: case D_INCONSISTENT: case D_OUTDATED: case D_CONSISTENT: case D_UP_TO_DATE: case D_FAILED: /* disk state is stable as well. */ break;
/* no new io accepted during transitional states */ case D_ATTACHING: case D_NEGOTIATING: case D_UNKNOWN: case D_MASK: /* not "stable" */ return0;
}
staticinlinebool may_inc_ap_bio(struct drbd_device *device)
{ int mxb = drbd_get_max_buffers(device);
if (drbd_suspended(device)) returnfalse; if (atomic_read(&device->suspend_cnt)) returnfalse;
/* to avoid potential deadlock or bitmap corruption, *invariousplaces,weonlyallownewapplicationio
* to start during "stable" states. */
/* no new io accepted when attaching or detaching the disk */ if (!drbd_state_is_stable(device)) returnfalse;
/* since some older kernels don't have atomic_add_unless,
* and we are within the spinlock anyways, we have this workaround. */ if (atomic_read(&device->ap_bio_cnt) > mxb) returnfalse; if (test_bit(BITMAP_IO, &device->flags)) returnfalse; returntrue;
}
spin_lock_irq(&device->resource->req_lock);
rv = may_inc_ap_bio(device); if (rv)
atomic_inc(&device->ap_bio_cnt);
spin_unlock_irq(&device->resource->req_lock);
return rv;
}
staticinlinevoid inc_ap_bio(struct drbd_device *device)
{ /* we wait here *aslongasthedeviceissuspended *untilthebitmapisnolongerontheflyduringconnection *handshakeaslongaswewouldexceedthemax_bufferlimit. * *toavoidraceswiththereconnectcode,
* we need to atomic_inc within the spinlock. */
staticinlinevoid dec_ap_bio(struct drbd_device *device)
{ int mxb = drbd_get_max_buffers(device); int ap_bio = atomic_dec_return(&device->ap_bio_cnt);
D_ASSERT(device, ap_bio >= 0);
if (ap_bio == 0 && test_bit(BITMAP_IO, &device->flags)) { if (!test_and_set_bit(BITMAP_IO_QUEUED, &device->flags))
drbd_queue_work(&first_peer_device(device)->
connection->sender_work,
&device->bm_io_work.w);
}
/* this currently does wake_up for every dec_ap_bio! *mayberatherintroducesometypeofhysteresis?
* e.g. (ap_bio == mxb/2 || ap_bio == 0) ? */ if (ap_bio < mxb)
wake_up(&device->misc_wait);
}
staticinlineint drbd_queue_order_type(struct drbd_device *device)
{ /* sorry, we currently have no working implementation
* of distributed TCQ stuff */ #ifndef QUEUE_ORDERED_NONE #define QUEUE_ORDERED_NONE 0 #endif return QUEUE_ORDERED_NONE;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.