/* used for synchronous meta data and bitmap IO *submittedbydrbd_md_sync_page_io()
*/ void drbd_md_endio(struct bio *bio)
{ struct drbd_device *device;
/* special case: drbd_md_read() during drbd_adm_attach() */ if (device->ldev)
put_ldev(device);
bio_put(bio);
/* We grabbed an extra reference in _drbd_md_sync_page_io() to be able *totimeoutonthelowerleveldevice,andeventuallydetachfromit. *Ifthisiocompletionrunsafterthattimeoutexpired,this *drbd_md_put_buffer()mayallowustofinallytryandre-attach. *Duringnormaloperation,thisonlyputsthatextrareference *downto1again. *Makesurewefirstdropthereference,andonlythensignal *completion,orwemay(indrbd_al_read_log())cyclesofastintothe *nextdrbd_md_sync_page_io(),thatwetriggerthe *ASSERT(atomic_read(&device->md_io_in_use)==1)there.
*/
drbd_md_put_buffer(device);
device->md_io.done = 1;
wake_up(&device->misc_wait);
}
/* reads on behalf of the partner, *"submitted"bythereceiver
*/ staticvoid drbd_endio_read_sec_final(struct drbd_peer_request *peer_req) __releases(local)
{ unsignedlong flags = 0; struct drbd_peer_device *peer_device = peer_req->peer_device; struct drbd_device *device = peer_device->device;
spin_lock_irqsave(&device->resource->req_lock, flags);
device->read_cnt += peer_req->i.size >> 9;
list_del(&peer_req->w.list); if (list_empty(&device->read_ee))
wake_up(&device->ee_wait); if (test_bit(__EE_WAS_ERROR, &peer_req->flags))
__drbd_chk_io_error(device, DRBD_READ_ERROR);
spin_unlock_irqrestore(&device->resource->req_lock, flags);
/* writes on behalf of the partner, or resync writes,
* "submitted" by the receiver, final stage. */ void drbd_endio_write_sec_final(struct drbd_peer_request *peer_req) __releases(local)
{ unsignedlong flags = 0; struct drbd_peer_device *peer_device = peer_req->peer_device; struct drbd_device *device = peer_device->device; struct drbd_connection *connection = peer_device->connection; struct drbd_interval i; int do_wake;
u64 block_id; int do_al_complete_io;
/* after we moved peer_req to done_ee, *wemaynolongeraccessit, *itmaybefreed/reusedalready!
* (as soon as we release the req_lock) */
i = peer_req->i;
do_al_complete_io = peer_req->flags & EE_CALL_AL_COMPLETE_IO;
block_id = peer_req->block_id;
peer_req->flags &= ~EE_CALL_AL_COMPLETE_IO;
if (peer_req->flags & EE_WAS_ERROR) { /* In protocol != C, we usually do not send write acks.
* In case of a write error, send the neg ack anyways. */ if (!__test_and_set_bit(__EE_SEND_WRITE_ACK, &peer_req->flags))
inc_unacked(device);
drbd_set_out_of_sync(peer_device, peer_req->i.sector, peer_req->i.size);
}
/* FIXME do we want to detach for failed REQ_OP_DISCARD?
* ((peer_req->flags & (EE_WAS_ERROR|EE_TRIM)) == EE_WAS_ERROR) */ if (peer_req->flags & EE_WAS_ERROR)
__drbd_chk_io_error(device, DRBD_WRITE_ERROR);
if (connection->cstate >= C_WF_REPORT_PARAMS) {
kref_get(&device->kref); /* put is in drbd_send_acks_wf() */ if (!queue_work(connection->ack_sender, &peer_device->send_acks_work))
kref_put(&device->kref, drbd_destroy_device);
}
spin_unlock_irqrestore(&device->resource->req_lock, flags);
if (block_id == ID_SYNCER)
drbd_rs_complete_io(device, i.sector);
if (do_wake)
wake_up(&device->ee_wait);
if (do_al_complete_io)
drbd_al_complete_io(device, &i);
put_ldev(device);
}
/* writes on behalf of the partner, or resync writes, *"submitted"bythereceiver.
*/ void drbd_peer_request_endio(struct bio *bio)
{ struct drbd_peer_request *peer_req = bio->bi_private; struct drbd_device *device = peer_req->peer_device->device; bool is_write = bio_data_dir(bio) == WRITE; bool is_discard = bio_op(bio) == REQ_OP_WRITE_ZEROES ||
bio_op(bio) == REQ_OP_DISCARD;
if (bio->bi_status)
set_bit(__EE_WAS_ERROR, &peer_req->flags);
bio_put(bio); /* no need for the bio anymore */ if (atomic_dec_and_test(&peer_req->pending_bios)) { if (is_write)
drbd_endio_write_sec_final(peer_req); else
drbd_endio_read_sec_final(peer_req);
}
}
staticvoid
drbd_panic_after_delayed_completion_of_aborted_request(struct drbd_device *device)
{
panic("drbd%u %s/%u potential random memory corruption caused by delayed completion of aborted local request\n",
device->minor, device->resource->name, device->vnr);
}
/* read, readA or write requests on R_PRIMARY coming from drbd_make_request
*/ void drbd_request_endio(struct bio *bio)
{ unsignedlong flags; struct drbd_request *req = bio->bi_private; struct drbd_device *device = req->device; struct bio_and_error m; enum drbd_req_event what;
/* If this request was aborted locally before, *butnowwascompleted"successfully", *chancesarethatthiscausedarbitrarydatacorruption. * *"aborting"requests,orforce-detachingthedisk,isintendedfor *completelyblocked/hunglocalbackingdeviceswhichdonolonger *completerequestsatall,notevendoerrorcompletions.Inthis *situation,usuallyahard-resetandfailoveristheonlywayout. * *By"aborting",basicallyfakingalocalerror-completion, *weallowforamoregracefulswichoverbycleanlymigratingservices. *Stilltheaffectednodehastoberebooted"soon". * *Bycompletingtheserequests,weallowtheupperlayerstore-use *theassociateddatapages. * *Iflaterthelocalbackingdevice"recovers",andnowDMAssomedata *fromdiskintotheoriginalrequestpages,inthebestcaseitwill *justputrandomdataintounusedpages;buttypicallyitwillcorrupt *meanwhilecompletelyunrelateddata,causingallsortsofdamage. * *Whichmeansdelayedsuccessfulcompletion, *especiallyforREADrequests, *isareasontopanic(). * *Weassumethatadelayed*error*completionisOK, *thoughwestillwillcomplainnoisilyaboutit.
*/ if (unlikely(req->rq_state & RQ_LOCAL_ABORTED)) { if (drbd_ratelimit())
drbd_emerg(device, "delayed completion of aborted local request; disk-timeout may be too aggressive\n");
if (!bio->bi_status)
drbd_panic_after_delayed_completion_of_aborted_request(device);
}
/* to avoid recursion in __req_mod */ if (unlikely(bio->bi_status)) { switch (bio_op(bio)) { case REQ_OP_WRITE_ZEROES: case REQ_OP_DISCARD: if (bio->bi_status == BLK_STS_NOTSUPP)
what = DISCARD_COMPLETED_NOTSUPP; else
what = DISCARD_COMPLETED_WITH_ERROR; break; case REQ_OP_READ: if (bio->bi_opf & REQ_RAHEAD)
what = READ_AHEAD_COMPLETED_WITH_ERROR; else
what = READ_COMPLETED_WITH_ERROR; break; default:
what = WRITE_COMPLETED_WITH_ERROR; break;
}
} else {
what = COMPLETED_OK;
}
src = kmap_atomic(page); while ((tmp = page_chain_next(page))) { /* all but the last page will be fully used */
crypto_shash_update(desc, src, PAGE_SIZE);
kunmap_atomic(src);
page = tmp;
src = kmap_atomic(page);
} /* and now the last, possibly only partially used page */
len = peer_req->i.size & (PAGE_SIZE - 1);
crypto_shash_update(desc, src, len ?: PAGE_SIZE);
kunmap_atomic(src);
/* GFP_TRY, because if there is no memory available right now, this may
* be rescheduled for later. It is "only" background resync, after all. */
peer_req = drbd_alloc_peer_req(peer_device, ID_SYNCER /* unused */, sector,
size, size, GFP_TRY); if (!peer_req) goto defer;
atomic_add(size >> 9, &device->rs_sect_ev); if (drbd_submit_peer_request(peer_req) == 0) return0;
/* If it failed because of ENOMEM, retry should help. If it failed *becausebio_add_pagefailed(probablybrokenlowerleveldriver), *retrymayormaynothelp.
* If it does not, you may need to force disconnect. */
spin_lock_irq(&device->resource->req_lock);
list_del(&peer_req->w.list);
spin_unlock_irq(&device->resource->req_lock);
staticint drbd_rs_controller(struct drbd_peer_device *peer_device, unsignedint sect_in)
{ struct drbd_device *device = peer_device->device; struct disk_conf *dc; unsignedint want; /* The number of sectors we want in-flight */ int req_sect; /* Number of sectors to request in this turn */ int correction; /* Number of sectors more we need in-flight */ int cps; /* correction per invocation of drbd_rs_controller() */ int steps; /* Number of time steps to plan ahead */ int curr_corr; int max_sect; struct fifo_buffer *plan;
dc = rcu_dereference(device->ldev->disk_conf);
plan = rcu_dereference(device->rs_plan_s);
staticint drbd_rs_number_requests(struct drbd_peer_device *peer_device)
{ struct drbd_device *device = peer_device->device; unsignedint sect_in; /* Number of sectors that came in since the last turn */ int number, mxb;
/* Don't have more than "max-buffers"/2 in-flight. *Otherwisewemaycausetheremotesitetostallondrbd_alloc_pages(), *potentiallycausingadistributeddeadlockoncongestionduring *online-verifyor(checksum-based)resync,ifmax-buffers,
* socket buffer sizes and resync rate settings are mis-configured. */
/* note that "number" is in units of "BM_BLOCK_SIZE" (which is 4k), *mxb(asusedhere,andindrbd_alloc_pagesonthepeer)is *"numberofpages"(typicallyalso4k),
* but "rs_in_flight" is in "sectors" (512 Byte). */ if (mxb - device->rs_in_flight/8 < number)
number = mxb - device->rs_in_flight/8;
return number;
}
staticint make_resync_request(struct drbd_peer_device *const peer_device, int cancel)
{ struct drbd_device *const device = peer_device->device; struct drbd_connection *const connection = peer_device ? peer_device->connection : NULL; unsignedlong bit;
sector_t sector; const sector_t capacity = get_capacity(device->vdisk); int max_bio_size; int number, rollback_i, size; int align, requeue = 0; int i = 0; int discard_granularity = 0;
if (!get_ldev(device)) { /* Since we only need to access device->rsync a get_ldev_if_state(device,D_FAILED)wouldbesufficient,but tocontinueresyncwithabrokendiskmakesnosenseat
all */
drbd_err(device, "Disk broke down during resync!\n"); return0;
}
max_bio_size = queue_max_hw_sectors(device->rq_queue) << 9;
number = drbd_rs_number_requests(peer_device); if (number <= 0) goto requeue;
for (i = 0; i < number; i++) { /* Stop generating RS requests when half of the send buffer is filled,
* but notify TCP that we'd like to have more space. */
mutex_lock(&connection->data.mutex); if (connection->data.socket) { struct sock *sk = connection->data.socket->sk; int queued = sk->sk_wmem_queued; int sndbuf = sk->sk_sndbuf; if (queued > sndbuf / 2) {
requeue = 1; if (sk->sk_socket)
set_bit(SOCK_NOSPACE, &sk->sk_socket->flags);
}
} else
requeue = 1;
mutex_unlock(&connection->data.mutex); if (requeue) goto requeue;
next_sector:
size = BM_BLOCK_SIZE;
bit = drbd_bm_find_next(device, device->bm_resync_fo);
#if DRBD_MAX_BIO_SIZE > BM_BLOCK_SIZE /* try to find some adjacent bits. *westopifwehavealreadythemaximumreqsize. * *Additionallyalwaysalignbiggerrequests,inorderto *bepreparedforallstripesizesofsoftwareRAIDs.
*/
align = 1;
rollback_i = i; while (i < number) { if (size + BM_BLOCK_SIZE > max_bio_size) break;
/* Be always aligned */ if (sector & ((1<<(align+3))-1)) break;
if (discard_granularity && size == discard_granularity) break;
/* do not cross extent boundaries */ if (((bit+1) & BM_BLOCKS_PER_BM_EXT_MASK) == 0) break; /* now, is it actually dirty, after all? *caution,drbd_bm_test_bitistri-stateforsome *obscurereason;(b==0)wouldgettheout-of-band *onlyaccidentallyrightbecauseofthe"oddlysized"
* adjustment below */ if (drbd_bm_test_bit(device, bit+1) != 1) break;
bit++;
size += BM_BLOCK_SIZE; if ((BM_BLOCK_SIZE << align) <= size)
align++;
i++;
} /* if we merged some,
* reset the offset to start the next drbd_bm_find_next from */ if (size > BM_BLOCK_SIZE)
device->bm_resync_fo = bit + 1; #endif
/* adjust very last sectors, in case we are oddly sized */ if (sector + (size>>9) > capacity)
size = (capacity-sector)<<9;
if (device->use_csums) { switch (read_for_csum(peer_device, sector, size)) { case -EIO: /* Disk failure */
put_ldev(device); return -EIO; case -EAGAIN: /* allocation failed, or ldev busy */
drbd_rs_complete_io(device, sector);
device->bm_resync_fo = BM_SECT_TO_BIT(sector);
i = rollback_i; goto requeue; case0: /* everything ok */ break; default:
BUG();
}
} else { int err;
staticint make_ov_request(struct drbd_peer_device *peer_device, int cancel)
{ struct drbd_device *device = peer_device->device; int number, i, size;
sector_t sector; const sector_t capacity = get_capacity(device->vdisk); bool stop_sector_reached = false;
if (unlikely(cancel)) return1;
number = drbd_rs_number_requests(peer_device);
sector = device->ov_position; for (i = 0; i < number; i++) { if (sector >= capacity) return1;
/* We check for "finished" only in the reply path: *w_e_end_ov_reply().
* We need to send at least one request out. */
stop_sector_reached = i > 0
&& verify_can_do_stop_sector(device)
&& sector >= device->ov_stop_sector; if (stop_sector_reached) break;
size = BM_BLOCK_SIZE;
if (drbd_try_rs_begin_io(peer_device, sector)) {
device->ov_position = sector; goto requeue;
}
if (sector + (size>>9) > capacity)
size = (capacity-sector)<<9;
/* Remove all elements from the resync LRU. Since future actions *mightsetbitsinthe(main)bitmap,thentheentriesinthe
* resync LRU would be wrong. */ if (drbd_rs_del_all(device)) { /* In case this is not possible now, most probably because *thereareP_RS_DATA_REPLYPacketslingeringontheworker's *queue(oreventhereadoperationsforthosepackets
* is not finished by now). Retry in 100ms. */
schedule_timeout_interruptible(HZ / 10);
dw = kmalloc(sizeof(struct drbd_device_work), GFP_ATOMIC); if (dw) {
dw->w.cb = w_resync_finished;
dw->device = device;
drbd_queue_work(&connection->sender_work, &dw->w); return1;
}
drbd_err(device, "Warn failed to drbd_rs_del_all() and to kmalloc(dw).\n");
}
db = device->rs_total; /* adjust for verify start and stop sectors, respective reached position */ if (device->state.conn == C_VERIFY_S || device->state.conn == C_VERIFY_T)
db -= device->ov_left;
dbdt = Bit2KB(db/dt);
device->rs_paused /= HZ;
if (!get_ldev(device)) goto out;
ping_peer(device);
spin_lock_irq(&device->resource->req_lock);
os = drbd_read_state(device);
/* This protects us against multiple calls (that can happen in the presence
of application IO), and against connectivity loss just before we arrive here. */ if (os.conn <= C_CONNECTED) goto out_unlock;
if (os.conn == C_SYNC_TARGET || os.conn == C_PAUSED_SYNC_T) { if (device->p_uuid) { int i; for (i = UI_BITMAP ; i <= UI_HISTORY_END ; i++)
_drbd_uuid_set(device, i, device->p_uuid[i]);
drbd_uuid_set(device, UI_BITMAP, device->ldev->md.uuid[UI_CURRENT]);
_drbd_uuid_set(device, UI_CURRENT, device->p_uuid[UI_CURRENT]);
} else {
drbd_err(device, "device->p_uuid is NULL! BUG\n");
}
}
if (!(os.conn == C_VERIFY_S || os.conn == C_VERIFY_T)) { /* for verify runs, we don't update uuids here,
* so there would be nothing to report. */
drbd_uuid_set_bm(device, 0UL);
drbd_print_uuids(device, "updated UUIDs"); if (device->p_uuid) { /* Now the two UUID sets are equal, update what we
* know of the peer. */ int i; for (i = UI_CURRENT ; i <= UI_HISTORY_END ; i++)
device->p_uuid[i] = device->ldev->md.uuid[i];
}
}
}
/* If we have been sync source, and have an effective fencing-policy,
* once *all* volumes are back in sync, call "unfence". */ if (os.conn == C_SYNC_SOURCE) { enum drbd_disk_state disk_state = D_MASK; enum drbd_disk_state pdsk_state = D_MASK; enum drbd_fencing_p fp = FP_DONT_CARE;
if (unlikely(cancel)) {
drbd_free_peer_req(device, peer_req);
dec_unacked(device); return0;
}
/* after "cancel", because after drbd_disconnect/drbd_rs_cancel_all
* the resync lru has been cleaned up already */ if (get_ldev(device)) {
drbd_rs_complete_io(device, peer_req->i.sector);
put_ldev(device);
}
di = peer_req->digest;
if (likely((peer_req->flags & EE_WAS_ERROR) == 0)) {
digest_size = crypto_shash_digestsize(peer_device->connection->verify_tfm);
digest = kmalloc(digest_size, GFP_NOIO); if (digest) {
drbd_csum_ee(peer_device->connection->verify_tfm, peer_req, digest);
/* let's advance progress step marks only for every other megabyte */ if ((device->ov_left & 0x200) == 0x200)
drbd_advance_rs_marks(peer_device, device->ov_left);
staticvoid maybe_send_barrier(struct drbd_connection *connection, unsignedint epoch)
{ /* re-init if first write on this connection */ if (!connection->send.seen_any_write_yet) return; if (connection->send.current_epoch_nr != epoch) { if (connection->send.current_epoch_writes)
drbd_send_barrier(connection);
connection->send.current_epoch_nr = epoch;
}
}
/* this time, no connection->send.current_epoch_writes++; *Ifitwassent,itwastheclosingbarrierforthelast *replicatedepoch,beforewewentintoAHEADmode.
* No more barriers will be sent, until we leave AHEAD mode again. */
maybe_send_barrier(connection, req->epoch);
/* caller must lock_all_resources() */ enum drbd_ret_code drbd_resync_after_valid(struct drbd_device *device, int o_minor)
{ struct drbd_device *odev; int resync_after;
if (o_minor == -1) return NO_ERROR; if (o_minor < -1 || o_minor > MINORMASK) return ERR_RESYNC_AFTER;
/* check for loops */
odev = minor_to_device(o_minor); while (1) { if (odev == device) return ERR_RESYNC_AFTER_CYCLE;
/* You are free to depend on diskless, non-existing, *ornotyet/nolongerexistingminors. *Weonlyrejectdependencyloops. *Wecannotfollowthedependencychainbeyondadetachedor *missingminor.
*/ if (!odev || !odev->ldev || odev->state.disk == D_DISKLESS) return NO_ERROR;
rcu_read_lock();
resync_after = rcu_dereference(odev->ldev->disk_conf)->resync_after;
rcu_read_unlock(); /* dependency chain ends here, no cycles. */ if (resync_after == -1) return NO_ERROR;
/* Updating the RCU protected object in place is necessary since thisfunctiongetscalledfromatomiccontext. Itisvalidsinceallotherupdatesalsoleadtoancompletely
empty fifo */
rcu_read_lock();
plan = rcu_dereference(device->rs_plan_s);
plan->total = 0;
fifo_set(plan, 0);
rcu_read_unlock();
}
if (!connection) {
drbd_err(device, "No connection to peer, aborting!\n"); return;
}
if (!test_bit(B_RS_H_DONE, &device->flags)) { if (side == C_SYNC_TARGET) { /* Since application IO was locked out during C_WF_BITMAP_T and C_WF_SYNC_UUIDwearestillunmodified.BeforegoingtoC_SYNC_TARGET
we check that we might make the data inconsistent. */
r = drbd_khelper(device, "before-resync-target");
r = (r >> 8) & 0xff; if (r > 0) {
drbd_info(device, "before-resync-target handler returned %d, " "dropping connection.\n", r);
conn_request_state(connection, NS(conn, C_DISCONNECTING), CS_HARD); return;
}
} else/* C_SYNC_SOURCE */ {
r = drbd_khelper(device, "before-resync-source");
r = (r >> 8) & 0xff; if (r > 0) { if (r == 3) {
drbd_info(device, "before-resync-source handler returned %d, " "ignoring. Old userland tools?", r);
} else {
drbd_info(device, "before-resync-source handler returned %d, " "dropping connection.\n", r);
conn_request_state(connection,
NS(conn, C_DISCONNECTING), CS_HARD); return;
}
}
}
}
if (current == connection->worker.task) { /* The worker should not sleep waiting for state_mutex,
that can take long */ if (!mutex_trylock(device->state_mutex)) {
set_bit(B_RS_H_DONE, &device->flags);
device->start_resync_timer.expires = jiffies + HZ/5;
add_timer(&device->start_resync_timer); return;
}
} else {
mutex_lock(device->state_mutex);
}
lock_all_resources();
clear_bit(B_RS_H_DONE, &device->flags); /* Did some connection breakage or IO error race with us? */ if (device->state.conn < C_CONNECTED
|| !get_ldev_if_state(device, D_NEGOTIATING)) {
unlock_all_resources(); goto out;
}
ns = drbd_read_state(device);
ns.aftr_isp = !_drbd_may_sync_now(device);
ns.conn = side;
if (side == C_SYNC_TARGET)
ns.disk = D_INCONSISTENT; else/* side == C_SYNC_SOURCE */
ns.pdsk = D_INCONSISTENT;
r = _drbd_set_state(device, ns, CS_VERBOSE, NULL);
ns = drbd_read_state(device);
if (ns.conn < C_CONNECTED)
r = SS_UNKNOWN_ERROR;
if (r == SS_SUCCESS) { unsignedlong tw = drbd_bm_total_weight(device); unsignedlong now = jiffies; int i;
device->rs_failed = 0;
device->rs_paused = 0;
device->rs_same_csum = 0;
device->rs_last_sect_ev = 0;
device->rs_total = tw;
device->rs_start = now; for (i = 0; i < DRBD_SYNC_MARKS; i++) {
device->rs_mark_left[i] = tw;
device->rs_mark_time[i] = now;
}
drbd_pause_after(device); /* Forget potentially stale cached per resync extent bit-counts. *Opencodeddrbd_rs_cancel_all(device),wealreadyhaveIRQs
* disabled, and know the disk state is ok. */
spin_lock(&device->al_lock);
lc_reset(device->resync);
device->resync_locked = 0;
device->resync_wenr = LC_FREE;
spin_unlock(&device->al_lock);
}
unlock_all_resources();
if (r == SS_SUCCESS) {
wake_up(&device->al_wait); /* for lc_reset() above */ /* reset rs_last_bcast when a resync or verify is started,
* to deal with potential jiffies wrap. */
device->rs_last_bcast = jiffies - HZ;
/* Since protocol 96, we must serialize drbd_gen_and_send_sync_uuid *withw_send_oos,orthesynctargetwillgetconfusedasto *howmuchbitstoresync.Wecannotdothatalways,becauseforan *emptyresyncandprotocol<95,weneedtodoithere,aswecall *drbd_resync_finishedfromhereinthatcase. *Wedrbd_gen_and_send_sync_uuidhereforprotocol<96,
* and from after_state_ch otherwise. */ if (side == C_SYNC_SOURCE && connection->agreed_pro_version < 96)
drbd_gen_and_send_sync_uuid(peer_device);
if (connection->agreed_pro_version < 95 && device->rs_total == 0) { /* This still has a race (about when exactly the peers *detectconnectionloss)thatcanleadtoafullsync *onnexthandshake.In8.3.9wefixedthiswithexplicit *resync-finishednotifications,butthefix *introducesaprotocolchange.Sleepingforsome *timelongerthanthepinginterval+timeoutonthe *SyncSource,togivetheSyncTargetthechanceto *detectconnectionloss,thenwaitingforaping *response(implicitindrbd_resync_finished)reduces
* the race considerably, but does not solve it. */ if (side == C_SYNC_SOURCE) { struct net_conf *nc; int timeo;
drbd_rs_controller_reset(peer_device); /* ns.conn may already be != device->state.conn, *wemayhavebeenpausedinbetween,orbecomepauseduntil *thetimertriggers.
* No matter, that is handled in resync_timer_fn() */ if (ns.conn == C_SYNC_TARGET)
mod_timer(&device->resync_timer, jiffies);
drbd_bm_write_lazy(device, 0); if (resync_done && is_sync_state(device->state.conn))
drbd_resync_finished(peer_device);
drbd_bcast_event(device, &sib); /* update timestamp, in case it took a while to write out stuff */
device->rs_last_bcast = jiffies;
put_ldev(device);
}
staticvoid go_diskless(struct drbd_device *device)
{ struct drbd_peer_device *peer_device = first_peer_device(device);
D_ASSERT(device, device->state.disk == D_FAILED); /* we cannot assert local_cnt == 0 here, as get_ldev_if_state will *inc/decitfrequently.OnceweareD_DISKLESS,noonewilltouch *theprotectedmembersanymore,though,soonceput_ldevreacheszero
* again, it will be safe to free them. */
/* Try to write changed bitmap pages, read errors may have just *setsomebitsoutsidetheareacoveredbytheactivitylog. * *IfwehaveanIOerrorduringthebitmapwriteout, *wewillwantafullsyncnexttime,justincase. *(Dowewantaspecificmetadataflagforthis?) * *Ifthatdoesnotmakeittostablestorageeither, *wecannotdoanythingaboutthatanymore. * *Westillneedtocheckifbothbitmapandldevarepresent,wemay *enduphereafterafailedattach,beforeldevwasevenassigned.
*/ if (device->bitmap && device->ldev) { /* An interrupted resync or similar is allowed to recounts bits *whilewedetach. *Anymodificationswouldnotbeexpectedanymore,though.
*/ if (drbd_bitmap_io_from_worker(device, drbd_bm_write, "detach", BM_LOCKED_TEST_ALLOWED, peer_device)) { if (test_bit(WAS_READ_ERROR, &device->flags)) {
drbd_md_set_flag(device, MDF_FULL_SYNC);
drbd_md_sync(device);
}
}
}
staticunsignedlong get_work_bits(unsignedlong *flags)
{ unsignedlong old, new; do {
old = *flags; new = old & ~DRBD_DEVICE_WORK_MASK;
} while (cmpxchg(flags, old, new) != old); return old & DRBD_DEVICE_WORK_MASK;
}
staticvoid do_unqueued_work(struct drbd_connection *connection)
{ struct drbd_peer_device *peer_device; int vnr;
rcu_read_lock();
idr_for_each_entry(&connection->peer_devices, peer_device, vnr) { struct drbd_device *device = peer_device->device; unsignedlong todo = get_work_bits(&device->flags); if (!todo) continue;
dequeue_work_batch(&connection->sender_work, work_list); if (!list_empty(work_list)) return;
/* Still nothing to do? *Maybewestillneedtoclosethecurrentepoch, *evenifnonewrequestsarequeuedyet. * *Also,pokeTCP,justincase.
* Then wait for new work (or signal). */
rcu_read_lock();
nc = rcu_dereference(connection->net_conf);
uncork = nc ? nc->tcp_cork : 0;
rcu_read_unlock(); if (uncork) {
mutex_lock(&connection->data.mutex); if (connection->data.socket)
tcp_sock_set_cork(connection->data.socket->sk, false);
mutex_unlock(&connection->data.mutex);
}
for (;;) { int send_barrier;
prepare_to_wait(&connection->sender_work.q_wait, &wait, TASK_INTERRUPTIBLE);
spin_lock_irq(&connection->resource->req_lock);
spin_lock(&connection->sender_work.q_lock); /* FIXME get rid of this one? */ if (!list_empty(&connection->sender_work.q))
list_splice_tail_init(&connection->sender_work.q, work_list);
spin_unlock(&connection->sender_work.q_lock); /* FIXME get rid of this one? */ if (!list_empty(work_list) || signal_pending(current)) {
spin_unlock_irq(&connection->resource->req_lock); break;
}
/* We found nothing new to do, no to-be-communicated request, *nootherworkitem.Wemaystillneedtoclosethelast *epoch.Nextincomingrequestepochwillbeconnection-> *currenttransferlogepochnumber.Ifthatisdifferent *fromtheepochofthelastrequestwecommunicated,itis *safetosendtheepochseparatingbarriernow.
*/
send_barrier =
atomic_read(&connection->current_tle_nr) !=
connection->send.current_epoch_nr;
spin_unlock_irq(&connection->resource->req_lock);
if (send_barrier)
maybe_send_barrier(connection,
connection->send.current_epoch_nr + 1);
if (test_bit(DEVICE_WORK_PENDING, &connection->flags)) break;
/* drbd_send() may have called flush_signals() */ if (get_t_state(&connection->worker) != RUNNING) break;
schedule(); /* may be woken up for other things but new work, too, *e.g.ifthecurrentepochgotclosed.
* In which case we send the barrier above. */
}
finish_wait(&connection->sender_work.q_wait, &wait);
/* someone may have changed the config while we have been waiting above. */
rcu_read_lock();
nc = rcu_dereference(connection->net_conf);
cork = nc ? nc->tcp_cork : 0;
rcu_read_unlock();
mutex_lock(&connection->data.mutex); if (connection->data.socket) { if (cork)
tcp_sock_set_cork(connection->data.socket->sk, true); elseif (!uncork)
tcp_sock_set_cork(connection->data.socket->sk, false);
}
mutex_unlock(&connection->data.mutex);
}
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.31Bemerkung:
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.