mirror of
https://github.com/nestriness/nestri.git
synced 2026-09-19 17:25:19 +03:00
fix(nesinit): the relay may not stall the session, and may not buffer without end
Three problems in the relay, all of them found in review. Handing an envelope over waited for room. That loop also carries stop, shutdown and the workload's exit, so a workload slow to read its own mail — or one that never connected — could hold the lifecycle layer still behind it. It never waits now: an envelope that will not fit is dropped, which costs nothing, because what crosses this layer is re-sent when it changes. Envelopes were queued for a workload that was not there. The queue filled with copies that would be stale by the time anyone connected, and filling it was what stalled the session. Nothing is held while the socket has nobody on it. A frame had no maximum length. The workload can write for as long as it likes without ever sending a newline, and the process assembling that is the one the kernel has been told not to kill, so the memory it takes comes out of everything else in the guest. Past 64 KiB the connection is dropped and the relay waits for the next one; the failure says how long the frame got and nothing about what was in it. Also, a tag or a mount point with a nul byte in it was quietly turned into an empty string, so an unmountable descriptor arrived later as a mount failure about something else, after the mount point had already been created. It is refused by name now, before anything is created. The relay's tests grew a harness that waits for the connection to be carried before sending anything down it, because dropping what arrives with nobody connected made "connected" something a test has to establish rather than assume.
This commit is contained in:
@@ -183,7 +183,7 @@ where
|
||||
}
|
||||
}
|
||||
}
|
||||
HostToGuest::Payload { payload: envelope } => hand_over(payload, envelope).await,
|
||||
HostToGuest::Payload { payload: envelope } => hand_over(payload, envelope),
|
||||
HostToGuest::Stop => workload.signal_stop(),
|
||||
HostToGuest::Shutdown => return Ok(Outcome::Shutdown),
|
||||
}
|
||||
@@ -199,13 +199,24 @@ enum Event {
|
||||
|
||||
/// Hand an envelope to the relay, and treat a relay that is not there as the
|
||||
/// caller's problem rather than a failure of this session.
|
||||
async fn hand_over(ports: &mut Ports, envelope: Payload) {
|
||||
tracing::debug!(envelope = %envelope.summary(), "handing an envelope over");
|
||||
if ports.to_workload.send(envelope).await.is_err() {
|
||||
// The workload is not on the relay. Dropped rather than queued: what
|
||||
// crosses here is re-sent when it changes, so a held copy is a stale
|
||||
// copy.
|
||||
tracing::warn!("dropped an envelope: nothing is on the relay");
|
||||
///
|
||||
/// It never waits. This loop also carries stop, shutdown and the workload's
|
||||
/// exit, and none of those may be held up by a workload that is slow to read
|
||||
/// its own mail — or by one that never connected at all. What crosses this
|
||||
/// layer is re-sent when it changes, so a dropped copy costs less than a
|
||||
/// stalled session.
|
||||
fn hand_over(ports: &mut Ports, envelope: Payload) {
|
||||
use tokio::sync::mpsc::error::TrySendError;
|
||||
|
||||
let summary = envelope.summary();
|
||||
match ports.to_workload.try_send(envelope) {
|
||||
Ok(()) => tracing::debug!(envelope = %summary, "handed an envelope over"),
|
||||
Err(TrySendError::Full(_)) => {
|
||||
tracing::warn!(envelope = %summary, "dropped an envelope: the relay is behind")
|
||||
}
|
||||
Err(TrySendError::Closed(_)) => {
|
||||
tracing::warn!(envelope = %summary, "dropped an envelope: the relay is gone")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -655,4 +666,55 @@ mod tests {
|
||||
|
||||
assert_eq!(session.await.unwrap(), Outcome::Shutdown);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn a_relay_that_is_not_draining_does_not_stall_the_session() {
|
||||
// Bounded, because the failure is a session that stops rather than one
|
||||
// that answers wrongly.
|
||||
tokio::time::timeout(std::time::Duration::from_secs(5), async {
|
||||
a_backed_up_relay().await
|
||||
})
|
||||
.await
|
||||
.expect("the session stalled on the relay");
|
||||
}
|
||||
|
||||
async fn a_backed_up_relay() {
|
||||
let (guest, host) = tokio::io::duplex(4096);
|
||||
let mut caller = Caller::new(host);
|
||||
let (down_tx, down_rx) = mpsc::channel(1);
|
||||
let (_up_tx, up_rx) = mpsc::channel::<Payload>(1);
|
||||
// Held and never read: a workload that is slow to read its own mail,
|
||||
// or one that connected and stopped.
|
||||
let _backed_up = down_rx;
|
||||
|
||||
let session = tokio::spawn(async move {
|
||||
let mut ports = Ports {
|
||||
to_workload: down_tx,
|
||||
from_workload: up_rx,
|
||||
};
|
||||
let mut workload = Double::exits_when_stopped(Exit::code(0));
|
||||
run(guest, &mut workload, &mut ports).await.unwrap()
|
||||
});
|
||||
|
||||
assert!(matches!(caller.expect().await, GuestToHost::Ready { .. }));
|
||||
for _ in 0..8 {
|
||||
caller
|
||||
.say(&HostToGuest::Payload {
|
||||
payload: Payload::new("identity", "backlog"),
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
// The lifecycle layer still moves: stop, shutdown and an exit are on
|
||||
// this loop too, and none of them may wait on the relay.
|
||||
caller
|
||||
.say(&HostToGuest::Boot {
|
||||
descriptor: Box::new(descriptor()),
|
||||
})
|
||||
.await;
|
||||
caller.expect_started().await;
|
||||
caller.say(&HostToGuest::Shutdown).await;
|
||||
|
||||
assert_eq!(session.await.unwrap(), Outcome::Shutdown);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user