Line data Source code
1 : //! Safekeeper communication endpoint to WAL proposer (compute node).
2 : //! Gets messages from the network, passes them down to consensus module and
3 : //! sends replies back.
4 :
5 : use std::{collections::HashMap, sync::Arc, time::Duration};
6 :
7 : use anyhow::{bail, Result};
8 : use bytes::{Bytes, BytesMut};
9 : use camino::Utf8PathBuf;
10 : use desim::{
11 : executor::{self, PollSome},
12 : network::TCP,
13 : node_os::NodeOs,
14 : proto::{AnyMessage, NetEvent, NodeEvent},
15 : };
16 : use http::Uri;
17 : use safekeeper::{
18 : safekeeper::{ProposerAcceptorMessage, SafeKeeper, ServerInfo, UNKNOWN_SERVER_VERSION},
19 : state::{TimelinePersistentState, TimelineState},
20 : timeline::TimelineError,
21 : wal_storage::Storage,
22 : SafeKeeperConf,
23 : };
24 : use tracing::{debug, info_span, warn};
25 : use utils::{
26 : id::{NodeId, TenantId, TenantTimelineId, TimelineId},
27 : lsn::Lsn,
28 : };
29 :
30 : use super::safekeeper_disk::{DiskStateStorage, DiskWALStorage, SafekeeperDisk, TimelineDisk};
31 :
32 : struct SharedState {
33 : sk: SafeKeeper<DiskStateStorage, DiskWALStorage>,
34 : disk: Arc<TimelineDisk>,
35 : }
36 :
37 : struct GlobalMap {
38 : timelines: HashMap<TenantTimelineId, SharedState>,
39 : conf: SafeKeeperConf,
40 : disk: Arc<SafekeeperDisk>,
41 : }
42 :
43 : impl GlobalMap {
44 : /// Restores global state from disk.
45 9491 : fn new(disk: Arc<SafekeeperDisk>, conf: SafeKeeperConf) -> Result<Self> {
46 9491 : let mut timelines = HashMap::new();
47 :
48 9491 : for (&ttid, disk) in disk.timelines.lock().iter() {
49 7120 : debug!("loading timeline {}", ttid);
50 7120 : let state = disk.state.lock().clone();
51 7120 :
52 7120 : if state.server.wal_seg_size == 0 {
53 0 : bail!(TimelineError::UninitializedWalSegSize(ttid));
54 7120 : }
55 7120 :
56 7120 : if state.server.pg_version == UNKNOWN_SERVER_VERSION {
57 0 : bail!(TimelineError::UninitialinzedPgVersion(ttid));
58 7120 : }
59 7120 :
60 7120 : if state.commit_lsn < state.local_start_lsn {
61 0 : bail!(
62 0 : "commit_lsn {} is smaller than local_start_lsn {}",
63 0 : state.commit_lsn,
64 0 : state.local_start_lsn
65 0 : );
66 7120 : }
67 7120 :
68 7120 : let control_store = DiskStateStorage::new(disk.clone());
69 7120 : let wal_store = DiskWALStorage::new(disk.clone(), &control_store)?;
70 :
71 7120 : let sk = SafeKeeper::new(TimelineState::new(control_store), wal_store, conf.my_id)?;
72 7120 : timelines.insert(
73 7120 : ttid,
74 7120 : SharedState {
75 7120 : sk,
76 7120 : disk: disk.clone(),
77 7120 : },
78 7120 : );
79 : }
80 :
81 9491 : Ok(Self {
82 9491 : timelines,
83 9491 : conf,
84 9491 : disk,
85 9491 : })
86 9491 : }
87 :
88 1402 : fn create(&mut self, ttid: TenantTimelineId, server_info: ServerInfo) -> Result<()> {
89 1402 : if self.timelines.contains_key(&ttid) {
90 0 : bail!("timeline {} already exists", ttid);
91 1402 : }
92 1402 :
93 1402 : debug!("creating new timeline {}", ttid);
94 :
95 1402 : let commit_lsn = Lsn::INVALID;
96 1402 : let local_start_lsn = Lsn::INVALID;
97 :
98 1402 : let state =
99 1402 : TimelinePersistentState::new(&ttid, server_info, vec![], commit_lsn, local_start_lsn)?;
100 :
101 1402 : let disk_timeline = self.disk.put_state(&ttid, state);
102 1402 : let control_store = DiskStateStorage::new(disk_timeline.clone());
103 1402 : let wal_store = DiskWALStorage::new(disk_timeline.clone(), &control_store)?;
104 :
105 1402 : let sk = SafeKeeper::new(
106 1402 : TimelineState::new(control_store),
107 1402 : wal_store,
108 1402 : self.conf.my_id,
109 1402 : )?;
110 :
111 1402 : self.timelines.insert(
112 1402 : ttid,
113 1402 : SharedState {
114 1402 : sk,
115 1402 : disk: disk_timeline,
116 1402 : },
117 1402 : );
118 1402 : Ok(())
119 1402 : }
120 :
121 30275 : fn get(&mut self, ttid: &TenantTimelineId) -> &mut SharedState {
122 30275 : self.timelines.get_mut(ttid).expect("timeline must exist")
123 30275 : }
124 :
125 19571 : fn has_tli(&self, ttid: &TenantTimelineId) -> bool {
126 19571 : self.timelines.contains_key(ttid)
127 19571 : }
128 : }
129 :
130 : /// State of a single connection to walproposer.
131 : struct ConnState {
132 : tcp: TCP,
133 :
134 : greeting: bool,
135 : ttid: TenantTimelineId,
136 : flush_pending: bool,
137 :
138 : runtime: tokio::runtime::Runtime,
139 : }
140 :
141 9491 : pub fn run_server(os: NodeOs, disk: Arc<SafekeeperDisk>) -> Result<()> {
142 9491 : let _enter = info_span!("safekeeper", id = os.id()).entered();
143 9491 : debug!("started server");
144 9491 : os.log_event("started;safekeeper".to_owned());
145 9491 : let conf = SafeKeeperConf {
146 9491 : workdir: Utf8PathBuf::from("."),
147 9491 : my_id: NodeId(os.id() as u64),
148 9491 : listen_pg_addr: String::new(),
149 9491 : listen_http_addr: String::new(),
150 9491 : no_sync: false,
151 9491 : broker_endpoint: "/".parse::<Uri>().unwrap(),
152 9491 : broker_keepalive_interval: Duration::from_secs(0),
153 9491 : heartbeat_timeout: Duration::from_secs(0),
154 9491 : remote_storage: None,
155 9491 : max_offloader_lag_bytes: 0,
156 9491 : wal_backup_enabled: false,
157 9491 : listen_pg_addr_tenant_only: None,
158 9491 : advertise_pg_addr: None,
159 9491 : availability_zone: None,
160 9491 : peer_recovery_enabled: false,
161 9491 : backup_parallel_jobs: 0,
162 9491 : pg_auth: None,
163 9491 : pg_tenant_only_auth: None,
164 9491 : http_auth: None,
165 9491 : sk_auth_token: None,
166 9491 : current_thread_runtime: false,
167 9491 : walsenders_keep_horizon: false,
168 9491 : partial_backup_timeout: Duration::from_secs(0),
169 9491 : disable_periodic_broker_push: false,
170 9491 : enable_offload: false,
171 9491 : delete_offloaded_wal: false,
172 9491 : control_file_save_interval: Duration::from_secs(1),
173 9491 : partial_backup_concurrency: 1,
174 9491 : eviction_min_resident: Duration::ZERO,
175 9491 : };
176 :
177 9491 : let mut global = GlobalMap::new(disk, conf.clone())?;
178 9491 : let mut conns: HashMap<usize, ConnState> = HashMap::new();
179 :
180 9491 : for (&_ttid, shared_state) in global.timelines.iter_mut() {
181 7120 : let flush_lsn = shared_state.sk.wal_store.flush_lsn();
182 7120 : let commit_lsn = shared_state.sk.state.commit_lsn;
183 7120 : os.log_event(format!("tli_loaded;{};{}", flush_lsn.0, commit_lsn.0));
184 7120 : }
185 :
186 9491 : let node_events = os.node_events();
187 9491 : let mut epoll_vec: Vec<Box<dyn PollSome>> = vec![];
188 9491 : let mut epoll_idx: Vec<usize> = vec![];
189 :
190 : // TODO: batch events processing (multiple events per tick)
191 : loop {
192 74965 : epoll_vec.clear();
193 74965 : epoll_idx.clear();
194 74965 :
195 74965 : // node events channel
196 74965 : epoll_vec.push(Box::new(node_events.clone()));
197 74965 : epoll_idx.push(0);
198 :
199 : // tcp connections
200 291391 : for conn in conns.values() {
201 291391 : epoll_vec.push(Box::new(conn.tcp.recv_chan()));
202 291391 : epoll_idx.push(conn.tcp.connection_id());
203 291391 : }
204 :
205 : // waiting for the next message
206 74965 : let index = executor::epoll_chans(&epoll_vec, -1).unwrap();
207 74965 :
208 74965 : if index == 0 {
209 : // got a new connection
210 24830 : match node_events.must_recv() {
211 24830 : NodeEvent::Accept(tcp) => {
212 24830 : conns.insert(
213 24830 : tcp.connection_id(),
214 24830 : ConnState {
215 24830 : tcp,
216 24830 : greeting: false,
217 24830 : ttid: TenantTimelineId::empty(),
218 24830 : flush_pending: false,
219 24830 : runtime: tokio::runtime::Builder::new_current_thread().build()?,
220 : },
221 : );
222 : }
223 0 : NodeEvent::Internal(_) => unreachable!(),
224 : }
225 24830 : continue;
226 50135 : }
227 50135 :
228 50135 : let connection_id = epoll_idx[index];
229 50135 : let conn = conns.get_mut(&connection_id).unwrap();
230 50135 : let mut next_event = Some(conn.tcp.recv_chan().must_recv());
231 :
232 : loop {
233 91674 : let event = match next_event {
234 51239 : Some(event) => event,
235 40435 : None => break,
236 : };
237 :
238 51239 : match event {
239 36711 : NetEvent::Message(msg) => {
240 36711 : let res = conn.process_any(msg, &mut global);
241 36711 : if res.is_err() {
242 9700 : let e = res.unwrap_err();
243 9700 : let estr = e.to_string();
244 9700 : if !estr.contains("finished processing START_REPLICATION") {
245 9491 : warn!("conn {:?} error: {:?}", connection_id, e);
246 0 : panic!("unexpected error at safekeeper: {:#}", e);
247 209 : }
248 209 : conns.remove(&connection_id);
249 209 : break;
250 27011 : }
251 : }
252 14528 : NetEvent::Closed => {
253 14528 : // TODO: remove from conns?
254 14528 : }
255 : }
256 :
257 41539 : next_event = conn.tcp.recv_chan().try_recv();
258 : }
259 :
260 194469 : conns.retain(|_, conn| {
261 194469 : let res = conn.flush(&mut global);
262 194469 : if res.is_err() {
263 0 : debug!("conn {:?} error: {:?}", conn.tcp, res);
264 194469 : }
265 194469 : res.is_ok()
266 194469 : });
267 : }
268 0 : }
269 :
270 : impl ConnState {
271 : /// Process a message from the network. It can be START_REPLICATION request or a valid ProposerAcceptorMessage message.
272 27220 : fn process_any(&mut self, any: AnyMessage, global: &mut GlobalMap) -> Result<()> {
273 27220 : if let AnyMessage::Bytes(copy_data) = any {
274 27220 : let repl_prefix = b"START_REPLICATION ";
275 27220 : if !self.greeting && copy_data.starts_with(repl_prefix) {
276 209 : self.process_start_replication(copy_data.slice(repl_prefix.len()..), global)?;
277 209 : bail!("finished processing START_REPLICATION")
278 27011 : }
279 :
280 27011 : let msg = ProposerAcceptorMessage::parse(copy_data)?;
281 27011 : debug!("got msg: {:?}", msg);
282 27011 : self.process(msg, global)
283 : } else {
284 0 : bail!("unexpected message, expected AnyMessage::Bytes");
285 : }
286 27220 : }
287 :
288 : /// Process START_REPLICATION request.
289 209 : fn process_start_replication(
290 209 : &mut self,
291 209 : copy_data: Bytes,
292 209 : global: &mut GlobalMap,
293 209 : ) -> Result<()> {
294 : // format is "<tenant_id> <timeline_id> <start_lsn> <end_lsn>"
295 209 : let str = String::from_utf8(copy_data.to_vec())?;
296 :
297 209 : let mut parts = str.split(' ');
298 209 : let tenant_id = parts.next().unwrap().parse::<TenantId>()?;
299 209 : let timeline_id = parts.next().unwrap().parse::<TimelineId>()?;
300 209 : let start_lsn = parts.next().unwrap().parse::<u64>()?;
301 209 : let end_lsn = parts.next().unwrap().parse::<u64>()?;
302 :
303 209 : let ttid = TenantTimelineId::new(tenant_id, timeline_id);
304 209 : let shared_state = global.get(&ttid);
305 209 :
306 209 : // read bytes from start_lsn to end_lsn
307 209 : let mut buf = vec![0; (end_lsn - start_lsn) as usize];
308 209 : shared_state.disk.wal.lock().read(start_lsn, &mut buf);
309 209 :
310 209 : // send bytes to the client
311 209 : self.tcp.send(AnyMessage::Bytes(Bytes::from(buf)));
312 209 : Ok(())
313 209 : }
314 :
315 : /// Get or create a timeline.
316 19571 : fn init_timeline(
317 19571 : &mut self,
318 19571 : ttid: TenantTimelineId,
319 19571 : server_info: ServerInfo,
320 19571 : global: &mut GlobalMap,
321 19571 : ) -> Result<()> {
322 19571 : self.ttid = ttid;
323 19571 : if global.has_tli(&ttid) {
324 18169 : return Ok(());
325 1402 : }
326 1402 :
327 1402 : global.create(ttid, server_info)
328 19571 : }
329 :
330 : /// Process a ProposerAcceptorMessage.
331 27011 : fn process(&mut self, msg: ProposerAcceptorMessage, global: &mut GlobalMap) -> Result<()> {
332 27011 : if !self.greeting {
333 19571 : self.greeting = true;
334 19571 :
335 19571 : match msg {
336 19571 : ProposerAcceptorMessage::Greeting(ref greeting) => {
337 19571 : tracing::info!(
338 0 : "start handshake with walproposer {:?} {:?}",
339 : self.tcp,
340 : greeting
341 : );
342 19571 : let server_info = ServerInfo {
343 19571 : pg_version: greeting.pg_version,
344 19571 : system_id: greeting.system_id,
345 19571 : wal_seg_size: greeting.wal_seg_size,
346 19571 : };
347 19571 : let ttid = TenantTimelineId::new(greeting.tenant_id, greeting.timeline_id);
348 19571 : self.init_timeline(ttid, server_info, global)?
349 : }
350 : _ => {
351 0 : bail!("unexpected message {msg:?} instead of greeting");
352 : }
353 : }
354 7440 : }
355 :
356 27011 : let tli = global.get(&self.ttid);
357 27011 :
358 27011 : match msg {
359 3667 : ProposerAcceptorMessage::AppendRequest(append_request) => {
360 3667 : self.flush_pending = true;
361 3667 : self.process_sk_msg(
362 3667 : tli,
363 3667 : &ProposerAcceptorMessage::NoFlushAppendRequest(append_request),
364 3667 : )?;
365 : }
366 23344 : other => {
367 23344 : self.process_sk_msg(tli, &other)?;
368 : }
369 : }
370 :
371 27011 : Ok(())
372 27011 : }
373 :
374 : /// Process FlushWAL if needed.
375 194469 : fn flush(&mut self, global: &mut GlobalMap) -> Result<()> {
376 194469 : // TODO: try to add extra flushes in simulation, to verify that extra flushes don't break anything
377 194469 : if !self.flush_pending {
378 191414 : return Ok(());
379 3055 : }
380 3055 : self.flush_pending = false;
381 3055 : let shared_state = global.get(&self.ttid);
382 3055 : self.process_sk_msg(shared_state, &ProposerAcceptorMessage::FlushWAL)
383 194469 : }
384 :
385 : /// Make safekeeper process a message and send a reply to the TCP
386 30066 : fn process_sk_msg(
387 30066 : &mut self,
388 30066 : shared_state: &mut SharedState,
389 30066 : msg: &ProposerAcceptorMessage,
390 30066 : ) -> Result<()> {
391 30066 : let mut reply = self.runtime.block_on(shared_state.sk.process_msg(msg))?;
392 30066 : if let Some(reply) = &mut reply {
393 : // TODO: if this is AppendResponse, fill in proper hot standby feedback and disk consistent lsn
394 :
395 25519 : let mut buf = BytesMut::with_capacity(128);
396 25519 : reply.serialize(&mut buf)?;
397 :
398 25519 : self.tcp.send(AnyMessage::Bytes(buf.into()));
399 4547 : }
400 30066 : Ok(())
401 30066 : }
402 : }
403 :
404 : impl Drop for ConnState {
405 24802 : fn drop(&mut self) {
406 24802 : debug!("dropping conn: {:?}", self.tcp);
407 24802 : if !std::thread::panicking() {
408 209 : self.tcp.close();
409 24593 : }
410 : // TODO: clean up non-fsynced WAL
411 24802 : }
412 : }
|