1 /* SPDX-License-Identifier: GPL-2.0 */
2 #ifndef __FS_CEPH_MESSENGER_H
3 #define __FS_CEPH_MESSENGER_H
5 #include <linux/bvec.h>
6 #include <linux/crypto.h>
7 #include <linux/kref.h>
8 #include <linux/mutex.h>
10 #include <linux/radix-tree.h>
11 #include <linux/uio.h>
12 #include <linux/workqueue.h>
13 #include <net/net_namespace.h>
15 #include <linux/ceph/types.h>
16 #include <linux/ceph/buffer.h>
19 struct ceph_connection;
22 * Ceph defines these callbacks for handling connection events.
24 struct ceph_connection_operations {
25 struct ceph_connection *(*get)(struct ceph_connection *);
26 void (*put)(struct ceph_connection *);
28 /* handle an incoming message. */
29 void (*dispatch) (struct ceph_connection *con, struct ceph_msg *m);
31 /* authorize an outgoing connection */
32 struct ceph_auth_handshake *(*get_authorizer) (
33 struct ceph_connection *con,
34 int *proto, int force_new);
35 int (*add_authorizer_challenge)(struct ceph_connection *con,
37 int challenge_buf_len);
38 int (*verify_authorizer_reply) (struct ceph_connection *con);
39 int (*invalidate_authorizer)(struct ceph_connection *con);
41 /* there was some error on the socket (disconnect, whatever) */
42 void (*fault) (struct ceph_connection *con);
44 /* a remote host as terminated a message exchange session, and messages
45 * we sent (or they tried to send us) may be lost. */
46 void (*peer_reset) (struct ceph_connection *con);
48 struct ceph_msg * (*alloc_msg) (struct ceph_connection *con,
49 struct ceph_msg_header *hdr,
52 void (*reencode_message) (struct ceph_msg *msg);
54 int (*sign_message) (struct ceph_msg *msg);
55 int (*check_message_signature) (struct ceph_msg *msg);
57 /* msgr2 authentication exchange */
58 int (*get_auth_request)(struct ceph_connection *con,
59 void *buf, int *buf_len,
60 void **authorizer, int *authorizer_len);
61 int (*handle_auth_reply_more)(struct ceph_connection *con,
62 void *reply, int reply_len,
63 void *buf, int *buf_len,
64 void **authorizer, int *authorizer_len);
65 int (*handle_auth_done)(struct ceph_connection *con,
66 u64 global_id, void *reply, int reply_len,
67 u8 *session_key, int *session_key_len,
68 u8 *con_secret, int *con_secret_len);
69 int (*handle_auth_bad_method)(struct ceph_connection *con,
70 int used_proto, int result,
71 const int *allowed_protos, int proto_cnt,
72 const int *allowed_modes, int mode_cnt);
75 /* use format string %s%lld */
76 #define ENTITY_NAME(n) ceph_entity_type_name((n).type), le64_to_cpu((n).num)
78 struct ceph_messenger {
79 struct ceph_entity_inst inst; /* my name+address */
80 struct ceph_entity_addr my_enc_addr;
86 * the global_seq counts connections i (attempt to) initiate
87 * in order to disambiguate certain connect race conditions.
90 spinlock_t global_seq_lock;
93 enum ceph_msg_data_type {
94 CEPH_MSG_DATA_NONE, /* message contains no data payload */
95 CEPH_MSG_DATA_PAGES, /* data source/destination is a page array */
96 CEPH_MSG_DATA_PAGELIST, /* data source/destination is a pagelist */
98 CEPH_MSG_DATA_BIO, /* data source/destination is a bio list */
99 #endif /* CONFIG_BLOCK */
100 CEPH_MSG_DATA_BVECS, /* data source/destination is a bio_vec array */
105 struct ceph_bio_iter {
107 struct bvec_iter iter;
110 #define __ceph_bio_iter_advance_step(it, n, STEP) do { \
111 unsigned int __n = (n), __cur_n; \
114 BUG_ON(!(it)->iter.bi_size); \
115 __cur_n = min((it)->iter.bi_size, __n); \
117 bio_advance_iter((it)->bio, &(it)->iter, __cur_n); \
118 if (!(it)->iter.bi_size && (it)->bio->bi_next) { \
119 dout("__ceph_bio_iter_advance_step next bio\n"); \
120 (it)->bio = (it)->bio->bi_next; \
121 (it)->iter = (it)->bio->bi_iter; \
128 * Advance @it by @n bytes.
130 #define ceph_bio_iter_advance(it, n) \
131 __ceph_bio_iter_advance_step(it, n, 0)
134 * Advance @it by @n bytes, executing BVEC_STEP for each bio_vec.
136 #define ceph_bio_iter_advance_step(it, n, BVEC_STEP) \
137 __ceph_bio_iter_advance_step(it, n, ({ \
139 struct bvec_iter __cur_iter; \
141 __cur_iter = (it)->iter; \
142 __cur_iter.bi_size = __cur_n; \
143 __bio_for_each_segment(bv, (it)->bio, __cur_iter, __cur_iter) \
147 #endif /* CONFIG_BLOCK */
149 struct ceph_bvec_iter {
150 struct bio_vec *bvecs;
151 struct bvec_iter iter;
154 #define __ceph_bvec_iter_advance_step(it, n, STEP) do { \
155 BUG_ON((n) > (it)->iter.bi_size); \
157 bvec_iter_advance((it)->bvecs, &(it)->iter, (n)); \
161 * Advance @it by @n bytes.
163 #define ceph_bvec_iter_advance(it, n) \
164 __ceph_bvec_iter_advance_step(it, n, 0)
167 * Advance @it by @n bytes, executing BVEC_STEP for each bio_vec.
169 #define ceph_bvec_iter_advance_step(it, n, BVEC_STEP) \
170 __ceph_bvec_iter_advance_step(it, n, ({ \
172 struct bvec_iter __cur_iter; \
174 __cur_iter = (it)->iter; \
175 __cur_iter.bi_size = (n); \
176 for_each_bvec(bv, (it)->bvecs, __cur_iter, __cur_iter) \
180 #define ceph_bvec_iter_shorten(it, n) do { \
181 BUG_ON((n) > (it)->iter.bi_size); \
182 (it)->iter.bi_size = (n); \
185 struct ceph_msg_data {
186 enum ceph_msg_data_type type;
190 struct ceph_bio_iter bio_pos;
193 #endif /* CONFIG_BLOCK */
194 struct ceph_bvec_iter bvec_pos;
197 size_t length; /* total # bytes */
198 unsigned int alignment; /* first page */
201 struct ceph_pagelist *pagelist;
205 struct ceph_msg_data_cursor {
206 size_t total_resid; /* across all data items */
208 struct ceph_msg_data *data; /* current data item */
209 size_t resid; /* bytes not yet consumed */
210 bool need_crc; /* crc update needed */
213 struct ceph_bio_iter bio_iter;
214 #endif /* CONFIG_BLOCK */
215 struct bvec_iter bvec_iter;
217 unsigned int page_offset; /* offset in page */
218 unsigned short page_index; /* index in array */
219 unsigned short page_count; /* pages in array */
221 struct { /* pagelist */
222 struct page *page; /* page from list */
223 size_t offset; /* bytes from list */
229 * a single message. it contains a header (src, dest, message type, etc.),
230 * footer (crc values, mainly), a "front" message body, and possibly a
231 * data payload (stored in some number of pages).
234 struct ceph_msg_header hdr; /* header */
236 struct ceph_msg_footer footer; /* footer */
237 struct ceph_msg_footer_old old_footer; /* old format footer */
239 struct kvec front; /* unaligned blobs of message */
240 struct ceph_buffer *middle;
243 struct ceph_msg_data *data;
246 struct ceph_msg_data_cursor cursor;
248 struct ceph_connection *con;
249 struct list_head list_head; /* links for connection lists */
256 struct ceph_msgpool *pool;
262 #define CEPH_CON_S_CLOSED 1
263 #define CEPH_CON_S_PREOPEN 2
264 #define CEPH_CON_S_V1_BANNER 3
265 #define CEPH_CON_S_V1_CONNECT_MSG 4
266 #define CEPH_CON_S_V2_BANNER_PREFIX 5
267 #define CEPH_CON_S_V2_BANNER_PAYLOAD 6
268 #define CEPH_CON_S_V2_HELLO 7
269 #define CEPH_CON_S_V2_AUTH 8
270 #define CEPH_CON_S_V2_AUTH_SIGNATURE 9
271 #define CEPH_CON_S_V2_SESSION_CONNECT 10
272 #define CEPH_CON_S_V2_SESSION_RECONNECT 11
273 #define CEPH_CON_S_OPEN 12
274 #define CEPH_CON_S_STANDBY 13
277 * ceph_connection flag bits
279 #define CEPH_CON_F_LOSSYTX 0 /* we can close channel or drop
280 messages on errors */
281 #define CEPH_CON_F_KEEPALIVE_PENDING 1 /* we need to send a keepalive */
282 #define CEPH_CON_F_WRITE_PENDING 2 /* we have data ready to send */
283 #define CEPH_CON_F_SOCK_CLOSED 3 /* socket state changed to closed */
284 #define CEPH_CON_F_BACKOFF 4 /* need to retry queuing delayed
287 /* ceph connection fault delay defaults, for exponential backoff */
288 #define BASE_DELAY_INTERVAL (HZ / 4)
289 #define MAX_DELAY_INTERVAL (15 * HZ)
291 struct ceph_connection_v1_info {
292 struct kvec out_kvec[8], /* sending header/footer data */
294 int out_kvec_left; /* kvec's left in out_kvec */
295 int out_skip; /* skip this many bytes */
296 int out_kvec_bytes; /* total bytes left */
297 bool out_more; /* there is more data after the kvecs */
300 struct ceph_auth_handshake *auth;
301 int auth_retry; /* true if we need a newer authorizer */
303 /* connection negotiation temps */
304 u8 in_banner[CEPH_BANNER_MAX_LEN];
305 struct ceph_entity_addr actual_peer_addr;
306 struct ceph_entity_addr peer_addr_for_me;
307 struct ceph_msg_connect out_connect;
308 struct ceph_msg_connect_reply in_reply;
310 int in_base_pos; /* bytes read */
312 /* message in temps */
313 u8 in_tag; /* protocol control byte */
314 struct ceph_msg_header in_hdr;
315 __le64 in_temp_ack; /* for reading an ack */
317 /* message out temps */
318 struct ceph_msg_header out_hdr;
319 __le64 out_temp_ack; /* for writing an ack */
320 struct ceph_timespec out_temp_keepalive2; /* for writing keepalive2
323 u32 connect_seq; /* identify the most recent connection
324 attempt for this session */
325 u32 peer_global_seq; /* peer's global seq for this connection */
328 #define CEPH_CRC_LEN 4
329 #define CEPH_GCM_KEY_LEN 16
330 #define CEPH_GCM_IV_LEN sizeof(struct ceph_gcm_nonce)
331 #define CEPH_GCM_BLOCK_LEN 16
332 #define CEPH_GCM_TAG_LEN 16
334 #define CEPH_PREAMBLE_LEN 32
335 #define CEPH_PREAMBLE_INLINE_LEN 48
336 #define CEPH_PREAMBLE_PLAIN_LEN CEPH_PREAMBLE_LEN
337 #define CEPH_PREAMBLE_SECURE_LEN (CEPH_PREAMBLE_LEN + \
338 CEPH_PREAMBLE_INLINE_LEN + \
340 #define CEPH_EPILOGUE_PLAIN_LEN (1 + 3 * CEPH_CRC_LEN)
341 #define CEPH_EPILOGUE_SECURE_LEN (CEPH_GCM_BLOCK_LEN + CEPH_GCM_TAG_LEN)
343 #define CEPH_FRAME_MAX_SEGMENT_COUNT 4
345 struct ceph_frame_desc {
346 int fd_tag; /* FRAME_TAG_* */
348 int fd_lens[CEPH_FRAME_MAX_SEGMENT_COUNT]; /* logical */
349 int fd_aligns[CEPH_FRAME_MAX_SEGMENT_COUNT];
352 struct ceph_gcm_nonce {
354 __le64 counter __packed;
357 struct ceph_connection_v2_info {
358 struct iov_iter in_iter;
359 struct kvec in_kvecs[5]; /* recvmsg */
360 struct bio_vec in_bvec; /* recvmsg (in_cursor) */
362 int in_state; /* IN_S_* */
364 struct iov_iter out_iter;
365 struct kvec out_kvecs[8]; /* sendmsg */
366 struct bio_vec out_bvec; /* sendpage (out_cursor, out_zero),
367 sendmsg (out_enc_pages) */
369 int out_state; /* OUT_S_* */
371 int out_zero; /* # of zero bytes to send */
372 bool out_iter_sendpage; /* use sendpage if possible */
374 struct ceph_frame_desc in_desc;
375 struct ceph_msg_data_cursor in_cursor;
376 struct ceph_msg_data_cursor out_cursor;
378 struct crypto_shash *hmac_tfm; /* post-auth signature */
379 struct crypto_aead *gcm_tfm; /* on-wire encryption */
380 struct aead_request *gcm_req;
381 struct crypto_wait gcm_wait;
382 struct ceph_gcm_nonce in_gcm_nonce;
383 struct ceph_gcm_nonce out_gcm_nonce;
385 struct page **in_enc_pages;
389 struct page **out_enc_pages;
390 int out_enc_page_cnt;
394 int con_mode; /* CEPH_CON_MODE_* */
399 struct kvec in_sign_kvecs[8];
400 struct kvec out_sign_kvecs[8];
401 int in_sign_kvec_cnt;
402 int out_sign_kvec_cnt;
410 u8 in_buf[CEPH_PREAMBLE_SECURE_LEN];
411 u8 out_buf[CEPH_PREAMBLE_SECURE_LEN];
413 u8 late_status; /* FRAME_LATE_STATUS_* */
420 u8 pad[CEPH_GCM_BLOCK_LEN - 1];
426 * A single connection with another host.
428 * We maintain a queue of outgoing messages, and some session state to
429 * ensure that we can preserve the lossless, ordered delivery of
430 * messages in the case of a TCP disconnect.
432 struct ceph_connection {
435 const struct ceph_connection_operations *ops;
437 struct ceph_messenger *msgr;
439 int state; /* CEPH_CON_S_* */
443 unsigned long flags; /* CEPH_CON_F_* */
444 const char *error_msg; /* error message, if any */
446 struct ceph_entity_name peer_name; /* peer name */
447 struct ceph_entity_addr peer_addr; /* peer address */
453 struct list_head out_queue;
454 struct list_head out_sent; /* sending or sent but unacked */
455 u64 out_seq; /* last message queued for send */
457 u64 in_seq, in_seq_acked; /* last message received, acked */
459 struct ceph_msg *in_msg;
460 struct ceph_msg *out_msg; /* sending message (== tail of
463 struct page *bounce_page;
464 u32 in_front_crc, in_middle_crc, in_data_crc; /* calculated crc */
466 struct timespec64 last_keepalive_ack; /* keepalive2 ack stamp */
468 struct delayed_work work; /* send|recv work */
469 unsigned long delay; /* current delay interval */
472 struct ceph_connection_v1_info v1;
473 struct ceph_connection_v2_info v2;
477 extern struct page *ceph_zero_page;
479 void ceph_con_flag_clear(struct ceph_connection *con, unsigned long con_flag);
480 void ceph_con_flag_set(struct ceph_connection *con, unsigned long con_flag);
481 bool ceph_con_flag_test(struct ceph_connection *con, unsigned long con_flag);
482 bool ceph_con_flag_test_and_clear(struct ceph_connection *con,
483 unsigned long con_flag);
484 bool ceph_con_flag_test_and_set(struct ceph_connection *con,
485 unsigned long con_flag);
487 void ceph_encode_my_addr(struct ceph_messenger *msgr);
489 int ceph_tcp_connect(struct ceph_connection *con);
490 int ceph_con_close_socket(struct ceph_connection *con);
491 void ceph_con_reset_session(struct ceph_connection *con);
493 u32 ceph_get_global_seq(struct ceph_messenger *msgr, u32 gt);
494 void ceph_con_discard_sent(struct ceph_connection *con, u64 ack_seq);
495 void ceph_con_discard_requeued(struct ceph_connection *con, u64 reconnect_seq);
497 void ceph_msg_data_cursor_init(struct ceph_msg_data_cursor *cursor,
498 struct ceph_msg *msg, size_t length);
499 struct page *ceph_msg_data_next(struct ceph_msg_data_cursor *cursor,
500 size_t *page_offset, size_t *length);
501 void ceph_msg_data_advance(struct ceph_msg_data_cursor *cursor, size_t bytes);
503 u32 ceph_crc32c_page(u32 crc, struct page *page, unsigned int page_offset,
504 unsigned int length);
506 bool ceph_addr_is_blank(const struct ceph_entity_addr *addr);
507 int ceph_addr_port(const struct ceph_entity_addr *addr);
508 void ceph_addr_set_port(struct ceph_entity_addr *addr, int p);
510 void ceph_con_process_message(struct ceph_connection *con);
511 int ceph_con_in_msg_alloc(struct ceph_connection *con,
512 struct ceph_msg_header *hdr, int *skip);
513 void ceph_con_get_out_msg(struct ceph_connection *con);
516 int ceph_con_v1_try_read(struct ceph_connection *con);
517 int ceph_con_v1_try_write(struct ceph_connection *con);
518 void ceph_con_v1_revoke(struct ceph_connection *con);
519 void ceph_con_v1_revoke_incoming(struct ceph_connection *con);
520 bool ceph_con_v1_opened(struct ceph_connection *con);
521 void ceph_con_v1_reset_session(struct ceph_connection *con);
522 void ceph_con_v1_reset_protocol(struct ceph_connection *con);
525 int ceph_con_v2_try_read(struct ceph_connection *con);
526 int ceph_con_v2_try_write(struct ceph_connection *con);
527 void ceph_con_v2_revoke(struct ceph_connection *con);
528 void ceph_con_v2_revoke_incoming(struct ceph_connection *con);
529 bool ceph_con_v2_opened(struct ceph_connection *con);
530 void ceph_con_v2_reset_session(struct ceph_connection *con);
531 void ceph_con_v2_reset_protocol(struct ceph_connection *con);
534 extern const char *ceph_pr_addr(const struct ceph_entity_addr *addr);
536 extern int ceph_parse_ips(const char *c, const char *end,
537 struct ceph_entity_addr *addr,
538 int max_count, int *count, char delim);
540 extern int ceph_msgr_init(void);
541 extern void ceph_msgr_exit(void);
542 extern void ceph_msgr_flush(void);
544 extern void ceph_messenger_init(struct ceph_messenger *msgr,
545 struct ceph_entity_addr *myaddr);
546 extern void ceph_messenger_fini(struct ceph_messenger *msgr);
547 extern void ceph_messenger_reset_nonce(struct ceph_messenger *msgr);
549 extern void ceph_con_init(struct ceph_connection *con, void *private,
550 const struct ceph_connection_operations *ops,
551 struct ceph_messenger *msgr);
552 extern void ceph_con_open(struct ceph_connection *con,
553 __u8 entity_type, __u64 entity_num,
554 struct ceph_entity_addr *addr);
555 extern bool ceph_con_opened(struct ceph_connection *con);
556 extern void ceph_con_close(struct ceph_connection *con);
557 extern void ceph_con_send(struct ceph_connection *con, struct ceph_msg *msg);
559 extern void ceph_msg_revoke(struct ceph_msg *msg);
560 extern void ceph_msg_revoke_incoming(struct ceph_msg *msg);
562 extern void ceph_con_keepalive(struct ceph_connection *con);
563 extern bool ceph_con_keepalive_expired(struct ceph_connection *con,
564 unsigned long interval);
566 void ceph_msg_data_add_pages(struct ceph_msg *msg, struct page **pages,
567 size_t length, size_t alignment, bool own_pages);
568 extern void ceph_msg_data_add_pagelist(struct ceph_msg *msg,
569 struct ceph_pagelist *pagelist);
571 void ceph_msg_data_add_bio(struct ceph_msg *msg, struct ceph_bio_iter *bio_pos,
573 #endif /* CONFIG_BLOCK */
574 void ceph_msg_data_add_bvecs(struct ceph_msg *msg,
575 struct ceph_bvec_iter *bvec_pos);
577 struct ceph_msg *ceph_msg_new2(int type, int front_len, int max_data_items,
578 gfp_t flags, bool can_fail);
579 extern struct ceph_msg *ceph_msg_new(int type, int front_len, gfp_t flags,
582 extern struct ceph_msg *ceph_msg_get(struct ceph_msg *msg);
583 extern void ceph_msg_put(struct ceph_msg *msg);
585 extern void ceph_msg_dump(struct ceph_msg *msg);