
CVE-2023-2598 का io_uring से संबंधित शोषण
io_uring लिनक्स के लिए एक सिस्टम कॉल इंटरफ़ेस है। यह अब तक लगभग सभी सिस्टम कॉल्स का समर्थन करता है, न केवल शुरुआती read() और write का। यह एप्लिकेशन को सिस्टम कॉल्स शुरू करने में सक्षम बनाता है जिन्हें एसिंक्रोनस रूप से निष्पादित किया जा सकता है।
हर io_uring इम्प्लीमेंटेशन के केंद्र में दो रिंग बफ़र होते हैं - सबमिशन क्यू (SQ) और कम्प्लीशन क्यू (CQ)। ये रिंग बफ़र एप्लिकेशन और कर्नेल के बीच साझा किए जाते हैं।
हम io_uring_get_sqe के माध्यम से एक सबमिशन क्यू एंट्री (SQE) प्राप्त कर सकते हैं जो उस syscall का वर्णन करती है जिसे आप io_uring से निष्पादित करवाना चाहते हैं। एप्लिकेशन फिर एक io_uring_enter syscall करता है जो कर्नेल को प्रभावी रूप से बताता है कि सबमिशन क्यू में किया जाने वाला कार्य प्रतीक्षारत है।
कर्नेल द्वारा ऑपरेशन निष्पादित करने के बाद यह एक कम्प्लीशन क्यू एंट्री (CQE) को कम्प्लीशन क्यू रिंग बफ़र में रखता है जिसे बाद में एप्लिकेशन द्वारा उपभोग किया जा सकता है।
फ़ंक्शन io_sqe_buffer_register वर्चुअल पेजों और भौतिक पतों के मैपिंग को लागू करता है।
हमें पहले कुछ अवधारणाओं को स्पष्ट करना चाहिए।
एप्लिकेशन io_uring_register के माध्यम से बफ़र के लिए एक अनुरोध शुरू करता है। कॉल श्रृंखला इस प्रकार है:
io_uring_register_buffers->io_uring_register->io_sqe_buffers_register
फ़ंक्शन io_sqe_buffers_register का स्रोत कोड इस प्रकार है:
int io_sqe_buffers_register(struct io_ring_ctx *ctx, void __user *arg,
unsigned int nr_args, u64 __user *tags)
{
struct page *last_hpage = NULL;
struct io_rsrc_data *data;
int i, ret;
struct iovec iov;
BUILD_BUG_ON(IORING_MAX_REG_BUFFERS >= (1u << 16));
if (ctx->user_bufs)
return -EBUSY;
if (!nr_args || nr_args > IORING_MAX_REG_BUFFERS)
return -EINVAL;
ret = io_rsrc_node_switch_start(ctx);
if (ret)
return ret;
ret = io_rsrc_data_alloc(ctx, io_rsrc_buf_put, tags, nr_args, &data);
if (ret)
return ret;
ret = io_buffers_map_alloc(ctx, nr_args);
if (ret) {
io_rsrc_data_free(data);
return ret;
}
for (i = 0; i < nr_args; i++, ctx->nr_user_bufs++) {
if (arg) {
ret = io_copy_iov(ctx, &iov, arg, i);
if (ret)
break;
ret = io_buffer_validate(&iov);
if (ret)
break;
} else {
memset(&iov, 0, sizeof(iov));
}
if (!iov.iov_base && *io_get_tag_slot(data, i)) {
ret = -EINVAL;
break;
}
ret = io_sqe_buffer_register(ctx, &iov, &ctx->user_bufs[i],
&last_hpage);
if (ret)
break;
}
WARN_ON_ONCE(ctx->buf_data);
ctx->buf_data = data;
if (ret)
__io_sqe_buffers_unregister(ctx);
else
io_rsrc_node_switch(ctx, NULL);
return ret;
}
इस फ़ंक्शन में, हम io_sqe_buffer_register तक पहुँचेंगे। और हमें एक लॉजिकल बग मिलेगा। फ़ंक्शन io_sqe_buffer_register का स्रोत कोड इस प्रकार है:
static int io_sqe_buffer_register(struct io_ring_ctx *ctx, struct iovec *iov,
struct io_mapped_ubuf **pimu,
struct page **last_hpage)
{
struct io_mapped_ubuf *imu = NULL;
struct page **pages = NULL;
unsigned long off;
size_t size;
int ret, nr_pages, i;
struct folio *folio = NULL;
*pimu = ctx->dummy_ubuf;
if (!iov->iov_base)
return 0;
ret = -ENOMEM;
pages = io_pin_pages((unsigned long) iov->iov_base, iov->iov_len,
&nr_pages);
if (IS_ERR(pages)) {
ret = PTR_ERR(pages);
pages = NULL;
goto done;
}
/* If it's a huge page, try to coalesce them into a single bvec entry */
if (nr_pages > 1) {
folio = page_folio(pages[0]);
for (i = 1; i < nr_pages; i++) {
if (page_folio(pages[i]) != folio) {
folio = NULL;
break;
}
}
if (folio) {
folio_put_refs(folio, nr_pages - 1);
nr_pages = 1;
}
}
imu = kvmalloc(struct_size(imu, bvec, nr_pages), GFP_KERNEL);
if (!imu)
goto done;
ret = io_buffer_account_pin(ctx, pages, nr_pages, imu, last_hpage);
if (ret) {
unpin_user_pages(pages, nr_pages);
goto done;
}
off = (unsigned long) iov->iov_base & ~PAGE_MASK;
size = iov->iov_len;
/* store original address for later verification */
imu->ubuf = (unsigned long) iov->iov_base;
imu->ubuf_end = imu->ubuf + iov->iov_len;
imu->nr_bvecs = nr_pages;
*pimu = imu;
ret = 0;
if (folio) {
bvec_set_page(&imu->bvec[0], pages[0], size, off);
goto done;
}
for (i = 0; i < nr_pages; i++) {
size_t vec_len;
vec_len = min_t(size_t, size, PAGE_SIZE - off);
bvec_set_page(&imu->bvec[i], pages[i], vec_len, off);
off = 0;
size -= vec_len;
}
done:
if (ret)
kvfree(imu);
kvfree(pages);
return ret;
}
यहाँ मैं केवल कुछ महत्वपूर्ण बिंदुओं का उल्लेख करता हूँ।
imu का अर्थ है वर्चुअल एड्रेस/पेज।page का अर्थ है भौतिक एड्रेस/पेज।folio का अर्थ है बहुत सारे पेज जो भौतिक रूप से सतत होते हैं, उस स्थिति को रोकते हुए जब किसी फ़ंक्शन को कॉल किया जाता है और उसके पैरामीटर में एक पेज होता है, लेकिन यह पेज पेजों की एक सतत श्रृंखला से संबंधित होता है, और हमें यह सुनिश्चित नहीं होता कि पूरे पेज का उपयोग करना है या एकल पेज का।struct iovec -> बस एक संरचना है जो एक बफ़र का वर्णन करती है, जिसमें बफ़र का प्रारंभिक पता और उसकी लंबाई होती है। और कुछ नहीं।io_mapped_ubuf एक संरचना है जो उस बफ़र के बारे में जानकारी रखती है जिसे io_uring इंस्टेंस में पंजीकृत किया गया है।struct io_mapped_ubuf {
u64 ubuf; // the address at which the buffer starts
u64 ubuf_end; // the address at which it ends
unsigned int nr_bvecs; // how many bio_vec(s) are needed to address the buffer
unsigned long acct_pages;
struct bio_vec bvec[]; // array of bio_vec(s)
};
सदस्य bio_ver एक struct है जैसे iovec लेकिन भौतिक मेमोरी के लिए।
...
/* If it's a huge page, try to coalesce them into a single bvec entry */
if (nr_pages > 1) { // if more than one page
folio = page_folio(pages[0]); // converts from page to folio
// returns the folio that contains this page
for (i = 1; i < nr_pages; i++) {
if (page_folio(pages[i]) != folio) { // different folios -> not physically contiguous
folio = NULL; // set folio to NULL as we cannot coalesce into a single entry
break;
}
}
if (folio) { // if all the pages are in the same folio
folio_put_refs(folio, nr_pages - 1);
nr_pages = 1; // sets nr_pages to 1 as it can be represented as a single folio page
}
}
...
यह कोड जो जाँचता है कि पेज एक ही folio से हैं, वास्तव में यह जाँच नहीं करता कि वे सतत हैं या नहीं। यह एक ही पेज हो सकता है जिसे कई बार मैप किया गया है। पुनरावृत्ति के दौरान page_folio(page) बार-बार एक ही folio लौटाएगा और जाँचों को पार करता रहेगा। यह एक स्पष्ट लॉजिक बग है। आइए io_sqe_buffer_register के साथ आगे बढ़ें और देखें कि इसका परिणाम क्या होता है।
...
imu = kvmalloc(struct_size(imu, bvec, nr_pages), GFP_KERNEL);
// allocates imu with an array for nr_pages bio_vec(s)
// bio_vec - a contiguous range of physical memory addresses
// we need a bio_vec for each (physical) page
// in the case of a folio - the array of bio_vec(s) will be of size 1
if (!imu)
goto done;
ret = io_buffer_account_pin(ctx, pages, nr_pages, imu, last_hpage);
if (ret) {
unpin_user_pages(pages, nr_pages);
goto done;
}
off = (unsigned long) iov->iov_base & ~PAGE_MASK;
size = iov->iov_len; // sets the size to that passed by the user!
/* store original address for later verification */
imu->ubuf = (unsigned long) iov->iov_base; // user-controlled
imu->ubuf_end = imu->ubuf + iov->iov_len; // calculates the end based on the length
imu->nr_bvecs = nr_pages; // this would be 1 in the case of folio
*pimu = imu;
ret = 0;
if (folio) { // in case of folio - we need just a single bio_vec (efficiant!)
bvec_set_page(&imu->bvec[0], pages[0], size, off);
goto done;
}
for (i = 0; i < nr_pages; i++) {
size_t vec_len;
vec_len = min_t(size_t, size, PAGE_SIZE - off);
bvec_set_page(&imu->bvec[i], pages[i], vec_len, off);
off = 0;
size -= vec_len;
}
done:
if (ret)
kvfree(imu);
kvfree(pages);
return ret;
}