-
Notifications
You must be signed in to change notification settings - Fork 1
/
syms.c
445 lines (378 loc) · 10.4 KB
/
syms.c
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
#include "syms.h"
/* ================================ START unmodified functions from madvise.c ================================ */
static bool
madvise_behavior_valid(int behavior)
{
switch (behavior) {
case MADV_DOFORK:
case MADV_DONTFORK:
case MADV_NORMAL:
case MADV_SEQUENTIAL:
case MADV_RANDOM:
case MADV_REMOVE:
case MADV_WILLNEED:
case MADV_DONTNEED:
case MADV_FREE:
case MADV_COLD:
case MADV_PAGEOUT:
case MADV_POPULATE_READ:
case MADV_POPULATE_WRITE:
#ifdef CONFIG_KSM
case MADV_MERGEABLE:
case MADV_UNMERGEABLE:
#endif
#ifdef CONFIG_TRANSPARENT_HUGEPAGE
case MADV_HUGEPAGE:
case MADV_NOHUGEPAGE:
#endif
case MADV_DONTDUMP:
case MADV_DODUMP:
case MADV_WIPEONFORK:
case MADV_KEEPONFORK:
#ifdef CONFIG_MEMORY_FAILURE
case MADV_SOFT_OFFLINE:
case MADV_HWPOISON:
#endif
return true;
default:
return false;
}
}
static int madvise_inject_error(int behavior,
unsigned long start, unsigned long end)
{
unsigned long size;
if (!capable(CAP_SYS_ADMIN))
return -EPERM;
for (; start < end; start += size) {
unsigned long pfn;
struct page *page;
int ret;
ret = get_user_pages_fast(start, 1, 0, &page);
if (ret != 1)
return ret;
pfn = page_to_pfn(page);
/*
* When soft offlining hugepages, after migrating the page
* we dissolve it, therefore in the second loop "page" will
* no longer be a compound page.
*/
size = page_size(compound_head(page));
if (behavior == MADV_SOFT_OFFLINE) {
pr_info("Soft offlining pfn %#lx at process virtual address %#lx\n",
pfn, start);
ret = soft_offline_page_sym(pfn, MF_COUNT_INCREASED);
} else {
pr_info("Injecting memory failure for pfn %#lx at process virtual address %#lx\n",
pfn, start);
ret = memory_failure(pfn, MF_COUNT_INCREASED);
}
if (ret)
return ret;
}
return 0;
}
static int madvise_need_mmap_write(int behavior)
{
switch (behavior) {
case MADV_REMOVE:
case MADV_WILLNEED:
case MADV_DONTNEED:
case MADV_COLD:
case MADV_PAGEOUT:
case MADV_FREE:
case MADV_POPULATE_READ:
case MADV_POPULATE_WRITE:
return 0;
default:
/* be safe, default to 1. list exceptions explicitly */
return 1;
}
}
static
int madvise_walk_vmas(struct mm_struct *mm, unsigned long start,
unsigned long end, unsigned long arg,
int (*visit)(struct vm_area_struct *vma,
struct vm_area_struct **prev, unsigned long start,
unsigned long end, unsigned long arg, struct mmu_gather *tlb),
struct mmu_gather *tlb)
{
struct vm_area_struct *vma;
struct vm_area_struct *prev;
unsigned long tmp;
int unmapped_error = 0;
/*
* If the interval [start,end) covers some unmapped address
* ranges, just ignore them, but return -ENOMEM at the end.
* - different from the way of handling in mlock etc.
*/
vma = find_vma_prev_sym(mm, start, &prev);
if (vma && start > vma->vm_start)
prev = vma;
for (;;) {
int error;
/* Still start < end. */
if (!vma)
return -ENOMEM;
/* Here start < (end|vma->vm_end). */
if (start < vma->vm_start) {
unmapped_error = -ENOMEM;
start = vma->vm_start;
if (start >= end)
break;
}
/* Here vma->vm_start <= start < (end|vma->vm_end) */
tmp = vma->vm_end;
if (end < tmp)
tmp = end;
/* Here vma->vm_start <= start < tmp <= (end|vma->vm_end). */
error = visit(vma, &prev, start, tmp, arg, tlb);
if (error)
return error;
start = tmp;
if (prev && start < prev->vm_end)
start = prev->vm_end;
if (start >= end)
break;
if (prev)
vma = prev->vm_next;
else /* madvise_remove dropped mmap_lock */
vma = find_vma(mm, start);
}
return unmapped_error;
}
/* ================================ END unmodified functions from madvise.c ================================ */
static inline void
custom_mmu_notifier_invalidate_range_start(struct mmu_notifier_range *range)
{
might_sleep();
lock_map_acquire(&__mmu_notifier_invalidate_range_start_map);
if (mm_has_notifiers(range->mm)) {
range->flags |= MMU_NOTIFIER_RANGE_BLOCKABLE;
__mmu_notifier_invalidate_range_start_sym(range);
}
lock_map_release(&__mmu_notifier_invalidate_range_start_map);
}
static inline void
custom_mmu_notifier_invalidate_range_end(struct mmu_notifier_range *range)
{
if (mmu_notifier_range_blockable(range))
might_sleep();
if (mm_has_notifiers(range->mm))
__mmu_notifier_invalidate_range_end_sym(range, false);
}
void zap_page_range_noflush(struct vm_area_struct *vma, unsigned long start,
unsigned long size, struct mmu_gather* tlb)
{
struct mmu_notifier_range range;
if (!tlb)
return;
lru_add_drain_sym();
mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, vma, vma->vm_mm,
start, start + size);
update_hiwater_rss(vma->vm_mm);
custom_mmu_notifier_invalidate_range_start(&range);
for ( ; vma && vma->vm_start < range.end; vma = vma->vm_next)
unmap_single_vma_sym(tlb, vma, start, range.end, NULL);
custom_mmu_notifier_invalidate_range_end(&range);
}
static long custom_madvise_dontneed_single_vma(struct vm_area_struct *vma,
unsigned long start, unsigned long end, struct mmu_gather *tlb)
{
if (tlb)
zap_page_range_noflush(vma, start, end - start, tlb);
else
zap_page_range_sym(vma, start, end - start);
return 0;
}
static long custom_madvise_dontneed_free(struct vm_area_struct *vma,
struct vm_area_struct **prev,
unsigned long start, unsigned long end,
int behavior, struct mmu_gather *tlb)
{
/* struct mm_struct *mm = vma->vm_mm; */
*prev = vma;
if (!can_madv_lru_vma(vma))
return -EINVAL;
/* NOTE: hack */
/* if (!userfaultfd_remove(vma, start, end)) { */
/* *prev = NULL; /\* mmap_lock has been dropped, prev is stale *\/ */
/* mmap_read_lock(mm); */
/* vma = find_vma(mm, start); */
/* if (!vma) */
/* return -ENOMEM; */
/* if (start < vma->vm_start) { */
/* /\* */
/* * This "vma" under revalidation is the one */
/* * with the lowest vma->vm_start where start */
/* * is also < vma->vm_end. If start < */
/* * vma->vm_start it means an hole materialized */
/* * in the user address space within the */
/* * virtual range passed to MADV_DONTNEED */
/* * or MADV_FREE. */
/* *\/ */
/* return -ENOMEM; */
/* } */
/* if (!can_madv_lru_vma(vma)) */
/* return -EINVAL; */
/* if (end > vma->vm_end) { */
/* /\* */
/* * Don't fail if end > vma->vm_end. If the old */
/* * vma was split while the mmap_lock was */
/* * released the effect of the concurrent */
/* * operation may not cause madvise() to */
/* * have an undefined result. There may be an */
/* * adjacent next vma that we'll walk */
/* * next. userfaultfd_remove() will generate an */
/* * UFFD_EVENT_REMOVE repetition on the */
/* * end-vma->vm_end range, but the manager can */
/* * handle a repetition fine. */
/* *\/ */
/* end = vma->vm_end; */
/* } */
/* VM_WARN_ON(start >= end); */
/* } */
if (behavior == MADV_DONTNEED)
return custom_madvise_dontneed_single_vma(vma, start, end, tlb);
/* else if (behavior == MADV_FREE) */
/* return madvise_free_single_vma(vma, start, end); */
else
return -EINVAL;
}
/* NOTE hack, ignores every behavior except MADV_DONTNEED */
static int custom_madvise_vma_behavior(struct vm_area_struct *vma,
struct vm_area_struct **prev,
unsigned long start, unsigned long end,
unsigned long behavior,
struct mmu_gather *tlb)
{
if (behavior == MADV_DONTNEED)
return custom_madvise_dontneed_free(vma, prev, start, end, behavior, tlb);
else
return -EINVAL;
}
static bool
custom_process_madvise_behavior_valid(int behavior)
{
switch (behavior) {
case MADV_DONTNEED:
case MADV_COLD:
case MADV_PAGEOUT:
case MADV_WILLNEED:
return true;
default:
return false;
}
}
int custom_do_madvise(struct mm_struct *mm, unsigned long start, size_t len_in, int behavior, struct mmu_gather *tlb)
{
unsigned long end;
int error;
int write;
size_t len;
struct blk_plug plug;
start = untagged_addr(start);
if (!madvise_behavior_valid(behavior))
return -EINVAL;
if (!PAGE_ALIGNED(start))
return -EINVAL;
len = PAGE_ALIGN(len_in);
/* Check to see whether len was rounded up from small -ve to zero */
if (len_in && !len)
return -EINVAL;
end = start + len;
if (end < start)
return -EINVAL;
if (end == start)
return 0;
#ifdef CONFIG_MEMORY_FAILURE
if (behavior == MADV_HWPOISON || behavior == MADV_SOFT_OFFLINE)
return madvise_inject_error(behavior, start, start + len_in);
#endif
write = madvise_need_mmap_write(behavior);
if (write) {
if (mmap_write_lock_killable(mm))
return -EINTR;
} else {
mmap_read_lock(mm);
}
blk_start_plug(&plug);
error = madvise_walk_vmas(mm, start, end, behavior,
custom_madvise_vma_behavior, tlb);
blk_finish_plug(&plug);
if (write)
mmap_write_unlock(mm);
else
mmap_read_unlock(mm);
return error;
}
asmlinkage ssize_t hooked_process_madvise(struct pt_regs *regs)
{
ssize_t ret;
struct iovec iovstack[UIO_FASTIOV], iovec;
struct iovec *iov = iovstack;
struct iov_iter iter;
struct task_struct *task;
struct mm_struct *mm;
size_t total_len;
unsigned int f_flags;
int pidfd = regs->di;
const struct iovec __user *vec = (const struct iovec __user *)regs->si;
size_t vlen = regs->dx;
int behavior = regs->r10;
unsigned int flags = regs->r8;
struct mmu_gather tlb;
if (flags != 0) {
ret = -EINVAL;
goto out;
}
ret = import_iovec(READ, vec, vlen, ARRAY_SIZE(iovstack), &iov, &iter);
if (ret < 0)
goto out;
task = pidfd_get_task_sym(pidfd, &f_flags);
if (IS_ERR(task)) {
ret = PTR_ERR(task);
goto free_iov;
}
if (!custom_process_madvise_behavior_valid(behavior)) {
ret = -EINVAL;
goto release_task;
}
/* Require PTRACE_MODE_READ to avoid leaking ASLR metadata. */
mm = mm_access_sym(task, PTRACE_MODE_READ_FSCREDS);
if (IS_ERR_OR_NULL(mm)) {
ret = IS_ERR(mm) ? PTR_ERR(mm) : -ESRCH;
goto release_task;
}
/*
* Require CAP_SYS_NICE for influencing process performance. Note that
* only non-destructive hints are currently supported.
*/
if (!capable(CAP_SYS_NICE)) {
ret = -EPERM;
goto release_mm;
}
total_len = iov_iter_count(&iter);
if (behavior == MADV_DONTNEED)
tlb_gather_mmu_sym(&tlb, mm);
while (iov_iter_count(&iter)) {
iovec = iov_iter_iovec(&iter);
ret = custom_do_madvise(mm, (unsigned long)iovec.iov_base,
iovec.iov_len, behavior, &tlb);
if (ret < 0)
break;
iov_iter_advance(&iter, iovec.iov_len);
}
if (behavior == MADV_DONTNEED)
tlb_finish_mmu_sym(&tlb);
if (ret == 0)
ret = total_len - iov_iter_count(&iter);
release_mm:
mmput(mm);
release_task:
put_task_struct(task);
free_iov:
kfree(iov);
out:
return ret;
}