-
Notifications
You must be signed in to change notification settings - Fork 7
Expand file tree
/
Copy pathsyscall.rs
More file actions
1484 lines (1352 loc) · 64.8 KB
/
Copy pathsyscall.rs
File metadata and controls
1484 lines (1352 loc) · 64.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
//! SYSCALL/SYSRET 系统调用入口配置
//!
//! 配置 x86_64 的快速系统调用机制,包括:
//! - IA32_STAR: 内核/用户代码段选择子
//! - IA32_LSTAR: 系统调用入口点地址
//! - IA32_SFMASK: RFLAGS 掩码
//! - IA32_EFER: 启用 SYSCALL/SYSRET 扩展
//!
//! # Phase 6: User Space Support
//!
//! 这是实现 Ring 3 用户态支持的关键组件。
//!
//! ## SYSCALL 寄存器约定
//!
//! 用户态调用 SYSCALL 时:
//! - RAX: 系统调用号
//! - RDI: arg0, RSI: arg1, RDX: arg2
//! - R10: arg3 (不是 RCX,因为 SYSCALL 会覆盖它)
//! - R8: arg4, R9: arg5
//!
//! SYSCALL 指令执行后:
//! - RCX = 用户态 RIP (返回地址)
//! - R11 = 用户态 RFLAGS
//! - CS/SS 根据 STAR MSR 切换
use crate::gdt;
use core::arch::asm;
/// IA32_STAR MSR 地址
const IA32_STAR: u32 = 0xC000_0081;
/// IA32_LSTAR MSR 地址 (64-bit SYSCALL 入口点)
const IA32_LSTAR: u32 = 0xC000_0082;
/// IA32_CSTAR MSR 地址 (32-bit 兼容模式,暂不使用)
#[allow(dead_code)]
const IA32_CSTAR: u32 = 0xC000_0083;
/// IA32_SFMASK MSR 地址 (SYSCALL RFLAGS 掩码)
const IA32_SFMASK: u32 = 0xC000_0084;
/// IA32_EFER MSR 地址
const IA32_EFER: u32 = 0xC000_0080;
/// IA32_GS_BASE MSR 地址 (用户态 GS 基址)
///
/// Used with SWAPGS instruction to swap between user and kernel GS base.
/// User-space programs can set this via arch_prctl(ARCH_SET_GS).
const IA32_GS_BASE: u32 = 0xC000_0101;
/// IA32_KERNEL_GS_BASE MSR 地址 (内核态 GS 基址)
///
/// Contains the kernel's GS base address. The SWAPGS instruction atomically
/// swaps IA32_GS_BASE and IA32_KERNEL_GS_BASE. In single-core mode, this is
/// set to 0. In SMP mode, this would point to per-CPU data structures.
const IA32_KERNEL_GS_BASE: u32 = 0xC000_0102;
/// EFER.SCE 位 (System Call Extensions)
const EFER_SCE: u64 = 1 << 0;
/// RFLAGS 中断标志位
const RFLAGS_IF: u64 = 1 << 9;
/// RFLAGS 单步标志位
const RFLAGS_TF: u64 = 1 << 8;
/// RFLAGS 方向标志位
const RFLAGS_DF: u64 = 1 << 10;
/// RFLAGS AC 标志位 (SMAP)
const RFLAGS_AC: u64 = 1 << 18;
/// RFLAGS IOPL 位 (I/O 特权级)
const RFLAGS_IOPL: u64 = 0b11 << 12;
/// RFLAGS NT 位 (嵌套任务)
const RFLAGS_NT: u64 = 1 << 14;
/// RFLAGS RF 位 (恢复标志)
const RFLAGS_RF: u64 = 1 << 16;
/// 用户代码段选择子 (SYSRET/IRET 回退用)
const USER_CODE_SELECTOR: u64 = 0x23;
/// 用户数据段选择子
const USER_DATA_SELECTOR: u64 = 0x1B;
// ============================================================================
// 系统调用帧定义
// ============================================================================
/// 系统调用保存帧中的寄存器数量
const SYSCALL_FRAME_QWORDS: usize = 16;
/// 系统调用帧大小(字节)
const SYSCALL_FRAME_SIZE: usize = SYSCALL_FRAME_QWORDS * 8;
/// Scratch stack size per logical CPU.
///
/// R104-7 DOC: Each CPU gets a private 4 KiB scratch stack used exclusively
/// inside the `syscall_entry` assembly trampoline. The stack is active only
/// while maskable interrupts are disabled (SFMASK clears IF), so it cannot be
/// re-entered by another syscall. NMIs and machine-check exceptions can still
/// fire on this stack; their handlers must be minimal.
///
/// **Budget breakdown (scratch stack usage before kernel-stack switch):**
/// - Register save frame: SYSCALL_FRAME_QWORDS × 8 = 128 bytes
/// - `cld` + nested-syscall detection (`lock bts`): negligible stack
/// - `call get_rsp0`: one return-address push = 8 bytes
/// - **Total estimated:** ~136 bytes ⇒ headroom ≈ 3.9 KiB
///
/// Note: The FXSAVE area (512 bytes) and the duplicated register frame are
/// allocated on the **kernel stack** (TSS RSP0), not here. See the assembly
/// at phase 4 of `syscall_entry` for the kernel-stack layout.
///
/// If additional work is added to the assembly trampoline before the kernel-
/// stack switch (e.g. shadow-stack CET verification), revisit this budget.
const SYSCALL_SCRATCH_SIZE: usize = 4096;
/// FPU/SIMD 保存区大小(FXSAVE/FXRSTOR 需要 512 字节且 16 字节对齐)
/// Z-1 fix: 用于在 syscall 路径中保存/恢复用户态 FPU 状态
const FPU_SAVE_AREA_SIZE: usize = 512;
// 帧内各寄存器的偏移量
const OFF_RAX: usize = 0; // 系统调用号 / 返回值
const OFF_RCX: usize = 8; // 用户 RIP
const OFF_RDX: usize = 16; // arg2
const OFF_RBX: usize = 24; // callee-saved
const OFF_RSP: usize = 32; // 用户 RSP
const OFF_RBP: usize = 40; // callee-saved
const OFF_RSI: usize = 48; // arg1
const OFF_RDI: usize = 56; // arg0
const OFF_R8: usize = 64; // arg4
const OFF_R9: usize = 72; // arg5
const OFF_R10: usize = 80; // arg3
const OFF_R11: usize = 88; // 用户 RFLAGS
const OFF_R12: usize = 96; // callee-saved
const OFF_R13: usize = 104; // callee-saved
const OFF_R14: usize = 112; // callee-saved
const OFF_R15: usize = 120; // callee-saved
/// 对齐的栈存储(确保 16 字节对齐满足 ABI 要求)
#[derive(Clone, Copy)]
#[repr(C, align(16))]
struct AlignedStack<const N: usize>([u8; N]);
// ============================================================================
// R23-2 fix: Per-CPU syscall 临时数据
// ============================================================================
// 将原来的全局变量改为 per-CPU 数组,为 SMP 支持做准备。
// 当前 current_cpu_id() 总是返回 0,所以实际行为与单核相同。
// 未来启用 SMP 时,只需实现真正的 CPU ID 获取逻辑即可。
//
// **SMP 升级路径**:
// 1. 实现 current_cpu_id() 读取 APIC ID
// 2. 在汇编中通过 GS 段或 APIC ID 计算 per-CPU 偏移
// 3. 将 `lea rsp, [{scratch_stacks}]` 改为 `lea rsp, [{scratch_stacks} + cpu_id * SCRATCH_SIZE]`
/// 最大支持的 CPU 数量(必须与 cpu_local crate 保持一致)
const SYSCALL_MAX_CPUS: usize = 64;
// 编译时断言:确保 SYSCALL_MAX_CPUS 与 cpu_local::max_cpus() 一致
const _: () = {
assert!(
SYSCALL_MAX_CPUS == 64,
"SYSCALL_MAX_CPUS must match cpu_local::max_cpus()"
);
// 注意:cpu_local::max_cpus() 是 const fn,但由于跨 crate 常量引用限制,
// 这里硬编码为 64。如果 cpu_local 修改了 MAX_CPUS,需要同步更新此处。
};
/// Per-CPU scratch 栈数组
///
/// 每个 CPU 有独立的 4KB 临时栈,用于 syscall 入口时保存用户寄存器。
/// 使用 `#[no_mangle]` 以便汇编代码可以直接引用符号地址。
///
/// # Safety
///
/// - 在中断禁用状态下使用(SFMASK 清除 IF),不会重入
/// - 每个 CPU 只访问自己的 slot,通过 CPU ID 索引
/// - 数组元素继承 AlignedStack 的 16 字节对齐属性
#[no_mangle]
static mut SYSCALL_SCRATCH_STACKS: [AlignedStack<SYSCALL_SCRATCH_SIZE>; SYSCALL_MAX_CPUS] =
[AlignedStack([0; SYSCALL_SCRATCH_SIZE]); SYSCALL_MAX_CPUS];
// ============================================================================
// R67-8 FIX: GS-based per-CPU syscall metadata
// ============================================================================
// Instead of using slot 0 for all CPUs, we use GS-relative addressing.
// After SWAPGS, GS points to this CPU's SyscallPerCpu structure.
// This avoids the race condition where multiple CPUs clobber slot 0.
/// R67-8 FIX: Per-CPU syscall metadata accessible via GS segment.
///
/// After SWAPGS in syscall entry, GS base points to this structure.
/// The assembly uses `gs:[offset]` to access per-CPU data without
/// needing to compute CPU ID.
///
/// R67-11 FIX: Added `syscall_active` field to detect nested syscalls.
/// This prevents stack corruption when an interrupt handler attempts
/// to execute a syscall while one is already in progress.
///
/// # R102-L7: SWAPGS State Machine Invariant
///
/// The per-CPU GS base follows a strict two-state invariant that must be
/// maintained by ALL entry/exit paths (syscall, interrupt, NMI, MCE):
///
/// **Kernel context** (after `init_syscall_percpu` SWAPGS or on syscall entry):
/// - `IA32_GS_BASE` = `&SYSCALL_PERCPU[cpu_id]` (per-CPU data, active in kernel)
/// - `IA32_KERNEL_GS_BASE` = user GS value (swapped out, for user restore)
///
/// **User context** (after SWAPGS in `enter_usermode` or SYSRETQ path):
/// - `IA32_GS_BASE` = user GS value (active in userspace)
/// - `IA32_KERNEL_GS_BASE` = `&SYSCALL_PERCPU[cpu_id]` (swapped out, for kernel restore)
///
/// A misplaced SWAPGS in any code path corrupts this state machine, causing:
/// 1. Null dereference on per-CPU access (kernel crash)
/// 2. Per-CPU data written to user-controlled memory (privilege escalation)
/// 3. User GS pointing to kernel memory (information leak)
///
/// **Critical rule**: Every path from user→kernel must execute exactly one SWAPGS,
/// and every path from kernel→user must execute exactly one SWAPGS. NMI/MCE handlers
/// must check whether they interrupted kernel or user context before deciding whether
/// to SWAPGS (using the IST-based entry mechanism or checking saved CS RPL).
#[derive(Clone, Copy)]
#[repr(C, align(64))]
pub struct SyscallPerCpu {
/// Top of this CPU's scratch stack (pre-computed for fast access)
pub scratch_top: u64,
/// User RSP shadow - saved on syscall entry, restored on exit
pub user_rsp_shadow: u64,
/// Pointer to current syscall frame on kernel stack
pub frame_ptr: u64,
/// R67-11 FIX: Per-CPU syscall active flag (0 = idle, 1 = active).
/// Accessed atomically via `lock bts` in assembly. Plain u64
/// (not AtomicU64) to maintain Copy trait for array initialization.
pub syscall_active: u64,
// ---- H.3 KPTI: Dual-CR3 values for syscall entry/exit CR3 switching ----
//
// When KPTI is active, `kpti_kernel_cr3 != kpti_user_cr3` and the syscall
// trampoline switches CR3 on entry (user → kernel) and exit (kernel → user).
// When KPTI is inactive, both values are identical and the `cmp/je` in the
// assembly skips the `mov cr3` entirely — zero overhead on non-KPTI systems.
//
// Written by `arch_set_kpti_cr3s()` during context switch (interrupts disabled).
// Read by the syscall entry/exit assembly via GS-relative addressing.
/// CR3 to load on user→kernel transition (full kernel page tables).
pub kpti_kernel_cr3: u64,
/// CR3 to load on kernel→user transition (user page tables + trampoline).
pub kpti_user_cr3: u64,
/// Scratch slot used by the exit trampoline to preserve a user register
/// during the CR3 switch (currently: RDX).
pub kpti_tmp: u64,
/// M0 item 5: the PID that owns the current `frame_ptr`. Set by the syscall
/// dispatcher at entry (via `set_frame_owner`) and validated by the MUTABLE
/// frame accessor (`get_current_syscall_frame_mut_inner`) so a signal-delivery
/// write can never target a STALE cross-task frame. The bare `frame_ptr != 0`
/// gate is already sufficient at the syscall-return tail (the tail is always
/// preceded by THIS syscall's entry, which sets `frame_ptr`; a block zeroes it
/// via `switch_context`), but the owner check is defense-in-depth and the
/// invariant the future preemptive-IRQ-delivery slice will rely on (where the
/// hook is NOT preceded by a syscall entry). Not touched by the ASM.
pub frame_owner_pid: u64,
/// R172-04 FIX: the FS_BASE / KERNEL_GS_BASE that the CURRENTLY-running task must
/// have on the MSRs when it returns to ring 3. STAGED (in Rust) by the closed
/// writer set — W1: the scheduler before `switch_context`/`switch_to_user`
/// switch-in; W3: `arch_prctl(SET_FS/SET_GS)`; W4: exec TLS reset — and COMMITTED
/// (in ASM) by the SYSRET fast-path epilogue with two `wrmsr`, BEFORE user
/// RCX/RDX become live and while kernel GS is still active. The scheduler also
/// installs the live pair before every incoming context, including timer IRET
/// and kernel continuations that can reblock before reaching this epilogue.
/// SYSRET keeps an idempotent boundary commit for the same staged task state.
/// CR4.FSGSBASE is never enabled, so the PCB (and hence this staged pair) is the
/// authoritative source for the task's FS/GS. Appended at the struct TAIL so the
/// existing GS-relative offsets (frame_ptr/syscall_active/kpti_*) are unchanged.
pub pending_fs_base: u64,
/// R172-04 FIX: companion to `pending_fs_base` (the user GS base, programmed into
/// IA32_KERNEL_GS_BASE so the epilogue SWAPGS makes it the active user GS).
pub pending_gs_base: u64,
}
impl SyscallPerCpu {
const fn new() -> Self {
Self {
scratch_top: 0,
user_rsp_shadow: 0,
frame_ptr: 0,
syscall_active: 0,
kpti_kernel_cr3: 0,
kpti_user_cr3: 0,
kpti_tmp: 0,
frame_owner_pid: 0,
pending_fs_base: 0,
pending_gs_base: 0,
}
}
}
/// Per-CPU syscall metadata array, indexed by CPU ID.
/// Each entry is 64-byte aligned for cache line isolation.
#[no_mangle]
static mut SYSCALL_PERCPU: [SyscallPerCpu; SYSCALL_MAX_CPUS] =
[SyscallPerCpu::new(); SYSCALL_MAX_CPUS];
/// H.3 KPTI: Update the per-CPU dual-CR3 pair used by the syscall trampoline.
///
/// Called from `activate_memory_space()` during context switch when the active
/// address space changes. The assembly entry/exit trampoline reads these values
/// via `gs:[PERCPU_KPTI_KERNEL_CR3_OFFSET]` and `gs:[PERCPU_KPTI_USER_CR3_OFFSET]`.
///
/// When KPTI is disabled (user_cr3 == kernel_cr3), the trampoline's `cmp/je` skips
/// the `mov cr3` instruction entirely — no overhead on non-KPTI configurations.
///
/// # Safety
///
/// Must be called with interrupts disabled (or from a context where no interrupt
/// handler on this CPU can observe a partially-written pair). Context switch paths
/// already satisfy this constraint.
pub fn arch_set_kpti_cr3s(user_cr3: u64, kernel_cr3: u64) {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return;
}
// SAFETY: single-writer (interrupts disabled during context switch), no aliasing.
// The syscall entry/exit assembly on this CPU reads these fields only while
// interrupts are disabled or after SFMASK has cleared IF, so there is no
// concurrent reader on the same CPU.
unsafe {
SYSCALL_PERCPU[cpu_id].kpti_kernel_cr3 = kernel_cr3;
SYSCALL_PERCPU[cpu_id].kpti_user_cr3 = user_cr3;
}
}
/// Offset of scratch_top in SyscallPerCpu (for GS-relative addressing)
///
/// R103-3 FIX: Made `pub` so `context_switch.rs` can import these offsets
/// instead of duplicating them as hard-coded literals. A single source of
/// truth prevents silent desync when fields are reordered.
pub const PERCPU_SCRATCH_TOP_OFFSET: usize = 0;
/// Offset of user_rsp_shadow in SyscallPerCpu
pub const PERCPU_USER_RSP_OFFSET: usize = 8;
/// Offset of frame_ptr in SyscallPerCpu
pub const PERCPU_FRAME_PTR_OFFSET: usize = 16;
/// R67-11 FIX: Offset of syscall_active flag in SyscallPerCpu
pub const PERCPU_SYSCALL_ACTIVE_OFFSET: usize = 24;
/// H.3 KPTI: Offset of kpti_kernel_cr3 in SyscallPerCpu
pub const PERCPU_KPTI_KERNEL_CR3_OFFSET: usize = 32;
/// H.3 KPTI: Offset of kpti_user_cr3 in SyscallPerCpu
pub const PERCPU_KPTI_USER_CR3_OFFSET: usize = 40;
/// H.3 KPTI: Offset of kpti_tmp scratch slot in SyscallPerCpu
pub const PERCPU_KPTI_TMP_OFFSET: usize = 48;
/// M0 item 5: Offset of frame_owner_pid in SyscallPerCpu (Rust-only, no ASM use).
pub const PERCPU_FRAME_OWNER_OFFSET: usize = 56;
/// R172-04: Offset of pending_fs_base (GS-relative; read by the SYSRET epilogue wrmsr).
pub const PERCPU_PENDING_FS_BASE_OFFSET: usize = 64;
/// R172-04: Offset of pending_gs_base (GS-relative; read by the SYSRET epilogue wrmsr).
pub const PERCPU_PENDING_GS_BASE_OFFSET: usize = 72;
// R103-3 FIX: Compile-time assertions that the offsets match the struct layout.
// If SyscallPerCpu is reordered, these will produce a build error.
const _: () = {
assert!(
core::mem::offset_of!(SyscallPerCpu, scratch_top) == PERCPU_SCRATCH_TOP_OFFSET,
"PERCPU_SCRATCH_TOP_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, user_rsp_shadow) == PERCPU_USER_RSP_OFFSET,
"PERCPU_USER_RSP_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, frame_ptr) == PERCPU_FRAME_PTR_OFFSET,
"PERCPU_FRAME_PTR_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, syscall_active) == PERCPU_SYSCALL_ACTIVE_OFFSET,
"PERCPU_SYSCALL_ACTIVE_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, kpti_kernel_cr3) == PERCPU_KPTI_KERNEL_CR3_OFFSET,
"PERCPU_KPTI_KERNEL_CR3_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, kpti_user_cr3) == PERCPU_KPTI_USER_CR3_OFFSET,
"PERCPU_KPTI_USER_CR3_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, kpti_tmp) == PERCPU_KPTI_TMP_OFFSET,
"PERCPU_KPTI_TMP_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, frame_owner_pid) == PERCPU_FRAME_OWNER_OFFSET,
"PERCPU_FRAME_OWNER_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, pending_fs_base) == PERCPU_PENDING_FS_BASE_OFFSET,
"PERCPU_PENDING_FS_BASE_OFFSET does not match struct layout"
);
assert!(
core::mem::offset_of!(SyscallPerCpu, pending_gs_base) == PERCPU_PENDING_GS_BASE_OFFSET,
"PERCPU_PENDING_GS_BASE_OFFSET does not match struct layout"
);
};
/// R67-11 FIX: Error code for nested syscall rejection.
/// Using -EBUSY (16) to indicate the syscall layer is busy.
const SYSCALL_NESTED_ERROR: i64 = -16;
/// A SYSCALL instruction does not push a return frame; it leaves the
/// continuation RIP in RCX. Ring-3 addresses in this kernel are restricted
/// to the low canonical half, so bit 47 is a conservative origin classifier:
/// set means the entry was issued from kernel/high-half code. The trampoline
/// uses this predicate before SWAPGS and treats every positive result as a
/// fail-closed nested entry. A non-canonical RCX with bit 47 set is therefore
/// rejected too, rather than being allowed to reach a user-GS swap.
#[inline]
#[allow(dead_code)]
const fn syscall_rcx_is_kernel_origin(rcx: u64) -> bool {
(rcx & (1u64 << 47)) != 0
}
// ============================================================================
// MSR 操作
// ============================================================================
/// 读取 MSR
#[inline]
unsafe fn rdmsr(msr: u32) -> u64 {
let low: u32;
let high: u32;
asm!(
"rdmsr",
in("ecx") msr,
out("eax") low,
out("edx") high,
options(nomem, nostack, preserves_flags)
);
((high as u64) << 32) | (low as u64)
}
/// 写入 MSR
#[inline]
unsafe fn wrmsr(msr: u32, value: u64) {
let low = value as u32;
let high = (value >> 32) as u32;
asm!(
"wrmsr",
in("ecx") msr,
in("eax") low,
in("edx") high,
options(nomem, nostack, preserves_flags)
);
}
// ============================================================================
// P1-A D1-ARC-ENTRY-STATE: kernel GS enforcement
// ============================================================================
/// P1-A: bit N set after CPU N has programmed `SYSCALL_PERCPU[N]` and performed
/// the boot SWAPGS so *this* CPU runs with kernel `IA32_GS_BASE`.
///
/// Per-CPU (not a single global) so an early AP cannot false-positive against
/// BSP's ready bit before the AP's own `init_syscall_percpu` + swapgs.
static SYSCALL_GS_READY_MASK: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0);
/// P1-A: Assert that `IA32_GS_BASE` points at **this CPU's** `SYSCALL_PERCPU` slot.
///
/// After a correct CPL3→kernel entry (`swapgs` on syscall/timer Ring-3 path, or
/// already-kernel GS on CPL0 IRQ), GS must address per-CPU syscall metadata.
/// Scheduling or KPTI CR3 loads with user GS active is the R178-1 class.
///
/// S-2 (D3-ARC-GS-ASSERT): the check is exact-slot, not array-range. A foreign
/// CPU's slot inside `SYSCALL_PERCPU` would still mean cross-CPU state
/// corruption on schedule, so range membership alone is insufficient.
///
/// No-op until **this** CPU has completed `init_syscall_percpu` (bit in
/// `SYSCALL_GS_READY_MASK`).
#[inline]
pub fn assert_kernel_gs_base() {
use core::sync::atomic::Ordering;
let cpu = cpu_local::current_cpu_id();
if cpu >= 64 {
return;
}
let mask = SYSCALL_GS_READY_MASK.load(Ordering::Acquire);
if (mask & (1u64 << cpu)) == 0 {
return;
}
let gs_base = unsafe { rdmsr(IA32_GS_BASE) };
assert!(
gs_base_is_exact_slot(gs_base, cpu),
"P1-A ENTRY-STATE: CPU {} IA32_GS_BASE=0x{:x} != exact SYSCALL_PERCPU slot 0x{:x} — \
missing swapgs / enter_kernel_state before schedule, or foreign-CPU GS slot?",
cpu,
gs_base,
expected_gs_slot(cpu)
);
}
/// S-2: the enforcement predicate — `gs_base` must be **exactly** this CPU's
/// `SYSCALL_PERCPU` slot. Factored out so the negative self-test exercises the
/// same predicate the production assertion uses (an array-range check would
/// wrongly return true for a foreign CPU's slot).
#[inline]
fn gs_base_is_exact_slot(gs_base: u64, cpu: usize) -> bool {
gs_base == expected_gs_slot(cpu)
}
/// S-2: address of the exact `SYSCALL_PERCPU[cpu]` slot this CPU must have in
/// `IA32_GS_BASE` after kernel entry.
///
/// # Panics
/// Panics if `cpu >= SYSCALL_MAX_CPUS`; callers gate on the ready mask first.
#[inline]
fn expected_gs_slot(cpu: usize) -> u64 {
assert!(
cpu < SYSCALL_MAX_CPUS,
"CPU index out of SYSCALL_PERCPU range"
);
// SAFETY: raw address computation only (addr_of! creates no reference to
// the mutable static); SYSCALL_PERCPU has fixed layout for the kernel's life.
unsafe {
(core::ptr::addr_of!(SYSCALL_PERCPU) as u64)
+ (core::mem::size_of::<SyscallPerCpu>() as u64) * (cpu as u64)
}
}
/// Boot self-check: kernel GS must already be active after `init_syscall_percpu(0)`.
///
/// S-2 negative probe: an adjacent CPU's slot must NOT satisfy the exact-slot
/// predicate — this pins the D3-ARC-GS-ASSERT hardening (exact ownership, not
/// array-range membership).
pub fn run_entry_state_gs_self_test() {
assert_kernel_gs_base();
let cpu = cpu_local::current_cpu_id();
if cpu + 1 < SYSCALL_MAX_CPUS {
// Feed the adjacent slot's address through the SAME enforcement
// predicate: it lies inside SYSCALL_PERCPU, so a regressed
// array-range check would accept it; the exact-slot check must not.
let foreign = expected_gs_slot(cpu + 1);
assert!(
!gs_base_is_exact_slot(foreign, cpu),
"S-2 GS self-test: adjacent-slot address must fail the exact-slot predicate"
);
}
}
/// 系统调用入口是否已初始化
/// R102-L3 FIX: Use AtomicBool instead of static mut to prevent data races
/// if multiple CPUs attempt concurrent init_syscall_msr calls.
static SYSCALL_INITIALIZED: core::sync::atomic::AtomicBool =
core::sync::atomic::AtomicBool::new(false);
/// 初始化 SYSCALL/SYSRET MSR
///
/// 配置快速系统调用机制,使用户态程序可以通过 SYSCALL 指令进入内核。
///
/// # Arguments
///
/// * `syscall_entry` - 系统调用入口函数地址(汇编存根)
///
/// # Safety
///
/// - 必须在每个 CPU 上调用,且在该 CPU 的 GDT 加载之后
/// (STAR/LSTAR/SFMASK 与 EFER.SCE 均为 per-CPU MSR,BSP 与每个 AP 都必须各自配置)
/// - syscall_entry 必须是有效的系统调用处理程序地址
///
/// # STAR MSR 布局 (64-bit 模式)
///
/// ```text
/// bits 63:48 = 用户代码段选择子基址(SYSRET 加载 CS = 此值 + 16, SS = 此值 + 8)
/// bits 47:32 = 内核代码段选择子(SYSCALL 加载 CS = 此值, SS = 此值 + 8)
/// bits 31:0 = 保留(32-bit 模式使用)
/// ```
pub unsafe fn init_syscall_msr(syscall_entry: u64) {
// R171-G1-01 FIX: STAR/LSTAR/SFMASK and EFER.SCE are PER-CPU MSRs — every
// logical CPU (the BSP and each AP) MUST program its own copy. The previous
// global `SYSCALL_INITIALIZED` early-return ran the MSR programming only on
// the first (BSP) call and skipped it on every AP, leaving APs with a
// reset-value LSTAR (=0): a `syscall` executed on an AP then transferred to
// RIP 0 in ring 0 (and SYSRET loaded reset-value selectors). The writes
// below are per-CPU (wrmsr affects only the executing CPU) and idempotent,
// so they now run UNCONDITIONALLY on every CPU. `SYSCALL_INITIALIZED` is kept
// only to (a) back `is_initialized()` and (b) gate the one-time, KASLR-
// sensitive MSR dump so the LSTAR address is not logged once per CPU.
let sel = gdt::selectors();
// 获取选择子的原始值(不含 RPL)
let kernel_cs = sel.kernel_code.0 as u64;
let user_data = sel.user_data.0 as u64;
// STAR 布局计算:
// SYSRET (64-bit): CS = STAR[63:48] + 16, SS = STAR[63:48] + 8
// 目标:CS = 0x23 (user_code | RPL=3), SS = 0x1b (user_data | RPL=3)
// 计算:STAR[63:48] = 0x23 - 16 = 0x13 = (user_data - 8) | 3
let sysret_base = (user_data - 8) | 3;
let star_value = (sysret_base << 48) | (kernel_cs << 32);
// 写入 STAR
wrmsr(IA32_STAR, star_value);
// 写入 LSTAR (系统调用入口点)
wrmsr(IA32_LSTAR, syscall_entry);
// 写入 SFMASK (SYSCALL 时清除的 RFLAGS 位)
// 清除 IF/TF/DF/AC 以及 IOPL/NT/RF,防止特权/调试位带入内核
let sfmask =
RFLAGS_IF | RFLAGS_TF | RFLAGS_DF | RFLAGS_AC | RFLAGS_IOPL | RFLAGS_NT | RFLAGS_RF;
wrmsr(IA32_SFMASK, sfmask);
// R67-8 FIX: GS base initialization moved to init_syscall_percpu()
// which is called per-CPU after stack allocation. The kernel GS base
// will point to each CPU's SyscallPerCpu structure for GS-relative
// addressing in syscall entry/exit.
// Note: For BSP, init_syscall_percpu(0) must be called after this function.
// 启用 EFER.SCE (System Call Extensions)
let efer = rdmsr(IA32_EFER);
wrmsr(IA32_EFER, efer | EFER_SCE);
// R171-G1-01 FIX: publish "first CPU done" exactly once. The CAS both backs
// `is_initialized()` and selects the single CPU allowed to emit the one-time
// dump/log below — so the per-CPU re-programming never re-logs the KASLR-
// sensitive LSTAR address (debug) or spams the line once per AP (release).
let first_cpu = SYSCALL_INITIALIZED
.compare_exchange(
false,
true,
core::sync::atomic::Ordering::AcqRel,
core::sync::atomic::Ordering::Acquire,
)
.is_ok();
// R102-7 FIX: Gate LSTAR/STAR/SFMASK values behind debug_assertions.
// These contain kernel code addresses that defeat KASLR if leaked
// to serial console or log output in production builds.
#[cfg(debug_assertions)]
if first_cpu {
kprintln!("SYSCALL MSR initialized:");
kprintln!(" STAR: 0x{:016x}", star_value);
kprintln!(" LSTAR: 0x{:016x}", syscall_entry);
kprintln!(" SFMASK: 0x{:016x}", sfmask);
kprintln!(
" Kernel CS: 0x{:x}, SYSRET base: 0x{:x}",
kernel_cs,
sysret_base
);
}
#[cfg(not(debug_assertions))]
if first_cpu {
klog_always!("SYSCALL MSR initialized");
}
}
/// 检查 SYSCALL/SYSRET 是否已初始化
pub fn is_initialized() -> bool {
SYSCALL_INITIALIZED.load(core::sync::atomic::Ordering::Acquire)
}
/// 获取当前 STAR MSR 值(调试用)
pub fn get_star() -> u64 {
unsafe { rdmsr(IA32_STAR) }
}
/// 获取当前 LSTAR MSR 值(调试用)
pub fn get_lstar() -> u64 {
unsafe { rdmsr(IA32_LSTAR) }
}
// ============================================================================
// Syscall 帧访问(供 clone/fork 使用)
// ============================================================================
/// Syscall 帧结构(与汇编保存布局一致)
///
/// 这个结构体表示 syscall_entry_stub 保存到内核栈上的寄存器帧。
/// 布局必须与汇编中的偏移量完全匹配。
#[repr(C)]
#[derive(Debug, Clone, Copy)]
pub struct SyscallFrame {
pub rax: u64, // 0x00: 系统调用号 / 返回值
pub rcx: u64, // 0x08: 用户 RIP (syscall 保存)
pub rdx: u64, // 0x10: arg2
pub rbx: u64, // 0x18: callee-saved
pub rsp: u64, // 0x20: 用户 RSP
pub rbp: u64, // 0x28: callee-saved
pub rsi: u64, // 0x30: arg1
pub rdi: u64, // 0x38: arg0
pub r8: u64, // 0x40: arg4
pub r9: u64, // 0x48: arg5
pub r10: u64, // 0x50: arg3
pub r11: u64, // 0x58: 用户 RFLAGS (syscall 保存)
pub r12: u64, // 0x60: callee-saved
pub r13: u64, // 0x68: callee-saved
pub r14: u64, // 0x70: callee-saved
pub r15: u64, // 0x78: callee-saved
}
/// 获取当前 CPU 的 syscall 帧指针
///
/// 在 syscall 处理期间调用,返回指向当前 syscall 帧的指针。
/// 用于 clone/fork 读取调用者的寄存器状态。
///
/// # Safety
///
/// 只能在 syscall 处理器内部调用(即 syscall_dispatcher 执行期间)。
/// 在 syscall 处理结束后,返回的指针将无效。
///
/// # R102-6 FIX: Lifetime Containment
///
/// The underlying data is only valid for the duration of the active syscall on
/// this CPU. This function is private (`fn`, not `pub fn`) and exists solely as
/// the registered callback for `kernel_core::register_syscall_frame_callback`.
/// External callers MUST use `with_current_syscall_frame()` which contains the
/// reference inside a closure, preventing escape.
///
/// # Returns
///
/// 返回 Some(&SyscallFrame) 如果在 syscall 上下文中,否则返回 None
fn get_current_syscall_frame_inner() -> Option<&'static kernel_core::SyscallFrame> {
// R67-8 FIX: Use current_cpu_id() instead of hardcoded slot 0
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return None;
}
unsafe {
let ptr = SYSCALL_PERCPU[cpu_id].frame_ptr;
if ptr == 0 {
None
} else {
// 类型转换安全:arch::SyscallFrame 和 kernel_core::SyscallFrame 布局完全相同
Some(&*(ptr as *const kernel_core::SyscallFrame))
}
}
}
/// R102-6 FIX: Execute a closure with the current syscall frame.
///
/// This is the public API for accessing the syscall frame during a syscall handler.
/// The closure-based design prevents the frame reference from escaping beyond the
/// active syscall window, eliminating the use-after-free hazard of the old
/// `get_current_syscall_frame() -> Option<&'static SyscallFrame>` API.
///
/// # Arguments
///
/// * `f` - Closure that receives `&SyscallFrame` and returns a value of type `R`.
///
/// # Returns
///
/// `Some(R)` if called during an active syscall on this CPU, `None` otherwise.
pub fn with_current_syscall_frame<F, R>(f: F) -> Option<R>
where
F: FnOnce(&kernel_core::SyscallFrame) -> R,
{
get_current_syscall_frame_inner().map(f)
}
/// M0 item 5: record the PID that owns the current syscall frame.
///
/// Called by the kernel_core syscall dispatcher at entry (via a registered callback)
/// so the MUTABLE frame accessor can reject a STALE cross-task `frame_ptr`. Runs on
/// the per-CPU GS structure with interrupts effectively disabled for the relevant
/// window (the dispatcher runs in the task's own context); a plain write is safe
/// (single writer per CPU, no concurrent reader on the same CPU).
fn set_frame_owner_inner(pid: u64) {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return;
}
// SAFETY: per-CPU slot, single-writer in this CPU's syscall context.
unsafe {
SYSCALL_PERCPU[cpu_id].frame_owner_pid = pid;
}
}
/// M0 item 5 (1b-2): read this CPU's raw `(frame_ptr, frame_owner_pid)` binding.
///
/// Used at the syscall dispatcher entry to SNAPSHOT the live frame binding into the
/// PCB so a blocked-and-resumed return tail can republish it (the per-CPU `frame_ptr`
/// is zeroed by `switch_context` on a block). Returns `(0, 0)` for an out-of-range
/// CPU index. Reads only — never mutates per-CPU state.
fn get_frame_binding_inner() -> (u64, u64) {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return (0, 0);
}
// SAFETY: per-CPU slot, this CPU's own syscall context (single reader/writer).
unsafe {
(
SYSCALL_PERCPU[cpu_id].frame_ptr,
SYSCALL_PERCPU[cpu_id].frame_owner_pid,
)
}
}
/// M0 item 5 (1b-2): REPUBLISH this CPU's `(frame_ptr, frame_owner_pid)` binding.
///
/// Called by `maybe_deliver_signal` from a PCB-saved binding when a blocked syscall
/// has resumed (its per-CPU `frame_ptr` was zeroed on the block). Writes BOTH fields
/// together so the owner-checked mutable accessor sees a consistent pair. The signal
/// delivery path runs in the task's own syscall context on this CPU with no concurrent
/// same-CPU reader.
///
/// LEAK BOUND (NOT "zeroed before any return"): the republished pointer is re-zeroed by
/// the ordinary syscall epilogue and by every `switch_context` switch-out. The dispatcher
/// tail's post-delivery `reschedule_if_needed` MAY switch away first via the
/// `save_context`+`enter_usermode` path, which does NOT zero `frame_ptr` — so a republished
/// value can briefly outlive this CPU's switch. That is harmless: the owner-checked MUTABLE
/// accessor rejects it for any other PID, and the only un-owner-checked reader
/// (`get_current_syscall_frame_inner`, used by clone/fork) runs ONLY mid-syscall, where the
/// entry ASM has already overwritten `frame_ptr` with that task's own live frame.
fn set_frame_binding_inner(frame_ptr: u64, owner_pid: u64) {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return;
}
// SAFETY: per-CPU slot, single-writer in this CPU's syscall context.
unsafe {
SYSCALL_PERCPU[cpu_id].frame_ptr = frame_ptr;
SYSCALL_PERCPU[cpu_id].frame_owner_pid = owner_pid;
}
}
/// R172-04 FIX: stage the FS/GS bases the currently-running task must have on the MSRs
/// at its next ring-3 return. The SYSRET fast-path epilogue commits these via two
/// `wrmsr` (and the enter_usermode/switch_to_user IRETQ paths write the MSRs directly).
/// Single-writer per CPU (scheduler switch-in W1 runs with IRQs off; arch_prctl W3 /
/// exec W4 run in the task's own syscall context on this CPU). Exposed BOTH as a pub fn
/// (the `sched` crate calls it directly for W1) AND registered as a `kernel_core`
/// callback (W3/W4, since `kernel_core` cannot depend on `arch`).
pub fn stage_pending_tls_bases(fs_base: u64, gs_base: u64) {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return;
}
// SAFETY: per-CPU slot, single-writer on this CPU; the epilogue reads it with IRQs
// off / kernel GS active.
unsafe {
SYSCALL_PERCPU[cpu_id].pending_fs_base = fs_base;
SYSCALL_PERCPU[cpu_id].pending_gs_base = gs_base;
}
}
/// M0 item 5: MUTABLE access to the current syscall frame, for signal-handler
/// delivery (it redirects RIP/RSP/args to the handler).
///
/// Returns the raw frame pointer ONLY when ALL hold: the CPU index is valid, the GS
/// `frame_ptr` is non-zero (a live syscall frame is present — zeroed on a blocking
/// `switch_context`), AND the recorded `frame_owner_pid` equals `expected_pid` (the
/// caller's `current_pid()`). The owner check rejects a stale `frame_ptr` left by the
/// Ring-3 schedule path (`save_context` + `enter_usermode`, which does NOT zero it);
/// at the syscall-return tail the bare non-zero check already suffices, but the owner
/// check is the invariant a future preemptive-IRQ-delivery slice will rely on.
///
/// # Safety contract for the caller
/// The returned pointer is the live kernel-stack `SyscallFrame`; it is valid only for
/// the duration of the active syscall on this CPU (the same lifetime contract as
/// `with_current_syscall_frame`). The FPU save area is the 512 bytes immediately
/// ABOVE the frame (`frame_ptr + SYSCALL_FRAME_SIZE`), unconditionally `fxsave64`d on
/// entry and `fxrstor64`d on exit.
fn get_current_syscall_frame_mut_inner(
expected_pid: u64,
) -> Option<*mut kernel_core::SyscallFrame> {
let cpu_id = cpu_local::current_cpu_id();
if cpu_id >= SYSCALL_MAX_CPUS {
return None;
}
unsafe {
let ptr = SYSCALL_PERCPU[cpu_id].frame_ptr;
let owner = SYSCALL_PERCPU[cpu_id].frame_owner_pid;
if ptr == 0 || owner != expected_pid {
None
} else {
Some(ptr as *mut kernel_core::SyscallFrame)
}
}
}
/// 注册 syscall 帧回调到 kernel_core
///
/// 在 syscall 初始化时调用,让 kernel_core 能访问当前 syscall 帧
pub fn register_frame_callback() {
kernel_core::register_syscall_frame_callback(get_current_syscall_frame_inner);
// M0 item 5: register the mutable-frame accessor + the owner-setter used by the
// syscall-return signal-delivery path.
kernel_core::syscall::register_syscall_frame_mut_callback(get_current_syscall_frame_mut_inner);
kernel_core::syscall::register_set_frame_owner_callback(set_frame_owner_inner);
// M0 item 5 (1b-2): the raw (frame_ptr, owner) get/set used to snapshot the live
// binding at syscall entry and republish it on the blocked-resume return tail.
kernel_core::syscall::register_get_frame_binding_callback(get_frame_binding_inner);
kernel_core::syscall::register_set_frame_binding_callback(set_frame_binding_inner);
// R172-04: the FS/GS staging writer used by arch_prctl (W3) and exec TLS reset (W4)
// in kernel_core (which cannot depend on arch directly).
kernel_core::syscall::register_stage_pending_tls_bases_callback(stage_pending_tls_bases);
}
/// R67-8 FIX: Initialize per-CPU syscall metadata and kernel GS base.
///
/// Must be called once per CPU after its scratch stacks are allocated.
/// This sets up the GS-based per-CPU addressing used in syscall entry.
///
/// # Arguments
///
/// * `cpu_id` - Logical CPU index (0 = BSP, 1+ = APs)
///
/// # Safety
///
/// - Must be called with interrupts disabled
/// - Must be called only once per CPU
/// - Must be called after SYSCALL_SCRATCH_STACKS is initialized
pub unsafe fn init_syscall_percpu(cpu_id: usize) {
assert!(
cpu_id < SYSCALL_MAX_CPUS,
"CPU ID {} out of range for syscall per-CPU init",
cpu_id
);
// Pre-compute scratch stack top for this CPU
let scratch_base = SYSCALL_SCRATCH_STACKS[cpu_id].0.as_ptr() as usize;
SYSCALL_PERCPU[cpu_id].scratch_top = (scratch_base + SYSCALL_SCRATCH_SIZE) as u64;
SYSCALL_PERCPU[cpu_id].user_rsp_shadow = 0;
SYSCALL_PERCPU[cpu_id].frame_ptr = 0;
// R67-11 FIX: Initialize syscall active flag to 0 (idle)
SYSCALL_PERCPU[cpu_id].syscall_active = 0;
// R172-04: no staged TLS at init (kernel has FS/GS base 0 until a user task runs).
SYSCALL_PERCPU[cpu_id].pending_fs_base = 0;
SYSCALL_PERCPU[cpu_id].pending_gs_base = 0;
// Program kernel GS base to point to this CPU's SyscallPerCpu
// After SWAPGS, GS:0 will point to SYSCALL_PERCPU[cpu_id]
let percpu_addr = &SYSCALL_PERCPU[cpu_id] as *const SyscallPerCpu as u64;
wrmsr(IA32_KERNEL_GS_BASE, percpu_addr);
// User GS base starts at 0 (user can set it via arch_prctl)
wrmsr(IA32_GS_BASE, 0);
// R100-2 FIX (double-fault regression): 立即执行 SWAPGS,使内核从启动
// 开始就运行在 "post-SWAPGS" 状态:
// IA32_GS_BASE = 内核 per-CPU 指针(内核态活跃使用)
// IA32_KERNEL_GS_BASE = 0(将由调度器写入用户 GS 值)
//
// 若不执行此 SWAPGS,首次 enter_usermode() 路径中的 SWAPGS 会将两个
// MSR 从 (GS_BASE=0, KERNEL_GS_BASE=0) 交换为 (0, 0),导致后续
// SYSCALL 入口的 SWAPGS 无法恢复 per-CPU 指针,引发 DOUBLE FAULT。
asm!("swapgs", options(nostack, preserves_flags));
// P1-A: mark THIS CPU ready for GS entry-state checks (per-CPU bit).
if cpu_id < 64 {
SYSCALL_GS_READY_MASK.fetch_or(1u64 << cpu_id, core::sync::atomic::Ordering::Release);
}
// Wire the checker into kernel_core schedule entry (kernel_core holds a fn
// pointer; arch registers it here — Once is idempotent across BSP/APs).
kernel_core::scheduler_hook::register_kernel_gs_assert(assert_kernel_gs_base);
}
// ============================================================================
// C ABI 辅助函数
// ============================================================================
/// 获取内核栈顶(TSS RSP0)
///
/// # Safety
///
/// 仅由 syscall_entry_stub 汇编代码调用
#[no_mangle]
extern "C" fn syscall_get_kernel_rsp0() -> u64 {
gdt::get_kernel_stack().as_u64()
}
/// 系统调用分发器桥接
///
/// 将 C ABI 调用转发到 Rust 的 syscall_dispatcher
///
/// # Safety
///
/// 仅由 syscall_entry_stub 汇编代码调用
#[no_mangle]
extern "C" fn syscall_dispatcher_bridge(
syscall_num: u64,
arg0: u64,
arg1: u64,
arg2: u64,
arg3: u64,
arg4: u64,
arg5: u64,
) -> i64 {
kernel_core::syscall::syscall_dispatcher(syscall_num, arg0, arg1, arg2, arg3, arg4, arg5)
}
// ============================================================================
// 系统调用入口点
// ============================================================================
/// 系统调用入口点
///
/// 处理从用户态通过 SYSCALL 指令进入内核的情况。
///
/// ## 执行流程
///
/// 1. 保存用户 RSP 到暂存区
/// 2. 切换到临时栈