diff --git a/.agentplug-kv/embed-query-spill/q53ac96e6f459fd7e-62.json b/.agentplug-kv/embed-query-spill/q53ac96e6f459fd7e-62.json new file mode 100644 index 0000000000..3f6881735c --- /dev/null +++ b/.agentplug-kv/embed-query-spill/q53ac96e6f459fd7e-62.json @@ -0,0 +1 @@ +{"t":"continue to work on it until it is ready to 100% work on macos","v":[-0.04946858808398247,0.025487715378403664,0.046876657754182816,-0.03933538496494293,-0.026571940630674362,-0.018723079934716225,-0.06791770458221436,-0.0126107856631279,-0.031108194962143898,-0.004078487399965525,0.04247472807765007,-0.007026381324976683,0.0001385782816214487,0.0030778679065406322,0.03853490203619003,0.046825721859931946,0.03305750712752342,-0.0131806880235672,-0.020261315628886223,-0.03139540180563927,0.025097107514739037,-0.008632967248558998,0.008130533620715141,-0.02547999657690525,0.016314025968313217,0.056534964591264725,0.029683703556656837,0.045407913625240326,-0.02955142967402935,-0.16794580221176147,-0.06592405587434769,-0.07876431941986084,0.06155725568532944,0.02752777747809887,0.03977612778544426,0.011427327990531921,0.01845213584601879,0.03069082461297512,-0.0005880215903744102,0.02332400158047676,-0.018214818090200424,-0.015527931042015553,-0.08146390318870544,-0.02949867583811283,0.013198805041611195,0.017310231924057007,-0.0034347381442785263,-0.04373553395271301,0.058489974588155746,-0.042048171162605286,0.059852201491594315,-0.0661325752735138,0.008374461904168129,-0.013329585082828999,-0.02734672836959362,0.039165180176496506,0.023981520906090736,0.04909239709377289,0.0694822147488594,0.008217645809054375,0.04300565645098686,0.019885554909706116,-0.1594385951757431,0.12200980633497238,0.07793409377336502,0.0006684894324280322,-0.005754722747951746,-0.11679813265800476,0.016583459451794624,0.04722205549478531,-0.06634001433849335,0.01000878494232893,-0.043159663677215576,0.1357562392950058,-0.007218150421977043,0.04090677201747894,0.07693798840045929,-0.043878600001335144,-0.03642570599913597,-0.00945187360048294,-0.011600509285926819,-0.013936543837189674,0.006996828597038984,0.0171966515481472,-0.030566325411200523,0.022592434659600258,-0.002212233142927289,0.06228175014257431,0.02183816023170948,-0.05152180418372154,-0.02795330435037613,-0.06461740285158157,0.061902303248643875,-0.01942499540746212,-0.03273085132241249,0.025290818884968758,0.05477483198046684,0.026179540902376175,-0.09102677553892136,0.07964396476745605,-0.01185759250074625,0.020750651136040688,-0.008686697110533714,0.002121030120179057,0.07419070601463318,0.009341996163129807,0.04918026924133301,0.03377210348844528,-0.061722397804260254,-0.024791263043880463,0.00853219535201788,0.02713337540626526,0.024423815310001373,0.011949908919632435,0.0924558937549591,0.003913873340934515,-0.011556534096598625,0.04857795313000679,-0.013838861137628555,-0.006262852810323238,0.025906842201948166,0.09616240113973618,0.0648876428604126,0.037203215062618256,-0.03764670714735985,-0.07606244832277298,0.022048555314540863,0.05532956123352051,-0.01196358259767294,-0.05904959514737129,0.05334857851266861,-0.05414276197552681,-0.02860729582607746,0.006951197981834412,-0.010629846714437008,0.0609266571700573,-0.02714603953063488,0.006002393085509539,0.07165531069040298,-0.015640711411833763,0.005561787635087967,-0.03696354478597641,0.05593627318739891,-0.06564810872077942,0.027524137869477272,0.060470934957265854,0.021713532507419586,0.09084261953830719,-0.038887929171323776,-0.05166736617684364,0.0204810481518507,0.010492809116840363,0.0099710151553154,-0.004788126330822706,-0.045874111354351044,-0.017859039828181267,0.07836338877677917,0.06908532977104187,-0.084260493516922,-0.0002789542486425489,-0.08717773854732513,-0.05628754198551178,-0.08460607379674911,0.04024004936218262,0.03716929629445076,0.008155854418873787,0.02929210662841797,0.07225070148706436,0.008658035658299923,0.0026497882790863514,-0.018216539174318314,-0.018364373594522476,0.03715316951274872,-0.08865087479352951,0.09380093961954117,-0.07319464534521103,-0.06806664168834686,0.016478417441248894,-0.009164310060441494,0.04905083030462265,0.04338988661766052,-0.06631264090538025,-0.0037472848780453205,0.03410261496901512,0.045534905046224594,-0.049260348081588745,-0.0037340251728892326,0.012272278778254986,0.087885282933712,-0.06623309850692749,-0.0015139775350689888,-0.008504151366651058,-0.04203953966498375,0.027592992410063744,0.008541819639503956,-0.08893901109695435,-0.06157282739877701,-0.032753393054008484,0.039799705147743225,-0.005379430018365383,-0.022619163617491722,0.028924819082021713,-0.007563663646578789,0.002964625833556056,0.011976849287748337,0.027307894080877304,-0.02556869201362133,-0.030687114223837852,0.09266790747642517,0.004791646264493465,-0.03352304920554161,-0.03548455238342285,0.08440502732992172,-0.04135138541460037,-0.028292395174503326,0.03996644169092178,-0.031206650659441948,0.027756838127970695,0.05419332534074783,0.061948537826538086,-0.03506489098072052,0.0599665492773056,-0.034656357020139694,-0.1997329741716385,-0.023900141939520836,-0.07818182557821274,-0.00210705678910017,0.024007419124245644,-0.019523750990629196,0.021082697436213493,-0.03224987909197807,-0.022216714918613434,0.0044430759735405445,0.06589090824127197,-0.024901174008846283,0.016178278252482414,0.0003342722193337977,-0.02289670705795288,-0.009298020042479038,0.05608883500099182,0.007926801219582558,-0.031040746718645096,0.05050443485379219,0.0003924795310012996,0.013237126171588898,-0.12540146708488464,-0.07436218857765198,-0.007061497773975134,0.025876617059111595,0.10577334463596344,0.01778118684887886,-0.0028864911291748285,-0.08581908047199249,0.0315895676612854,0.021115154027938843,-0.04401420056819916,-0.1384372115135193,-0.021538887172937393,0.09218405187129974,0.024331072345376015,-0.03483130782842636,0.01710362173616886,0.009637303650379181,0.020128881558775902,0.11272837966680527,-0.04867664724588394,-0.10068169236183167,-0.06431397050619125,-0.02762911282479763,-0.0418139286339283,0.05515392869710922,-0.10545677691698074,-0.053053852170705795,0.0641922727227211,0.06994674354791641,0.0001606803125469014,0.02402474172413349,0.05689508467912674,0.005297868512570858,-0.09065324068069458,0.02547844685614109,-0.0065065110102295876,-0.037345945835113525,0.05572983995079994,-0.051312364637851715,-0.01213575154542923,0.0581943653523922,0.00921756960451603,0.019033532589673996,-0.009922919794917107,0.004349716007709503,0.028932293877005577,-0.03127605468034744,0.02737472578883171,0.04744908586144447,-0.013066908344626427,0.00790165551006794,0.0401325523853302,-0.04211841896176338,-0.019920121878385544,-0.005333133973181248,-0.04549294710159302,-0.011047394946217537,0.04155195876955986,-0.02495569922029972,0.061336781829595566,0.020569568499922752,0.05089324712753296,0.07311622053384781,0.01085831131786108,-0.020350787788629532,0.031270191073417664,-0.06866388767957687,-0.06619127839803696,-0.0038163242861628532,0.02289402112364769,0.034899741411209106,0.01543836947530508,0.013718018308281898,-0.18059809505939484,0.001861597876995802,-0.019229864701628685,0.07481343299150467,-0.08799027651548386,-0.038502298295497894,0.07084502279758453,-0.01417457778006792,-0.10040577501058578,0.06471937149763107,-0.07451769709587097,0.07629438489675522,0.011390878818929195,-0.01633593440055847,0.003757808357477188,0.00761036854237318,0.05621941760182381,-0.07682589441537857,0.009874165989458561,-0.08319663256406784,-0.013679291121661663,-0.0270924661308527,0.16818591952323914,-0.06392393261194229,-0.003042748896405101,0.008369702845811844,-0.024655643850564957,0.07865365594625473,0.0867457389831543,-0.023221084848046303,-0.027484741061925888,0.0014651104575023055,-0.06700734049081802,-0.03374172002077103,-0.0548006109893322,-0.025374310091137886,-0.07940462231636047,0.05086546763777733,0.03301694989204407,-0.017701687291264534,0.002485712757334113,0.026128217577934265,-0.014705476351082325,-0.009771629236638546,0.1307714730501175,-0.02665679156780243,-0.02602994441986084,-0.04534025117754936,-0.005196345038712025,-0.06204879656434059,-0.021309679374098778,-0.05913672223687172,0.01933889091014862,-0.012037335895001888,0.0317094624042511,0.03677907586097717,0.008748319000005722,0.11208406835794449,-0.08951223641633987,-0.004189499653875828,0.04192401096224785,0.024570563808083534,0.09048343449831009,0.05962062627077103,0.00879011768847704]} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-101751788295946195.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-101751788295946195.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-102831788295949791.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-102831788295949791.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-102881788199266918.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-102881788199266918.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-103341788215691811.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-103341788215691811.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-108121788295955136.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-108121788295955136.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-111271788295958296.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-111271788295958296.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-117191788295963215.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-117191788295963215.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-118211788199290918.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-118211788199290918.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-120491788199291875.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-120491788199291875.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-120491788295965591.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-120491788295965591.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-122881788295970594.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-122881788295970594.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-128621788199303300.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-128621788199303300.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-129251788295981727.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-129251788295981727.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-131411788215745702.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-131411788215745702.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-133151788199310835.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-133151788199310835.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-135481788215751635.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-135481788215751635.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-137601788215756223.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-137601788215756223.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-140551788295993094.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-140551788295993094.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-143041788215761288.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-143041788215761288.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-143721788295999445.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-143721788295999445.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-144761788215765051.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-144761788215765051.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-148881788215770530.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-148881788215770530.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-149941788296011708.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-149941788296011708.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-155201788296017304.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-155201788296017304.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-156701788296019652.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-156701788296019652.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-156721788199331913.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-156721788199331913.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-164251788296034463.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-164251788296034463.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-17481788204829211.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-17481788204829211.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-186201788215839559.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-186201788215839559.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-193111788199360944.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-193111788199360944.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-207101788215875374.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-207101788215875374.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-210301788199373715.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-210301788199373715.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-21051788199158374.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-21051788199158374.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-230041788215908268.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-230041788215908268.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-27471788215581175.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-27471788215581175.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-395461788294028113.json b/.agentplug/plugin-dispatch/out/gm-codesearch-395461788294028113.json new file mode 100644 index 0000000000..0abb4b5ed3 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-395461788294028113.json @@ -0,0 +1 @@ +{"data":{"degraded":false,"mode":"root_scoped","root":"litebox_platform_macos_userland","vector_hits":[]},"dispatch_id":"1788294065428-500-a089e7a88481673a","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"8c2605ec2e6addbc","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-395461788294028113.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-395461788294028113.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-397621788374127712.json b/.agentplug/plugin-dispatch/out/gm-codesearch-397621788374127712.json new file mode 100644 index 0000000000..e265ac861a --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-397621788374127712.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-39f1a0ff-d0115eff-1","score":37.43905397395181,"symbol":{"kind":"function_item","line_end":12,"line_start":8,"name":"main","path":"litebox_runner_linux_on_macos_userland/src/main.rs"},"text":"litebox_runner_linux_on_macos_userland/src/main.rs:8:12 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_linux_on_macos_userland::CliArgs;\n litebox_runner_linux_on_macos_userland::run(CliArgs::parse())\n}"},{"key":"ci-b8807bce-1130b664-1","score":31.38026972609122,"symbol":{"kind":"function_item","line_end":9,"line_start":5,"name":"main","path":"litebox_runner_optee_on_linux_userland/src/main.rs"},"text":"litebox_runner_optee_on_linux_userland/src/main.rs:5:9 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_optee_on_linux_userland::CliArgs;\n litebox_runner_optee_on_linux_userland::run(CliArgs::parse())\n}"},{"key":"ci-30003d83-48919216-0","score":27.691114318580812,"symbol":{"kind":"section","line_end":130,"line_start":1,"name":"","path":"docs/macos.md"},"text":"docs/macos.md:1:130 \n# LiteBox on macOS (Apple Silicon)\n\nLiteBox runs guest instructions natively; only the *system* interface is\nvirtualized. On an Apple Silicon Mac that means the only sensible configuration\nis an **AArch64 Linux guest on an AArch64 macOS host** — no emulation anywhere.\nThere is deliberately no x86-64 macOS platform: an x86-64 guest would need\ninstruction emulation, which is the thing this design exists to avoid.\n\nThis document covers what works today, what the host imposes, and what is left\nbefore a guest can actually execute.\n\n## What is in the tree\n\n| Piece | State |\n| --- | --- |\n| `litebox_platform_macos_userland` | The macOS \"South\" platform: memory, locking, time, signals, timers, threads, TLS, randomness, derived keys, stdio, `utun` networking, fault recovery. |\n| `litebox` core | Builds for `aarch64-apple-darwin`, including the Mach-O exception table. |\n| `litebox_shim_linux` | The Linux \"North\" shim, ported to AArch64: signal frames, syscall entry/return, thread-pointer handling, `stat`/`uname` ABI, exception decoding. |\n| `litebox_syscall_rewriter` | Already had AArch64 support (`arm64.rs`) for rewriting `SVC` and `TPIDR_EL0` accesses in Linux ELF images. |\n| `litebox_packager` | OCI mode now pulls the image matching the host architecture, and builds on Apple Silicon. |\n| Guest entry | **Implemented** (context switch + syscall dispatch), tested on real hardware. A syscall-only guest runs end to end; the guest thread-pointer plumbing and non-syscall event paths remain. See [Remaining work](#remaining-work). |\n\n## Building\n\n```sh\nrustup target add aarch64-apple-darwin\ncargo build --workspace --exclude litebox_runner_lvbs --exclude litebox_runner_snp\n```\n\n`litebox_runner_lvbs` and `litebox_runner_snp` are freestanding images for\ncustom targets and are not built for a hosted target on any platform.\n\nCI covers this in the `Build and Test macOS (Apple Silicon)` job, which also\ncompiles and runs `litebox_platform_macos_userland/tests/darwin_abi_probe.c`\nagainst the runner's real SDK headers -- the only check in this repo that\nverifies the crate's hand-written Mach/BSD struct layouts (used by the fault\nhandler to read `ucontext_t::uc_mcontext`) against an actual Darwin toolchain,\nsince nothing else in a Linux-hosted development loop can.\n\n### Hypervisor.framework release boundary\n\nThe stock-code backend requires Apple Silicon, macOS 26 or newer, and a macOS\n26 SDK or newer. The combined runner retains its macOS 11 deployment target so\nthe native backend still launches on older hosts; post-macOS-11 HVF imports are\nweak and the production boundary checks `__builtin_available(macOS 26.0, *)`\nbefore making any such call. `litebox_platform_macos_userland/build.rs`\ncompiles the narrow C boundary and linked EL1 monitor against the active SDK\nheaders, so Apple enum, object, and structure layouts do not get copied into\nRust constants. A release runner is built and signed with the checked-in\nentitlement manifest as follows:\n\n```sh\ncargo build --release --locked -p litebox_runner_linux_on_macos_userland\ncodesign --force --options runtime --sign - \\\n --entitlements litebox_runner_linux_on_macos_userland/entitlements.plist \\\n target/release/litebox_runner_linux_on_macos_userland\ncodesign --display --entitlements - \\\n target/release/litebox_runner_linux_on_macos_userland\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-boundary\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-memory\n```\n\nThe boundary diagnostic validates the active-SDK configuration, maps and\nunmaps the monitor, and creates, verifies, and destroys a vCPU. The compact\nmemory diagnostic does not run guest instructions. It proves that IPA zero is\nreserved for the monitor, GVA=HVA mappings receive compact non-identity IPAs,\n16 KiB stage-one roots software-walk to the same pages, stage-two permissions\ncan move from RW to RX without RWX, rejected overlap transactions publish no\nnew root generation, and released IPA is reused without poisoning the VM. The\nlinked 16 KiB monitor remains the source of truth, but the VM owns an aligned\nallocator-backed copy: Hypervisor.framework rejects file-backed Mach-O\n`__TEXT` pages as stage-two backing.\n\n`com.apple.security.hypervisor=true` is mandatory for\nHypervisor.framework VM creation. `com.apple.security.cs.allow-jit=true` remains\nrequired when the native rewritten backend is packaged with Hardened Runtime.\nThe legacy `com.apple.vm.hypervisor` entitlement belongs only to deployment\ntargets through macOS 10.15 and is deliberately absent. Notarization and\nHardened Runtime do not grant Hypervisor.framework authorization; the final\nexecutable still needs the hypervisor entitlement.\n\n## What the host imposes\n\n### 16 KiB pages\n\nApple Silicon's page size is 16 KiB. Every fixed mapping and every protection\nchange must be aligned to it, so `litebox::mm::linux::PAGE_SIZE` is 16384 on\nthis target rather than 4096. The guest sees the same value through `AT_PAGESZ`,\nwhich is exactly how a Linux kernel configured for 16 KiB or 64 KiB pages\nreports itself.\n\nAArch64 ELF images are conventionally linked with a 64 KiB maximum page size, so\ntheir `PT_LOAD` segments stay aligned either way. An image built with 4 KiB\nsegment alignment will not map cleanly.\n\n### The first 4 GiB is unusable\n\nAn arm64 Mach-O process reserves `[0, 4 GiB)` as the `__PAGEZERO` segment:\nunmapped and impossible to map over. `TASK_ADDR_MIN` is therefore `0x1_0000_0000`.\n\nThe practical consequence is that guest images must be position-independent, or\nlinked above 4 GiB. An `ET_EXEC` binary linked at the customary `0x400000`\ncannot be loaded at its preferred address on this host.\n\n### W^X, `MAP_JIT`, and code signing\n\nmacOS refuses to make anonymous memory executable through the ordinary path, and\nrefuses to add `PROT_EXEC` to anything that was ever writable. The supported\nescape hatch is `MAP_JIT`, which the platform passes whenever a mapping requests\n`EXEC`. Using it has two consequences:\n\n1. **The JIT entitlement is only load-bearing under the Hardened Runtime.**\n Per Apple's own documentation, `com.apple.security.cs.allow-jit` is required\n only when a binary has the Hardened Runtime enabled (`codesign --options\n runtime`, which in turn is what notarization requires); without it,\n `MAP_JIT` works with or without the entitlement present. The command below\n ad-hoc-signs with the entitlement anyway -- it costs nothing and future-proofs\n a later `--options runtime`, notarized build -- but for local development\n outside Gatekeeper, neither the entitlement nor notarization is actually\n required for `MAP_JIT` itself to work. Create an entitlements file:\n\n ```xml\n \n \n \n \n com.apple.security.cs.allow-jit"},{"key":"ci-f0a0109-776f5635-1","score":24.009278367098528,"symbol":{"kind":"function_item","line_end":106,"line_start":50,"name":"run","path":"litebox_runner_optee_on_linux_userland/src/lib.rs"},"text":"litebox_runner_optee_on_linux_userland/src/lib.rs:50:106 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let ldelf_data: Vec = {\n let ldelf = PathBuf::from(&cli_args.ldelf);\n let data =\n std::fs::read(&ldelf).with_context(|| format!(\"failed to read {}\", cli_args.ldelf))?;\n if cli_args.rewrite_syscalls {\n litebox_syscall_rewriter::hook_syscalls_in_elf(&data, None)\n .with_context(|| format!(\"failed to rewrite {}\", cli_args.ldelf))?\n } else {\n data\n }\n };\n\n let prog_data: Vec = {\n let prog = PathBuf::from(&cli_args.program);\n let data =\n std::fs::read(&prog).with_context(|| format!(\"failed to read {}\", cli_args.program))?;\n if cli_args.rewrite_syscalls {\n litebox_syscall_rewriter::hook_syscalls_in_elf(&data, None)\n .with_context(|| format!(\"failed to rewrite {}\", cli_args.program))?\n } else {\n data\n }\n };\n\n // TODO(jb): Clean up platform initialization once we have https://github.com/MSRSSP/litebox/issues/24\n let platform = Platform::new(None);\n litebox_platform_multiplex::set_platform(platform);\n let shim_builder = litebox_shim_optee::OpteeShimBuilder::new();\n let _litebox = shim_builder.litebox();\n let shim = shim_builder.build();\n\n platform.initialize_boot_specific_kdf_support();\n\n if cli_args.command_sequence.is_empty() {\n run_ta_with_default_commands(&shim, ldelf_data.as_slice(), prog_data.as_slice());\n } else {\n tests::run_ta_with_test_commands(\n &shim,\n ldelf_data.as_slice(),\n prog_data.as_slice(),\n cli_args.program.as_str(),\n &PathBuf::from(&cli_args.command_sequence),\n );\n }\n Ok(())\n}"},{"key":"ci-ea2224bb-9e5d768c-1","score":23.7434334902661,"symbol":{"kind":"struct_item","line_end":44,"line_start":21,"name":"CliArgs","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:21:44 CliArgs\npub struct CliArgs {\n /// The program and arguments passed to it (e.g., `/bin/ls --color`).\n ///\n /// The program path refers to a path inside the tar archive provided via\n /// `--initial-files`. All binaries must be pre-rewritten with the syscall\n /// rewriter.\n #[arg(required = true, trailing_var_arg = true, value_hint = clap::ValueHint::CommandWithArguments)]\n pub program_and_arguments: Vec,\n /// Environment variables passed to the program (`K=V` pairs; can be invoked multiple times)\n #[arg(long = \"env\")]\n pub environment_variables: Vec,\n /// Forward the existing environment variables\n #[arg(long = \"forward-env\")]\n pub forward_environment_variables: bool,\n /// Allow using unstable options\n #[arg(short = 'Z', long = \"unstable\")]\n pub unstable: bool,\n /// Tar archive containing the program and its shared libraries.\n ///\n /// All ELF binaries should be pre-rewritten with the syscall rewriter\n /// (e.g., via `litebox-packager`).\n #[arg(long = \"initial-files\", value_name = \"PATH_TO_TAR\", value_hint = clap::ValueHint::FilePath)]\n pub initial_files: PathBuf,\n}"},{"key":"ci-ea2224bb-9e5d768c-0","score":22.091965008132732,"symbol":{"kind":"function_item","line_end":147,"line_start":53,"name":"run","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:53:147 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let tar_file = &cli_args.initial_files;\n if tar_file.extension().and_then(|x| x.to_str()) != Some(\"tar\") {\n anyhow::bail!(\"Expected a .tar file, found {}\", tar_file.display());\n }\n let tar_data = std::fs::read(tar_file)\n .map_err(|e| anyhow!(\"Could not read tar file at {}: {}\", tar_file.display(), e))?;\n\n let platform = Platform::new();\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n // The program path is a Unix-style path inside the tar archive.\n let prog_path = &cli_args.program_and_arguments[0];\n\n let initial_file_system = {\n let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem.with_root_privileges(|fs| {\n use litebox::fs::FileSystem as _;\n fs.mkdir(\n \"/tmp\",\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap();\n fs.chown(\"/tmp\", Some(1000), Some(1000)).unwrap();\n\n // Standard FHS directories that guest tools expect to already exist (e.g. `apk`\n // opens a log file under `/var/log`) but that don't survive as empty-directory\n // entries when an OCI image's rootfs is scanned into a file-based tar: an empty\n // directory has no file contents, so it produces no tar entry, and `TarRo`'s\n // directory tree is inferred purely from file paths.\n for dir in [\"/run\", \"/var\", \"/var/log\", \"/var/cache\", \"/var/tmp\"] {\n fs.mkdir(\n dir,\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|_| {\n panic!(\"{dir} creation cannot fail on a fresh in-memory file system\")\n });\n }\n });\n\n shim_builder.default_fs(in_mem, tar_data.into())\n };\n let initial_file_system = std::sync::Arc::new(initial_file_system);\n\n let shim = shim_builder.build();\n let argv = cli_args\n .program_and_arguments\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp: Vec<_> = cli_args\n .environment_variables\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp = if cli_args.forward_environment_variables {\n envp.into_iter()\n .chain(std::env::vars().map(|(k, v)| {\n std::ffi::CString::new(k.bytes().chain(*b\"=\").chain(v.bytes()).collect::>())\n .unwrap()\n }))\n .collect()\n } else {\n envp\n };\n\n let program = shim\n .load_program(\n initial_file_system,\n platform.init_task(),\n prog_path,\n argv,\n envp,\n )\n .unwrap();\n unsafe {\n litebox_platform_windows_userland::run_thread(\n program.entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n std::process::exit(program.process.wait())\n}"},{"key":"ci-33f01406-6265afd1-0","score":21.143838129008635,"symbol":{"kind":"function_item","line_end":8,"line_start":4,"name":"main","path":"litebox_packager/src/main.rs"},"text":"litebox_packager/src/main.rs:4:8 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_packager::CliArgs;\n litebox_packager::run(CliArgs::parse())\n}"},{"key":"ci-b44f76e0-a87fbecf-0","score":19.13446242568561,"symbol":{"kind":"function_item","line_end":102,"line_start":77,"name":"main","path":"litebox_syscall_rewriter/src/main.rs"},"text":"litebox_syscall_rewriter/src/main.rs:77:102 main\nfn main() -> anyhow::Result<()> {\n let cli_args = CliArgs::parse();\n let mut input_binary = std::fs::File::open(&cli_args.input_binary)?;\n let mut input_binary_bytes = vec![];\n input_binary.read_to_end(&mut input_binary_bytes)?;\n let output_binary = litebox_syscall_rewriter::rewrite_binary_for_host(\n &input_binary_bytes,\n cli_args.trampoline_addr,\n cli_args.host.into(),\n )?;\n let output_path = cli_args.output_binary.unwrap_or_else(|| {\n cli_args.input_binary.with_file_name(\n cli_args\n .input_binary\n .file_name()\n .unwrap()\n .to_string_lossy()\n .into_owned()\n + \".hooked\",\n )\n });\n let mut file = std::fs::File::create(output_path)?;\n copy_file_permissions(&input_binary, &file)?;\n file.write_all(&output_binary)?;\n Ok(())\n}"},{"key":"ci-b30dc81f-d5084132-0","score":18.37720061577283,"symbol":{"kind":"function_item","line_end":112,"line_start":57,"name":"main","path":"litebox_platform_macos_userland/build.rs"},"text":"litebox_platform_macos_userland/build.rs:57:112 main\nfn main() -> Result<(), Box> {\n println!(\"cargo:rerun-if-changed=src/hvf_sdk.c\");\n println!(\"cargo:rerun-if-changed=src/hvf_monitor.S\");\n for variable in [\n \"MACOSX_DEPLOYMENT_TARGET\",\n \"DEVELOPER_DIR\",\n \"SDKROOT\",\n \"TOOLCHAINS\",\n ] {\n println!(\"cargo:rerun-if-env-changed={variable}\");\n }\n\n if env::var(\"CARGO_CFG_TARGET_OS\")? != \"macos\"\n || env::var(\"CARGO_CFG_TARGET_ARCH\")? != \"aarch64\"\n {\n return Ok(());\n }\n\n let sdk = String::from_utf8(xcrun([\"--sdk\", \"macosx\", \"--show-sdk-path\"])?.stdout)?;\n let sdk = sdk.trim();\n let deployment_target =\n env::var(\"MACOSX_DEPLOYMENT_TARGET\").unwrap_or_else(|_| \"11.0\".to_owned());\n let out = PathBuf::from(env::var_os(\"OUT_DIR\").ok_or(\"OUT_DIR is not set\")?);\n let manifest =\n PathBuf::from(env::var_os(\"CARGO_MANIFEST_DIR\").ok_or(\"CARGO_MANIFEST_DIR is not set\")?);\n let sdk_object = out.join(\"hvf_sdk.o\");\n let monitor_object = out.join(\"hvf_monitor.o\");\n let archive = out.join(\"liblitebox_hvf_sdk.a\");\n\n compile(\n &manifest.join(\"src/hvf_sdk.c\"),\n &sdk_object,\n sdk,\n &deployment_target,\n )?;\n compile(\n &manifest.join(\"src/hvf_monitor.S\"),\n &monitor_object,\n sdk,\n &deployment_target,\n )?;\n xcrun([\n OsStr::new(\"--sdk\"),\n OsStr::new(\"macosx\"),\n OsStr::new(\"ar\"),\n OsStr::new(\"rcs\"),\n archive.as_os_str(),\n sdk_object.as_os_str(),\n monitor_object.as_os_str(),\n ])?;\n\n println!(\"cargo:rustc-link-search=native={}\", out.display());\n println!(\"cargo:rustc-link-lib=static=litebox_hvf_sdk\");\n println!(\"cargo:rustc-link-lib=framework=Hypervisor\");\n Ok(())\n}"},{"key":"ci-adbeac4c-1cb07d98-0","score":16.758748936264965,"symbol":{"kind":"section","line_end":43,"line_start":4,"name":"","path":"litebox_packager/examples/xfce/README.md"},"text":"litebox_packager/examples/xfce/README.md:4:43 \n# XFCE desktop guest\n\nThis recipe builds an Alpine 3.24 XFCE desktop for the macOS userland runner\nand serves its 1024×768 framebuffer and input through the built-in browser\nviewer.\n\nApple's kernel clears AArch64 `x18` whenever it returns to userspace. Stock\nAlpine allocates that register, which silently corrupts ld.so, Xorg, GTK, and\nXFCE hot loops under native guest execution. The build therefore first runs\n`../../scripts/build-x18-desktop-repo.sh`; that rebuilds the desktop's loaded\ncode closure with x18 reserved and refuses to publish any runtime ELF that\nstill disassembles to an `x18`/`w18` operand.\n\n```sh\n./litebox_packager/examples/xfce/build-xfce-image.sh /tmp/litebox-xfce.tar\n\ncargo run --release -p litebox_runner_linux_on_macos_userland -- \\\n --unstable --guest-root \\\n --initial-files /tmp/litebox-xfce.tar \\\n --vnc-web 6080 -- \\\n /usr/bin/start-desktop.sh\n```\n\nOpen . The canvas accepts pointer, wheel, and keyboard\ninput.\n\nThe first package build is intentionally substantial and resumable. Successful\naports origins remain in the retained `litebox-x18-repo-build` container;\nrerunning retries only unfinished origins. Override paths and names with:\n\n- `LITEBOX_X18_DESKTOP_REPO`\n- `LITEBOX_ALPINE_BRANCH`\n- `LITEBOX_XFCE_IMAGE_TAG`\n\nThe recipe disables GLX and fbdev ShadowFB because those unimplemented paths\ndo not update litebox's browser framebuffer. It also appends the synthetic\n`/sys/class/graphics/fb0/device/subsystem` link required by Xorg's fbdevhw\nprobe; the packager otherwise intentionally omits `/sys` from OCI rootfs\nimages."}],"commits":[{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":92.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":41.0},{"hash":"624f0ceedf841b147c5e972104d7cfed355ccc90","message":"feat(net): guest web access without root via an in-stack HTTP proxy","score":37.0},{"hash":"dbc767aa1920e9a3c60c3602e6e899b4d37eeeb4","message":"Wire fchown via a new fd_chown, completing the chown family","score":36.0},{"hash":"5251e6a45413ae95629010a02e13a7959a9e3517","message":"feat(evdev): /dev/input/event* emulation wired to VNC input","score":35.0},{"hash":"366634828dae54bd2ff0c64e83f5d1cb9ade7c46","message":"fix(ci): clear clippy 1.98 lints, fmt drift, and vendored-crate no_std check","score":35.0},{"hash":"4a62d24fb8dba3be51e29bf89b38449d8bbd7d19","message":"feat(terminal): honor TCSETSW/TCSETSF, add per-guest IP override, npm 0.1.1","score":35.0},{"hash":"b97641713732df7cb18e20ebbd80e675e777fba6","message":"feat(rfb): runner-side VNC server presenting the guest /dev/fb0 framebuffer","score":33.0},{"hash":"d06199534c3cf1090fdb261b8c800d380dea941e","message":"fix: getrusage/wait4 guest-stack overrun, doc gate, pipe-hang probes","score":32.0},{"hash":"323d13318ea10946e74e99a7e24d593c9ef6e7ed","message":"feat(shim): the syscall surface an X11 desktop stack needs","score":31.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788374539319-500-e646470881a64e45","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"de5c21d06de20560","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-397621788374127712.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-397621788374127712.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-41581788215586266.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-41581788215586266.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-418341788372528614.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-418341788372528614.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-420721788372533381.json b/.agentplug/plugin-dispatch/out/gm-codesearch-420721788372533381.json new file mode 100644 index 0000000000..b7de46e682 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-420721788372533381.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-30003d83-48919216-0","score":26.026689446123292,"symbol":{"kind":"section","line_end":130,"line_start":1,"name":"","path":"docs/macos.md"},"text":"docs/macos.md:1:130 \n# LiteBox on macOS (Apple Silicon)\n\nLiteBox runs guest instructions natively; only the *system* interface is\nvirtualized. On an Apple Silicon Mac that means the only sensible configuration\nis an **AArch64 Linux guest on an AArch64 macOS host** — no emulation anywhere.\nThere is deliberately no x86-64 macOS platform: an x86-64 guest would need\ninstruction emulation, which is the thing this design exists to avoid.\n\nThis document covers what works today, what the host imposes, and what is left\nbefore a guest can actually execute.\n\n## What is in the tree\n\n| Piece | State |\n| --- | --- |\n| `litebox_platform_macos_userland` | The macOS \"South\" platform: memory, locking, time, signals, timers, threads, TLS, randomness, derived keys, stdio, `utun` networking, fault recovery. |\n| `litebox` core | Builds for `aarch64-apple-darwin`, including the Mach-O exception table. |\n| `litebox_shim_linux` | The Linux \"North\" shim, ported to AArch64: signal frames, syscall entry/return, thread-pointer handling, `stat`/`uname` ABI, exception decoding. |\n| `litebox_syscall_rewriter` | Already had AArch64 support (`arm64.rs`) for rewriting `SVC` and `TPIDR_EL0` accesses in Linux ELF images. |\n| `litebox_packager` | OCI mode now pulls the image matching the host architecture, and builds on Apple Silicon. |\n| Guest entry | **Implemented** (context switch + syscall dispatch), tested on real hardware. A syscall-only guest runs end to end; the guest thread-pointer plumbing and non-syscall event paths remain. See [Remaining work](#remaining-work). |\n\n## Building\n\n```sh\nrustup target add aarch64-apple-darwin\ncargo build --workspace --exclude litebox_runner_lvbs --exclude litebox_runner_snp\n```\n\n`litebox_runner_lvbs` and `litebox_runner_snp` are freestanding images for\ncustom targets and are not built for a hosted target on any platform.\n\nCI covers this in the `Build and Test macOS (Apple Silicon)` job, which also\ncompiles and runs `litebox_platform_macos_userland/tests/darwin_abi_probe.c`\nagainst the runner's real SDK headers -- the only check in this repo that\nverifies the crate's hand-written Mach/BSD struct layouts (used by the fault\nhandler to read `ucontext_t::uc_mcontext`) against an actual Darwin toolchain,\nsince nothing else in a Linux-hosted development loop can.\n\n### Hypervisor.framework release boundary\n\nThe stock-code backend requires Apple Silicon, macOS 26 or newer, and a macOS\n26 SDK or newer. The combined runner retains its macOS 11 deployment target so\nthe native backend still launches on older hosts; post-macOS-11 HVF imports are\nweak and the production boundary checks `__builtin_available(macOS 26.0, *)`\nbefore making any such call. `litebox_platform_macos_userland/build.rs`\ncompiles the narrow C boundary and linked EL1 monitor against the active SDK\nheaders, so Apple enum, object, and structure layouts do not get copied into\nRust constants. A release runner is built and signed with the checked-in\nentitlement manifest as follows:\n\n```sh\ncargo build --release --locked -p litebox_runner_linux_on_macos_userland\ncodesign --force --options runtime --sign - \\\n --entitlements litebox_runner_linux_on_macos_userland/entitlements.plist \\\n target/release/litebox_runner_linux_on_macos_userland\ncodesign --display --entitlements - \\\n target/release/litebox_runner_linux_on_macos_userland\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-boundary\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-memory\n```\n\nThe boundary diagnostic validates the active-SDK configuration, maps and\nunmaps the monitor, and creates, verifies, and destroys a vCPU. The compact\nmemory diagnostic does not run guest instructions. It proves that IPA zero is\nreserved for the monitor, GVA=HVA mappings receive compact non-identity IPAs,\n16 KiB stage-one roots software-walk to the same pages, stage-two permissions\ncan move from RW to RX without RWX, rejected overlap transactions publish no\nnew root generation, and released IPA is reused without poisoning the VM. The\nlinked 16 KiB monitor remains the source of truth, but the VM owns an aligned\nallocator-backed copy: Hypervisor.framework rejects file-backed Mach-O\n`__TEXT` pages as stage-two backing.\n\n`com.apple.security.hypervisor=true` is mandatory for\nHypervisor.framework VM creation. `com.apple.security.cs.allow-jit=true` remains\nrequired when the native rewritten backend is packaged with Hardened Runtime.\nThe legacy `com.apple.vm.hypervisor` entitlement belongs only to deployment\ntargets through macOS 10.15 and is deliberately absent. Notarization and\nHardened Runtime do not grant Hypervisor.framework authorization; the final\nexecutable still needs the hypervisor entitlement.\n\n## What the host imposes\n\n### 16 KiB pages\n\nApple Silicon's page size is 16 KiB. Every fixed mapping and every protection\nchange must be aligned to it, so `litebox::mm::linux::PAGE_SIZE` is 16384 on\nthis target rather than 4096. The guest sees the same value through `AT_PAGESZ`,\nwhich is exactly how a Linux kernel configured for 16 KiB or 64 KiB pages\nreports itself.\n\nAArch64 ELF images are conventionally linked with a 64 KiB maximum page size, so\ntheir `PT_LOAD` segments stay aligned either way. An image built with 4 KiB\nsegment alignment will not map cleanly.\n\n### The first 4 GiB is unusable\n\nAn arm64 Mach-O process reserves `[0, 4 GiB)` as the `__PAGEZERO` segment:\nunmapped and impossible to map over. `TASK_ADDR_MIN` is therefore `0x1_0000_0000`.\n\nThe practical consequence is that guest images must be position-independent, or\nlinked above 4 GiB. An `ET_EXEC` binary linked at the customary `0x400000`\ncannot be loaded at its preferred address on this host.\n\n### W^X, `MAP_JIT`, and code signing\n\nmacOS refuses to make anonymous memory executable through the ordinary path, and\nrefuses to add `PROT_EXEC` to anything that was ever writable. The supported\nescape hatch is `MAP_JIT`, which the platform passes whenever a mapping requests\n`EXEC`. Using it has two consequences:\n\n1. **The JIT entitlement is only load-bearing under the Hardened Runtime.**\n Per Apple's own documentation, `com.apple.security.cs.allow-jit` is required\n only when a binary has the Hardened Runtime enabled (`codesign --options\n runtime`, which in turn is what notarization requires); without it,\n `MAP_JIT` works with or without the entitlement present. The command below\n ad-hoc-signs with the entitlement anyway -- it costs nothing and future-proofs\n a later `--options runtime`, notarized build -- but for local development\n outside Gatekeeper, neither the entitlement nor notarization is actually\n required for `MAP_JIT` itself to work. Create an entitlements file:\n\n ```xml\n \n \n \n \n com.apple.security.cs.allow-jit"},{"key":"ci-39f1a0ff-d0115eff-0","score":23.972234590465803,"symbol":{"kind":"function_item","line_end":18,"line_start":15,"name":"main","path":"litebox_runner_linux_on_macos_userland/src/main.rs"},"text":"litebox_runner_linux_on_macos_userland/src/main.rs:15:18 main\nfn main() {\n eprintln!(\"This program is only supported on macOS on Apple Silicon\");\n std::process::exit(1);\n}"},{"key":"ci-b30dc81f-d5084132-0","score":20.398320760791407,"symbol":{"kind":"function_item","line_end":112,"line_start":57,"name":"main","path":"litebox_platform_macos_userland/build.rs"},"text":"litebox_platform_macos_userland/build.rs:57:112 main\nfn main() -> Result<(), Box> {\n println!(\"cargo:rerun-if-changed=src/hvf_sdk.c\");\n println!(\"cargo:rerun-if-changed=src/hvf_monitor.S\");\n for variable in [\n \"MACOSX_DEPLOYMENT_TARGET\",\n \"DEVELOPER_DIR\",\n \"SDKROOT\",\n \"TOOLCHAINS\",\n ] {\n println!(\"cargo:rerun-if-env-changed={variable}\");\n }\n\n if env::var(\"CARGO_CFG_TARGET_OS\")? != \"macos\"\n || env::var(\"CARGO_CFG_TARGET_ARCH\")? != \"aarch64\"\n {\n return Ok(());\n }\n\n let sdk = String::from_utf8(xcrun([\"--sdk\", \"macosx\", \"--show-sdk-path\"])?.stdout)?;\n let sdk = sdk.trim();\n let deployment_target =\n env::var(\"MACOSX_DEPLOYMENT_TARGET\").unwrap_or_else(|_| \"11.0\".to_owned());\n let out = PathBuf::from(env::var_os(\"OUT_DIR\").ok_or(\"OUT_DIR is not set\")?);\n let manifest =\n PathBuf::from(env::var_os(\"CARGO_MANIFEST_DIR\").ok_or(\"CARGO_MANIFEST_DIR is not set\")?);\n let sdk_object = out.join(\"hvf_sdk.o\");\n let monitor_object = out.join(\"hvf_monitor.o\");\n let archive = out.join(\"liblitebox_hvf_sdk.a\");\n\n compile(\n &manifest.join(\"src/hvf_sdk.c\"),\n &sdk_object,\n sdk,\n &deployment_target,\n )?;\n compile(\n &manifest.join(\"src/hvf_monitor.S\"),\n &monitor_object,\n sdk,\n &deployment_target,\n )?;\n xcrun([\n OsStr::new(\"--sdk\"),\n OsStr::new(\"macosx\"),\n OsStr::new(\"ar\"),\n OsStr::new(\"rcs\"),\n archive.as_os_str(),\n sdk_object.as_os_str(),\n monitor_object.as_os_str(),\n ])?;\n\n println!(\"cargo:rustc-link-search=native={}\", out.display());\n println!(\"cargo:rustc-link-lib=static=litebox_hvf_sdk\");\n println!(\"cargo:rustc-link-lib=framework=Hypervisor\");\n Ok(())\n}"},{"key":"ci-f1d76c9a-279b43d3-0","score":18.759543938704265,"symbol":{"kind":"section","line_end":126,"line_start":1,"name":"","path":"docs/roadmap.md"},"text":"docs/roadmap.md:1:126 \n# Roadmap: known gaps and follow-up work\n\nThis is a working list of gaps found while porting LiteBox to macOS/Apple\nSilicon and auditing the rest of the tree for related issues. Each entry\nbelow was deliberately **not** implemented in that pass, because doing it\ncorrectly needs either real hardware/kernel verification this repo's CI\ncannot provide from a Linux-hosted sandbox, or a genuine design decision\nrather than a mechanical fix. Implementing any of these without that\nverification risks the exact kind of half-finished, silently-wrong change\nthis list exists to avoid.\n\nItems are grouped by how much verification they need before landing, not by\nsubsystem.\n\n## Resolved on real hardware this pass\n\n* **The `TPIDR_EL0` anchor question is answered.** Measured on an Apple M3\n Pro (macOS 26.3.1): `TPIDR_EL0` does not survive a context switch (XNU\n overwrites it with its own value, not merely leaves it stale) and cannot\n anchor the guest thread pointer. `TPIDRRO_EL0` is stable across a reschedule\n and distinct per thread, matching Apple's documented pthread-self-pointer\n use. See [`docs/macos.md`](./macos.md#remaining-work) for the full\n measurement and the resulting design (a reserved pthread TSD slot read via\n a `TPIDRRO_EL0`-relative direct-TSD sequence, mirroring libSystem's own\n fast accessors). What's left is implementation, not research:\n\n## Needs real Apple Silicon hardware (implementation, not open questions)\n\n* **`Host::MacOs`'s anchor register is right; the fixed TSD slot number is\n not, and the whole \"bake one number in at packaging time\" approach has a\n deeper problem than the number being wrong.** Gates anchor on `TPIDRRO_EL0`\n (real, tested) and address the guest thread pointer at pthread TSD slot\n `MACOS_GUEST_TPIDR_TSD_SLOT` (hardcoded to 256, sourced from\n apple-oss-distributions/libpthread as \"the first dynamic\n `pthread_key_create` key\") -- a LiteBox-owned slot rather than a raw offset\n into Apple's own pthread structure, so it no longer risks corrupting\n libpthread state the way the earlier design did.\n \n `litebox_platform_macos_userland::new` calls `pthread_key_create` at startup\n and records the key. It originally *asserted* the key equalled the baked slot\n -- **which always fails on real hardware**, making the platform\n unconstructable -- so that was softened (this pass) to a loud warning that\n leaves construction working (regression test\n `reserving_the_tsd_slot_does_not_panic_on_mismatch`); a syscall-only guest is\n unaffected, a `TPIDR_EL0`-using guest is unsupported until the real fix below.\n Measured on this M3 Pro (macOS 26.3.1): a minimal Rust binary's first\n `pthread_key_create` call returns 259; a plain C `main`'s first call\n returns 258. Neither is 256. Something in libSystem's startup path claims a\n few dynamic keys before user code runs, undocumented and not guaranteed\n stable across macOS versions or across binaries with different statically\n linked dependencies (each with their own static initializers, potentially\n claiming more). This means the actual slot a real runner binary gets is a\n property of *that specific binary's* full startup sequence -- not knowable\n by the rewriter, which runs separately, earlier, packaging the guest image\n with no visibility into what the eventual runner process will look like.\n \n The failure mode is safe (a loud warning at `MacOsUserland::new()`, not\n silent corruption), so this does not need the same \"keep it out of anything\n that runs for real\" mitigation the previous corruption bug did.\n\n **The rewriter half of the fix has landed; the loader half has not.**\n `Host::MacOs` gates no longer bake the slot number in. They read a byte offset\n from the trampoline header slot `HEADER_GUEST_TP_OFFSET_MACOS` and address\n `[TPIDRRO_EL0 + offset]`, which is what makes the number a load-time rather\n than a packaging-time decision. `Host::Linux` is untouched and still bakes its\n immediate, since its offset is genuine compile-time ABI; the two are now\n distinguished explicitly by `GuestTpAddressing`.\n\n The slot holds an *offset*, never a thread-pointer value. The loader maps the\n trampoline writable, fills the header, then flips it to read+execute\n (`litebox_common_linux`'s `load_trampoline`), so nothing can rewrite that word\n once a guest is running -- and one word could not serve two threads anyway. The\n per-thread part comes from `TPIDRRO_EL0`, which is already per-thread, so this\n design stays compatible with per-thread guest TPs rather than foreclosing them.\n\n **The loader half has landed too.** `SystemInfoProvider::get_guest_tp_slot_offset`\n reports the offset a host decides at run time (`None` on every host that bakes\n it in); `litebox_platform_macos_userland` answers with\n `guest_tp_slot_byte_offset()`, the reserved `pthread_key_create` key scaled by\n 8. `litebox_common_linux`'s `load_trampoline` publishes it into the header slot\n in the same window it already writes the syscall entry point -- while the\n trampoline is still writable and before the flip to read+execute. That window\n is the only correct place for it. `litebox_common_linux` cannot depend on the\n rewriter, so `litebox_shim_linux` holds the two slot constants together with a\n `const` assertion rather than a comment.\n\n What remains for a `TPIDR_EL0`-using guest: `pthread_setspecific` of each guest\n thread's pointer into the reserved key, and a macOS runner to exercise any of\n it -- none wires `MacOsUserland` into `litebox_shim_linux` today, so this whole\n path is still unexercised end to end on hardware.\n* **The platform's *own* per-thread context-switch bookkeeping** —\n **RESOLVED on real hardware.** A separate problem from the rewriter's guest\n slot above, and the thing that limited this platform to one guest thread at a\n time. `litebox_platform_linux_userland`'s x86_64\n `run_thread_arch`/`switch_to_guest`/`syscall_callback` (the closest thing to\n a template) does not only virtualize the *guest's* thread pointer -- it also\n stashes its own bookkeeping (`host_sp`, `host_bp`, `guest_context_top`,\n `in_guest`) in `fs:`-relative TLS slots, because by the time\n `syscall_callback` runs, every general-purpose register holds live guest\n state and there is nothing else durable to read \"where was the host stack\"\n from. That mechanism is entirely x86_64-ELF-specific (raw `@tpoff`-relative\n local-exec TLS addressing, resolved to a link-time-fixed offset with no\n function call and no runtime-determined value at all) and has no Mach-O\n equivalent to copy directly.\n\n **The building block, confirmed on real Apple M3 Pro hardware:** a raw\n `mrs tpidrro_el0` (masked, `& ~7`, matching libSystem's own\n `_os_tsd_get_base`) plus a `[base, #(key * 8)]` read/write reaches the *same*\n per-thread storage `pthread_getspecific`/`pthread_setspecific` do, for a\n **second**, independently `pthread_key_create`-reserved dynamic TSD key (not\n just the one already relied on for the guest's own `TPIDR_EL0` shadow) -- in\n both directions, across the full `usize` range, and disjointly across two\n genuinely concurrent OS threads. Measured again this pass: the first dynamic\n key a Rust binary gets is 259 and the pool is exhausted at key 767.\n\n **The blocker, and how it was actually solved.** Of the six naked functions\n in `litebox_platform_macos_userland::guest`, five are reached with registers\n to spare (three of them by a signal handler's `pc` redirect, so *every*\n register is free) and can simply do a two-register lookup: one register for\n the run-time-determined TSD byte offset, one for the `TPIDRRO_EL0` value. The\n syscall callback cannot: the rewriter's `SVC` gate leaves exactly **one**\n register free (`X16`), because `X17` still holds the guest's real value,\n which real Linux AArch64 preserves across a syscall and which this file's own\n fidelity philosophy (`preserves_registers_across_capture_and_resume`) commits\n to capturing faithfully. One register is enough for\n `mrs`/`and`/`ldr [x16, #imm]` **only if the immediate is a compile-time"},{"key":"ci-e7e5ab84-69e5da90-0","score":14.744628146931204,"symbol":{"kind":"impl_item","line_end":93,"line_start":17,"name":"","path":"litebox_runner_linux_on_windows_userland/tests/common/mod.rs"},"text":"litebox_runner_linux_on_windows_userland/tests/common/mod.rs:17:93 \nimpl TestLauncher {\n pub fn init_platform(\n tar_data: &'static [u8],\n initial_dirs: &[&str],\n initial_files: &[&str],\n ) -> Self {\n let platform = Platform::new();\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n let mut in_mem_fs = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem_fs.with_root_privileges(|fs| {\n fs.chmod(\"/\", Mode::RWXU | Mode::RWXG | Mode::RWXO)\n .expect(\"Failed to set permissions on root\");\n });\n let tar_data = if tar_data.is_empty() {\n litebox::fs::tar_ro::EMPTY_TAR_FILE.into()\n } else {\n tar_data.into()\n };\n let fs = shim_builder.default_fs(in_mem_fs, tar_data);\n let mut this = Self {\n platform,\n shim_builder,\n fs,\n };\n\n for each in initial_dirs {\n this.install_dir(each);\n }\n for each in initial_files {\n let data = std::fs::read(each).unwrap();\n this.install_file(data, each);\n }\n\n this\n }\n\n pub fn install_dir(&mut self, path: &str) {\n self.fs\n .mkdir(path, Mode::RWXU | Mode::RWXG | Mode::RWXO)\n .expect(\"Failed to create directory\");\n }\n\n pub fn install_file(&mut self, contents: Vec, out: &str) {\n let fd = self\n .fs\n .open(\n out,\n OFlags::CREAT | OFlags::WRONLY,\n Mode::RWXG | Mode::RWXO | Mode::RWXU,\n )\n .unwrap();\n self.fs.write(&fd, &contents, None).unwrap();\n self.fs.close(&fd).unwrap();\n }\n\n pub fn test_load_exec_common(self, executable_path: &str) {\n let fs = std::sync::Arc::new(self.fs);\n let argv = vec![\n CString::new(executable_path).unwrap(),\n CString::new(\"hello\").unwrap(),\n ];\n let envp = vec![CString::new(\"PATH=/bin\").unwrap()];\n let shim = self.shim_builder.build();\n let program = shim\n .load_program(fs, self.platform.init_task(), executable_path, argv, envp)\n .unwrap();\n unsafe {\n litebox_platform_windows_userland::run_thread(\n program.entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n assert_eq!(program.process.wait(), 0);\n }\n}"},{"key":"ci-f0a0109-776f5635-0","score":14.64823575396014,"symbol":{"kind":"function_item","line_end":153,"line_start":111,"name":"run_ta_with_default_commands","path":"litebox_runner_optee_on_linux_userland/src/lib.rs"},"text":"litebox_runner_optee_on_linux_userland/src/lib.rs:111:153 run_ta_with_default_commands\nfn run_ta_with_default_commands(\n shim: &litebox_shim_optee::OpteeShim,\n ldelf_bin: &[u8],\n ta_bin: &[u8],\n) {\n for func_id in [UteeEntryFunc::OpenSession, UteeEntryFunc::CloseSession] {\n let params = [const { UteeParamOwned::None }; UteeParamOwned::TEE_NUM_PARAMS];\n\n if func_id == UteeEntryFunc::OpenSession {\n let session_token = session_manager().try_acquire_open_session_token().unwrap();\n let session_id = session_token.session_id().unwrap();\n let loaded_program = shim\n .load_ldelf(ldelf_bin, TeeUuid::default(), Some(ta_bin))\n .map_err(|_| {\n panic!(\"Failed to load ldelf\");\n })\n .unwrap();\n let entrypoints = loaded_program.entrypoints.as_ref().unwrap();\n unsafe {\n litebox_platform_linux_userland::run_thread_ref(\n entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n\n // In OP-TEE TA, each command invocation is like (re)starting the TA with a new stack with\n // loaded binary and heap. In that sense, we can create (and destroy) a stack\n // for each command freely.\n let _ = entrypoints\n .load_ta_context(params.as_slice(), session_id, func_id as u32, None)\n .map_err(|_| {\n panic!(\"Failed to load TA context\");\n });\n unsafe {\n litebox_platform_linux_userland::reenter_thread(\n entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n } else if func_id == UteeEntryFunc::CloseSession {\n }\n }\n}"},{"key":"ci-a981cb7b-ec3c0462-0","score":13.372835927742676,"symbol":{"kind":"function_item","line_end":628,"line_start":571,"name":"et_exec_interpreter_loads_top_down_above_low_heap","path":"litebox_shim_linux/src/loader/elf.rs"},"text":"litebox_shim_linux/src/loader/elf.rs:571:628 et_exec_interpreter_loads_top_down_above_low_heap\n fn et_exec_interpreter_loads_top_down_above_low_heap() {\n let _guard = crate::syscalls::tests::address_space_guard();\n let task = crate::syscalls::tests::init_platform(None);\n write_file(&task, \"/main\", &minimal_elf(ET_EXEC, Some(INTERP_PATH)));\n write_file(&task, \"/ld.so\", &minimal_elf(ET_DYN, None));\n\n let mut loader = ElfLoader::new(&task, \"/main\").expect(\"loader should parse test ELFs\");\n let main = loader\n .main\n .load_mapped(task.global.platform)\n .expect(\"main should load\");\n assert_eq!(main.base_addr, 0);\n\n let interp = loader\n .interp\n .as_mut()\n .expect(\"test main should have PT_INTERP\")\n .load_mapped(task.global.platform)\n .expect(\"interpreter should load\");\n\n // The interpreter must land high — via the top-down search — so the\n // low ET_EXEC brk heap below it is not capped. The exact address is\n // not asserted: `get_unmmaped_area` returns the highest free gap, and\n // host mappings seeded into the userland VMA tree can sit near the top\n // and push that gap below the very top slot (see `mm/linux.rs`). Assert\n // the invariant that matters — placement in the high half of the\n // address space, far above the low-heap region — not one exact slot.\n let addr_max = >::TASK_ADDR_MAX;\n assert!(\n interp.base_addr >= addr_max / 2,\n \"ET_EXEC interpreter loaded at {:#x}, near the low-heap region {:#x} rather than top-down high (>= {:#x})\",\n interp.base_addr,\n crate::loader::DEFAULT_LOW_ADDR,\n addr_max / 2,\n );\n\n // Release both images before returning. Every test in this binary shares\n // one host address space, but each builds its own task with its own VMM,\n // and a VMM models only its own mappings -- so anything this test leaves\n // mapped is invisible to the next test's placement search and collides\n // with whatever it picks. That is easy to miss on a host whose guest\n // range sits well clear of the host's own image; on arm64 macOS both\n // live above the 4 GiB `__PAGEZERO` floor, so the collision is routine.\n // Each synthetic image maps exactly one PT_LOAD page (`minimal_elf`\n // sets filesz == memsz == PAGE_SIZE). Do NOT derive the length from\n // `brk`: on a platform that requires syscall rewriting, `load_mapped`\n // pushes brk DEFAULT_RESERVED_SPACE_SIZE (16 MiB) past the image\n // without mapping that space, so a brk-derived munmap overshoots --\n // the top-down interpreter ends exactly at TASK_ADDR_MAX, which on\n // Linux x86-64 is the host TASK_SIZE (munmap EINVAL panics\n // deallocate_pages), and Windows' region walk asserts on the\n // never-committed tail.\n let exec_start = usize::try_from(EXEC_LOAD_ADDR).expect(\"load address fits usize\");\n task.sys_munmap(UserPtrMut::from_usize(exec_start), PAGE_SIZE)\n .expect(\"main image should unmap\");\n task.sys_munmap(UserPtrMut::from_usize(interp.base_addr), PAGE_SIZE)\n .expect(\"interpreter image should unmap\");\n }"},{"key":"ci-c5d8753d-d68df601-0","score":13.19179016907935,"symbol":{"kind":"section","line_end":160,"line_start":1,"name":"","path":"docs/benchmarks/awk-performance-investigation.md"},"text":"docs/benchmarks/awk-performance-investigation.md:1:160 \n# AWK performance and `vfork`/`time` investigation\n\nThis document records a performance and correctness investigation triggered by:\n\n```sh\n/usr/bin/time -p npx @openclew/litebox -- \\\n /bin/busybox awk 'BEGIN {a=0;b=1;for(i=0;i<100000000;i++){c=(a+b)%1000000007;a=b;b=c} print a}'\n```\n\nreporting `real 240.53` against `user 83.92` / `sys 1.39` (a wall-clock time\nroughly 2.8x the reported CPU time), and:\n\n```sh\nnpx @openclew/litebox -- /bin/busybox sh -c \\\n 'time /bin/busybox awk \"BEGIN {a=0;b=1;for(i=0;i<10000000;i++){c=(a+b)%1000000007;a=b;b=c} print a}\"'\n```\n\nfailing with `time: vfork: Invalid argument`.\n\nEnvironment for every measurement below: Apple M3 Pro, 11 cores, macOS\n26.3.1 (Darwin 25.3.0), built from a HEAD checkout of this repo (not the\n`npx @openclew/litebox` published package -- see \"npm package is stale\"\nbelow). This is a shared, actively-used development machine; several\nmeasurements below were taken under heavy, uncontrolled concurrent load from\nunrelated work (other terminal sessions building and testing this same\nrepo). Every number is labeled with the load average at the time it was\ntaken so it can be weighed accordingly -- this investigation treats ambient\ncontention as a variable to control for, not something to hide.\n\n## Summary of findings\n\n1. **The `vfork: Invalid argument` failure does not reproduce on current\n HEAD.** It was already fixed as a side effect of the prior\n delayed-address-space-handoff `fork`/`vfork` rework (commit `691fd87` and\n related), which predates this investigation. The `npx @openclew/litebox`\n package still fails because its pinned revision\n (`npm/lib/platform.js`'s `PINNED_REV`) is stale and predates that fix --\n see \"npm package is stale\" below.\n2. **A real, separate bug was found and fixed in this pass:** `wait4(...,\n &rusage)` left the caller's `rusage` buffer completely uninitialized\n whenever one was requested. Combined with (1) now succeeding, this is\n exactly what produced the user-visible symptom: `busybox time` prints\n whatever garbage was already in that guest memory, e.g. `sys\n 2367004162h 16m 32s`. Fixed by actually populating `ru_utime` from\n real, host-measured per-thread CPU time, and zeroing every other field.\n3. **The wall-clock-vs-CPU-time gap is real, not purely an accounting\n artifact, but it is highly sensitive to ambient system load** -- and this\n machine had extreme, uncontrolled load (verified up to ~18x\n oversubscription on an 11-core box) for parts of this investigation. A\n controlled, same-load A/B against native macOS `awk` shows litebox's own\n `user` CPU time is genuinely ~8x native's for the identical computation\n producing the identical result -- a real litebox-attributable CPU cost,\n independent of scheduling delay.\n4. **Root cause of that 8x, per `sample(1)` + live disassembly:** roughly\n half of all CPU samples during the hot loop land not in the guest's own\n code, but in a private JIT-allocated executable region holding\n AOT-rewritten guest code and PLT-style call stubs -- consistent with\n every arithmetic operation in the AWK script going through a real,\n dynamically-linked call into musl libc (`fmod`-shaped double modulo,\n allocation-shaped bit-twiddling) rather than being inlined. This is\n deferred as follow-up work (see \"Remaining overhead\" below) rather than\n attempted in this pass, because a fix would mean changing\n `litebox_syscall_rewriter`'s AOT rewriting itself, which needs more time\n to verify safely across the guest compatibility matrix than this pass\n had.\n\n## 1. The `vfork` failure: already fixed upstream of this investigation\n\nReproducing the exact original repro against a HEAD build:\n\n```sh\n$ target/release/litebox_runner_linux_on_macos_userland --initial-files alpine.tar -- \\\n /bin/busybox sh -c 'time /bin/busybox awk \"BEGIN {...}\"'\n490189494\nreal\t0m 12.01s\nuser\t0m 0.2741907030s\nsys\t2367004162h 16m 32s\n```\n\nIt no longer fails with `EINVAL` -- `vfork()`'s underlying\n`clone(CLONE_VM|CLONE_VFORK, ...)` is routed correctly by `do_clone` in\n`litebox_shim_linux/src/syscalls/process.rs` to `do_fork`, which already\nimplements the \"delayed address-space handoff\" model described in that\nfile's doc comments. That work landed before this investigation started.\n\nWhat's visibly broken instead is the *time* it prints, which is finding 2.\n\n### npm package is stale\n\n`npm/lib/platform.js`'s `PINNED_REV` is `497433858b0f8c52ea335df3576afb3e23e3a2e3`,\nwhich predates the fork/vfork rework entirely. Anyone running\n`npx @openclew/litebox` still gets the old `EINVAL` failure. This needs a\n`PINNED_REV` bump and republish to actually reach users -- tracked\nseparately, not done as part of this change (out of scope for a\ncorrectness/performance investigation; a version bump is its own\nreviewable, low-risk change).\n\n## 2. The `rusage` bug (fixed in this pass)\n\n### Root cause\n\n`Task::sys_wait4` in `litebox_shim_linux/src/syscalls/process.rs` used to\nhandle a non-null `rusage` pointer like this:\n\n```rust\nif rusage != 0 {\n // Reporting zeroed usage would be a lie that some callers act on; refusing is not,\n // and no caller in sight asks for it.\n log_unsupported!(\"wait4 with a rusage buffer\");\n}\n```\n\nIt logged and then did *nothing else* -- no error was returned, and the\nbuffer was never written. `wait4` reports success, and the guest's `struct\nrusage` is left exactly as it was before the call: whatever bytes happened\nto already be in that stack or heap allocation. `busybox time` reads\n`ru_utime`/`ru_stime` straight out of that memory and prints them, so the\nobserved `sys 2367004162h 16m 32s` is simply uninitialized memory\nreinterpreted as a `timeval`. This is also, independent of how silly the\noutput looks, an information-disclosure bug: guest code that requests\n`rusage` gets back bytes of guest memory it never wrote, with no relation\nto its own execution.\n\n### Fix\n\n- `litebox_common_linux::Rusage` -- a `#[repr(C)]` struct matching musl's\n LP64 `struct rusage` layout (`ru_utime`/`ru_stime` as the existing\n `TimeVal` type, then the fourteen POSIX `long` fields, then musl's\n 16-`long` reserved tail), written into guest memory the same way\n `Sysinfo`/`Statfs`/etc. already are.\n- `Process::cpu_time_nanos`, an `AtomicU64` accumulator. Each thread of a\n process adds its own `ShimPlatform::thread_cpu_time()` reading (real,\n host-measured, per-thread CPU time -- already used for\n `CLOCK_THREAD_CPUTIME_ID`, and already covered by an existing test that\n `thread_cpu_time` tracks real CPU usage, not wall-clock time) as it exits,\n in `Task::prepare_for_exit`. This has to happen on the exiting thread\n itself: `CLOCK_THREAD_CPUTIME_ID`-style clocks only ever read the calling\n thread's own counter.\n- `ProcessTable::record_exit`/`reap` now carry that accumulated value\n alongside the exit status.\n- `Task::sys_wait4` now writes a real `Rusage` value when a caller passes a\n non-null pointer: `ru_utime` is the process's real accumulated CPU time,\n every other field (including `ru_stime`) is explicitly zero rather than\n fabricated -- guest syscalls run as ordinary host user-mode Rust in this\n shim, so there is no meaningful \"kernel time\" of its own to attribute to\n `ru_stime`, and reporting a fabricated nonzero value would trade one lie\n for another. Zero, clearly labeled as \"unmeasured\" in the surrounding\n comment, is the honest answer.\n\n### After\n\n```sh\n$ target/release/litebox_runner_linux_on_macos_userland --initial-files alpine.tar -- \\\n /bin/busybox sh -c 'time /bin/busybox awk \"BEGIN {...10M iters...}\"'\n490189494\nreal\t0m 11.29s\nuser\t0m 7.18s\nsys\t0m 0.00s\n```\n"},{"key":"ci-ea2224bb-9e5d768c-0","score":12.579849872153302,"symbol":{"kind":"function_item","line_end":147,"line_start":53,"name":"run","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:53:147 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let tar_file = &cli_args.initial_files;\n if tar_file.extension().and_then(|x| x.to_str()) != Some(\"tar\") {\n anyhow::bail!(\"Expected a .tar file, found {}\", tar_file.display());\n }\n let tar_data = std::fs::read(tar_file)\n .map_err(|e| anyhow!(\"Could not read tar file at {}: {}\", tar_file.display(), e))?;\n\n let platform = Platform::new();\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n // The program path is a Unix-style path inside the tar archive.\n let prog_path = &cli_args.program_and_arguments[0];\n\n let initial_file_system = {\n let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem.with_root_privileges(|fs| {\n use litebox::fs::FileSystem as _;\n fs.mkdir(\n \"/tmp\",\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap();\n fs.chown(\"/tmp\", Some(1000), Some(1000)).unwrap();\n\n // Standard FHS directories that guest tools expect to already exist (e.g. `apk`\n // opens a log file under `/var/log`) but that don't survive as empty-directory\n // entries when an OCI image's rootfs is scanned into a file-based tar: an empty\n // directory has no file contents, so it produces no tar entry, and `TarRo`'s\n // directory tree is inferred purely from file paths.\n for dir in [\"/run\", \"/var\", \"/var/log\", \"/var/cache\", \"/var/tmp\"] {\n fs.mkdir(\n dir,\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|_| {\n panic!(\"{dir} creation cannot fail on a fresh in-memory file system\")\n });\n }\n });\n\n shim_builder.default_fs(in_mem, tar_data.into())\n };\n let initial_file_system = std::sync::Arc::new(initial_file_system);\n\n let shim = shim_builder.build();\n let argv = cli_args\n .program_and_arguments\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp: Vec<_> = cli_args\n .environment_variables\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp = if cli_args.forward_environment_variables {\n envp.into_iter()\n .chain(std::env::vars().map(|(k, v)| {\n std::ffi::CString::new(k.bytes().chain(*b\"=\").chain(v.bytes()).collect::>())\n .unwrap()\n }))\n .collect()\n } else {\n envp\n };\n\n let program = shim\n .load_program(\n initial_file_system,\n platform.init_task(),\n prog_path,\n argv,\n envp,\n )\n .unwrap();\n unsafe {\n litebox_platform_windows_userland::run_thread(\n program.entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n std::process::exit(program.process.wait())\n}"},{"key":"ci-99368d6-ad38df33-0","score":11.889958423421906,"symbol":{"kind":"section","line_end":47,"line_start":1,"name":"","path":"README.md"},"text":"README.md:1:47 \n# LiteBox\n\n> A security-focused library OS\n\n> [!NOTE] \n> This project is currently actively evolving and improving. While we are\n> working toward a stable release, some APIs and interfaces may change as the\n> design continues to mature. You are welcome to explore and experiment, but if\n> you need long-term stability, it may be best to wait for a stable release, or\n> be prepared to adapt to updates along the way.\n\nLiteBox is a sandboxing library OS that drastically cuts down the interface to the host, thereby reducing attack surface. It focuses on easy interop of various \"North\" shims and \"South\" platforms. LiteBox is designed for usage in both kernel and non-kernel scenarios.\n\nLiteBox exposes a Rust-y [`nix`](https://docs.rs/nix)/[`rustix`](https://docs.rs/rustix)-inspired \"North\" interface when it is provided a `Platform` interface at its \"South\". These interfaces allow for a wide variety of use-cases, easily allowing for connection between any of the North--South pairs.\n\nExample use cases include:\n- Running unmodified Linux programs on Windows\n- Running unmodified Linux programs on macOS (Apple Silicon) -- see [docs/macos.md](./docs/macos.md)\n- Sandboxing Linux applications on Linux\n- Run programs on top of SEV SNP\n- Running OP-TEE programs on Linux\n- Running on LVBS\n\n![LiteBox and related projects](./.figures/litebox.svg)\n\n## Contributing\n\nSee the following files for details:\n\n- [CONTRIBUTING.md](./CONTRIBUTING.md)\n- [CODE_OF_CONDUCT.md](./CODE_OF_CONDUCT.md)\n- [SECURITY.md](./SECURITY.md)\n- [SUPPORT.md](./SUPPORT.md)\n- [docs/roadmap.md](./docs/roadmap.md) for known gaps and follow-up work\n\n## License\n\nMIT License. See [./LICENSE](./LICENSE) for details.\n\n## Trademarks\n\nThis project may contain trademarks or logos for projects, products, or services. Authorized use of Microsoft \ntrademarks or logos is subject to and must follow \n[Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general).\nUse of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship.\nAny use of third-party trademarks or logos are subject to those third-party's policies."}],"commits":[{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":13.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":10.0},{"hash":"624f0ceedf841b147c5e972104d7cfed355ccc90","message":"feat(net): guest web access without root via an in-stack HTTP proxy","score":10.0},{"hash":"323d13318ea10946e74e99a7e24d593c9ef6e7ed","message":"feat(shim): the syscall surface an X11 desktop stack needs","score":8.0},{"hash":"d06199534c3cf1090fdb261b8c800d380dea941e","message":"fix: getrusage/wait4 guest-stack overrun, doc gate, pipe-hang probes","score":8.0},{"hash":"de65eb0c45b76e017976a322bf14be458c4239d0","message":"fix(ci): clear remaining clippy pedantic errors on macOS and rewriter","score":8.0},{"hash":"4a62d24fb8dba3be51e29bf89b38449d8bbd7d19","message":"feat(terminal): honor TCSETSW/TCSETSF, add per-guest IP override, npm 0.1.1","score":8.0},{"hash":"b97641713732df7cb18e20ebbd80e675e777fba6","message":"feat(rfb): runner-side VNC server presenting the guest /dev/fb0 framebuffer","score":7.0},{"hash":"fa5317648180150f75c2cb23e00d2ea16f13c843","message":"shim: log guest hardware exceptions; harness: surface runner errors","score":6.0},{"hash":"de0d9a8b391bb610126b398a2619989025675cf9","message":"fix(ci): green the cross-platform test suite","score":6.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788372947079-500-e87aa6c62c15763d","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"dd939f61039bb2ca","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-420721788372533381.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-420721788372533381.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-43801788215590477.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-43801788215590477.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-45351788215596032.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-45351788215596032.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-51911788287969870.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-51911788287969870.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-519201788294203098.json b/.agentplug/plugin-dispatch/out/gm-codesearch-519201788294203098.json new file mode 100644 index 0000000000..f1fe8907b7 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-519201788294203098.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-4503fc80-eff84b10-0","score":33.4684802751792,"symbol":{"kind":"function_item","line_end":2349,"line_start":2262,"name":"test_page_provider","path":"litebox_platform_windows_userland/src/lib.rs"},"text":"litebox_platform_windows_userland/src/lib.rs:2262:2349 test_page_provider\n fn test_page_provider() {\n let collect_regions = |r| {\n let mut regions = Vec::new();\n process_memory_range_by_regions(\n r,\n |region, state| -> Result {\n regions.push((region, state));\n Ok(true)\n },\n )\n .unwrap();\n regions\n };\n\n let platform = WindowsUserland::new();\n let system_allocation_granularity =\n platform.sys_info.read().unwrap().dwAllocationGranularity as usize;\n // Allocate some pages: it should reserve `system_allocation_granularity` bytes but only commit 0x1000 bytes\n let addr = >::allocate_pages(\n platform,\n 0..0x1000,\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n assert_eq!(\n collect_regions(addr..addr + system_allocation_granularity),\n vec![\n (\n addr..addr + 0x1000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n (\n addr + 0x1000..addr + system_allocation_granularity,\n windows_sys::Win32::System::Memory::MEM_RESERVE\n ),\n ]\n );\n\n assert!(system_allocation_granularity >= 0x1_0000);\n // We should be able to allocate [addr + 0x8000, addr + 0x1_0000)\n let addr2 = >::allocate_pages(\n platform,\n (addr + 0x8000)..(addr + 0x1_0000),\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n // Even though `fixed_address` is false, we should still get the requested address if it's free.\n assert_eq!(addr2, addr + 0x8000);\n assert_eq!(\n collect_regions(addr..addr + 0x1_0000),\n vec![\n (\n addr..addr + 0x1000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n (\n addr + 0x1000..addr + 0x8000,\n windows_sys::Win32::System::Memory::MEM_RESERVE\n ),\n (\n addr + 0x8000..addr + 0x1_0000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n ]\n );\n\n // Try to allocate [addr + 0x4000, addr + 0x1_0000), which overlaps with existing committed pages.\n // OS should allocate a new region instead of the requested one (as `fixed_address` is false)\n let addr3 = >::allocate_pages(\n platform,\n (addr + 0x4000)..(addr + 0x1_0000),\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n assert_ne!(addr3, addr + 0x4000);\n }"},{"key":"ci-26a63893-31523872-0","score":15.966694977189968,"symbol":{"kind":"trait_item","line_end":49,"line_start":14,"name":"MemoryProvider","path":"litebox_platform_linux_kernel/src/mm/mod.rs"},"text":"litebox_platform_linux_kernel/src/mm/mod.rs:14:49 MemoryProvider\npub trait MemoryProvider {\n /// Global virtual address offset for one-to-one mapping of physical memory\n /// to kernel virtual memory.\n const GVA_OFFSET: VirtAddr;\n /// Mask for private page table entry (e.g., SNP encryption bit).\n /// For simplicity, we assume the mask is constant.\n const PRIVATE_PTE_MASK: u64;\n\n /// Allocate (1 << `order`) virtually and physically contiguous pages from global allocator.\n fn mem_allocate_pages(order: u32) -> Option<*mut u8>;\n\n /// De-allocates virtually and physically contiguous pages returned from [`Self::mem_allocate_pages`].\n ///\n /// # Safety\n ///\n /// The caller must ensure that the `ptr` is valid and was allocated by this allocator.\n ///\n /// `order` must be the same as the one used during allocation.\n unsafe fn mem_free_pages(ptr: *mut u8, order: u32);\n\n /// Obtain physical address (PA) of a page given its VA\n fn va_to_pa(va: VirtAddr) -> PhysAddr {\n PhysAddr::new_truncate(va - Self::GVA_OFFSET)\n }\n\n /// Obtain virtual address (VA) of a page given its PA\n fn pa_to_va(pa: PhysAddr) -> VirtAddr {\n let pa = pa.as_u64() & !Self::PRIVATE_PTE_MASK;\n VirtAddr::new_truncate(pa + Self::GVA_OFFSET.as_u64())\n }\n\n /// Set physical address as private via mask.\n fn make_pa_private(pa: PhysAddr) -> PhysAddr {\n PhysAddr::new_truncate(pa.as_u64() | Self::PRIVATE_PTE_MASK)\n }\n}"},{"key":"ci-6c20d5c9-d836403-0","score":15.378706843769072,"symbol":{"kind":"trait_item","line_end":61,"line_start":15,"name":"MemoryProvider","path":"litebox_platform_lvbs/src/mm/mod.rs"},"text":"litebox_platform_lvbs/src/mm/mod.rs:15:61 MemoryProvider\npub trait MemoryProvider {\n /// Global virtual address offset for one-to-one mapping of physical memory\n /// to kernel virtual memory.\n const GVA_OFFSET: VirtAddr;\n /// Mask for private page table entry (e.g., SNP encryption bit).\n /// For simplicity, we assume the mask is constant.\n const PRIVATE_PTE_MASK: u64;\n\n /// Allocate (1 << `order`) virtually and physically contiguous pages from global allocator.\n fn mem_allocate_pages(order: u32) -> Option<*mut u8>;\n\n /// De-allocates virtually and physically contiguous pages returned from [`Self::mem_allocate_pages`].\n ///\n /// # Safety\n ///\n /// The caller must ensure that the `ptr` is valid and was allocated by this allocator.\n ///\n /// `order` must be the same as the one used during allocation.\n unsafe fn mem_free_pages(ptr: *mut u8, order: u32);\n\n /// Add a range of memory to global allocator.\n /// Morally, the global allocator takes ownership of this range of memory.\n ///\n /// # Safety\n ///\n /// The caller must ensure that the memory range is valid and not used by any others.\n unsafe fn mem_fill_pages(start: usize, size: usize);\n\n /// Obtain physical address (PA) of a page given its kernel VA.\n ///\n /// The VTL1 kernel region maps kernel memory via `VA = PA + KERNEL_OFFSET`.\n fn va_to_pa(va: VirtAddr) -> PhysAddr {\n PhysAddr::new_truncate(va.as_u64() - crate::KERNEL_OFFSET)\n }\n\n /// Obtain the kernel virtual address (VA) of a page given its PA.\n ///\n /// The VTL1 kernel region maps kernel memory via `VA = PA + KERNEL_OFFSET`.\n fn pa_to_va(pa: PhysAddr) -> VirtAddr {\n VirtAddr::new_truncate(pa.as_u64() + crate::KERNEL_OFFSET)\n }\n\n /// Set physical address as private via mask.\n fn make_pa_private(pa: PhysAddr) -> PhysAddr {\n PhysAddr::new_truncate(pa.as_u64() | Self::PRIVATE_PTE_MASK)\n }\n}"},{"key":"ci-b7eac667-5b2d5edf-0","score":15.264753061052634,"symbol":{"kind":"function_item","line_end":4311,"line_start":4279,"name":"with_signal_alt_stack_actually_registers_one","path":"litebox_platform_macos_userland/src/lib.rs"},"text":"litebox_platform_macos_userland/src/lib.rs:4279:4311 with_signal_alt_stack_actually_registers_one\n/// disabled.\n#[cfg(test)]\n#[test]\nfn with_signal_alt_stack_actually_registers_one() {\n // `with_signal_alt_stack` does a real host `mmap`/`munmap` of its own,\n // which can race `allocate_jit_pages_hint_honors_the_suggested_address`'s\n // address-space probing under the parallel test harness; serialize\n // against it the same way guest-entry tests already do.\n let _serial = guest::tests::TEST_SERIAL\n .lock()\n .unwrap_or_else(std::sync::PoisonError::into_inner);\n\n let before = current_signal_stack();\n let during = with_signal_alt_stack(current_signal_stack);\n let after = current_signal_stack();\n\n assert_ne!(\n during.ss_sp, before.ss_sp,\n \"with_signal_alt_stack must install a stack distinct from whatever \\\n (if any) the thread already had\"\n );\n assert_eq!(\n during.ss_flags & libc::SS_DISABLE,\n 0,\n \"the installed stack must actually be enabled\"\n );\n assert!(\n during.ss_size >= libc::SIGSTKSZ,\n \"the installed stack must be large enough to actually run a handler\"\n );\n assert_eq!(\n (after.ss_sp, after.ss_size, after.ss_flags),\n (before.ss_sp, before.ss_size, before.ss_flags),"},{"key":"ci-909e246d-f4d42a4e-0","score":14.207123633013945,"symbol":{"kind":"function_item","line_end":297,"line_start":215,"name":"test_vmm_page_fault","path":"litebox_platform_linux_kernel/src/mm/tests.rs"},"text":"litebox_platform_linux_kernel/src/mm/tests.rs:215:297 test_vmm_page_fault\nfn test_vmm_page_fault() {\n let start_addr: usize = 0x1_0000;\n let p4 = PageTableAllocator::::allocate_frame(true).unwrap();\n let platform = MockKernel::new(p4.start_address());\n let litebox = LiteBox::new(platform);\n let vmm = PageManager::<_, PAGE_SIZE>::new(&litebox);\n unsafe {\n assert_eq!(\n vmm.create_writable_pages(\n Some(NonZeroAddress::new(start_addr).unwrap()),\n NonZeroPageSize::new(4 * PAGE_SIZE).unwrap(),\n CreatePagesFlags::FIXED_ADDR,\n |_: UserMutPtr| Ok(0),\n )\n .unwrap()\n .as_usize(),\n start_addr\n );\n }\n // [0x1_0000, 0x1_4000)\n\n // Access page w/o mapping\n assert!(matches!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 6 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n },\n Err(PageFaultError::AccessError(_))\n ));\n\n // Access non-present page w/ mapping\n assert!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 2 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n }\n .is_ok()\n );\n\n // insert stack mapping\n let stack_addr: usize = 0x1000_0000;\n unsafe {\n assert_eq!(\n vmm.create_stack_pages(\n Some(NonZeroAddress::new(stack_addr).unwrap()),\n NonZeroPageSize::new(4 * PAGE_SIZE).unwrap(),\n CreatePagesFlags::FIXED_ADDR,\n )\n .unwrap()\n .as_usize(),\n stack_addr\n );\n }\n // [0x1_0000, 0x1_4000), [0x1000_0000, 0x1000_4000)\n // Test stack growth\n assert!(\n unsafe {\n vmm.handle_page_fault(stack_addr - PAGE_SIZE, PageFaultErrorCode::USER_MODE.bits())\n }\n .is_ok()\n );\n assert_eq!(\n vmm.mappings()\n .iter()\n .map(|v| v.0.clone())\n .collect::>(),\n vec![0x1_0000..0x1_4000, 0x0fff_f000..0x1000_4000]\n );\n // Cannot grow stack too far\n assert!(matches!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 100 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n },\n Err(PageFaultError::AllocationFailed)\n ));\n}"},{"key":"ci-fc090d9e-dc553b19-0","score":12.590946147236258,"symbol":{"kind":"function_item","line_end":304,"line_start":257,"name":"sys_madvise","path":"litebox_common_linux/src/mm.rs"},"text":"litebox_common_linux/src/mm.rs:257:304 sys_madvise\npub fn sys_madvise<\n Platform: litebox::platform::RawPointerProvider\n + litebox::sync::RawSyncPrimitivesProvider\n + litebox::platform::PageManagementProvider<{ litebox::mm::linux::PAGE_SIZE }>,\n>(\n pm: &litebox::mm::PageManager,\n addr: UserPtrMut,\n len: usize,\n advice: crate::MadviseBehavior,\n) -> Result<(), Errno> {\n if addr.as_usize() & !PAGE_MASK != 0 {\n return Err(Errno::EINVAL);\n }\n if len == 0 {\n return Ok(());\n }\n let aligned_len = len.next_multiple_of(PAGE_SIZE);\n if aligned_len == 0 {\n // overflow\n return Err(Errno::EINVAL);\n }\n let Some(_end) = addr.as_usize().checked_add(aligned_len) else {\n return Err(Errno::EINVAL);\n };\n\n let addr = addr.to_platform_ptr::();\n match advice {\n crate::MadviseBehavior::Normal\n | crate::MadviseBehavior::DontFork\n | crate::MadviseBehavior::DoFork => {\n // No-op for now, as we don't support fork yet.\n Ok(())\n }\n crate::MadviseBehavior::DontNeed => {\n // After a successful MADV_DONTNEED operation, the semantics of memory access in the specified region are changed:\n // subsequent accesses of pages in the range will succeed, but will result in either repopulating the memory contents\n // from the up-to-date contents of the underlying mapped file (for shared file mappings, shared anonymous mappings,\n // and shmem-based techniques such as System V shared memory segments) or zero-fill-on-demand pages for anonymous private mappings.\n //\n // Note we do not support shared memory yet, so this is just to discard the pages without removing the mapping.\n unsafe { pm.reset_pages(addr, aligned_len, false) }.map_err(Errno::from)\n }\n crate::MadviseBehavior::Free => {\n unsafe { pm.reset_pages(addr, aligned_len, true) }.map_err(Errno::from)\n }\n _ => unimplemented!(\"Unsupported madvise behavior {:?}\", advice),\n }\n}"},{"key":"ci-30003d83-26521427-0","score":12.415994645339476,"symbol":{"kind":"section","line_end":131,"line_start":1,"name":"","path":"docs/macos.md"},"text":"docs/macos.md:1:131 \n# LiteBox on macOS (Apple Silicon)\n\nLiteBox runs guest instructions natively; only the *system* interface is\nvirtualized. On an Apple Silicon Mac that means the only sensible configuration\nis an **AArch64 Linux guest on an AArch64 macOS host** — no emulation anywhere.\nThere is deliberately no x86-64 macOS platform: an x86-64 guest would need\ninstruction emulation, which is the thing this design exists to avoid.\n\nThis document covers what works today, what the host imposes, and what is left\nbefore a guest can actually execute.\n\n## What is in the tree\n\n| Piece | State |\n| --- | --- |\n| `litebox_platform_macos_userland` | The macOS \"South\" platform: memory, locking, time, signals, timers, threads, TLS, randomness, derived keys, stdio, `utun` networking, fault recovery. |\n| `litebox` core | Builds for `aarch64-apple-darwin`, including the Mach-O exception table. |\n| `litebox_shim_linux` | The Linux \"North\" shim, ported to AArch64: signal frames, syscall entry/return, thread-pointer handling, `stat`/`uname` ABI, exception decoding. |\n| `litebox_syscall_rewriter` | Already had AArch64 support (`arm64.rs`) for rewriting `SVC` and `TPIDR_EL0` accesses in Linux ELF images. |\n| `litebox_packager` | OCI mode now pulls the image matching the host architecture, and builds on Apple Silicon. |\n| Guest entry | **Implemented** (context switch + syscall dispatch), tested on real hardware. A syscall-only guest runs end to end; the guest thread-pointer plumbing and non-syscall event paths remain. See [Remaining work](#remaining-work). |\n\n## Building\n\n```sh\nrustup target add aarch64-apple-darwin\ncargo build --workspace --exclude litebox_runner_lvbs --exclude litebox_runner_snp\n```\n\n`litebox_runner_lvbs` and `litebox_runner_snp` are freestanding images for\ncustom targets and are not built for a hosted target on any platform.\n\nCI covers this in the `Build and Test macOS (Apple Silicon)` job, which also\ncompiles and runs `litebox_platform_macos_userland/tests/darwin_abi_probe.c`\nagainst the runner's real SDK headers -- the only check in this repo that\nverifies the crate's hand-written Mach/BSD struct layouts (used by the fault\nhandler to read `ucontext_t::uc_mcontext`) against an actual Darwin toolchain,\nsince nothing else in a Linux-hosted development loop can.\n\n### Hypervisor.framework release boundary\n\nThe stock-code backend requires Apple Silicon, macOS 26 or newer, and a macOS\n26 SDK or newer. `litebox_platform_macos_userland/build.rs` compiles the narrow\nC boundary and linked EL1 monitor against the active SDK headers, so Apple enum,\nobject, and structure layouts do not get copied into Rust constants. A release\nrunner is built and signed with the checked-in entitlement manifest as follows:\n\n```sh\ncargo build --release --locked -p litebox_runner_linux_on_macos_userland\ncodesign --force --options runtime --sign - \\\n --entitlements litebox_runner_linux_on_macos_userland/entitlements.plist \\\n target/release/litebox_runner_linux_on_macos_userland\ncodesign --display --entitlements - \\\n target/release/litebox_runner_linux_on_macos_userland\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-boundary\n```\n\nThe diagnostic validates the active-SDK configuration, maps and unmaps the\nmonitor, and creates, verifies, and destroys a vCPU. The linked 16 KiB monitor\nremains the source of truth, but the VM owns an aligned allocator-backed copy:\nHypervisor.framework rejects file-backed Mach-O `__TEXT` pages as stage-two\nbacking.\n\n`com.apple.security.hypervisor=true` is mandatory for\nHypervisor.framework VM creation. `com.apple.security.cs.allow-jit=true` remains\nrequired when the native rewritten backend is packaged with Hardened Runtime.\nThe legacy `com.apple.vm.hypervisor` entitlement belongs only to deployment\ntargets through macOS 10.15 and is deliberately absent. Notarization and\nHardened Runtime do not grant Hypervisor.framework authorization; the final\nexecutable still needs the hypervisor entitlement.\n\n## What the host imposes\n\n### 16 KiB pages\n\nApple Silicon's page size is 16 KiB. Every fixed mapping and every protection\nchange must be aligned to it, so `litebox::mm::linux::PAGE_SIZE` is 16384 on\nthis target rather than 4096. The guest sees the same value through `AT_PAGESZ`,\nwhich is exactly how a Linux kernel configured for 16 KiB or 64 KiB pages\nreports itself.\n\nAArch64 ELF images are conventionally linked with a 64 KiB maximum page size, so\ntheir `PT_LOAD` segments stay aligned either way. An image built with 4 KiB\nsegment alignment will not map cleanly.\n\n### The first 4 GiB is unusable\n\nAn arm64 Mach-O process reserves `[0, 4 GiB)` as the `__PAGEZERO` segment:\nunmapped and impossible to map over. `TASK_ADDR_MIN` is therefore `0x1_0000_0000`.\n\nThe practical consequence is that guest images must be position-independent, or\nlinked above 4 GiB. An `ET_EXEC` binary linked at the customary `0x400000`\ncannot be loaded at its preferred address on this host.\n\n### W^X, `MAP_JIT`, and code signing\n\nmacOS refuses to make anonymous memory executable through the ordinary path, and\nrefuses to add `PROT_EXEC` to anything that was ever writable. The supported\nescape hatch is `MAP_JIT`, which the platform passes whenever a mapping requests\n`EXEC`. Using it has two consequences:\n\n1. **The JIT entitlement is only load-bearing under the Hardened Runtime.**\n Per Apple's own documentation, `com.apple.security.cs.allow-jit` is required\n only when a binary has the Hardened Runtime enabled (`codesign --options\n runtime`, which in turn is what notarization requires); without it,\n `MAP_JIT` works with or without the entitlement present. The command below\n ad-hoc-signs with the entitlement anyway -- it costs nothing and future-proofs\n a later `--options runtime`, notarized build -- but for local development\n outside Gatekeeper, neither the entitlement nor notarization is actually\n required for `MAP_JIT` itself to work. Create an entitlements file:\n\n ```xml\n \n \n \n \n com.apple.security.cs.allow-jit\n \n \n \n ```\n\n and sign the runner with it:\n\n ```sh\n codesign --sign - --entitlements litebox.entitlements --force \n ```\n\n2. **Writes must be bracketed.** A `MAP_JIT` mapping is writable *or* executable"},{"key":"ci-4e44ed81-7a49d500-0","score":11.944629536874348,"symbol":{"kind":"function_item","line_end":272,"line_start":258,"name":"test_guard_page_gap","path":"litebox_platform_lvbs/src/mm/vmap.rs"},"text":"litebox_platform_lvbs/src/mm/vmap.rs:258:272 test_guard_page_gap\n fn test_guard_page_gap() {\n let allocator = VmapRegionAllocator::new();\n\n let va_a = allocator.allocate_va(1).unwrap();\n let va_b = allocator.allocate_va(1).unwrap();\n\n // Allocations should be separated by at least GUARD_PAGES unmapped pages\n let gap_pages = (va_b.as_u64() - va_a.as_u64()) / PAGE_SIZE as u64;\n assert!(\n gap_pages >= (1 + GUARD_PAGES as u64),\n \"expected at least {} pages between allocations, got {}\",\n 1 + GUARD_PAGES,\n gap_pages\n );\n }"},{"key":"ci-5e03415c-efa79bde-0","score":10.676593508331,"symbol":{"kind":"function_item","line_end":314,"line_start":292,"name":"allocate_stack","path":"litebox_shim_optee/src/loader/ta_stack.rs"},"text":"litebox_shim_optee/src/loader/ta_stack.rs:292:314 allocate_stack\npub(crate) fn allocate_stack(task: &crate::Task, stack_base: Option) -> Option {\n let sp = if let Some(stack_base) = stack_base {\n UserMutPtr::from_usize(stack_base)\n } else {\n let length = litebox::mm::linux::NonZeroPageSize::new(super::DEFAULT_STACK_SIZE)\n .expect(\"DEFAULT_STACK_SIZE is not page-aligned\");\n unsafe {\n task.global\n .pm\n .create_stack_pages(\n None,\n length,\n // Pre-populate: stack initialization runs before run_thread_arch\n // sets up the kernel-mode demand paging infrastructure.\n CreatePagesFlags::POPULATE_PAGES_IMMEDIATELY,\n )\n .ok()?\n }\n };\n let stack = TaStack::new(sp, super::DEFAULT_STACK_SIZE)?;\n\n Some(stack)\n}"},{"key":"ci-e7145f0-4f97b5e8-0","score":10.541822121941449,"symbol":{"kind":"function_item","line_end":317,"line_start":231,"name":"test_vmm_page_fault","path":"litebox_platform_lvbs/src/mm/tests.rs"},"text":"litebox_platform_lvbs/src/mm/tests.rs:231:317 test_vmm_page_fault\nfn test_vmm_page_fault() {\n let start_addr: usize = 0x1_0000;\n let platform = MockKernel::new(\n x86_64::PhysAddr::new(0),\n x86_64::PhysAddr::new(0),\n x86_64::PhysAddr::new(0),\n x86_64::PhysAddr::new(0),\n );\n let litebox = LiteBox::new(platform);\n let vmm = PageManager::<_, PAGE_SIZE>::new(&litebox);\n unsafe {\n assert_eq!(\n vmm.create_writable_pages(\n Some(litebox::mm::linux::NonZeroAddress::new(start_addr).unwrap()),\n litebox::mm::linux::NonZeroPageSize::new(4 * PAGE_SIZE).unwrap(),\n litebox::mm::linux::CreatePagesFlags::FIXED_ADDR,\n |_: UserMutPtr| Ok(0),\n )\n .unwrap()\n .as_usize(),\n start_addr\n );\n }\n // [0x1_0000, 0x1_4000)\n\n // Access page w/o mapping\n assert!(matches!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 6 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n },\n Err(PageFaultError::AccessError(_))\n ));\n\n // Access non-present page w/ mapping\n assert!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 2 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n }\n .is_ok()\n );\n\n // insert stack mapping\n let stack_addr: usize = 0x1000_0000;\n unsafe {\n assert_eq!(\n vmm.create_stack_pages(\n Some(litebox::mm::linux::NonZeroAddress::new(stack_addr).unwrap()),\n litebox::mm::linux::NonZeroPageSize::new(4 * PAGE_SIZE).unwrap(),\n litebox::mm::linux::CreatePagesFlags::FIXED_ADDR,\n )\n .unwrap()\n .as_usize(),\n stack_addr\n );\n }\n // [0x1_0000, 0x1_4000), [0x1000_0000, 0x1000_4000)\n // Test stack growth\n assert!(\n unsafe {\n vmm.handle_page_fault(stack_addr - PAGE_SIZE, PageFaultErrorCode::USER_MODE.bits())\n }\n .is_ok()\n );\n assert_eq!(\n vmm.mappings()\n .iter()\n .map(|v| v.0.clone())\n .collect::>(),\n vec![0x1_0000..0x1_4000, 0x0fff_f000..0x1000_4000]\n );\n // Cannot grow stack too far\n assert!(matches!(\n unsafe {\n vmm.handle_page_fault(\n start_addr + 100 * PAGE_SIZE,\n PageFaultErrorCode::USER_MODE.bits(),\n )\n },\n Err(PageFaultError::AllocationFailed)\n ));\n}"}],"commits":[{"hash":"f027100797e13e383f78e52604d4d7f021932271","message":"fix(xfce): serialize Thunar and xterm launch to avoid X client startup race","score":2.0},{"hash":"7c20fe35f3d54465097941ffd6c2b26f124848aa","message":"xfce fixed","score":2.0},{"hash":"d7a6857d0394e7c83ddffeb599d9d402f04dd6f8","message":"fix(signals): implement tkill/tgkill delivery to sibling threads","score":2.0},{"hash":"52253fdae98d1a0ddbd8f8e5119c22c1e9fc3ef8","message":"fix(packager): scope x18 gate to the live smoke-test closure, add simdutf","score":2.0},{"hash":"a2236699e12fbed0afad6a2671d5ee5b30e262ec","message":"fix(packager): scope the x18 residual gate to what the image ships","score":2.0},{"hash":"fe91ffff8cbe504aace32fc9d428d0620989c2b2","message":"fix(vnc-web): bound stalled pre-upgrade clients to a 5s read timeout","score":2.0},{"hash":"ae07970ae0fd60be4e6df5204f4279f0d683359d","message":"feat(packager): script to rebuild the desktop package closure with -ffixed-x18","score":2.0},{"hash":"a9c705ba5318889309d3111b1a02e4cc1bc110d0","message":"fix(channel): replace ringbuf caching halves with one shared locked deque","score":2.0},{"hash":"76183166f32114f8deaff3b94510f97d28031693","message":"fix(test): retry reserve-alignment host-fidelity probe on stale address","score":2.0},{"hash":"5251e6a45413ae95629010a02e13a7959a9e3517","message":"feat(evdev): /dev/input/event* emulation wired to VNC input","score":2.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788294642340-500-86e442c6a1436862","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"9cdea24b57b464f6","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-519201788294203098.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-519201788294203098.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-58931788199194315.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-58931788199194315.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-603421788200113306.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-603421788200113306.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-622231788200135237.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-622231788200135237.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-63781788199200961.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-63781788199200961.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-648941788200188623.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-648941788200188623.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-64971788199202674.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-64971788199202674.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-68281788199209977.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-68281788199209977.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-692381788302938563.json b/.agentplug/plugin-dispatch/out/gm-codesearch-692381788302938563.json new file mode 100644 index 0000000000..7ea60ccf6e --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-692381788302938563.json @@ -0,0 +1 @@ +{"data":{"degraded":false,"mode":"root_scoped","root":"/Users/dylanwong/litebox/litebox_platform_macos_userland","vector_hits":[{"distance":0.4694764316082001,"kind":"function_item","line_end":112,"line_start":57,"name":"main","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/build.rs","snippet":"fn main() -> Result<(), Box> {\n println!(\"cargo:rerun-if-changed=src/hvf_sdk.c\");\n println!(\"cargo:rerun-if-changed=src/hvf_monitor.S\");\n for variable in [\n \"MACOSX_DEPLOYMENT_TARGET\",\n \"DEVELOPER_DIR\",\n \"SDKROOT\",\n \"TOOLCHAINS\",\n ] {\n println!(\"cargo:rerun-if-env-changed={variable}\");\n }\n\n if env::var(\"CARGO_CFG_TARGET_OS\")? != \"maco"},{"distance":0.47474658489227295,"kind":"function_item","line_end":55,"line_start":28,"name":"compile","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/build.rs","snippet":"fn compile(\n source: &Path,\n object: &Path,\n sdk: &str,\n deployment_target: &str,\n) -> Result<(), Box> {\n let minimum_version = format!(\"-mmacosx-version-min={deployment_target}\");\n xcrun([\n OsStr::new(\"--sdk\"),\n OsStr::new(\"macosx\"),\n OsStr::new(\"clang\"),\n OsStr::new(\"-arch\"),\n OsStr::new(\"arm64\"),\n OsStr::new(\"-isysroot\"),\n "},{"distance":0.4892887771129608,"kind":"function_item","line_end":4331,"line_start":4299,"name":"with_signal_alt_stack_actually_registers_one","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/lib.rs","snippet":"fn with_signal_alt_stack_actually_registers_one() {\n // `with_signal_alt_stack` does a real host `mmap`/`munmap` of its own,\n // which can race `allocate_jit_pages_hint_honors_the_suggested_address`'s\n // address-space probing under the parallel test harness; serialize\n // against it the same way guest-entry tests already do.\n let _serial = guest::tests::TEST_SERIAL\n .lock()\n"},{"distance":0.501526415348053,"kind":"function_definition","line_end":329,"line_start":304,"name":"","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/hvf_sdk.c","snippet":"hv_return_t litebox_hvf_vcpu_verify_feature_regs(uint64_t identifier,\n const uint64_t *expected,\n size_t count,\n size_t *mismatch_index,\n uint64_t *actual_value) {\n if (count != litebox_hvf_feature_reg_cou"},{"distance":0.5098949074745178,"kind":"function_item","line_end":906,"line_start":704,"name":"hvf_smoke_probe","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/hvf.rs","snippet":"pub fn hvf_smoke_probe() -> Result {\n let _exclusive = HVF_SMOKE_LOCK\n .lock()\n .unwrap_or_else(std::sync::PoisonError::into_inner);\n if sdk::production_vm_is_live() {\n return Err(HvfSmokeError::ProductionVmActive);\n }\n\n // SAFETY: `_SC_PAGESIZE` has no pointer arguments or side effects.\n let host_page_size = unsafe { libc::sysconf"},{"distance":0.5149121880531311,"kind":"impl_item","line_end":1391,"line_start":1377,"name":"","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/hvf_sdk.rs","snippet":"impl Drop for HvfVcpu {\n fn drop(&mut self) {\n if self.live {\n self.live = false;\n let Ok(_operation) = self.vm.begin_cleanup_operation() else {\n self.vm.poison();\n return;\n };\n let result = unsafe { litebox_hvf_vcpu_destroy(self.identifier) };\n if !succeeded(result) {\n self.vm.poison();\n "},{"distance":0.5278261303901672,"kind":"function_item","line_end":4290,"line_start":4282,"name":"current_signal_stack","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/lib.rs","snippet":"fn current_signal_stack() -> libc::stack_t {\n // SAFETY: a null `ss` argument only queries the current stack, it does not\n // install one.\n unsafe {\n let mut oss: libc::stack_t = core::mem::zeroed();\n libc::sigaltstack(core::ptr::null(), &raw mut oss);\n oss\n }\n}"},{"distance":0.5702703595161438,"kind":"function_item","line_end":400,"line_start":398,"name":"last_errno","path":"Users/dylanwong/litebox/litebox_platform_macos_userland/src/hvf_backing.rs","snippet":"fn last_errno() -> i32 {\n std::io::Error::last_os_error().raw_os_error().unwrap_or(-1)\n}"}]},"dispatch_id":"1788302978144-500-7eabeb8db433676","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"76cde332094bd8ed","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-692381788302938563.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-692381788302938563.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-704331788372957347.json b/.agentplug/plugin-dispatch/out/gm-codesearch-704331788372957347.json new file mode 100644 index 0000000000..7673a58ff2 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-704331788372957347.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-30003d83-48919216-0","score":55.641735391374425,"symbol":{"kind":"section","line_end":130,"line_start":1,"name":"","path":"docs/macos.md"},"text":"docs/macos.md:1:130 \n# LiteBox on macOS (Apple Silicon)\n\nLiteBox runs guest instructions natively; only the *system* interface is\nvirtualized. On an Apple Silicon Mac that means the only sensible configuration\nis an **AArch64 Linux guest on an AArch64 macOS host** — no emulation anywhere.\nThere is deliberately no x86-64 macOS platform: an x86-64 guest would need\ninstruction emulation, which is the thing this design exists to avoid.\n\nThis document covers what works today, what the host imposes, and what is left\nbefore a guest can actually execute.\n\n## What is in the tree\n\n| Piece | State |\n| --- | --- |\n| `litebox_platform_macos_userland` | The macOS \"South\" platform: memory, locking, time, signals, timers, threads, TLS, randomness, derived keys, stdio, `utun` networking, fault recovery. |\n| `litebox` core | Builds for `aarch64-apple-darwin`, including the Mach-O exception table. |\n| `litebox_shim_linux` | The Linux \"North\" shim, ported to AArch64: signal frames, syscall entry/return, thread-pointer handling, `stat`/`uname` ABI, exception decoding. |\n| `litebox_syscall_rewriter` | Already had AArch64 support (`arm64.rs`) for rewriting `SVC` and `TPIDR_EL0` accesses in Linux ELF images. |\n| `litebox_packager` | OCI mode now pulls the image matching the host architecture, and builds on Apple Silicon. |\n| Guest entry | **Implemented** (context switch + syscall dispatch), tested on real hardware. A syscall-only guest runs end to end; the guest thread-pointer plumbing and non-syscall event paths remain. See [Remaining work](#remaining-work). |\n\n## Building\n\n```sh\nrustup target add aarch64-apple-darwin\ncargo build --workspace --exclude litebox_runner_lvbs --exclude litebox_runner_snp\n```\n\n`litebox_runner_lvbs` and `litebox_runner_snp` are freestanding images for\ncustom targets and are not built for a hosted target on any platform.\n\nCI covers this in the `Build and Test macOS (Apple Silicon)` job, which also\ncompiles and runs `litebox_platform_macos_userland/tests/darwin_abi_probe.c`\nagainst the runner's real SDK headers -- the only check in this repo that\nverifies the crate's hand-written Mach/BSD struct layouts (used by the fault\nhandler to read `ucontext_t::uc_mcontext`) against an actual Darwin toolchain,\nsince nothing else in a Linux-hosted development loop can.\n\n### Hypervisor.framework release boundary\n\nThe stock-code backend requires Apple Silicon, macOS 26 or newer, and a macOS\n26 SDK or newer. The combined runner retains its macOS 11 deployment target so\nthe native backend still launches on older hosts; post-macOS-11 HVF imports are\nweak and the production boundary checks `__builtin_available(macOS 26.0, *)`\nbefore making any such call. `litebox_platform_macos_userland/build.rs`\ncompiles the narrow C boundary and linked EL1 monitor against the active SDK\nheaders, so Apple enum, object, and structure layouts do not get copied into\nRust constants. A release runner is built and signed with the checked-in\nentitlement manifest as follows:\n\n```sh\ncargo build --release --locked -p litebox_runner_linux_on_macos_userland\ncodesign --force --options runtime --sign - \\\n --entitlements litebox_runner_linux_on_macos_userland/entitlements.plist \\\n target/release/litebox_runner_linux_on_macos_userland\ncodesign --display --entitlements - \\\n target/release/litebox_runner_linux_on_macos_userland\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-boundary\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-memory\n```\n\nThe boundary diagnostic validates the active-SDK configuration, maps and\nunmaps the monitor, and creates, verifies, and destroys a vCPU. The compact\nmemory diagnostic does not run guest instructions. It proves that IPA zero is\nreserved for the monitor, GVA=HVA mappings receive compact non-identity IPAs,\n16 KiB stage-one roots software-walk to the same pages, stage-two permissions\ncan move from RW to RX without RWX, rejected overlap transactions publish no\nnew root generation, and released IPA is reused without poisoning the VM. The\nlinked 16 KiB monitor remains the source of truth, but the VM owns an aligned\nallocator-backed copy: Hypervisor.framework rejects file-backed Mach-O\n`__TEXT` pages as stage-two backing.\n\n`com.apple.security.hypervisor=true` is mandatory for\nHypervisor.framework VM creation. `com.apple.security.cs.allow-jit=true` remains\nrequired when the native rewritten backend is packaged with Hardened Runtime.\nThe legacy `com.apple.vm.hypervisor` entitlement belongs only to deployment\ntargets through macOS 10.15 and is deliberately absent. Notarization and\nHardened Runtime do not grant Hypervisor.framework authorization; the final\nexecutable still needs the hypervisor entitlement.\n\n## What the host imposes\n\n### 16 KiB pages\n\nApple Silicon's page size is 16 KiB. Every fixed mapping and every protection\nchange must be aligned to it, so `litebox::mm::linux::PAGE_SIZE` is 16384 on\nthis target rather than 4096. The guest sees the same value through `AT_PAGESZ`,\nwhich is exactly how a Linux kernel configured for 16 KiB or 64 KiB pages\nreports itself.\n\nAArch64 ELF images are conventionally linked with a 64 KiB maximum page size, so\ntheir `PT_LOAD` segments stay aligned either way. An image built with 4 KiB\nsegment alignment will not map cleanly.\n\n### The first 4 GiB is unusable\n\nAn arm64 Mach-O process reserves `[0, 4 GiB)` as the `__PAGEZERO` segment:\nunmapped and impossible to map over. `TASK_ADDR_MIN` is therefore `0x1_0000_0000`.\n\nThe practical consequence is that guest images must be position-independent, or\nlinked above 4 GiB. An `ET_EXEC` binary linked at the customary `0x400000`\ncannot be loaded at its preferred address on this host.\n\n### W^X, `MAP_JIT`, and code signing\n\nmacOS refuses to make anonymous memory executable through the ordinary path, and\nrefuses to add `PROT_EXEC` to anything that was ever writable. The supported\nescape hatch is `MAP_JIT`, which the platform passes whenever a mapping requests\n`EXEC`. Using it has two consequences:\n\n1. **The JIT entitlement is only load-bearing under the Hardened Runtime.**\n Per Apple's own documentation, `com.apple.security.cs.allow-jit` is required\n only when a binary has the Hardened Runtime enabled (`codesign --options\n runtime`, which in turn is what notarization requires); without it,\n `MAP_JIT` works with or without the entitlement present. The command below\n ad-hoc-signs with the entitlement anyway -- it costs nothing and future-proofs\n a later `--options runtime`, notarized build -- but for local development\n outside Gatekeeper, neither the entitlement nor notarization is actually\n required for `MAP_JIT` itself to work. Create an entitlements file:\n\n ```xml\n \n \n \n \n com.apple.security.cs.allow-jit"},{"key":"ci-ea2224bb-9e5d768c-1","score":23.40853588490289,"symbol":{"kind":"struct_item","line_end":44,"line_start":21,"name":"CliArgs","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:21:44 CliArgs\npub struct CliArgs {\n /// The program and arguments passed to it (e.g., `/bin/ls --color`).\n ///\n /// The program path refers to a path inside the tar archive provided via\n /// `--initial-files`. All binaries must be pre-rewritten with the syscall\n /// rewriter.\n #[arg(required = true, trailing_var_arg = true, value_hint = clap::ValueHint::CommandWithArguments)]\n pub program_and_arguments: Vec,\n /// Environment variables passed to the program (`K=V` pairs; can be invoked multiple times)\n #[arg(long = \"env\")]\n pub environment_variables: Vec,\n /// Forward the existing environment variables\n #[arg(long = \"forward-env\")]\n pub forward_environment_variables: bool,\n /// Allow using unstable options\n #[arg(short = 'Z', long = \"unstable\")]\n pub unstable: bool,\n /// Tar archive containing the program and its shared libraries.\n ///\n /// All ELF binaries should be pre-rewritten with the syscall rewriter\n /// (e.g., via `litebox-packager`).\n #[arg(long = \"initial-files\", value_name = \"PATH_TO_TAR\", value_hint = clap::ValueHint::FilePath)]\n pub initial_files: PathBuf,\n}"},{"key":"ci-39f1a0ff-d0115eff-1","score":22.30582800341835,"symbol":{"kind":"function_item","line_end":12,"line_start":8,"name":"main","path":"litebox_runner_linux_on_macos_userland/src/main.rs"},"text":"litebox_runner_linux_on_macos_userland/src/main.rs:8:12 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_linux_on_macos_userland::CliArgs;\n litebox_runner_linux_on_macos_userland::run(CliArgs::parse())\n}"},{"key":"ci-b30dc81f-d5084132-0","score":19.360635703938684,"symbol":{"kind":"function_item","line_end":112,"line_start":57,"name":"main","path":"litebox_platform_macos_userland/build.rs"},"text":"litebox_platform_macos_userland/build.rs:57:112 main\nfn main() -> Result<(), Box> {\n println!(\"cargo:rerun-if-changed=src/hvf_sdk.c\");\n println!(\"cargo:rerun-if-changed=src/hvf_monitor.S\");\n for variable in [\n \"MACOSX_DEPLOYMENT_TARGET\",\n \"DEVELOPER_DIR\",\n \"SDKROOT\",\n \"TOOLCHAINS\",\n ] {\n println!(\"cargo:rerun-if-env-changed={variable}\");\n }\n\n if env::var(\"CARGO_CFG_TARGET_OS\")? != \"macos\"\n || env::var(\"CARGO_CFG_TARGET_ARCH\")? != \"aarch64\"\n {\n return Ok(());\n }\n\n let sdk = String::from_utf8(xcrun([\"--sdk\", \"macosx\", \"--show-sdk-path\"])?.stdout)?;\n let sdk = sdk.trim();\n let deployment_target =\n env::var(\"MACOSX_DEPLOYMENT_TARGET\").unwrap_or_else(|_| \"11.0\".to_owned());\n let out = PathBuf::from(env::var_os(\"OUT_DIR\").ok_or(\"OUT_DIR is not set\")?);\n let manifest =\n PathBuf::from(env::var_os(\"CARGO_MANIFEST_DIR\").ok_or(\"CARGO_MANIFEST_DIR is not set\")?);\n let sdk_object = out.join(\"hvf_sdk.o\");\n let monitor_object = out.join(\"hvf_monitor.o\");\n let archive = out.join(\"liblitebox_hvf_sdk.a\");\n\n compile(\n &manifest.join(\"src/hvf_sdk.c\"),\n &sdk_object,\n sdk,\n &deployment_target,\n )?;\n compile(\n &manifest.join(\"src/hvf_monitor.S\"),\n &monitor_object,\n sdk,\n &deployment_target,\n )?;\n xcrun([\n OsStr::new(\"--sdk\"),\n OsStr::new(\"macosx\"),\n OsStr::new(\"ar\"),\n OsStr::new(\"rcs\"),\n archive.as_os_str(),\n sdk_object.as_os_str(),\n monitor_object.as_os_str(),\n ])?;\n\n println!(\"cargo:rustc-link-search=native={}\", out.display());\n println!(\"cargo:rustc-link-lib=static=litebox_hvf_sdk\");\n println!(\"cargo:rustc-link-lib=framework=Hypervisor\");\n Ok(())\n}"},{"key":"ci-39f1a0ff-d0115eff-0","score":18.89500984338972,"symbol":{"kind":"function_item","line_end":18,"line_start":15,"name":"main","path":"litebox_runner_linux_on_macos_userland/src/main.rs"},"text":"litebox_runner_linux_on_macos_userland/src/main.rs:15:18 main\nfn main() {\n eprintln!(\"This program is only supported on macOS on Apple Silicon\");\n std::process::exit(1);\n}"},{"key":"ci-adbeac4c-1cb07d98-0","score":18.28824548865797,"symbol":{"kind":"section","line_end":43,"line_start":4,"name":"","path":"litebox_packager/examples/xfce/README.md"},"text":"litebox_packager/examples/xfce/README.md:4:43 \n# XFCE desktop guest\n\nThis recipe builds an Alpine 3.24 XFCE desktop for the macOS userland runner\nand serves its 1024×768 framebuffer and input through the built-in browser\nviewer.\n\nApple's kernel clears AArch64 `x18` whenever it returns to userspace. Stock\nAlpine allocates that register, which silently corrupts ld.so, Xorg, GTK, and\nXFCE hot loops under native guest execution. The build therefore first runs\n`../../scripts/build-x18-desktop-repo.sh`; that rebuilds the desktop's loaded\ncode closure with x18 reserved and refuses to publish any runtime ELF that\nstill disassembles to an `x18`/`w18` operand.\n\n```sh\n./litebox_packager/examples/xfce/build-xfce-image.sh /tmp/litebox-xfce.tar\n\ncargo run --release -p litebox_runner_linux_on_macos_userland -- \\\n --unstable --guest-root \\\n --initial-files /tmp/litebox-xfce.tar \\\n --vnc-web 6080 -- \\\n /usr/bin/start-desktop.sh\n```\n\nOpen . The canvas accepts pointer, wheel, and keyboard\ninput.\n\nThe first package build is intentionally substantial and resumable. Successful\naports origins remain in the retained `litebox-x18-repo-build` container;\nrerunning retries only unfinished origins. Override paths and names with:\n\n- `LITEBOX_X18_DESKTOP_REPO`\n- `LITEBOX_ALPINE_BRANCH`\n- `LITEBOX_XFCE_IMAGE_TAG`\n\nThe recipe disables GLX and fbdev ShadowFB because those unimplemented paths\ndo not update litebox's browser framebuffer. It also appends the synthetic\n`/sys/class/graphics/fb0/device/subsystem` link required by Xorg's fbdevhw\nprobe; the packager otherwise intentionally omits `/sys` from OCI rootfs\nimages."},{"key":"ci-b8807bce-1130b664-1","score":16.247043755557762,"symbol":{"kind":"function_item","line_end":9,"line_start":5,"name":"main","path":"litebox_runner_optee_on_linux_userland/src/main.rs"},"text":"litebox_runner_optee_on_linux_userland/src/main.rs:5:9 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_optee_on_linux_userland::CliArgs;\n litebox_runner_optee_on_linux_userland::run(CliArgs::parse())\n}"},{"key":"ci-f1d76c9a-279b43d3-0","score":14.582338012773024,"symbol":{"kind":"section","line_end":126,"line_start":1,"name":"","path":"docs/roadmap.md"},"text":"docs/roadmap.md:1:126 \n# Roadmap: known gaps and follow-up work\n\nThis is a working list of gaps found while porting LiteBox to macOS/Apple\nSilicon and auditing the rest of the tree for related issues. Each entry\nbelow was deliberately **not** implemented in that pass, because doing it\ncorrectly needs either real hardware/kernel verification this repo's CI\ncannot provide from a Linux-hosted sandbox, or a genuine design decision\nrather than a mechanical fix. Implementing any of these without that\nverification risks the exact kind of half-finished, silently-wrong change\nthis list exists to avoid.\n\nItems are grouped by how much verification they need before landing, not by\nsubsystem.\n\n## Resolved on real hardware this pass\n\n* **The `TPIDR_EL0` anchor question is answered.** Measured on an Apple M3\n Pro (macOS 26.3.1): `TPIDR_EL0` does not survive a context switch (XNU\n overwrites it with its own value, not merely leaves it stale) and cannot\n anchor the guest thread pointer. `TPIDRRO_EL0` is stable across a reschedule\n and distinct per thread, matching Apple's documented pthread-self-pointer\n use. See [`docs/macos.md`](./macos.md#remaining-work) for the full\n measurement and the resulting design (a reserved pthread TSD slot read via\n a `TPIDRRO_EL0`-relative direct-TSD sequence, mirroring libSystem's own\n fast accessors). What's left is implementation, not research:\n\n## Needs real Apple Silicon hardware (implementation, not open questions)\n\n* **`Host::MacOs`'s anchor register is right; the fixed TSD slot number is\n not, and the whole \"bake one number in at packaging time\" approach has a\n deeper problem than the number being wrong.** Gates anchor on `TPIDRRO_EL0`\n (real, tested) and address the guest thread pointer at pthread TSD slot\n `MACOS_GUEST_TPIDR_TSD_SLOT` (hardcoded to 256, sourced from\n apple-oss-distributions/libpthread as \"the first dynamic\n `pthread_key_create` key\") -- a LiteBox-owned slot rather than a raw offset\n into Apple's own pthread structure, so it no longer risks corrupting\n libpthread state the way the earlier design did.\n \n `litebox_platform_macos_userland::new` calls `pthread_key_create` at startup\n and records the key. It originally *asserted* the key equalled the baked slot\n -- **which always fails on real hardware**, making the platform\n unconstructable -- so that was softened (this pass) to a loud warning that\n leaves construction working (regression test\n `reserving_the_tsd_slot_does_not_panic_on_mismatch`); a syscall-only guest is\n unaffected, a `TPIDR_EL0`-using guest is unsupported until the real fix below.\n Measured on this M3 Pro (macOS 26.3.1): a minimal Rust binary's first\n `pthread_key_create` call returns 259; a plain C `main`'s first call\n returns 258. Neither is 256. Something in libSystem's startup path claims a\n few dynamic keys before user code runs, undocumented and not guaranteed\n stable across macOS versions or across binaries with different statically\n linked dependencies (each with their own static initializers, potentially\n claiming more). This means the actual slot a real runner binary gets is a\n property of *that specific binary's* full startup sequence -- not knowable\n by the rewriter, which runs separately, earlier, packaging the guest image\n with no visibility into what the eventual runner process will look like.\n \n The failure mode is safe (a loud warning at `MacOsUserland::new()`, not\n silent corruption), so this does not need the same \"keep it out of anything\n that runs for real\" mitigation the previous corruption bug did.\n\n **The rewriter half of the fix has landed; the loader half has not.**\n `Host::MacOs` gates no longer bake the slot number in. They read a byte offset\n from the trampoline header slot `HEADER_GUEST_TP_OFFSET_MACOS` and address\n `[TPIDRRO_EL0 + offset]`, which is what makes the number a load-time rather\n than a packaging-time decision. `Host::Linux` is untouched and still bakes its\n immediate, since its offset is genuine compile-time ABI; the two are now\n distinguished explicitly by `GuestTpAddressing`.\n\n The slot holds an *offset*, never a thread-pointer value. The loader maps the\n trampoline writable, fills the header, then flips it to read+execute\n (`litebox_common_linux`'s `load_trampoline`), so nothing can rewrite that word\n once a guest is running -- and one word could not serve two threads anyway. The\n per-thread part comes from `TPIDRRO_EL0`, which is already per-thread, so this\n design stays compatible with per-thread guest TPs rather than foreclosing them.\n\n **The loader half has landed too.** `SystemInfoProvider::get_guest_tp_slot_offset`\n reports the offset a host decides at run time (`None` on every host that bakes\n it in); `litebox_platform_macos_userland` answers with\n `guest_tp_slot_byte_offset()`, the reserved `pthread_key_create` key scaled by\n 8. `litebox_common_linux`'s `load_trampoline` publishes it into the header slot\n in the same window it already writes the syscall entry point -- while the\n trampoline is still writable and before the flip to read+execute. That window\n is the only correct place for it. `litebox_common_linux` cannot depend on the\n rewriter, so `litebox_shim_linux` holds the two slot constants together with a\n `const` assertion rather than a comment.\n\n What remains for a `TPIDR_EL0`-using guest: `pthread_setspecific` of each guest\n thread's pointer into the reserved key, and a macOS runner to exercise any of\n it -- none wires `MacOsUserland` into `litebox_shim_linux` today, so this whole\n path is still unexercised end to end on hardware.\n* **The platform's *own* per-thread context-switch bookkeeping** —\n **RESOLVED on real hardware.** A separate problem from the rewriter's guest\n slot above, and the thing that limited this platform to one guest thread at a\n time. `litebox_platform_linux_userland`'s x86_64\n `run_thread_arch`/`switch_to_guest`/`syscall_callback` (the closest thing to\n a template) does not only virtualize the *guest's* thread pointer -- it also\n stashes its own bookkeeping (`host_sp`, `host_bp`, `guest_context_top`,\n `in_guest`) in `fs:`-relative TLS slots, because by the time\n `syscall_callback` runs, every general-purpose register holds live guest\n state and there is nothing else durable to read \"where was the host stack\"\n from. That mechanism is entirely x86_64-ELF-specific (raw `@tpoff`-relative\n local-exec TLS addressing, resolved to a link-time-fixed offset with no\n function call and no runtime-determined value at all) and has no Mach-O\n equivalent to copy directly.\n\n **The building block, confirmed on real Apple M3 Pro hardware:** a raw\n `mrs tpidrro_el0` (masked, `& ~7`, matching libSystem's own\n `_os_tsd_get_base`) plus a `[base, #(key * 8)]` read/write reaches the *same*\n per-thread storage `pthread_getspecific`/`pthread_setspecific` do, for a\n **second**, independently `pthread_key_create`-reserved dynamic TSD key (not\n just the one already relied on for the guest's own `TPIDR_EL0` shadow) -- in\n both directions, across the full `usize` range, and disjointly across two\n genuinely concurrent OS threads. Measured again this pass: the first dynamic\n key a Rust binary gets is 259 and the pool is exhausted at key 767.\n\n **The blocker, and how it was actually solved.** Of the six naked functions\n in `litebox_platform_macos_userland::guest`, five are reached with registers\n to spare (three of them by a signal handler's `pc` redirect, so *every*\n register is free) and can simply do a two-register lookup: one register for\n the run-time-determined TSD byte offset, one for the `TPIDRRO_EL0` value. The\n syscall callback cannot: the rewriter's `SVC` gate leaves exactly **one**\n register free (`X16`), because `X17` still holds the guest's real value,\n which real Linux AArch64 preserves across a syscall and which this file's own\n fidelity philosophy (`preserves_registers_across_capture_and_resume`) commits\n to capturing faithfully. One register is enough for\n `mrs`/`and`/`ldr [x16, #imm]` **only if the immediate is a compile-time"},{"key":"ci-667c1eaa-978941ff-0","score":14.376137420393862,"symbol":{"kind":"function_item","line_end":749,"line_start":653,"name":"run_rewritten_iperf3","path":"dev_bench/src/main.rs"},"text":"dev_bench/src/main.rs:653:749 run_rewritten_iperf3\nfn run_rewritten_iperf3(ctx: BenchCtx<'_>) -> Result<()> {\n let BenchCtx {\n sh,\n cli_args: _,\n project_root,\n is_init,\n lock_tracing,\n } = ctx;\n let tar_file = sh.current_dir().join(\"iperf3_rootfs.tar\");\n let release_mode = true;\n if is_init {\n rewriter_iperf3(ctx.with_init(true))?;\n rewriter_iperf3(ctx.with_init(false))?;\n\n let tar_base_dir = sh.current_dir().join(\"iperf3_tar_base\");\n sh.create_dir(&tar_base_dir)?;\n let libs = find_dependencies(sh, \"iperf3\")?;\n for lib in libs {\n let dest_path = tar_base_dir\n .join(lib.strip_prefix(\"/\").unwrap_or_else(|_| {\n panic!(\"Library path '{}' is not absolute\", lib.display())\n }));\n if let Some(parent) = dest_path.parent() {\n sh.create_dir(parent)?;\n }\n cmd!(\n sh,\n \"{project_root}/target/release/litebox_syscall_rewriter {lib} -o {dest_path}\"\n )\n .run()?;\n }\n\n sh.remove_path(&tar_file)?;\n cmd!(sh, \"tar --format=ustar -C {tar_base_dir} -cvf {tar_file} .\").run()?;\n let release = release_mode.then_some(\"--release\");\n let features: &[&str] = if lock_tracing {\n &[\"--features\", \"lock_tracing\"]\n } else {\n &[]\n };\n cmd!(\n sh,\n \"cargo build -p litebox_runner_linux_userland {release...} {features...}\"\n )\n .run()?;\n } else {\n let mode = if release_mode { \"release\" } else { \"debug\" };\n let iperf3_host = locate_command(sh, \"iperf3\")?;\n let runner = format!(\n \"{}/target/{mode}/litebox_runner_linux_userland\",\n project_root.display()\n );\n let iperf3_rewritten = sh.current_dir().join(\"iperf3_rewritten\");\n\n // Spawn the sandboxed iperf3 server in a background thread so we can run the client\n // from the host side. The server uses `-1` to exit after handling one client.\n let server_handle = std::thread::spawn(move || -> Result<()> {\n let sh = xshell::Shell::new()?;\n cmd!(\n sh,\n \"{runner} --unstable --env LD_LIBRARY_PATH=/lib64:/lib32:/lib --env HOME=/ --tun-device-name tun99 --initial-files {tar_file} {iperf3_rewritten} -s -1 -B 10.0.0.2\"\n ).run()?;\n Ok(())\n });\n\n // Retry the client connection until the server is ready, using a short\n // connect-timeout so we don't waste time sleeping for a fixed duration.\n debug!(\"Connecting iperf3 client to sandboxed server\");\n let client_sh = xshell::Shell::new()?;\n let max_attempts = 50;\n for attempt in 1..=max_attempts {\n let result = cmd!(\n client_sh,\n \"{iperf3_host} -c 10.0.0.2 --bytes 1G --connect-timeout 50\"\n )\n .quiet()\n .ignore_stdout()\n .ignore_stderr()\n .run();\n if result.is_ok() {\n break;\n }\n if attempt == max_attempts {\n return Err(anyhow!(\n \"iperf3 client failed to connect after {max_attempts} attempts\"\n ));\n }\n debug!(attempt, \"iperf3 client connection failed, retrying\");\n std::thread::sleep(Duration::from_millis(100));\n }\n\n server_handle\n .join()\n .map_err(|e| anyhow!(\"iperf3 server thread panicked: {e:?}\"))??;\n }\n Ok(())\n}"},{"key":"ci-99368d6-ad38df33-0","score":14.302668521684474,"symbol":{"kind":"section","line_end":47,"line_start":1,"name":"","path":"README.md"},"text":"README.md:1:47 \n# LiteBox\n\n> A security-focused library OS\n\n> [!NOTE] \n> This project is currently actively evolving and improving. While we are\n> working toward a stable release, some APIs and interfaces may change as the\n> design continues to mature. You are welcome to explore and experiment, but if\n> you need long-term stability, it may be best to wait for a stable release, or\n> be prepared to adapt to updates along the way.\n\nLiteBox is a sandboxing library OS that drastically cuts down the interface to the host, thereby reducing attack surface. It focuses on easy interop of various \"North\" shims and \"South\" platforms. LiteBox is designed for usage in both kernel and non-kernel scenarios.\n\nLiteBox exposes a Rust-y [`nix`](https://docs.rs/nix)/[`rustix`](https://docs.rs/rustix)-inspired \"North\" interface when it is provided a `Platform` interface at its \"South\". These interfaces allow for a wide variety of use-cases, easily allowing for connection between any of the North--South pairs.\n\nExample use cases include:\n- Running unmodified Linux programs on Windows\n- Running unmodified Linux programs on macOS (Apple Silicon) -- see [docs/macos.md](./docs/macos.md)\n- Sandboxing Linux applications on Linux\n- Run programs on top of SEV SNP\n- Running OP-TEE programs on Linux\n- Running on LVBS\n\n![LiteBox and related projects](./.figures/litebox.svg)\n\n## Contributing\n\nSee the following files for details:\n\n- [CONTRIBUTING.md](./CONTRIBUTING.md)\n- [CODE_OF_CONDUCT.md](./CODE_OF_CONDUCT.md)\n- [SECURITY.md](./SECURITY.md)\n- [SUPPORT.md](./SUPPORT.md)\n- [docs/roadmap.md](./docs/roadmap.md) for known gaps and follow-up work\n\n## License\n\nMIT License. See [./LICENSE](./LICENSE) for details.\n\n## Trademarks\n\nThis project may contain trademarks or logos for projects, products, or services. Authorized use of Microsoft \ntrademarks or logos is subject to and must follow \n[Microsoft's Trademark & Brand Guidelines](https://www.microsoft.com/en-us/legal/intellectualproperty/trademarks/usage/general).\nUse of Microsoft trademarks or logos in modified versions of this project must not cause confusion or imply Microsoft sponsorship.\nAny use of third-party trademarks or logos are subject to those third-party's policies."}],"commits":[{"hash":"de65eb0c45b76e017976a322bf14be458c4239d0","message":"fix(ci): clear remaining clippy pedantic errors on macOS and rewriter","score":7.0},{"hash":"624f0ceedf841b147c5e972104d7cfed355ccc90","message":"feat(net): guest web access without root via an in-stack HTTP proxy","score":6.0},{"hash":"b97641713732df7cb18e20ebbd80e675e777fba6","message":"feat(rfb): runner-side VNC server presenting the guest /dev/fb0 framebuffer","score":6.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":4.0},{"hash":"4d8bcd25f1a64dcd91c6db94dff2dd3cddb9c0ec","message":"fix(runner): propagate CString NUL errors and name infallible /tmp setup","score":4.0},{"hash":"6e1021490837e7c36b235b56df9c77ce25a26cd9","message":"fix(runner): name the port and likely cause in viewer bind errors","score":4.0},{"hash":"78062e8e46abede70b010fdb00c5811fb8ea203b","message":"test(macos-userland): add live X16 sentinel probe, refuting the roadmap's X16-corruption hypothesis","score":3.0},{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":3.0},{"hash":"e44b8068ac22eb0e2ee862dbcfff6a45d492da4c","message":"fix(doc): restore run()'s doc block and document new_with_options panics","score":3.0},{"hash":"3f7d8ae6f7321e89889f02aa3efa373d673e02e3","message":"feat(vnc): bridge RFB keyboard into guest stdin","score":3.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788373381218-500-6050c91e2835f1be","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"ff13920d5d49558a","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-704331788372957347.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-704331788372957347.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-71721788295910360.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-71721788295910360.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-73321788199217733.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-73321788199217733.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-75281788215650991.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-75281788215650991.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-76931788295918957.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-76931788295918957.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-83641788215663090.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-83641788215663090.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-87421788215671661.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-87421788215671661.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-882091788375033189.json b/.agentplug/plugin-dispatch/out/gm-codesearch-882091788375033189.json new file mode 100644 index 0000000000..919a066c7e --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-882091788375033189.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-a132790f-6d766ade-0","score":44.09914782498046,"symbol":{"kind":"impl_item","line_end":97,"line_start":69,"name":"","path":"litebox_shim_linux/src/loader/auxv.rs"},"text":"litebox_shim_linux/src/loader/auxv.rs:69:97 \nimpl Task {\n /// Initialize the auxiliary vector with the credentials that will be published with this image.\n pub fn init_auxv(&self, credentials: &Credentials, secure: bool) -> AuxVec {\n let mut aux = AuxVec::new();\n\n aux.insert(AuxKey::AT_UID, credentials.uid as usize);\n aux.insert(AuxKey::AT_EUID, credentials.euid as usize);\n aux.insert(AuxKey::AT_GID, credentials.gid as usize);\n aux.insert(AuxKey::AT_EGID, credentials.egid as usize);\n aux.insert(AuxKey::AT_SECURE, usize::from(secure));\n\n if let Some(vdso_base) = self.global.platform.get_vdso_address() {\n aux.insert(AuxKey::AT_SYSINFO_EHDR, vdso_base);\n }\n\n let (hwcap, hwcap2) = self.global.platform.get_hwcap();\n // `AT_HWCAP`/`AT_HWCAP2` are each a 32-bit-wide Linux kernel ABI concept (the value a\n // 32-bit ARM `getauxval` caller would also see); every bit `SystemInfoProvider::get_hwcap`\n // implementations set is below bit 32, so this never actually truncates even on a\n // 32-bit-`usize` target.\n #[allow(clippy::cast_possible_truncation)]\n {\n aux.insert(AuxKey::AT_HWCAP, hwcap as usize);\n aux.insert(AuxKey::AT_HWCAP2, hwcap2 as usize);\n }\n\n aux\n }\n}"},{"key":"ci-a132790f-6d766ade-1","score":36.83411498815168,"symbol":{"kind":"function_item","line_end":96,"line_start":71,"name":"init_auxv","path":"litebox_shim_linux/src/loader/auxv.rs"},"text":"litebox_shim_linux/src/loader/auxv.rs:71:96 init_auxv\n pub fn init_auxv(&self, credentials: &Credentials, secure: bool) -> AuxVec {\n let mut aux = AuxVec::new();\n\n aux.insert(AuxKey::AT_UID, credentials.uid as usize);\n aux.insert(AuxKey::AT_EUID, credentials.euid as usize);\n aux.insert(AuxKey::AT_GID, credentials.gid as usize);\n aux.insert(AuxKey::AT_EGID, credentials.egid as usize);\n aux.insert(AuxKey::AT_SECURE, usize::from(secure));\n\n if let Some(vdso_base) = self.global.platform.get_vdso_address() {\n aux.insert(AuxKey::AT_SYSINFO_EHDR, vdso_base);\n }\n\n let (hwcap, hwcap2) = self.global.platform.get_hwcap();\n // `AT_HWCAP`/`AT_HWCAP2` are each a 32-bit-wide Linux kernel ABI concept (the value a\n // 32-bit ARM `getauxval` caller would also see); every bit `SystemInfoProvider::get_hwcap`\n // implementations set is below bit 32, so this never actually truncates even on a\n // 32-bit-`usize` target.\n #[allow(clippy::cast_possible_truncation)]\n {\n aux.insert(AuxKey::AT_HWCAP, hwcap as usize);\n aux.insert(AuxKey::AT_HWCAP2, hwcap2 as usize);\n }\n\n aux\n }"},{"key":"ci-4f57d83a-fc7671f5-0","score":12.50007270691777,"symbol":{"kind":"impl_item","line_end":215,"line_start":64,"name":"","path":"litebox_shim_linux/src/loader/stack.rs"},"text":"litebox_shim_linux/src/loader/stack.rs:64:215 \nimpl UserStack {\n /// Stack alignment required by libc ABI\n const STACK_ALIGNMENT: usize = 16;\n\n /// Create a new stack for the user process.\n ///\n /// `stack_top` and `len` must be aligned to [`Self::STACK_ALIGNMENT`]\n pub(super) fn new(stack_top: UserPtrMut, len: usize) -> Option {\n if !stack_top.as_usize().is_multiple_of(Self::STACK_ALIGNMENT) {\n return None;\n }\n if !len.is_multiple_of(Self::STACK_ALIGNMENT) {\n return None;\n }\n Some(Self {\n stack_top,\n len,\n pos: len,\n _platform: PhantomData,\n })\n }\n\n /// Get the current stack pointer.\n pub(super) fn get_cur_stack_top(&self) -> usize {\n self.stack_top.as_usize() + self.pos\n }\n\n /// Push `bytes` to the stack.\n ///\n /// Returns `None` if the stack has insufficient space.\n fn push_bytes(&mut self, bytes: &[u8]) -> Option<()> {\n let _end = isize::try_from(self.pos).ok()?;\n self.pos = self.pos.checked_sub(bytes.len())?;\n self.stack_top\n .copy_from_slice::(self.pos, bytes)?;\n Some(())\n }\n\n /// Push a value to the stack.\n ///\n /// Returns `None` if the stack has insufficient space.\n fn push_usize(&mut self, val: usize) -> Option<()> {\n self.push_bytes(&val.to_le_bytes())\n }\n\n /// Push a string with a null terminator to the stack.\n ///\n /// Returns `None` if the stack has insufficient space.\n fn push_cstring(&mut self, val: &CString) -> Option<()> {\n let bytes = val.as_bytes_with_nul();\n self.push_bytes(bytes)\n }\n\n /// Push a vector of strings with null terminators to the stack.\n ///\n /// Returns the offsets of the strings in the stack.\n /// Returns `None` if the stack has insufficient space.\n fn push_cstrings(&mut self, vals: &[CString]) -> Option> {\n // Push in reverse so that -- with the stack growing down -- `vals[0]`\n // lands at the LOWEST address and the whole block is contiguous in\n // increasing address order. That is the exact layout the Linux kernel\n // produces, and the one libuv's `uv_setup_args` relies on: it walks\n // `argv[0]..argv[n]` then `environ[0]..` requiring each string to abut\n // the previous at a higher address, and sizes the process-title buffer\n // from that contiguous span. Pushing forward reversed each block, so the\n // walk broke immediately and libuv computed a garbage `process_title.len`\n // -- and the first `process.title = ...` (which `npm` does at startup)\n // then `memset`s that bogus length and SIGSEGVs.\n let mut ptrs = alloc::vec![0usize; vals.len()];\n for (i, val) in vals.iter().enumerate().rev() {\n self.push_cstring(val)?;\n ptrs[i] = self.pos;\n }\n Some(ptrs)\n }\n\n /// Push a vector of stack pointers to the stack.\n ///\n /// `offsets` are the offsets of the pointers in the stack.\n ///\n /// Returns `None` if the stack has insufficient space.\n fn push_pointers(&mut self, offsets: Vec) -> Option<()> {\n // write end marker\n self.push_usize(0)?;\n let size = offsets.len().checked_mul(size_of::())?;\n self.pos = self.pos.checked_sub(size)?;\n let ptr: UserPtrMut = UserPtrMut::from_usize(self.stack_top.as_usize() + self.pos);\n for (i, p) in offsets.iter().enumerate() {\n let addr: usize = self.stack_top.as_usize() + *p;\n ptr.write_at_offset::(i.reinterpret_as_signed(), addr)?;\n }\n Some(())\n }\n\n /// Push an auxiliary vector to the stack.\n ///\n /// Returns `None` if the stack has insufficient space.\n fn push_aux(&mut self, aux: AuxVec) -> Option<()> {\n // write end marker\n self.push_usize(0)?;\n self.push_usize(AuxKey::AT_NULL as usize)?;\n for (key, val) in aux {\n self.push_usize(val)?;\n self.push_usize(key as usize)?;\n }\n Some(())\n }\n\n /// Initialize the stack for the new process.\n pub(super) fn init(\n &mut self,\n argv: Vec,\n env: Vec,\n mut aux: BTreeMap,\n platform: &impl litebox::platform::CrngProvider,\n ) -> Option<()> {\n // end markers\n self.pos = self.pos.checked_sub(size_of::())?;\n self.stack_top\n .write_at_offset::(isize::try_from(self.pos).ok()?, 0)?;\n\n let envp = self.push_cstrings(&env)?;\n let argvp = self.push_cstrings(&argv)?;\n\n // AT_RANDOM: 16 bytes of real randomness (libc's stack-canary seed).\n let mut random_bytes = [0u8; 16];\n <_ as litebox::platform::CrngProvider>::fill_bytes_crng(platform, &mut random_bytes);\n self.push_bytes(&random_bytes)?;\n aux.insert(AuxKey::AT_RANDOM, self.stack_top.as_usize() + self.pos);\n\n let align_down = |pos: usize, alignment: usize| -> usize {\n debug_assert!(alignment.is_power_of_two());\n pos & !(alignment - 1)\n };\n\n // ensure stack is aligned\n self.pos = align_down(self.pos, size_of::());\n // to ensure the final pos is aligned, we need to add some padding\n let len = (aux.len() + 1) * 2 + envp.len() + 1 + argvp.len() + 1 + /* argc */ 1;\n let size = len * size_of::();\n let final_pos = self.pos.checked_sub(size)?;\n self.pos -= final_pos - align_down(final_pos, Self::STACK_ALIGNMENT);\n\n self.push_aux(aux)?;\n self.push_pointers(envp)?;\n self.push_pointers(argvp)?;\n\n self.push_usize(argv.len())?;\n assert_eq!(self.pos, align_down(self.pos, Self::STACK_ALIGNMENT));\n Some(())\n }\n}"},{"key":"ci-f1d76c9a-279b43d3-0","score":11.560231797407845,"symbol":{"kind":"section","line_end":126,"line_start":1,"name":"","path":"docs/roadmap.md"},"text":"docs/roadmap.md:1:126 \n# Roadmap: known gaps and follow-up work\n\nThis is a working list of gaps found while porting LiteBox to macOS/Apple\nSilicon and auditing the rest of the tree for related issues. Each entry\nbelow was deliberately **not** implemented in that pass, because doing it\ncorrectly needs either real hardware/kernel verification this repo's CI\ncannot provide from a Linux-hosted sandbox, or a genuine design decision\nrather than a mechanical fix. Implementing any of these without that\nverification risks the exact kind of half-finished, silently-wrong change\nthis list exists to avoid.\n\nItems are grouped by how much verification they need before landing, not by\nsubsystem.\n\n## Resolved on real hardware this pass\n\n* **The `TPIDR_EL0` anchor question is answered.** Measured on an Apple M3\n Pro (macOS 26.3.1): `TPIDR_EL0` does not survive a context switch (XNU\n overwrites it with its own value, not merely leaves it stale) and cannot\n anchor the guest thread pointer. `TPIDRRO_EL0` is stable across a reschedule\n and distinct per thread, matching Apple's documented pthread-self-pointer\n use. See [`docs/macos.md`](./macos.md#remaining-work) for the full\n measurement and the resulting design (a reserved pthread TSD slot read via\n a `TPIDRRO_EL0`-relative direct-TSD sequence, mirroring libSystem's own\n fast accessors). What's left is implementation, not research:\n\n## Needs real Apple Silicon hardware (implementation, not open questions)\n\n* **`Host::MacOs`'s anchor register is right; the fixed TSD slot number is\n not, and the whole \"bake one number in at packaging time\" approach has a\n deeper problem than the number being wrong.** Gates anchor on `TPIDRRO_EL0`\n (real, tested) and address the guest thread pointer at pthread TSD slot\n `MACOS_GUEST_TPIDR_TSD_SLOT` (hardcoded to 256, sourced from\n apple-oss-distributions/libpthread as \"the first dynamic\n `pthread_key_create` key\") -- a LiteBox-owned slot rather than a raw offset\n into Apple's own pthread structure, so it no longer risks corrupting\n libpthread state the way the earlier design did.\n \n `litebox_platform_macos_userland::new` calls `pthread_key_create` at startup\n and records the key. It originally *asserted* the key equalled the baked slot\n -- **which always fails on real hardware**, making the platform\n unconstructable -- so that was softened (this pass) to a loud warning that\n leaves construction working (regression test\n `reserving_the_tsd_slot_does_not_panic_on_mismatch`); a syscall-only guest is\n unaffected, a `TPIDR_EL0`-using guest is unsupported until the real fix below.\n Measured on this M3 Pro (macOS 26.3.1): a minimal Rust binary's first\n `pthread_key_create` call returns 259; a plain C `main`'s first call\n returns 258. Neither is 256. Something in libSystem's startup path claims a\n few dynamic keys before user code runs, undocumented and not guaranteed\n stable across macOS versions or across binaries with different statically\n linked dependencies (each with their own static initializers, potentially\n claiming more). This means the actual slot a real runner binary gets is a\n property of *that specific binary's* full startup sequence -- not knowable\n by the rewriter, which runs separately, earlier, packaging the guest image\n with no visibility into what the eventual runner process will look like.\n \n The failure mode is safe (a loud warning at `MacOsUserland::new()`, not\n silent corruption), so this does not need the same \"keep it out of anything\n that runs for real\" mitigation the previous corruption bug did.\n\n **The rewriter half of the fix has landed; the loader half has not.**\n `Host::MacOs` gates no longer bake the slot number in. They read a byte offset\n from the trampoline header slot `HEADER_GUEST_TP_OFFSET_MACOS` and address\n `[TPIDRRO_EL0 + offset]`, which is what makes the number a load-time rather\n than a packaging-time decision. `Host::Linux` is untouched and still bakes its\n immediate, since its offset is genuine compile-time ABI; the two are now\n distinguished explicitly by `GuestTpAddressing`.\n\n The slot holds an *offset*, never a thread-pointer value. The loader maps the\n trampoline writable, fills the header, then flips it to read+execute\n (`litebox_common_linux`'s `load_trampoline`), so nothing can rewrite that word\n once a guest is running -- and one word could not serve two threads anyway. The\n per-thread part comes from `TPIDRRO_EL0`, which is already per-thread, so this\n design stays compatible with per-thread guest TPs rather than foreclosing them.\n\n **The loader half has landed too.** `SystemInfoProvider::get_guest_tp_slot_offset`\n reports the offset a host decides at run time (`None` on every host that bakes\n it in); `litebox_platform_macos_userland` answers with\n `guest_tp_slot_byte_offset()`, the reserved `pthread_key_create` key scaled by\n 8. `litebox_common_linux`'s `load_trampoline` publishes it into the header slot\n in the same window it already writes the syscall entry point -- while the\n trampoline is still writable and before the flip to read+execute. That window\n is the only correct place for it. `litebox_common_linux` cannot depend on the\n rewriter, so `litebox_shim_linux` holds the two slot constants together with a\n `const` assertion rather than a comment.\n\n What remains for a `TPIDR_EL0`-using guest: `pthread_setspecific` of each guest\n thread's pointer into the reserved key, and a macOS runner to exercise any of\n it -- none wires `MacOsUserland` into `litebox_shim_linux` today, so this whole\n path is still unexercised end to end on hardware.\n* **The platform's *own* per-thread context-switch bookkeeping** —\n **RESOLVED on real hardware.** A separate problem from the rewriter's guest\n slot above, and the thing that limited this platform to one guest thread at a\n time. `litebox_platform_linux_userland`'s x86_64\n `run_thread_arch`/`switch_to_guest`/`syscall_callback` (the closest thing to\n a template) does not only virtualize the *guest's* thread pointer -- it also\n stashes its own bookkeeping (`host_sp`, `host_bp`, `guest_context_top`,\n `in_guest`) in `fs:`-relative TLS slots, because by the time\n `syscall_callback` runs, every general-purpose register holds live guest\n state and there is nothing else durable to read \"where was the host stack\"\n from. That mechanism is entirely x86_64-ELF-specific (raw `@tpoff`-relative\n local-exec TLS addressing, resolved to a link-time-fixed offset with no\n function call and no runtime-determined value at all) and has no Mach-O\n equivalent to copy directly.\n\n **The building block, confirmed on real Apple M3 Pro hardware:** a raw\n `mrs tpidrro_el0` (masked, `& ~7`, matching libSystem's own\n `_os_tsd_get_base`) plus a `[base, #(key * 8)]` read/write reaches the *same*\n per-thread storage `pthread_getspecific`/`pthread_setspecific` do, for a\n **second**, independently `pthread_key_create`-reserved dynamic TSD key (not\n just the one already relied on for the guest's own `TPIDR_EL0` shadow) -- in\n both directions, across the full `usize` range, and disjointly across two\n genuinely concurrent OS threads. Measured again this pass: the first dynamic\n key a Rust binary gets is 259 and the pool is exhausted at key 767.\n\n **The blocker, and how it was actually solved.** Of the six naked functions\n in `litebox_platform_macos_userland::guest`, five are reached with registers\n to spare (three of them by a signal handler's `pc` redirect, so *every*\n register is free) and can simply do a two-register lookup: one register for\n the run-time-determined TSD byte offset, one for the `TPIDRRO_EL0` value. The\n syscall callback cannot: the rewriter's `SVC` gate leaves exactly **one**\n register free (`X16`), because `X17` still holds the guest's real value,\n which real Linux AArch64 preserves across a syscall and which this file's own\n fidelity philosophy (`preserves_registers_across_capture_and_resume`) commits\n to capturing faithfully. One register is enough for\n `mrs`/`and`/`ldr [x16, #imm]` **only if the immediate is a compile-time"},{"key":"ci-4503fc80-eff84b10-0","score":11.346984316670138,"symbol":{"kind":"function_item","line_end":2349,"line_start":2262,"name":"test_page_provider","path":"litebox_platform_windows_userland/src/lib.rs"},"text":"litebox_platform_windows_userland/src/lib.rs:2262:2349 test_page_provider\n fn test_page_provider() {\n let collect_regions = |r| {\n let mut regions = Vec::new();\n process_memory_range_by_regions(\n r,\n |region, state| -> Result {\n regions.push((region, state));\n Ok(true)\n },\n )\n .unwrap();\n regions\n };\n\n let platform = WindowsUserland::new();\n let system_allocation_granularity =\n platform.sys_info.read().unwrap().dwAllocationGranularity as usize;\n // Allocate some pages: it should reserve `system_allocation_granularity` bytes but only commit 0x1000 bytes\n let addr = >::allocate_pages(\n platform,\n 0..0x1000,\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n assert_eq!(\n collect_regions(addr..addr + system_allocation_granularity),\n vec![\n (\n addr..addr + 0x1000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n (\n addr + 0x1000..addr + system_allocation_granularity,\n windows_sys::Win32::System::Memory::MEM_RESERVE\n ),\n ]\n );\n\n assert!(system_allocation_granularity >= 0x1_0000);\n // We should be able to allocate [addr + 0x8000, addr + 0x1_0000)\n let addr2 = >::allocate_pages(\n platform,\n (addr + 0x8000)..(addr + 0x1_0000),\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n // Even though `fixed_address` is false, we should still get the requested address if it's free.\n assert_eq!(addr2, addr + 0x8000);\n assert_eq!(\n collect_regions(addr..addr + 0x1_0000),\n vec![\n (\n addr..addr + 0x1000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n (\n addr + 0x1000..addr + 0x8000,\n windows_sys::Win32::System::Memory::MEM_RESERVE\n ),\n (\n addr + 0x8000..addr + 0x1_0000,\n windows_sys::Win32::System::Memory::MEM_COMMIT\n ),\n ]\n );\n\n // Try to allocate [addr + 0x4000, addr + 0x1_0000), which overlaps with existing committed pages.\n // OS should allocate a new region instead of the requested one (as `fixed_address` is false)\n let addr3 = >::allocate_pages(\n platform,\n (addr + 0x4000)..(addr + 0x1_0000),\n MemoryRegionPermissions::WRITE,\n false,\n true,\n FixedAddressBehavior::Hint,\n )\n .unwrap()\n .as_usize();\n assert_ne!(addr3, addr + 0x4000);\n }"},{"key":"ci-a87bfb4d-219d0a8d-0","score":10.260236172469131,"symbol":{"kind":"function_item","line_end":41,"line_start":35,"name":"default_low_addr","path":"litebox_shim_linux/src/loader/mod.rs"},"text":"litebox_shim_linux/src/loader/mod.rs:35:41 default_low_addr\npub(crate) fn default_low_addr() -> usize {\n DEFAULT_LOW_ADDR.max(\n >::TASK_ADDR_MIN,\n )\n}"},{"key":"ci-26120153-df996502-0","score":10.201592570671137,"symbol":{"kind":"function_item","line_end":1689,"line_start":1645,"name":"nt_convert_between_auxiliary_counter_status_matches_host_ntdll","path":"litebox_shim_windows/src/syscalls/sysinfo.rs"},"text":"litebox_shim_windows/src/syscalls/sysinfo.rs:1645:1689 nt_convert_between_auxiliary_counter_status_matches_host_ntdll\n fn nt_convert_between_auxiliary_counter_status_matches_host_ntdll() {\n run_with_test_platform_pointers(|| {\n let source = 0u64;\n let mut destination = 0u64;\n let mut conversion_error = 0u64;\n\n // SAFETY: Passing a null source pointer intentionally probes host ntdll's invalid\n // input behavior; the output pointers are valid local scalars for the duration.\n let host_null_source_status = unsafe {\n host_status(NtConvertBetweenAuxiliaryCounterAndPerformanceCounter(\n 0,\n core::ptr::null(),\n &raw mut destination,\n &raw mut conversion_error,\n ))\n };\n let guest_null_source_status =\n TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter(\n 0,\n null_const_ptr(),\n mut_ptr(&mut destination),\n Some(mut_ptr(&mut conversion_error)),\n );\n assert_eq!(guest_null_source_status, host_null_source_status);\n\n // SAFETY: All pointers passed to host ntdll point at local scalar variables that live\n // for the whole call; the function does not retain them.\n let host_valid_source_status = unsafe {\n host_status(NtConvertBetweenAuxiliaryCounterAndPerformanceCounter(\n 0,\n &raw const source,\n &raw mut destination,\n &raw mut conversion_error,\n ))\n };\n let guest_valid_source_status =\n TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter(\n 0,\n const_ptr(&source),\n mut_ptr(&mut destination),\n Some(mut_ptr(&mut conversion_error)),\n );\n assert_eq!(guest_valid_source_status, host_valid_source_status);\n });\n }"},{"key":"ci-f4597cff-da47046c-1","score":9.838240599746811,"symbol":{"kind":"function_item","line_end":2576,"line_start":2478,"name":"test_prepatched_mmap_publishes_trampoline_header","path":"litebox_shim_linux/src/syscalls/mm.rs"},"text":"litebox_shim_linux/src/syscalls/mm.rs:2478:2576 test_prepatched_mmap_publishes_trampoline_header\n fn test_prepatched_mmap_publishes_trampoline_header() {\n use litebox::platform::SystemInfoProvider as _;\n\n // Values the packager might have seeded; the runtime must replace them.\n const SEED_ENTRY: u64 = 0x1111_1111_1111_1111;\n const SEED_TP_OFFSET: u64 = 0x2222_2222_2222_2222;\n const TRAMP_SIZE: usize = 32;\n\n let _guard = crate::syscalls::tests::address_space_guard();\n let task = init_platform(None);\n\n // Synthetic pre-patched ET_DYN:\n // [ELF header + one PT_LOAD phdr | pad to PAGE_SIZE]\n // [trampoline code (TRAMP_SIZE bytes)] [32-byte LITEBOX0 trailer]\n let mut file = alloc::vec![0u8; PAGE_SIZE + TRAMP_SIZE + 32];\n file[0..4].copy_from_slice(b\"\\x7fELF\");\n file[4] = 2; // ELFCLASS64\n file[5] = 1; // ELFDATA2LSB\n file[6] = 1; // EV_CURRENT\n file[16..18].copy_from_slice(&3u16.to_le_bytes()); // e_type = ET_DYN\n #[cfg(target_arch = \"x86_64\")]\n let e_machine: u16 = 62; // EM_X86_64\n #[cfg(target_arch = \"aarch64\")]\n let e_machine: u16 = 183; // EM_AARCH64\n file[18..20].copy_from_slice(&e_machine.to_le_bytes());\n file[20..24].copy_from_slice(&1u32.to_le_bytes()); // e_version\n file[32..40].copy_from_slice(&64u64.to_le_bytes()); // e_phoff\n file[52..54].copy_from_slice(&64u16.to_le_bytes()); // e_ehsize\n file[54..56].copy_from_slice(&56u16.to_le_bytes()); // e_phentsize\n file[56..58].copy_from_slice(&1u16.to_le_bytes()); // e_phnum\n // PT_LOAD at p_offset 0, p_vaddr 0, R+X, one page.\n let ph = 64;\n file[ph..ph + 4].copy_from_slice(&1u32.to_le_bytes()); // p_type\n file[ph + 4..ph + 8].copy_from_slice(&5u32.to_le_bytes()); // p_flags R|X\n file[ph + 32..ph + 40].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_filesz\n file[ph + 40..ph + 48].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_memsz\n file[ph + 48..ph + 56].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_align\n // Trampoline code, seeded like the packager leaves it.\n file[PAGE_SIZE..PAGE_SIZE + 8].copy_from_slice(&SEED_ENTRY.to_le_bytes());\n file[PAGE_SIZE + 8..PAGE_SIZE + 16].copy_from_slice(&SEED_TP_OFFSET.to_le_bytes());\n // Trailer: magic, trampoline file offset, vaddr (just past PT_LOAD), size.\n let t = PAGE_SIZE + TRAMP_SIZE;\n file[t..t + 8].copy_from_slice(b\"LITEBOX0\");\n file[t + 8..t + 16].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes());\n file[t + 16..t + 24].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes());\n file[t + 24..t + 32].copy_from_slice(&(TRAMP_SIZE as u64).to_le_bytes());\n\n let fd = task\n .sys_open(\n \"prepatched_test.so\",\n OFlags::RDWR | OFlags::CREAT,\n Mode::RWXU,\n )\n .unwrap();\n let fd = i32::try_from(fd).unwrap();\n assert_eq!(task.sys_write(fd, &file, None).unwrap(), file.len());\n\n // Map two pages so the trampoline's MAP_FIXED landing zone (page 1,\n // per the trailer's vaddr) is this test's own mapping, not whatever\n // else the harness put there.\n let addr = task\n .sys_mmap(\n 0,\n 2 * PAGE_SIZE,\n ProtFlags::PROT_READ | ProtFlags::PROT_EXEC,\n MapFlags::MAP_PRIVATE,\n fd,\n 0,\n )\n .unwrap();\n\n let tramp = UserPtrMut::::from_usize(addr.as_usize() + PAGE_SIZE);\n let header = tramp.to_owned_slice::(16).unwrap();\n let platform = task.global.platform;\n\n let entry = platform.get_syscall_entry_point();\n assert_ne!(entry, 0, \"test platform must expose a syscall entry point\");\n assert_eq!(\n header[..8],\n entry.to_le_bytes(),\n \"mmap path must publish the runtime syscall entry into trampoline word 0\"\n );\n\n match platform.get_guest_tp_slot_offset() {\n Some(offset) => assert_eq!(\n header[8..16],\n offset.to_ne_bytes(),\n \"mmap path must publish the runtime guest TP slot offset into trampoline word 1\"\n ),\n None => assert_eq!(\n header[8..16],\n SEED_TP_OFFSET.to_le_bytes(),\n \"platforms that bake the TP offset into gates must leave the seeded word alone\"\n ),\n }\n\n task.sys_munmap(addr, 2 * PAGE_SIZE).unwrap();\n task.sys_close(fd).unwrap();\n }"},{"key":"ci-36cd1f98-af2dccfd-0","score":9.634784098868156,"symbol":{"kind":"function_item","line_end":345,"line_start":330,"name":"panic","path":"litebox_runner_snp/src/main.rs"},"text":"litebox_runner_snp/src/main.rs:330:345 panic\nfn panic(info: &core::panic::PanicInfo) -> ! {\n let msg = info.message();\n ghcb_prints(msg.as_str().unwrap_or(\"empty panic message\"));\n\n if let Some(location) = info.location() {\n ghcb_prints(\"panic occurred at \");\n ghcb_prints(location.file());\n litebox_platform_linux_kernel::print_str_and_int!(\":\", u64::from(location.line()), 10);\n } else {\n ghcb_prints(\"panic occurred but can't get location information...\");\n }\n litebox_platform_linux_kernel::host::snp::snp_impl::HostSnpInterface::terminate(\n globals::SM_SEV_TERM_SET,\n globals::SM_TERM_GENERAL,\n );\n}"},{"key":"ci-a981cb7b-ec3c0462-0","score":9.314891819169782,"symbol":{"kind":"function_item","line_end":628,"line_start":571,"name":"et_exec_interpreter_loads_top_down_above_low_heap","path":"litebox_shim_linux/src/loader/elf.rs"},"text":"litebox_shim_linux/src/loader/elf.rs:571:628 et_exec_interpreter_loads_top_down_above_low_heap\n fn et_exec_interpreter_loads_top_down_above_low_heap() {\n let _guard = crate::syscalls::tests::address_space_guard();\n let task = crate::syscalls::tests::init_platform(None);\n write_file(&task, \"/main\", &minimal_elf(ET_EXEC, Some(INTERP_PATH)));\n write_file(&task, \"/ld.so\", &minimal_elf(ET_DYN, None));\n\n let mut loader = ElfLoader::new(&task, \"/main\").expect(\"loader should parse test ELFs\");\n let main = loader\n .main\n .load_mapped(task.global.platform)\n .expect(\"main should load\");\n assert_eq!(main.base_addr, 0);\n\n let interp = loader\n .interp\n .as_mut()\n .expect(\"test main should have PT_INTERP\")\n .load_mapped(task.global.platform)\n .expect(\"interpreter should load\");\n\n // The interpreter must land high — via the top-down search — so the\n // low ET_EXEC brk heap below it is not capped. The exact address is\n // not asserted: `get_unmmaped_area` returns the highest free gap, and\n // host mappings seeded into the userland VMA tree can sit near the top\n // and push that gap below the very top slot (see `mm/linux.rs`). Assert\n // the invariant that matters — placement in the high half of the\n // address space, far above the low-heap region — not one exact slot.\n let addr_max = >::TASK_ADDR_MAX;\n assert!(\n interp.base_addr >= addr_max / 2,\n \"ET_EXEC interpreter loaded at {:#x}, near the low-heap region {:#x} rather than top-down high (>= {:#x})\",\n interp.base_addr,\n crate::loader::DEFAULT_LOW_ADDR,\n addr_max / 2,\n );\n\n // Release both images before returning. Every test in this binary shares\n // one host address space, but each builds its own task with its own VMM,\n // and a VMM models only its own mappings -- so anything this test leaves\n // mapped is invisible to the next test's placement search and collides\n // with whatever it picks. That is easy to miss on a host whose guest\n // range sits well clear of the host's own image; on arm64 macOS both\n // live above the 4 GiB `__PAGEZERO` floor, so the collision is routine.\n // Each synthetic image maps exactly one PT_LOAD page (`minimal_elf`\n // sets filesz == memsz == PAGE_SIZE). Do NOT derive the length from\n // `brk`: on a platform that requires syscall rewriting, `load_mapped`\n // pushes brk DEFAULT_RESERVED_SPACE_SIZE (16 MiB) past the image\n // without mapping that space, so a brk-derived munmap overshoots --\n // the top-down interpreter ends exactly at TASK_ADDR_MAX, which on\n // Linux x86-64 is the host TASK_SIZE (munmap EINVAL panics\n // deallocate_pages), and Windows' region walk asserts on the\n // never-committed tail.\n let exec_start = usize::try_from(EXEC_LOAD_ADDR).expect(\"load address fits usize\");\n task.sys_munmap(UserPtrMut::from_usize(exec_start), PAGE_SIZE)\n .expect(\"main image should unmap\");\n task.sys_munmap(UserPtrMut::from_usize(interp.base_addr), PAGE_SIZE)\n .expect(\"interpreter image should unmap\");\n }"}],"commits":[{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":9.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":6.0},{"hash":"323d13318ea10946e74e99a7e24d593c9ef6e7ed","message":"feat(shim): the syscall surface an X11 desktop stack needs","score":6.0},{"hash":"366634828dae54bd2ff0c64e83f5d1cb9ade7c46","message":"fix(ci): clear clippy 1.98 lints, fmt drift, and vendored-crate no_std check","score":5.0},{"hash":"624f0ceedf841b147c5e972104d7cfed355ccc90","message":"feat(net): guest web access without root via an in-stack HTTP proxy","score":4.0},{"hash":"5251e6a45413ae95629010a02e13a7959a9e3517","message":"feat(evdev): /dev/input/event* emulation wired to VNC input","score":4.0},{"hash":"d06199534c3cf1090fdb261b8c800d380dea941e","message":"fix: getrusage/wait4 guest-stack overrun, doc gate, pipe-hang probes","score":4.0},{"hash":"de0d9a8b391bb610126b398a2619989025675cf9","message":"fix(ci): green the cross-platform test suite","score":4.0},{"hash":"876e41c093728c831e524c56c2decb4a784dbade","message":"Add an AF_NETLINK route socket, so os.networkInterfaces() works","score":4.0},{"hash":"dbc767aa1920e9a3c60c3602e6e899b4d37eeeb4","message":"Wire fchown via a new fd_chown, completing the chown family","score":4.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788375444850-500-dd93e25d6dc892e9","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"c769462f66429da4","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-882091788375033189.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-882091788375033189.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-88241788295926913.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-88241788295926913.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-89721788215675643.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-89721788215675643.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-89841788295932523.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-89841788295932523.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-92521788373655553.json b/.agentplug/plugin-dispatch/out/gm-codesearch-92521788373655553.json new file mode 100644 index 0000000000..5095c4c98b --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-92521788373655553.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-ea2224bb-9e5d768c-1","score":39.34267990834939,"symbol":{"kind":"struct_item","line_end":44,"line_start":21,"name":"CliArgs","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:21:44 CliArgs\npub struct CliArgs {\n /// The program and arguments passed to it (e.g., `/bin/ls --color`).\n ///\n /// The program path refers to a path inside the tar archive provided via\n /// `--initial-files`. All binaries must be pre-rewritten with the syscall\n /// rewriter.\n #[arg(required = true, trailing_var_arg = true, value_hint = clap::ValueHint::CommandWithArguments)]\n pub program_and_arguments: Vec,\n /// Environment variables passed to the program (`K=V` pairs; can be invoked multiple times)\n #[arg(long = \"env\")]\n pub environment_variables: Vec,\n /// Forward the existing environment variables\n #[arg(long = \"forward-env\")]\n pub forward_environment_variables: bool,\n /// Allow using unstable options\n #[arg(short = 'Z', long = \"unstable\")]\n pub unstable: bool,\n /// Tar archive containing the program and its shared libraries.\n ///\n /// All ELF binaries should be pre-rewritten with the syscall rewriter\n /// (e.g., via `litebox-packager`).\n #[arg(long = \"initial-files\", value_name = \"PATH_TO_TAR\", value_hint = clap::ValueHint::FilePath)]\n pub initial_files: PathBuf,\n}"},{"key":"ci-39f1a0ff-d0115eff-1","score":39.30384275976628,"symbol":{"kind":"function_item","line_end":12,"line_start":8,"name":"main","path":"litebox_runner_linux_on_macos_userland/src/main.rs"},"text":"litebox_runner_linux_on_macos_userland/src/main.rs:8:12 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_linux_on_macos_userland::CliArgs;\n litebox_runner_linux_on_macos_userland::run(CliArgs::parse())\n}"},{"key":"ci-b8807bce-1130b664-1","score":33.2450585119057,"symbol":{"kind":"function_item","line_end":9,"line_start":5,"name":"main","path":"litebox_runner_optee_on_linux_userland/src/main.rs"},"text":"litebox_runner_optee_on_linux_userland/src/main.rs:5:9 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_runner_optee_on_linux_userland::CliArgs;\n litebox_runner_optee_on_linux_userland::run(CliArgs::parse())\n}"},{"key":"ci-30003d83-48919216-0","score":32.376689517595665,"symbol":{"kind":"section","line_end":130,"line_start":1,"name":"","path":"docs/macos.md"},"text":"docs/macos.md:1:130 \n# LiteBox on macOS (Apple Silicon)\n\nLiteBox runs guest instructions natively; only the *system* interface is\nvirtualized. On an Apple Silicon Mac that means the only sensible configuration\nis an **AArch64 Linux guest on an AArch64 macOS host** — no emulation anywhere.\nThere is deliberately no x86-64 macOS platform: an x86-64 guest would need\ninstruction emulation, which is the thing this design exists to avoid.\n\nThis document covers what works today, what the host imposes, and what is left\nbefore a guest can actually execute.\n\n## What is in the tree\n\n| Piece | State |\n| --- | --- |\n| `litebox_platform_macos_userland` | The macOS \"South\" platform: memory, locking, time, signals, timers, threads, TLS, randomness, derived keys, stdio, `utun` networking, fault recovery. |\n| `litebox` core | Builds for `aarch64-apple-darwin`, including the Mach-O exception table. |\n| `litebox_shim_linux` | The Linux \"North\" shim, ported to AArch64: signal frames, syscall entry/return, thread-pointer handling, `stat`/`uname` ABI, exception decoding. |\n| `litebox_syscall_rewriter` | Already had AArch64 support (`arm64.rs`) for rewriting `SVC` and `TPIDR_EL0` accesses in Linux ELF images. |\n| `litebox_packager` | OCI mode now pulls the image matching the host architecture, and builds on Apple Silicon. |\n| Guest entry | **Implemented** (context switch + syscall dispatch), tested on real hardware. A syscall-only guest runs end to end; the guest thread-pointer plumbing and non-syscall event paths remain. See [Remaining work](#remaining-work). |\n\n## Building\n\n```sh\nrustup target add aarch64-apple-darwin\ncargo build --workspace --exclude litebox_runner_lvbs --exclude litebox_runner_snp\n```\n\n`litebox_runner_lvbs` and `litebox_runner_snp` are freestanding images for\ncustom targets and are not built for a hosted target on any platform.\n\nCI covers this in the `Build and Test macOS (Apple Silicon)` job, which also\ncompiles and runs `litebox_platform_macos_userland/tests/darwin_abi_probe.c`\nagainst the runner's real SDK headers -- the only check in this repo that\nverifies the crate's hand-written Mach/BSD struct layouts (used by the fault\nhandler to read `ucontext_t::uc_mcontext`) against an actual Darwin toolchain,\nsince nothing else in a Linux-hosted development loop can.\n\n### Hypervisor.framework release boundary\n\nThe stock-code backend requires Apple Silicon, macOS 26 or newer, and a macOS\n26 SDK or newer. The combined runner retains its macOS 11 deployment target so\nthe native backend still launches on older hosts; post-macOS-11 HVF imports are\nweak and the production boundary checks `__builtin_available(macOS 26.0, *)`\nbefore making any such call. `litebox_platform_macos_userland/build.rs`\ncompiles the narrow C boundary and linked EL1 monitor against the active SDK\nheaders, so Apple enum, object, and structure layouts do not get copied into\nRust constants. A release runner is built and signed with the checked-in\nentitlement manifest as follows:\n\n```sh\ncargo build --release --locked -p litebox_runner_linux_on_macos_userland\ncodesign --force --options runtime --sign - \\\n --entitlements litebox_runner_linux_on_macos_userland/entitlements.plist \\\n target/release/litebox_runner_linux_on_macos_userland\ncodesign --display --entitlements - \\\n target/release/litebox_runner_linux_on_macos_userland\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-boundary\ntarget/release/litebox_runner_linux_on_macos_userland \\\n --unstable --hvf-memory\n```\n\nThe boundary diagnostic validates the active-SDK configuration, maps and\nunmaps the monitor, and creates, verifies, and destroys a vCPU. The compact\nmemory diagnostic does not run guest instructions. It proves that IPA zero is\nreserved for the monitor, GVA=HVA mappings receive compact non-identity IPAs,\n16 KiB stage-one roots software-walk to the same pages, stage-two permissions\ncan move from RW to RX without RWX, rejected overlap transactions publish no\nnew root generation, and released IPA is reused without poisoning the VM. The\nlinked 16 KiB monitor remains the source of truth, but the VM owns an aligned\nallocator-backed copy: Hypervisor.framework rejects file-backed Mach-O\n`__TEXT` pages as stage-two backing.\n\n`com.apple.security.hypervisor=true` is mandatory for\nHypervisor.framework VM creation. `com.apple.security.cs.allow-jit=true` remains\nrequired when the native rewritten backend is packaged with Hardened Runtime.\nThe legacy `com.apple.vm.hypervisor` entitlement belongs only to deployment\ntargets through macOS 10.15 and is deliberately absent. Notarization and\nHardened Runtime do not grant Hypervisor.framework authorization; the final\nexecutable still needs the hypervisor entitlement.\n\n## What the host imposes\n\n### 16 KiB pages\n\nApple Silicon's page size is 16 KiB. Every fixed mapping and every protection\nchange must be aligned to it, so `litebox::mm::linux::PAGE_SIZE` is 16384 on\nthis target rather than 4096. The guest sees the same value through `AT_PAGESZ`,\nwhich is exactly how a Linux kernel configured for 16 KiB or 64 KiB pages\nreports itself.\n\nAArch64 ELF images are conventionally linked with a 64 KiB maximum page size, so\ntheir `PT_LOAD` segments stay aligned either way. An image built with 4 KiB\nsegment alignment will not map cleanly.\n\n### The first 4 GiB is unusable\n\nAn arm64 Mach-O process reserves `[0, 4 GiB)` as the `__PAGEZERO` segment:\nunmapped and impossible to map over. `TASK_ADDR_MIN` is therefore `0x1_0000_0000`.\n\nThe practical consequence is that guest images must be position-independent, or\nlinked above 4 GiB. An `ET_EXEC` binary linked at the customary `0x400000`\ncannot be loaded at its preferred address on this host.\n\n### W^X, `MAP_JIT`, and code signing\n\nmacOS refuses to make anonymous memory executable through the ordinary path, and\nrefuses to add `PROT_EXEC` to anything that was ever writable. The supported\nescape hatch is `MAP_JIT`, which the platform passes whenever a mapping requests\n`EXEC`. Using it has two consequences:\n\n1. **The JIT entitlement is only load-bearing under the Hardened Runtime.**\n Per Apple's own documentation, `com.apple.security.cs.allow-jit` is required\n only when a binary has the Hardened Runtime enabled (`codesign --options\n runtime`, which in turn is what notarization requires); without it,\n `MAP_JIT` works with or without the entitlement present. The command below\n ad-hoc-signs with the entitlement anyway -- it costs nothing and future-proofs\n a later `--options runtime`, notarized build -- but for local development\n outside Gatekeeper, neither the entitlement nor notarization is actually\n required for `MAP_JIT` itself to work. Create an entitlements file:\n\n ```xml\n \n \n \n \n com.apple.security.cs.allow-jit"},{"key":"ci-f0a0109-776f5635-1","score":31.378261622509427,"symbol":{"kind":"function_item","line_end":106,"line_start":50,"name":"run","path":"litebox_runner_optee_on_linux_userland/src/lib.rs"},"text":"litebox_runner_optee_on_linux_userland/src/lib.rs:50:106 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let ldelf_data: Vec = {\n let ldelf = PathBuf::from(&cli_args.ldelf);\n let data =\n std::fs::read(&ldelf).with_context(|| format!(\"failed to read {}\", cli_args.ldelf))?;\n if cli_args.rewrite_syscalls {\n litebox_syscall_rewriter::hook_syscalls_in_elf(&data, None)\n .with_context(|| format!(\"failed to rewrite {}\", cli_args.ldelf))?\n } else {\n data\n }\n };\n\n let prog_data: Vec = {\n let prog = PathBuf::from(&cli_args.program);\n let data =\n std::fs::read(&prog).with_context(|| format!(\"failed to read {}\", cli_args.program))?;\n if cli_args.rewrite_syscalls {\n litebox_syscall_rewriter::hook_syscalls_in_elf(&data, None)\n .with_context(|| format!(\"failed to rewrite {}\", cli_args.program))?\n } else {\n data\n }\n };\n\n // TODO(jb): Clean up platform initialization once we have https://github.com/MSRSSP/litebox/issues/24\n let platform = Platform::new(None);\n litebox_platform_multiplex::set_platform(platform);\n let shim_builder = litebox_shim_optee::OpteeShimBuilder::new();\n let _litebox = shim_builder.litebox();\n let shim = shim_builder.build();\n\n platform.initialize_boot_specific_kdf_support();\n\n if cli_args.command_sequence.is_empty() {\n run_ta_with_default_commands(&shim, ldelf_data.as_slice(), prog_data.as_slice());\n } else {\n tests::run_ta_with_test_commands(\n &shim,\n ldelf_data.as_slice(),\n prog_data.as_slice(),\n cli_args.program.as_str(),\n &PathBuf::from(&cli_args.command_sequence),\n );\n }\n Ok(())\n}"},{"key":"ci-ea2224bb-9e5d768c-0","score":25.50478006654966,"symbol":{"kind":"function_item","line_end":147,"line_start":53,"name":"run","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:53:147 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let tar_file = &cli_args.initial_files;\n if tar_file.extension().and_then(|x| x.to_str()) != Some(\"tar\") {\n anyhow::bail!(\"Expected a .tar file, found {}\", tar_file.display());\n }\n let tar_data = std::fs::read(tar_file)\n .map_err(|e| anyhow!(\"Could not read tar file at {}: {}\", tar_file.display(), e))?;\n\n let platform = Platform::new();\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n // The program path is a Unix-style path inside the tar archive.\n let prog_path = &cli_args.program_and_arguments[0];\n\n let initial_file_system = {\n let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem.with_root_privileges(|fs| {\n use litebox::fs::FileSystem as _;\n fs.mkdir(\n \"/tmp\",\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap();\n fs.chown(\"/tmp\", Some(1000), Some(1000)).unwrap();\n\n // Standard FHS directories that guest tools expect to already exist (e.g. `apk`\n // opens a log file under `/var/log`) but that don't survive as empty-directory\n // entries when an OCI image's rootfs is scanned into a file-based tar: an empty\n // directory has no file contents, so it produces no tar entry, and `TarRo`'s\n // directory tree is inferred purely from file paths.\n for dir in [\"/run\", \"/var\", \"/var/log\", \"/var/cache\", \"/var/tmp\"] {\n fs.mkdir(\n dir,\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|_| {\n panic!(\"{dir} creation cannot fail on a fresh in-memory file system\")\n });\n }\n });\n\n shim_builder.default_fs(in_mem, tar_data.into())\n };\n let initial_file_system = std::sync::Arc::new(initial_file_system);\n\n let shim = shim_builder.build();\n let argv = cli_args\n .program_and_arguments\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp: Vec<_> = cli_args\n .environment_variables\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp = if cli_args.forward_environment_variables {\n envp.into_iter()\n .chain(std::env::vars().map(|(k, v)| {\n std::ffi::CString::new(k.bytes().chain(*b\"=\").chain(v.bytes()).collect::>())\n .unwrap()\n }))\n .collect()\n } else {\n envp\n };\n\n let program = shim\n .load_program(\n initial_file_system,\n platform.init_task(),\n prog_path,\n argv,\n envp,\n )\n .unwrap();\n unsafe {\n litebox_platform_windows_userland::run_thread(\n program.entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n std::process::exit(program.process.wait())\n}"},{"key":"ci-cd2594be-4b0edc44-1","score":25.0845850097393,"symbol":{"kind":"function_item","line_end":337,"line_start":198,"name":"run_dynamic_linked_prog_with_rewriter","path":"litebox_runner_linux_on_windows_userland/tests/loader.rs"},"text":"litebox_runner_linux_on_windows_userland/tests/loader.rs:198:337 run_dynamic_linked_prog_with_rewriter\nfn run_dynamic_linked_prog_with_rewriter(\n libs_to_rewrite: &[(&str, &str)],\n exec_name: &str,\n cmd_args: &[&str],\n install_files: fn(std::path::PathBuf),\n) {\n // Use the already compiled executable from the tests folder (same dir as this file)\n let mut test_dir = std::path::PathBuf::from(env!(\"CARGO_MANIFEST_DIR\"));\n test_dir.push(\"tests/test-bins\");\n\n let prog_name = exec_name;\n let prog_name_hooked = format!(\"{prog_name}.hooked\");\n\n let path = test_dir.join(prog_name);\n let hooked_path = test_dir.join(&prog_name_hooked);\n\n let out_path = std::env::var(\"OUT_DIR\").unwrap();\n\n // Rewrite the target ELF executable file\n let _ = std::fs::remove_file(hooked_path.clone());\n let cargo = std::env::var(\"CARGO\").unwrap_or_else(|_| \"cargo\".to_string());\n let output = std::process::Command::new(&cargo)\n .args([\n \"run\",\n \"-p\",\n \"litebox_syscall_rewriter\",\n \"--\",\n path.to_str().unwrap(),\n \"-o\",\n hooked_path.to_str().unwrap(),\n ])\n .output()\n .expect(\"Failed to run syscall rewriter\");\n assert!(\n output.status.success(),\n \"failed to run syscall rewriter {:?}\",\n std::str::from_utf8(output.stderr.as_slice()).unwrap()\n );\n\n // Create tar file containing all dependencies\n let tar_src_path = std::path::Path::new(&out_path).join(\"test_program_tar\");\n println!(\n \"Creating tar source directory path: {}\",\n tar_src_path.to_str().unwrap()\n );\n\n std::fs::create_dir_all(tar_src_path.join(\"out\")).unwrap();\n\n // Rewrite all libraries that are required for initialization\n for (file, prefix) in libs_to_rewrite {\n let src = test_dir.join(file);\n let dst_dir = tar_src_path.join(prefix.trim_start_matches('/'));\n let dst = dst_dir.join(file);\n std::fs::create_dir_all(&dst_dir).unwrap();\n let _ = std::fs::remove_file(&dst);\n println!(\n \"Running `cargo run -p litebox_syscall_rewriter -- {} -o {}`\",\n src.to_str().unwrap(),\n dst.to_str().unwrap(),\n );\n let output = std::process::Command::new(&cargo)\n .args([\n \"run\",\n \"-p\",\n \"litebox_syscall_rewriter\",\n \"--\",\n src.to_str().unwrap(),\n \"-o\",\n dst.to_str().unwrap(),\n ])\n .output()\n .expect(\"Failed to run syscall rewriter\");\n assert!(\n output.status.success(),\n \"failed to run syscall rewriter {:?}\",\n std::str::from_utf8(output.stderr.as_slice()).unwrap()\n );\n }\n\n // Install the required files (e.g., scripts) to tar directory's /out\n install_files(tar_src_path.join(\"out\"));\n\n // Copy the hooked binary into the tar source directory\n let hooked_tar_dir = tar_src_path.join(\"bin\");\n std::fs::create_dir_all(&hooked_tar_dir).unwrap();\n std::fs::copy(&hooked_path, hooked_tar_dir.join(&prog_name_hooked)).unwrap();\n\n // tar\n let tar_target_file = std::path::Path::new(&out_path).join(\"rootfs_rewriter.tar\");\n let tar_data = std::process::Command::new(\"tar\")\n .args([\n \"-cvf\",\n tar_target_file.to_str().unwrap(),\n \"bin\",\n \"lib\",\n \"lib64\",\n \"out\",\n ])\n .current_dir(&tar_src_path)\n .output()\n .expect(\"Failed to create tar file\");\n assert!(\n tar_data.status.success(),\n \"failed to create tar file {:?}\",\n std::str::from_utf8(tar_data.stderr.as_slice()).unwrap()\n );\n println!(\"Tar file created at: {}\", tar_target_file.to_str().unwrap());\n\n let binary_path = std::env::var(\"NEXTEST_BIN_EXE_litebox_runner_linux_on_windows_userland\")\n .unwrap_or_else(|_| {\n env!(\"CARGO_BIN_EXE_litebox_runner_linux_on_windows_userland\").to_string()\n });\n\n // The program path refers to the tar-internal path.\n let prog_tar_path = format!(\"/bin/{prog_name_hooked}\");\n\n // Run litebox_runner_linux_on_windows_userland with the tar file\n let mut args = vec![\n // Tell ld where to find the libraries.\n // See https://man7.org/linux/man-pages/man8/ld.so.8.html for how ld works.\n // Alternatively, we could add a `/etc/ld.so.cache` file to the rootfs.\n \"--env\",\n \"LD_LIBRARY_PATH=/lib64:/lib32:/lib\",\n \"--initial-files\",\n tar_target_file.to_str().unwrap(),\n ];\n args.push(&prog_tar_path);\n args.extend_from_slice(cmd_args);\n\n let mut command = std::process::Command::new(&binary_path);\n command.args(&args);\n println!(\"Running `{command:?}`\");\n let status = command\n .status()\n .expect(\"Failed to run litebox_runner_linux_on_windows_userland\");\n assert!(\n status.success(),\n \"failed to run litebox_runner_linux_on_windows_userland: {status}\",\n );\n}"},{"key":"ci-33f01406-6265afd1-0","score":23.08634154079395,"symbol":{"kind":"function_item","line_end":8,"line_start":4,"name":"main","path":"litebox_packager/src/main.rs"},"text":"litebox_packager/src/main.rs:4:8 main\nfn main() -> anyhow::Result<()> {\n use clap::Parser as _;\n use litebox_packager::CliArgs;\n litebox_packager::run(CliArgs::parse())\n}"},{"key":"ci-46ccd3db-66d3ccf6-0","score":23.081579357370416,"symbol":{"kind":"function_definition","line_end":768,"line_start":578,"name":"main","path":"dev_bench/unixbench/run_unixbench.py"},"text":"dev_bench/unixbench/run_unixbench.py:578:768 main\ndef main():\n parser = argparse.ArgumentParser(\n description=\"Run UnixBench benchmarks natively and under LiteBox, then compare results.\",\n )\n parser.add_argument(\n \"--benchmarks\", nargs=\"+\", default=DEFAULT_BENCHMARKS,\n choices=list(BENCHMARKS.keys()),\n help=\"Which benchmarks to run (default: all supported)\",\n )\n parser.add_argument(\n \"--mode\", choices=[\"both\", \"native\", \"litebox\"], default=\"both\",\n help=\"Run mode: 'native', 'litebox', or 'both' (default: both)\",\n )\n parser.add_argument(\n \"--duration\", type=int, default=None,\n help=\"Override duration in seconds for each benchmark run \"\n \"(default: use official per-benchmark durations)\",\n )\n parser.add_argument(\n \"--iterations\", type=int, default=None,\n help=\"Override number of iterations per benchmark \"\n \"(default: use official per-benchmark iteration counts)\",\n )\n parser.add_argument(\n \"--release\", action=\"store_true\",\n help=\"Use release build of litebox binaries\",\n )\n parser.add_argument(\n \"--runner-path\", type=str, default=None,\n help=\"Path to litebox_runner_linux_userland binary (auto-detected if not given)\",\n )\n parser.add_argument(\n \"--packager-path\", type=str, default=None,\n help=\"Path to litebox_packager binary (auto-detected if not given)\",\n )\n parser.add_argument(\n \"--no-build\", action=\"store_true\",\n help=\"Skip building litebox binaries (use existing binaries as-is)\",\n )\n parser.add_argument(\n \"--windows\", action=\"store_true\",\n help=\"Run on Windows using litebox_runner_linux_on_windows_userland. \"\n \"Requires --prepared-dir with artifacts from prepare_unixbench.py\",\n )\n parser.add_argument(\n \"--prepared-dir\", type=str, default=None,\n help=\"Path to directory of pre-prepared artifacts from prepare_unixbench.py \"\n \"(required for --windows mode)\",\n )\n parser.add_argument(\n \"--output\", type=str, default=None,\n help=\"Save results to a JSON file\",\n )\n parser.add_argument(\n \"--work-dir\", type=str, default=None,\n help=\"Working directory for intermediate files (default: temp dir)\",\n )\n\n args = parser.parse_args()\n\n # ── Validate flags ──────────────────────────────────────────────────\n\n is_windows_mode = args.windows\n prepared_dir = Path(args.prepared_dir) if args.prepared_dir else None\n\n if is_windows_mode:\n if args.mode == \"native\":\n print(\"Error: --windows cannot be combined with --mode native\")\n sys.exit(1)\n if args.mode == \"both\":\n # On Windows, there are no native Linux binaries to run\n print(\"Note: --windows mode only supports LiteBox runs (no native).\")\n args.mode = \"litebox\"\n if prepared_dir is None:\n # Default to benchmark/unixbench/prepared/ if it exists\n default_prepared = Path(__file__).resolve().parent / \"prepared\"\n if default_prepared.exists():\n prepared_dir = default_prepared\n else:\n print(\"Error: --windows requires --prepared-dir (or a prepared/ directory)\")\n print(\"Run prepare_unixbench.py on Linux/WSL first.\")\n sys.exit(1)\n if not prepared_dir.exists():\n print(f\"Error: prepared directory not found at {prepared_dir}\")\n sys.exit(1)\n\n workspace_root = find_workspace_root()\n unixbench_dir = find_unixbench_dir(workspace_root)\n pgms_dir = unixbench_dir / \"pgms\"\n\n # On Windows with --prepared-dir, we don't need UnixBench source\n if not is_windows_mode:\n ensure_unixbench_downloaded(workspace_root)\n ensure_unixbench_built(unixbench_dir)\n\n # Resolve litebox binaries\n run_litebox_mode = args.mode in (\"both\", \"litebox\")\n runner_path = None\n packager_path = None\n\n if run_litebox_mode:\n if args.runner_path:\n runner_path = Path(args.runner_path)\n if args.packager_path:\n packager_path = Path(args.packager_path)\n elif is_windows_mode:\n # On Windows, build (or locate) the runner\n runner_name = \"litebox_runner_linux_on_windows_userland\"\n if sys.platform == \"win32\":\n runner_name += \".exe\"\n build_type = \"release\" if args.release else \"debug\"\n runner_path = workspace_root / \"target\" / build_type / runner_name\n\n if not args.no_build:\n build_cmd = [\n \"cargo\", \"build\",\n \"-p\", \"litebox_runner_linux_on_windows_userland\",\n ]\n if args.release:\n build_cmd.append(\"--release\")\n print(f\"Building Windows runner ({build_type})...\")\n result = subprocess.run(build_cmd, cwd=str(workspace_root))\n if result.returncode != 0:\n print(f\"Error: cargo build failed (exit {result.returncode})\")\n sys.exit(1)\n print(\"Build complete.\")\n\n if not runner_path.exists():\n print(f\"Error: Windows runner not found at {runner_path}\")\n print(\"Build it on Windows first:\")\n print(f\" cargo build -p litebox_runner_linux_on_windows_userland\"\n + (\" --release\" if args.release else \"\"))\n sys.exit(1)\n elif args.no_build:\n # Use existing binaries without building\n runner_path, packager_path = find_litebox_binaries(\n workspace_root, args.release,\n )\n if runner_path is None:\n print(\"Error: litebox_runner_linux_userland not found.\")\n print(\"Run without --no-build, or build manually:\")\n print(\" cargo build -p litebox_runner_linux_userland\"\n + (\" --release\" if args.release else \"\"))\n sys.exit(1)\n else:\n # Build fresh binaries to ensure they are up-to-date\n runner_path, packager_path = build_litebox_binaries(\n workspace_root, args.release,\n )\n\n # Working directory\n if args.work_dir:\n work_dir = Path(args.work_dir)\n work_dir.mkdir(parents=True, exist_ok=True)\n cleanup_work_dir = False\n else:\n work_dir = Path(tempfile.mkdtemp(prefix=\"litebox_bench_\"))\n cleanup_work_dir = True\n\n print(f\"Workspace root: {workspace_root}\")\n if not is_windows_mode:\n print(f\"UnixBench dir: {unixbench_dir}\")\n print(f\"Work dir: {work_dir}\")\n if is_windows_mode:\n print(f\"Platform: Windows (litebox_runner_linux_on_windows)\")\n print(f\"Prepared dir: {prepared_dir}\")\n if run_litebox_mode:\n print(f\"Runner: {runner_path}\")\n if not is_windows_mode:\n print(f\"Packager: {packager_path or 'cargo run'}\")\n print(f\"Benchmarks: {', '.join(args.benchmarks)}\")\n if args.duration is not None:\n print(f\"Duration: {args.duration}s (override)\")\n else:\n print(f\"Duration: per-benchmark defaults\")\n if args.iterations is not None:\n print(f\"Iterations: {args.iterations} (override)\")\n else:\n print(f\"Iterations: per-benchmark defaults\")\n print(f\"Mode: {args.mode}\")\n print()\n\n # ── Run benchmarks ──────────────────────────────────────────────────\n\n all_results: dict[str, ComparisonRow] = {}\n\n for bench_name in args.benchmarks:\n bench = BENCHMARKS[bench_name]\n row = ComparisonRow(name=bench_name, unit=\"\")\n\n # Use per-benchmark defaults unless overridden by CLI"},{"key":"ci-b44f76e0-a87fbecf-0","score":21.142268792015216,"symbol":{"kind":"function_item","line_end":102,"line_start":77,"name":"main","path":"litebox_syscall_rewriter/src/main.rs"},"text":"litebox_syscall_rewriter/src/main.rs:77:102 main\nfn main() -> anyhow::Result<()> {\n let cli_args = CliArgs::parse();\n let mut input_binary = std::fs::File::open(&cli_args.input_binary)?;\n let mut input_binary_bytes = vec![];\n input_binary.read_to_end(&mut input_binary_bytes)?;\n let output_binary = litebox_syscall_rewriter::rewrite_binary_for_host(\n &input_binary_bytes,\n cli_args.trampoline_addr,\n cli_args.host.into(),\n )?;\n let output_path = cli_args.output_binary.unwrap_or_else(|| {\n cli_args.input_binary.with_file_name(\n cli_args\n .input_binary\n .file_name()\n .unwrap()\n .to_string_lossy()\n .into_owned()\n + \".hooked\",\n )\n });\n let mut file = std::fs::File::create(output_path)?;\n copy_file_permissions(&input_binary, &file)?;\n file.write_all(&output_binary)?;\n Ok(())\n}"}],"commits":[{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":44.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":25.0},{"hash":"624f0ceedf841b147c5e972104d7cfed355ccc90","message":"feat(net): guest web access without root via an in-stack HTTP proxy","score":22.0},{"hash":"b97641713732df7cb18e20ebbd80e675e777fba6","message":"feat(rfb): runner-side VNC server presenting the guest /dev/fb0 framebuffer","score":20.0},{"hash":"d06199534c3cf1090fdb261b8c800d380dea941e","message":"fix: getrusage/wait4 guest-stack overrun, doc gate, pipe-hang probes","score":17.0},{"hash":"366634828dae54bd2ff0c64e83f5d1cb9ade7c46","message":"fix(ci): clear clippy 1.98 lints, fmt drift, and vendored-crate no_std check","score":17.0},{"hash":"4a62d24fb8dba3be51e29bf89b38449d8bbd7d19","message":"feat(terminal): honor TCSETSW/TCSETSF, add per-guest IP override, npm 0.1.1","score":17.0},{"hash":"323d13318ea10946e74e99a7e24d593c9ef6e7ed","message":"feat(shim): the syscall surface an X11 desktop stack needs","score":16.0},{"hash":"5251e6a45413ae95629010a02e13a7959a9e3517","message":"feat(evdev): /dev/input/event* emulation wired to VNC input","score":16.0},{"hash":"dbc767aa1920e9a3c60c3602e6e899b4d37eeeb4","message":"Wire fchown via a new fd_chown, completing the chown family","score":14.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788374067269-500-a6592b57416eda6c","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"fe94973eb31db1a1","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-92521788373655553.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-92521788373655553.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-92901788215680536.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-92901788215680536.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-933841788170794871.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-933841788170794871.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-93971788295937679.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-93971788295937679.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-95501788295938803.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-95501788295938803.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-957511788294725676.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-957511788294725676.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-96141788215684034.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-96141788215684034.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-974901788215478850.json b/.agentplug/plugin-dispatch/out/gm-codesearch-974901788215478850.json new file mode 100644 index 0000000000..12ca81e6ae --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-codesearch-974901788215478850.json @@ -0,0 +1 @@ +{"data":{"bm25_hits":[{"key":"ci-bf0ce1c3-2c8c8b10-0","score":25.577487020316156,"symbol":{"kind":"function_item","line_end":880,"line_start":851,"name":"query_symbolic_link_requires_space_for_trailing_nul","path":"litebox_shim_windows/src/syscalls/symlink.rs"},"text":"litebox_shim_windows/src/syscalls/symlink.rs:851:880 query_symbolic_link_requires_space_for_trailing_nul\n fn query_symbolic_link_requires_space_for_trailing_nul() {\n run_with_test_platform_pointers(|| {\n let task = test_task();\n let target = r\"\\BaseNamedObjects\\ExactLengthTarget\";\n let handle = create_link(&task, r\"\\BaseNamedObjects\\LiteBoxExactSymlink\", target);\n let mut output = alloc::vec![0xeeeeu16; target.encode_utf16().count()];\n let mut target_string = UnicodeString {\n length: 0x1234,\n maximum_length: size_of_val(output.as_slice()).trunc(),\n padding_0: [0; 4],\n buffer: output.as_mut_ptr() as usize,\n };\n let mut returned_length = 0;\n\n assert_eq!(\n task.sys_nt_query_symbolic_link_object(\n handle,\n mut_ptr(&mut target_string),\n Some(mut_ptr(&mut returned_length)),\n ),\n NtStatus::BUFFER_TOO_SMALL\n );\n assert_eq!(\n returned_length,\n ((target.encode_utf16().count() + 1) * 2).trunc()\n );\n assert!(output.iter().all(|unit| *unit == 0xeeee));\n assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS);\n });\n }"},{"key":"ci-26120153-df996502-1","score":12.525145315586531,"symbol":{"kind":"function_item","line_end":1641,"line_start":1601,"name":"nt_query_performance_counter_duration_tracks_sleep_duration","path":"litebox_shim_windows/src/syscalls/sysinfo.rs"},"text":"litebox_shim_windows/src/syscalls/sysinfo.rs:1601:1641 nt_query_performance_counter_duration_tracks_sleep_duration\n fn nt_query_performance_counter_duration_tracks_sleep_duration() {\n run_with_test_platform_pointers(|| {\n let task = crate::tests::test_task();\n let mut guest_frequency = 0i64;\n let mut guest_start = 0i64;\n let mut guest_end = 0i64;\n\n let guest_start_status = task.sys_nt_query_performance_counter(\n mut_ptr(&mut guest_start),\n Some(mut_ptr(&mut guest_frequency)),\n );\n\n std::thread::sleep(QPC_SLEEP_DURATION);\n\n let guest_end_status = task.sys_nt_query_performance_counter(\n mut_ptr(&mut guest_end),\n Some(mut_ptr(&mut guest_frequency)),\n );\n\n assert_eq!(guest_start_status, NtStatus::SUCCESS);\n assert_eq!(guest_end_status, NtStatus::SUCCESS);\n assert_eq!(guest_frequency, QPC_FREQUENCY_HZ);\n\n let guest_duration_nanos = qpc_delta_nanos(guest_start, guest_end);\n let minimum_duration_nanos = QPC_SLEEP_DURATION\n .saturating_sub(QPC_SLEEP_TOLERANCE)\n .as_nanos();\n let maximum_duration_nanos = QPC_SLEEP_DURATION\n .saturating_add(QPC_SLEEP_TOLERANCE)\n .as_nanos();\n\n assert!(\n guest_duration_nanos >= minimum_duration_nanos,\n \"guest duration {guest_duration_nanos}ns was shorter than requested sleep minus tolerance {minimum_duration_nanos}ns\",\n );\n assert!(\n guest_duration_nanos <= maximum_duration_nanos,\n \"guest duration {guest_duration_nanos}ns was longer than requested sleep plus tolerance {maximum_duration_nanos}ns\",\n );\n });\n }"},{"key":"ci-7bd73c9-d6ae523a-0","score":12.18670762042432,"symbol":{"kind":"function_definition","line_end":736,"line_start":596,"name":"install_x18_gcc_runtimes","path":"litebox_packager/scripts/build-x18-desktop-repo.sh"},"text":"litebox_packager/scripts/build-x18-desktop-repo.sh:596:736 install_x18_gcc_runtimes\ninstall_x18_gcc_runtimes() {\n info \"Installing the verified x18-clean GCC runtime build inputs...\"\n # shellcheck disable=SC2016\n \"$CONTAINER_ENGINE\" exec \"$BUILD_CONTAINER\" sh -c '\n set -e\n origin=/root/verified-origins/gcc\n [ -f \"$origin/.completed\" ] && [ -f \"$origin/SHA256SUMS\" ] || {\n echo \"verified GCC origin is unavailable\" >&2\n exit 1\n }\n (cd \"$origin/artifacts\" && sha256sum -c ../SHA256SUMS > /dev/null)\n\n libgcc=\"$(find \"$origin/artifacts\" -type f \\\n -name \"libgcc-[0-9]*.apk\" -print)\"\n libgcc_static=\"$(find \"$origin/artifacts\" -type f \\\n -name \"libgcc-static-[0-9]*.apk\" -print)\"\n libstdcxx=\"$(find \"$origin/artifacts\" -type f \\\n -name \"libstdc++-[0-9]*.apk\" -print)\"\n libgomp=\"$(find \"$origin/artifacts\" -type f \\\n -name \"libgomp-[0-9]*.apk\" -print)\"\n [ -n \"$libgcc\" ] && [ \"$(printf \"%s\\n\" \"$libgcc\" | wc -l)\" -eq 1 ]\n [ -n \"$libgcc_static\" ] && \\\n [ \"$(printf \"%s\\n\" \"$libgcc_static\" | wc -l)\" -eq 1 ]\n [ -n \"$libstdcxx\" ] && [ \"$(printf \"%s\\n\" \"$libstdcxx\" | wc -l)\" -eq 1 ]\n [ -n \"$libgomp\" ] && [ \"$(printf \"%s\\n\" \"$libgomp\" | wc -l)\" -eq 1 ]\n [ \"$(wc -l < \"$origin/outputs\")\" -eq 4 ]\n apk verify \"$libgcc\" > /dev/null\n apk verify \"$libgcc_static\" > /dev/null\n apk verify \"$libstdcxx\" > /dev/null\n apk verify \"$libgomp\" > /dev/null\n\n plan=\"$(apk add --simulate --no-network \\\n --repositories-file /dev/null \\\n \"$libgcc\" \"$libgcc_static\" \"$libstdcxx\" \"$libgomp\" 2>&1)\"\n printf \"%s\\n\" \"$plan\"\n replacements=\"$(printf \"%s\\n\" \"$plan\" | grep -c \"Replacing \" || true)\"\n actions=\"$(printf \"%s\\n\" \"$plan\" | grep -Ec \\\n \"(^| )(Installing|Upgrading|Downgrading|Replacing|Purging) \" || true)\"\n case \"$replacements:$actions\" in\n 4:4|0:0) ;;\n *)\n echo \"unexpected GCC runtime installation plan\" >&2\n exit 1\n ;;\n esac\n\n world=/tmp/apk-world.before-x18-gcc-runtimes\n cp /etc/apk/world \"$world\"\n restore_world() {\n cp \"$world\" /etc/apk/world\n cmp \"$world\" /etc/apk/world\n rm -f \"$world\"\n }\n trap restore_world 0\n apk add --no-network --repositories-file /dev/null \\\n \"$libgcc\" \"$libgcc_static\" \"$libstdcxx\" \"$libgomp\" > /dev/null\n restore_world\n trap - 0\n\n installed_payload=/tmp/litebox-x18-gcc-installed-payload\n rm -rf \"$installed_payload\"\n mkdir \"$installed_payload\"\n for artifact in \"$libgcc\" \"$libgcc_static\" \"$libstdcxx\" \"$libgomp\"; do\n tar -xzf \"$artifact\" -C \"$installed_payload\" 2>/dev/null\n done\n payload_count=0\n while IFS= read -r payload; do\n relative=\"${payload#$installed_payload}\"\n case \"$relative\" in\n /.PKGINFO|/.SIGN.*) continue;;\n esac\n [ -f \"$relative\" ] && cmp -s \"$payload\" \"$relative\" || {\n echo \"installed GCC runtime payload mismatch: $relative\" >&2\n exit 1\n }\n payload_count=$((payload_count + 1))\n done <&2\n exit 1\n }\n done < \"$disassembly\" || return 1\n awk -F \"\\t\" \"NF >= 3 {\n ops = \\$3\n gsub(/\\\\[/, \\\" \\\", ops); gsub(/\\\\]/, \\\" \\\", ops)\n gsub(/[,{}!]/, \\\" \\\", ops)\n n = split(ops, a, /[[:space:]]+/)\n hit = 0\n for (i = 1; i <= n; i++)\n if (a[i] == \\\"x18\\\" || a[i] == \\\"w18\\\") hit = 1\n count += hit\n } END { print count + 0 }\" \"$disassembly\"\n rm -f \"$disassembly\"\n }\n\n archive_total=0\n archive_dir=\"$(dirname \"$(gcc -print-libgcc-file-name)\")\"\n for archive in \"$archive_dir\"/libgcc*.a; do\n [ -f \"$archive\" ] || continue\n n=\"$(count_x18_instructions \"$archive\")\" || exit 1\n if [ \"$n\" -gt 0 ]; then\n echo \"residual x18 instructions in installed compiler archive: $archive ($n)\" >&2\n archive_total=$((archive_total + n))\n fi\n done\n echo \"x18 instructions in installed libgcc archives: $archive_total\"\n [ \"$archive_total\" -eq 0 ]\n\n runtime_total=0\n runtime_count=0\n while IFS= read -r runtime; do\n runtime_count=$((runtime_count + 1))\n n=\"$(count_x18_instructions \"$runtime\")\" || exit 1\n if [ \"$n\" -gt 0 ]; then\n echo \"residual x18 instructions in installed GCC runtime: $runtime ($n)\" >&2\n runtime_total=$((runtime_total + n))\n fi\n done <(\n litebox: &LiteBox,\n in_mem_fs: litebox::fs::in_mem::FileSystem,\n tar_data: Cow<'static, [u8]>,\n) -> WindowsFS\nwhere\n Platform: ShimPlatform + CrngProvider + StdioProvider,\n{\n let devices = litebox::fs::resolver::Resolver::new(\n litebox,\n litebox::fs::composer::Composer::builder()\n .mount(\"/dev\", |allocator| {\n litebox::fs::devices::Devices::new(litebox, allocator)\n })\n .build()\n .unwrap(),\n );\n let tar_ro = litebox::fs::resolver::Resolver::new(\n litebox,\n litebox::fs::composer::Composer::builder()\n .mount(\"/\", |allocator| {\n litebox::fs::tar_ro::TarRo::new(tar_data, allocator)\n })\n .build()\n .unwrap(),\n );\n litebox::fs::layered::FileSystem::new(\n litebox,\n in_mem_fs,\n litebox::fs::layered::FileSystem::new(\n litebox,\n devices,\n tar_ro,\n litebox::fs::layered::LayeringSemantics::LowerLayerReadOnly,\n ),\n litebox::fs::layered::LayeringSemantics::LowerLayerWritableFiles,\n )\n}"},{"key":"ci-c5d8753d-d68df601-0","score":10.507319271514248,"symbol":{"kind":"section","line_end":160,"line_start":1,"name":"","path":"docs/benchmarks/awk-performance-investigation.md"},"text":"docs/benchmarks/awk-performance-investigation.md:1:160 \n# AWK performance and `vfork`/`time` investigation\n\nThis document records a performance and correctness investigation triggered by:\n\n```sh\n/usr/bin/time -p npx @openclew/litebox -- \\\n /bin/busybox awk 'BEGIN {a=0;b=1;for(i=0;i<100000000;i++){c=(a+b)%1000000007;a=b;b=c} print a}'\n```\n\nreporting `real 240.53` against `user 83.92` / `sys 1.39` (a wall-clock time\nroughly 2.8x the reported CPU time), and:\n\n```sh\nnpx @openclew/litebox -- /bin/busybox sh -c \\\n 'time /bin/busybox awk \"BEGIN {a=0;b=1;for(i=0;i<10000000;i++){c=(a+b)%1000000007;a=b;b=c} print a}\"'\n```\n\nfailing with `time: vfork: Invalid argument`.\n\nEnvironment for every measurement below: Apple M3 Pro, 11 cores, macOS\n26.3.1 (Darwin 25.3.0), built from a HEAD checkout of this repo (not the\n`npx @openclew/litebox` published package -- see \"npm package is stale\"\nbelow). This is a shared, actively-used development machine; several\nmeasurements below were taken under heavy, uncontrolled concurrent load from\nunrelated work (other terminal sessions building and testing this same\nrepo). Every number is labeled with the load average at the time it was\ntaken so it can be weighed accordingly -- this investigation treats ambient\ncontention as a variable to control for, not something to hide.\n\n## Summary of findings\n\n1. **The `vfork: Invalid argument` failure does not reproduce on current\n HEAD.** It was already fixed as a side effect of the prior\n delayed-address-space-handoff `fork`/`vfork` rework (commit `691fd87` and\n related), which predates this investigation. The `npx @openclew/litebox`\n package still fails because its pinned revision\n (`npm/lib/platform.js`'s `PINNED_REV`) is stale and predates that fix --\n see \"npm package is stale\" below.\n2. **A real, separate bug was found and fixed in this pass:** `wait4(...,\n &rusage)` left the caller's `rusage` buffer completely uninitialized\n whenever one was requested. Combined with (1) now succeeding, this is\n exactly what produced the user-visible symptom: `busybox time` prints\n whatever garbage was already in that guest memory, e.g. `sys\n 2367004162h 16m 32s`. Fixed by actually populating `ru_utime` from\n real, host-measured per-thread CPU time, and zeroing every other field.\n3. **The wall-clock-vs-CPU-time gap is real, not purely an accounting\n artifact, but it is highly sensitive to ambient system load** -- and this\n machine had extreme, uncontrolled load (verified up to ~18x\n oversubscription on an 11-core box) for parts of this investigation. A\n controlled, same-load A/B against native macOS `awk` shows litebox's own\n `user` CPU time is genuinely ~8x native's for the identical computation\n producing the identical result -- a real litebox-attributable CPU cost,\n independent of scheduling delay.\n4. **Root cause of that 8x, per `sample(1)` + live disassembly:** roughly\n half of all CPU samples during the hot loop land not in the guest's own\n code, but in a private JIT-allocated executable region holding\n AOT-rewritten guest code and PLT-style call stubs -- consistent with\n every arithmetic operation in the AWK script going through a real,\n dynamically-linked call into musl libc (`fmod`-shaped double modulo,\n allocation-shaped bit-twiddling) rather than being inlined. This is\n deferred as follow-up work (see \"Remaining overhead\" below) rather than\n attempted in this pass, because a fix would mean changing\n `litebox_syscall_rewriter`'s AOT rewriting itself, which needs more time\n to verify safely across the guest compatibility matrix than this pass\n had.\n\n## 1. The `vfork` failure: already fixed upstream of this investigation\n\nReproducing the exact original repro against a HEAD build:\n\n```sh\n$ target/release/litebox_runner_linux_on_macos_userland --initial-files alpine.tar -- \\\n /bin/busybox sh -c 'time /bin/busybox awk \"BEGIN {...}\"'\n490189494\nreal\t0m 12.01s\nuser\t0m 0.2741907030s\nsys\t2367004162h 16m 32s\n```\n\nIt no longer fails with `EINVAL` -- `vfork()`'s underlying\n`clone(CLONE_VM|CLONE_VFORK, ...)` is routed correctly by `do_clone` in\n`litebox_shim_linux/src/syscalls/process.rs` to `do_fork`, which already\nimplements the \"delayed address-space handoff\" model described in that\nfile's doc comments. That work landed before this investigation started.\n\nWhat's visibly broken instead is the *time* it prints, which is finding 2.\n\n### npm package is stale\n\n`npm/lib/platform.js`'s `PINNED_REV` is `497433858b0f8c52ea335df3576afb3e23e3a2e3`,\nwhich predates the fork/vfork rework entirely. Anyone running\n`npx @openclew/litebox` still gets the old `EINVAL` failure. This needs a\n`PINNED_REV` bump and republish to actually reach users -- tracked\nseparately, not done as part of this change (out of scope for a\ncorrectness/performance investigation; a version bump is its own\nreviewable, low-risk change).\n\n## 2. The `rusage` bug (fixed in this pass)\n\n### Root cause\n\n`Task::sys_wait4` in `litebox_shim_linux/src/syscalls/process.rs` used to\nhandle a non-null `rusage` pointer like this:\n\n```rust\nif rusage != 0 {\n // Reporting zeroed usage would be a lie that some callers act on; refusing is not,\n // and no caller in sight asks for it.\n log_unsupported!(\"wait4 with a rusage buffer\");\n}\n```\n\nIt logged and then did *nothing else* -- no error was returned, and the\nbuffer was never written. `wait4` reports success, and the guest's `struct\nrusage` is left exactly as it was before the call: whatever bytes happened\nto already be in that stack or heap allocation. `busybox time` reads\n`ru_utime`/`ru_stime` straight out of that memory and prints them, so the\nobserved `sys 2367004162h 16m 32s` is simply uninitialized memory\nreinterpreted as a `timeval`. This is also, independent of how silly the\noutput looks, an information-disclosure bug: guest code that requests\n`rusage` gets back bytes of guest memory it never wrote, with no relation\nto its own execution.\n\n### Fix\n\n- `litebox_common_linux::Rusage` -- a `#[repr(C)]` struct matching musl's\n LP64 `struct rusage` layout (`ru_utime`/`ru_stime` as the existing\n `TimeVal` type, then the fourteen POSIX `long` fields, then musl's\n 16-`long` reserved tail), written into guest memory the same way\n `Sysinfo`/`Statfs`/etc. already are.\n- `Process::cpu_time_nanos`, an `AtomicU64` accumulator. Each thread of a\n process adds its own `ShimPlatform::thread_cpu_time()` reading (real,\n host-measured, per-thread CPU time -- already used for\n `CLOCK_THREAD_CPUTIME_ID`, and already covered by an existing test that\n `thread_cpu_time` tracks real CPU usage, not wall-clock time) as it exits,\n in `Task::prepare_for_exit`. This has to happen on the exiting thread\n itself: `CLOCK_THREAD_CPUTIME_ID`-style clocks only ever read the calling\n thread's own counter.\n- `ProcessTable::record_exit`/`reap` now carry that accumulated value\n alongside the exit status.\n- `Task::sys_wait4` now writes a real `Rusage` value when a caller passes a\n non-null pointer: `ru_utime` is the process's real accumulated CPU time,\n every other field (including `ru_stime`) is explicitly zero rather than\n fabricated -- guest syscalls run as ordinary host user-mode Rust in this\n shim, so there is no meaningful \"kernel time\" of its own to attribute to\n `ru_stime`, and reporting a fabricated nonzero value would trade one lie\n for another. Zero, clearly labeled as \"unmeasured\" in the surrounding\n comment, is the honest answer.\n\n### After\n\n```sh\n$ target/release/litebox_runner_linux_on_macos_userland --initial-files alpine.tar -- \\\n /bin/busybox sh -c 'time /bin/busybox awk \"BEGIN {...10M iters...}\"'\n490189494\nreal\t0m 11.29s\nuser\t0m 7.18s\nsys\t0m 0.00s\n```\n"},{"key":"ci-daf8ce5d-4dec50f7-0","score":9.73838398676028,"symbol":{"kind":"function_item","line_end":1354,"line_start":1329,"name":"resolve_symlink_in_rootfs_edge_cases","path":"litebox_packager/src/oci.rs"},"text":"litebox_packager/src/oci.rs:1329:1354 resolve_symlink_in_rootfs_edge_cases\n fn resolve_symlink_in_rootfs_edge_cases() {\n let tmp = tempfile::tempdir().unwrap();\n let rootfs = tmp.path();\n std::fs::write(rootfs.join(\"hello.txt\"), b\"hi\").unwrap();\n\n // Cycle: a -> b -> a\n let mut cycle_map = HashMap::new();\n cycle_map.insert(PathBuf::from(\"a\"), PathBuf::from(\"b\"));\n cycle_map.insert(PathBuf::from(\"b\"), PathBuf::from(\"a\"));\n assert!(resolve_symlink_in_rootfs(Path::new(\"a\"), rootfs, &cycle_map, 32).is_none());\n\n let empty_map = HashMap::new();\n\n // Empty path\n assert!(resolve_symlink_in_rootfs(Path::new(\"\"), rootfs, &empty_map, 32).is_none());\n\n // Nonexistent path\n assert!(\n resolve_symlink_in_rootfs(Path::new(\"does/not/exist\"), rootfs, &empty_map, 32)\n .is_none()\n );\n\n // Regular file (not a symlink) returns host path directly\n let r = resolve_symlink_in_rootfs(Path::new(\"hello.txt\"), rootfs, &empty_map, 32);\n assert_eq!(r, Some(rootfs.join(\"hello.txt\")));\n }"},{"key":"ci-36cd1f98-af2dccfd-1","score":9.708242902571747,"symbol":{"kind":"function_item","line_end":326,"line_start":304,"name":"sandbox_tun_read_write","path":"litebox_runner_snp/src/main.rs"},"text":"litebox_runner_snp/src/main.rs:304:326 sandbox_tun_read_write\npub extern \"C\" fn sandbox_tun_read_write() {\n // wait until shim is initialized\n let shim = loop {\n if let Some(shim) = SHIM.get() {\n break shim;\n }\n core::hint::spin_loop();\n };\n #[cfg(debug_assertions)]\n litebox_util_log::debug!(\"sandbox_tun_read_write started\");\n while !litebox_platform_linux_kernel::host::snp::snp_impl::all_threads_exited() {\n let _timeout = loop {\n match shim\n .perform_network_interaction() {\n litebox::net::PlatformInteractionReinvocationAdvice::CallAgainImmediately => {},\n litebox::net::PlatformInteractionReinvocationAdvice::WaitOnDeviceOrSocketInteraction { timeout } => break timeout,\n }\n };\n // TODO: use timeout to wait on host events\n }\n\n litebox_platform_linux_kernel::host::snp::snp_impl::HostSnpInterface::return_to_host();\n}"},{"key":"ci-c0c7fddd-77e5aa1f-0","score":9.46387737453801,"symbol":{"kind":"function_item","line_end":618,"line_start":446,"name":"run","path":"litebox_runner_linux_on_macos_userland/src/lib.rs"},"text":"litebox_runner_linux_on_macos_userland/src/lib.rs:446:618 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let tar_file = &cli_args.initial_files;\n if tar_file.extension().and_then(|x| x.to_str()) != Some(\"tar\") {\n anyhow::bail!(\"Expected a .tar file, found {}\", tar_file.display());\n }\n let tar_data = std::fs::read(tar_file)\n .map_err(|e| anyhow!(\"Could not read tar file at {}: {}\", tar_file.display(), e))?;\n\n // `--vnc`/`--vnc-web` make the viewer keyboard a second stdin producer, so a\n // closed/redirected host stdin must not read as EOF to the guest (a console program\n // would exit on it).\n let platform = Platform::new_with_options(\n cli_args.tun_device_name.as_deref(),\n cli_args.vnc || cli_args.vnc_web.is_some(),\n );\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n // The program path is a Unix-style path inside the tar archive.\n let prog_path = &cli_args.program_and_arguments[0];\n\n let initial_file_system = {\n let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem.with_root_privileges(|fs| {\n use litebox::fs::FileSystem as _;\n fs.mkdir(\n \"/tmp\",\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|e| {\n panic!(\"/tmp creation cannot fail on a fresh in-memory file system: {e}\")\n });\n fs.chown(\"/tmp\", Some(1000), Some(1000))\n .unwrap_or_else(|e| {\n panic!(\"/tmp chown cannot fail on a fresh in-memory file system: {e}\")\n });\n\n // Standard FHS directories that guest tools expect to already exist (e.g. `apk`\n // opens a log file under `/var/log`) but that don't survive as empty-directory\n // entries when an OCI image's rootfs is scanned into a file-based tar: an empty\n // directory has no file contents, so it produces no tar entry, and `TarRo`'s\n // directory tree is inferred purely from file paths.\n for dir in [\"/run\", \"/var\", \"/var/log\", \"/var/cache\", \"/var/tmp\"] {\n fs.mkdir(\n dir,\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|_| {\n panic!(\"{dir} creation cannot fail on a fresh in-memory file system\")\n });\n }\n });\n\n if cli_args.guest_root {\n // Files the guest creates must be owned by the identity the guest runs as; see\n // `set_current_user`'s doc comment for the X/dbus failures a mismatch causes.\n in_mem.set_current_user(0, 0);\n }\n\n shim_builder.default_fs(in_mem, tar_data.into())\n };\n let initial_file_system = std::sync::Arc::new(initial_file_system);\n\n // Per-invocation network identity override (mirrors the Linux host\n // runner's fleet-hive patch): lets many concurrent litebox processes on\n // one host each own a distinct, independently-reachable address instead\n // of all defaulting to the same hardcoded 10.0.0.2/10.0.0.1. Unset =\n // identical to upstream behavior. CLI flags rather than env vars\n // (`--guest-ip`/`--gateway-ip`, `CliArgs`) — a `sudo`-invoked launch\n // under the default `env_reset` policy strips arbitrary env vars\n // (confirmed live: \"sudo: sorry, you are not allowed to set the\n // following environment variables\") but always passes argv through.\n let shim = shim_builder.build_with_net_config(cli_args.guest_ip, cli_args.gateway_ip);\n\n // Bind AND run the VNC server's whole accept loop before the Seatbelt sandbox below, which\n // denies every syscall not explicitly allowed -- and unlike a plain read/write on an\n // already-open fd (stdio, the `utun` device -- see the sandbox call's own comment below),\n // `accept()` on a listening socket is itself a mediated network operation with no `allow`\n // rule in this profile, so it cannot run post-sandbox. This mirrors the tar-file read and\n // `utun` device open above: every host resource the process will ever need must be acquired\n // -- and here, USED -- before `enable_seatbelt_sandbox()` runs. Concretely: spawn the whole\n // `RfbServer::run` (bind already happened in `RfbServer::bind`, accept-loop-and-serve\n // happens in the spawned thread) before the sandbox call; Seatbelt restrictions apply\n // process-wide and are inherited by every thread (per this crate's own seatbelt module doc\n // comment), so a thread spawned pre-sandbox keeps running exactly as before after the main\n // thread sandboxes itself.\n let input_handler = build_input_handler(shim.input_registry(), shim.framebuffer(), platform);\n let vnc_worker = if cli_args.vnc {\n let framebuffer = shim\n .framebuffer()\n .ok_or_else(|| anyhow!(\"--vnc requires a filesystem that mounts /dev/fb0\"))?;\n let bind_addr = cli_args\n .vnc_bind_all\n .then_some(std::net::IpAddr::V4(std::net::Ipv4Addr::UNSPECIFIED));\n let server = litebox_rfb::RfbServer::bind(\n bind_addr,\n cli_args.vnc_port,\n std::sync::Arc::new(FramebufferAdapter(framebuffer)),\n )\n .map_err(|e| {\n anyhow!(\n \"failed to bind VNC listener on port {}: {e}\\n\\\n (most often another runner instance is still running -- stop it or pick a \\\n different --vnc-port)\",\n cli_args.vnc_port\n )\n })?;\n litebox_util_log::info!(\n addr:% = server.local_addr().map_err(|e| anyhow!(\"{e}\"))?;\n \"vnc server listening\"\n );\n let shutdown_handle = server.shutdown_handle();\n let on_input = input_handler.clone();\n let worker = std::thread::spawn(move || {\n if let Err(e) = server.run(on_input) {\n litebox_util_log::warn!(error:% = e; \"vnc server stopped\");\n }\n });\n Some((worker, shutdown_handle))\n } else {\n None\n };\n\n // The browser-based viewer: identical lifecycle to the VNC server (bind + spawn before\n // the sandbox; Seatbelt denies post-sandbox `accept()`), identical input plumbing.\n if let Some(port) = cli_args.vnc_web {\n let framebuffer = shim\n .framebuffer()\n .ok_or_else(|| anyhow!(\"--vnc-web requires a filesystem that mounts /dev/fb0\"))?;\n let server = litebox_rfb::web::WebServer::bind(\n None,\n port,\n std::sync::Arc::new(FramebufferAdapter(framebuffer)),\n )\n .map_err(|e| {\n anyhow!(\n \"failed to bind the web viewer listener on port {port}: {e}\\n\\\n (most often another runner instance is still running -- stop it or pick a \\\n different --vnc-web port)\"\n )\n })?;\n litebox_util_log::info!(\n addr:% = server.local_addr().map_err(|e| anyhow!(\"{e}\"))?;\n \"web viewer listening -- open http://127.0.0.1 at this port\"\n );\n let on_input = input_handler.clone();\n std::thread::spawn(move || {\n if let Err(e) = server.run(on_input) {\n litebox_util_log::warn!(error:% = e; \"web viewer server stopped\");\n }\n });\n }\n\n // The guest-web-access bridge. Same lifecycle position and reasoning as the VNC server\n // above: the in-guest listener and the resolver snapshot (`/etc/resolv.conf` becomes\n // unreadable under the sandbox) must both exist before `enable_seatbelt_sandbox*` runs;\n // the widened profile then keeps the bridge's outbound `connect`s working after.\n if cli_args.net_proxy {\n let listener = shim\n .listen_in_"},{"key":"ci-ea2224bb-9e5d768c-0","score":9.399571538071775,"symbol":{"kind":"function_item","line_end":147,"line_start":53,"name":"run","path":"litebox_runner_linux_on_windows_userland/src/lib.rs"},"text":"litebox_runner_linux_on_windows_userland/src/lib.rs:53:147 run\npub fn run(cli_args: CliArgs) -> Result<()> {\n tracing_subscriber::fmt()\n .with_timer(tracing_subscriber::fmt::time::uptime())\n .with_level(true)\n .with_env_filter(\n tracing_subscriber::EnvFilter::builder()\n .with_env_var(\"LITEBOX_LOG\")\n .from_env_lossy(),\n )\n .init();\n\n let tar_file = &cli_args.initial_files;\n if tar_file.extension().and_then(|x| x.to_str()) != Some(\"tar\") {\n anyhow::bail!(\"Expected a .tar file, found {}\", tar_file.display());\n }\n let tar_data = std::fs::read(tar_file)\n .map_err(|e| anyhow!(\"Could not read tar file at {}: {}\", tar_file.display(), e))?;\n\n let platform = Platform::new();\n let shim_builder = litebox_shim_linux::LinuxShimBuilder::new(platform);\n let litebox = shim_builder.litebox();\n\n // The program path is a Unix-style path inside the tar archive.\n let prog_path = &cli_args.program_and_arguments[0];\n\n let initial_file_system = {\n let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox);\n in_mem.with_root_privileges(|fs| {\n use litebox::fs::FileSystem as _;\n fs.mkdir(\n \"/tmp\",\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap();\n fs.chown(\"/tmp\", Some(1000), Some(1000)).unwrap();\n\n // Standard FHS directories that guest tools expect to already exist (e.g. `apk`\n // opens a log file under `/var/log`) but that don't survive as empty-directory\n // entries when an OCI image's rootfs is scanned into a file-based tar: an empty\n // directory has no file contents, so it produces no tar entry, and `TarRo`'s\n // directory tree is inferred purely from file paths.\n for dir in [\"/run\", \"/var\", \"/var/log\", \"/var/cache\", \"/var/tmp\"] {\n fs.mkdir(\n dir,\n litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO,\n )\n .unwrap_or_else(|_| {\n panic!(\"{dir} creation cannot fail on a fresh in-memory file system\")\n });\n }\n });\n\n shim_builder.default_fs(in_mem, tar_data.into())\n };\n let initial_file_system = std::sync::Arc::new(initial_file_system);\n\n let shim = shim_builder.build();\n let argv = cli_args\n .program_and_arguments\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp: Vec<_> = cli_args\n .environment_variables\n .iter()\n .map(|x| std::ffi::CString::new(x.bytes().collect::>()).unwrap())\n .collect();\n let envp = if cli_args.forward_environment_variables {\n envp.into_iter()\n .chain(std::env::vars().map(|(k, v)| {\n std::ffi::CString::new(k.bytes().chain(*b\"=\").chain(v.bytes()).collect::>())\n .unwrap()\n }))\n .collect()\n } else {\n envp\n };\n\n let program = shim\n .load_program(\n initial_file_system,\n platform.init_task(),\n prog_path,\n argv,\n envp,\n )\n .unwrap();\n unsafe {\n litebox_platform_windows_userland::run_thread(\n program.entrypoints,\n &mut litebox_common_linux::PtRegs::default(),\n );\n }\n std::process::exit(program.process.wait())\n}"},{"key":"ci-2791005d-e47a3f17-0","score":9.152618858618515,"symbol":{"kind":"function_item","line_end":253,"line_start":199,"name":"root_kind_controls_relative_path_resolution","path":"litebox_shim_windows/src/syscalls/file_path.rs"},"text":"litebox_shim_windows/src/syscalls/file_path.rs:199:253 root_kind_controls_relative_path_resolution\n fn root_kind_controls_relative_path_resolution() {\n let task = crate::tests::test_task();\n let resolver = FilePathResolver::new(&task.process.object_manager);\n\n assert_eq!(\n resolver.resolve(\n FilePathRoot::Filesystem {\n path: \"/tmp/root\",\n is_directory: true,\n },\n r\"child\\file.txt\",\n ),\n Ok(FileTarget::Filesystem(String::from(\n \"/tmp/root/child/file.txt\"\n )))\n );\n assert_eq!(\n resolver.resolve(\n FilePathRoot::Filesystem {\n path: \"/tmp/root\",\n is_directory: true,\n },\n r\"C:\\Windows\\System32\\ntdll.dll\",\n ),\n Ok(FileTarget::Filesystem(String::from(\n \"/Windows/System32/ntdll.dll\"\n )))\n );\n assert_eq!(\n resolver.resolve(\n FilePathRoot::Filesystem {\n path: \"/tmp/root\",\n is_directory: true,\n },\n r\"MixedCase\\File.TXT\",\n ),\n Ok(FileTarget::Filesystem(String::from(\n \"/tmp/root/MixedCase/File.TXT\"\n )))\n );\n assert_eq!(\n resolver.resolve(\n FilePathRoot::Filesystem {\n path: \"/tmp/root.txt\",\n is_directory: false,\n },\n \"child.txt\",\n ),\n Err(NtStatus::NOT_A_DIRECTORY)\n );\n assert_eq!(\n resolver.resolve(FilePathRoot::Condrv(CondrvObject::Reference), r\"\\Connect\"),\n Ok(FileTarget::Condrv(CondrvObject::Connect))\n );\n }"}],"commits":[{"hash":"85045d2a1c10ac40cddc6fec88e18bbe1a39fc9d","message":"Add symbolic-link storage to the in-memory fs, behind the FileSystem trait","score":8.0},{"hash":"f0b6d7b6e2987927b17245eb27f74a5dd6e42499","message":"xfce","score":7.0},{"hash":"ea064557a45dc288201698de0ed46b9f3eb2c804","message":"fix(shim/net/xfce): DNS resolution, ping raw sockets, ifconfig struct size, xfce panel freeze","score":6.0},{"hash":"323d13318ea10946e74e99a7e24d593c9ef6e7ed","message":"feat(shim): the syscall surface an X11 desktop stack needs","score":6.0},{"hash":"876e41c093728c831e524c56c2decb4a784dbade","message":"Add an AF_NETLINK route socket, so os.networkInterfaces() works","score":6.0},{"hash":"c115726a900ad553f780a19eb6e868b75beadce3","message":"Wire symlink/symlinkat and follow trailing links, so require() resolves symlinks","score":6.0},{"hash":"481e70a176519d5cee1883f18211a448677c0e19","message":"fix(ci): unmask shim_windows unused imports, absorb alarm lateness","score":5.0},{"hash":"5abdf00dc91dd840102c7f3e7f9246d2571e5e0b","message":"fix(fbdev): cargo fmt + resolve rustdoc broken-link/bare-URL warnings","score":4.0},{"hash":"b854f87ceac04c00959531678d22917bd6a66a78","message":"shim_windows: fix CI clippy lints from toolchain 1.98","score":4.0},{"hash":"f37c4cb98f8b2776ec96dff847fe538db110e6a4","message":"Implement rename/renameat/renameat2, so fs.renameSync stops returning ENOSYS","score":4.0}],"mode":"dual","vector_hits":[]},"dispatch_id":"1788215903901-500-eaa892597f938ac6","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"1daa17f075c7f502","verb":"codesearch"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-974901788215478850.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-974901788215478850.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-99161788295942639.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-99161788295942639.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-99601788215687559.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-99601788215687559.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-codesearch-99971788199258294.json.ready b/.agentplug/plugin-dispatch/out/gm-codesearch-99971788199258294.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-1011788295817560.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-1011788295817560.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-109701788358236144.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-109701788358236144.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-127621788295977149.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-127621788295977149.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-14571788294809493.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-14571788294809493.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-169401788296040692.json b/.agentplug/plugin-dispatch/out/gm-exec_js-169401788296040692.json new file mode 100644 index 0000000000..3a7d7d11d8 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-exec_js-169401788296040692.json @@ -0,0 +1 @@ +{"data":"{\"decision_required\":\"this call hit its timeoutMs still running -- it was NOT killed, it is alive in the background task registry as task_id. Decide: `task-output {id}` to keep it running and poll progress/result later (the queue is already free, this worker returned immediately), or `task-stop {id}` to kill it now. It does not run forever unattended -- dispatch one of those two, do not leave it un-decided.\",\"elapsed_ms\":120000,\"in_progress\":true,\"ok\":true,\"task_id\":\"task-1a05ec050c4\",\"timed_out\":true}","dispatch_id":"1788296161040-500-757566362b598670","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"b478dda5773f4501","verb":"exec_js"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-169401788296040692.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-169401788296040692.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-171601788360602838.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-171601788360602838.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-18301788358028055.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-18301788358028055.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-203981788365562035.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-203981788365562035.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-208411788372141377.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-208411788372141377.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-213551788372150973.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-213551788372150973.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-219921788372164261.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-219921788372164261.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-22151788294817665.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-22151788294817665.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-226791788372175240.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-226791788372175240.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-231961788372184151.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-231961788372184151.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-236071788372195000.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-236071788372195000.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-23771788294818910.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-23771788294818910.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-23851788358038415.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-23851788358038415.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-241351788372205896.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-241351788372205896.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-246091788363268908.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-246091788363268908.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-264491788358570911.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-264491788358570911.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-273171788358589609.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-273171788358589609.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-28381788373535682.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-28381788373535682.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-285181788365731843.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-285181788365731843.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-286321788358613798.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-286321788358613798.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-28901788294826639.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-28901788294826639.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-30261788294829630.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-30261788294829630.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-32461788373543178.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-32461788373543178.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-33491788294835545.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-33491788294835545.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-336351788374067462.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-336351788374067462.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-338581788374067469.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-338581788374067469.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-34011788373543520.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-34011788373543520.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-34471788373543725.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-34471788373543725.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-34771788295850429.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-34771788295850429.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-357321788374081852.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-357321788374081852.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-358471788374082869.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-358471788374082869.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-365771788374094256.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-365771788374094256.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-367601788374096389.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-367601788374096389.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-36851788373546041.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-36851788373546041.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-370251788374100383.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-370251788374100383.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-373911788374104973.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-373911788374104973.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-376301788374108350.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-376301788374108350.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-390661788374115737.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-390661788374115737.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-392391788374117397.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-392391788374117397.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-395181788374125219.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-395181788374125219.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-396321788374125655.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-396321788374125655.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-396631788372514429.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-396631788372514429.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-403061788294039969.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-403061788294039969.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-440111788294093527.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-440111788294093527.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-449891788294105358.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-449891788294105358.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-460691788294122182.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-460691788294122182.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-461911788294122654.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-461911788294122654.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-46381788373568664.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-46381788373568664.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-468891788294132993.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-468891788294132993.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-46961788294857949.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-46961788294857949.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-472151788294135975.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-472151788294135975.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-47351788373568835.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-47351788373568835.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-475021788294141834.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-475021788294141834.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-479901788294148589.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-479901788294148589.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-48021788373570258.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-48021788373570258.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-481661788294150870.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-481661788294150870.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-48461788373571345.json b/.agentplug/plugin-dispatch/out/gm-exec_js-48461788373571345.json new file mode 100644 index 0000000000..5ef8b0dc9c --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-exec_js-48461788373571345.json @@ -0,0 +1 @@ +{"data":"{\"decision_required\":\"this call hit its timeoutMs still running -- it was NOT killed, it is alive in the background task registry as task_id. Decide: `task-output {id}` to keep it running and poll progress/result later (the queue is already free, this worker returned immediately), or `task-stop {id}` to kill it now. It does not run forever unattended -- dispatch one of those two, do not leave it un-decided.\",\"elapsed_ms\":30001,\"in_progress\":true,\"ok\":true,\"task_id\":\"task-1a0635f956e\",\"timed_out\":true}","dispatch_id":"1788373628960-500-dd56bbd19eac5cd9","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"4d0a37786058ed44","verb":"exec_js"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-48461788373571345.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-48461788373571345.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-486821788294158089.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-486821788294158089.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-488061788294159309.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-488061788294159309.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-491231788294165421.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-491231788294165421.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-505561788294186948.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-505561788294186948.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-511011788294195277.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-511011788294195277.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-512791788294197938.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-512791788294197938.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-545371788296590186.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-545371788296590186.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-553301788311653692.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-553301788311653692.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-59021788294876676.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-59021788294876676.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-604201788295267748.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-604201788295267748.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-605181788361648840.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-605181788361648840.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-60821788360383810.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-60821788360383810.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-631811788295302417.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-631811788295302417.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-63361788294889053.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-63361788294889053.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-636831788295307144.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-636831788295307144.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-64541788373605139.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-64541788373605139.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-66981788373611090.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-66981788373611090.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-682441788372947095.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-682441788372947095.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-685221788372947266.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-685221788372947266.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-685251788372947177.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-685251788372947177.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-70601788373618254.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-70601788373618254.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-712921788357378172.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-712921788357378172.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-724241788357402088.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-724241788357402088.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-73631788294907500.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-73631788294907500.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-747421788357454801.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-747421788357454801.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-747471788374731215.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-747471788374731215.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-781681788357523624.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-781681788357523624.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-78221788373637851.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-78221788373637851.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-786341788364580789.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-786341788364580789.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-79311788373638706.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-79311788373638706.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-80071788294915761.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-80071788294915761.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-8061788294797306.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-8061788294797306.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-81041788373642514.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-81041788373642514.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-812731788374888418.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-812731788374888418.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-826911788364676470.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-826911788364676470.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-82691788294918927.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-82691788294918927.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-82921788373643775.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-82921788373643775.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-830231788364684070.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-830231788364684070.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-830411788374898769.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-830411788374898769.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-833901788374910981.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-833901788374910981.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-834881788357633141.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-834881788357633141.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-839821788374922030.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-839821788374922030.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-840991788357650762.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-840991788357650762.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-852811788364743185.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-852811788364743185.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-852861788295623428.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-852861788295623428.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-85501788373645178.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-85501788373645178.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-864561788357706636.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-864561788357706636.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-867531788357715910.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-867531788357715910.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-871851788357726124.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-871851788357726124.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-880111788295636334.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-880111788295636334.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-88281788373649799.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-88281788373649799.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-884231788357759577.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-884231788357759577.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-889981788294650278.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-889981788294650278.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-890361788295659203.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-890361788295659203.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-891381788357769017.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-891381788357769017.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-894611788294658668.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-894611788294658668.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-89661788373653260.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-89661788373653260.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-896911788357778308.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-896911788357778308.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-904301788295676962.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-904301788295676962.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-909961788357789866.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-909961788357789866.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-915651788357801643.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-915651788357801643.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-921281788357811616.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-921281788357811616.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-923031788294687420.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-923031788294687420.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-930571788294698083.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-930571788294698083.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-931621788357838920.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-931621788357838920.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-931661788373381229.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-931661788373381229.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-941471788360190958.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-941471788360190958.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-953641788373397561.json b/.agentplug/plugin-dispatch/out/gm-exec_js-953641788373397561.json new file mode 100644 index 0000000000..139901ba79 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-exec_js-953641788373397561.json @@ -0,0 +1 @@ +{"data":"{\"decision_required\":\"this call hit its timeoutMs still running -- it was NOT killed, it is alive in the background task registry as task_id. Decide: `task-output {id}` to keep it running and poll progress/result later (the queue is already free, this worker returned immediately), or `task-stop {id}` to kill it now. It does not run forever unattended -- dispatch one of those two, do not leave it un-decided.\",\"elapsed_ms\":60002,\"in_progress\":true,\"ok\":true,\"task_id\":\"task-1a0635d0424\",\"timed_out\":true}","dispatch_id":"1788373457647-500-af7364f7fd1fca4f","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"6d6f66f349500994","verb":"exec_js"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-953641788373397561.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-953641788373397561.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-955511788294724923.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-955511788294724923.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-961261788294731824.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-961261788294731824.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-969061788294740812.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-969061788294740812.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-976711788294752368.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-976711788294752368.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-977971788357935729.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-977971788357935729.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-9781788294798328.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-9781788294798328.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-981621788294756937.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-981621788294756937.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-984691788373461013.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-984691788373461013.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-986271788294762271.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-986271788294762271.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-991011788373471327.json b/.agentplug/plugin-dispatch/out/gm-exec_js-991011788373471327.json new file mode 100644 index 0000000000..04eacd7714 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-exec_js-991011788373471327.json @@ -0,0 +1 @@ +{"data":"{\"decision_required\":\"this call hit its timeoutMs still running -- it was NOT killed, it is alive in the background task registry as task_id. Decide: `task-output {id}` to keep it running and poll progress/result later (the queue is already free, this worker returned immediately), or `task-stop {id}` to kill it now. It does not run forever unattended -- dispatch one of those two, do not leave it un-decided.\",\"elapsed_ms\":60001,\"in_progress\":true,\"ok\":true,\"task_id\":\"task-1a0635c122f\",\"timed_out\":true}","dispatch_id":"1788373531638-500-1f74338cf065fb96","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"b1a9d6fa030d3af0","verb":"exec_js"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-991011788373471327.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-991011788373471327.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-exec_js-994961788294777942.json.ready b/.agentplug/plugin-dispatch/out/gm-exec_js-994961788294777942.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-785951788254770610.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-785951788254770610.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-803201788254776863.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-803201788254776863.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-804301788254777393.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-804301788254777393.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-804631788254777926.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-804631788254777926.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-805071788254778360.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-805071788254778360.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-805411788254778789.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-805411788254778789.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-805701788254778901.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-805701788254778901.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-805981788254779019.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-805981788254779019.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-806721788254779464.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-806721788254779464.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-fetch-835221788251924200.json.ready b/.agentplug/plugin-dispatch/out/gm-fetch-835221788251924200.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-git_diff-646801788295323194.json b/.agentplug/plugin-dispatch/out/gm-git_diff-646801788295323194.json new file mode 100644 index 0000000000..8eeb5c2ba6 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-git_diff-646801788295323194.json @@ -0,0 +1 @@ +{"dispatch_id":"1788295623306-500-805137f55136ad15","error":"git [\"diff\", \"--no-color\"] timed out after 300000ms, killed","error_code":"failed","hint":"git rejected the range; an empty diff must never be inferred from a rejected argument -- check both endpoints exist locally","next_dispatch_hint":"instruction","ok":false,"range":"","request_fingerprint":"5e376e3a7e351171","verb":"git_diff"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-git_diff-646801788295323194.json.ready b/.agentplug/plugin-dispatch/out/gm-git_diff-646801788295323194.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-git_show-606271788143254607.json b/.agentplug/plugin-dispatch/out/gm-git_show-606271788143254607.json new file mode 100644 index 0000000000..3425495955 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-git_show-606271788143254607.json @@ -0,0 +1 @@ +{"data":{"output":"commit 88d7a152430aa44b120c410cc6852904b844df77\nAuthor: dylanwongtencent \nDate: Sun Aug 30 18:55:25 2026 -0700\n\n diag(xfce): add heartbeat proving xfce4-panel, not Xorg/xfwm4, is the freeze culprit\n \n Live-verified this session, decisively narrowing the intermittent XFCE\n freeze investigated across many prior commits: built an instrumented test\n tar with a background heartbeat loop logging \"xset q: OK/FAILED\" to the\n runner's own stdout every 5s, then drove a real freeze on it. Result: xset\n q (an Xorg core protocol round trip) kept succeeding every single cycle\n for 3.5+ minutes straight while the panel clock stayed frozen and every\n xfce4-panel widget (Applications menu, Show Desktop, taskbar icons) was\n completely unresponsive to clicks. A titlebar drag -- an xfwm4-owned\n operation -- worked perfectly during the same freeze window.\n \n This rules out Xorg and xfwm4 as the wedged party and isolates the bug to\n xfce4-panel specifically. Combined with this session's other live\n findings (no litebox host thread ever caught spinning or with an\n unreturned syscall across three independent lldb/trace capture rounds; a\n sentinel-register test refuting XNU X16 corruption; a clean re-read of\n litebox_shim_linux's unix-socket/epoll code finding no staleness bug),\n the freeze is conclusively a guest-side application deadlock inside\n xfce4-panel itself -- not a litebox bug, not an Xorg bug, not an xfwm4\n bug.\n \n Keeps the heartbeat as a standing, low-cost diagnostic (purely\n observational, never calls fail, ~12 bytes/5s of log growth) so this\n confirmation is available for free from any future runner log instead of\n needing a bespoke instrumented rebuild each time. xfce4-panel's own\n startup warning (\"liblauncher-CRITICAL: Failed to start file monitor:\n Unable to find default local file monitor type\", from inotify being\n entirely unimplemented in this guest) is the leading remaining suspect\n for what eventually wedges it, but that mechanism is not yet confirmed --\n next step for whoever continues this is tracing xfce4-panel's own GLib/\n GIO/GDBus codepaths specifically, now that the search is scoped to one\n process instead of the whole desktop stack.\n \n Co-Authored-By: Claude Sonnet 5 \n Claude-Session: https://claude.ai/code/session_01NHnUPKUowhVamZfWgokv7E\n\ndiff --git a/litebox_packager/examples/xfce/start-desktop.sh b/litebox_packager/examples/xfce/start-desktop.sh\nindex 444d0af..867b1bf 100755\n--- a/litebox_packager/examples/xfce/start-desktop.sh\n+++ b/litebox_packager/examples/xfce/start-desktop.sh\n@@ -199,6 +199,30 @@ print_log thunar.log \"$THUNAR_LOG\"\n print_log xterm.log \"$XTERM_LOG\"\n printf '%s\\n' \"DESKTOP UP\"\n \n+# Diagnostic-only heartbeat (never calls fail -- purely observational): a\n+# host-visible timestamped record of whether Xorg's own core protocol round\n+# trip (the same xset q the health-check loop below already gates on) keeps\n+# succeeding through a freeze. CONFIRMED LIVE (2026-08-30/31): during an\n+# actual panel-click freeze (clock static, Applications menu and taskbar\n+# buttons both dead), this heartbeat kept logging \"xset q: OK\" every 5s for\n+# 3.5+ minutes straight, and a titlebar drag (an xfwm4-owned operation) still\n+# worked during the same freeze -- Xorg and xfwm4 are NOT the wedged party.\n+# Only xfce4-panel-owned widgets (Applications menu, Show Desktop, taskbar\n+# icons) stopped responding. Kept as a standing diagnostic so any future\n+# freeze investigation gets this confirmation for free from the ordinary\n+# runner log, instead of needing a bespoke instrumented rebuild each time.\n+(\n+ while :; do\n+ sleep 5\n+ if xset q >/dev/null 2>&1; then\n+ printf '%s heartbeat: xset q OK\\n' \"$(date -u +%H:%M:%S)\"\n+ else\n+ printf '%s heartbeat: xset q FAILED\\n' \"$(date -u +%H:%M:%S)\"\n+ fi\n+ done\n+) &\n+heartbeat_pid=$!\n+\n while :; do\n sleep 5\n require_alive Xorg \"$xorg_pid\" \"$XORG_LOG\"\n"},"dispatch_id":"1788151689005-500-c811d5ceabda5b7","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"d7fec05ddfb47174","verb":"git_show"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-git_show-606271788143254607.json.ready b/.agentplug/plugin-dispatch/out/gm-git_show-606271788143254607.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-git_status-162661788296031999.json.ready b/.agentplug/plugin-dispatch/out/gm-git_status-162661788296031999.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-git_status-573631788361598376.json.ready b/.agentplug/plugin-dispatch/out/gm-git_status-573631788361598376.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-git_status-638861788295311873.json.ready b/.agentplug/plugin-dispatch/out/gm-git_status-638861788295311873.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-health-958711788290670773.json b/.agentplug/plugin-dispatch/out/gm-health-958711788290670773.json new file mode 100644 index 0000000000..4363832e01 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-health-958711788290670773.json @@ -0,0 +1 @@ +{"data":{"crate_version":"0.1.1281","effective_config":{"budget":{"default_k":10,"default_limit":8,"pool_floor":20,"pool_multiplier":5},"bulk_embed":{"flat_json_migration_budget_ms":2000,"git_commit_embed_budget_ms":30000},"db_path":{"db_filename":"gm.db","resolved":"/Users/dylanwong/litebox/.gm/gm.db","state_root_dir":".gm"},"embed_cache":{"plain_cache_max_text_bytes":4096,"query_cache_capacity":64,"query_cache_ttl_ms":600000},"embed_dim":384,"guards":{"conformance":{"conformant":true,"declared_verbs":12,"note":"A non-empty `undispatchable` means the capability table names a verb dispatch cannot reach -- the published guard surface would be describing protection over nothing.","undispatchable":[]},"env_get":{"allowed_exact":["CLAUDE_PROJECT_DIR","GITHUB_TOKEN","GH_TOKEN"],"allowed_prefixes":["PLUGKIT_","GM_"],"applied_to":["env_get"]},"kv_put":{"allowed_namespaces":["default","session","config","cache","user"],"applied_to":["kv_put"]},"path_within_project":{"applied_to":["fs_read","fs_write","fs_stat","fs_readdir","scan_deps"],"rejects":["any .. segment","absolute paths","paths containing a drive colon"]},"unguarded":{"note":"These reach the network, a Node process, and a real browser with no allowlist of their own. That is deliberate -- they exist to run arbitrary caller-supplied work -- but it means the trust boundary for them is the CALLER, not this layer.","verbs":["fetch","exec_js","serp","browser","cdp"]}},"index":{"digest_max_files":2000,"max_chunks_per_file_per_pass":64,"max_file_bytes":262144,"pessimistic_ms_per_chunk":16000,"prune_enumeration_file_cap":20000,"prune_pass_file_limit_ceiling":2000,"prune_pass_file_limit_floor":50,"split_chunk_above_bytes":8192,"wall_budget_ms":45000},"memory_sync":{"embed_budget_ms":1500,"rekey_batch_max":25,"rekey_rows_deadline_ms":1800,"rename_batch_chunk":60,"shadow_abort_threshold":5,"total_budget_ms":2000},"persisted_paths":{"note":"these paths are a compatibility surface -- a reader outside this crate depends on each staying where it is","paths":[{"exists":true,"path":".gm/turn-state.json"},{"exists":true,"path":".gm/prd.yml"},{"exists":true,"path":".gm/mutables.yml"},{"exists":false,"path":".gm/residual-check-fired"},{"exists":true,"path":".gm/claim-audit-fired"},{"exists":true,"path":".gm/last-instruction-ts"},{"exists":true,"path":".gm/exec-spool/.ci-validated"},{"exists":false,"path":".gm/exec-spool/.turn-browser-edits.json"},{"exists":false,"path":".gm/exec-spool/.turn-browser-witnessed"},{"exists":true,"path":".gm/exec-spool/.gate-deviation-repeats.json"}]},"pipeline":{"summarize_input_char_cap":8192,"summarize_max_summary_chars":800,"summarize_preserve":["entities","numbers","ids"],"summarize_target_chars":400,"summarize_threshold":2048,"ttl_ms":120000},"rejected_tiers":["/Users/dylanwong/.gm/config.source.json: every source in this tier failed to load (1 entries): /Users/dylanwong/.gm/config.source.json: config repo https://github.com/dywongcloud/gm refreshed, but no config file at /Users/dylanwong/.gm/config-source-cache/bcbf16fb5ef31124/.gm"],"scoring":{"bm25_b":0.75,"bm25_k1":1.2,"cos_floor":0.0,"dedup_jaccard":0.7,"fusion_identifier_boost":2.0,"fusion_rrf_k":60.0,"half_life_ms":2592000000.0,"recency_floor":0.4},"shared_store_contract":{"aggregate_hazard":"an UNFILTERED aggregate over an F32_BLOB vector table answers 0 even when the table is full. Count with a predicate, or with SUM over a GROUP BY subquery, or against the _vec_shadow companion. A bare COUNT(*) reporting 0 is not evidence of an empty store.","busy_timeout_ceiling_rule":"must stay well under the host dispatch deadline (120s). A wait that outlives the epoch budget TRAPS the instance instead of returning a reportable SQLITE_BUSY, and a trap carries no error text -- contention then looks like a crash.","busy_timeout_ms":20000,"identity":"the resolved file path IS the identity -- libsql routes by path alone and ignores the `db` handle name for any real file (that field is consulted only for :memory:), so two callers naming the same file share one store and two projects with different files cannot collide","isolation":"none beyond the path. There is no per-plugin namespace and no ownership claim; any plugin handed the path reaches the whole store, including tables another plugin created.","journal_mode":"WAL where the conversion succeeds. It needs an exclusive lock, so it is attempted once per process per path and memoized only on a verified read-back -- PRAGMA journal_mode=WAL returns OK while silently doing nothing when it cannot take the lock.","locking":"the WASI VFS has no OS file locking and serializes with a .lock DIRECTORY. It is created before a write and removed after, so an unclean exit leaves it behind and every later write returns SQLITE_BUSY forever -- no holder to find, no timeout that expires it, survives reboots. A locked error now names the directory and the remedy."},"tables":{"code_chunks":"code_chunks","git_commits":"git_commit_vectors","memory_md_files":"memories_md_files","memory_md_meta":"memories_md_meta","rssearch":"rssearch_vectors"},"tier":"implicit_default_repo","version":1,"why":"gm's own shared default config repo at https://github.com/AnEntrypoint/gm-config (no project or user config.source.json configured)"},"error_codes":["failed","retired_verb","unsupported_by_design","unknown_verb","invalid_args","panic","gate_denied"],"imports":["host_cwd","host_fs_allow_root","host_fs_read","host_fs_write","host_fs_cas_write","host_fs_remove","host_fs_readdir","host_fs_stat","host_fetch","host_kv_get","host_kv_put","host_kv_delete","host_kv_query","host_vec_search","host_vec_embed","host_exec_js","host_log","host_now_ms","host_env_get","host_random_fill","host_browser_exec","host_oxi_exec","host_task_proc","host_git","host_plugin_call"],"imports_count":25,"lifecycle_liveness":{"last_event":"config_resolved","last_event_age_ms":57,"stale":false},"loaded_module_is_compiled_version_above_not_project_pin":true,"now":1788290671051,"ok":false,"plugin_failure_codes":["unknown_plugin","plugin_not_loaded_yet","deadline_exceeded","malformed_response","host_returned_empty","plugin_error","plugin_response_lost"],"plugin_response_envelope":{"failure":"{ok: false, error: string}; a missing `error` on a failed response is classified by plugin_failure_code","hazards":{"unfiltered_aggregate_on_vector_table":"DRIVER-SPECIFIC, measured under node:sqlite and NOT reproducing through the agentplug-libsql wasm plugin (2026-08-02: bare COUNT(*) on rssearch_vectors answers 278 there, matching the predicated form exactly). Under node:sqlite an UNFILTERED aggregate over a libsql F32_BLOB vector table answers 0 even when the table is full. Measured on this repo's store: `SELECT COUNT(*) FROM rssearch_vectors` returns 0, and the subquery form without a WHERE also returns 0 -- but the same aggregate WITH any predicate returns 755, reconciling exactly against 428 live plus 327 tombstoned rows. The missing predicate is the trigger, not the aggregate. The query verbs pass SQL through untouched, so nothing stops a caller reading that 0 as an empty knowledgebase and acting on it (dropping the table, re-indexing, reporting data loss). Always count with a predicate, or count the
_vec_shadow companion. PRAGMA integrity_check is separately unreliable on these tables -- it fails with `unknown function: libsql_vector_idx()`, which is not corruption."},"payload_field_by_verb":{"bert.embed":"embedding (plus `dim`); takes {text, kind} where kind=\"query\" selects the query conditioning, anything else is treated as a passage","bert.embed_batch":"embeddings (array, one per input); takes {texts}","libsql.list_dbs":"dbs (always empty: connections are no longer tracked by name)","libsql.open/close/begin/commit/rollback":"ACCEPTED BUT INERT -- every exec/query is its own open-operate-close cycle, so these are no-ops kept only so existing callers do not break. A caller must not assume begin/commit gives it a transaction.","libsql.query":"rows","libsql.query_params":"rows","libsql.serialize":"bytes_b64 (also mirrored as `data` for older callers; `bytes_b64` is the contract field)","treesitter.extract_chunks":"chunks (plus `lang`)","treesitter.lang_for_ext":"lang (null when the extension is unrecognized)","treesitter.parse":"nodes (plus `lang`, null when neither ext nor lang resolves -- an unresolved language is ok:true with an empty nodes array, NOT an error)"},"shape":"flat: {ok: bool, : ...} -- NOT wrapped under `data` the way a gm verb response is"},"project_gm_json_pinned_version":null,"source_sha":"cab7985b1af9101ceefe3aab6c04b8d2c47fab7d","subsystem_probes":{"codesearch":{"error":null,"ok":true},"recall":{"error":"semantic retrieval unavailable (embedder failed: exec rc=5 ext=5 msg=database is locked -- if no process holds this db, check for a stale dot-lock directory at /Users/dylanwong/litebox/.gm/gm.db.lock and remove it) and the keyword fallback matched nothing -- this is NOT an empty-knowledgebase result","ok":false}},"subsystems":[{"subsystem":"fs","verbs":["fs_read","fs_write","fs_readdir","fs_stat","fetch","env_get","kv_get","kv_put","kv_query"]},{"subsystem":"git","verbs":["git_status","branch_status","git_push","git_add","git_commit","git_finalize","git_log","git_diff","git_show","git_fetch","git_branch","git_checkout","git_merge","git_merge_abort","git_branch_delete","git_rm","git_revert","git_reset"]},{"subsystem":"sql","verbs":["sql_open","sql_close","sql_list_dbs","sql_exec","sql_query","sql_smoke","sql_serialize","sql_deserialize"]},{"subsystem":"memory","verbs":["memorize","memorize-prune","recall","codeinsight_index","codesearch","forget","discipline"]},{"subsystem":"exec","verbs":["exec_js","lang","python","bash","powershell","ssh","go","rust","c","cpp","java","deno"]},{"subsystem":"browser","verbs":["browser","cdp"]},{"subsystem":"orchestrator","verbs":["transition","transition-revert","mutable-resolve","mutable-add","mutable-list","mutable-defer","memorize-fire","discipline-note","discipline-check-removal","discipline-audit","capability-resolve","memory-namespace-audit","codeinsight-namespace-audit","calculus-model-check","phase-status","residual-scan","auto-recall","instruction","prd-add","prd-resolve","prd-list","prd-defer","task-spawn","task-list","task-stop","task-output","memorize-continue","fsm-vendor","fsm-validate","fsm-propose-override","claim-audit","submodule-check","component-loader-reconcile","component-loader-hmr"]},{"subsystem":"meta","verbs":["health","config_resolve","dataflow_resolve","status","close","filter","cache_get","cache_put","cache_invalidate","cache_stats","learn"]}],"verb_aliases":[{"alias":"nodejs","canonical":"exec_js","lang_preserved":true},{"alias":"javascript","canonical":"exec_js","lang_preserved":true},{"alias":"node","canonical":"exec_js","lang_preserved":true},{"alias":"js","canonical":"exec_js","lang_preserved":true},{"alias":"memorize_prune","canonical":"memorize-prune","lang_preserved":true},{"alias":"py","canonical":"python","lang_preserved":false},{"alias":"sh","canonical":"bash","lang_preserved":false},{"alias":"shell","canonical":"bash","lang_preserved":false},{"alias":"zsh","canonical":"bash","lang_preserved":false},{"alias":"ps1","canonical":"powershell","lang_preserved":false}],"version":"0.1.1281"},"dispatch_id":"1788291378667-500-ab5b97fcb27f780","next_dispatch_hint":"instruction","ok":true,"request_fingerprint":"76047f156cb2a5c5","verb":"health"} \ No newline at end of file diff --git a/.agentplug/plugin-dispatch/out/gm-health-958711788290670773.json.ready b/.agentplug/plugin-dispatch/out/gm-health-958711788290670773.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-instruction-100001788360455682.json.ready b/.agentplug/plugin-dispatch/out/gm-instruction-100001788360455682.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-instruction-137081788360516720.json.ready b/.agentplug/plugin-dispatch/out/gm-instruction-137081788360516720.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-instruction-160501788296028273.json.ready b/.agentplug/plugin-dispatch/out/gm-instruction-160501788296028273.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-instruction-178311788360608315.json.ready b/.agentplug/plugin-dispatch/out/gm-instruction-178311788360608315.json.ready new file mode 100644 index 0000000000..e69de29bb2 diff --git a/.agentplug/plugin-dispatch/out/gm-instruction-230751788367919461.json b/.agentplug/plugin-dispatch/out/gm-instruction-230751788367919461.json new file mode 100644 index 0000000000..4d522744a9 --- /dev/null +++ b/.agentplug/plugin-dispatch/out/gm-instruction-230751788367919461.json @@ -0,0 +1 @@ +{"body_parse_error":true,"data":{"await_result":null,"codeinsight_overview":{"by_kind":[],"counts_error":"prepare rc=5 msg=database is locked (vm_steps=0)","counts_unavailable":true,"digest":"v3:9726956689684305:files=480","file_count":0,"largest_files":[],"likely_orphaned":[],"symbol_count":0},"config_changed":[],"config_repo_unreachable":null,"discipline_policies":[],"epistemic_gap":0,"fsm_gates_weaker_than_default":[],"fsm_graph_rejected":null,"gm_plugkit_stale":null,"instruction":"# ORCHESTRATOR\n\nYOU are the state machine. Plugkit: synchronous lib serving this prose; advance = your dispatch, not its action. Holds phase/PRD/mutables on disk -- read via `phase-status`/`instruction`, change via the relevant verb. Nothing advances while you wait.\n\nYour authorization = the request. Your receipt = the PRD you write. Trajectory SPECIFY -> PROVE -> EMIT -> STATE -> CONC -> SEC -> RES -> DECIDE -> COMPLETE, each transition a verb you dispatch. The graph is NOT linear: feedback edges route every later stage's discoveries back -- PROVE/EMIT/STATE/CONC/SEC/RES/DECIDE can each return to SPECIFY (reshaping), STATE/CONC/SEC/RES return to EMIT (repair), CONC and SEC return to STATE (boundary enforcement), DECIDE returns to SPECIFY, PROVE, STATE, CONC, SEC, or RES (empirical fitness feedback, routed to whichever phase owns the failing obligation's kind). Stage ownership: SPECIFY = alignment/research/PRD density; PROVE = typed dependency-DAG proof obligations (precondition/invariant/postcondition/resource-bound/type-shape), gated by mutables-all-resolved + mutables-all-typed; EMIT = AST/source emission, gated by no-synthetic-test-files + no-graphical-symbols-in-diff + no-admit-deferral-markers; STATE = typed totality/ownership/replay/effect-boundary obligations, gated by idempotent-dispatch-replay-safe + state-obligations-ready; CONC = typed happens-before/disjointness/contention obligations, gated by conc-obligations-ready; SEC = typed secrets/injection/identity-authority/message-timing obligations, gated by no-secrets-in-diff + sec-obligations-ready; RES = typed exception-model/partial-failure/degradation/crucible obligations, gated by no-unchecked-panics-in-diff + res-obligations-ready; DECIDE = adversarial verification + push/CI/commitment, gated by the full closure set into COMPLETE. Every stage's obligations live in one dependency-tracked DAG (`.gm/mutables.yml`, `depends_on` field) spanning all five typed phases -- a CONC-kind row may legitimately depend on an already-resolved STATE-kind row, matching how a Lean proof reuses an earlier lemma regardless of which section it lives in. Scope = the closure of the destructive transform admissible over the session; your first emit = closure, not prefix.\n\n**Why the 9-stage shape stayed put when the obligation system went non-linear.** The FSM's `Edge{from,to,gates}` primitive was already an arbitrary directed graph before this change -- 12 non-linear feedback edges (PROVE->SPECIFY, STATE->SPECIFY, DECIDE->PROVE, etc.) existed already, so nothing about adopting a Lean-style dependency graph required reordering or collapsing the named stages. The analogy: Lean's non-linearity lives in its lemma/theorem dependency graph, not in reordering `section`/`namespace` blocks -- a lemma in one section can freely depend on a lemma from an earlier section without the sections themselves needing to move. gm's stages are the equivalent of Lean's sections: coarse-grain review boundaries naming WHICH KIND of obligation is being worked (a human/agent context switch), while `depends_on` on individual mutables carries the actual non-linear structure, cross-phase-boundary included. Reordering the stages would have been solving a problem that does not exist; deepening the obligation graph inside the existing stage skeleton is the change that Lean's model actually calls for.\n\n**Continuation invariant (the brick wall).** Turn without tool call = stop -- harness reads only tool calls. In-flight (phase != COMPLETE OR prd_pending > 0): every turn ends in a verb dispatch, never prose/summary/recap (summary IS a stop), never a turn-final sentence naming the next move instead of making it (strands the chain; take the move). Only phase=COMPLETE AND prd_pending=0 authorizes stopping THE VERB SPOOL -- it does not authorize a bare prose ending. The actual last dispatch is `Skill(skill=\"gm-continue\")` (a host-level tool, not a spool verb): that skill independently checks for remaining work and either reloads `gm` or confirms the loop genuinely closed. Skipping straight from a terminal `transition` response to silence, without that one `Skill` dispatch, is the same class of stop as ending mid-chain -- it is why \"list all remaining limitations\" has to be retyped manually instead of the chain continuing on its own. Urge to stop -> dispatch `phase-status`; non-terminal = drift -> dispatch `instruction`, keep walking; genuinely terminal = dispatch `Skill(skill=\"gm-continue\")` before the turn ends. Depends only on the verb spool -- holds on every agent. Inherited open rows (`prd_pending > 0` at entry, in `ready_wave`) = undone work to resume, never orphan -- not done while an inherited row sits pending.\n\n**There is no next session where a \"ready to resume\" turn actually resumes -- writing that sentence ends the conversation as surely as never writing anything again.** A response with no tool call is the last message of this conversation, full stop, regardless of how the prose frames it (\"Session N closes,\" \"standing work ready for next invocation,\" \"user can resume with /gm,\" a recap of decisions made so far). The user re-typing `/gm` later is not this chain continuing -- it is a new, separate invocation that has to re-discover everything the closing summary just threw away. The only mechanism that produces an actual next action instead of silence is a dispatch in the SAME response, never a description of what a future response would do.\n\n## Admission Filter\n\n```\ncandidate -> [L1 witness] -> [L2 single-writer] -> [L3 direction] -> execute\n```\n\n- **L1.** Admit on witness, not cheapness. Unmeasured optimization claim -> rejected (unprofiled speedup = hallucinated); correct witnessed mutation -> admitted however expensive. Only cost weighed: correctness-cost of unverified claim, never effort. Work envelope unbounded; \"too much work\" never rejects.\n- **L2.** Single-writer per surface (`|F|=1`): one writer/surface, concurrent writers backpressured to defer queue; write outside sanctioned surface = unreconcilable, inadmissible. Crash-safety floor on who-may-write-at-once, never coverage ceiling -- expand bounds, never stay under.\n- **L3.** Lyapunov: `Delta d >= 0` rejects dispatch. Audit tuple `(id, hash, ts)` per accepted write. Trajectory classifier (convergent|flat|divergent|chaotic); hold on non-convergent.\n\nFive phases = scheduling; filter = engine on every candidate, gating witness/writer-safety/direction, never effort.\n\n## Invariants\n\n- **Measurement gates optimization** *claims*, not effort -- a measured-correct change ships however costly.\n- **Bounds prevent cascades:** explicit per-surface writer capacity converts crash to graceful degradation -- bounds writers, not coverage.\n- **Effort is unbounded:** the maximal-effort fully-destructive run is the default; the only costs weighed are maintenance-surface left behind (net-smaller wins, a heavy dep for a few lines loses) and the correctness-cost of an unverified claim.\n- **Direction eliminates waste:** motion that does not reduce distance is dead.\n- **Monotonic closure on first emit:** a partial emit externalizes residual cost as unaudited state; mature artifact = first artifact.\n- **Witness is the audit primitive:** a claim without `(id, hash, ts)` is not in the system.\n\n## Code Invariants (every possible emission)\n\nThe named-principle canon lives distributed across the stage prose files (Correctness & Reliability + Idempotency at STATE, Performance at CONC, Architecture + Workflow + XY at SPECIFY, Code Quality at EMIT, Security at SEC, Definition of Done at DECIDE, Chain-of-Thought at PROVE); those names are the wide preferences with narrow selection text, and they govern every emission. What remains here is the gm-specific operational residue the canon does not cover:\n\n- **Naming by scale:** <50 lines single-letter algebraic; 50-200 short descriptors; >200 full names; public APIs explicit.\n- **Binary transport, append-only persistence:** varint fields; lexical cursors for sparse reads; append-only sequence for replay; chunked by lexical range, modify only the touched chunk.\n- **Single focused task per session:** no drive-by refactors; pre-compute and inline.\n- **Async boundary explicit:** sequential awaitable primitives; no implicit callback ordering; unified error channel, never swallow rejections.\n\n## Token Discipline\n\nEnglish describing intent = liability when code encodes it; comments = liability when names+structure encode the same; duplication-that-must-sync = liability. Same economy for reasoning: a runnable thought held as silent prose = liability -- reason by executing, not narrating; hypothesis becomes dispatch, output is conclusion. Prose enacts the discipline structurally, never narrates scenarios. Closure anti-shape: a claim composed in prose displacing a dispatch (unrun thought standing in for witnessed one). Response body is not a mutation surface.\n\n## Install\n\n`npx gm-skill install` copies the skill directory into `~/.claude/skills/gm/` (and `~/.agents/skills/gm/`), installed as `/gm`; `--yes` is the non-interactive form. No `skills` library.\n\n## Bootstrap\n\nFirst dispatch checks `~/.gm-tools/plugkit.wasm` (or `~/.claude/gm-tools/plugkit.wasm` on legacy installs). Absent -> write `.gm/exec-spool/in/bootstrap/0.txt`; plugkit fetches, sha-verifies, writes `.bootstrap-status.json`. On pin mismatch it writes `.bootstrap-error.json` and you pause the chain.\n\n## Hook denials throw, never mutate\n\nA hook that blocks a tool call throws an error carrying an imperative instruction string as its whole denial surface -- it never rewrites the call's own arguments into a form that then fails on its own, never a shell command exiting 1, never a one-liner writing to stderr and exiting. A thrown error reads to the model as a policy refusal (\"try a different tool\"); an args-mutation producing the same failure reads as \"the tool is broken,\" so the model retries the same tool in the same shape, a loop that never converges. Every denial-issuing hook: throw, never mutate.\n\n## Supervisor drift and version updates\n\nA supervisor respawns the watcher under fresh code on `wrapper.drift`/`version.drift` or a stale `.status.json`. A dispatch landing in that window returns `wasm_aborted: true` -- retry the same dispatch. `update.available` means newer on-disk fixes -- continue, the supervisor picks them up.\n\n**Sideload protection can silently and permanently pin a stale gm plugin build.** The real, currently-loaded plugin binaries live at `~/.agentplug/plugins/.wasm` (per-plugin, e.g. `gm.wasm`, `bert.wasm`), not the `~/.gm-tools/plugkit.wasm` bootstrap path above -- that bootstrap path is the initial-fetch target only; the live agentplug-runner daemon (`~/.agentplug/`) serves from its own plugins directory once running. If `~/.agentplug/plugins/.version` holds a non-release-semver string (a hand-built dev tag, e.g. `local-dev-sideload-
= spin::Mutex::new(Table { + rows: ArrayVec::new_const(), + overflow_seen: [0; SYSNO_BITMAP_WORDS], + overflow_dropped: 0, + }); + + /// What [`record`] found: the caller logs a `First` loudly and a `Repeat` + /// only at trace level. + pub(crate) enum Sighting { + First, + Repeat(u64), + } + + /// Format `args` into a bounded key; truncation is silent by design. + pub(crate) fn shape_of(args: core::fmt::Arguments<'_>) -> Shape { + let mut shape = Shape::new_const(); + let _ = write!(shape, "{args}"); + shape + } + + /// Record one sighting of (`nr`, `shape`). + pub(crate) fn record(nr: Option, shape: &Shape) -> Sighting { + let mut table = TABLE.lock(); + if let Some(row) = table + .rows + .iter_mut() + .find(|row| row.nr == nr && row.shape == *shape) + { + row.count = row.count.saturating_add(1); + return Sighting::Repeat(row.count); + } + if table + .rows + .try_push(Row { + nr, + shape: *shape, + count: 1, + }) + .is_ok() + { + return Sighting::First; + } + // Table full: fall back to once-per-syscall-number. + if let Some(nr) = nr + && let Some(word) = table.overflow_seen.get_mut(nr / 64) + { + let bit = 1u64 << (nr % 64); + if *word & bit == 0 { + *word |= bit; + return Sighting::First; + } + } + table.overflow_dropped = table.overflow_dropped.saturating_add(1); + Sighting::Repeat(table.overflow_dropped) + } + + /// The Linux name of syscall `nr`, from the `syscalls` crate's table, with + /// a local tail for numbers newer than that table (`mseal` is the one a + /// current Chromium probes on every large mapping). + pub(crate) fn syscall_name(nr: usize) -> &'static str { + if let Some(sysno) = ::syscalls::Sysno::new(nr) { + return sysno.name(); + } + match nr { + 462 => "mseal", + 463 => "setxattrat", + 464 => "getxattrat", + 465 => "listxattrat", + 466 => "removexattrat", + 467 => "open_tree_attr", + 468 => "file_getattr", + 469 => "file_setattr", + _ => "?", + } } } @@ -138,6 +276,36 @@ pub struct LinuxShimEntrypoints { _not_send: core::marker::PhantomData<*const ()>, } +/// Decodes a host exception into the pair the memory manager needs to service a +/// demand fault -- the faulting address and the architecture's raw fault status +/// word -- or `None` when the exception is not a memory fault at all. +/// +/// x86-64 reports the address in `CR2` and the status in the hardware error +/// code; aarch64 reports them in `FAR_EL1` and `ESR_EL1`. Both are opaque here: +/// the platform's [`VmemPageFaultHandler`](litebox::mm::linux::VmemPageFaultHandler) +/// is what decodes the status word. +#[cfg(target_arch = "x86_64")] +fn page_fault_info(info: &litebox::shim::ExceptionInfo) -> Option<(usize, u64)> { + (info.exception == litebox::shim::Exception::PAGE_FAULT) + .then(|| (info.cr2, u64::from(info.error_code))) +} + +#[cfg(target_arch = "aarch64")] +fn page_fault_info(info: &litebox::shim::ExceptionInfo) -> Option<(usize, u64)> { + use litebox::shim::Exception; + + // Both abort classes are memory faults; the current-EL variants are the + // ones raised by LiteBox's own accesses to guest memory. + let is_abort = matches!( + info.exception, + Exception::DATA_ABORT_CURRENT_EL + | Exception::DATA_ABORT_LOWER_EL + | Exception::INSTRUCTION_ABORT_CURRENT_EL + | Exception::INSTRUCTION_ABORT_LOWER_EL + ); + is_abort.then_some((info.fault_address, info.esr)) +} + impl litebox::shim::EnterShim for LinuxShimEntrypoints { @@ -156,12 +324,14 @@ impl litebox::shim::EnterShim ctx: &mut Self::ExecutionContext, info: &litebox::shim::ExceptionInfo, ) -> ContinueOperation { - if info.kernel_mode && info.exception == litebox::shim::Exception::PAGE_FAULT { + if info.kernel_mode + && let Some((fault_address, error_code)) = page_fault_info(info) + { if unsafe { self.task .global .pm - .handle_page_fault(info.cr2, info.error_code.into()) + .handle_page_fault(fault_address, error_code) } .is_ok() { @@ -170,6 +340,44 @@ impl litebox::shim::EnterShim return ContinueOperation::Terminate; } } + // Best-effort symbolization of a genuine guest fault: name the guest + // ELF image (and image-relative offset) containing the fault PC and + // the return address, in the `path+0xoffset` form `llvm-symbolizer` + // resolves directly against the guest's own binaries. Debug level so + // it is inert unless logging is enabled -- guests also take faults on + // purpose (e.g. OpenSSL's SIGILL CPU-feature probes). + { + let symbolize = |addr: usize| match self.task.symbolize_guest_address(addr) { + Some((path, offset)) => alloc::format!("{path}+{offset:#x}"), + None => alloc::format!("{addr:#x} (no image)"), + }; + #[cfg(target_arch = "aarch64")] + { + litebox_util_log::debug!( + pc:% = symbolize(ctx.pc), x30:% = symbolize(ctx.regs[30]), + exception:? = info.exception; + "guest fault location" + ); + // The integer register file, so a data abort's faulting address can be + // reconstructed from the disassembly at `pc` (the platform's `fault_address` + // is not always populated). + let mut regs = alloc::string::String::new(); + for (i, r) in ctx.regs.iter().enumerate() { + use core::fmt::Write as _; + let _ = write!(regs, "x{i}={r:#x} "); + } + litebox_util_log::debug!( + sp:% = alloc::format!("{:#x}", ctx.sp), regs:% = regs; + "guest fault registers" + ); + } + #[cfg(target_arch = "x86_64")] + litebox_util_log::debug!( + rip:% = symbolize(ctx.rip), rsp:% = alloc::format!("{:#x}", ctx.rsp), + exception:? = info.exception; + "guest fault location" + ); + } self.enter_shim(false, ctx, |task, _ctx| task.handle_exception_request(info)) } @@ -188,6 +396,11 @@ impl LinuxShimEntrypoints { if !is_init { self.task.enter_from_guest(); } + // Recorded on every entry so that a snapshot taken later -- at a blocking point deep + // inside a syscall, where no `PtRegs` is in reach -- knows where this task's live guest + // stack starts. See `syscalls::process::Task::save_address_space`. + self.task + .record_guest_sp(syscalls::process::guest_stack_pointer(ctx)); f(&self.task, ctx); if self.task.prepare_to_run_guest(ctx) { ContinueOperation::Resume @@ -201,14 +414,35 @@ impl LinuxShimEntrypoints { pub struct LinuxShimBuilder { platform: &'static Platform, litebox: LiteBox, + /// Handle to the `/proc` backend mounted by [`Self::default_fs`], if it was called. + /// [`Self::build`] moves this into [`GlobalState`] so the shim can publish the guest task's + /// identity into it as that becomes known (see `syscalls::process::Task::set_task_comm`). + proc_handle: Cell>>, + /// Handle to the `/dev/fb0` framebuffer mounted by [`Self::default_fs`], if it was called. + /// [`Self::build`] moves this into [`GlobalState`] so `sys_ioctl` can service `FBIO*` + /// requests directly, without threading the framebuffer through the generic `FS` type. + framebuffer: Cell>>, + /// Handle to the `/dev/input` event-device registry mounted by [`Self::default_fs`], if it + /// was called. Same lifecycle as `framebuffer`: [`Self::build`] moves it into + /// [`GlobalState`] for `sys_ioctl`/`sys_read`/poll interception, and the runner takes a + /// clone (via [`LinuxShim::input_registry`]) to inject RFB input events through. + input_registry: Cell>>, } impl LinuxShimBuilder { /// Returns a new shim builder using the given platform. pub fn new(platform: &'static Platform) -> Self { + Self::new_with_litebox(platform, LiteBox::new(platform)) + } + + /// Returns a new shim builder using an already-created LiteBox instance. + pub fn new_with_litebox(platform: &'static Platform, litebox: LiteBox) -> Self { Self { platform, - litebox: LiteBox::new(platform), + litebox, + proc_handle: Cell::new(None), + framebuffer: Cell::new(None), + input_registry: Cell::new(None), } } @@ -218,29 +452,62 @@ impl LinuxShimBuilder { } /// Create a default layered file system with the given in-memory layer and tar data. + /// + /// Also mounts a `/proc` backend and a `/dev/fb0` framebuffer, stashing handles to both on + /// `self`; [`Self::build`] moves them into the built shim's `GlobalState` so the guest task's + /// identity can be published into `/proc//*` once it's known, and `sys_ioctl` can + /// service `FBIO*` requests. Calling this more than once replaces the stashed handles with + /// the most recent call's -- only the filesystem actually passed to + /// `LinuxShim::load_program` should be kept live. pub fn default_fs( &self, in_mem_fs: litebox::fs::in_mem::FileSystem, tar_data: Cow<'static, [u8]>, ) -> DefaultFS { - default_fs(&self.litebox, in_mem_fs, tar_data) + let (fs, proc_handle, framebuffer, input_registry) = + default_fs(&self.litebox, in_mem_fs, tar_data); + self.proc_handle.set(Some(proc_handle)); + self.framebuffer.set(Some(framebuffer)); + self.input_registry.set(Some(input_registry)); + fs } /// Build the shim. pub fn build(self) -> LinuxShim { - let mut net = Network::new(&self.litebox); + self.build_with_net_config(None, None) + } + + /// Same as [`Self::build`], but lets the caller override this instance's + /// interface/gateway addresses (`None` = use `Network::new`'s default of + /// `10.0.0.2`/`10.0.0.1`). Needed to run more than one shim on the same + /// host at once, each independently reachable. + pub fn build_with_net_config( + self, + interface_ip: Option, + gateway_ip: Option, + ) -> LinuxShim { + let mut net = Network::new_with_optional_addrs(&self.litebox, interface_ip, gateway_ip); net.set_platform_interaction(litebox::net::PlatformInteraction::Manual); let global = Arc::new(GlobalState { platform: self.platform, pm: PageManager::new(&self.litebox), - futex_manager: FutexManager::new(), pipes: Pipes::new(&self.litebox), net: litebox::sync::Mutex::new(net), boot_time: self.platform.now(), next_thread_id: 2.into(), // start from 2, as 1 is used by the main thread + proc_handle: self.proc_handle.take(), + framebuffer: self.framebuffer.take(), + input_registry: self.input_registry.take(), litebox: self.litebox, unix_addr_table: litebox::sync::RwLock::new(syscalls::unix::UnixAddrTable::new()), - elf_patch_cache: litebox::sync::Mutex::new(alloc::collections::BTreeMap::new()), + pty_registry: Arc::new(syscalls::file::PtyRegistry::new()), + guest_images: litebox::sync::Mutex::new(alloc::vec::Vec::new()), + loaded_images: litebox::sync::Mutex::new(alloc::vec::Vec::new()), + shared_file_backings: litebox::sync::Mutex::new(alloc::collections::BTreeMap::new()), + termios: litebox::sync::Mutex::new(litebox_common_linux::Termios::default_cooked()), + stdio_foreground_pgid: core::sync::atomic::AtomicI32::new(0), + processes: syscalls::process::ProcessTable::new(), + brk_lock: litebox::sync::Mutex::new(()), }); LinuxShim(global) } @@ -254,6 +521,23 @@ impl Clone for LinuxShim { } impl LinuxShim { + /// A cheap handle to this shim's `/dev/fb0` framebuffer, if [`LinuxShimBuilder::default_fs`] + /// mounted one -- for a runner-side reader (e.g. an RFB server) to read guest-painted pixels + /// independently of any guest fd. `None` when the shim was built with a filesystem that + /// doesn't mount `/dev/fb0`. + #[must_use] + pub fn framebuffer(&self) -> Option> { + self.0.framebuffer.clone() + } + + /// A cheap handle to this shim's `/dev/input` event-device registry, if + /// [`LinuxShimBuilder::default_fs`] mounted one -- for a runner-side injector (e.g. the RFB + /// server's input events) to feed guest-visible keyboard/pointer events through. + #[must_use] + pub fn input_registry(&self) -> Option> { + self.0.input_registry.clone() + } + /// Loads the program at `path` as the shim's initial task, returning the /// initial register state. pub fn load_program( @@ -274,30 +558,42 @@ impl LinuxShim { } = task; let files = syscalls::file::FilesState::new(fs); - files.set_max_fd(syscalls::process::RLIMIT_NOFILE_CUR - 1); + // Actual fd-allocation ceiling stays at the hard limit regardless of the reported soft + // `RLIMIT_NOFILE` default (see that constant's own doc comment): a process that never + // calls `setrlimit` can still open as many fds as it needs, matching this shim's existing + // choice not to enforce rlimits, while `getrlimit`/`prlimit64` still *report* a realistic + // low soft default so close-every-fd-up-to-the-limit startup loops (real daemons do this, + // `dbus-daemon` observed live) stay fast instead of scanning up to a million fds. + files.set_max_fd(syscalls::process::RLIMIT_NOFILE_MAX - 1); let files = Arc::new(files); files.initialize_stdio_in_shared_descriptors_table(&self.0); + // Keep the pid/tid allocator clear of the initial task's own pid, so that no `fork`ed + // child can ever collide with it. + self.0 + .next_thread_id + .fetch_max(pid.saturating_add(1), core::sync::atomic::Ordering::Relaxed); + self.0 + .stdio_foreground_pgid + .store(pid, core::sync::atomic::Ordering::Release); + let entrypoints = crate::LinuxShimEntrypoints { _not_send: core::marker::PhantomData, task: Task { global: self.0.clone(), - thread: syscalls::process::ThreadState::new_process(pid), + thread: syscalls::process::ThreadState::new_process(pid, pid), wait_state: wait::WaitState::new(self.0.platform), pid, ppid, tid: pid, - credentials: syscalls::process::Credentials { - uid, - euid, - gid, - egid, - } - .into(), + credentials: RefCell::new( + syscalls::process::Credentials::new(uid, euid, gid, egid).into(), + ), comm: [0; litebox_common_linux::TASK_COMM_LEN].into(), // set at load time fs: Arc::new(syscalls::file::FsState::new()).into(), files: files.into(), signals: syscalls::signal::SignalState::new_process(), + guest_sp: Cell::new(0), }, }; @@ -347,6 +643,37 @@ impl LinuxShim { &self.0.litebox } + /// Create a host-owned TCP listener inside the guest's network stack (typically on the + /// guest's loopback), whose accepted connections host code services directly -- the + /// listening-side counterpart of [`Self::tcp_connection`]. The guest never sees an fd for + /// any of these sockets. + /// + /// # Errors + /// + /// Fails if the socket cannot be created, bound (e.g. the guest already owns the port), or + /// put into the listening state. + pub fn listen_in_guest( + &self, + addr: core::net::SocketAddr, + backlog: u16, + ) -> Result, Errno> { + host_service::listen_in_guest(&self.0, addr, backlog) + } + + /// Create a host-owned UDP socket bound inside the guest's network stack (typically on the + /// guest's loopback) -- the datagram counterpart of [`Self::listen_in_guest`]. The guest + /// never sees an fd for it. + /// + /// # Errors + /// + /// Fails if the socket cannot be created or bound (e.g. the guest already owns the port). + pub fn bind_udp_in_guest( + &self, + addr: core::net::SocketAddr, + ) -> Result, Errno> { + host_service::bind_udp_in_guest(&self.0, addr) + } + /// Returns the platform this shim was built with. pub fn platform(&self) -> &'static Platform { self.0.platform @@ -375,21 +702,55 @@ impl LinuxShimProcess { } /// Create a default layered file system with the given in-memory layer and tar data. +/// +/// Also returns a handle to the mounted `/proc` backend, and to the mounted `/dev/fb0` +/// framebuffer; the caller (`LinuxShimBuilder`) is responsible for keeping the `/proc` handle +/// reachable so the guest task's identity can be published into it once known -- see +/// `syscalls::process::Task::set_task_comm` -- and stashes the framebuffer handle into +/// `GlobalState` so `sys_ioctl` can service `FBIO*` requests without threading it through the +/// generic `FS` type. fn default_fs( litebox: &LiteBox, in_mem_fs: litebox::fs::in_mem::FileSystem, tar_data: Cow<'static, [u8]>, -) -> LinuxFS { - let dev_stdio = litebox::fs::resolver::Resolver::new( +) -> ( + LinuxFS, + litebox::fs::proc::Proc, + litebox::fs::devices::Framebuffer, + litebox::fs::devices::InputRegistry, +) { + let mut proc_handle = None; + let mut framebuffer = None; + let input_registry = litebox::fs::devices::InputRegistry::new(); + let input_registry_for_mount = input_registry.clone(); + let current_user = in_mem_fs.current_user(); + let dev_stdio = litebox::fs::resolver::Resolver::new_with_user( litebox, litebox::fs::composer::Composer::builder() .mount("/dev", |allocator| { - litebox::fs::devices::Devices::new(litebox, allocator) + let devices = litebox::fs::devices::Devices::new(litebox, allocator); + framebuffer = Some(devices.framebuffer()); + devices + }) + .mount("/dev/input", |allocator| { + litebox::fs::devices::InputDevices::new(allocator, input_registry_for_mount) + }) + .mount("/proc", |allocator| { + let proc = litebox::fs::proc::Proc::new(allocator); + proc_handle = Some(proc.clone()); + proc }) .build() .unwrap(), + current_user, ); - let tar_ro = litebox::fs::resolver::Resolver::new( + let proc_handle = proc_handle.expect("mounted immediately above"); + // `Composer::builder().mount("/dev", ..)`'s closure runs synchronously inside `.build()` + // above, so `framebuffer` is always `Some` here in practice; falling back to a fresh, + // unmounted `Framebuffer` rather than panicking keeps this path total even if that + // invariant is ever violated by a future refactor. + let framebuffer = framebuffer.unwrap_or_else(litebox::fs::devices::Framebuffer::new); + let tar_ro = litebox::fs::resolver::Resolver::new_with_user( litebox, litebox::fs::composer::Composer::builder() .mount("/", |allocator| { @@ -397,18 +758,22 @@ fn default_fs( }) .build() .unwrap(), + current_user, ); - litebox::fs::layered::FileSystem::new( + let fs = litebox::fs::layered::FileSystem::new_with_user( litebox, in_mem_fs, - litebox::fs::layered::FileSystem::new( + litebox::fs::layered::FileSystem::new_with_user( litebox, dev_stdio, tar_ro, litebox::fs::layered::LayeringSemantics::LowerLayerReadOnly, + current_user, ), litebox::fs::layered::LayeringSemantics::LowerLayerWritableFiles, - ) + current_user, + ); + (fs, proc_handle, framebuffer, input_registry) } // Special override so that `GETFL` can return stdio-specific flags @@ -461,6 +826,18 @@ impl Task { } } } + + /// Explicitly closes every fd still alive in this (process-wide-last) file table. + /// + /// See `syscalls::process::Task::prepare_for_exit` for why this has to be explicit rather + /// than relying on `FilesState`'s `Drop`. + pub(crate) fn close_all_fds_on_exit(&self) { + let files = self.files.borrow(); + let alive_fds: Vec = files.raw_descriptor_store.read().iter_alive().collect(); + for raw_fd in alive_fds { + let _ = self.do_close(raw_fd); + } + } } impl syscalls::file::FilesState { @@ -474,6 +851,7 @@ impl syscalls::file::FilesState>) -> R, epoll: impl FnOnce(&TypedFd>) -> R, unix: impl FnOnce(&TypedFd>) -> R, + netlink: impl FnOnce(&TypedFd>) -> R, ) -> Result { let rds = self.raw_descriptor_store.read(); if let Ok(fd) = rds.fd_from_raw_integer(fd) { @@ -500,6 +878,10 @@ impl syscalls::file::FilesState { self.map(|v| v as usize) } } +impl ToSyscallResult for Result { + fn to_syscall_result(self) -> Result { + // `inotify_add_watch(2)`'s only current caller of this impl returns a small positive + // watch descriptor; the raw syscall ABI still reinterprets it as an unsigned register + // value exactly like every other non-negative-int-returning syscall here. + self.map(|v| v.reinterpret_as_unsigned() as usize) + } +} impl Task { /// A wrapper function around `sys_pread64` that copies data in chunks to avoid OOMing. @@ -561,13 +951,82 @@ impl Task { Ok(read_total) } + /// A wrapper around `sys_write`/`sys_pwrite64` that copies the guest buffer + /// in bounded chunks to avoid a single unbounded allocation for a huge + /// guest-supplied `count`, mirroring [`Self::pread_with_user_buf`]. + /// + /// Unlike the read direction, a write is not itself retried past a short + /// result: `sys_write` may legitimately write fewer bytes than asked (a + /// pipe or socket at capacity), and real `write(2)` semantics leave + /// retrying a short write to the caller, not the kernel. So only the + /// copy-from-guest-memory step is chunked; a chunk that is not fully + /// consumed ends the loop, exactly as a single unchunked write to that + /// same destination would have. + fn write_with_user_buf( + &self, + fd: i32, + buf: UserPtr, + count: usize, + offset: Option, + ) -> Result { + // A zero-length write must still dispatch: Linux checks fd validity + // and writability before it looks at count, so write(read_end, buf, 0) + // is EBADF, not a silent 0 (witnessed live by pipe_broker's lifecycle + // sub-test under the broker runner). The descriptor layers already + // handle empty buffers correctly past that check. + if count == 0 { + return self.sys_write(fd, &[], offset); + } + let mut written_total = 0; + while written_total < count { + let to_write = (count - written_total).min(MAX_KERNEL_BUF_SIZE); + let chunk_ptr = UserPtr::::from_usize(buf.as_usize() + written_total); + let Some(chunk) = chunk_ptr.to_owned_slice::(to_write) else { + return if written_total > 0 { + Ok(written_total) + } else { + Err(Errno::EFAULT) + }; + }; + match self.sys_write(fd, &chunk, offset.map(|o| o + written_total)) { + Ok(size) => { + written_total += size; + if size < to_write { + // A short write: the destination could not currently + // accept the full chunk. Stop here, matching what a + // single unchunked write to the same destination + // would have returned. + break; + } + } + Err(e) => { + return if written_total > 0 { + Ok(written_total) + } else { + Err(e) + }; + } + } + } + assert!(written_total <= count); + Ok(written_total) + } + /// Handle Linux syscalls and dispatch them to LiteBox implementations. /// /// # Panics /// /// Unsupported syscalls or arguments would trigger a panic for development purposes. fn handle_syscall_request(&self, ctx: &mut litebox_common_linux::PtRegs) { - let return_value = match self.do_syscall(ctx) { + let result = self.do_syscall(ctx); + // The request-side twin of this line lives in `do_syscall` (the + // `req=` trace). Logging the result too is what turns the trace into + // a usable differential record: a guest that aborts after a burst of + // syscalls (libuv's `uv_loop_init` cleanup was the motivating case) + // is undiagnosable from requests alone, because the failing call and + // the cleanup that follows it look identical without return values. + litebox_util_log::trace!(pid:? = self.pid, tid:? = self.tid, ret:? = result; "sysret"); + let return_value = match result { Ok(v) => v, Err(err) => (err.as_neg() as isize).reinterpret_as_unsigned(), }; @@ -575,6 +1034,11 @@ impl Task { { ctx.rax = return_value; } + #[cfg(target_arch = "aarch64")] + { + // The aarch64 Linux syscall ABI returns in x0. + ctx.regs[0] = return_value; + } } fn do_syscall(&self, ctx: &mut litebox_common_linux::PtRegs) -> Result { @@ -587,7 +1051,49 @@ impl Task { #[cfg(target_arch = "x86_64")] let syscall_number = ctx.orig_rax; - let request = SyscallRequest::try_from_raw(syscall_number, ctx, log_unsupported_fmt)?; + // The aarch64 Linux syscall ABI passes the number in x8, which the entry + // path records in `pt_regs::syscallno`. Sign-extending keeps an + // out-of-range value (the kernel writes -1 for "no syscall") looking the + // same as it does in x86-64's `orig_rax`, so the dispatch below rejects + // it identically on both architectures. + #[cfg(target_arch = "aarch64")] + let syscall_number = (ctx.syscallno as isize).reinterpret_as_unsigned(); + // The decoder reports what it could not decode through the callback; keep + // that message out of band so the one line logged below can carry the + // syscall number, its name, the decoder's own description of the + // unsupported shape, and the errno the guest is about to see -- and so + // that a decode failure the decoder does *not* flag (a bare `EINVAL` + // for an unknown `prctl` option, say) is still visible, once. + let decode_note: RefCell> = RefCell::new(None); + let request = SyscallRequest::try_from_raw(syscall_number, ctx, |args| { + *decode_note.borrow_mut() = Some(unsupported::shape_of(args)); + }); + let request = match request { + Ok(request) => { + if let Some(shape) = decode_note.take() { + self.report_unsupported_syscall(syscall_number, shape, Ok(())); + } + request + } + Err(errno) => { + let shape = decode_note.take().unwrap_or_else(|| { + let args = raw_syscall_args(ctx); + unsupported::shape_of(format_args!( + "decode failed, args=[{:#x}, {:#x}, {:#x}, {:#x}, {:#x}, {:#x}]", + args[0], args[1], args[2], args[3], args[4], args[5] + )) + }); + self.report_unsupported_syscall(syscall_number, shape, Err(errno)); + return Err(errno); + } + }; + // A permanent, trace-gated record of every decoded syscall + // (`LITEBOX_LOG=litebox_shim_linux=trace`). Off by default and a single level check when + // it is off, but it is the only view of what a real guest is actually asking for: it is + // what showed that busybox's blocking `wait` is a `sigsuspend` loop, and that the shim + // was answering `sigsuspend` with an unimplemented-syscall error that release builds did + // not even log (`log_unsupported_fmt` is `debug_assertions`-only). + litebox_util_log::trace!(pid:? = self.pid, tid:? = self.tid, req:? = request; "syscall"); match request { SyscallRequest::Exit { status } => { @@ -646,11 +1152,9 @@ impl Task { }) } } - SyscallRequest::Write { fd, buf, count } => match buf.to_owned_slice::(count) - { - Some(buf) => self.sys_write(fd, &buf, None), - None => Err(Errno::EFAULT), - }, + SyscallRequest::Write { fd, buf, count } => { + self.write_with_user_buf(fd, buf, count, None) + } SyscallRequest::Close { fd } => syscall!(sys_close(fd)), SyscallRequest::Lseek { fd, offset, whence } => { use litebox::utils::TruncateExt as _; @@ -670,6 +1174,10 @@ impl Task { SyscallRequest::Chdir { pathname } => pathname .to_cstring::() .map_or(Err(Errno::EINVAL), |path| syscall!(sys_chdir(path))), + SyscallRequest::Chroot { path } => path + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| syscall!(sys_chroot(path))), + SyscallRequest::Fchdir { fd } => syscall!(sys_fchdir(fd)), SyscallRequest::RtSigprocmask { how, set, @@ -683,6 +1191,9 @@ impl Task { sigsetsize, } => self.sys_rt_sigaction(signum, act, oldact, sigsetsize), SyscallRequest::RtSigreturn => self.sys_rt_sigreturn(ctx), + SyscallRequest::RtSigsuspend { mask, sigsetsize } => { + self.sys_rt_sigsuspend(mask, sigsetsize) + } SyscallRequest::Ioctl { fd, arg } => syscall!(sys_ioctl(fd, arg)), SyscallRequest::Pread64 { fd, @@ -695,10 +1206,10 @@ impl Task { buf, count, offset, - } => match buf.to_owned_slice::(count) { - Some(buf) => self.sys_pwrite64(fd, &buf, offset), - None => Err(Errno::EFAULT), - }, + } => { + let pos = usize::try_from(offset).map_err(|_| Errno::EINVAL)?; + self.write_with_user_buf(fd, buf, count, Some(pos)) + } SyscallRequest::Sendfile { out_fd, in_fd, @@ -712,11 +1223,30 @@ impl Task { flags, fd, offset, - } => self - .sys_mmap(addr, length, prot, flags, fd, offset) - .map(|ptr| ptr.as_usize()), + } => { + // GUARD (litebox-ordinary-syscall-cross-process-clobber): see + // `Task::touches_another_process`'s doc comment. Only `MAP_FIXED` is destructive + // to whatever is already there; an ordinary hinted/hint-free mapping always goes + // through the placement allocator, which cannot land on another process's live + // memory (`Vmem::reserve_external` already covers the one gap that existed there). + if flags.contains(litebox_common_linux::MapFlags::MAP_FIXED) + && self.touches_another_process(addr, length) + { + Err(Errno::ENOMEM) + } else { + self.sys_mmap(addr, length, prot, flags, fd, offset) + .map(|ptr| ptr.as_usize()) + } + } SyscallRequest::Mprotect { addr, length, prot } => { - syscall!(sys_mprotect(addr, length, prot)) + // GUARD (litebox-ordinary-syscall-cross-process-clobber): see + // `Task::touches_another_process`'s doc comment. Mirrors real Linux's own + // `mprotect` response to a range that is not entirely this process's own mapping. + if self.touches_another_process(addr.as_usize(), length) { + Err(Errno::ENOMEM) + } else { + syscall!(sys_mprotect(addr, length, prot)) + } } SyscallRequest::Mremap { old_addr, @@ -724,10 +1254,47 @@ impl Task { new_size, flags, new_addr, - } => self - .sys_mremap(old_addr, old_size, new_size, flags, new_addr) - .map(|ptr| ptr.as_usize()), - SyscallRequest::Munmap { addr, length } => syscall!(sys_munmap(addr, length)), + } => { + // GUARD (litebox-ordinary-syscall-cross-process-clobber): see + // `Task::touches_another_process`'s doc comment. This was the last + // unguarded hole, and the one that actually reproduced live: a + // node process walks a descending run of what it believes are its + // own adjacent pages with `mremap(addr, 1 page -> 2 pages, 0)` and + // takes the first success. On real Linux those neighbours are its + // own (a process's successive mmaps are contiguous) and a foreign + // address is unmappable, so the loop only ever sees success or + // ENOMEM. Under one shared address space the neighbour is another + // process's live page; the grow "succeeded", `record_mapped` handed + // that page to the caller, and the caller's exit unmapped it under + // its real owner -- a translation fault in musl's malloc for the + // victim, several seconds later. ENOMEM rather than EFAULT on + // purpose: it is the answer the same loop already gets for a page + // of its own that cannot grow in place, so it keeps walking to + // pages it really owns instead of aborting on an error real Linux + // never produces for a valid pointer. + let foreign_old = self.touches_another_process(old_addr.as_usize(), old_size); + let foreign_new = flags.contains(litebox_common_linux::MRemapFlags::MREMAP_FIXED) + && self.touches_another_process(new_addr, new_size); + if foreign_old || foreign_new { + Err(Errno::ENOMEM) + } else { + self.sys_mremap(old_addr, old_size, new_size, flags, new_addr) + .map(|ptr| ptr.as_usize()) + } + } + SyscallRequest::Munmap { addr, length } => { + // GUARD (litebox-ordinary-syscall-cross-process-clobber): see + // `Task::touches_another_process`'s doc comment. Real Linux's own `munmap` is a + // silent no-op over a range this process never had mapped; treating a range that + // turns out to be a stranger's memory the same way is the closest honest match -- + // no case exists where correctly unmapping this process's own memory requires + // touching another process's. + if self.touches_another_process(addr.as_usize(), length) { + Ok(0) + } else { + syscall!(sys_munmap(addr, length)) + } + } SyscallRequest::Brk { addr } => self.sys_brk(addr), SyscallRequest::Readv { fd, iovec, iovcnt } => self.sys_readv(fd, iovec, iovcnt), SyscallRequest::Writev { fd, iovec, iovcnt } => self.sys_writev(fd, iovec, iovcnt), @@ -853,6 +1420,7 @@ impl Task { } => syscall!(sys_getpeername(sockfd, addr, addrlen)), SyscallRequest::Uname { buf } => syscall!(sys_uname(buf)), SyscallRequest::Fcntl { fd, arg } => syscall!(sys_fcntl(fd, arg)), + SyscallRequest::Flock { fd, operation } => syscall!(sys_flock(fd, operation)), SyscallRequest::Getcwd { buf, size: count } => { let mut kernel_buf = vec![0u8; count.min(MAX_KERNEL_BUF_SIZE)]; self.sys_getcwd(&mut kernel_buf).and_then(|size| { @@ -883,7 +1451,40 @@ impl Task { sigmask, sigsetsize, } => self.sys_epoll_pwait(epfd, events, maxevents, timeout, sigmask, sigsetsize), + // PR_SET_VMA (PartitionAlloc names every large anonymous mapping): decoded so it + // is visible in the syscall trace, answered EINVAL exactly like a kernel built + // without CONFIG_ANON_VMA_NAME -- callers already tolerate that. + SyscallRequest::Prctl { + args: litebox_common_linux::PrctlArg::SetVma { + opcode, addr, len, .. + }, + } => { + // One line per opcode, not per mapping: the per-call `addr`/`len` are + // already in the trace-level `syscall req=` record. + let _ = (addr, len); + log_unsupported!("prctl(PR_SET_VMA, opcode = {opcode}) -> EINVAL"); + Err(Errno::EINVAL) + } SyscallRequest::Prctl { args } => self.sys_prctl(args), + // seccomp(2) and prctl(PR_SET_SECCOMP): no BPF filtering exists in the shim, so + // `sys_seccomp` answers an honest ENOSYS (a kernel without CONFIG_SECCOMP), never a + // fake success, and Chromium's layer-2 sandbox degrades instead of believing itself + // enforced. + SyscallRequest::Seccomp { + operation, + flags, + args, + } => syscall!(sys_seccomp(operation, flags, args)), + // mseal(2): nothing seals mappings yet; ENOSYS is what a pre-6.10 kernel says and + // PartitionAlloc's probe tolerates it. Decoded (rather than "unknown syscall 462") + // so the trace shows what was asked. + SyscallRequest::Mseal { addr, len, flags } => { + // One line per flags value, not per call (Chromium probes this on every + // large mapping); `addr`/`len` are in the trace-level `syscall req=` record. + let _ = (addr, len); + log_unsupported!("mseal(flags = {flags}) -> ENOSYS"); + Err(Errno::ENOSYS) + } SyscallRequest::ArchPrctl { arg } => syscall!(sys_arch_prctl(arg)), SyscallRequest::Readlink { pathname, @@ -974,6 +1575,39 @@ impl Task { syscall!(sys_openat(dirfd, path, flags, mode)) }), SyscallRequest::Ftruncate { fd, length } => syscall!(sys_ftruncate(fd, length)), + SyscallRequest::Fallocate { + fd, + mode, + offset, + len, + } => syscall!(sys_fallocate(fd, mode, offset, len)), + SyscallRequest::Fadvise64 { + fd, + offset, + len, + advice, + } => syscall!(sys_fadvise64(fd, offset, len, advice)), + SyscallRequest::Preadv2 { + fd, + iovec, + iovcnt, + pos_l, + pos_h, + flags, + } => self.sys_preadv2(fd, iovec, iovcnt, preadv_pwritev_offset(pos_l, pos_h), flags), + SyscallRequest::Pwritev2 { + fd, + iovec, + iovcnt, + pos_l, + pos_h, + flags, + } => self.sys_pwritev2(fd, iovec, iovcnt, preadv_pwritev_offset(pos_l, pos_h), flags), + SyscallRequest::MemfdCreate { name, flags } => name + .to_cstring::() + .map_or(Err(Errno::EFAULT), |name| { + syscall!(sys_memfd_create(&name, flags)) + }), SyscallRequest::Mknodat { dirfd, pathname, @@ -993,6 +1627,98 @@ impl Task { .map_or(Err(Errno::EFAULT), |path| { syscall!(sys_unlinkat(dirfd, path, flags)) }), + SyscallRequest::Symlinkat { + target, + newdirfd, + linkpath, + } => match ( + target.to_cstring::(), + linkpath.to_cstring::(), + ) { + (Some(target), Some(linkpath)) => { + syscall!(sys_symlinkat(target, newdirfd, linkpath)) + } + _ => Err(Errno::EFAULT), + }, + SyscallRequest::Linkat { + olddirfd, + oldpath, + newdirfd, + newpath, + flags, + } => match ( + oldpath.to_cstring::(), + newpath.to_cstring::(), + ) { + (Some(oldpath), Some(newpath)) => { + syscall!(sys_linkat(olddirfd, oldpath, newdirfd, newpath, flags)) + } + _ => Err(Errno::EFAULT), + }, + SyscallRequest::Renameat2 { + olddirfd, + oldpath, + newdirfd, + newpath, + flags, + } => match ( + oldpath.to_cstring::(), + newpath.to_cstring::(), + ) { + (Some(oldpath), Some(newpath)) => { + syscall!(sys_renameat2(olddirfd, oldpath, newdirfd, newpath, flags)) + } + _ => Err(Errno::EFAULT), + }, + SyscallRequest::Fchmodat { + dirfd, + pathname, + mode, + flags, + } => pathname + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| { + syscall!(sys_fchmodat(dirfd, path, mode, flags)) + }), + SyscallRequest::Fchownat { + dirfd, + pathname, + owner, + group, + flags, + } => pathname + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| { + syscall!(sys_fchownat(dirfd, path, owner, group, flags)) + }), + SyscallRequest::Fchmod { fd, mode } => syscall!(sys_fchmod(fd, mode)), + SyscallRequest::Fchown { fd, owner, group } => { + syscall!(sys_fchown(fd, owner, group)) + } + SyscallRequest::Fsync { fd } => syscall!(sys_fsync(fd)), + SyscallRequest::Utimensat { + dirfd, + pathname, + times, + flags, + } => { + let times = times + .map(|ptr| -> Result<_, Errno> { + let a = ptr.read_at_offset::(0).ok_or(Errno::EFAULT)?; + let b = ptr.read_at_offset::(1).ok_or(Errno::EFAULT)?; + Ok([a, b]) + }) + .transpose()?; + match pathname { + Some(pathname) => pathname + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| { + syscall!(sys_utimensat(dirfd, path, times, flags)) + }), + // `futimens(fd, times)`, emulated by glibc as `utimensat(fd, NULL, times, 0)`. + None => syscall!(sys_futimens(dirfd, times)), + } + } SyscallRequest::Stat { pathname, buf } => { pathname .to_cstring::() @@ -1020,7 +1746,8 @@ impl Task { .ok_or(Errno::EFAULT) .map(|()| 0) }), - #[cfg(target_arch = "x86_64")] + // Reached through `newfstatat` on x86-64 and `fstatat` on aarch64, + // where it is the only path-based stat syscall the kernel offers. SyscallRequest::Newfstatat { dirfd, pathname, @@ -1059,9 +1786,20 @@ impl Task { }) }) } + SyscallRequest::Statfs { pathname, buf } => pathname + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| syscall!(sys_statfs(path, buf))), + SyscallRequest::Fstatfs { fd, buf } => syscall!(sys_fstatfs(fd, buf)), SyscallRequest::Eventfd2 { initval, flags } => { syscall!(sys_eventfd2(initval, flags)) } + SyscallRequest::InotifyInit1 { flags } => syscall!(sys_inotify_init1(flags)), + SyscallRequest::InotifyAddWatch { fd, pathname, mask } => pathname + .to_cstring::() + .map_or(Err(Errno::EFAULT), |path| { + syscall!(sys_inotify_add_watch(fd, path, mask)) + }), + SyscallRequest::InotifyRmWatch { fd, wd } => syscall!(sys_inotify_rm_watch(fd, wd)), SyscallRequest::Pipe2 { pipefd, flags } => { self.sys_pipe2(flags).and_then(|(read_fd, write_fd)| { pipefd @@ -1075,12 +1813,21 @@ impl Task { } SyscallRequest::Clone { args } => self.sys_clone(ctx, &args), SyscallRequest::Clone3 { args } => self.sys_clone3(ctx, args), + SyscallRequest::Unshare { flags } => self.sys_unshare(flags), SyscallRequest::SetThreadArea { user_desc } => { #[cfg(target_arch = "x86_64")] { let _ = user_desc; Err(Errno::ENOSYS) // x86_64 does not support set_thread_area } + #[cfg(target_arch = "aarch64")] + { + // aarch64 has no `set_thread_area` either; the thread + // pointer is `TPIDR_EL0`, set through `clone`'s `tls` + // argument. + let _ = user_desc; + Err(Errno::ENOSYS) + } } SyscallRequest::SetTidAddress { tidptr } => { Ok(self.sys_set_tid_address(tidptr).reinterpret_as_unsigned() as usize) @@ -1117,17 +1864,56 @@ impl Task { } SyscallRequest::Getpid => Ok(self.sys_getpid().reinterpret_as_unsigned() as usize), SyscallRequest::Getppid => Ok(self.sys_getppid().reinterpret_as_unsigned() as usize), + SyscallRequest::Getpgid { pid } => self + .sys_getpgid(pid) + .map(|pgid| pgid.reinterpret_as_unsigned() as usize), + SyscallRequest::Setpgid { pid, pgid } => syscall!(sys_setpgid(pid, pgid)), + SyscallRequest::Setsid => self + .sys_setsid() + .map(|sid| sid.reinterpret_as_unsigned() as usize), + SyscallRequest::Wait4 { + pid, + wstatus, + options, + rusage, + } => self + .sys_wait4(pid, wstatus, options, rusage) + .map(|pid| pid.reinterpret_as_unsigned() as usize), SyscallRequest::Getuid => Ok(self.sys_getuid() as usize), SyscallRequest::Getgid => Ok(self.sys_getgid() as usize), SyscallRequest::Geteuid => Ok(self.sys_geteuid() as usize), SyscallRequest::Getegid => Ok(self.sys_getegid() as usize), + SyscallRequest::Getgroups { size, list } => syscall!(sys_getgroups(size, list)), + SyscallRequest::Setgroups { size, list } => syscall!(sys_setgroups(size, list)), + SyscallRequest::Setuid { uid } => syscall!(sys_setuid(uid)), + SyscallRequest::Setgid { gid } => syscall!(sys_setgid(gid)), + SyscallRequest::Setresuid { ruid, euid, suid } => { + syscall!(sys_setresuid(ruid, euid, suid)) + } + SyscallRequest::Setresgid { rgid, egid, sgid } => { + syscall!(sys_setresgid(rgid, egid, sgid)) + } + SyscallRequest::Getresuid { ruid, euid, suid } => { + syscall!(sys_getresuid(ruid, euid, suid)) + } + SyscallRequest::Getresgid { rgid, egid, sgid } => { + syscall!(sys_getresgid(rgid, egid, sgid)) + } SyscallRequest::Sysinfo { buf } => { let sysinfo = self.sys_sysinfo(); buf.write_at_offset::(0, sysinfo) .ok_or(Errno::EFAULT) .map(|()| 0) } + SyscallRequest::Getrusage { who, usage } => { + let rusage = self.sys_getrusage(who); + usage + .write_at_offset::(0, rusage) + .ok_or(Errno::EFAULT) + .map(|()| 0) + } SyscallRequest::CapGet { header, data } => syscall!(sys_capget(header, data)), + SyscallRequest::CapSet { header, data } => syscall!(sys_capset(header, data)), SyscallRequest::GetDirent64 { fd, dirp, count } => { self.sys_getdirent64(fd, dirp, count) } @@ -1150,6 +1936,16 @@ impl Task { // platform. Ok(0) } + SyscallRequest::SchedGetParam { pid, param } => { + syscall!(sys_sched_getparam(pid, param)) + } + SyscallRequest::SchedSetParam { pid, param } => { + syscall!(sys_sched_setparam(pid, param)) + } + SyscallRequest::SchedGetScheduler { pid } => syscall!(sys_sched_getscheduler(pid)), + SyscallRequest::SchedSetScheduler { pid, policy, param } => { + syscall!(sys_sched_setscheduler(pid, policy, param)) + } SyscallRequest::Futex { args } => self.sys_futex(args), SyscallRequest::Umask { mask } => { let old_mask = self.sys_umask(mask); @@ -1158,6 +1954,30 @@ impl Task { SyscallRequest::Kill { pid, sig } => self.sys_kill(pid, sig), SyscallRequest::Tkill { tid, sig } => self.sys_tkill(tid, sig), SyscallRequest::Tgkill { tgid, tid, sig } => self.sys_tgkill(tgid, tid, sig), + SyscallRequest::RtSigtimedwait { + set, + info, + timeout, + sigsetsize, + } => self.sys_rt_sigtimedwait(set, info, timeout, sigsetsize), + SyscallRequest::RtSigqueueinfo { pid, sig, info } => { + self.sys_rt_sigqueueinfo(pid, sig, info) + } + SyscallRequest::RtTgsigqueueinfo { + tgid, + tid, + sig, + info, + } => self.sys_rt_tgsigqueueinfo(tgid, tid, sig, info), + SyscallRequest::Getpriority { which, who } => self.sys_getpriority(which, who), + SyscallRequest::Setpriority { + which, + who, + niceval, + } => syscall!(sys_setpriority(which, who, niceval)), + SyscallRequest::Membarrier { cmd, flags, cpu_id } => { + self.sys_membarrier(cmd, flags, cpu_id) + } SyscallRequest::Sigaltstack { ss, old_ss } => self.sys_sigaltstack(ss, old_ss, ctx), SyscallRequest::Alarm { seconds } => syscall!(sys_alarm(seconds)), SyscallRequest::Pause => syscall!(sys_pause()), @@ -1169,6 +1989,15 @@ impl Task { new_value, old_value, } => syscall!(sys_setitimer(which, new_value, old_value)), + #[cfg(target_arch = "aarch64")] + SyscallRequest::Ptrace { + request, + pid, + addr, + data, + } => self.sys_ptrace(request, pid, addr, data), + #[cfg(target_arch = "x86_64")] + SyscallRequest::Ptrace { .. } => Err(Errno::ENOSYS), _ => { log_unsupported!("{request:?}"); Err(Errno::ENOSYS) @@ -1177,6 +2006,149 @@ impl Task { } } +/// The six syscall arguments as the guest passed them, for naming an undecodable +/// request by its shape when the decoder itself said nothing about it (the argument +/// values are what tell an unknown `madvise` advice or `prctl` option apart). +#[cfg(target_arch = "aarch64")] +fn raw_syscall_args(ctx: &litebox_common_linux::PtRegs) -> [usize; 6] { + [ + ctx.regs[0], + ctx.regs[1], + ctx.regs[2], + ctx.regs[3], + ctx.regs[4], + ctx.regs[5], + ] +} +#[cfg(target_arch = "x86_64")] +fn raw_syscall_args(ctx: &litebox_common_linux::PtRegs) -> [usize; 6] { + [ctx.rdi, ctx.rsi, ctx.rdx, ctx.r10, ctx.r8, ctx.r9] +} + +/// A guest ELF image placed by the shim's own loader -- the main executable +/// and its `PT_INTERP` -- recorded at load time for fault symbolization. +/// +/// Kept apart from [`syscalls::mm::GuestImage`] deliberately: that list is +/// filled from the `mmap` path and named through the mapping fd's recorded +/// path, but the loader opens its images with `open_executable_as` and inserts +/// the descriptor raw, so no path ever exists for it there -- and the main +/// executable, where an official Chromium's `NOTREACHED()` traps land, was +/// precisely the image the fault line printed as "(no image)". Rows carry the +/// loading pid so that a later `execve` by the same process replaces its own +/// rows (exec replaces the whole address space); a forked child shares its +/// parent's placement and resolves through the parent's row by address. +pub(crate) struct LoadedImage { + pid: i32, + /// Lowest guest address covered by the image's `PT_LOAD` segments. + lo: usize, + /// One past the highest guest address covered by the image. + hi: usize, + /// The load bias: guest address minus ELF vaddr. `addr - base` is the + /// image-relative offset `llvm-symbolizer` resolves against the file. + base: usize, + /// The path the loader opened the image with. + path: alloc::string::String, +} + +/// Rows kept in [`GlobalState::loaded_images`] before the oldest are dropped. +/// Two rows per exec, so this covers thousands of short-lived processes while +/// bounding a desktop session that forks forever. +const LOADED_IMAGES_CAP: usize = 4096; + +impl Task { + /// Drop the loader-placed rows of this pid: an image load replaces the + /// whole address space, so whatever was recorded for it is gone. + pub(crate) fn forget_loaded_images(&self) { + self.global + .loaded_images + .lock() + .retain(|img| img.pid != self.pid); + } + + /// Record an image the loader has just placed at bias `base`, spanning + /// guest addresses `lo..hi`. + pub(crate) fn record_loaded_image(&self, path: &str, base: usize, lo: usize, hi: usize) { + let mut images = self.global.loaded_images.lock(); + if images.len() >= LOADED_IMAGES_CAP { + let excess = images.len() + 1 - LOADED_IMAGES_CAP; + images.drain(..excess); + } + images.push(LoadedImage { + pid: self.pid, + lo, + hi, + base, + path: alloc::string::String::from(path), + }); + } + + /// Name the guest ELF image containing `addr` as `(path, image-relative + /// offset)`. This task's own loader-placed images win (they are exact for + /// the live process), then any loader-placed image by address (a forked + /// child), then the `mmap`-recorded shared libraries, latest first. + pub(crate) fn symbolize_guest_address( + &self, + addr: usize, + ) -> Option<(alloc::string::String, usize)> { + { + let images = self.global.loaded_images.lock(); + let contains = |img: &&LoadedImage| (img.lo..img.hi).contains(&addr); + let hit = images + .iter() + .rev() + .filter(|img| img.pid == self.pid) + .find(contains) + .or_else(|| images.iter().rev().find(contains)); + if let Some(img) = hit { + return Some((img.path.clone(), addr - img.base)); + } + } + self.find_guest_image(addr) + } + + /// Log one undecodable syscall, once per distinct (number, shape). + /// + /// `warn` when the decoder flagged the request as unsupported or unknown, + /// `debug` when it merely refused the arguments (`result` is the errno the + /// guest gets); repeats of the same row are shown only at `trace`. + fn report_unsupported_syscall( + &self, + nr: usize, + shape: unsupported::Shape, + result: Result<(), Errno>, + ) { + let name = unsupported::syscall_name(nr); + // Only the fallback shape built in `do_syscall` starts this way; every + // other shape came from the decoder naming what it does not support. + let flagged = !shape.starts_with("decode failed"); + let result = match result { + Ok(()) => "decoded", + Err(errno) => errno.as_str(), + }; + match unsupported::record(Some(nr), &shape) { + unsupported::Sighting::First if flagged => { + litebox_util_log::warn!( + nr:? = nr, name:% = name, what:% = shape, result:% = result, pid:? = self.pid; + "unsupported syscall (first sighting; repeats logged at trace level)" + ); + } + unsupported::Sighting::First => { + litebox_util_log::debug!( + nr:? = nr, name:% = name, what:% = shape, result:% = result, pid:? = self.pid; + "undecodable syscall arguments (first sighting; repeats logged at trace level)" + ); + } + unsupported::Sighting::Repeat(count) => { + litebox_util_log::trace!( + nr:? = nr, name:% = name, what:% = shape, result:% = result, count:? = count, + pid:? = self.pid; + "unsupported syscall (repeat)" + ); + } + } + } +} + /// Global shim state, shared across all tasks. struct GlobalState { /// The platform instance used throughout the shim. @@ -1185,8 +2157,6 @@ struct GlobalState { litebox: litebox::LiteBox, /// The page manager for managing virtual memory. pm: litebox::mm::PageManager, - /// The futex manager for handling futex operations. - futex_manager: FutexManager, /// The anonymous pipe implementation. pipes: Pipes, /// The network subsystem. @@ -1198,8 +2168,52 @@ struct GlobalState { next_thread_id: core::sync::atomic::AtomicI32, /// UNIX domain socket address table unix_addr_table: litebox::sync::RwLock>, - /// Per-process collection of ELF patching state for runtime syscall rewriting. - elf_patch_cache: litebox::sync::Mutex, + /// Unix98 pseudoterminal namespace shared by every guest task. + pty_registry: Arc>, + /// Guest ELF images recorded at map time, for fault symbolization. Grows + /// monotonically (never pruned on unmap) and survives the mapping fd's + /// close, unlike [`Self::elf_patch_cache`]. See `Task::find_guest_image`. + guest_images: litebox::sync::Mutex>, + /// Guest ELF images placed by the shim's own loader, for fault + /// symbolization; see [`LoadedImage`] for why these are not in + /// [`Self::guest_images`]. + loaded_images: litebox::sync::Mutex>, + /// Stable shared-page identities keyed by filesystem device/inode, so aliases opened through + /// different file descriptions still converge on one backing object. + shared_file_backings: litebox::sync::Mutex< + Platform, + alloc::collections::BTreeMap<(usize, usize), litebox::mm::linux::SharedFutexBacking>, + >, + /// Handle to the `/proc` backend mounted by [`LinuxShimBuilder::default_fs`], if any -- + /// `None` when the shim was built with a filesystem that doesn't mount one. + /// `Task::set_task_comm` publishes the guest task's identity here as it becomes known. + proc_handle: Option>, + /// Handle to the `/dev/fb0` framebuffer mounted by [`LinuxShimBuilder::default_fs`], if any + /// -- `None` when the shim was built with a filesystem that doesn't mount one. + /// `syscalls::file::Task::sys_ioctl` services `FBIO*` requests through this handle directly, + /// rather than by routing through the generic `FS` backend trait: the ioctl structs + /// (`FbVarScreeninfo`/`FbFixScreeninfo`) live on the concrete [`litebox::fs::devices::Framebuffer`] + /// type, not on the `FileSystem`/`Backend` traits, so there's no generic path from an `FS`-typed + /// fd to them. + framebuffer: Option>, + /// Handle to the `/dev/input` event-device registry mounted by + /// [`LinuxShimBuilder::default_fs`], if any -- `None` when the shim was built with a + /// filesystem that doesn't mount one. `sys_read`/`sys_ioctl`/poll intercept evdev fds + /// through this, and the runner injects input events into it. + input_registry: Option>, + /// Real termios state for the process's controlling terminal (shared by stdin/stdout/stderr, + /// like a real Linux `tty_struct`), as read by `TCGETS` and written by `TCSETS`. + termios: litebox::sync::Mutex, + /// Foreground process group of the host-backed stdin/stdout/stderr terminal. This belongs to + /// the terminal, not to any guest process; pseudoterminals keep the same state in `PtyState`. + stdio_foreground_pgid: core::sync::atomic::AtomicI32, + /// Parent/child relationships and exit statuses for every guest process, so that `fork`ed + /// children can be reaped by `wait4`. + processes: syscalls::process::ProcessTable, + /// Serializes the swap-operate-restore sequence that gives each guest process its own program + /// break on top of the single break [`litebox::mm::PageManager`] tracks. See + /// `Task::sys_brk`. + brk_lock: litebox::sync::Mutex, } struct Task { @@ -1213,8 +2227,12 @@ struct Task { /// Thread ID tid: i32, /// Task credentials. These are set per task but are Arc'd to save space - /// since most tasks never change their credentials. - credentials: Arc, + /// since most tasks never change their credentials. `setuid`/`setgid` + /// replace the `Arc` rather than mutate through it, so a thread that + /// still shares the old one (e.g. a sibling from `clone`) is unaffected + /// -- matching the raw syscall, which (unlike glibc's thread-broadcasting + /// wrapper) only ever updates the calling thread's credentials. + credentials: RefCell>, /// Command name (usually the executable name, excluding the path) comm: Cell<[u8; litebox_common_linux::TASK_COMM_LEN]>, /// Filesystem state. `RefCell` to support `unshare` in the future. @@ -1223,6 +2241,12 @@ struct Task { files: RefCell>>, /// Signal state signals: syscalls::signal::SignalState, + /// The guest stack pointer as of the most recent entry into the shim. + /// + /// Needed because a snapshot of this task's memory can be taken at any blocking point, not + /// just at a syscall that was handed a `PtRegs`; see + /// `syscalls::process::Task::save_address_space` for what it is used for. + guest_sp: Cell, } impl Drop for Task { @@ -1249,20 +2273,18 @@ mod test_utils { files.initialize_stdio_in_shared_descriptors_table(&self); Task { wait_state: wait::WaitState::new(self.platform), - thread: syscalls::process::ThreadState::new_process(pid), + thread: syscalls::process::ThreadState::new_process(pid, 0), pid, ppid: 0, tid: pid, - credentials: Arc::new(syscalls::process::Credentials { - uid: 0, - euid: 0, - gid: 0, - egid: 0, - }), + credentials: RefCell::new(Arc::new(syscalls::process::Credentials::new( + 0, 0, 0, 0, + ))), comm: Cell::new(*b"test\0\0\0\0\0\0\0\0\0\0\0\0"), fs: Arc::new(syscalls::file::FsState::new()).into(), files: files.into(), signals: syscalls::signal::SignalState::new_process(), + guest_sp: Cell::new(0), global: self, } } @@ -1282,11 +2304,12 @@ mod test_utils { pid: self.pid, ppid: self.ppid, tid, - credentials: self.credentials.clone(), + credentials: RefCell::new(self.credentials.borrow().clone()), comm: self.comm.clone(), fs: self.fs.clone(), files: self.files.clone(), signals: self.signals.clone_for_new_task(), + guest_sp: Cell::new(0), }; Some(task) } diff --git a/litebox_shim_linux/src/loader/auxv.rs b/litebox_shim_linux/src/loader/auxv.rs index d23b87953d..97e7bf7ec4 100644 --- a/litebox_shim_linux/src/loader/auxv.rs +++ b/litebox_shim_linux/src/loader/auxv.rs @@ -3,7 +3,7 @@ //! Auxiliary vector support. -use crate::{ShimFS, ShimPlatform, Task}; +use crate::{ShimFS, ShimPlatform, Task, syscalls::process::Credentials}; #[allow(non_camel_case_types)] #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, PartialOrd, Ord)] @@ -67,20 +67,31 @@ pub enum AuxKey { pub type AuxVec = alloc::collections::btree_map::BTreeMap; impl Task { - /// Initialize the auxiliary vector with user information and VDSO address. - pub fn init_auxv(&self) -> AuxVec { + /// Initialize the auxiliary vector with the credentials that will be published with this image. + pub fn init_auxv(&self, credentials: &Credentials, secure: bool) -> AuxVec { let mut aux = AuxVec::new(); - let user_info = &self.credentials; - aux.insert(AuxKey::AT_UID, user_info.uid as usize); - aux.insert(AuxKey::AT_EUID, user_info.euid as usize); - aux.insert(AuxKey::AT_GID, user_info.gid as usize); - aux.insert(AuxKey::AT_EGID, user_info.egid as usize); + aux.insert(AuxKey::AT_UID, credentials.uid as usize); + aux.insert(AuxKey::AT_EUID, credentials.euid as usize); + aux.insert(AuxKey::AT_GID, credentials.gid as usize); + aux.insert(AuxKey::AT_EGID, credentials.egid as usize); + aux.insert(AuxKey::AT_SECURE, usize::from(secure)); if let Some(vdso_base) = self.global.platform.get_vdso_address() { aux.insert(AuxKey::AT_SYSINFO_EHDR, vdso_base); } + let (hwcap, hwcap2) = self.global.platform.get_hwcap(); + // `AT_HWCAP`/`AT_HWCAP2` are each a 32-bit-wide Linux kernel ABI concept (the value a + // 32-bit ARM `getauxval` caller would also see); every bit `SystemInfoProvider::get_hwcap` + // implementations set is below bit 32, so this never actually truncates even on a + // 32-bit-`usize` target. + #[allow(clippy::cast_possible_truncation)] + { + aux.insert(AuxKey::AT_HWCAP, hwcap as usize); + aux.insert(AuxKey::AT_HWCAP2, hwcap2 as usize); + } + aux } } diff --git a/litebox_shim_linux/src/loader/elf.rs b/litebox_shim_linux/src/loader/elf.rs index b0449c25b6..78b4f7a60b 100644 --- a/litebox_shim_linux/src/loader/elf.rs +++ b/litebox_shim_linux/src/loader/elf.rs @@ -5,13 +5,23 @@ use alloc::{ffi::CString, vec::Vec}; use litebox::{ - fs::{Mode, OFlags}, + fs::{AccessCredentials, FileStatus}, mm::linux::{CreatePagesFlags, MappingError, PAGE_SIZE}, - utils::{ReinterpretSignedExt, TruncateExt}, + utils::TruncateExt, }; use litebox_common_linux::{MapFlags, errno::Errno, loader::ElfParsedFile}; use thiserror::Error; +/// The loader and the rewriter must name the same word for the guest +/// thread-pointer offset. `litebox_common_linux` cannot depend on the rewriter, +/// so this crate -- which depends on both -- is where the two are held together. +/// A drift here would make the loader publish the offset into the middle of an +/// instruction instead of into the slot the gates read. +const _: () = assert!( + litebox_common_linux::loader::TRAMPOLINE_GUEST_TP_SLOT_OFFSET + == litebox_syscall_rewriter::TRAMPOLINE_GUEST_TP_SLOT_OFFSET +); + use crate::{ UserPtrMut, loader::auxv::{AuxKey, AuxVec}, @@ -24,22 +34,43 @@ use crate::{ShimFS, ShimPlatform, Task}; struct ElfFile<'a, Platform: ShimPlatform, FS: ShimFS> { task: &'a Task, fd: i32, + status: FileStatus, load_high: bool, } impl<'a, Platform: ShimPlatform, FS: ShimFS> ElfFile<'a, Platform, FS> { fn new(task: &'a Task, path: impl litebox::path::Arg) -> Result { - let fd = task - .sys_open(path, OFlags::RDONLY, Mode::empty())? - .reinterpret_as_signed(); - Ok(ElfFile { + let credentials = task.credentials.borrow().clone(); + let access = AccessCredentials::new( + credentials.euid, + credentials.egid, + credentials.supplementary_groups(), + ); + let files = task.files.borrow(); + let (fd, status) = files.fs.open_executable_as(access, path)?; + let fd = files.insert_raw_fd(fd).map_err(|fd| { + let _ = files.fs.close(&fd); + Errno::EMFILE + })?; + let fd = i32::try_from(fd).expect("RLIMIT_NOFILE keeps guest descriptors within i32"); + Ok(Self { task, fd, + status, load_high: false, }) } } +pub(crate) fn read_executable_header( + task: &Task, + path: impl litebox::path::Arg, + header: &mut [u8], +) -> Result { + let file = ElfFile::new(task, path)?; + task.sys_read(file.fd, header, Some(0)) +} + impl Drop for ElfFile<'_, Platform, FS> { fn drop(&mut self) { self.task.sys_close(self.fd).expect("failed to close fd"); @@ -70,7 +101,11 @@ impl litebox_common_linux::loader::ReadAt } fn size(&mut self) -> Result { - Ok(self.task.sys_fstat(self.fd)?.st_size as u64) + // `st_size` is unsigned and pointer-width in the x86-64 `struct stat` + // and a signed 64-bit field in the generic layout aarch64 uses; a + // negative file size is not representable either way. + let size = self.task.sys_fstat(self.fd)?.st_size; + u64::try_from(size).map_err(|_| Errno::EINVAL) } } @@ -93,7 +128,7 @@ impl litebox_common_linux::loader::MapMemory // platform honoring an out-of-range hint. 0 } else { - super::DEFAULT_LOW_ADDR + super::default_low_addr::() }; let mapping_ptr = self .task @@ -190,6 +225,15 @@ pub(crate) struct ElfLoader<'a, Platform: ShimPlatform, FS: ShimFS> { struct FileAndParsed<'a, Platform: ShimPlatform, FS: ShimFS> { file: ElfFile<'a, Platform, FS>, parsed: ElfParsedFile, + /// The path the image was opened with, kept for fault symbolization. + path: alloc::string::String, +} + +/// The `PT_LOAD` span of an ELF, in page-aligned vaddrs relative to its load +/// bias: `lo..hi` covers every loadable segment. +struct LoadSpan { + lo: usize, + hi: usize, } impl<'a, Platform: ShimPlatform, FS: ShimFS> FileAndParsed<'a, Platform, FS> { @@ -197,6 +241,7 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> FileAndParsed<'a, Platform, FS> { task: &'a Task, path: impl litebox::path::Arg, ) -> Result { + let path_name = path.to_rust_str_lossy().into_owned(); let file = ElfFile::new(task, path).map_err(ElfLoaderError::OpenError)?; let mut parsed = litebox_common_linux::loader::ElfParsedFile::parse(&mut &file) .map_err(ElfLoaderError::ParseError)?; @@ -208,7 +253,8 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> FileAndParsed<'a, Platform, FS> { // (UnpatchedBinary error), the runtime patching during mmap will patch // code segments as they are mapped. if syscall_entry_point != 0 { - match parsed.parse_trampoline(&mut &file, syscall_entry_point) { + let guest_tp_slot_offset = task.global.platform.get_guest_tp_slot_offset(); + match parsed.parse_trampoline(&mut &file, syscall_entry_point, guest_tp_slot_offset) { Ok(()) | Err(litebox_common_linux::loader::ElfParseError::UnpatchedBinary) => { // Ok: pre-patched trampoline found, or unpatched binary // that the runtime mmap hook will handle. @@ -217,7 +263,71 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> FileAndParsed<'a, Platform, FS> { } } - Ok(Self { file, parsed }) + Ok(Self { + file, + parsed, + path: path_name, + }) + } + + /// The image's `PT_LOAD` span, read back from its program headers. + /// + /// `ElfParsedFile` keeps its headers private and `MappingInfo` reports + /// only the bias, so the span is re-derived from the file the same way the + /// `mmap` path derives it for shared libraries (`init_elf_patch_state`). + /// Best-effort: an unreadable or degenerate table simply leaves the image + /// unnamed in a fault line. + fn load_span(&self) -> Option { + use litebox_common_linux::loader::ReadAt as _; + use object::elf::{FileHeader64, PT_LOAD, ProgramHeader64}; + use object::endian::LittleEndian; + const ENDIAN: LittleEndian = LittleEndian; + + let mut file = &self.file; + let mut ehdr_buf = [0u8; core::mem::size_of::>()]; + file.read_at(0, &mut ehdr_buf).ok()?; + let (ehdr, _) = object::from_bytes::>(&ehdr_buf).ok()?; + let e_phoff = ehdr.e_phoff.get(ENDIAN); + let e_phentsize = usize::from(ehdr.e_phentsize.get(ENDIAN)); + let e_phnum = usize::from(ehdr.e_phnum.get(ENDIAN)); + if e_phentsize < core::mem::size_of::>() { + return None; + } + let phdrs_size = e_phentsize.checked_mul(e_phnum)?; + if phdrs_size == 0 || phdrs_size > 0x10000 { + return None; + } + let mut phdrs_buf = alloc::vec![0u8; phdrs_size]; + file.read_at(e_phoff, &mut phdrs_buf).ok()?; + + let mut lo = usize::MAX; + let mut hi = 0usize; + for chunk in phdrs_buf.chunks_exact(e_phentsize) { + let Ok((ph, _)) = object::from_bytes::>(chunk) else { + continue; + }; + if ph.p_type.get(ENDIAN) != PT_LOAD { + continue; + } + let start: usize = ph.p_vaddr.get(ENDIAN).trunc(); + let end = start.checked_add(ph.p_memsz.get(ENDIAN).trunc())?; + lo = lo.min(start & !(PAGE_SIZE - 1)); + hi = hi.max(end.checked_next_multiple_of(PAGE_SIZE)?); + } + (lo < hi).then_some(LoadSpan { lo, hi }) + } + + /// Publish where this image landed for fault symbolization. + fn record_loaded(&self, info: &litebox_common_linux::loader::MappingInfo) { + if let Some(span) = self.load_span() { + let base = info.base_addr; + self.file.task.record_loaded_image( + &self.path, + base, + base.wrapping_add(span.lo), + base.wrapping_add(span.hi), + ); + } } /// Load the ELF into guest memory. @@ -247,8 +357,17 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> ElfLoader<'a, Platform, FS> { // Parse the interpreter ELF file, if any. let interp = if let Some(interp_name) = main.parsed.interp(&mut &main.file)? { - // e.g., /lib64/ld-linux-x86-64.so.2 - let mut interp = FileAndParsed::new(task, interp_name)?; + // e.g., /lib64/ld-linux-x86-64.so.2 -- a guest-visible path, which `execve`'s own + // path was too before `resolve_shebang` resolved it. Resolve it the same way: beneath + // the process's `chroot` root, following symlinks with an absolute target restarting + // from that root (Linux's `open_exec` walks `nd->root` for the interpreter exactly as + // for the main image), so a jail loads its own `ld.so` or fails with `ENOENT`, and + // never reaches the interpreter outside it. + let interp_path = task + .resolve_path(interp_name.as_c_str()) + .and_then(|path| task.follow_open_path(path, litebox::fs::OFlags::RDONLY)) + .map_err(ElfLoaderError::OpenError)?; + let mut interp = FileAndParsed::new(task, interp_path)?; // Linux places the ET_EXEC interpreter high so brk can grow above // the fixed-address main image without hitting ld.so. interp.file.load_high = true; @@ -260,6 +379,10 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> ElfLoader<'a, Platform, FS> { Ok(Self { path, main, interp }) } + pub(crate) fn main_status(&self) -> &FileStatus { + &self.main.file.status + } + /// Load an ELF file and prepare the stack for the new process. pub fn load( &mut self, @@ -269,12 +392,19 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> ElfLoader<'a, Platform, FS> { ) -> Result { let global = &self.main.file.task.global; + // This load replaces the address space, so anything recorded for the + // previous image of this process is stale from here on. + self.main.file.task.forget_loaded_images(); + // Load the main ELF file first so that it gets privileged addresses. let info = self.main.load_mapped(global.platform)?; + self.main.record_loaded(&info); // Load the interpreter ELF file, if any. let interp = if let Some(interp) = &mut self.interp { - Some(interp.load_mapped(global.platform)?) + let interp_info = interp.load_mapped(global.platform)?; + interp.record_loaded(&interp_info); + Some(interp_info) } else { None }; @@ -300,13 +430,29 @@ impl<'a, Platform: ShimPlatform, FS: ShimFS> ElfLoader<'a, Platform, FS> { .create_stack_pages(None, length, CreatePagesFlags::empty()) .map_err(ElfLoaderError::MappingError)? }; + // Mapped directly through the page manager rather than through `sys_mmap`, so record it + // as this process's the same way `sys_mmap` would (see `Process::owned_ranges`). + self.main.file.task.record_mapped( + litebox::platform::RawConstPointer::as_usize(&sp), + super::DEFAULT_STACK_SIZE, + ); + // Where each image landed, and where the stack landed: exactly the + // placements a cross-process teardown investigation needs, and + // invisible in the syscall trace (these are shim-internal mappings). + litebox_util_log::debug!( + main_base:? = info.base_addr, + interp_base:? = interp.as_ref().map(|i| i.base_addr), + stack:? = litebox::platform::RawConstPointer::as_usize(&sp), + stack_size:? = super::DEFAULT_STACK_SIZE; + "loaded program image" + ); let mut stack = UserStack::::new( UserPtrMut::from_platform_ptr::(sp), super::DEFAULT_STACK_SIZE, ) .ok_or(ElfLoaderError::InvalidStackAddr)?; stack - .init(argv, envp, aux) + .init(argv, envp, aux, global.platform) .ok_or(ElfLoaderError::InvalidStackAddr)?; Ok(ElfLoadInfo { @@ -367,12 +513,32 @@ mod tests { const PROGRAM_HEADER_SIZE_U16: u16 = 56; const ET_EXEC: u16 = 2; const ET_DYN: u16 = 3; - const EM_X86_64: u16 = 62; + /// The synthetic ELFs below must claim the host's own machine, because the + /// loader rejects any other with `UnsupportedType` before it reaches the + /// placement logic under test. + const EM_HOST: u16 = if cfg!(target_arch = "x86_64") { + 62 // EM_X86_64 + } else { + 183 // EM_AARCH64 + }; const PT_LOAD: u32 = 1; const PT_INTERP: u32 = 3; const PF_X: u32 = 1; const PF_R: u32 = 4; - const EXEC_LOAD_ADDR: u64 = 0x400000; + /// Where the synthetic `ET_EXEC` asks to be loaded. + /// + /// Linux's customary `0x400000` is not usable on every host: an arm64 Mach-O + /// process reserves the first 4 GiB as `__PAGEZERO`, so a fixed mapping + /// there is refused outright. Anchoring to the host's own floor is still not + /// enough, because the host binary is itself mapped just above that floor -- + /// this test process's own code sits within the first few MiB of it -- so a + /// small offset lands inside the running image and the fixed mapping fails. + /// The gap below is therefore large enough to clear any plausible host + /// image, while staying far below `TASK_ADDR_MAX` on every host, since what + /// this test asserts is that the *interpreter* lands in the high half. + const EXEC_LOAD_ADDR: u64 = + >::TASK_ADDR_MIN as u64 + + 0x8_0000_0000; const INTERP_PATH_OFFSET: usize = 0x200; const INTERP_PATH: &[u8] = b"/ld.so\0"; @@ -404,7 +570,7 @@ mod tests { buf.extend_from_slice(&[2, 1, 1, 0]); buf.extend_from_slice(&[0; 8]); push_u16(buf, elf_type); - push_u16(buf, EM_X86_64); + push_u16(buf, EM_HOST); push_u32(buf, 1); push_u64(buf, entry); push_u64(buf, u64::from(ELF_HEADER_SIZE_U16)); @@ -493,6 +659,7 @@ mod tests { #[test] fn et_exec_interpreter_loads_top_down_above_low_heap() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = crate::syscalls::tests::init_platform(None); write_file(&task, "/main", &minimal_elf(ET_EXEC, Some(INTERP_PATH))); write_file(&task, "/ld.so", &minimal_elf(ET_DYN, None)); @@ -526,5 +693,27 @@ mod tests { crate::loader::DEFAULT_LOW_ADDR, addr_max / 2, ); + + // Release both images before returning. Every test in this binary shares + // one host address space, but each builds its own task with its own VMM, + // and a VMM models only its own mappings -- so anything this test leaves + // mapped is invisible to the next test's placement search and collides + // with whatever it picks. That is easy to miss on a host whose guest + // range sits well clear of the host's own image; on arm64 macOS both + // live above the 4 GiB `__PAGEZERO` floor, so the collision is routine. + // Each synthetic image maps exactly one PT_LOAD page (`minimal_elf` + // sets filesz == memsz == PAGE_SIZE). Do NOT derive the length from + // `brk`: on a platform that requires syscall rewriting, `load_mapped` + // pushes brk DEFAULT_RESERVED_SPACE_SIZE (16 MiB) past the image + // without mapping that space, so a brk-derived munmap overshoots -- + // the top-down interpreter ends exactly at TASK_ADDR_MAX, which on + // Linux x86-64 is the host TASK_SIZE (munmap EINVAL panics + // deallocate_pages), and Windows' region walk asserts on the + // never-committed tail. + let exec_start = usize::try_from(EXEC_LOAD_ADDR).expect("load address fits usize"); + task.sys_munmap(UserPtrMut::from_usize(exec_start), PAGE_SIZE) + .expect("main image should unmap"); + task.sys_munmap(UserPtrMut::from_usize(interp.base_addr), PAGE_SIZE) + .expect("interpreter image should unmap"); } } diff --git a/litebox_shim_linux/src/loader/mod.rs b/litebox_shim_linux/src/loader/mod.rs index a7e370cd38..19a407f1be 100644 --- a/litebox_shim_linux/src/loader/mod.rs +++ b/litebox_shim_linux/src/loader/mod.rs @@ -2,8 +2,14 @@ // Licensed under the MIT license. //! This module contains the loader for the LiteBox shim. +//! +//! Nothing in here is architecture-specific: segment mapping, the auxiliary +//! vector and the initial stack layout are all defined by the generic +//! System V/Linux ABI, and the ELF machine type is carried by the image rather +//! than assumed. The module is therefore built for every architecture the rest +//! of the shim supports, so an aarch64 host gets the same loader an x86-64 host +//! does. -#![cfg(target_arch = "x86_64")] pub mod auxv; pub mod elf; mod stack; @@ -12,4 +18,24 @@ pub(crate) const DEFAULT_STACK_SIZE: usize = 8 * 1024 * 1024; // 8 MB /// A default low address is used for the binary (which grows upwards) to avoid /// conflicts with the kernel's memory mappings (which grows downwards). +/// +/// This is a preference, not a floor. Use [`default_low_addr`] rather than the +/// constant directly: on a host whose lowest mappable address is higher than +/// this, asking for it is not merely ignored but rejected. pub(crate) const DEFAULT_LOW_ADDR: usize = 0x1000_0000; + +/// [`DEFAULT_LOW_ADDR`], raised to the host's lowest mappable address. +/// +/// An arm64 Mach-O process reserves the first 4 GiB as `__PAGEZERO`, which puts +/// `TASK_ADDR_MIN` above the constant. A mapping requested below that floor +/// fails with `BelowMinAddress`, surfaced to the guest as `EPERM`, so on such a +/// host the bare constant makes every image -- including a +/// position-independent one, which is otherwise free to land anywhere -- fail +/// to load. +pub(crate) fn default_low_addr() -> usize { + DEFAULT_LOW_ADDR.max( + >::TASK_ADDR_MIN, + ) +} diff --git a/litebox_shim_linux/src/loader/stack.rs b/litebox_shim_linux/src/loader/stack.rs index bcd10bce70..7d5dc612f6 100644 --- a/litebox_shim_linux/src/loader/stack.rs +++ b/litebox_shim_linux/src/loader/stack.rs @@ -119,12 +119,22 @@ impl UserStack { /// Returns the offsets of the strings in the stack. /// Returns `None` if the stack has insufficient space. fn push_cstrings(&mut self, vals: &[CString]) -> Option> { - let mut envp = Vec::with_capacity(vals.len()); - for val in vals { + // Push in reverse so that -- with the stack growing down -- `vals[0]` + // lands at the LOWEST address and the whole block is contiguous in + // increasing address order. That is the exact layout the Linux kernel + // produces, and the one libuv's `uv_setup_args` relies on: it walks + // `argv[0]..argv[n]` then `environ[0]..` requiring each string to abut + // the previous at a higher address, and sizes the process-title buffer + // from that contiguous span. Pushing forward reversed each block, so the + // walk broke immediately and libuv computed a garbage `process_title.len` + // -- and the first `process.title = ...` (which `npm` does at startup) + // then `memset`s that bogus length and SIGSEGVs. + let mut ptrs = alloc::vec![0usize; vals.len()]; + for (i, val) in vals.iter().enumerate().rev() { self.push_cstring(val)?; - envp.push(self.pos); + ptrs[i] = self.pos; } - Some(envp) + Some(ptrs) } /// Push a vector of stack pointers to the stack. @@ -165,6 +175,7 @@ impl UserStack { argv: Vec, env: Vec, mut aux: BTreeMap, + platform: &impl litebox::platform::CrngProvider, ) -> Option<()> { // end markers self.pos = self.pos.checked_sub(size_of::())?; @@ -174,11 +185,10 @@ impl UserStack { let envp = self.push_cstrings(&env)?; let argvp = self.push_cstrings(&argv)?; - // TODO: generate a random value - self.push_bytes(&[ - 0xDE, 0xAD, 0xBE, 0xEF, 0xDE, 0xAD, 0xBE, 0xEF, 0xDE, 0xAD, 0xBE, 0xEF, 0xDE, 0xAD, - 0xBE, 0xEF, - ])?; + // AT_RANDOM: 16 bytes of real randomness (libc's stack-canary seed). + let mut random_bytes = [0u8; 16]; + <_ as litebox::platform::CrngProvider>::fill_bytes_crng(platform, &mut random_bytes); + self.push_bytes(&random_bytes)?; aux.insert(AuxKey::AT_RANDOM, self.stack_top.as_usize() + self.pos); let align_down = |pos: usize, alignment: usize| -> usize { diff --git a/litebox_shim_linux/src/syscalls/epoll.rs b/litebox_shim_linux/src/syscalls/epoll.rs index ec656593d7..ee9dc62fde 100644 --- a/litebox_shim_linux/src/syscalls/epoll.rs +++ b/litebox_shim_linux/src/syscalls/epoll.rs @@ -15,7 +15,10 @@ use litebox::{ polling::{Pollee, TryOpError}, wait::{WaitContext, WaitError, Waker}, }, - fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry, TypedFd}, + fd::{ + EntryHandle, EntryIdentity, FdEnabledSubsystem, FdEnabledSubsystemEntry, TypedFd, + WeakEntryHandle, + }, utils::ReinterpretUnsignedExt, }; use litebox_common_linux::{EpollEvent, EpollOp, errno::Errno}; @@ -23,6 +26,17 @@ use litebox_common_linux::{EpollEvent, EpollOp, errno::Errno}; use super::file::FilesState; use crate::{GlobalState, ShimFS, ShimPlatform}; +/// Serializes every nested-epoll `epoll_ctl(ADD)` across the whole process, mirroring real +/// Linux's `epmutex`. Cycle detection (walking the nested-epoll DAG) and the edge insertion it +/// guards have to happen as one atomic step: checking and inserting under separate locks lets two +/// concurrent adds that each individually look cycle-free still complete a cycle together (e.g. +/// thread 1 adds B into A, thread 2 concurrently adds A into B; neither sees the other's +/// not-yet-committed edge during its own check). A single global lock removes the race by only +/// ever allowing one such check-then-insert to be in flight anywhere in the process. It is not +/// taken for plain (non-nested) adds or for readiness polling, so the common case pays nothing +/// for it. +static EPOLL_NEST_LOCK: spin::Mutex<()> = spin::Mutex::new(()); + pub(crate) struct EpollSubsystem( core::marker::PhantomData<(Platform, FS)>, ); @@ -43,72 +57,146 @@ bitflags::bitflags! { } pub(crate) enum EpollDescriptor { - Eventfd(Arc>>), - Epoll(Arc>>), - File(Arc>), - Socket(Arc>), - Pipe(Arc>), - Unix(Arc>>), + Eventfd(WeakEntryHandle>), + Epoll(WeakEntryHandle>), + File(WeakEntryHandle), + Socket(WeakEntryHandle>), + Pipe(WeakEntryHandle>), + Unix(WeakEntryHandle>), + Netlink(WeakEntryHandle>), + Inotify(WeakEntryHandle>), +} + +impl Clone for EpollDescriptor { + fn clone(&self) -> Self { + match self { + Self::Eventfd(file) => Self::Eventfd(file.clone()), + Self::Epoll(file) => Self::Epoll(file.clone()), + Self::File(file) => Self::File(file.clone()), + Self::Socket(socket) => Self::Socket(socket.clone()), + Self::Pipe(pipe) => Self::Pipe(pipe.clone()), + Self::Unix(unix) => Self::Unix(unix.clone()), + Self::Netlink(netlink) => Self::Netlink(netlink.clone()), + Self::Inotify(inotify) => Self::Inotify(inotify.clone()), + } + } } impl EpollDescriptor { - pub fn try_from(files: &FilesState, raw_fd: usize) -> Result { + pub fn try_from( + global: &GlobalState, + files: &FilesState, + raw_fd: usize, + ) -> Result { let rds = files.raw_descriptor_store.read(); if let Ok(fd) = rds.fd_from_raw_integer::(raw_fd) { - return Ok(EpollDescriptor::File(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::File(handle.downgrade())); } if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) { - return Ok(EpollDescriptor::Socket(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Socket(handle.downgrade())); } if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) { - return Ok(EpollDescriptor::Pipe(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Pipe(handle.downgrade())); } if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) { - return Ok(EpollDescriptor::Eventfd(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Eventfd(handle.downgrade())); } if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) { - return Ok(EpollDescriptor::Epoll(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Epoll(handle.downgrade())); } if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) { - return Ok(EpollDescriptor::Unix(fd)); + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Unix(handle.downgrade())); + } + if let Ok(fd) = + rds.fd_from_raw_integer::>(raw_fd) + { + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Netlink(handle.downgrade())); + } + if let Ok(fd) = rds.fd_from_raw_integer::>(raw_fd) + { + let handle = global + .litebox + .descriptor_table() + .entry_handle(&fd) + .ok_or(Errno::EBADF)?; + return Ok(Self::Inotify(handle.downgrade())); } Err(Errno::EBADF) } -} -enum DescriptorRef { - Eventfd(Weak>>), - Epoll(Weak>>), - File(Weak>), - Socket(Weak>), - Pipe(Weak>), - Unix(Weak>>), -} + fn identity(&self) -> EntryIdentity { + match self { + Self::Eventfd(file) => file.identity(), + Self::Epoll(file) => file.identity(), + Self::File(file) => file.identity(), + Self::Socket(socket) => socket.identity(), + Self::Pipe(pipe) => pipe.identity(), + Self::Unix(unix) => unix.identity(), + Self::Netlink(netlink) => netlink.identity(), + Self::Inotify(inotify) => inotify.identity(), + } + } -impl DescriptorRef { - fn from(value: &EpollDescriptor) -> Self { - match value { - EpollDescriptor::Eventfd(file) => Self::Eventfd(Arc::downgrade(file)), - EpollDescriptor::Epoll(file) => Self::Epoll(Arc::downgrade(file)), - EpollDescriptor::File(file) => Self::File(Arc::downgrade(file)), - EpollDescriptor::Socket(socket) => Self::Socket(Arc::downgrade(socket)), - EpollDescriptor::Pipe(pipe) => Self::Pipe(Arc::downgrade(pipe)), - EpollDescriptor::Unix(unix) => Self::Unix(Arc::downgrade(unix)), + fn is_alive(&self) -> bool { + match self { + Self::Eventfd(file) => file.upgrade().is_some(), + Self::Epoll(file) => file.upgrade().is_some(), + Self::File(file) => file.upgrade().is_some(), + Self::Socket(socket) => socket.upgrade().is_some(), + Self::Pipe(pipe) => pipe.upgrade().is_some(), + Self::Unix(unix) => unix.upgrade().is_some(), + Self::Netlink(netlink) => netlink.upgrade().is_some(), + Self::Inotify(inotify) => inotify.upgrade().is_some(), } } - fn upgrade(&self) -> Option> { + pub(crate) fn is_epoll_identity(&self, identity: EntryIdentity) -> bool { + matches!(self, Self::Epoll(epoll) if epoll.identity() == identity) + } + + fn epoll_handle(&self) -> Option>> { match self { - DescriptorRef::Eventfd(eventfd) => eventfd.upgrade().map(EpollDescriptor::Eventfd), - DescriptorRef::Epoll(epoll) => epoll.upgrade().map(EpollDescriptor::Epoll), - DescriptorRef::File(file) => file.upgrade().map(EpollDescriptor::File), - DescriptorRef::Socket(socket) => socket.upgrade().map(EpollDescriptor::Socket), - DescriptorRef::Pipe(pipe) => pipe.upgrade().map(EpollDescriptor::Pipe), - DescriptorRef::Unix(unix) => unix.upgrade().map(EpollDescriptor::Unix), + Self::Epoll(epoll) => epoll.upgrade(), + _ => None, } } } @@ -122,6 +210,22 @@ impl EpollDescriptor { mask: Events, observer: Option>>, ) -> Option { + // `/dev/input/event*` fds have real queue-backed readiness through the input registry + // (X11/libinput poll these and only read after `IN` -- dummy always-ready would spin + // them on empty reads). Register before checking the queue: an input record arriving + // between a check and a later registration would otherwise leave Xorg asleep until the + // next record, making every click or key appear one event late. + if let EpollDescriptor::File(file) = self + && let Some(file) = file.upgrade() + && let Some(registry) = global.input_registry.as_ref() + && let Ok(minor) = file.with_metadata(|meta: &super::file::InputEventMinor| meta.minor) + { + if let Some(observer) = observer { + registry.register_observer(minor, observer, mask); + } + let events = registry.check_io_events(minor)?; + return Some(events & (mask | Events::ALWAYS_POLLED)); + } let poll = |iop: &dyn IOPollable| { if let Some(observer) = observer { iop.register_observer(observer, mask); @@ -130,43 +234,168 @@ impl EpollDescriptor { }; match self { EpollDescriptor::Eventfd(fd) => { - let handle = global.litebox.descriptor_table().entry_handle(fd)?; + let handle = fd.upgrade()?; + Some(handle.with_entry(|entry| poll(entry))) + } + EpollDescriptor::Epoll(fd) => { + let handle = fd.upgrade()?; Some(handle.with_entry(|entry| poll(entry))) } - EpollDescriptor::Epoll(_file) => unimplemented!(), EpollDescriptor::File(file) => { - // TODO: File polling returns dummy events for now, but distinguish stdio enough for REPLs. - let events = match global - .litebox - .descriptor_table() - .with_metadata(file, |stream: &litebox::platform::StdioStream| *stream) - { - Ok(litebox::platform::StdioStream::Stdin) => Events::IN, - Ok( - litebox::platform::StdioStream::Stdout - | litebox::platform::StdioStream::Stderr, - ) - | Err(_) => Events::OUT, - }; + let file = file.upgrade()?; + // Real files in general still get dummy "always ready" events -- only stdin has + // a real, epoll-observable readiness signal (see `StdioProvider::stdin_pollable` + // and `litebox::platform::StdinPump`); stdout/stderr writes to a real terminal + // essentially never block in practice, so `Events::OUT` dummy readiness for them + // remains a reasonable approximation. + let events = + match file.with_metadata(|stream: &litebox::platform::StdioStream| *stream) { + Ok(litebox::platform::StdioStream::Stdin) => { + match global.platform.stdin_pollable() { + Some(pollable) => poll(pollable), + // Platform can't distinguish real readiness: fall back to the + // pre-existing dummy "always ready" behavior. + None => Events::IN, + } + } + Ok( + litebox::platform::StdioStream::Stdout + | litebox::platform::StdioStream::Stderr, + ) + | Err(_) => Events::OUT, + }; Some(events & mask) } EpollDescriptor::Socket(fd) => { - let proxy = match global.get_proxy(fd) { - Ok(p) => p, - Err(e) => { - log_unsupported!("epoll poll with socket fd: {:?}", e); - return None; - } - }; - Some(poll(&proxy)) + let handle = fd.upgrade()?; + let proxy = handle + .with_metadata(|proxy: &super::net::SocketProxy| proxy.0.clone()) + .ok()?; + Some(poll(proxy.as_ref())) + } + EpollDescriptor::Pipe(fd) => { + let handle = fd.upgrade()?; + Some(global.pipes.with_iopollable_handle(&handle, poll)) } - EpollDescriptor::Pipe(fd) => global.with_linux_pipe_iopollable(fd, poll).ok(), EpollDescriptor::Unix(fd) => { - let handle = global.litebox.descriptor_table().entry_handle(fd)?; + let handle = fd.upgrade()?; + Some(handle.with_entry(|entry| poll(entry))) + } + EpollDescriptor::Netlink(fd) => { + let handle = fd.upgrade()?; + Some(handle.with_entry(|entry| poll(entry))) + } + EpollDescriptor::Inotify(fd) => { + let handle = fd.upgrade()?; Some(handle.with_entry(|entry| poll(entry))) } } } + + /// Resolves where a transient observer must be removed before registering it. + /// + /// Most registrations live in the open file description and can be found again through this + /// descriptor's weak handle. Input queues and stdin are different: their subjects outlive the + /// filesystem entry. Remember those special routes now so closing the last descriptor while + /// `poll` is asleep cannot strand a dead observer in a long-lived subject. + fn transient_registration_target( + &self, + global: &GlobalState, + ) -> PollRegistrationTarget { + if let Self::File(file) = self + && let Some(file) = file.upgrade() + { + if global.input_registry.is_some() + && let Ok(minor) = + file.with_metadata(|meta: &super::file::InputEventMinor| meta.minor) + { + return PollRegistrationTarget::InputDevice(minor); + } + if let Ok(litebox::platform::StdioStream::Stdin) = + file.with_metadata(|stream: &litebox::platform::StdioStream| *stream) + && global.platform.stdin_pollable().is_some() + { + return PollRegistrationTarget::Stdin; + } + } + PollRegistrationTarget::Descriptor(self.clone()) + } + + /// Removes a registration previously made by [`Self::poll`]. The descriptor only keeps weak + /// open-file-description handles, so cleanup never extends the lifetime of the polled object. + fn unregister_observer( + &self, + global: &GlobalState, + observer: Weak>, + ) { + // Input devices bypass the filesystem's dummy readiness path and register directly on the + // queue owned by InputRegistry, so their cleanup must take that same path. + if let EpollDescriptor::File(file) = self + && let Some(file) = file.upgrade() + && let Some(registry) = global.input_registry.as_ref() + && let Ok(minor) = file.with_metadata(|meta: &super::file::InputEventMinor| meta.minor) + { + registry.unregister_observer(minor, observer); + return; + } + + let unregister = |iop: &dyn IOPollable| { + iop.unregister_observer(observer.clone()); + }; + match self { + EpollDescriptor::Eventfd(fd) => { + if let Some(handle) = fd.upgrade() { + handle.with_entry(|entry| unregister(entry)); + } + } + EpollDescriptor::Epoll(fd) => { + if let Some(handle) = fd.upgrade() { + handle.with_entry(|entry| unregister(entry)); + } + } + EpollDescriptor::File(file) => { + let Some(file) = file.upgrade() else { + return; + }; + if let Ok(litebox::platform::StdioStream::Stdin) = + file.with_metadata(|stream: &litebox::platform::StdioStream| *stream) + && let Some(pollable) = global.platform.stdin_pollable() + { + unregister(pollable); + } + } + EpollDescriptor::Socket(fd) => { + let Some(handle) = fd.upgrade() else { + return; + }; + if let Ok(proxy) = handle + .with_metadata(|proxy: &super::net::SocketProxy| proxy.0.clone()) + { + unregister(proxy.as_ref()); + } + } + EpollDescriptor::Pipe(fd) => { + if let Some(handle) = fd.upgrade() { + global.pipes.with_iopollable_handle(&handle, unregister); + } + } + EpollDescriptor::Unix(fd) => { + if let Some(handle) = fd.upgrade() { + handle.with_entry(|entry| unregister(entry)); + } + } + EpollDescriptor::Netlink(fd) => { + if let Some(handle) = fd.upgrade() { + handle.with_entry(|entry| unregister(entry)); + } + } + EpollDescriptor::Inotify(fd) => { + if let Some(handle) = fd.upgrade() { + handle.with_entry(|entry| unregister(entry)); + } + } + } + } } pub(crate) struct EpollFile { @@ -210,16 +439,16 @@ impl EpollFile { pub(crate) fn epoll_ctl( &self, global: &GlobalState, + self_fd: &Arc>>, op: EpollOp, fd: u32, file: &EpollDescriptor, event: Option, ) -> Result<(), Errno> { match op { - EpollOp::EpollCtlAdd => self.add_interest(global, fd, file, event.unwrap()), + EpollOp::EpollCtlAdd => self.add_interest(global, self_fd, fd, file, event.unwrap()), EpollOp::EpollCtlMod => { - log_unsupported!("epoll_ctl mod"); - Err(Errno::EINVAL) + self.mod_interest(global, fd, file, event.ok_or(Errno::EINVAL)?) } EpollOp::EpollCtlDel => { let mut interests = self.interests.lock(); @@ -234,14 +463,34 @@ impl EpollFile { fn add_interest( &self, global: &GlobalState, + self_fd: &Arc>>, fd: u32, file: &EpollDescriptor, event: EpollEvent, ) -> Result<(), Errno> { + // A cycle can only be formed by nesting one epoll inside another, so only that case needs + // the global lock; a plain fd add can't create one and stays as cheap as before. The guard + // is held across both the cycle check and the insert below -- see `EPOLL_NEST_LOCK` for why + // splitting those into separate critical sections would reopen the race this closes. + let _nest_guard = matches!(file, EpollDescriptor::Epoll(_)).then(|| EPOLL_NEST_LOCK.lock()); + if let Some(inner_handle) = file.epoll_handle() { + let self_handle = global + .litebox + .descriptor_table() + .entry_handle(self_fd) + .ok_or(Errno::EBADF)?; + if self_handle.identity() == inner_handle.identity() { + return Err(Errno::EINVAL); + } + if Self::nested_epoll_reaches(&self_handle, &inner_handle, 1)? { + return Err(Errno::ELOOP); + } + } + let mut interests = self.interests.lock(); let key = EpollEntryKey::new(fd, file); if let Some(entry) = interests.get(&key) - && entry.desc.upgrade().is_some() + && entry.desc.is_alive() { return Err(Errno::EEXIST); } @@ -250,7 +499,7 @@ impl EpollFile { let mask = Events::from_bits_truncate(event.events); let entry = EpollEntry::new( - DescriptorRef::from(file), + file.clone(), mask, EpollFlags::from_bits_truncate(event.events), event.data, @@ -267,7 +516,41 @@ impl EpollFile { Ok(()) } - #[expect(dead_code, reason = "currently unused, but might want to use soon")] + /// Returns whether `self_fd` is reachable by following already-registered nested-epoll + /// interests starting at `fd`, i.e. whether accepting `fd` as a new interest of `self_fd` + /// would close a cycle. + /// + /// Must be called with `EPOLL_NEST_LOCK` held. Under that lock, every edge in the existing + /// nested-epoll graph got there by passing this same check, so the graph is acyclic by + /// induction going in -- the walk below can therefore only ever revisit `self_fd` itself + /// (caught up front by open-file-description identity, before `self_fd`'s own entry is ever + /// locked), never an intermediate node, so it can't re-lock an entry it is already holding on + /// this call stack. Depth is also capped, mirroring real Linux's nesting limit, so a long + /// acyclic chain can't blow the stack either. + fn nested_epoll_reaches( + self_fd: &EntryHandle>, + fd: &EntryHandle>, + depth: u32, + ) -> Result { + const MAX_NESTED_EPOLL_DEPTH: u32 = 5; + if self_fd.identity() == fd.identity() { + return Ok(true); + } + if depth > MAX_NESTED_EPOLL_DEPTH { + return Err(Errno::ELOOP); + } + fd.with_entry(|entry: &Self| { + for nested in entry.interests.lock().values() { + if let Some(inner_fd) = nested.desc.epoll_handle() + && Self::nested_epoll_reaches(self_fd, &inner_fd, depth + 1)? + { + return Ok(true); + } + } + Ok(false) + }) + } + fn mod_interest( &self, global: &GlobalState, @@ -284,7 +567,7 @@ impl EpollFile { let mut interests = self.interests.lock(); let key = EpollEntryKey::new(fd, file); let entry = interests.get(&key).ok_or(Errno::ENOENT)?; - if entry.desc.upgrade().is_none() { + if !entry.desc.is_alive() { // The file descriptor is closed, remove the entry interests.remove(&key); return Err(Errno::ENOENT); @@ -326,27 +609,37 @@ impl EpollFile { super::common_functions_for_file_status!(); } +impl IOPollable for EpollFile { + fn check_io_events(&self) -> Events { + if self.ready.entries.lock().is_empty() { + Events::empty() + } else { + Events::IN + } + } + + fn register_observer(&self, observer: Weak>, mask: Events) { + self.ready.pollee.register_observer(observer, mask); + } + + fn unregister_observer(&self, observer: Weak>) { + self.ready.pollee.unregister_observer(observer); + } +} + #[derive(PartialEq, Eq, PartialOrd, Ord)] -struct EpollEntryKey(u32, usize); +struct EpollEntryKey(u32, EntryIdentity); impl EpollEntryKey { fn new( fd: u32, desc: &EpollDescriptor, ) -> Self { - let ptr = match desc { - EpollDescriptor::Eventfd(file) => Arc::as_ptr(file).addr(), - EpollDescriptor::Epoll(file) => Arc::as_ptr(file).addr(), - EpollDescriptor::File(file) => Arc::as_ptr(file).addr(), - EpollDescriptor::Socket(socket_fd) => Arc::as_ptr(socket_fd).addr(), - EpollDescriptor::Pipe(pipe_fd) => Arc::as_ptr(pipe_fd).addr(), - EpollDescriptor::Unix(unix) => Arc::as_ptr(unix).addr(), - }; - Self(fd, ptr) + Self(fd, desc.identity()) } } struct EpollEntry { - desc: DescriptorRef, + desc: EpollDescriptor, inner: litebox::sync::Mutex, ready: Arc>, is_ready: AtomicBool, @@ -362,7 +655,7 @@ struct EpollEntryInner { impl EpollEntry { fn new( - desc: DescriptorRef, + desc: EpollDescriptor, mask: Events, flags: EpollFlags, data: u64, @@ -379,7 +672,6 @@ impl EpollEntry { } fn poll(&self, global: &GlobalState) -> Option<(Option, bool)> { - let file = self.desc.upgrade()?; let inner = self.inner.lock(); if !self.is_enabled.load(core::sync::atomic::Ordering::Relaxed) { @@ -387,14 +679,11 @@ impl EpollEntry { return None; } - let events = file.poll(global, inner.mask, None)?; + let events = self.desc.poll(global, inner.mask, None)?; if events.is_empty() { Some((None, false)) } else { - let event = Some(EpollEvent { - events: events.bits(), - data: inner.data, - }); + let event = Some(EpollEvent::new(events.bits(), inner.data)); // keep the entry in the ready list if it is not edge-triggered or one-shot let is_still_ready = event.is_some() @@ -516,6 +805,35 @@ struct PollEntry { struct PollEntryObserver(Waker); +enum PollRegistrationTarget { + Descriptor(EpollDescriptor), + InputDevice(usize), + Stdin, +} + +impl PollRegistrationTarget { + fn unregister(self, global: &GlobalState, observer: Weak>) { + match self { + Self::Descriptor(descriptor) => descriptor.unregister_observer(global, observer), + Self::InputDevice(minor) => { + if let Some(registry) = global.input_registry.as_ref() { + registry.unregister_observer(minor, observer); + } + } + Self::Stdin => { + if let Some(pollable) = global.platform.stdin_pollable() { + pollable.unregister_observer(observer); + } + } + } + } +} + +struct PollRegistration { + target: PollRegistrationTarget, + observer: Weak>, +} + impl Clone for PollEntryObserver { fn clone(&self) -> Self { Self(self.0.clone()) @@ -547,35 +865,73 @@ impl PollSet { global: &GlobalState, files: &FilesState, waker: Option<&Waker>, + registrations: &mut Vec>, ) -> bool { let mut is_ready = false; for entry in &mut self.entries { entry.revents = if entry.fd < 0 { continue; - } else if let Ok(poll_descriptor) = - EpollDescriptor::try_from(files, entry.fd.reinterpret_as_unsigned() as usize) - { - let observer = if !is_ready && let Some(waker) = waker { - // TODO: a separate allocation is necessary here - // because registering an observer twice with two - // different event masks results in the last one - // replacing the first. If this is changed to - // instead combine the new event mask into the existing - // registration's mask, then we can use a single observer - // for all entries. - let observer = Arc::new(PollEntryObserver(waker.clone())); - let weak = Arc::downgrade(&observer); - entry.observer = Some(observer); - Some(weak as _) + } else if let Ok(poll_descriptor) = EpollDescriptor::try_from( + global, + files, + entry.fd.reinterpret_as_unsigned() as usize, + ) { + let observer: Option>> = + if !is_ready && let Some(waker) = waker { + // A separate allocation is necessary here because registering an observer + // twice with two different event masks results in the last one replacing + // the first. If registration instead combines masks, this can become one + // observer shared by all entries. + let observer = Arc::new(PollEntryObserver(waker.clone())); + let weak = Arc::downgrade(&observer); + entry.observer = Some(observer); + Some(weak) + } else { + // The poll set is already ready, or this scan is only checking readiness. + None + }; + let registration_target = observer + .as_ref() + .map(|_| poll_descriptor.transient_registration_target(global)); + // poll(2) on a regular file or directory is always ready -- Linux's + // `DEFAULT_POLLMASK` (IN | OUT | RDNORM | WRNORM) for any node without its own + // `poll` op. Every other FS-backed node keeps its subsystem readiness: stdin's + // pump, `/dev/input/event*` queues, and the OUT-only approximation for the rest, + // so a queue-backed device never spins on empty reads. BusyBox's `read` builtin + // polls its fd before every byte, so an FS-backed file that never reported IN + // hung `read x < file` (and every `while read` loop over a file) forever. + let always_ready_file = matches!(poll_descriptor, EpollDescriptor::File(_)) + && files + .run_on_raw_fd( + entry.fd.reinterpret_as_unsigned() as usize, + |fd| { + files.fs.fd_file_status(fd).is_ok_and(|status| { + matches!( + status.file_type, + litebox::fs::FileType::RegularFile + | litebox::fs::FileType::Directory + ) + }) + }, + |_| false, + |_| false, + |_| false, + |_| false, + |_| false, + |_| false, + ) + .unwrap_or(false); + let events = if always_ready_file { + (Events::IN | Events::OUT) & entry.mask } else { - // The poll set is already ready, or we have already - // registered the observer for this entry. - None + poll_descriptor + .poll(global, entry.mask, observer.clone()) + .unwrap_or(Events::NVAL) }; - // TODO: add machinery to unregister the observer to avoid leaks. - poll_descriptor - .poll(global, entry.mask, observer) - .unwrap_or(Events::NVAL) + if let (Some(observer), Some(target)) = (observer, registration_target) { + registrations.push(PollRegistration { target, observer }); + } + events } else { Events::NVAL }; @@ -592,7 +948,9 @@ impl PollSet { global: &GlobalState, files: &FilesState, ) { - self.scan_once(global, files, None); + let mut registrations = Vec::new(); + self.scan_once(global, files, None, &mut registrations); + debug_assert!(registrations.is_empty()); } /// Waits for any of the fds in the poll set to become ready. @@ -602,19 +960,38 @@ impl PollSet { cx: &WaitContext<'_, Platform>, files: &FilesState, ) -> Result<(), WaitError> { - if self.scan_once(global, files, None) { + let mut registrations = Vec::new(); + if self.scan_once(global, files, None, &mut registrations) { return Ok(()); } let mut register = true; - cx.wait_until(|| { - if self.scan_once(global, files, register.then_some(cx.waker())) { + let result = cx.wait_until(|| { + if self.scan_once( + global, + files, + register.then_some(cx.waker()), + &mut registrations, + ) { return true; } // Don't register observers again in the next iteration. register = false; false - }) + }); + + // Every registration above belongs only to this wait. Remove it on readiness, timeout, or + // interruption before dropping the strong observers, leaving permanent epoll interests + // untouched. + for registration in registrations.drain(..) { + registration + .target + .unregister(global, registration.observer); + } + for entry in &mut self.entries { + entry.observer = None; + } + result } /// Returns the accumulated `revents` for each entry in the poll set. @@ -644,9 +1021,11 @@ mod test { use alloc::sync::Arc; use litebox::event::Events; use litebox::event::wait::WaitState; - use litebox_common_linux::{EfdFlags, EpollEvent}; + use litebox::fd::TypedFd; + use litebox_common_linux::EpollEvent; + use litebox_common_linux::errno::Errno; - use super::EpollFile; + use super::{EpollFile, EpollSubsystem}; use crate::syscalls::file::FilesState; extern crate std; @@ -655,84 +1034,56 @@ mod test { crate::syscalls::tests::test_platform(None) } + type TestEpollFd = Arc>>>; + + fn new_epoll_fd( + task: &crate::Task>, + ) -> TestEpollFd { + Arc::new( + task.global + .litebox + .descriptor_table_mut() + .insert::>>( + EpollFile::new(), + ), + ) + } + fn setup_epoll() -> ( crate::Task>, - EpollFile>, + TestEpollFd, ) { let task = crate::syscalls::tests::init_platform(None); - - let epoll = EpollFile::new(); - (task, epoll) + let epoll_fd = new_epoll_fd(&task); + (task, epoll_fd) } #[test] - fn test_epoll_with_eventfd() { - let (task, epoll) = setup_epoll(); - let eventfd = crate::syscalls::eventfd::EventFile::new(0, EfdFlags::CLOEXEC); - let typed = task + fn test_epoll_with_pipe() { + let (task, epoll_fd) = setup_epoll(); + let (producer, consumer) = task .global - .litebox - .descriptor_table_mut() - .insert::>(eventfd); - let files = Arc::new(FilesState::new(task.files.borrow().fs.clone())); - let Ok(raw_fd) = files.insert_raw_fd(typed) else { - unreachable!() - }; - let descriptor = super::EpollDescriptor::try_from(&files, raw_fd).unwrap(); - epoll - .add_interest( - &task.global, - 10, - &descriptor, - EpollEvent { - events: Events::IN.bits(), - data: 0, - }, - ) - .unwrap(); - - // spawn a thread to write to the eventfd - { - let global = task.global.clone(); - let files = Arc::clone(&files); - std::thread::spawn(move || { - let typed = files - .raw_descriptor_store - .read() - .fd_from_raw_integer::>(raw_fd) - .unwrap(); - let _ = global - .litebox - .descriptor_table() - .with_entry(&typed, |entry| { - entry.write(&WaitState::new(platform()).context(), 1) - }); - }); - } - epoll - .wait(&task.global, &WaitState::new(platform()).context(), 1024) + .pipes + .create_pipe(2, litebox::pipes::Flags::empty(), None) .unwrap(); - } - - #[test] - fn test_epoll_with_pipe() { - let (task, epoll) = setup_epoll(); - let (producer, consumer) = - task.global - .pipes - .create_pipe(2, litebox::pipes::Flags::empty(), None); let consumer = Arc::new(consumer); let reader = super::EpollDescriptor::Pipe(Arc::clone(&consumer)); - epoll - .add_interest( - &task.global, - 10, - &reader, - EpollEvent { - events: Events::IN.bits(), - data: 0, - }, - ) + let handle = task + .global + .litebox + .descriptor_table() + .entry_handle(&epoll_fd) + .unwrap(); + handle + .with_entry(|epoll| { + epoll.add_interest( + &task.global, + &epoll_fd, + 10, + &reader, + EpollEvent::new(Events::IN.bits(), 0), + ) + }) .unwrap(); // spawn a thread to write to the pipe @@ -747,8 +1098,10 @@ mod test { 2 ); }); - epoll - .wait(&task.global, &WaitState::new(platform()).context(), 1024) + handle + .with_entry(|epoll| { + epoll.wait(&task.global, &WaitState::new(platform()).context(), 1024) + }) .unwrap(); let mut buf = [0; 2]; task.global @@ -759,24 +1112,279 @@ mod test { } #[test] - fn test_poll() { + fn test_epoll_ctl_mod_updates_registered_fd_instead_of_failing() { + // Regression: `EPOLL_CTL_MOD` used to return `EINVAL` unconditionally. + // libuv's `uv__io_poll` registers a watcher with `ADD`, and on the + // `EEXIST` that a re-add returns it issues `MOD` to swap the event + // mask; the stray `EINVAL` there made libuv `abort()` (guest SIGABRT), + // which stalled every Node `http` loopback connection. `MOD` on a + // registered fd must succeed; `MOD` on an unregistered fd is `ENOENT`, + // never `EINVAL`. + use litebox_common_linux::EpollOp; + let (task, epoll_fd) = setup_epoll(); + let (_producer, consumer) = task + .global + .pipes + .create_pipe(2, litebox::pipes::Flags::empty(), None) + .unwrap(); + let consumer = Arc::new(consumer); + let reader = super::EpollDescriptor::Pipe(Arc::clone(&consumer)); + let handle = task + .global + .litebox + .descriptor_table() + .entry_handle(&epoll_fd) + .unwrap(); + + // MOD before the fd is registered: not present, so ENOENT (not EINVAL). + let before_add = handle.with_entry(|epoll| { + epoll.epoll_ctl( + &task.global, + &epoll_fd, + EpollOp::EpollCtlMod, + 10, + &reader, + Some(EpollEvent::new(Events::OUT.bits(), 0)), + ) + }); + assert_eq!(before_add, Err(Errno::ENOENT)); + + // ADD, then MOD to a fresh mask: the MOD must succeed. + handle + .with_entry(|epoll| { + epoll.epoll_ctl( + &task.global, + &epoll_fd, + EpollOp::EpollCtlAdd, + 10, + &reader, + Some(EpollEvent::new(Events::IN.bits(), 0)), + ) + }) + .unwrap(); + handle + .with_entry(|epoll| { + epoll.epoll_ctl( + &task.global, + &epoll_fd, + EpollOp::EpollCtlMod, + 10, + &reader, + Some(EpollEvent::new(Events::OUT.bits(), 5)), + ) + }) + .expect("MOD on a registered fd must succeed, not return EINVAL"); + } + + #[test] + fn test_epoll_nested() { let task = crate::syscalls::tests::init_platform(None); - let mut set = super::PollSet::with_capacity(0); - let eventfd = crate::syscalls::eventfd::EventFile::new(0, EfdFlags::empty()); + let inner_fd = new_epoll_fd(&task); + let (producer, consumer) = task + .global + .pipes + .create_pipe(2, litebox::pipes::Flags::empty(), None) + .unwrap(); + let consumer = Arc::new(consumer); + let reader = super::EpollDescriptor::Pipe(Arc::clone(&consumer)); + let inner_handle = task + .global + .litebox + .descriptor_table() + .entry_handle(&inner_fd) + .unwrap(); + inner_handle + .with_entry(|inner| { + inner.add_interest( + &task.global, + &inner_fd, + 20, + &reader, + EpollEvent::new(Events::IN.bits(), 0), + ) + }) + .unwrap(); + + let outer_fd = new_epoll_fd(&task); + let nested = super::EpollDescriptor::Epoll(Arc::clone(&inner_fd)); + let outer_handle = task + .global + .litebox + .descriptor_table() + .entry_handle(&outer_fd) + .unwrap(); + outer_handle + .with_entry(|outer| { + outer.add_interest( + &task.global, + &outer_fd, + 10, + &nested, + EpollEvent::new(Events::IN.bits(), 42), + ) + }) + .unwrap(); + + // Writing to the pipe should make the inner epoll ready, which in turn should make the + // outer epoll (which has the inner epoll nested inside it) ready. + task.global + .pipes + .write(&WaitState::new(platform()).context(), &producer, &[1, 2]) + .unwrap(); + + let events = outer_handle + .with_entry(|outer| { + outer.wait(&task.global, &WaitState::new(platform()).context(), 1024) + }) + .unwrap(); + assert_eq!(events.len(), 1); + let data = events[0].data; + assert_eq!(data, 42); + } + + #[test] + fn test_epoll_nested_cycle_rejected() { + let task = crate::syscalls::tests::init_platform(None); + + let a_fd = new_epoll_fd(&task); + let b_fd = new_epoll_fd(&task); + + let a_handle = task + .global + .litebox + .descriptor_table() + .entry_handle(&a_fd) + .unwrap(); + a_handle + .with_entry(|a| { + a.add_interest( + &task.global, + &a_fd, + 20, + &super::EpollDescriptor::Epoll(Arc::clone(&b_fd)), + EpollEvent::new(Events::IN.bits(), 0), + ) + }) + .unwrap(); - let typed = task + // B adding A back would close a 2-fd cycle; this must be rejected synchronously with + // ELOOP rather than being allowed to form (which would only surface as a hang later, + // on the first event delivered into the cycle). + let b_handle = task .global .litebox - .descriptor_table_mut() - .insert::>(eventfd); + .descriptor_table() + .entry_handle(&b_fd) + .unwrap(); + let result = b_handle.with_entry(|b| { + b.add_interest( + &task.global, + &b_fd, + 10, + &super::EpollDescriptor::Epoll(Arc::clone(&a_fd)), + EpollEvent::new(Events::IN.bits(), 0), + ) + }); + assert_eq!(result, Err(Errno::ELOOP)); + } + + /// Reproduces, under real concurrency, the exact race a prior cycle-detection attempt + /// missed: thread 1 adds B into A while thread 2 concurrently adds A into B. Checking for a + /// cycle and committing the new edge are two different critical sections unless a single + /// process-wide lock spans both, so each thread's check can run before the other's insert is + /// visible -- both threads see an acyclic graph, both commit, and together they still close + /// the cycle. Since A adding B and B adding A are reciprocal, the only two correct outcomes + /// per iteration are "exactly one add wins, the other gets ELOOP" -- never both winning + /// (that would be the cycle itself), never both losing, and never neither thread returning at + /// all. A `Barrier` lines both threads up right before their `add_interest` call to maximize + /// the chance of hitting the race, and `recv_timeout` bounds each attempt so a regression + /// that reintroduces the deadlock fails this test quickly instead of hanging the run. + #[test] + fn test_epoll_nested_concurrent_add_never_forms_cycle() { + let task = crate::syscalls::tests::init_platform(None); + let global = task.global.clone(); + + for iteration in 0..30u32 { + let a_fd = new_epoll_fd(&task); + let b_fd = new_epoll_fd(&task); + let barrier = Arc::new(std::sync::Barrier::new(2)); + + let (tx_a, rx_a) = std::sync::mpsc::channel(); + let g = global.clone(); + let (a, b) = (Arc::clone(&a_fd), Arc::clone(&b_fd)); + let bar = Arc::clone(&barrier); + std::thread::spawn(move || { + let handle = g.litebox.descriptor_table().entry_handle(&a).unwrap(); + bar.wait(); + let result = handle.with_entry(|entry| { + entry.add_interest( + &g, + &a, + 1000 + iteration, + &super::EpollDescriptor::Epoll(Arc::clone(&b)), + EpollEvent::new(Events::IN.bits(), 0), + ) + }); + let _ = tx_a.send(result); + }); + + let (tx_b, rx_b) = std::sync::mpsc::channel(); + let g = global.clone(); + let (a, b) = (Arc::clone(&a_fd), Arc::clone(&b_fd)); + let bar = Arc::clone(&barrier); + std::thread::spawn(move || { + let handle = g.litebox.descriptor_table().entry_handle(&b).unwrap(); + bar.wait(); + let result = handle.with_entry(|entry| { + entry.add_interest( + &g, + &b, + 2000 + iteration, + &super::EpollDescriptor::Epoll(Arc::clone(&a)), + EpollEvent::new(Events::IN.bits(), 0), + ) + }); + let _ = tx_b.send(result); + }); + + let timeout = core::time::Duration::from_secs(5); + let Ok(result_a) = rx_a.recv_timeout(timeout) else { + panic!( + "iteration {iteration}: thread adding B into A never returned -- \ + a cycle likely formed and something is stuck on it" + ); + }; + let Ok(result_b) = rx_b.recv_timeout(timeout) else { + panic!( + "iteration {iteration}: thread adding A into B never returned -- \ + a cycle likely formed and something is stuck on it" + ); + }; + + match (result_a, result_b) { + (Ok(()), Err(Errno::ELOOP)) | (Err(Errno::ELOOP), Ok(())) => {} + other => panic!( + "iteration {iteration}: expected exactly one add to win and the other to be \ + rejected with ELOOP, got {other:?} instead" + ), + } + } + } + + #[test] + fn test_poll() { + let task = crate::syscalls::tests::init_platform(None); + + let mut set = super::PollSet::with_capacity(0); + let (rfd_u, wfd_u) = task + .sys_pipe2(litebox::fs::OFlags::empty()) + .expect("pipe2 failed"); + let rfd = i32::try_from(rfd_u).unwrap(); + let wfd = i32::try_from(wfd_u).unwrap(); let no_fds = FilesState::new(task.files.borrow().fs.clone()); - let fds = Arc::new(FilesState::new(task.files.borrow().fs.clone())); - let Ok(raw_fd) = fds.insert_raw_fd(typed) else { - unreachable!() - }; - let fd = i32::try_from(raw_fd).unwrap(); - set.add_fd(fd, Events::IN); + let fds = task.files.borrow().clone(); + set.add_fd(rfd, Events::IN); let revents = |set: &super::PollSet| { let revents: std::vec::Vec<_> = set.revents().collect(); @@ -788,40 +1396,14 @@ mod test { .unwrap(); assert_eq!(revents(&set), Events::NVAL); - { - let typed = fds - .raw_descriptor_store - .read() - .fd_from_raw_integer::>( - raw_fd, - ) - .unwrap(); - task.global - .litebox - .descriptor_table() - .with_entry(&typed, |entry| { - entry.write(&WaitState::new(platform()).context(), 1) - }); - } + task.sys_write(wfd, &[1], None).unwrap(); set.wait(&task.global, &WaitState::new(platform()).context(), &fds) .unwrap(); assert_eq!(revents(&set), Events::IN); - { - let typed = fds - .raw_descriptor_store - .read() - .fd_from_raw_integer::>( - raw_fd, - ) - .unwrap(); - task.global - .litebox - .descriptor_table() - .with_entry(&typed, |entry| { - entry.read(&WaitState::new(platform()).context()) - }); - } + let mut buf = [0; 1]; + assert_eq!(task.sys_read(rfd, &mut buf, None).unwrap(), 1); + assert_eq!(buf, [1]); set.wait( &task.global, &WaitState::new(platform()) @@ -832,29 +1414,17 @@ mod test { .unwrap_err(); assert!(revents(&set).is_empty()); - // spawn a thread to write to the eventfd - let global = task.global.clone(); - let fds_for_thread = Arc::clone(&fds); - std::thread::spawn(move || { - let typed = fds_for_thread - .raw_descriptor_store - .read() - .fd_from_raw_integer::>( - raw_fd, - ) - .unwrap(); - let handle = global - .litebox - .descriptor_table() - .entry_handle(&typed) - .unwrap(); - let _ = - handle.with_entry(|entry| entry.write(&WaitState::new(platform()).context(), 1)); + task.spawn_clone_for_test(move |task| { + std::thread::sleep(core::time::Duration::from_millis(100)); + assert_eq!(task.sys_write(wfd, &[1], None).unwrap(), 1); }); set.wait(&task.global, &WaitState::new(platform()).context(), &fds) .unwrap(); assert_eq!(revents(&set), Events::IN); + + let _ = task.sys_close(rfd); + let _ = task.sys_close(wfd); } #[test] diff --git a/litebox_shim_linux/src/syscalls/eventfd.rs b/litebox_shim_linux/src/syscalls/eventfd.rs index 1149b4d7f7..3355d5ee4e 100644 --- a/litebox_shim_linux/src/syscalls/eventfd.rs +++ b/litebox_shim_linux/src/syscalls/eventfd.rs @@ -8,6 +8,7 @@ use core::sync::atomic::AtomicU32; use litebox::{ event::{ Events, IOPollable, + counter::{EventCounter, EventCounterReadMode}, observer::Observer, polling::{Pollee, TryOpError}, wait::WaitContext, @@ -19,7 +20,7 @@ use litebox::{ }; use litebox_common_linux::{EfdFlags, errno::Errno}; -use crate::ShimPlatform; +use crate::{GlobalState, ShimFS, ShimPlatform}; pub(crate) struct EventfdSubsystem(core::marker::PhantomData); impl FdEnabledSubsystem for EventfdSubsystem { @@ -27,222 +28,218 @@ impl FdEnabledSubsystem for EventfdSubsystem { } impl FdEnabledSubsystemEntry for EventFile {} +/// Where the eventfd's counter actually lives. +/// +/// With a broker connected, the counter is a broker object +/// ([`EventCounter`]), which is what lets a brokered deployment share the +/// eventfd across guest processes. Without one -- every macOS run today, and +/// any Linux run started without `--broker-control-socket` -- there is no +/// broker to host that object, and `eventfd2` used to fail outright with +/// `EIO`, which took down every real `libuv` consumer at `uv_loop_init` +/// (Node aborts in `LegacyTracingAgent`'s constructor before running a line +/// of JS). The local variant is a plain in-shim counter with the exact +/// `eventfd(2)` semantics, sufficient for everything a single guest process +/// can observe. +enum Backend { + Brokered(EventCounter), + Local { + counter: litebox::sync::Mutex, + pollee: Pollee, + }, +} + pub(crate) struct EventFile { - counter: litebox::sync::Mutex, + backend: Backend, /// File status flags (see [`OFlags::STATUS_FLAGS_MASK`]) status: AtomicU32, semaphore: bool, - pollee: Pollee, } impl EventFile { - pub(crate) fn new(count: u64, flags: EfdFlags) -> Self { + fn new(backend: Backend, flags: EfdFlags) -> Self { let mut status = OFlags::RDWR; status.set(OFlags::NONBLOCK, flags.contains(EfdFlags::NONBLOCK)); - Self { - counter: litebox::sync::Mutex::new(count), + backend, status: AtomicU32::new(status.bits()), semaphore: flags.contains(EfdFlags::SEMAPHORE), - pollee: Pollee::new(), } } - fn try_read(&self) -> Result> { - let mut counter = self.counter.lock(); - if *counter == 0 { - return Err(TryOpError::TryAgain); - } - - let res = if self.semaphore { 1 } else { *counter }; - *counter -= res; - - drop(counter); - self.pollee.notify_observers(Events::OUT); - Ok(res) - } - pub(crate) fn read(&self, cx: &WaitContext<'_, Platform>) -> Result { - self.pollee - .wait( - cx, - self.get_status().contains(OFlags::NONBLOCK), - Events::IN, - || self.try_read(), - ) - .map_err(Errno::from) - } - - fn try_write(&self, value: u64) -> Result> { - let mut counter = self.counter.lock(); - if let Some(new_value) = (*counter).checked_add(value) { - // The maximum value that may be stored in the counter is the largest unsigned - // 64-bit value minus 1 (i.e., 0xfffffffffffffffe) - if new_value != u64::MAX { - *counter = new_value; - drop(counter); - self.pollee.notify_observers(Events::IN); - return Ok(8); - } + match &self.backend { + Backend::Brokered(counter) => counter + .read( + cx, + self.is_nonblocking(), + if self.semaphore { + EventCounterReadMode::One + } else { + EventCounterReadMode::All + }, + ) + .map_err(Errno::from), + Backend::Local { counter, pollee } => pollee + .wait(cx, self.is_nonblocking(), Events::IN, || { + let mut counter = counter.lock(); + if *counter == 0 { + return Err(TryOpError::::TryAgain); + } + let res = if self.semaphore { 1 } else { *counter }; + *counter -= res; + drop(counter); + pollee.notify_observers(Events::OUT); + Ok(res) + }) + .map_err(Errno::from), } - - Err(TryOpError::TryAgain) } pub(crate) fn write(&self, cx: &WaitContext<'_, Platform>, value: u64) -> Result { - self.pollee - .wait( - cx, - self.get_status().contains(OFlags::NONBLOCK), - Events::OUT, - || self.try_write(value), - ) - .map_err(Errno::from) + match &self.backend { + Backend::Brokered(counter) => counter + .write(cx, self.is_nonblocking(), value) + .map_err(Errno::from), + Backend::Local { counter, pollee } => pollee + .wait(cx, self.is_nonblocking(), Events::OUT, || { + let mut counter = counter.lock(); + // The counter's maximum is `u64::MAX - 1`; a write that + // would exceed it blocks (or `EAGAIN`s), per eventfd(2). + if let Some(new_value) = (*counter).checked_add(value) + && new_value != u64::MAX + { + *counter = new_value; + drop(counter); + pollee.notify_observers(Events::IN); + return Ok(8); + } + Err(TryOpError::::TryAgain) + }) + .map_err(Errno::from), + } } super::common_functions_for_file_status!(); + + fn is_nonblocking(&self) -> bool { + self.get_status().contains(OFlags::NONBLOCK) + } } impl IOPollable for EventFile { fn check_io_events(&self) -> Events { - let counter = self.counter.lock(); - let mut events = Events::empty(); - if *counter != 0 { - events |= Events::IN; + match &self.backend { + Backend::Brokered(counter) => counter.check_io_events(), + Backend::Local { counter, .. } => { + let counter = counter.lock(); + let mut events = Events::empty(); + if *counter != 0 { + events |= Events::IN; + } + // Writable whenever at least a value of 1 fits. + if *counter < u64::MAX - 1 { + events |= Events::OUT; + } + events + } } - // if it is possible to write a value of at least "1" - // without blocking, the file is writable - let is_writable = *counter < u64::MAX - 1; - if is_writable { - events |= Events::OUT; + } + + fn register_observer(&self, observer: alloc::sync::Weak>, mask: Events) { + match &self.backend { + Backend::Brokered(counter) => counter.register_observer(observer, mask), + Backend::Local { pollee, .. } => pollee.register_observer(observer, mask), } + } - events + fn unregister_observer(&self, observer: alloc::sync::Weak>) { + match &self.backend { + Backend::Brokered(counter) => counter.unregister_observer(observer), + Backend::Local { pollee, .. } => pollee.unregister_observer(observer), + } } +} - fn register_observer(&self, observer: alloc::sync::Weak>, mask: Events) { - self.pollee.register_observer(observer, mask); +impl GlobalState { + pub(crate) fn create_linux_eventfd( + &self, + initval: u32, + flags: EfdFlags, + ) -> Result, Errno> { + if flags + .intersects((EfdFlags::SEMAPHORE | EfdFlags::CLOEXEC | EfdFlags::NONBLOCK).complement()) + { + return Err(Errno::EINVAL); + } + + let count = u64::from(initval); + // Prefer the brokered counter (shareable across guest processes in a + // brokered deployment); `Unavailable` means no broker is connected at + // all -- fall back to the local backend rather than failing the + // syscall. Any other creation error is a real broker fault and is + // reported as such. + let backend = match EventCounter::new(&self.litebox, count) { + Ok(counter) => Backend::Brokered(counter), + Err(litebox::event::counter::EventCounterError::Unavailable) => Backend::Local { + counter: litebox::sync::Mutex::new(count), + pollee: Pollee::new(), + }, + Err(err) => return Err(Errno::from(err)), + }; + Ok(EventFile::new(backend, flags)) } } #[cfg(test)] mod tests { - use crate::syscalls::tests::TestPlatform; use litebox::event::wait::WaitState; use litebox_common_linux::{EfdFlags, errno::Errno}; extern crate std; - fn platform() -> &'static TestPlatform { - crate::syscalls::tests::test_platform(None) - } - + /// Without a broker, `eventfd2` must still work via the local backend -- + /// this exact gap aborted Node at `uv_loop_init` (its `LegacyTracingAgent` + /// asserts on the result) before any JS ran, while `--version` worked. #[test] - fn test_semaphore_eventfd() { - let _task = crate::syscalls::tests::init_platform(None); - - let eventfd = alloc::sync::Arc::new(super::EventFile::new(0, EfdFlags::SEMAPHORE)); - let total = 8; - for _ in 0..total { - let copied_eventfd = eventfd.clone(); - std::thread::spawn(move || { - copied_eventfd - .read(&WaitState::new(platform()).context()) - .unwrap(); - }); - } - - std::thread::sleep(core::time::Duration::from_millis(500)); - eventfd - .write(&WaitState::new(platform()).context(), total) - .unwrap(); - } - - #[test] - fn test_blocking_eventfd() { - let _task = crate::syscalls::tests::init_platform(None); - - let eventfd = alloc::sync::Arc::new(super::EventFile::new(0, EfdFlags::empty())); - let copied_eventfd = eventfd.clone(); - std::thread::spawn(move || { - copied_eventfd - .write(&WaitState::new(platform()).context(), 1) - .unwrap(); - // block until the first read finishes - copied_eventfd - .write(&WaitState::new(platform()).context(), u64::MAX - 1) - .unwrap(); - }); - - // block until the first write - let ret = eventfd.read(&WaitState::new(platform()).context()).unwrap(); - assert_eq!(ret, 1); - - // block until the second write - let ret = eventfd.read(&WaitState::new(platform()).context()).unwrap(); - assert_eq!(ret, u64::MAX - 1); + fn test_eventfd_works_without_broker() { + let task = crate::syscalls::tests::init_platform(None); + let platform = crate::syscalls::tests::test_platform(None); + + let eventfd = task + .global + .create_linux_eventfd(3, EfdFlags::NONBLOCK) + .expect("brokerless eventfd must fall back to the local backend"); + + // The initial count reads back in one shot, then the empty counter + // reports EAGAIN rather than blocking (NONBLOCK is set). + assert_eq!(eventfd.read(&WaitState::new(platform).context()), Ok(3)); + assert_eq!( + eventfd.read(&WaitState::new(platform).context()), + Err(Errno::EAGAIN) + ); + + // A write of 5 wakes the counter back up; semaphore mode is off, so + // the next read drains it whole. + assert_eq!(eventfd.write(&WaitState::new(platform).context(), 5), Ok(8)); + assert_eq!(eventfd.read(&WaitState::new(platform).context()), Ok(5)); } + /// Semaphore mode decrements by exactly one per read. #[test] - fn test_blocking_eventfd_no_race_on_massive_readwrite() { - let _task = crate::syscalls::tests::init_platform(None); - - let eventfd = alloc::sync::Arc::new(super::EventFile::new(0, EfdFlags::empty())); - let copied_eventfd = eventfd.clone(); - std::thread::spawn(move || { - for _ in 0..10000 { - copied_eventfd - .write(&WaitState::new(platform()).context(), u64::MAX - 1) - .unwrap(); - } - }); - - for _ in 0..10000 { - let ret = eventfd.read(&WaitState::new(platform()).context()).unwrap(); - assert_eq!(ret, u64::MAX - 1); - } - } - - #[test] - fn test_nonblocking_eventfd() { - let _task = crate::syscalls::tests::init_platform(None); - - let eventfd = alloc::sync::Arc::new(super::EventFile::new(0, EfdFlags::NONBLOCK)); - let copied_eventfd = eventfd.clone(); - std::thread::spawn(move || { - // first write should succeed immediately - copied_eventfd - .write(&WaitState::new(platform()).context(), 1) - .unwrap(); - // block until the first read finishes - while let Err(e) = - copied_eventfd.write(&WaitState::new(platform()).context(), u64::MAX - 1) - { - assert_eq!(e, Errno::EAGAIN, "Unexpected error: {e:?}"); - core::hint::spin_loop(); - } - }); - - let read = |eventfd: &super::EventFile, expected_value: u64| { - loop { - match eventfd.read(&WaitState::new(platform()).context()) { - Ok(ret) => { - assert_eq!(ret, expected_value); - break; - } - Err(Errno::EAGAIN) => { - // busy wait - // TODO: use poll rather than busy wait - } - Err(e) => panic!("Unexpected error: {e:?}"), - } - core::hint::spin_loop(); - } - }; - - // block until the first write - read(&eventfd, 1); - // block until the second write - read(&eventfd, u64::MAX - 1); + fn test_eventfd_local_semaphore_mode() { + let task = crate::syscalls::tests::init_platform(None); + let platform = crate::syscalls::tests::test_platform(None); + + let eventfd = task + .global + .create_linux_eventfd(2, EfdFlags::SEMAPHORE | EfdFlags::NONBLOCK) + .expect("brokerless eventfd must fall back to the local backend"); + + assert_eq!(eventfd.read(&WaitState::new(platform).context()), Ok(1)); + assert_eq!(eventfd.read(&WaitState::new(platform).context()), Ok(1)); + assert_eq!( + eventfd.read(&WaitState::new(platform).context()), + Err(Errno::EAGAIN) + ); } } diff --git a/litebox_shim_linux/src/syscalls/file.rs b/litebox_shim_linux/src/syscalls/file.rs index 3973aa1f47..bc83deeea8 100644 --- a/litebox_shim_linux/src/syscalls/file.rs +++ b/litebox_shim_linux/src/syscalls/file.rs @@ -4,51 +4,74 @@ //! Implementation of file related syscalls, e.g., `open`, `read`, `write`, etc. use alloc::{ + collections::BTreeMap, ffi::CString, string::{String, ToString as _}, + sync::{Arc, Weak}, vec, }; use litebox::{ event::{Events, wait::WaitError}, - fd::{FdEnabledSubsystem, MetadataError, TypedFd}, - fs::{Mode, OFlags, SeekWhence}, + fd::{EntryHandle, FdEnabledSubsystem, MetadataError, TypedFd}, + fs::{AccessCredentials, Mode, OFlags, SeekWhence}, mm::linux::PAGE_SIZE, + net::Network, path, + pipes::Pipes, platform::StdioStream, utils::{ReinterpretSignedExt as _, ReinterpretUnsignedExt as _, TruncateExt as _}, }; use litebox_common_linux::{ AccessFlags, AtFlags, EfdFlags, EpollCreateFlags, FcntlArg, FileDescriptorFlags, FileStat, - InodeType, IoReadVec, IoWriteVec, IoctlArg, Statx, StatxMask, TimeParam, errno::Errno, - signal::Signal, + FlockOperation, InodeType, InotifyInitFlags, InotifyMask, IoReadVec, IoWriteVec, IoctlArg, + SockFlags, SockType, Statx, StatxMask, TimeParam, errno::Errno, signal::Signal, }; use thiserror::Error; use crate::{GlobalState, ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut, syscalls::signal}; -use core::sync::atomic::{AtomicUsize, Ordering}; +use core::{ + ffi::CStr, + sync::atomic::{AtomicBool, AtomicI32, AtomicU32, AtomicUsize, Ordering}, +}; #[derive(Clone, Copy)] -struct AccessUserInfo { +struct AccessUserInfo<'a> { user: u32, group: u32, + supplementary_groups: &'a [u32], } -impl From for AccessUserInfo { +impl From for AccessUserInfo<'static> { fn from(value: litebox::fs::UserInfo) -> Self { Self { user: u32::from(value.user), group: u32::from(value.group), + supplementary_groups: &[], } } } +impl<'a> AccessUserInfo<'a> { + fn as_fs_credentials(self) -> AccessCredentials<'a> { + AccessCredentials::new(self.user, self.group, self.supplementary_groups) + } +} + +static NEXT_MEMFD_ID: AtomicUsize = AtomicUsize::new(0); + /// Task state shared by `CLONE_FS`. pub(crate) struct FsState { umask: core::sync::atomic::AtomicU32, - /// The current working directory + /// The current working directory, as a real (root-inclusive, see `root`) absolute path. /// /// Must end with a '/'. cwd: litebox::sync::RwLock, + /// The process's root directory (`chroot(2)`): the real absolute path beneath which every + /// guest-visible absolute path is resolved (see `Task::map_absolute_to_root`). `/` until + /// the first `chroot`; inherited by `fork`, shared by `CLONE_FS`, kept across `execve`. + /// + /// Must end with a '/'. + root: litebox::sync::RwLock, } impl Clone for FsState { @@ -56,6 +79,7 @@ impl Clone for FsState { Self { umask: self.umask.load(Ordering::Relaxed).into(), cwd: litebox::sync::RwLock::new(self.cwd.read().clone()), + root: litebox::sync::RwLock::new(self.root.read().clone()), } } } @@ -65,6 +89,7 @@ impl FsState { Self { umask: (Mode::WGRP | Mode::WOTH).bits().into(), cwd: litebox::sync::RwLock::new(String::from("/")), + root: litebox::sync::RwLock::new(String::from("/")), } } @@ -79,9 +104,26 @@ pub(crate) struct FilesState { pub(crate) fs: alloc::sync::Arc, pub(crate) raw_descriptor_store: litebox::sync::RwLock, + pub(crate) shared_file_mappings: + litebox::sync::Mutex>>, max_fd: AtomicUsize, } +/// A reference to an open file description carried in an `SCM_RIGHTS` message. +/// +/// The sender's descriptor number and descriptor-local flags are intentionally absent. +/// The strong entry handle keeps the description alive until it is installed by the +/// receiver or discarded with the message. +pub(crate) enum TransferredFd { + Fs(EntryHandle), + Network(EntryHandle>), + Pipe(EntryHandle>), + EventFd(EntryHandle>), + Epoll(EntryHandle>), + Unix(EntryHandle>), + Netlink(EntryHandle>), +} + impl FilesState { pub(crate) fn new(fs: alloc::sync::Arc) -> Self { Self { @@ -89,6 +131,7 @@ impl FilesState { raw_descriptor_store: litebox::sync::RwLock::new( litebox::fd::RawDescriptorStorage::new(), ), + shared_file_mappings: litebox::sync::Mutex::new(alloc::vec::Vec::new()), max_fd: AtomicUsize::new(usize::MAX), } } @@ -97,6 +140,84 @@ impl FilesState { self.max_fd.store(max_fd, Ordering::Relaxed); } + /// Returns the file-descriptor table a `fork`ed child starts with: every descriptor of this + /// table, duplicated at the same number. + /// + /// "Duplicated" is `dup(2)`'s sense, which is `fork(2)`'s too: the new descriptor refers to + /// the same open file description, so the file offset and status flags stay shared with the + /// parent, while the descriptor itself -- and, crucially, the number it is filed under -- is + /// the child's alone. That independence is the whole point: a shell between `fork` and `exec` + /// rearranges fds 0/1/2 for the command it is about to run, and none of that may reach back + /// into the shell. + /// + /// `FD_CLOEXEC` is per descriptor rather than per description, so it is copied explicitly. + pub(crate) fn fork_copy(&self, task: &Task) -> Result { + fn dup_into( + task: &Task, + new: &FilesState, + fd: &TypedFd, + raw_fd: usize, + cloexec: bool, + ) -> Result<(), Errno> { + let mut dt = task.global.litebox.descriptor_table_mut(); + let fd: TypedFd = dt.duplicate(fd).ok_or(Errno::EBADF)?; + note_pty_slave_descriptor::(&dt, &fd); + if cloexec { + let old = dt.set_fd_metadata(&fd, FileDescriptorFlags::FD_CLOEXEC); + assert!(old.is_none()); + } + drop(dt); + let inserted = new + .raw_descriptor_store + .write() + .fd_into_specific_raw_integer(fd, raw_fd); + assert!(inserted, "the new table cannot already have fd {raw_fd}"); + Ok(()) + } + + let new = Self::new(self.fs.clone()); + new.set_max_fd(self.max_fd.load(Ordering::Relaxed)); + (*new.shared_file_mappings.lock()).clone_from(&self.shared_file_mappings.lock()); + let alive_fds: alloc::vec::Vec = + self.raw_descriptor_store.read().iter_alive().collect(); + for raw_fd in alive_fds { + let cloexec = get_file_descriptor_flags(raw_fd, &task.global, self) + .is_ok_and(|flags| flags.contains(FileDescriptorFlags::FD_CLOEXEC)); + // Inotify fds aren't one of `run_on_raw_fd`'s hand-enumerated subsystem parameters + // (see that function's doc comment, and the matching pre-check in `do_read`/ + // `do_close`): resolve them directly first so a live inotify fd (e.g. dbus-daemon's + // own `inotify_init1()` result) can be duplicated into a forked child exactly like + // every other fd, rather than `run_on_raw_fd`'s `EBADF` fallthrough aborting the + // *entire* fork on the first such fd it meets -- live-verified: this exact gap + // turned `dbus-daemon`'s `fork()` for D-Bus service activation into + // `Errno::EBADF`, breaking every activated service (`xfconfd` included) the moment + // inotify support gave dbus-daemon a real inotify fd to carry across the fork. + if let Ok(inotify_fd) = self + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(raw_fd) + { + dup_into(task, &new, &inotify_fd, raw_fd, cloexec)?; + continue; + } + let dup_result = self.run_on_raw_fd( + raw_fd, + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + |fd| dup_into(task, &new, fd, raw_fd, cloexec), + ); + if !matches!(dup_result, Ok(Ok(()))) { + litebox_util_log::debug!(raw_fd:% = raw_fd, result:? = dup_result; "fork_copy: fd duplication failed"); + } + dup_result??; + } + Ok(new) + } + // Returns Ok(raw_fd) if it fits within the max limits already set up; otherwise returns the // Err(typed_fd) pub(crate) fn insert_raw_fd( @@ -116,6 +237,347 @@ impl FilesState { } } +impl Task { + /// Capture an open file description for an `SCM_RIGHTS` message without + /// allocating a descriptor in either process. + pub(crate) fn transfer_fd(&self, raw_fd: i32) -> Result, Errno> { + let raw_fd = usize::try_from(raw_fd).map_err(|_| Errno::EBADF)?; + let files = self.files.borrow(); + let result = files.run_on_raw_fd( + raw_fd, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Fs) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Network) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Pipe) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::EventFd) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Epoll) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Unix) + .ok_or(Errno::EBADF) + }, + |fd| { + self.global + .litebox + .descriptor_table() + .entry_handle(fd) + .map(TransferredFd::Netlink) + .ok_or(Errno::EBADF) + }, + ); + result.flatten() + } + + /// Install an `SCM_RIGHTS` open file description at the receiver's lowest + /// available descriptor number. + pub(crate) fn install_transferred_fd( + &self, + transferred: TransferredFd, + cloexec: bool, + ) -> Result { + fn install( + task: &Task, + handle: EntryHandle, + cloexec: bool, + ) -> Result + where + Platform: ShimPlatform, + FS: ShimFS, + Subsystem: FdEnabledSubsystem, + { + let typed = { + let mut descriptors = task.global.litebox.descriptor_table_mut(); + let typed = descriptors.insert_handle(handle); + note_pty_slave_descriptor::(&descriptors, &typed); + if cloexec { + let old = descriptors.set_fd_metadata(&typed, FileDescriptorFlags::FD_CLOEXEC); + assert!(old.is_none()); + } + typed + }; + task.files.borrow().insert_raw_fd(typed).map_err(|typed| { + let _ = task.global.litebox.descriptor_table_mut().remove(&typed); + Errno::EMFILE + }) + } + + match transferred { + TransferredFd::Fs(handle) => install(self, handle, cloexec), + TransferredFd::Network(handle) => install(self, handle, cloexec), + TransferredFd::Pipe(handle) => install(self, handle, cloexec), + TransferredFd::EventFd(handle) => install(self, handle, cloexec), + TransferredFd::Epoll(handle) => install(self, handle, cloexec), + TransferredFd::Unix(handle) => install(self, handle, cloexec), + TransferredFd::Netlink(handle) => install(self, handle, cloexec), + } + } +} + +const F_SEAL_SEAL: u32 = 0x0001; +const F_SEAL_SHRINK: u32 = 0x0002; +const F_SEAL_GROW: u32 = 0x0004; +const F_SEAL_WRITE: u32 = 0x0008; +const F_SEAL_FUTURE_WRITE: u32 = 0x0010; +const F_SEAL_ALL: u32 = + F_SEAL_SEAL | F_SEAL_SHRINK | F_SEAL_GROW | F_SEAL_WRITE | F_SEAL_FUTURE_WRITE; + +/// Entry metadata identifying an unlinked file created by `memfd_create(2)` and +/// storing its inode-scoped seal set. +#[derive(Debug)] +pub(crate) struct MemfdBacking { + seals: AtomicU32, + shared_futex_backing: litebox::mm::linux::SharedFutexBacking, +} + +impl Clone for MemfdBacking { + fn clone(&self) -> Self { + Self { + seals: AtomicU32::new(self.seals()), + shared_futex_backing: self.shared_futex_backing, + } + } +} + +impl MemfdBacking { + fn new(allow_sealing: bool) -> Self { + Self { + seals: AtomicU32::new(if allow_sealing { 0 } else { F_SEAL_SEAL }), + shared_futex_backing: litebox::mm::linux::SharedFutexBacking::new(), + } + } + + pub(crate) fn shared_futex_backing(&self) -> litebox::mm::linux::SharedFutexBacking { + self.shared_futex_backing + } + + fn seals(&self) -> u32 { + self.seals.load(Ordering::Acquire) + } + + fn add_seals(&self, seals: u32) -> Result<(), Errno> { + if seals & !F_SEAL_ALL != 0 { + return Err(Errno::EINVAL); + } + let mut current = self.seals(); + loop { + if current & F_SEAL_SEAL != 0 { + return Err(Errno::EPERM); + } + match self.seals.compare_exchange_weak( + current, + current | seals, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => return Ok(()), + Err(updated) => current = updated, + } + } + } +} + +impl Task { + /// Returns the stable backing identity for a regular file. Memfds carry their identity as + /// descriptor metadata; ordinary files converge through their filesystem device/inode pair. + /// With `create == false`, an ordinary file that has never had a shared mapping returns `None`. + pub(crate) fn shared_file_backing( + &self, + fd: &TypedFd, + create: bool, + ) -> Option { + if let Ok(backing) = self + .global + .litebox + .descriptor_table() + .with_metadata(fd, MemfdBacking::shared_futex_backing) + { + return Some(backing); + } + let status = self.files.borrow().fs.fd_file_status(fd).ok()?; + if status.file_type != litebox::fs::FileType::RegularFile { + return None; + } + let key = (status.node_info.dev, status.node_info.ino); + let mut backings = self.global.shared_file_backings.lock(); + if let Some(backing) = backings.get(&key) { + return Some(*backing); + } + if !create { + return None; + } + let backing = litebox::mm::linux::SharedFutexBacking::new(); + backings.insert(key, backing); + Some(backing) + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum PtySide { + Master, + Slave, +} + +struct PtyState { + unlocked: AtomicBool, + termios: litebox::sync::Mutex, + winsize: litebox::sync::Mutex, + foreground_pgid: AtomicI32, + /// Guest descriptors currently referring to the slave end, across every process and every + /// `dup`/`fork`/`SCM_RIGHTS` copy; see [`PtyRegistry::release_slave_descriptor`]. + slave_descriptors: AtomicUsize, + /// Linux's `TTY_OTHER_CLOSED` as the master sees it: set when the last slave descriptor + /// closes, cleared when a slave is (re)opened. While set, a master `read` that would + /// otherwise block reports end-of-file, which is what a terminal emulator waits for once + /// its shell exits -- reversibly, so the slave path stays reopenable (Unix98 semantics). + other_closed: AtomicBool, +} + +impl PtyState { + fn new(foreground_pgid: i32) -> Self { + Self { + unlocked: AtomicBool::new(false), + termios: litebox::sync::Mutex::new(litebox_common_linux::Termios::default_cooked()), + winsize: litebox::sync::Mutex::new(litebox_common_linux::Winsize { + row: 24, + col: 80, + xpixel: 0, + ypixel: 0, + }), + foreground_pgid: AtomicI32::new(foreground_pgid), + slave_descriptors: AtomicUsize::new(0), + other_closed: AtomicBool::new(false), + } + } +} + +struct PendingPtySlave { + /// The registry's own reference to the slave endpoint, cloned into every `/dev/pts/` + /// open. Kept alive until the master closes (`PtyMasterLease::drop` removes this entry), + /// so the slave can be closed and reopened any number of times meanwhile. + handle: EntryHandle>, + state: Arc>, +} + +/// Unix98 pseudoterminals allocated by `/dev/ptmx`, shared by all guest tasks. +pub(crate) struct PtyRegistry { + next_number: AtomicU32, + slaves: litebox::sync::Mutex>>, +} + +impl PtyRegistry { + pub(crate) fn new() -> Self { + Self { + next_number: AtomicU32::new(0), + slaves: litebox::sync::Mutex::new(BTreeMap::new()), + } + } + + /// One descriptor to slave `number` closed. When it was the last, mark the line as hung up + /// *reversibly* (`PtyState::other_closed`): the registry keeps its own reference to the + /// slave endpoint until the master closes, so a later `/dev/pts/` open succeeds -- the + /// Unix98 pattern a terminal emulator relies on (the parent opens the slave to probe it, + /// closes it, then the session-leader child reopens it as its controlling terminal). The + /// master's readers observe the hangup through `other_closed` (see `do_read`); shutting the + /// socket pair down here, as an earlier revision did, made every reopen `EIO` for good. + fn release_slave_descriptor(&self, number: u32, state: &PtyState) { + let _ = number; + let previous = state + .slave_descriptors + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |n| n.checked_sub(1)); + if previous != Ok(1) { + return; + } + state.other_closed.store(true, Ordering::Release); + } +} + +/// A descriptor was just created for `fd`'s open file description; if that is a pty slave end, +/// count it (see [`PtyRegistry::release_slave_descriptor`]). +fn note_pty_slave_descriptor( + descriptors: &litebox::fd::Descriptors, + fd: &TypedFd, +) { + let _ = descriptors.with_metadata(fd, |endpoint: &PtyEndpoint| { + if endpoint.side == PtySide::Slave { + endpoint + .state + .slave_descriptors + .fetch_add(1, Ordering::AcqRel); + } + }); +} + +struct PtyMasterLease { + number: u32, + registry: Weak>, +} + +impl Drop for PtyMasterLease { + fn drop(&mut self) { + if let Some(registry) = self.registry.upgrade() { + registry.slaves.lock().remove(&self.number); + } + } +} + +/// Entry metadata that turns an AF_UNIX connected stream endpoint into one side +/// of a pseudoterminal while retaining the socket subsystem's blocking I/O and +/// poll behavior. +struct PtyEndpoint { + number: u32, + side: PtySide, + state: Arc>, + master_lease: Option>>, +} + +impl Clone for PtyEndpoint { + fn clone(&self) -> Self { + Self { + number: self.number, + side: self.side, + state: self.state.clone(), + master_lease: self.master_lease.clone(), + } + } +} + /// Path in the file system #[derive(Debug)] enum FsPath { @@ -124,7 +586,6 @@ enum FsPath { /// Current working directory Cwd, /// Path is relative to a file descriptor - #[expect(dead_code, reason = "currently unused, might want to use later")] FdRelative { fd: u32, path: CString }, /// Fd Fd(u32), @@ -133,6 +594,211 @@ enum FsPath { /// Maximum size of a file path pub const PATH_MAX: usize = 4096; +/// The absolute path a file-backed fd was opened with, attached as entry metadata (see +/// [`litebox::fd::Descriptors::set_entry_metadata`]) so `openat`/`fstatat`-family syscalls can +/// resolve a path given relative to that fd (`dirfd`-relative resolution). +/// +/// Entry metadata -- unlike fd metadata -- is shared across every descriptor that refers to the +/// same open file description, so a `dup`/`dup2`/`dup3`/`fcntl(F_DUPFD)` copy of a `dirfd` +/// resolves relative paths identically to the original without any extra propagation code. +#[derive(Clone, Debug)] +struct FdPath(CString); + +/// The status and access-mode flags a filesystem fd was opened with, attached as entry metadata +/// for `/proc//fdinfo`'s `flags:` line. (`fcntl(F_GETFL)` on a plain file still answers from +/// the `Backend` layer -- see `sys_fcntl`.) +#[derive(Clone, Copy, Debug)] +struct FdOpenFlags(OFlags); + +/// The calling task's descriptor table as `/proc//fd` and `fdinfo` describe it (see +/// [`litebox::fs::proc::ProcFdTable`]). Weak on both ends: it is published into the `/proc` +/// backend, which outlives any one task. +struct ProcFdView { + global: Weak>, + files: Weak>, +} + +impl litebox::fs::proc::ProcFdTable + for ProcFdView +{ + fn fds(&self) -> alloc::vec::Vec { + self.files + .upgrade() + .map_or_else(alloc::vec::Vec::new, |files| { + files + .raw_descriptor_store + .read() + .iter_alive() + .filter_map(|fd| u32::try_from(fd).ok()) + .collect() + }) + } + + fn entry(&self, fd: u32) -> Option { + let global = self.global.upgrade()?; + let files = self.files.upgrade()?; + describe_raw_fd(&global, &files, usize::try_from(fd).ok()?) + } +} + +/// Every live guest process as `/proc` lists it (see [`litebox::fs::proc::ProcTaskTable`]): +/// a window onto the shim-wide process table. Weak, like [`ProcFdView`]: it is published into +/// the `/proc` backend, which outlives any one task. +struct ProcTaskView { + global: Weak>, +} + +impl litebox::fs::proc::ProcTaskTable + for ProcTaskView +{ + fn pids(&self) -> alloc::vec::Vec { + self.global + .upgrade() + .map_or_else(alloc::vec::Vec::new, |global| global.processes.live_pids()) + } + + fn task(&self, pid: i32) -> Option { + self.global.upgrade()?.processes.proc_task_info(pid) + } +} + +/// How `/proc//fd/` and `fdinfo/` describe one live descriptor: the path a +/// filesystem fd was opened by (the stdio device names for the fixed stdio fds), `/dev/ptmx` or +/// `/dev/pts/` for a pseudoterminal end, and the kernel's `socket:[..]`/`pipe:[..]`/ +/// `anon_inode:..` spellings for the rest -- numbered by the descriptor itself, since those +/// subsystems have no inode of their own here. +fn describe_raw_fd( + global: &GlobalState, + files: &FilesState, + raw_fd: usize, +) -> Option { + use litebox::fs::proc::ProcFdEntry; + + let anonymous = |target: String, flags: OFlags| ProcFdEntry { + target, + pos: 0, + flags: flags.bits(), + ino: raw_fd as u64, + }; + if files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(raw_fd) + .is_ok() + { + return Some(anonymous(String::from("anon_inode:inotify"), OFlags::RDONLY)); + } + files + .run_on_raw_fd( + raw_fd, + |fd| { + // The global descriptor-table guard is scoped to these metadata + // lookups and dropped *before* the `fs` calls below: `seek` / + // `fd_file_status` re-enter that same non-recursive `RwLock` + // (layered's `fd_file_status` does its own `descriptor_table()` + // lookup), and if a writer queued in between, the nested read + // would wait behind the writer that is itself waiting on this + // very guard -- a same-thread self-deadlock that then wedges + // every fd operation guest-wide (live-verified: one leaked-era + // read guard here froze a whole desktop boot once a concurrent + // `open` needed `descriptor_table_mut`). + let (target, flags) = { + let dt = global.litebox.descriptor_table(); + let target = dt + .with_metadata(fd, |FdPath(path)| path.to_string_lossy().into_owned()) + .or_else(|_| { + dt.with_metadata(fd, |stream: &StdioStream| { + String::from(match stream { + StdioStream::Stdin => "/dev/stdin", + StdioStream::Stdout => "/dev/stdout", + StdioStream::Stderr => "/dev/stderr", + }) + }) + }) + .or_else(|_| { + dt.with_metadata(fd, |_: &MemfdBacking| String::from("/memfd: (deleted)")) + }) + .unwrap_or_else(|_| String::from("anon_inode:[file]")); + let flags = dt + .with_metadata(fd, |crate::StdioStatusFlags(flags)| *flags) + .or_else(|_| dt.with_metadata(fd, |FdOpenFlags(flags)| *flags)) + .unwrap_or(OFlags::RDONLY); + (target, flags) + }; + ProcFdEntry { + target, + pos: files + .fs + .seek(fd, 0, SeekWhence::RelativeToCurrentOffset) + .map_or(0, |pos| pos as u64), + flags: (flags & OFlags::STATUS_FLAGS_MASK).bits(), + ino: files + .fs + .fd_file_status(fd) + .map_or(0, |status| status.node_info.ino as u64), + } + }, + |_fd| anonymous(alloc::format!("socket:[{raw_fd}]"), OFlags::RDWR), + |fd| { + anonymous( + alloc::format!("pipe:[{raw_fd}]"), + global.linux_pipe_status_flags(fd).unwrap_or(OFlags::RDWR), + ) + }, + |_fd| anonymous(String::from("anon_inode:[eventfd]"), OFlags::RDWR), + |_fd| anonymous(String::from("anon_inode:[eventpoll]"), OFlags::RDWR), + |fd| { + let target = { + let dt = global.litebox.descriptor_table(); + dt.with_metadata(fd, |endpoint: &PtyEndpoint| match endpoint.side { + PtySide::Master => String::from("/dev/ptmx"), + PtySide::Slave => alloc::format!("/dev/pts/{}", endpoint.number), + }) + .unwrap_or_else(|_| alloc::format!("socket:[{raw_fd}]")) + }; + anonymous(target, OFlags::RDWR) + }, + |_fd| anonymous(alloc::format!("socket:[{raw_fd}]"), OFlags::RDWR), + ) + .ok() +} + +/// Entry metadata tagging a `/dev/input/event*` fd with its evdev minor number, attached at +/// open time (see `insert_raw_file_fd`). Entry-scoped (not fd-scoped) so `dup`ed copies share +/// it, and reachable from epoll's descriptor-table-only context where no filesystem handle is +/// in scope. +#[derive(Clone, Copy, Debug)] +pub(crate) struct InputEventMinor { + pub(crate) minor: usize, + /// `O_NONBLOCK`/`O_NDELAY` at open time. A later `fcntl(F_SETFL)` is NOT reflected here + /// (the fs-backend SETFL arm has no per-entry flag store yet); every real evdev consumer + /// observed (Xorg's evdev driver, libevdev, links2's mice path) picks blocking-ness at + /// `open(2)` and never toggles it. + pub(crate) nonblock: bool, +} + +/// The evdev minor for `file`, if it was tagged as a `/dev/input/event*` device at open time -- +/// the descriptor-table-only lookup epoll's poll path uses (it has `GlobalState` but no +/// filesystem access). +pub(crate) fn input_event_minor_of( + global: &crate::GlobalState, + file: &TypedFd, +) -> Option { + input_event_meta_of(global, file).map(|m| m.minor) +} + +/// [`input_event_minor_of`], with the open-time `O_NONBLOCK` flag alongside. +pub(crate) fn input_event_meta_of( + global: &crate::GlobalState, + file: &TypedFd, +) -> Option { + global + .litebox + .descriptor_table() + .with_metadata(file, |m: &InputEventMinor| *m) + .ok() +} + impl FsPath { /// Create a new `FsPath` from a dirfd and path. /// @@ -177,56 +843,577 @@ impl FsPath { } } +/// The `flock(2)`-holder identity used throughout this module: the guest-visible raw fd number. +/// +/// This is the one place that convention is spelled out, so `sys_flock` and the close-time lock +/// release (in `do_close_and_replace`) can never drift apart on how a holder is identified. See +/// [`litebox::fs::flock::FlockTable`]'s doc comment for what this convention does and doesn't +/// model correctly (in particular, around `dup`). +fn flock_holder_for_raw_fd(raw_fd: usize) -> u64 { + u64::try_from(raw_fd).unwrap_or(u64::MAX) +} + impl Task { + fn credentials_snapshot(&self) -> Arc { + self.credentials.borrow().clone() + } + + fn access_user_from_snapshot( + credentials: &crate::syscalls::process::Credentials, + effective: bool, + ) -> AccessUserInfo<'_> { + AccessUserInfo { + user: if effective { + credentials.euid + } else { + credentials.uid + }, + group: if effective { + credentials.egid + } else { + credentials.gid + }, + supplementary_groups: credentials.supplementary_groups(), + } + } + fn get_umask(&self) -> Mode { self.fs.borrow().umask() } - /// Resolve a path against the current working directory. - pub(crate) fn resolve_path(&self, path: impl path::Arg) -> Result { - let path_str = path.as_rust_str().map_err(|_| Errno::EINVAL)?; - if path_str.is_empty() { - return Err(Errno::ENOENT); - } - if path_str.starts_with('/') { - CString::new(path_str.to_string()).map_err(|_| Errno::EINVAL) - } else { - let mut cwd = self.fs.borrow().cwd.read().clone(); - cwd.push_str(path_str); - CString::new(cwd).map_err(|_| Errno::EINVAL) + /// `/proc` describes "the task looking at it" (see `litebox::fs::proc`): ahead of any lookup + /// under it, hand the backend this task's identity (what `self`/`thread-self` name) and + /// descriptor table, the table every other live process is looked up through, and the + /// network's addresses. + fn publish_proc_view(&self, path: &str) { + if !path.starts_with("/proc") { + return; } + let Some(proc) = self.global.proc_handle.as_ref() else { + return; + }; + proc.set_caller(self.tid, self.proc_task_info()); + proc.set_task_table(Arc::new(ProcTaskView { + global: Arc::downgrade(&self.global), + })); + proc.set_fd_table(Arc::new(ProcFdView { + global: Arc::downgrade(&self.global), + files: Arc::downgrade(&self.files.borrow()), + })); + let net = self.global.net.lock(); + proc.set_net_addrs(net.interface_ip(), net.gateway_ip()); } - /// Resolve a path relative to a dirfd. - /// - /// Note that an empty path is not valid for this function, and will be rejected with `ENOENT`. - fn resolve_path_at(&self, dirfd: i32, pathname: impl path::Arg) -> Result { - let get_cwd = || self.fs.borrow().cwd.read().clone(); - let fs_path = FsPath::new(dirfd, pathname, get_cwd)?; - match fs_path { - FsPath::Absolute { path } => Ok(path), - FsPath::Cwd | FsPath::Fd(_) => Err(Errno::ENOENT), - FsPath::FdRelative { fd: _, path: _ } => { - log_unsupported!("path resolution with FsPath::FdRelative"); - Err(Errno::EINVAL) - } - } + /// The calling process's root directory (see `FsState::root`), with its trailing '/'. + fn fs_root(&self) -> String { + self.fs.borrow().root.read().clone() } - pub(crate) fn do_open( - &self, - path: impl path::Arg, - flags: OFlags, + /// The real path a guest-visible absolute path names beneath `root`. Outside a `chroot` + /// (`root == "/"`) the path is returned untouched. Beneath one, `.` and `..` are collapsed + /// lexically first, so `..` can never climb above the root -- Linux stops `..` at the root + /// during its walk; collapsing ahead of the walk is what this shim already does for + /// cwd-relative paths -- and then the root is prefixed. A trailing '/' is kept, since a + /// backend may answer `ENOTDIR` differently for one. + fn map_absolute_to_root(root: &str, guest_abs: &str) -> String { + if root == "/" { + return String::from(guest_abs); + } + let mut real = String::from(root.trim_end_matches('/')); + let mut components: alloc::vec::Vec<&str> = alloc::vec::Vec::new(); + for component in guest_abs.split('/') { + match component { + "" | "." => {} + ".." => { + components.pop(); + } + component => components.push(component), + } + } + for component in components { + real.push('/'); + real.push_str(component); + } + if real.is_empty() || (guest_abs.ends_with('/') && !real.ends_with('/')) { + real.push('/'); + } + real + } + + /// The guest-visible spelling of the real path `real` beneath `root`: `real` with the root + /// prefix stripped, or -- for a path outside the root, which a cwd left behind by a + /// `chroot` without a `chdir` can be -- Linux's `(unreachable)` marker ahead of the real + /// path, exactly as its `getcwd(2)` reports one. + fn guest_visible_path(root: &str, real: &str) -> String { + if root == "/" { + return String::from(real); + } + let root_dir = root.trim_end_matches('/'); + match real.strip_prefix(root_dir) { + Some("") => String::from("/"), + Some(rest) if rest.starts_with('/') => String::from(rest), + _ => alloc::format!("(unreachable){real}"), + } + } + + /// [`FsPath::new`] with `chroot` applied: absolute and cwd-relative paths resolve beneath + /// the process's root the same way [`Self::resolve_path`] resolves them; `dirfd`-relative + /// ones start from the descriptor's own real path (see [`Self::join_dir_relative_path`]). + fn fs_path(&self, dirfd: i32, pathname: impl path::Arg) -> Result { + let get_cwd = || self.fs.borrow().cwd.read().clone(); + if self.fs_root() == "/" { + return FsPath::new(dirfd, pathname, get_cwd); + } + let path_str = pathname.as_rust_str()?; + if path_str.len() > PATH_MAX { + return Err(Errno::ENAMETOOLONG); + } + if path_str.starts_with('/') + || (dirfd == litebox_common_linux::AT_FDCWD && !path_str.is_empty()) + { + return Ok(FsPath::Absolute { + path: self.resolve_path(path_str)?, + }); + } + FsPath::new(dirfd, pathname, get_cwd) + } + + /// Handle syscall `chroot`: make `pathname` -- a directory the caller may search -- the + /// calling process's root directory (see `FsState::root`). Linux gates this on + /// `CAP_SYS_CHROOT`; LiteBox models no capability beyond root's, so an effective uid of 0. + /// The working directory is left where it is, as Linux leaves it: a caller that wants it + /// inside the new root follows up with `chdir("/")`. + pub(crate) fn sys_chroot(&self, pathname: impl path::Arg) -> Result<(), Errno> { + use litebox::fs::FileType; + use litebox::fs::errors::{FileStatusError, PathError}; + use litebox::path::Arg as _; + + let credentials = self.credentials_snapshot(); + if credentials.euid != 0 { + return Err(Errno::EPERM); + } + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let resolved = self.resolve_path(pathname)?; + let abs_path = self.resolve_syscall_path_as(fs_credentials, &resolved, true)?; + match self + .files + .borrow() + .fs + .file_status_as(fs_credentials, abs_path.as_str()) + { + Ok(status) => { + if status.file_type != FileType::Directory { + return Err(Errno::ENOTDIR); + } + Self::do_access_mode(&status, caller, &AccessFlags::X_OK)?; + } + Err(FileStatusError::PathError(PathError::NoSuchFileOrDirectory)) => { + return Err(Errno::ENOENT); + } + Err(FileStatusError::PathError(_)) => { + return Err(Errno::EACCES); + } + Err(_) => { + return Err(Errno::ENOENT); + } + } + let mut root = abs_path; + if !root.ends_with('/') { + root.push('/'); + } + *self.fs.borrow().root.write() = root; + Ok(()) + } + + /// Resolve a path against the current working directory (and, beneath a `chroot`, the + /// process's root -- see [`Self::map_absolute_to_root`]). + pub(crate) fn resolve_path(&self, path: impl path::Arg) -> Result { + let path_str = path.as_rust_str().map_err(|_| Errno::EINVAL)?; + if path_str.is_empty() { + return Err(Errno::ENOENT); + } + let root = self.fs_root(); + if path_str.starts_with('/') { + CString::new(Self::map_absolute_to_root(&root, path_str)).map_err(|_| Errno::EINVAL) + } else { + let cwd = self.fs.borrow().cwd.read().clone(); + let guest_cwd = Self::guest_visible_path(&root, &cwd); + if guest_cwd.starts_with('/') { + // Rebase on the guest-visible cwd, so that `..` is floored at the root, then map + // the result back beneath it (a no-op outside a `chroot`). + let mut guest = guest_cwd; + guest.push_str(path_str); + CString::new(Self::map_absolute_to_root(&root, &guest)).map_err(|_| Errno::EINVAL) + } else { + // A cwd outside the root (`chroot` without `chdir`): Linux walks relative paths + // from the real directory, where `..` never meets the root. + let mut real = cwd; + real.push_str(path_str); + CString::new(real).map_err(|_| Errno::EINVAL) + } + } + } + + /// Join a directory's absolute path with a path given relative to it, matching the semantics + /// `openat`/`fstatat`-family syscalls need for a `dirfd`-relative lookup. + fn join_dir_relative_path(&self, dir_path: &CString, relative: &CString) -> Result { + let mut joined = dir_path.to_str().map_err(|_| Errno::EINVAL)?.to_string(); + if !joined.ends_with('/') { + joined.push('/'); + } + joined.push_str(relative.to_str().map_err(|_| Errno::EINVAL)?); + // Beneath a `chroot`, `..` from a descriptor inside the root is floored at the root + // the same way an absolute path's is (Linux stops `..` at the root during its walk); a + // descriptor outside the root -- opened before the `chroot` -- walks the real tree. + let root = self.fs_root(); + if root != "/" { + let guest = Self::guest_visible_path(&root, &joined); + if guest.starts_with('/') { + joined = Self::map_absolute_to_root(&root, &guest); + } + } + CString::new(joined).map_err(|_| Errno::EINVAL) + } + + /// Resolve `dirfd` to the absolute path it was opened with (see [`FdPath`]), for + /// `dirfd`-relative resolution. A closed descriptor is `EBADF`; a live descriptor from a + /// non-filesystem subsystem (socket, pipe, eventfd, and so on) is `ENOTDIR`. + fn resolve_dirfd_path(&self, fd: u32) -> Result { + let files = self.files.borrow(); + files + .run_on_raw_fd( + fd as usize, + |fd| { + let status = files.fs.fd_file_status(fd).map_err(Errno::from)?; + if status.file_type != litebox::fs::FileType::Directory { + return Err(Errno::ENOTDIR); + } + self.global + .litebox + .descriptor_table() + .with_metadata(fd, |path: &FdPath| path.0.clone()) + .map_err(|error| match error { + MetadataError::ClosedFd => Errno::EBADF, + MetadataError::NoSuchMetadata => Errno::ENOTDIR, + }) + }, + |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), + ) + .flatten() + } + + /// The absolute path `fd` was opened with, if one was recorded (see + /// [`FdPath`]). Best-effort by design: sockets, pipes, and fds inherited + /// without a path resolve to `None`. Used by the ELF mapping code to name + /// guest images for fault symbolization. + pub(crate) fn fd_abs_path(&self, fd: i32) -> Option { + let raw_fd = usize::try_from(fd).ok()?; + let files = self.files.borrow(); + let raw_descriptors = files.raw_descriptor_store.read(); + let file = raw_descriptors.fd_from_raw_integer::(raw_fd).ok()?; + self.global + .litebox + .descriptor_table() + .with_metadata(&file, |path: &FdPath| path.0.clone()) + .ok() + } + + /// Handle `fchdir(2)` using the opened directory's entry-scoped path metadata. + pub(crate) fn sys_fchdir(&self, fd: i32) -> Result<(), Errno> { + use litebox::path::Arg as _; + + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let raw_fd = usize::try_from(fd).map_err(|_| Errno::EBADF)?; + let files = self.files.borrow(); + let file = { + let descriptors = files.raw_descriptor_store.read(); + if !descriptors.is_alive(raw_fd) { + return Err(Errno::EBADF); + } + descriptors + .fd_from_raw_integer::(raw_fd) + .map_err(|error| match error { + litebox::fd::ErrRawIntFd::NotFound => Errno::EBADF, + litebox::fd::ErrRawIntFd::InvalidSubsystem => Errno::ENOTDIR, + })? + }; + let status = files + .fs + .fd_file_status(&file) + .map_err(|error| match error { + litebox::fs::errors::FileStatusError::ClosedFd => Errno::EBADF, + litebox::fs::errors::FileStatusError::Io + | litebox::fs::errors::FileStatusError::PathError(_) => Errno::EIO, + _ => Errno::EIO, + })?; + if status.file_type != litebox::fs::FileType::Directory { + return Err(Errno::ENOTDIR); + } + Self::do_access_mode(&status, caller, &AccessFlags::X_OK)?; + let path = self + .global + .litebox + .descriptor_table() + .with_metadata(&file, |path: &FdPath| path.0.clone()) + .map_err(|error| match error { + MetadataError::ClosedFd => Errno::EBADF, + MetadataError::NoSuchMetadata => Errno::EIO, + })?; + drop(files); + + let mut cwd = path.normalized().map_err(|_| Errno::EINVAL)?; + if cwd.is_empty() { + cwd.push('/'); + } else if !cwd.starts_with('/') { + return Err(Errno::EIO); + } + if !cwd.ends_with('/') { + cwd.push('/'); + } + *self.fs.borrow().cwd.write() = cwd; + Ok(()) + } + + /// Resolve a path relative to a dirfd. + /// + /// Note that an empty path is not valid for this function, and will be rejected with `ENOENT`. + fn resolve_path_at(&self, dirfd: i32, pathname: impl path::Arg) -> Result { + let fs_path = self.fs_path(dirfd, pathname)?; + match fs_path { + FsPath::Absolute { path } => Ok(path), + FsPath::Cwd | FsPath::Fd(_) => Err(Errno::ENOENT), + FsPath::FdRelative { fd, path } => { + let dir_path = self.resolve_dirfd_path(fd)?; + self.join_dir_relative_path(&dir_path, &path) + } + } + } + + /// Resolve a cwd/`dirfd`-joined path for a path-taking syscall other than `open`: every + /// intermediate symlink is followed (a symlinked directory such as the image's + /// `/var/run -> ../run` is transparent, as Linux's walk makes it), the final component only + /// when `follow_final`, and a `..` right after an intermediate link applies to the link's + /// target rather than being collapsed lexically ahead of link expansion. Without this, + /// `lstat`/`readlink`/`mkdir`/`unlink`/`rename`/`symlink`/`chown`/`utimensat` answered + /// `ENOTDIR` through any symlinked directory and `stat(/tmp/bindir/../etc/hosts)` + /// (bindir -> /bin) answered `ENOENT`, while `open` of the same names worked. + fn resolve_syscall_path_as( + &self, + credentials: AccessCredentials<'_>, + path: &CString, + follow_final: bool, + ) -> Result { + self.resolve_path_symlinks_with_final_as( + credentials, + path.to_str().map_err(|_| Errno::EINVAL)?, + follow_final, + ) + } + + fn do_open_raw_as( + &self, + credentials: AccessCredentials<'_>, + path: impl path::Arg, + flags: OFlags, mode: Mode, - ) -> Result, Errno> { + ) -> Result, litebox::fs::errors::OpenError> { let mode = mode & !self.get_umask(); self.files .borrow() .fs - .open(path, flags - OFlags::CLOEXEC, mode) + .open_as(credentials, path, flags - OFlags::CLOEXEC, mode) + } + + fn do_open_as( + &self, + credentials: AccessCredentials<'_>, + path: impl path::Arg, + flags: OFlags, + mode: Mode, + ) -> Result, Errno> { + self.do_open_raw_as(credentials, path, flags, mode) .map_err(Errno::from) } + /// Linux caps a single path resolution at `MAXSYMLINKS` (40) followed links. + const MAX_SYMLINK_HOPS: usize = 40; + + /// Resolve every symbolic link on an already-cwd-resolved absolute `path`, + /// per `path_resolution(7)`: walk it component by component and, whenever a + /// component is a symlink, splice in its target (an absolute target restarts + /// from `/`, a relative one is interpreted from the directory that contains + /// the link) and keep going -- so a symlink used as an *intermediate* + /// directory component is followed, not only the final one. + /// + /// A component that does not exist stops resolution and is returned verbatim + /// with whatever is still pending, so `O_CREAT` can still create a missing + /// final component and a genuinely missing path yields `ENOENT` from the real + /// operation rather than here. `ELOOP` once more than + /// [`Self::MAX_SYMLINK_HOPS`] links are followed. + /// + /// `..` components stay in the pending queue until the walk reaches them, so + /// one immediately after an intermediate symlink is applied to the followed + /// target rather than being collapsed against the link's lexical parent. + fn resolve_path_symlinks_with_final_as( + &self, + credentials: AccessCredentials<'_>, + path: &str, + follow_final: bool, + ) -> Result { + use alloc::collections::VecDeque; + use alloc::vec::Vec; + use litebox::fs::FileType; + use litebox::fs::errors::{FileStatusError, PathError}; + + self.publish_proc_view(path); + + let into_components = |s: &str| -> VecDeque { + s.split('/') + .filter(|component| !component.is_empty() && *component != ".") + .map(String::from) + .collect() + }; + // Beneath a `chroot`, `..` stops at the root and an absolute link target restarts from + // it, exactly as Linux's walk does against `nd->root`. Outside one the root is `/`, + // whose component list is empty, and both rules reduce to the plain ones. + let root_components: Vec = into_components(&self.fs_root()).into(); + let mut pending = into_components(path); + let mut resolved: Vec = Vec::new(); + let mut hops = 0usize; + + while let Some(name) = pending.pop_front() { + if name == ".." { + // Applied to the already-resolved prefix -- i.e. after any symlink + // in it was followed -- which is the correct base component. + if resolved != root_components { + resolved.pop(); + } + continue; + } + let mut candidate = String::new(); + for component in &resolved { + candidate.push('/'); + candidate.push_str(component); + } + candidate.push('/'); + candidate.push_str(&name); + + let file_type = match self + .files + .borrow() + .fs + .file_status_as(credentials, candidate.as_str()) + { + Ok(status) => status.file_type, + Err(FileStatusError::PathError( + PathError::NoSuchFileOrDirectory | PathError::MissingComponent, + )) => { + // This component does not exist: keep it and the rest verbatim + // and let the real operation decide (ENOENT vs O_CREAT). + resolved.push(name); + resolved.extend(pending); + return Ok(Self::join_absolute(&resolved)); + } + Err(e) => return Err(Errno::from(e)), + }; + + // A `/proc//fd/` magic link names an *open file*, not a path: following it + // must land on that file even when its `readlink` text is `socket:[7]`, + // `anon_inode:[eventfd]` or a since-unlinked temporary (Chromium's shared-memory + // files). Stop here with the magic path itself; `stat`/`open` of such a path answer + // from the descriptor (see `Task::proc_fd_magic_link`). An intermediate use + // (`/proc/self/fd/3/child`, a directory fd) is followed by text as before. + if file_type == FileType::SymLink + && pending.is_empty() + && self.proc_fd_magic_link(&candidate).is_some() + { + resolved.push(name); + continue; + } + if file_type == FileType::SymLink && (follow_final || !pending.is_empty()) { + hops += 1; + if hops > Self::MAX_SYMLINK_HOPS { + return Err(Errno::ELOOP); + } + let target = self + .files + .borrow() + .fs + .readlink_as(credentials, candidate.as_str()) + .map_err(Errno::from)?; + if target.is_empty() { + return Err(Errno::ENOENT); + } + // An absolute target restarts resolution from the root; a relative + // one continues from `resolved` (the link's directory, since the + // link's own name was not pushed). + if target.starts_with('/') { + resolved.clone_from(&root_components); + } + for component in target + .split('/') + .filter(|component| !component.is_empty() && *component != ".") + .rev() + { + pending.push_front(String::from(component)); + } + } else { + resolved.push(name); + } + } + Ok(Self::join_absolute(&resolved)) + } + + /// Join resolved path components into an absolute path (`/` when empty). + fn join_absolute(components: &[String]) -> String { + if components.is_empty() { + return String::from("/"); + } + let mut path = String::new(); + for component in components { + path.push('/'); + path.push_str(component); + } + path + } + + /// Apply `open(2)` default symlink-following to an already-resolved absolute + /// path. Two flag combinations keep only the final link opaque: `O_NOFOLLOW` (the backend + /// answers `ELOOP`) and `O_CREAT|O_EXCL` (an existing final link is `EEXIST`, never followed). + /// Intermediate links are followed in every mode, as Linux requires. + pub(crate) fn follow_open_path( + &self, + path: CString, + flags: OFlags, + ) -> Result { + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + self.follow_open_path_as(caller.as_fs_credentials(), path, flags) + } + + fn follow_open_path_as( + &self, + credentials: AccessCredentials<'_>, + path: CString, + flags: OFlags, + ) -> Result { + let follow_final = !flags.contains(OFlags::NOFOLLOW) + && !flags.contains(OFlags::CREAT | OFlags::EXCL); + let resolved = self.resolve_path_symlinks_with_final_as( + credentials, + path.to_str().map_err(|_| Errno::EINVAL)?, + follow_final, + )?; + CString::new(resolved).map_err(|_| Errno::EINVAL) + } + fn do_openat( &self, dirfd: i32, @@ -234,11 +1421,23 @@ impl Task { flags: OFlags, mode: Mode, ) -> Result, Errno> { + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); let path = self.resolve_path_at(dirfd, pathname)?; - self.do_open(path, flags, mode) + let path = self.follow_open_path_as(fs_credentials, path, flags)?; + self.do_open_as(fs_credentials, path, flags, mode) } - fn insert_raw_file_fd(&self, file: TypedFd, flags: OFlags) -> Result { + /// Insert a freshly-opened file into the raw fd table, optionally recording the absolute + /// path it was opened with (see [`FdPath`]) so it can later serve as a `dirfd` for + /// `openat`/`fstatat`-family syscalls. + fn insert_raw_file_fd( + &self, + file: TypedFd, + flags: OFlags, + path: Option, + ) -> Result { if flags.contains(OFlags::CLOEXEC) { let None = self .global @@ -249,10 +1448,54 @@ impl Task { unreachable!() }; } + if let Some(path) = path { + let old = self + .global + .litebox + .descriptor_table_mut() + .set_entry_metadata(&file, FdPath(path)); + debug_assert!(old.is_none()); + } + let old = self + .global + .litebox + .descriptor_table_mut() + .set_entry_metadata(&file, FdOpenFlags(flags & OFlags::STATUS_FLAGS_MASK)); + debug_assert!(old.is_none()); + // Tag `/dev/input/event*` fds with their evdev minor at open time (recognized by the + // input-core rdev major, same idea as `is_stdio`'s major check), so the read/ioctl/poll + // paths -- epoll in particular, which has no filesystem access, only the descriptor + // table -- can identify them by metadata lookup alone. Mirrors the `StdioStream` + // metadata the stdio fds carry. + { + let files = self.files.borrow(); + if !flags.contains(OFlags::PATH) + && let Ok(status) = files.fs.fd_file_status(&file) + && status.file_type == litebox::fs::FileType::CharacterDevice + && let Some(rdev) = status.node_info.rdev + && rdev.get() >> 8 == litebox::fs::devices::INPUT_MAJOR + { + let old = self + .global + .litebox + .descriptor_table_mut() + .set_entry_metadata( + &file, + InputEventMinor { + minor: rdev.get() & 0xff, + nonblock: flags.intersects(OFlags::NONBLOCK | OFlags::NDELAY), + }, + ); + debug_assert!(old.is_none()); + } + } let files = self.files.borrow(); let raw_fd = files.insert_raw_fd(file).map_err(|file| { - files.fs.close(&file).unwrap(); - Errno::EMFILE + if files.fs.close(&file).is_err() { + Errno::EIO + } else { + Errno::EMFILE + } })?; Ok(u32::try_from(raw_fd).unwrap()) } @@ -268,11 +1511,205 @@ impl Task { Mode::from_bits_retain(old_mask) } - /// Handle syscall `open` - pub fn sys_open(&self, path: impl path::Arg, flags: OFlags, mode: Mode) -> Result { - let path = self.resolve_path(path)?; - let file = self.do_open(path, flags, mode)?; - self.insert_raw_file_fd(file, flags) + /// Open `/dev/ptmx`, one allocated `/dev/pts/` endpoint, or the calling process's + /// `/dev/tty` alias. Returning `None` means the path is not in the pseudoterminal namespace. + fn open_pty_path(&self, path: &CStr, flags: OFlags) -> Option> { + if flags.contains(OFlags::PATH) { + return None; + } + let Ok(path) = path.to_str() else { + return Some(Err(Errno::EINVAL)); + }; + if path == "/dev/ptmx" { + return Some(self.open_ptmx(flags)); + } + if path == "/dev/tty" { + return Some(self.open_controlling_tty(flags)); + } + let number = path + .strip_prefix("/dev/pts/") + .and_then(|number| number.parse::().ok())?; + Some(self.open_pty_slave(number, flags)) + } + + fn open_ptmx(&self, flags: OFlags) -> Result { + let mut socket_flags = SockFlags::empty(); + socket_flags.set( + SockFlags::NONBLOCK, + flags.intersects(OFlags::NONBLOCK | OFlags::NDELAY), + ); + let (master, slave) = + super::unix::UnixSocket::new_connected_pair(SockType::Stream, socket_flags, self) + .ok_or(Errno::ENOSPC)?; + let number = self + .global + .pty_registry + .next_number + .fetch_add(1, Ordering::Relaxed); + let state = Arc::new(PtyState::new(self.process().process_group_id())); + let lease = Arc::new(PtyMasterLease { + number, + registry: Arc::downgrade(&self.global.pty_registry), + }); + + let (master_fd, slave_handle) = { + let mut descriptors = self.global.litebox.descriptor_table_mut(); + let master_fd = + descriptors.insert::>(master); + let slave_fd = + descriptors.insert::>(slave); + let old = descriptors.set_entry_metadata( + &master_fd, + PtyEndpoint { + number, + side: PtySide::Master, + state: state.clone(), + master_lease: Some(lease), + }, + ); + debug_assert!(old.is_none()); + let old = descriptors.set_entry_metadata( + &slave_fd, + PtyEndpoint:: { + number, + side: PtySide::Slave, + state: state.clone(), + master_lease: None, + }, + ); + debug_assert!(old.is_none()); + if flags.contains(OFlags::CLOEXEC) { + let old = descriptors.set_fd_metadata(&master_fd, FileDescriptorFlags::FD_CLOEXEC); + debug_assert!(old.is_none()); + } + let slave_handle = descriptors.entry_handle(&slave_fd).unwrap(); + let removed = descriptors.remove(&slave_fd); + debug_assert!(removed.is_none()); + (master_fd, slave_handle) + }; + let old = self.global.pty_registry.slaves.lock().insert( + number, + PendingPtySlave { + handle: slave_handle, + state, + }, + ); + debug_assert!(old.is_none()); + + let files = self.files.borrow(); + files + .insert_raw_fd(master_fd) + .map(u32::try_from) + .map_err(|master_fd| { + let _ = self + .global + .litebox + .descriptor_table_mut() + .remove(&master_fd); + Errno::EMFILE + }) + .and_then(|raw| raw.map_err(|_| Errno::EMFILE)) + } + + fn open_controlling_tty(&self, flags: OFlags) -> Result { + match self.process().controlling_pty() { + Some(number) => self + .open_pty_slave(number, flags | OFlags::NOCTTY) + .map_err(|error| { + if error == Errno::ENOENT { + Errno::ENXIO + } else { + error + } + }), + None => self.open_host_terminal(flags), + } + } + + /// `/dev/tty` for a process whose controlling terminal is the host's own stdio terminal -- + /// the initial task and whatever it forked without a pseudoterminal in between: the + /// `/dev/tty` device node, tagged like the fixed stdio fds so `TCGETS`/`TIOCGWINSZ`/... answer + /// on it. `ENXIO` when none of the runner's stdio streams is a terminal, as on Linux for a + /// process without a controlling terminal. `setsid()` is not tracked here: a session leader + /// that gave the host terminal up still reaches it, where Linux would say `ENXIO`. + fn open_host_terminal(&self, flags: OFlags) -> Result { + let stream = [StdioStream::Stdin, StdioStream::Stdout, StdioStream::Stderr] + .into_iter() + .find(|stream| self.global.platform.is_a_tty(*stream)) + .ok_or(Errno::ENXIO)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let path = CString::from(c"/dev/tty"); + let file = self.do_open_as(caller.as_fs_credentials(), path.clone(), flags, Mode::empty())?; + { + let mut status = OFlags::RDWR; + status.set( + OFlags::NONBLOCK, + flags.intersects(OFlags::NONBLOCK | OFlags::NDELAY), + ); + let mut dt = self.global.litebox.descriptor_table_mut(); + let old = dt.set_entry_metadata(&file, stream); + debug_assert!(old.is_none()); + let old = dt.set_entry_metadata(&file, crate::StdioStatusFlags(status)); + debug_assert!(old.is_none()); + } + self.insert_raw_file_fd(file, flags, Some(path)) + } + + fn open_pty_slave(&self, number: u32, flags: OFlags) -> Result { + let (handle, state) = { + let slaves = self.global.pty_registry.slaves.lock(); + let slave = slaves.get(&number).ok_or(Errno::ENOENT)?; + (slave.handle.clone(), slave.state.clone()) + }; + if !state.unlocked.load(Ordering::Acquire) { + return Err(Errno::EIO); + } + let slave_fd = { + let mut descriptors = self.global.litebox.descriptor_table_mut(); + let slave_fd = descriptors.insert_handle(handle); + if flags.contains(OFlags::CLOEXEC) { + let old = descriptors.set_fd_metadata(&slave_fd, FileDescriptorFlags::FD_CLOEXEC); + debug_assert!(old.is_none()); + } + slave_fd + }; + let files = self.files.borrow(); + let raw = files + .insert_raw_fd(slave_fd) + .map(u32::try_from) + .map_err(|slave_fd| { + let _ = self.global.litebox.descriptor_table_mut().remove(&slave_fd); + Errno::EMFILE + }) + .and_then(|raw| raw.map_err(|_| Errno::EMFILE))?; + state.slave_descriptors.fetch_add(1, Ordering::AcqRel); + // A (re)opened slave un-hangs the line for the master (Linux clears + // `TTY_OTHER_CLOSED` in `pty_open`). + state.other_closed.store(false, Ordering::Release); + if !flags.contains(OFlags::NOCTTY) { + // Opening a slave without O_NOCTTY acquires it only for a session leader that has no + // controlling terminal. Failure to acquire never makes the open itself fail. Only a + // new acquisition makes the caller's process group the terminal's foreground group; + // reopening an existing controlling terminal must not reset a later TIOCSPGRP choice. + if self.process().acquire_controlling_pty(self.pid, number) == Ok(true) { + state + .foreground_pgid + .store(self.process().process_group_id(), Ordering::Release); + } + } + Ok(raw) + } + + fn pty_endpoint( + &self, + fd: &TypedFd>, + ) -> Option> { + self.global + .litebox + .descriptor_table() + .with_metadata(fd, |endpoint: &PtyEndpoint| endpoint.clone()) + .ok() } /// Handle syscall `openat` @@ -283,12 +1720,182 @@ impl Task { flags: OFlags, mode: Mode, ) -> Result { - let file = self.do_openat(dirfd, pathname, flags, mode)?; - self.insert_raw_file_fd(file, flags) + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let flags = flags.normalized_for_open(); + let path = self.resolve_path_at(dirfd, pathname)?; + let path = self.follow_open_path_as(fs_credentials, path, flags)?; + let result = match self.open_pty_path(&path, flags) { + Some(result) => result, + None => self + .do_open_as(fs_credentials, path.clone(), flags, mode) + .and_then(|file| self.insert_raw_file_fd(file, flags, Some(path.clone()))), + }; + // The `req=Openat` trace line above this only shows the user pointer; the resolved + // path with the outcome is what a syscall-level diagnosis actually needs. + litebox_util_log::trace!(path:? = path, result:? = result; "openat"); + result } /// Handle syscall `ftruncate` + /// Handle syscall `fadvise64` (`posix_fadvise`). + /// + /// Every `POSIX_FADV_*` advice is a readahead/page-cache hint, and memory-backed files have + /// neither, so the accepted advice is a no-op; the argument checks are Linux's + /// `ksys_fadvise64_64` ones (`EBADF`, `EINVAL` for unknown advice or a negative length, + /// `ESPIPE` for a pipe or FIFO). + pub(crate) fn sys_fadvise64( + &self, + fd: i32, + _offset: usize, + len: usize, + advice: i32, + ) -> Result<(), Errno> { + // POSIX_FADV_NORMAL..=POSIX_FADV_NOREUSE (0..=5); aarch64 uses the generic numbering. + if !(0..=5).contains(&advice) { + return Err(Errno::EINVAL); + } + if len.reinterpret_as_signed() < 0 { + return Err(Errno::EINVAL); + } + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |_fd| Ok(()), + |_fd| Err(Errno::ESPIPE), + |_fd| Err(Errno::ESPIPE), + |_fd| Ok(()), + |_fd| Ok(()), + |_fd| Err(Errno::ESPIPE), + |_fd| Err(Errno::ESPIPE), + ) + .flatten() + } + + /// Handle syscall `fallocate`. + /// + /// The guest filesystems here are memory-backed like `tmpfs`, and this models exactly + /// `tmpfs`'s `fallocate`: mode `0` extends the file (zero-filled) when the range reaches past + /// EOF and is otherwise a no-op (space is always "allocated"); `FALLOC_FL_KEEP_SIZE` alone is + /// a no-op; `FALLOC_FL_PUNCH_HOLE | FALLOC_FL_KEEP_SIZE` zeroes the in-file part of the range; + /// every other mode (`ZERO_RANGE`, `COLLAPSE_RANGE`, `INSERT_RANGE`, `UNSHARE_RANGE`) is + /// `EOPNOTSUPP`, as `tmpfs` reports it. Argument checks are Linux's `vfs_fallocate` ones. + pub(crate) fn sys_fallocate( + &self, + fd: i32, + mode: i32, + offset: usize, + len: usize, + ) -> Result<(), Errno> { + const FALLOC_FL_KEEP_SIZE: i32 = 0x01; + const FALLOC_FL_PUNCH_HOLE: i32 = 0x02; + const FALLOC_FL_NO_HIDE_STALE: i32 = 0x04; + const FALLOC_FL_COLLAPSE_RANGE: i32 = 0x08; + const FALLOC_FL_ZERO_RANGE: i32 = 0x10; + const FALLOC_FL_INSERT_RANGE: i32 = 0x20; + const FALLOC_FL_UNSHARE_RANGE: i32 = 0x40; + const KNOWN: i32 = FALLOC_FL_KEEP_SIZE + | FALLOC_FL_PUNCH_HOLE + | FALLOC_FL_NO_HIDE_STALE + | FALLOC_FL_COLLAPSE_RANGE + | FALLOC_FL_ZERO_RANGE + | FALLOC_FL_INSERT_RANGE + | FALLOC_FL_UNSHARE_RANGE; + + let offset = usize::try_from(offset.reinterpret_as_signed()).map_err(|_| Errno::EINVAL)?; + let len = usize::try_from(len.reinterpret_as_signed()).map_err(|_| Errno::EINVAL)?; + if len == 0 { + return Err(Errno::EINVAL); + } + let end = offset.checked_add(len).ok_or(Errno::EFBIG)?; + if mode & !KNOWN != 0 { + return Err(Errno::EOPNOTSUPP); + } + // `PUNCH_HOLE` must come with `KEEP_SIZE`, and the exclusive modes cannot be combined. + if mode & FALLOC_FL_PUNCH_HOLE != 0 && mode & FALLOC_FL_KEEP_SIZE == 0 { + return Err(Errno::EOPNOTSUPP); + } + let exclusive = mode + & (FALLOC_FL_PUNCH_HOLE + | FALLOC_FL_COLLAPSE_RANGE + | FALLOC_FL_ZERO_RANGE + | FALLOC_FL_INSERT_RANGE); + if exclusive.count_ones() > 1 { + return Err(Errno::EINVAL); + } + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + // Linux: the fd must be open for writing (`EBADF`), and be a regular file (`ENODEV`) + // or a directory (`EISDIR`); pipes and sockets are `ESPIPE`. + let files = self.files.borrow(); + let (open_flags, status) = files + .run_on_raw_fd( + raw_fd, + |fd| { + let flags = self + .global + .litebox + .descriptor_table() + .with_metadata(fd, |FdOpenFlags(flags)| *flags) + .map_err(|_| Errno::EBADF)?; + let status = files.fs.fd_file_status(fd).map_err(Errno::from)?; + Ok((flags, status)) + }, + |_fd| Err(Errno::ESPIPE), + |_fd| Err(Errno::ESPIPE), + |_fd| Err(Errno::ENODEV), + |_fd| Err(Errno::ENODEV), + |_fd| Err(Errno::ESPIPE), + |_fd| Err(Errno::ESPIPE), + ) + .flatten()?; + if open_flags.contains(OFlags::PATH) || !open_flags.intersects(OFlags::WRONLY | OFlags::RDWR) { + return Err(Errno::EBADF); + } + match status.file_type { + litebox::fs::FileType::RegularFile => {} + litebox::fs::FileType::Directory => return Err(Errno::EISDIR), + _ => return Err(Errno::ENODEV), + } + drop(files); + let size = status.size; + if mode & (FALLOC_FL_COLLAPSE_RANGE | FALLOC_FL_INSERT_RANGE | FALLOC_FL_UNSHARE_RANGE) != 0 + || mode & FALLOC_FL_ZERO_RANGE != 0 + { + // `tmpfs` (`shmem_fallocate`) only implements the plain and hole-punching modes. + return Err(Errno::EOPNOTSUPP); + } + if mode & FALLOC_FL_PUNCH_HOLE != 0 { + // Zero the part of the range that lies inside the file; the file size is untouched. + let stop = end.min(size); + let zeros = vec![0u8; PAGE_SIZE]; + let mut cur = offset; + while cur < stop { + let chunk = (stop - cur).min(zeros.len()); + let written = self.sys_write(fd, &zeros[..chunk], Some(cur))?; + if written == 0 { + return Err(Errno::EIO); + } + cur += written; + } + return Ok(()); + } + if mode & FALLOC_FL_KEEP_SIZE == 0 && end > size { + // Plain preallocation past EOF grows the file, zero-filled -- the same path as + // `ftruncate`, including zeroing any shared file mapping of the new tail. + self.sys_ftruncate(fd, end)?; + } + Ok(()) + } + pub(crate) fn sys_ftruncate(&self, fd: i32, length: usize) -> Result<(), Errno> { + let length = usize::try_from(length.reinterpret_as_signed()).map_err(|_| Errno::EINVAL)?; let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { return Err(Errno::EBADF); }; @@ -296,9 +1903,33 @@ impl Task { files .run_on_raw_fd( raw_fd, - |fd| files.fs.truncate(fd, length, false).map_err(Errno::from), - |_fd| todo!("net"), - |_fd| todo!("pipes"), + |fd| { + let old_size = files + .fs + .fd_file_status(fd) + .map_err(Errno::from)? + .size; + let shared_backing = self.shared_file_backing(fd, false); + files + .fs + .truncate(fd, length, false) + .map_err(Errno::from)?; + if let Some(backing) = shared_backing + && old_size != length + { + self.global + .platform + .zero_shared_pages( + backing.identity(), + old_size.min(length)..old_size.max(length), + ) + .map_err(|_| Errno::EIO)?; + } + Ok(()) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), |_fd| Err(Errno::EINVAL), |_fd| Err(Errno::EINVAL), |_fd| Err(Errno::EINVAL), @@ -306,6 +1937,78 @@ impl Task { .flatten() } + /// Handle syscall `memfd_create`. + pub(crate) fn sys_memfd_create(&self, name: &CStr, flags: u32) -> Result { + const MFD_CLOEXEC: u32 = 0x0001; + const MFD_ALLOW_SEALING: u32 = 0x0002; + const MAX_NAME_LEN: usize = 249; + + if flags & !(MFD_CLOEXEC | MFD_ALLOW_SEALING) != 0 { + return Err(Errno::EINVAL); + } + if name.to_bytes().len() > MAX_NAME_LEN { + return Err(Errno::ENAMETOOLONG); + } + + let mut open_flags = OFlags::CREAT | OFlags::EXCL | OFlags::RDWR; + if flags & MFD_CLOEXEC != 0 { + open_flags |= OFlags::CLOEXEC; + } + + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + // The generic filesystem interface has no anonymous-inode constructor. Create a private + // regular file and unlink it before publishing its descriptor; the open file description + // keeps the inode alive and supplies the read/write/truncate/mmap behavior memfds need. + for _ in 0..64 { + let id = NEXT_MEMFD_ID.fetch_add(1, Ordering::Relaxed); + let path = alloc::format!("/tmp/.litebox-memfd-{}-{id}", self.pid); + let file = match self.do_open_raw_as( + fs_credentials, + path.as_str(), + open_flags, + Mode::RUSR | Mode::WUSR, + ) { + Ok(file) => file, + Err(litebox::fs::errors::OpenError::AlreadyExists) => continue, + Err(error) => { + litebox_util_log::error!( + pid:? = self.pid, + tid:? = self.tid, + attempt_id = id, + path:% = path, + error:? = error; + "memfd backing open failed" + ); + return Err(Errno::from(error)); + } + }; + let unlink_result = { + let files = self.files.borrow(); + files.fs.unlink_as(fs_credentials, path.as_str()) + }; + if let Err(error) = unlink_result { + let close_failed = { + let files = self.files.borrow(); + files.fs.close(&file).is_err() + }; + if close_failed { + return Err(Errno::EIO); + } + return Err(Errno::from(error)); + } + let old = self + .global + .litebox + .descriptor_table_mut() + .set_entry_metadata(&file, MemfdBacking::new(flags & MFD_ALLOW_SEALING != 0)); + assert!(old.is_none()); + return self.insert_raw_file_fd(file, open_flags, None); + } + Err(Errno::EAGAIN) + } + /// Handle syscall `mknodat` — create a filesystem node. pub(crate) fn sys_mknodat( &self, @@ -357,11 +2060,401 @@ impl Task { return Err(Errno::EINVAL); } + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); let path = self.resolve_path_at(dirfd, pathname)?; + // `unlink(2)`/`rmdir(2)` act on the final name itself, never on a link's target. + let path = self.resolve_syscall_path_as(fs_credentials, &path, false)?; if flags.contains(AtFlags::AT_REMOVEDIR) { - self.files.borrow().fs.rmdir(path).map_err(Errno::from) + self.files + .borrow() + .fs + .rmdir_as(fs_credentials, path.as_str()) + .map_err(Errno::from) + } else { + self.files + .borrow() + .fs + .unlink_as(fs_credentials, path.as_str()) + .map_err(Errno::from) + } + } + + /// Handle syscall `renameat2` (and `renameat`/`rename`, which the dispatcher + /// forwards here with the absent dirfds/flags defaulted to `AT_FDCWD`/0). + /// + /// Only the default (flags 0) and `RENAME_NOREPLACE` behaviours are + /// implemented; `RENAME_EXCHANGE` and `RENAME_WHITEOUT` are rejected with + /// `EINVAL`, which is what a backend that does not support them reports. + /// Neither path's trailing component is dereferenced -- `rename(2)` acts on a + /// symlink itself, never its target -- matching `sys_unlinkat` above. + pub(crate) fn sys_renameat2( + &self, + olddirfd: i32, + oldpath: impl path::Arg, + newdirfd: i32, + newpath: impl path::Arg, + flags: u32, + ) -> Result<(), Errno> { + const RENAME_NOREPLACE: u32 = 1 << 0; + const RENAME_EXCHANGE: u32 = 1 << 1; + const RENAME_WHITEOUT: u32 = 1 << 2; + + // Reject unknown bits outright, and the two behaviours LiteBox does not + // model. + if flags & !(RENAME_NOREPLACE | RENAME_EXCHANGE | RENAME_WHITEOUT) != 0 + || flags & (RENAME_EXCHANGE | RENAME_WHITEOUT) != 0 + { + return Err(Errno::EINVAL); + } + let noreplace = flags & RENAME_NOREPLACE != 0; + + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let oldpath = self.resolve_path_at(olddirfd, oldpath)?; + let oldpath = self.resolve_syscall_path_as(fs_credentials, &oldpath, false)?; + let newpath = self.resolve_path_at(newdirfd, newpath)?; + let newpath = self.resolve_syscall_path_as(fs_credentials, &newpath, false)?; + self.files + .borrow() + .fs + .rename_as(fs_credentials, oldpath.as_str(), newpath.as_str(), noreplace) + .map_err(Errno::from) + } + + /// Handle syscall `fchmodat`. + /// + /// `chmod` has no wrapper of its own here, matching this file's existing convention for the + /// other legacy no-dirfd syscalls that have an `*at` sibling (compare `sys_mkdirat`, which + /// likewise has no separate `sys_mkdir`): `chmod` is reached by the syscall dispatcher + /// constructing this same [`litebox_common_linux::SyscallRequest::Fchmodat`] with `dirfd` + /// forced to `AT_FDCWD`. The raw `fchmodat(2)` syscall (unlike `fchmodat2(2)`) takes no + /// `flags` argument, so callers reached through it always pass `AtFlags::empty()`. + /// + /// A trailing symlink is followed unless `AT_SYMLINK_NOFOLLOW` (`fchmodat2(2)`) says not to, + /// in which case naming the link itself is `EOPNOTSUPP`, as on Linux: symlinks have no mode. + pub fn sys_fchmodat( + &self, + dirfd: i32, + pathname: impl path::Arg, + mode: u32, + flags: AtFlags, + ) -> Result<(), Errno> { + if flags.intersects(AtFlags::AT_SYMLINK_NOFOLLOW.complement()) { + return Err(Errno::EINVAL); + } + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let path = self.resolve_path_at(dirfd, pathname)?; + let follow_final = !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW); + let path = self.resolve_path_symlinks_with_final_as( + fs_credentials, + path.to_str().map_err(|_| Errno::EINVAL)?, + follow_final, + )?; + let files = self.files.borrow(); + if !follow_final + && files + .fs + .file_status_as(fs_credentials, path.as_str())? + .file_type + == litebox::fs::FileType::SymLink + { + return Err(Errno::EOPNOTSUPP); + } + files + .fs + .chmod_as(fs_credentials, path.as_str(), Mode::from_bits_retain(mode)) + .map_err(Errno::from) + } + + fn chown_id(id: u32) -> Result, Errno> { + if id == u32::MAX { + Ok(None) } else { - self.files.borrow().fs.unlink(path).map_err(Errno::from) + u16::try_from(id).map(Some).map_err(|_| Errno::EINVAL) + } + } + + /// Handle syscall `fchownat` (and `chown`/`lchown`, which the dispatcher + /// forwards here with `dirfd` forced to `AT_FDCWD` and `flags` set to + /// `AT_SYMLINK_NOFOLLOW` for `lchown`). + /// + /// `owner`/`group` are the raw `uid_t`/`gid_t`; `(uid_t)-1` (`u32::MAX`) means + /// "leave this id unchanged". Other ids outside LiteBox's `u16` ownership model + /// are rejected rather than silently treated as unchanged. + pub(crate) fn sys_fchownat( + &self, + dirfd: i32, + pathname: impl path::Arg, + owner: u32, + group: u32, + flags: AtFlags, + ) -> Result<(), Errno> { + if flags.intersects(AtFlags::AT_SYMLINK_NOFOLLOW.complement()) { + return Err(Errno::EINVAL); + } + let owner = Self::chown_id(owner)?; + let group = Self::chown_id(group)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let path = self.resolve_path_at(dirfd, pathname)?; + // `chown(2)` dereferences a trailing symlink; `lchown(2)`/`AT_SYMLINK_NOFOLLOW` name + // the link itself. Intermediate links are followed either way. + let path = self.resolve_syscall_path_as( + fs_credentials, + &path, + !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW), + )?; + self.files + .borrow() + .fs + .chown_as(fs_credentials, path.as_str(), owner, group) + .map_err(Errno::from) + } + + /// Handle syscall `fchmod` + pub fn sys_fchmod(&self, fd: i32, mode: u32) -> Result<(), Errno> { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let mode = Mode::from_bits_retain(mode); + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |fd| { + files + .fs + .fd_chmod_as(caller.as_fs_credentials(), fd, mode) + .map_err(Errno::from) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten() + } + + /// Handle syscall `fchown`. + /// + /// `owner`/`group` are the raw `uid_t`/`gid_t`; only `u32::MAX` means + /// "unchanged". Other ids outside LiteBox's `u16` ownership model are rejected. + pub fn sys_fchown(&self, fd: i32, owner: u32, group: u32) -> Result<(), Errno> { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let owner = Self::chown_id(owner)?; + let group = Self::chown_id(group)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |fd| { + files + .fs + .fd_chown_as(caller.as_fs_credentials(), fd, owner, group) + .map_err(Errno::from) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten() + } + + /// Handle syscalls `fsync`, `fdatasync`, and `syncfs`: every LiteBox filesystem is + /// memory-resident, so there is nothing to flush and only the fd is checked -- `EBADF` when + /// closed, `EINVAL` for the descriptor kinds Linux refuses to sync (pipes, sockets, ...). + pub fn sys_fsync(&self, fd: i32) -> Result<(), Errno> { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + self.files + .borrow() + .run_on_raw_fd( + raw_fd, + |_fd| Ok(()), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten() + } + + /// Resolve a single raw `timespec` from `utimensat`/`futimens` into fs-layer semantics: `None` + /// means "leave unchanged" (`UTIME_OMIT`), `Some` carries a concrete timestamp (resolving + /// `UTIME_NOW` against the current wall-clock time). + fn resolve_utime( + &self, + ts: litebox_common_linux::Timespec, + ) -> Result, Errno> { + match ts.tv_nsec { + litebox_common_linux::UTIME_OMIT => Ok(None), + litebox_common_linux::UTIME_NOW => Ok(Some(self.now_as_fs_timestamp())), + nsec if nsec < 1_000_000_000 => Ok(Some(litebox::fs::Timestamp { + sec: ts.tv_sec, + nsec: nsec.reinterpret_as_signed(), + })), + _ => Err(Errno::EINVAL), + } + } + + /// Resolve the raw two-element `times` array from `utimensat`/`futimens` (`None` meaning a + /// `NULL` pointer, i.e., both timestamps set to "now") into `(atime, mtime)`, per + /// [`litebox::fs::FileSystem::utimensat`]'s `None`/`Some` semantics. + fn resolve_utimes( + &self, + times: Option<[litebox_common_linux::Timespec; 2]>, + ) -> Result< + ( + Option, + Option, + ), + Errno, + > { + let Some([atime, mtime]) = times else { + let now = self.now_as_fs_timestamp(); + return Ok((Some(now), Some(now))); + }; + Ok((self.resolve_utime(atime)?, self.resolve_utime(mtime)?)) + } + + fn now_as_fs_timestamp(&self) -> litebox::fs::Timestamp { + let now = self.real_time_as_duration_since_epoch(); + litebox::fs::Timestamp { + sec: now.as_secs().reinterpret_as_signed(), + nsec: i64::from(now.subsec_nanos()), + } + } + + /// Handle syscall `utimensat` + pub fn sys_utimensat( + &self, + dirfd: i32, + pathname: impl path::Arg, + times: Option<[litebox_common_linux::Timespec; 2]>, + flags: AtFlags, + ) -> Result<(), Errno> { + if flags.intersects(AtFlags::AT_SYMLINK_NOFOLLOW.complement()) { + return Err(Errno::EINVAL); + } + let (atime, mtime) = self.resolve_utimes(times)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let path = self.resolve_path_at(dirfd, pathname)?; + let path = self.resolve_syscall_path_as( + fs_credentials, + &path, + !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW), + )?; + self.files + .borrow() + .fs + .utimensat_as(fs_credentials, path.as_str(), atime, mtime) + .map_err(Errno::from) + } + + /// Handle syscall `futimens`. + /// + /// `futimens` has no syscall of its own: glibc implements it as + /// `utimensat(fd, NULL, times, 0)`, which LiteBox's syscall dispatcher routes here (see + /// [`litebox_common_linux::SyscallRequest::Utimensat`]'s doc comment). + pub fn sys_futimens( + &self, + fd: i32, + times: Option<[litebox_common_linux::Timespec; 2]>, + ) -> Result<(), Errno> { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let (atime, mtime) = self.resolve_utimes(times)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |fd| { + files + .fs + .fd_utimensat_as(caller.as_fs_credentials(), fd, atime, mtime) + .map_err(Errno::from) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten() + } + + /// Handle syscall `flock`. + /// + /// See [`litebox::fs::flock::FlockTable`]'s doc comment for exactly what whole-file advisory + /// locking means in LiteBox's single-process-but-multi-threaded model, and for the + /// fd-number-based holder-identity simplification this relies on. + pub fn sys_flock(&self, fd: i32, operation: FlockOperation) -> Result<(), Errno> { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let nonblock = operation.contains(FlockOperation::LOCK_NB); + let kind = match operation - FlockOperation::LOCK_NB { + FlockOperation::LOCK_SH => Some(litebox::fs::flock::FlockKind::Shared), + FlockOperation::LOCK_EX => Some(litebox::fs::flock::FlockKind::Exclusive), + FlockOperation::LOCK_UN => None, + _ => return Err(Errno::EINVAL), + }; + + let files = self.files.borrow(); + let node = files + .run_on_raw_fd( + raw_fd, + |fd| files.fs.fd_file_status(fd).map_err(Errno::from), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten()? + .node_info; + drop(files); + + let holder = litebox::fs::flock::FlockHolder(flock_holder_for_raw_fd(raw_fd)); + let flock_table = self.global.litebox.flock_table(); + match kind { + None => { + flock_table.unlock(node, holder); + Ok(()) + } + Some(kind) if nonblock => flock_table + .try_lock(node, holder, kind) + .map_err(|_| Errno::EWOULDBLOCK), + Some(kind) => flock_table + .lock(&self.wait_cx(), node, holder, kind) + .map_err(|_| Errno::EINTR), } } @@ -375,13 +2468,71 @@ impl Task { }; self.do_read(raw_fd, buf, offset) } + /// Whether a non-blocking `read()` on `fd` must return `EAGAIN` immediately instead of + /// falling through to the (potentially real-stdin-blocking) `Backend::read` path. + /// + /// Only ever true for the fixed stdin fd (0): it's the only raw fd this shim attaches both + /// `StdioStream` and `StdioStatusFlags` metadata to (see + /// `initialize_stdio_in_shared_descriptors_table`), and it's the only one backed by a real, + /// potentially-slow host resource that `Backend::read` has no non-blocking story for on its + /// own -- see `litebox::platform::StdioProvider::stdin_pollable`. + fn stdin_read_would_block(&self, fd: &TypedFd) -> bool { + let dt = self.global.litebox.descriptor_table(); + let Ok(stream) = dt.with_metadata(fd, |s: &StdioStream| *s) else { + return false; + }; + if stream != StdioStream::Stdin { + return false; + } + let Ok(nonblock) = dt.with_metadata(fd, |crate::StdioStatusFlags(flags)| { + flags.contains(OFlags::NONBLOCK) + }) else { + return false; + }; + if !nonblock { + return false; + } + // A platform with no real stdin-readiness signal can't tell us "not ready" for real, so + // fall back to the pre-existing (blocking) behavior rather than spuriously EAGAIN-ing. + self.global + .platform + .stdin_pollable() + .is_some_and(|pollable| !pollable.check_io_events().contains(Events::IN)) + } + pub(crate) fn do_read( &self, fd: u32, buf: &mut [u8], offset: Option, ) -> Result { + // A `read()` from a `NETLINK_ROUTE` socket (iproute2/busybox `ip`) drains + // pending dump bytes; `pread` (an offset) never targets a socket. The + // `&mut buf` is auto-reborrowed, so it stays usable on the non-netlink path. + if offset.is_none() + && let Some(res) = self.netlink_recv(fd, buf) + { + return res; + } let files = self.files.borrow(); + // Inotify fds aren't one of `run_on_raw_fd`'s hand-enumerated subsystem parameters (see + // that function's doc comment); resolve them directly first and fall through to the + // existing dispatch below on a miss, exactly mirroring how a subsystem absent from that + // enumeration would report `EBADF` today -- this changes no existing fd's behavior. + if let Ok(inotify_fd) = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(fd as usize) + { + espipe_for_non_seekable_offset(offset)?; + let handle = self + .global + .litebox + .descriptor_table() + .entry_handle(&inotify_fd) + .ok_or(Errno::EBADF)?; + return handle.with_entry(|file| file.read(&self.wait_cx(), buf)); + } // We need to do this cell dance because otherwise Rust can't recognize that the two // closures are mutually exclusive. let buf: core::cell::RefCell<&mut [u8]> = core::cell::RefCell::new(buf); @@ -389,10 +2540,38 @@ impl Task { .run_on_raw_fd( fd as usize, |fd| { - files + if self.stdin_read_would_block(fd) { + return Err(Errno::EAGAIN); + } + if let Some(meta) = input_event_meta_of(&self.global, fd) { + return self.read_input_events(meta, &mut buf.borrow_mut()); + } + let shared_backing = self.shared_file_backing(fd, false); + let read_offset = match (shared_backing, offset) { + (Some(_), None) => Some( + files + .fs + .seek(fd, 0, SeekWhence::RelativeToCurrentOffset) + .map_err(Errno::from)?, + ), + (_, explicit) => explicit, + }; + let mut output = buf.borrow_mut(); + let size = files .fs - .read(fd, &mut buf.borrow_mut(), offset) - .map_err(Errno::from) + .read(fd, &mut output, offset) + .map_err(Errno::from)?; + if let (Some(backing), Some(read_offset)) = (shared_backing, read_offset) { + self.global + .platform + .read_shared_pages( + backing.identity(), + read_offset, + &mut output[..size], + ) + .map_err(|_| Errno::EIO)?; + } + Ok(size) }, |fd| { espipe_for_non_seekable_offset(offset)?; @@ -436,7 +2615,25 @@ impl Task { .entry_handle(fd) .ok_or(Errno::EBADF)?; espipe_for_non_seekable_offset(offset)?; + // A pty master whose slave side has closed (and not been reopened since) + // reads end-of-file once buffered output is drained, instead of blocking + // forever: this is what a terminal emulator waits for after its shell exits. + let pty_other_closed = self.pty_endpoint(fd).is_some_and(|endpoint| { + endpoint.side == PtySide::Master + && endpoint.state.other_closed.load(Ordering::Acquire) + }); handle.with_entry(|file| { + if pty_other_closed { + return match file.recvfrom( + &self.wait_cx(), + &mut buf.borrow_mut(), + litebox_common_linux::ReceiveFlags::DONTWAIT, + None, + ) { + Err(Errno::EAGAIN) => Ok(0), + other => other, + }; + } file.recvfrom( &self.wait_cx(), &mut buf.borrow_mut(), @@ -445,6 +2642,16 @@ impl Task { ) }) }, + |fd| { + espipe_for_non_seekable_offset(offset)?; + let handle = self + .global + .litebox + .descriptor_table() + .entry_handle(fd) + .ok_or(Errno::EBADF)?; + handle.with_entry(|file| file.handle_recv(&mut buf.borrow_mut())) + }, ) .flatten()?; // For datagrams, the returned size represents the actual size of the message, @@ -458,14 +2665,46 @@ impl Task { /// `offset` is an optional offset to write to. If `None`, it will write to the current file position. /// If `Some`, it will write to the specified offset without changing the current file position. pub fn sys_write(&self, fd: i32, buf: &[u8], offset: Option) -> Result { - let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + let Ok(fd_u32) = u32::try_from(fd) else { return Err(Errno::EBADF); }; + let raw_fd = fd_u32 as usize; + // A `write()` to a `NETLINK_ROUTE` socket (as iproute2/busybox `ip` do, + // rather than `send()`) enqueues a dump; `pwrite` (an offset) never targets + // a socket. + if offset.is_none() + && let Some(res) = self.netlink_send(fd_u32, buf) + { + return res; + } let files = self.files.borrow(); let res = files .run_on_raw_fd( raw_fd, - |fd| files.fs.write(fd, buf, offset).map_err(Errno::from), + |fd| { + let shared_backing = self.shared_file_backing(fd, false); + let size = files.fs.write(fd, buf, offset).map_err(Errno::from)?; + if let Some(backing) = shared_backing { + let write_offset = match offset { + Some(offset) => offset, + None => files + .fs + .seek(fd, 0, SeekWhence::RelativeToCurrentOffset) + .map_err(Errno::from)? + .checked_sub(size) + .ok_or(Errno::EIO)?, + }; + self.global + .platform + .write_shared_pages( + backing.identity(), + write_offset, + &buf[..size], + ) + .map_err(|_| Errno::EIO)?; + } + Ok(size) + }, |fd| { espipe_for_non_seekable_offset(offset)?; self.global.sendto( @@ -500,18 +2739,71 @@ impl Task { file.write(&self.wait_cx(), value) }) }, - |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |fd| { + let handle = self + .global + .litebox + .descriptor_table() + .entry_handle(fd) + .ok_or(Errno::EBADF)?; + espipe_for_non_seekable_offset(offset)?; + let send = |payload: &[u8]| { + handle.with_entry(|file| { + file.sendto( + self, + payload, + litebox_common_linux::SendFlags::empty(), + None, + ) + }) + }; + let Some(endpoint) = self.pty_endpoint(fd) else { + return send(buf); + }; + if endpoint.side != PtySide::Slave { + return send(buf); + } + let output_flags = litebox_common_linux::OFlag::from_bits_truncate( + endpoint.state.termios.lock().c_oflag, + ); + if !output_flags.contains( + litebox_common_linux::OFlag::OPOST | litebox_common_linux::OFlag::ONLCR, + ) { + return send(buf); + } + let newline_count = buf.iter().copied().filter(|byte| *byte == b'\n').count(); + if newline_count == 0 { + return send(buf); + } + let output_len = buf + .len() + .checked_add(newline_count) + .ok_or(Errno::EOVERFLOW)?; + let mut output = alloc::vec::Vec::new(); + output + .try_reserve_exact(output_len) + .map_err(|_| Errno::ENOMEM)?; + for &byte in buf { + if byte == b'\n' { + output.push(b'\r'); + } + output.push(byte); + } + if send(&output)? != output.len() { + return Err(Errno::EIO); + } + Ok(buf.len()) + }, |fd| { + espipe_for_non_seekable_offset(offset)?; let handle = self .global .litebox .descriptor_table() .entry_handle(fd) .ok_or(Errno::EBADF)?; - espipe_for_non_seekable_offset(offset)?; - handle.with_entry(|file| { - file.sendto(self, buf, litebox_common_linux::SendFlags::empty(), None) - }) + Ok(handle.with_entry(|file| file.handle_send(buf))) }, ) .flatten(); @@ -527,12 +2819,6 @@ impl Task { self.sys_read(fd, buf, Some(pos)) } - /// Handle syscall `pwrite64` - pub fn sys_pwrite64(&self, fd: i32, buf: &[u8], offset: i64) -> Result { - let pos = usize::try_from(offset).map_err(|_| Errno::EINVAL)?; - self.sys_write(fd, buf, Some(pos)) - } - fn rewind_sendfile_in_fd(&self, in_raw_fd: usize, unread_n: usize) -> Result<(), Errno> { if unread_n == 0 { return Ok(()); @@ -555,6 +2841,7 @@ impl Task { |_fd| Err(Errno::EINVAL), |_fd| Err(Errno::EINVAL), |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), ) .flatten() } @@ -602,7 +2889,36 @@ impl Task { files .run_on_raw_fd( in_raw_fd, - |fd| files.fs.read(fd, buf_slice, cur_off).map_err(Errno::from), + |fd| { + let shared_backing = self.shared_file_backing(fd, false); + let read_offset = match (shared_backing, cur_off) { + (Some(_), None) => Some( + files + .fs + .seek(fd, 0, SeekWhence::RelativeToCurrentOffset) + .map_err(Errno::from)?, + ), + (_, explicit) => explicit, + }; + let size = files + .fs + .read(fd, buf_slice, cur_off) + .map_err(Errno::from)?; + if let (Some(backing), Some(read_offset)) = + (shared_backing, read_offset) + { + self.global + .platform + .read_shared_pages( + backing.identity(), + read_offset, + &mut buf_slice[..size], + ) + .map_err(|_| Errno::EIO)?; + } + Ok(size) + }, + |_fd| Err(non_fs_err), |_fd| Err(non_fs_err), |_fd| Err(non_fs_err), |_fd| Err(non_fs_err), @@ -684,30 +3000,8 @@ impl Task { files .run_on_raw_fd( raw_fd, - |fd| match files.fs.seek(fd, offset, whence) { - Ok(pos) => Ok(pos), - Err(litebox::fs::errors::SeekError::NotAFile) => { - let base: usize = match whence { - SeekWhence::RelativeToBeginning => 0, - SeekWhence::RelativeToCurrentOffset => self - .global - .litebox - .descriptor_table() - .with_metadata(fd, |off: &Diroff| off.0) - .unwrap_or(0), - SeekWhence::RelativeToEnd => { - return Err(Errno::EINVAL); - } - }; - let new_pos = base.checked_add_signed(offset).ok_or(Errno::EINVAL)?; - self.global - .litebox - .descriptor_table_mut() - .set_fd_metadata(fd, Diroff(new_pos)); - Ok(new_pos) - } - Err(e) => Err(Errno::from(e)), - }, + |fd| files.fs.seek(fd, offset, whence).map_err(Errno::from), + |_| Err(Errno::ESPIPE), |_| Err(Errno::ESPIPE), |_| Err(Errno::ESPIPE), |_| Err(Errno::ESPIPE), @@ -718,11 +3012,13 @@ impl Task { } fn do_mkdir(&self, pathname: impl path::Arg, mode: Mode) -> Result<(), Errno> { + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); let mode = mode & !self.get_umask(); self.files .borrow() .fs - .mkdir(pathname, mode) + .mkdir_as(caller.as_fs_credentials(), pathname, mode) .map_err(Errno::from) } @@ -734,10 +3030,303 @@ impl Task { mode: u32, ) -> Result<(), Errno> { let pathname = self.resolve_path_at(dirfd, pathname)?; - self.do_mkdir(pathname, Mode::from_bits_retain(mode)) + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let pathname = + self.resolve_syscall_path_as(caller.as_fs_credentials(), &pathname, false)?; + self.do_mkdir(pathname.as_str(), Mode::from_bits_retain(mode)) + } + + /// Handle syscall `symlinkat` (and `symlink`, which the dispatcher forwards + /// here with `newdirfd` = `AT_FDCWD`). + /// + /// `target` is the link's contents and is stored verbatim -- it is neither + /// resolved nor required to exist (a dangling link is valid). Only `linkpath` + /// is resolved, against `newdirfd`. + pub(crate) fn sys_symlinkat( + &self, + target: impl path::Arg, + newdirfd: i32, + linkpath: impl path::Arg, + ) -> Result<(), Errno> { + let target = target.as_rust_str().map_err(|_| Errno::EINVAL)?; + // `symlink(2)`: an empty target is ENOENT. + if target.is_empty() { + return Err(Errno::ENOENT); + } + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let linkpath = self.resolve_path_at(newdirfd, linkpath)?; + // The new name is never dereferenced (an existing final link is `EEXIST`), but a + // symlinked parent directory is. + let linkpath = self.resolve_syscall_path_as(fs_credentials, &linkpath, false)?; + self.files + .borrow() + .fs + .symlink_as(fs_credentials, target, linkpath.as_str()) + .map_err(Errno::from) + } + + /// Handle syscalls `link` and `linkat`. + /// + /// DEVIATION, disclosed: the layered filesystem has no inode-sharing hard links, so this + /// creates an exclusive *copy* of the source file at the new path. The dominant real-world + /// caller shape -- write a finished file, `link` it into place as an atomic + /// create-if-absent, `unlink` the original (Xorg's `/tmp/.X0-lock`, mail spools, lock + /// files generally) -- observes identical behavior: `EEXIST` when the name is taken, the + /// full content when it wins. What differs from real `link(2)`: post-link writes through + /// one name are not visible through the other, and `st_nlink`/inode identity stay + /// separate. A guest that round-trips those semantics needs real hard-link support in + /// `litebox::fs` first. + pub(crate) fn sys_linkat( + &self, + olddirfd: i32, + oldpath: impl path::Arg, + newdirfd: i32, + newpath: impl path::Arg, + flags: u32, + ) -> Result<(), Errno> { + const AT_SYMLINK_FOLLOW: u32 = 0x400; + const AT_EMPTY_PATH: u32 = 0x1000; + const COPY_BUFFER_SIZE: usize = 64 * 1024; + const MAX_STAGE_ATTEMPTS: usize = 64; + static NEXT_STAGE_ID: AtomicUsize = AtomicUsize::new(0); + + if flags & AT_EMPTY_PATH != 0 || flags & !(AT_SYMLINK_FOLLOW | AT_EMPTY_PATH) != 0 { + return Err(Errno::EINVAL); + } + + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let oldpath = self.resolve_path_at(olddirfd, oldpath)?; + let oldpath = self.resolve_path_symlinks_with_final_as( + fs_credentials, + oldpath.to_str().map_err(|_| Errno::EINVAL)?, + flags & AT_SYMLINK_FOLLOW != 0, + )?; + let newpath = self.resolve_path_at(newdirfd, newpath)?; + let newpath = self.resolve_path_symlinks_with_final_as( + fs_credentials, + newpath.to_str().map_err(|_| Errno::EINVAL)?, + false, + )?; + let parent_end = newpath.rfind('/').ok_or(Errno::EINVAL)?; + let parent = &newpath[..parent_end]; + let next_stage_path = || { + let id = NEXT_STAGE_ID.fetch_add(1, Ordering::Relaxed); + if parent.is_empty() { + alloc::format!("/.litebox-linkat-{}-{id}", self.pid) + } else { + alloc::format!("{parent}/.litebox-linkat-{}-{id}", self.pid) + } + }; + + let files = self.files.borrow(); + let map_status_error = |error| match error { + litebox::fs::errors::FileStatusError::ClosedFd => Errno::EBADF, + litebox::fs::errors::FileStatusError::Io => Errno::EIO, + litebox::fs::errors::FileStatusError::PathError(error) => error.into(), + _ => Errno::EIO, + }; + let close = |fd: &TypedFd| files.fs.close(fd).map_err(|_| Errno::EIO); + let cleanup = |stage: &str, failure: Errno| { + files + .fs + .unlink_as(fs_credentials, stage) + .map_err(Errno::from)?; + Err(failure) + }; + let publish = |stage: &str| { + match files + .fs + .rename_as(fs_credentials, stage, newpath.as_str(), true) + { + Ok(()) => Ok(()), + Err(error) => cleanup(stage, error.into()), + } + }; + + let source_status = files + .fs + .file_status_as(fs_credentials, oldpath.as_str()) + .map_err(map_status_error)?; + match source_status.file_type { + litebox::fs::FileType::SymLink => { + let target = files + .fs + .readlink_as(fs_credentials, oldpath.as_str()) + .map_err(Errno::from)?; + let stage = 'create: { + for _ in 0..MAX_STAGE_ATTEMPTS { + let stage = next_stage_path(); + if stage == newpath { + continue; + } + match files.fs.symlink_as( + fs_credentials, + target.as_str(), + stage.as_str(), + ) { + Ok(()) => break 'create stage, + Err(litebox::fs::errors::SymlinkError::AlreadyExists) => continue, + Err(error) => return Err(error.into()), + } + } + return Err(Errno::EAGAIN); + }; + publish(stage.as_str()) + } + litebox::fs::FileType::RegularFile => { + let src = files + .fs + .open_as( + fs_credentials, + oldpath.as_str(), + OFlags::RDONLY, + Mode::empty(), + ) + .map_err(Errno::from)?; + let status = match files.fs.fd_file_status(&src) { + Ok(status) => status, + Err(error) => { + let error = map_status_error(error); + close(&src)?; + return Err(error); + } + }; + if status.file_type != litebox::fs::FileType::RegularFile { + close(&src)?; + return Err(Errno::EPERM); + } + + let mut buffer = alloc::vec::Vec::new(); + if buffer.try_reserve_exact(COPY_BUFFER_SIZE).is_err() { + close(&src)?; + return Err(Errno::ENOMEM); + } + buffer.resize(COPY_BUFFER_SIZE, 0); + + let (stage, dst) = 'create: { + for _ in 0..MAX_STAGE_ATTEMPTS { + let stage = next_stage_path(); + if stage == newpath { + continue; + } + match files.fs.open_as( + fs_credentials, + stage.as_str(), + OFlags::WRONLY | OFlags::CREAT | OFlags::EXCL, + Mode::RUSR | Mode::WUSR, + ) { + Ok(dst) => break 'create (stage, dst), + Err(litebox::fs::errors::OpenError::AlreadyExists) => continue, + Err(error) => { + let error = Errno::from(error); + close(&src)?; + return Err(error); + } + } + } + close(&src)?; + return Err(Errno::EAGAIN); + }; + + let copy_result = (|| { + let mut offset = 0usize; + loop { + let read = + files + .fs + .read(&src, &mut buffer, Some(offset)) + .map_err(|error| match error { + litebox::fs::errors::ReadError::ClosedFd => Errno::EBADF, + litebox::fs::errors::ReadError::NotAFile => Errno::EISDIR, + litebox::fs::errors::ReadError::NotForReading => Errno::EBADF, + litebox::fs::errors::ReadError::Io => Errno::EIO, + _ => Errno::EIO, + })?; + if read == 0 { + break; + } + if read > buffer.len() { + return Err(Errno::EIO); + } + let mut written = 0usize; + while written < read { + let write_offset = + offset.checked_add(written).ok_or(Errno::EOVERFLOW)?; + let write = files + .fs + .write(&dst, &buffer[written..read], Some(write_offset)) + .map_err(|error| match error { + litebox::fs::errors::WriteError::ClosedFd => Errno::EBADF, + litebox::fs::errors::WriteError::NotAFile => Errno::EISDIR, + litebox::fs::errors::WriteError::NotForWriting => Errno::EBADF, + litebox::fs::errors::WriteError::ReadOnlyFileSystem => { + Errno::EROFS + } + litebox::fs::errors::WriteError::Io => Errno::EIO, + _ => Errno::EIO, + })?; + if write == 0 || write > read - written { + return Err(Errno::EIO); + } + written += write; + } + offset = offset.checked_add(read).ok_or(Errno::EOVERFLOW)?; + } + files + .fs + .fd_chown_as( + fs_credentials, + &dst, + Some(status.owner.user), + Some(status.owner.group), + ) + .map_err(Errno::from)?; + files + .fs + .fd_chmod_as(fs_credentials, &dst, status.mode) + .map_err(Errno::from)?; + files + .fs + .fd_utimensat_as( + fs_credentials, + &dst, + Some(status.atime), + Some(status.mtime), + ) + .map_err(Errno::from) + })(); + let src_close_result = close(&src); + let dst_close_result = close(&dst); + if let Err(error) = copy_result.and(src_close_result).and(dst_close_result) { + return cleanup(stage.as_str(), error); + } + publish(stage.as_str()) + } + _ => Err(Errno::EPERM), + } } pub(crate) fn do_close(&self, raw_fd: usize) -> Result<(), Errno> { + // See the matching comment in `do_read`: inotify fds sit outside `do_close_and_replace`'s + // `ConsumedFd` enumeration, so resolve them directly first. Inotify fds have no + // replace-into-same-slot use case in this codebase (unlike `do_close_and_replace`'s + // general `S`-typed `replace` parameter, only ever driven by `dup2`-style callers), so a + // direct consume-and-drop is the whole close. + { + let files = self.files.borrow(); + let mut rds = files.raw_descriptor_store.write(); + if rds + .fd_consume_raw_integer::>(raw_fd) + .is_ok() + { + return Ok(()); + } + } self.do_close_and_replace::(raw_fd, None) } @@ -756,6 +3345,7 @@ impl Task { Eventfd(alloc::sync::Arc>>), Epoll(alloc::sync::Arc>>), Unix(alloc::sync::Arc>>), + Netlink(alloc::sync::Arc>>), } let files = self.files.borrow(); @@ -792,6 +3382,10 @@ impl Task { ) { ConsumedFd::Unix(fd) + } else if let Ok(fd) = + rds.fd_consume_raw_integer::>(raw_fd) + { + ConsumedFd::Netlink(fd) } else { unreachable!("all subsystems covered") } @@ -813,6 +3407,16 @@ impl Task { if let Ok(raw_fd) = i32::try_from(raw_fd) { self.finalize_elf_patch(raw_fd); } + // Release any `flock(2)` lock this fd number holds before the fd goes away, so a + // guest that never calls `LOCK_UN` doesn't leak the lock for the rest of this + // LiteBox instance's lifetime. See `sys_flock`'s doc comment for the fd-number-based + // holder-identity simplification this relies on. + if let Ok(node) = files.fs.fd_file_status(&fd) { + self.global.litebox.flock_table().unlock( + node.node_info, + litebox::fs::flock::FlockHolder(flock_holder_for_raw_fd(raw_fd)), + ); + } files.fs.close(&fd).map_err(Errno::from) } ConsumedFd::Network(fd) => self.global.close_socket(&self.wait_cx(), fd), @@ -836,6 +3440,27 @@ impl Task { Ok(()) } ConsumedFd::Unix(fd) => { + let (entry, slave) = { + let mut dt = self.global.litebox.descriptor_table_mut(); + let slave = dt + .with_metadata(&fd, |endpoint: &PtyEndpoint| { + (endpoint.side == PtySide::Slave) + .then(|| (endpoint.number, endpoint.state.clone())) + }) + .ok() + .flatten(); + (dt.remove(&fd), slave) + }; + // do not hold any locks while dropping the entry + drop(entry); + if let Some((number, state)) = slave { + self.global + .pty_registry + .release_slave_descriptor(number, &state); + } + Ok(()) + } + ConsumedFd::Netlink(fd) => { let entry = { let mut dt = self.global.litebox.descriptor_table_mut(); dt.remove(&fd) @@ -897,6 +3522,72 @@ impl Task { }) } + /// Handle syscall `preadv2`. + /// + /// `preadv` with an `RWF_*` flag word: an unknown flag is `EOPNOTSUPP` (Linux's + /// `kiocb_set_rw_flags`), an offset of `-1` means "the current file offset" (`readv` + /// semantics), and the hint flags (`HIPRI`, `DSYNC`, `SYNC`, `NOWAIT`, `DONTCACHE`) are + /// accepted as no-ops -- memory-backed files never block or need syncing. `RWF_ATOMIC` is + /// `EOPNOTSUPP` as on any filesystem without atomic-write support. + pub(crate) fn sys_preadv2( + &self, + fd: i32, + iovec: UserPtr, + iovcnt: usize, + offset: i64, + flags: u32, + ) -> Result { + check_rwf_flags(flags)?; + if offset == -1 { + return self.sys_readv(fd, iovec, iovcnt); + } + self.sys_preadv(fd, iovec, iovcnt, offset) + } + + /// Handle syscall `pwritev2`; see [`Self::sys_preadv2`] for the flag rules. + /// + /// `RWF_APPEND` writes at end-of-file whatever `offset` says; `RWF_NOAPPEND` (what + /// `base::File::Write` passes so an `O_APPEND` file still honours the requested offset) + /// is the behaviour positional writes here already have, so it needs nothing extra. + pub(crate) fn sys_pwritev2( + &self, + fd: i32, + iovec: UserPtr, + iovcnt: usize, + offset: i64, + flags: u32, + ) -> Result { + check_rwf_flags(flags)?; + if flags & RWF_APPEND != 0 && flags & RWF_NOAPPEND != 0 { + return Err(Errno::EINVAL); + } + if offset == -1 { + return self.sys_writev(fd, iovec, iovcnt); + } + if flags & RWF_APPEND != 0 { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return Err(Errno::EBADF); + }; + let files = self.files.borrow(); + let size = files + .run_on_raw_fd( + raw_fd, + |fd| files.fs.fd_file_status(fd).map(|s| s.size).map_err(Errno::from), + |_fd| Ok(0), + |_fd| Ok(0), + |_fd| Ok(0), + |_fd| Ok(0), + |_fd| Ok(0), + |_fd| Ok(0), + ) + .flatten()?; + drop(files); + let at_end = i64::try_from(size).map_err(|_| Errno::EOVERFLOW)?; + return self.sys_pwritev(fd, iovec, iovcnt, at_end); + } + self.sys_pwritev(fd, iovec, iovcnt, offset) + } + /// Handle syscall `readv` pub(crate) fn sys_readv( &self, @@ -941,6 +3632,36 @@ impl Task { const IOV_MAX: usize = 1024; const SSIZE_MAX: usize = isize::MAX as usize; +/// `RWF_*` flags `preadv2`/`pwritev2` accept (`include/uapi/linux/fs.h`). +const RWF_HIPRI: u32 = 0x01; +const RWF_DSYNC: u32 = 0x02; +const RWF_SYNC: u32 = 0x04; +const RWF_NOWAIT: u32 = 0x08; +const RWF_APPEND: u32 = 0x10; +const RWF_NOAPPEND: u32 = 0x20; +const RWF_ATOMIC: u32 = 0x40; +const RWF_DONTCACHE: u32 = 0x80; +const RWF_SUPPORTED: u32 = RWF_HIPRI + | RWF_DSYNC + | RWF_SYNC + | RWF_NOWAIT + | RWF_APPEND + | RWF_NOAPPEND + | RWF_ATOMIC + | RWF_DONTCACHE; + +/// Linux's `kiocb_set_rw_flags` acceptance rules for a `preadv2`/`pwritev2` flag word. +fn check_rwf_flags(flags: u32) -> Result<(), Errno> { + if flags & !RWF_SUPPORTED != 0 { + return Err(Errno::EOPNOTSUPP); + } + if flags & RWF_ATOMIC != 0 { + // No filesystem here advertises `FMODE_CAN_ATOMIC_WRITE`. + return Err(Errno::EOPNOTSUPP); + } + Ok(()) +} + fn check_iovcnt(iovcnt: usize) -> Result<(), Errno> { if iovcnt > IOV_MAX { Err(Errno::EINVAL) @@ -1084,10 +3805,12 @@ impl Task { Ok(()) } - fn do_access_mode( + fn do_access_mode_values( mode: Mode, - owner: AccessUserInfo, - caller: AccessUserInfo, + owner_user: u32, + owner_group: u32, + is_directory: bool, + caller: AccessUserInfo<'_>, access_mode: &AccessFlags, ) -> Result<(), Errno> { if access_mode.is_empty() { @@ -1095,17 +3818,18 @@ impl Task { } if caller.user == 0 { if access_mode.contains(AccessFlags::X_OK) + && !is_directory && !mode.intersects(Mode::XUSR | Mode::XGRP | Mode::XOTH) { return Err(Errno::EACCES); } return Ok(()); } - // TODO: Linux also uses group bits when `owner.group` is in the caller's supplementary - // group list. `AccessUserInfo` only carries the real/effective primary group today. - let (read, write, execute) = if caller.user == owner.user { + let (read, write, execute) = if caller.user == owner_user { (Mode::RUSR, Mode::WUSR, Mode::XUSR) - } else if caller.group == owner.group { + } else if caller.group == owner_group + || caller.supplementary_groups.contains(&owner_group) + { (Mode::RGRP, Mode::WGRP, Mode::XGRP) } else { (Mode::ROTH, Mode::WOTH, Mode::XOTH) @@ -1122,29 +3846,42 @@ impl Task { Ok(()) } - fn access_user(&self, flags: &AtFlags) -> AccessUserInfo { - if flags.contains(AtFlags::AT_EACCESS) { - AccessUserInfo { - user: self.credentials.euid, - group: self.credentials.egid, - } - } else { - AccessUserInfo { - user: self.credentials.uid, - group: self.credentials.gid, - } - } + fn do_access_mode( + status: &litebox::fs::FileStatus, + caller: AccessUserInfo<'_>, + access_mode: &AccessFlags, + ) -> Result<(), Errno> { + Self::do_access_mode_values( + status.mode, + u32::from(status.owner.user), + u32::from(status.owner.group), + status.file_type == litebox::fs::FileType::Directory, + caller, + access_mode, + ) } fn do_access( &self, pathname: impl path::Arg, mode: AccessFlags, - caller: AccessUserInfo, + caller: AccessUserInfo<'_>, + follow_final: bool, ) -> Result<(), Errno> { - let status = self.files.borrow().fs.file_status(pathname)?; - let owner = status.owner.into(); - Self::do_access_mode(status.mode, owner, caller, &mode) + // `access(2)` and ordinary `faccessat(2)` dereference a trailing symlink, so a dangling + // link reports as absent and the target's permissions are checked. `faccessat2` with + // `AT_SYMLINK_NOFOLLOW` instead checks the link node itself. + let resolved = self.resolve_path_symlinks_with_final_as( + caller.as_fs_credentials(), + pathname.as_rust_str().map_err(|_| Errno::EINVAL)?, + follow_final, + )?; + let status = self + .files + .borrow() + .fs + .file_status_as(caller.as_fs_credentials(), resolved.as_str())?; + Self::do_access_mode(&status, caller, &mode) } /// Handle syscall `faccessat` @@ -1157,39 +3894,41 @@ impl Task { ) -> Result<(), Errno> { let supported_flags = AtFlags::AT_EACCESS | AtFlags::AT_SYMLINK_NOFOLLOW | AtFlags::AT_EMPTY_PATH; - // TODO: `AT_SYMLINK_NOFOLLOW` is accepted for Linux compatibility, but LiteBox file - // status lookups do not currently follow symlinks in any backend. if flags.intersects(supported_flags.complement()) { return Err(Errno::EINVAL); } Self::validate_access_mode(&mode)?; - let caller = self.access_user(&flags); + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot( + &credentials, + flags.contains(AtFlags::AT_EACCESS), + ); let get_cwd = || self.fs.borrow().cwd.read().clone(); - let fs_path = FsPath::new(dirfd, pathname, get_cwd)?; + let fs_path = self.fs_path(dirfd, pathname)?; + let follow_final = !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW); match fs_path { - FsPath::Absolute { path } => self.do_access(path, mode, caller), + FsPath::Absolute { path } => self.do_access(path, mode, caller, follow_final), FsPath::Cwd if flags.contains(AtFlags::AT_EMPTY_PATH) => { let cwd = get_cwd(); - self.do_access(cwd, mode, caller) + self.do_access(cwd, mode, caller, follow_final) } FsPath::Fd(fd) if flags.contains(AtFlags::AT_EMPTY_PATH) => { let stat: FileStat = descriptor_stat(fd as usize, self)?; - let owner = AccessUserInfo { - user: stat.st_uid, - group: stat.st_gid, - }; - Self::do_access_mode( + Self::do_access_mode_values( Mode::from_bits_truncate(stat.st_mode & 0o7777), - owner, + stat.st_uid, + stat.st_gid, + stat.st_mode & 0o170000 == InodeType::Dir as u32, caller, &mode, ) } FsPath::Cwd | FsPath::Fd(_) => Err(Errno::ENOENT), - FsPath::FdRelative { .. } => { - log_unsupported!("fd-relative faccessat is not supported yet"); - Err(Errno::EINVAL) + FsPath::FdRelative { fd, path } => { + let dir_path = self.resolve_dirfd_path(fd)?; + let joined = self.join_dir_relative_path(&dir_path, &path)?; + self.do_access(joined, mode, caller, follow_final) } } } @@ -1197,22 +3936,30 @@ impl Task { /// Read the target of a symbolic link /// /// The caller must pass an absolute path. - /// - /// Note that this function only handles the following cases that we hardcoded: - /// - `/proc/self/fd/` fn do_readlink(&self, fullpath: &str) -> Result { - if let Some(stripped) = fullpath.strip_prefix("/proc/self/fd/") { - let fd = stripped.parse::().map_err(|_| Errno::EINVAL)?; - match fd { - 0 => return Ok("/dev/stdin".to_string()), - 1 => return Ok("/dev/stdout".to_string()), - 2 => return Ok("/dev/stderr".to_string()), - _ => unimplemented!(), - } - } + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + self.do_readlink_as(caller.as_fs_credentials(), fullpath) + } - // TODO: we do not support symbolic links other than stdio yet. - Err(Errno::ENOENT) + fn do_readlink_as( + &self, + credentials: AccessCredentials<'_>, + fullpath: &str, + ) -> Result { + // Follow every intermediate link (`/proc/self/...` included -- `self` is itself a + // link) but never the final one, which is the object being read. + let fullpath = self.resolve_path_symlinks_with_final_as(credentials, fullpath, false)?; + self.publish_proc_view(&fullpath); + + // A symbolic link in the filesystem (`/proc//fd/` included): return its target + // verbatim. `readlink(2)` is EINVAL on a non-symlink and ENOENT on a missing path, + // which is exactly how `FileSystem::readlink` maps. + self.files + .borrow() + .fs + .readlink_as(credentials, fullpath.as_str()) + .map_err(Errno::from) } /// Handle syscall `readlink` @@ -1228,7 +3975,17 @@ impl Task { buf: &mut [u8], ) -> Result { let pathname = self.resolve_path_at(dirfd, pathname)?; - let path = self.do_readlink(pathname.to_str().map_err(|_| Errno::EINVAL)?)?; + let path = self.do_readlink(pathname.to_str().map_err(|_| Errno::EINVAL)?); + // Same rationale as `sys_openat`'s trace line: the raw request only carries a user + // pointer, and readlink targets are load-bearing for sysfs-probing guests. + litebox_util_log::trace!( + pid:? = self.pid, + tid:? = self.tid, + path:? = pathname, + result:? = path; + "readlinkat" + ); + let path = path?; let bytes = path.as_bytes(); let min_len = core::cmp::min(buf.len(), bytes.len()); buf[..min_len].copy_from_slice(&bytes[..min_len]); @@ -1236,12 +3993,91 @@ impl Task { } } +/// Block size used for the synthetic `statfs` figures below, as both the `i64` the ABI struct's +/// fields need and the `u64` the byte-count constants below need to divide by. +const SYNTHETIC_DISK_BLOCK_SIZE: i64 = 4096; +const SYNTHETIC_DISK_BLOCK_SIZE_U64: u64 = 4096; +/// Total synthetic "disk" space, matching the scale of `sys_sysinfo`'s synthetic RAM figures +/// (`litebox::fs::proc::SYNTHETIC_TOTAL_RAM_BYTES`) rather than anything measured -- LiteBox does +/// not model real per-mount disk usage. Kept a distinct constant since disk and RAM are unrelated +/// figures on any real system. +const SYNTHETIC_DISK_TOTAL_BYTES: u64 = 8 * 1024 * 1024 * 1024; +/// Free synthetic "disk" space; half of [`SYNTHETIC_DISK_TOTAL_BYTES`], the same +/// total/free ratio `sys_sysinfo`'s synthetic RAM figures use. +const SYNTHETIC_DISK_FREE_BYTES: u64 = SYNTHETIC_DISK_TOTAL_BYTES / 2; +/// `TMPFS_MAGIC` from ``: the closest real filesystem-type magic to LiteBox's own +/// synthetic, in-memory-backed filesystem. +const SYNTHETIC_STATFS_MAGIC: i64 = 0x0102_1994; + +/// The same synthetic `statfs` figures for every path/fd -- see `sys_statfs`/`sys_fstatfs`. +fn synthetic_statfs() -> litebox_common_linux::Statfs { + litebox_common_linux::Statfs { + f_type: SYNTHETIC_STATFS_MAGIC, + f_bsize: SYNTHETIC_DISK_BLOCK_SIZE, + f_blocks: SYNTHETIC_DISK_TOTAL_BYTES / SYNTHETIC_DISK_BLOCK_SIZE_U64, + f_bfree: SYNTHETIC_DISK_FREE_BYTES / SYNTHETIC_DISK_BLOCK_SIZE_U64, + f_bavail: SYNTHETIC_DISK_FREE_BYTES / SYNTHETIC_DISK_BLOCK_SIZE_U64, + f_files: 0, + f_ffree: 0, + f_fsid: [0, 0], + f_namelen: 255, + f_frsize: SYNTHETIC_DISK_BLOCK_SIZE, + f_flags: 0, + f_spare: [0; 4], + } +} + +/// `st_nlink`/`stx_nlink` of a `stat` result, so a link count the generic `From` +/// conversion cannot know (it reports 1) can be filled in afterwards -- see +/// [`Task::proc_dir_link_count`]. +trait StatLinkCount { + fn set_link_count(&mut self, count: u64); +} + +impl StatLinkCount for FileStat { + fn set_link_count(&mut self, count: u64) { + #[cfg(target_arch = "x86_64")] + { + self.st_nlink = count; + } + #[cfg(target_arch = "aarch64")] + { + self.st_nlink = count.trunc(); + } + } +} + +impl StatLinkCount for Statx { + fn set_link_count(&mut self, count: u64) { + self.stx_nlink = count.trunc(); + } +} + +/// Convert a [`litebox::fs::FileStatus`] into a `stat` result, filling in the link count of a +/// `/proc` directory (see [`litebox::fs::proc::Proc::dir_link_count_at`]) -- `path` is the +/// real absolute path `status` describes, when known. +fn stat_from_status( + task: &Task, + status: litebox::fs::FileStatus, + path: Option<&str>, +) -> T +where + T: From + StatLinkCount, +{ + let link_count = path.and_then(|path| task.proc_dir_link_count(&status, path)); + let mut stat = T::from(status); + if let Some(link_count) = link_count { + stat.set_link_count(link_count); + } + stat +} + fn descriptor_stat( raw_fd: usize, task: &Task, ) -> Result where - T: From + From, + T: From + From + StatLinkCount, { // TODO: give correct values for the synthesized branches. let synthetic = |mode_bits: u32, blksize: usize| FileStat { @@ -1253,7 +4089,13 @@ where st_gid: 0, st_rdev: 0, st_size: 0, + // The x86-64 `struct stat` declares `st_blksize` as a signed word the + // width of a pointer; the generic layout aarch64 uses declares it as a + // plain `int`. Both are wide enough for any block size LiteBox reports. + #[cfg(target_arch = "x86_64")] st_blksize: blksize, + #[cfg(target_arch = "aarch64")] + st_blksize: blksize.reinterpret_as_signed().trunc(), st_blocks: 0, ..Default::default() }; @@ -1265,10 +4107,18 @@ where .run_on_raw_fd( raw_fd, |fd| { + let path = task + .global + .litebox + .descriptor_table() + .with_metadata(fd, |path: &FdPath| path.0.clone()) + .ok(); files .fs .fd_file_status(fd) - .map(T::from) + .map(|status| { + stat_from_status(task, status, path.as_ref().and_then(|p| p.to_str().ok())) + }) .map_err(Errno::from) }, |_fd| Ok(T::from(synthetic(socket_mode, 4096))), @@ -1281,6 +4131,7 @@ where |_fd| Ok(T::from(synthetic(rw_user_mode, 4096))), |_fd| Ok(T::from(synthetic(rw_user_mode, 0))), |_fd| Ok(T::from(synthetic(socket_mode, 4096))), + |_fd| Ok(T::from(synthetic(socket_mode, 4096))), ) .flatten() } @@ -1302,6 +4153,15 @@ pub(crate) fn get_file_descriptor_flags( .with_metadata(fd, |flags: &FileDescriptorFlags| *flags) .unwrap_or(FileDescriptorFlags::empty()) } + // See the matching pre-check in `fork_copy`/`do_read`/`do_close`: inotify fds sit outside + // `run_on_raw_fd`'s hand-enumerated subsystem list, so resolve them directly first. + if let Ok(inotify_fd) = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(raw_fd) + { + return Ok(get_flags(global, &inotify_fd)); + } files.run_on_raw_fd( raw_fd, |fd| get_flags(global, fd), @@ -1310,6 +4170,7 @@ pub(crate) fn get_file_descriptor_flags( |fd| get_flags(global, fd), |fd| get_flags(global, fd), |fd| get_flags(global, fd), + |fd| get_flags(global, fd), ) } @@ -1330,6 +4191,15 @@ fn set_file_descriptor_flags( .set_fd_metadata(fd, flags); } + // See the matching pre-check in `get_file_descriptor_flags` just above. + if let Ok(inotify_fd) = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(raw_fd) + { + set_flags(global, &inotify_fd, flags); + return Ok(()); + } files.run_on_raw_fd( raw_fd, |fd| set_flags(global, fd, flags), @@ -1338,6 +4208,7 @@ fn set_file_descriptor_flags( |fd| set_flags(global, fd, flags), |fd| set_flags(global, fd, flags), |fd| set_flags(global, fd, flags), + |fd| set_flags(global, fd, flags), )?; Ok(()) } @@ -1346,20 +4217,88 @@ impl Task { /// Get the file status of `pathname`. /// /// The `pathname` must be absolute. - fn do_stat>( + /// The `st_nlink` a directory of the mounted `/proc` reports: Linux's `2 + `, + /// which for `/proc//task` counts the live threads (Chromium's + /// `ThreadHelpers::IsSingleThreaded` reads exactly this). `None` for anything else, which + /// keeps the generic conversion's answer. `path` is the real absolute path `status` + /// describes; `/proc` is mounted at the real `/proc`, so beneath a `chroot` nothing matches. + fn proc_dir_link_count(&self, status: &litebox::fs::FileStatus, path: &str) -> Option { + if status.file_type != litebox::fs::FileType::Directory { + return None; + } + let under_proc = path.strip_prefix("/proc")?; + if !(under_proc.is_empty() || under_proc.starts_with('/')) { + return None; + } + let components: alloc::vec::Vec<&str> = under_proc + .split('/') + .filter(|component| !component.is_empty() && *component != ".") + .collect(); + self.global + .proc_handle + .as_ref()? + .dir_link_count_at(&components) + } + + fn do_stat + From + StatLinkCount>( &self, pathname: impl path::Arg, follow_symlink: bool, ) -> Result { - let normalized_path = pathname.normalized()?; - let path = if follow_symlink { - self.do_readlink(normalized_path.as_str()) - .unwrap_or(normalized_path) - } else { - normalized_path - }; - let status = self.files.borrow().fs.file_status(path)?; - Ok(T::from(status)) + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + self.do_stat_as(caller.as_fs_credentials(), pathname, follow_symlink) + } + + /// If `path` is `/proc/self/fd/`, `/proc/thread-self/fd/` or `/proc//fd/` + /// for an open descriptor `n` of this task, that descriptor. Linux resolves such a magic + /// link to the open file itself, whatever its `readlink` text says. + fn proc_fd_magic_link(&self, path: &str) -> Option { + let rest = path.strip_prefix("/proc/")?; + let (who, rest) = rest.split_once('/')?; + let mine = who == "self" + || who == "thread-self" + || who.parse::().ok() == Some(self.pid); + if !mine { + return None; + } + let n = rest.strip_prefix("fd/")?; + if n.is_empty() || n.contains('/') || !n.bytes().all(|b| b.is_ascii_digit()) { + return None; + } + let fd = n.parse::().ok()?; + self.check_raw_fd_exists(fd).ok()?; + Some(fd) + } + + fn do_stat_as + From + StatLinkCount>( + &self, + credentials: AccessCredentials<'_>, + pathname: impl path::Arg, + follow_symlink: bool, + ) -> Result { + // `stat` follows a trailing symlink, `lstat` reports the link itself; both follow every + // intermediate link, so the same component walk `open` uses runs for either -- on the + // raw path, never a lexically normalized one: collapsing `..` ahead of link expansion + // turned `/tmp/bindir/../etc/hosts` (bindir -> /bin) into `/tmp/etc/hosts`, and skipping + // the walk for `lstat` made `ls -l /var/run/x` (run -> ../run) answer `ENOTDIR`. + let path = self.resolve_path_symlinks_with_final_as( + credentials, + pathname.as_rust_str().map_err(|_| Errno::EINVAL)?, + follow_symlink, + )?; + if follow_symlink && let Some(fd) = self.proc_fd_magic_link(&path) { + // `stat("/proc/self/fd/N")` is `fstat(N)` on Linux (the link resolves to the open + // file, not to its path), which is what lets a sandbox's "no open directories" + // sweep `fstatat` every descriptor, sockets and unlinked files included. + return descriptor_stat(fd.cast_unsigned() as usize, self); + } + let status = self + .files + .borrow() + .fs + .file_status_as(credentials, path.as_str())?; + Ok(stat_from_status(self, status, Some(path.as_str()))) } /// Handle syscall `stat` @@ -1393,24 +4332,37 @@ impl Task { flags: AtFlags, ) -> Result where - T: From + From, + T: From + From + StatLinkCount, { + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); let get_cwd = || self.fs.borrow().cwd.read().clone(); - let fs_path = FsPath::new(dirfd, pathname, get_cwd)?; + let fs_path = self.fs_path(dirfd, pathname)?; match fs_path { - FsPath::Absolute { path } => { - self.do_stat(path, !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW)) - } - FsPath::Cwd if flags.contains(AtFlags::AT_EMPTY_PATH) => { - Ok(T::from(self.files.borrow().fs.file_status(get_cwd())?)) - } + FsPath::Absolute { path } => self.do_stat_as( + fs_credentials, + path, + !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW), + ), + FsPath::Cwd if flags.contains(AtFlags::AT_EMPTY_PATH) => Ok(T::from( + self.files + .borrow() + .fs + .file_status_as(fs_credentials, get_cwd())?, + )), FsPath::Fd(fd) if flags.contains(AtFlags::AT_EMPTY_PATH) => { descriptor_stat(fd as usize, self) } FsPath::Cwd | FsPath::Fd(_) => Err(Errno::ENOENT), - FsPath::FdRelative { .. } => { - log_unsupported!("relative fstatat with AT_EMPTY_PATH unset is not supported yet"); - Err(Errno::EINVAL) + FsPath::FdRelative { fd, path } => { + let dir_path = self.resolve_dirfd_path(fd)?; + let joined = self.join_dir_relative_path(&dir_path, &path)?; + self.do_stat_as( + fs_credentials, + joined, + !flags.contains(AtFlags::AT_SYMLINK_NOFOLLOW), + ) } } } @@ -1422,7 +4374,15 @@ impl Task { pathname: impl path::Arg, flags: AtFlags, ) -> Result { - let current_support_flags = AtFlags::AT_EMPTY_PATH; + // `AT_SYMLINK_NOFOLLOW` is the flag `ls` and every other directory + // walker passes, and `do_fstatat` already acts on it -- it selects + // whether `do_stat` resolves the final component. Rejecting it here + // while `statx` and `faccessat` both accept it made `lstat` fail with + // `EINVAL` on a path that `stat` handled. `AT_NO_AUTOMOUNT` is accepted + // as the same no-op `statx` treats it as, since no LiteBox filesystem + // automounts. + let current_support_flags = + AtFlags::AT_EMPTY_PATH | AtFlags::AT_SYMLINK_NOFOLLOW | AtFlags::AT_NO_AUTOMOUNT; if flags.intersects(current_support_flags.complement()) { log_unsupported!("unsupported flags: {flags:?}"); return Err(Errno::EINVAL); @@ -1465,6 +4425,45 @@ impl Task { self.do_fstatat(dirfd, pathname, flags) } + /// Handle syscall `statfs`. + /// + /// LiteBox does not model per-mount free/total space, so every path (on any mount) reports + /// the same synthetic figures -- just enough for `df`'s `statvfs` call (via + /// `/proc/mounts`-enumerated mount points, see `litebox::fs::proc`) to succeed rather than + /// fail outright. The path only needs to resolve to *something*; real Linux behaves the same + /// way for any path on the same filesystem. + pub(crate) fn sys_statfs( + &self, + pathname: impl path::Arg, + buf: UserPtrMut, + ) -> Result<(), Errno> { + use litebox::path::Arg as _; + + let resolved = self.resolve_path(pathname)?; + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + let abs_path = self.resolve_syscall_path_as(fs_credentials, &resolved, true)?; + self.files + .borrow() + .fs + .file_status_as(fs_credentials, abs_path.as_str()) + .map_err(Errno::from)?; + buf.write_at_offset::(0, synthetic_statfs()) + .ok_or(Errno::EFAULT) + } + + /// Handle syscall `fstatfs`. See [`Self::sys_statfs`] on why every result is the same. + pub(crate) fn sys_fstatfs( + &self, + fd: i32, + buf: UserPtrMut, + ) -> Result<(), Errno> { + self.sys_fstat(fd)?; + buf.write_at_offset::(0, synthetic_statfs()) + .ok_or(Errno::EFAULT) + } + pub(crate) fn sys_fcntl(&self, fd: i32, arg: FcntlArg) -> Result { let Ok(desc) = u32::try_from(fd).and_then(usize::try_from) else { return Err(Errno::EBADF); @@ -1504,12 +4503,35 @@ impl Task { Ok(files .run_on_raw_fd( desc, - |fd| getfl_from_metadata!(fd, crate::StdioStatusFlags), + |fd| { + // Stdio fds carry `StdioStatusFlags`; every fd `sys_openat` opened + // carries `FdOpenFlags` (its access mode plus the status flags it was + // opened with, kept current by `SETFL` below). Linux answers + // `F_GETFL` from exactly that open-file-description state -- an + // `O_RDWR` file must report `O_RDWR`, which Chromium's shared-memory + // `CheckFDAccessMode` `CHECK`s -- so answer from the record and only + // fall back to "no flags" for an fd that has neither. + let dt = self.global.litebox.descriptor_table(); + match dt + .with_metadata(fd, |crate::StdioStatusFlags(flags)| { + *flags & OFlags::STATUS_FLAGS_MASK + }) + .or_else(|_| { + dt.with_metadata(fd, |FdOpenFlags(flags)| { + *flags & OFlags::STATUS_FLAGS_MASK + }) + }) { + Ok(flags) => Ok(flags), + Err(MetadataError::ClosedFd) => Err(Errno::EBADF), + Err(MetadataError::NoSuchMetadata) => Ok(OFlags::empty()), + } + }, |fd| getfl_from_metadata!(fd, crate::syscalls::net::SocketOFlags), |fd| self.global.linux_pipe_status_flags(fd), |fd| getfl_from_handle!(fd), |fd| getfl_from_handle!(fd), |fd| getfl_from_handle!(fd), + |fd| getfl_from_metadata!(fd, crate::syscalls::net::SocketOFlags), ) .flatten()? .bits()) @@ -1566,11 +4588,39 @@ impl Task { files.run_on_raw_fd( desc, |fd| { - setfl_in_metadata!( - fd, - crate::StdioStatusFlags, - unimplemented!("SETFL on non-stdio") - ) + // Only stdio raw fds carry `StdioStatusFlags` metadata (see + // `initialize_stdio_in_shared_descriptors_table`); other regular files + // have no status-flags story at the `Backend` layer (mirroring GETFL's + // fallback to `OFlags::empty()` above), so `NoSuchMetadata` here is + // expected and silently ignored, matching this ioctl's precedent for + // FIONBIO and real Linux's no-op SETFL on regular files. + let mut dt = self.global.litebox.descriptor_table_mut(); + match dt.with_metadata_mut(fd, |crate::StdioStatusFlags(f)| { + let diff = (*f & setfl_mask) ^ flags; + if diff.intersects(OFlags::APPEND | OFlags::DIRECT | OFlags::NOATIME) { + log_unsupported!("unsupported flags"); + } + f.toggle(diff); + }) { + Ok(()) => Ok(()), + Err(MetadataError::ClosedFd) => Err(Errno::EBADF), + // Not a stdio fd: keep the open-time record (`FdOpenFlags`, what + // `F_GETFL` and `/proc//fdinfo` answer from) in step with the + // request, so a later `F_GETFL` reports the `O_NONBLOCK`/`O_APPEND` + // the guest just set, as Linux does. `O_APPEND` toggled here is + // recorded but not yet honoured by the write path (see + // `sys_pwritev`'s note); that is the same no-op SETFL a regular file + // already got, now visible instead of silently dropped. + Err(MetadataError::NoSuchMetadata) => { + match dt.with_metadata_mut(fd, |FdOpenFlags(f)| { + let diff = (*f & setfl_mask) ^ flags; + f.toggle(diff); + }) { + Ok(()) | Err(MetadataError::NoSuchMetadata) => Ok(()), + Err(MetadataError::ClosedFd) => Err(Errno::EBADF), + } + } + } }, |fd| { setfl_in_metadata!( @@ -1592,9 +4642,59 @@ impl Task { toggle_flags!(fd); Ok(()) }, + |fd| { + setfl_in_metadata!( + fd, + crate::syscalls::net::SocketOFlags, + unreachable!("all netlink sockets have SocketOFlags when created") + ) + }, )??; Ok(0) } + FcntlArg::GET_SEALS => files + .run_on_raw_fd( + desc, + |fd| { + self.global + .litebox + .descriptor_table() + .with_metadata(fd, |memfd: &MemfdBacking| memfd.seals()) + .map_err(|error| match error { + MetadataError::ClosedFd => Errno::EBADF, + MetadataError::NoSuchMetadata => Errno::EINVAL, + }) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten(), + FcntlArg::ADD_SEALS(seals) => files + .run_on_raw_fd( + desc, + |fd| { + self.global + .litebox + .descriptor_table() + .with_metadata(fd, |memfd: &MemfdBacking| memfd.add_seals(seals)) + .map_err(|error| match error { + MetadataError::ClosedFd => Errno::EBADF, + MetadataError::NoSuchMetadata => Errno::EINVAL, + })? + .map(|()| 0) + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + ) + .flatten(), FcntlArg::GETLK(lock) => { self.files .borrow() @@ -1616,8 +4716,9 @@ impl Task { .ok_or(Errno::EFAULT)?; Ok(0) }, - |_fd| todo!("net"), - |_fd| todo!("pipes"), + |_fd| Err(Errno::EBADF), + |_fd| Err(Errno::EBADF), + |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), @@ -1638,8 +4739,9 @@ impl Task { // can always acquire the lock it owns, so we don't need to maintain anything. Ok(0) }, - |_fd| todo!("net"), - |_fd| todo!("pipes"), + |_fd| Err(Errno::EBADF), + |_fd| Err(Errno::EBADF), + |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), |_fd| Err(Errno::EBADF), @@ -1670,7 +4772,7 @@ impl Task { /// Handle syscall `getcwd` pub fn sys_getcwd(&self, buf: &mut [u8]) -> Result { - let cwd = self.fs.borrow().cwd.read().clone(); + let cwd = Self::guest_visible_path(&self.fs_root(), &self.fs.borrow().cwd.read()); // need to account for the null terminator if cwd.len() >= buf.len() { return Err(Errno::ERANGE); @@ -1690,16 +4792,28 @@ impl Task { use litebox::fs::errors::{FileStatusError, PathError}; use litebox::path::Arg as _; - // Resolve relative paths against CWD, then normalize (handle `.` / `..`). + let credentials = self.credentials_snapshot(); + let caller = Self::access_user_from_snapshot(&credentials, true); + let fs_credentials = caller.as_fs_credentials(); + // Resolve relative paths against CWD; `.`/`..` are applied by the component walk + // below, after any intermediate link they follow has been expanded. let resolved = self.resolve_path(pathname)?; - let abs_path = resolved.normalized().map_err(|_| Errno::EINVAL)?; + // `chdir(2)` dereferences a trailing symlink, so `cd` into a symlinked + // directory works (and lands the cwd on the link's target). + let abs_path = self.resolve_syscall_path_as(fs_credentials, &resolved, true)?; - // Verify the path exists and is a directory. - match self.files.borrow().fs.file_status(abs_path.as_str()) { + // Verify the path exists, is a directory, and is searchable by the caller. + match self + .files + .borrow() + .fs + .file_status_as(fs_credentials, abs_path.as_str()) + { Ok(status) => { if status.file_type != FileType::Directory { return Err(Errno::ENOTDIR); } + Self::do_access_mode(&status, caller, &AccessFlags::X_OK)?; } Err(FileStatusError::PathError(PathError::NoSuchFileOrDirectory)) => { return Err(Errno::ENOENT); @@ -1753,7 +4867,7 @@ impl Task { return Err(Errno::EINVAL); } - let eventfd = super::eventfd::EventFile::new(u64::from(initval), flags); + let eventfd = self.global.create_linux_eventfd(initval, flags)?; let mut dt = self.global.litebox.descriptor_table_mut(); let typed = dt.insert::>(eventfd); if flags.contains(EfdFlags::CLOEXEC) { @@ -1773,31 +4887,144 @@ impl Task { Ok(raw_fd.try_into().unwrap()) } - fn stdio_ioctl(&self, arg: &IoctlArg) -> Result { + /// Handle syscall `inotify_init1` (and plain `inotify_init`, dispatched with `flags` forced + /// empty -- see `SyscallRequest::InotifyInit1`'s construction from `Sysno::inotify_init`). + pub fn sys_inotify_init1(&self, flags: InotifyInitFlags) -> Result { + if flags.intersects((InotifyInitFlags::NONBLOCK | InotifyInitFlags::CLOEXEC).complement()) { + return Err(Errno::EINVAL); + } + + let inotify = self.global.create_linux_inotify(flags); + let mut dt = self.global.litebox.descriptor_table_mut(); + let typed = dt.insert::>(inotify); + if flags.contains(InotifyInitFlags::CLOEXEC) { + let old = dt.set_fd_metadata(&typed, FileDescriptorFlags::FD_CLOEXEC); + assert!(old.is_none()); + } + drop(dt); + let files = self.files.borrow(); + let raw_fd = files.insert_raw_fd(typed).map_err(|typed| { + self.global + .litebox + .descriptor_table_mut() + .remove(&typed) + .unwrap(); + Errno::EMFILE + })?; + Ok(raw_fd.try_into().unwrap()) + } + + /// Handle syscall `inotify_add_watch`. `pathname` is accepted (matching this shim's + /// no-change-notification-producer honesty note in `syscalls::inotify`'s doc comment: a + /// watch is only ever validated as a syntactically well-formed request, never resolved + /// against the filesystem, since it will never actually observe that path change) but is not + /// otherwise inspected. + pub fn sys_inotify_add_watch( + &self, + fd: i32, + _pathname: alloc::ffi::CString, + mask: InotifyMask, + ) -> Result { + let files = self.files.borrow(); + let inotify_fd = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(fd as usize) + .map_err(|_| Errno::EBADF)?; + let handle = self + .global + .litebox + .descriptor_table() + .entry_handle(&inotify_fd) + .ok_or(Errno::EBADF)?; + handle.with_entry(|file| file.add_watch(mask)) + } + + /// Handle syscall `inotify_rm_watch`. + pub fn sys_inotify_rm_watch(&self, fd: i32, wd: i32) -> Result<(), Errno> { + let files = self.files.borrow(); + let inotify_fd = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(fd as usize) + .map_err(|_| Errno::EBADF)?; + let handle = self + .global + .litebox + .descriptor_table() + .entry_handle(&inotify_fd) + .ok_or(Errno::EBADF)?; + handle.with_entry(|file| file.rm_watch(wd)) + } + + fn stdio_ioctl(&self, stream: StdioStream, arg: &IoctlArg) -> Result { match arg { IoctlArg::TCGETS(termios) => { + let current = self.global.termios.lock().clone(); termios - .write_at_offset::( - 0, - litebox_common_linux::Termios { - c_iflag: 0, - c_oflag: 0, - c_cflag: 0, - c_lflag: 0, - c_line: 0, - c_cc: [0; 19], - }, - ) + .write_at_offset::(0, current) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::TCSETS(termios, action) => { + let new_termios = termios.read_at_offset::(0).ok_or(Errno::EFAULT)?; + let lflag = litebox_common_linux::LFlag::from_bits_truncate(new_termios.c_lflag); + let raw = !lflag.contains(litebox_common_linux::LFlag::ICANON); + let echo = lflag.contains(litebox_common_linux::LFlag::ECHO); + *self.global.termios.lock() = new_termios; + // Mirror the raw/echo-relevant bits onto the real host terminal so keystrokes + // actually arrive byte-at-a-time once the guest disables canonical mode, honoring + // TCSETS/TCSETSW/TCSETSF's NOW/DRAIN/FLUSH distinction -- a no-op on + // platforms/streams without a real backing terminal. + let platform_action = match action { + litebox_common_linux::TerminalSetAction::Now => { + litebox::platform::TerminalSetAction::Now + } + litebox_common_linux::TerminalSetAction::Drain => { + litebox::platform::TerminalSetAction::Drain + } + litebox_common_linux::TerminalSetAction::Flush => { + litebox::platform::TerminalSetAction::Flush + } + }; + self.global.platform.set_terminal_raw_mode_with_action( + stream, + raw, + echo, + platform_action, + ); + Ok(0) + } + IoctlArg::TIOCGPGRP(pgrp) => { + let pgid = self.global.stdio_foreground_pgid.load(Ordering::Acquire); + pgrp.write_at_offset::(0, pgid) .ok_or(Errno::EFAULT)?; Ok(0) } - IoctlArg::TCSETS(_) => Ok(0), // TODO: implement + IoctlArg::TIOCSPGRP(pgrp) => { + let pgid = pgrp.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if pgid <= 0 { + return Err(Errno::EINVAL); + } + self.global + .stdio_foreground_pgid + .store(pgid, Ordering::Release); + Ok(0) + } IoctlArg::TIOCGWINSZ(ws) => { + // Query the real terminal size where the platform can provide it (e.g. via + // `TIOCGWINSZ` on the real host fd, or `GetConsoleScreenBufferInfo` on Windows); + // fall back to the traditional 80x24 default otherwise. A guest's own line + // editor (e.g. `ash`'s `lineedit.c`) uses this to decide the column width at + // which to wrap its own echoed-input redisplay, so returning a fake, too-narrow + // size here (previously hardcoded to 20x20) caused spurious wraps in the echo of + // typed input well before the real terminal would ever need to wrap. + let (row, col) = self.global.platform.tty_window_size().unwrap_or((24, 80)); ws.write_at_offset::( 0, litebox_common_linux::Winsize { - row: 20, - col: 20, + row, + col, xpixel: 0, ypixel: 0, }, @@ -1805,21 +5032,166 @@ impl Task { .ok_or(Errno::EFAULT)?; Ok(0) } - IoctlArg::TIOCGPTN(_) => Err(Errno::ENOTTY), - _ => todo!(), + // The host terminal already is this session's controlling terminal, and its window + // is the host's to size: both are accepted as no-ops. + IoctlArg::TIOCSCTTY(_) | IoctlArg::TIOCSWINSZ(_) => Ok(0), + _ => Err(Errno::ENOTTY), + } + } + + fn pty_ioctl( + &self, + endpoint: &PtyEndpoint, + arg: &IoctlArg, + ) -> Result { + match arg { + IoctlArg::TCGETS(termios) => { + let current = endpoint.state.termios.lock().clone(); + termios + .write_at_offset::(0, current) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::TCSETS(termios, _action) => { + let termios = termios.read_at_offset::(0).ok_or(Errno::EFAULT)?; + *endpoint.state.termios.lock() = termios; + Ok(0) + } + IoctlArg::TIOCSCTTY(_force) => { + if endpoint.side != PtySide::Slave { + return Err(Errno::EINVAL); + } + self.process() + .acquire_controlling_pty(self.pid, endpoint.number)?; + endpoint + .state + .foreground_pgid + .store(self.process().process_group_id(), Ordering::Release); + Ok(0) + } + IoctlArg::TIOCGPGRP(pgrp) => { + let pgid = endpoint.state.foreground_pgid.load(Ordering::Acquire); + pgrp.write_at_offset::(0, pgid) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::TIOCSPGRP(pgrp) => { + let pgid = pgrp.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if pgid <= 0 { + return Err(Errno::EINVAL); + } + endpoint + .state + .foreground_pgid + .store(pgid, Ordering::Release); + Ok(0) + } + IoctlArg::TIOCGWINSZ(winsize) => { + let current = endpoint.state.winsize.lock().clone(); + winsize + .write_at_offset::(0, current) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::TIOCSWINSZ(winsize) => { + let winsize = winsize.read_at_offset::(0).ok_or(Errno::EFAULT)?; + *endpoint.state.winsize.lock() = winsize; + Ok(0) + } + IoctlArg::TIOCGPTN(number) if endpoint.side == PtySide::Master => { + number + .write_at_offset::(0, endpoint.number) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::TIOCSPTLCK(locked) if endpoint.side == PtySide::Master => { + let locked = locked.read_at_offset::(0).ok_or(Errno::EFAULT)?; + endpoint + .state + .unlocked + .store(locked == 0, Ordering::Release); + Ok(0) + } + _ => Err(Errno::ENOTTY), } } fn is_stdio(&self, fs: &FS, fd: &TypedFd) -> Result { match fs.fd_file_status(fd) { Ok(status) => { - // See https://www.kernel.org/doc/Documentation/admin-guide/devices.txt + // See https://www.kernel.org/doc/Documentation/admin-guide/devices.txt: the + // Unix98 pty slave majors the stdio devices report, plus `/dev/tty` itself. + let rdev = status.node_info.rdev.map_or(0, core::num::NonZeroUsize::get); + let major = rdev >> 8; + Ok(status.file_type == litebox::fs::FileType::CharacterDevice + && ((136..=143).contains(&major) + || rdev == litebox::fs::devices::TTYAUX_MAJOR << 8)) + } + Err(litebox::fs::errors::FileStatusError::ClosedFd) => Err(Errno::EBADF), + Err(_) => unimplemented!(), + } + } + + /// If `fd` names a `/dev/input/event*` device, its evdev minor number -- read from the + /// [`InputEventMinor`] metadata attached at open time. `None` for every other fd (the + /// read/ioctl paths then fall through to their normal handling). + fn input_event_minor(&self, _fs: &FS, fd: &TypedFd) -> Option { + input_event_minor_of(&self.global, fd) + } + + /// Blocking-capable `read()` for a `/dev/input/event*` fd: waits on the device's event + /// queue with this task's wait context (which the `Backend` trait's `read` cannot do), per + /// evdev semantics -- whole 24-byte events only, `EINVAL` for a short buffer, `EAGAIN` + /// only under `O_NONBLOCK`. + fn read_input_events(&self, meta: InputEventMinor, buf: &mut [u8]) -> Result { + let InputEventMinor { minor, nonblock } = meta; + // `/dev/input/mice` is a plain byte stream (its handshake reads 1 byte at a time); + // only the evdev event devices insist on whole 24-byte events. + if minor != litebox::fs::devices::MICE_MINOR + && buf.len() < litebox::fs::devices::INPUT_EVENT_SIZE + { + return Err(Errno::EINVAL); + } + let Some(registry) = self.global.input_registry.as_ref() else { + return Err(Errno::ENODEV); + }; + // `nonblock` is the open-time `O_NONBLOCK`/`O_NDELAY` (see `InputEventMinor`); honoring + // it is load-bearing: Xorg's evdev driver opens the device `O_NDELAY` and drains it + // with reads it expects to `EAGAIN` when empty -- a blocking read here wedged the X + // server's whole main loop (observed live as the desktop-wide deadlock). + registry + .read_blocking(&self.wait_cx(), minor, buf, nonblock) + .map_err(|e| match e { + // A timeout maps to EAGAIN like an empty non-blocking read would -- though with + // no timeout on this wait context it never actually fires. + litebox::event::polling::TryOpError::TryAgain + | litebox::event::polling::TryOpError::WaitError( + litebox::event::wait::WaitError::TimedOut, + ) => Errno::EAGAIN, + litebox::event::polling::TryOpError::WaitError( + litebox::event::wait::WaitError::Interrupted, + ) => Errno::EINTR, + litebox::event::polling::TryOpError::Other(infallible) => match infallible {}, + }) + } + + /// Whether `fd` names `/dev/fb0` -- recognized the same way [`Self::is_stdio`] recognizes a + /// tty (by the `rdev` major number [`litebox::fs::devices`] assigns it), rather than by + /// requiring a distinct fd-table subsystem for one device. + pub(crate) fn is_fb0(&self, fs: &FS, fd: &TypedFd) -> Result { + match fs.fd_file_status(fd) { + Ok(status) => { let major = status.node_info.rdev.map_or(0, |v| v.get() >> 8); - Ok((136..=143).contains(&major) + Ok(major == litebox::fs::devices::FB_MAJOR && status.file_type == litebox::fs::FileType::CharacterDevice) } Err(litebox::fs::errors::FileStatusError::ClosedFd) => Err(Errno::EBADF), - Err(_) => unimplemented!(), + // `fd` was already resolved to an open fs-backend fd by the caller's + // `run_on_raw_fd`, so a status lookup on it failing with anything other than + // `ClosedFd` (`Io`, `PathError`, or any variant `#[non_exhaustive]` may add later) + // cannot happen in practice; report "not recognized as fb0" rather than assume a + // specific unreachable shape. + Err(_) => Ok(false), } } @@ -1837,11 +5209,22 @@ impl Task { .borrow() .run_on_raw_fd( desc, - |_file_fd| { - // TODO: stdio NONBLOCK? - #[cfg(debug_assertions)] - litebox_util_log::debug!("set non-blocking on raw fd unimplemented"); - Ok(()) + |file_fd| { + // Only stdio raw fds carry `StdioStatusFlags` metadata (see + // `initialize_stdio_in_shared_descriptors_table`); other regular + // files have no non-blocking story at the `Backend` layer, so + // `NoSuchMetadata` there is expected and silently ignored, matching + // this ioctl's pre-existing (no-op) behavior for non-stdio raw fds. + match self + .global + .litebox + .descriptor_table_mut() + .with_metadata_mut(file_fd, |crate::StdioStatusFlags(flags)| { + flags.set(OFlags::NONBLOCK, val != 0); + }) { + Ok(()) | Err(MetadataError::NoSuchMetadata) => Ok(()), + Err(MetadataError::ClosedFd) => Err(Errno::EBADF), + } }, |socket_fd| { if let Err(e) = self @@ -1904,10 +5287,30 @@ impl Task { }); Ok(()) }, + |fd| { + self.global + .litebox + .descriptor_table_mut() + .with_metadata_mut( + fd, + |crate::syscalls::net::SocketOFlags(flags)| { + flags.set(OFlags::NONBLOCK, val != 0); + }, + ) + .map_err(|error| match error { + MetadataError::ClosedFd => Errno::EBADF, + MetadataError::NoSuchMetadata => { + unreachable!("all netlink sockets have SocketOFlags when created") + } + }) + }, ) .flatten()?; Ok(0) } + // `FD_CLOEXEC` lives on the descriptor-table entry, not on + // anything specific to a given fd kind, so every kind sets it the + // same way. IoctlArg::FIOCLEX => files.run_on_raw_fd( desc, |fd| { @@ -1918,8 +5321,30 @@ impl Task { .set_fd_metadata(fd, FileDescriptorFlags::FD_CLOEXEC); Ok(0) }, - |_fd| todo!("net"), - |_fd| todo!("pipes"), + |fd| { + let _old = self + .global + .litebox + .descriptor_table_mut() + .set_fd_metadata(fd, FileDescriptorFlags::FD_CLOEXEC); + Ok(0) + }, + |fd| { + let _old = self + .global + .litebox + .descriptor_table_mut() + .set_fd_metadata(fd, FileDescriptorFlags::FD_CLOEXEC); + Ok(0) + }, + |fd| { + let _old = self + .global + .litebox + .descriptor_table_mut() + .set_fd_metadata(fd, FileDescriptorFlags::FD_CLOEXEC); + Ok(0) + }, |fd| { let _old = self .global @@ -1947,8 +5372,13 @@ impl Task { )?, IoctlArg::TCGETS(..) | IoctlArg::TCSETS(..) + | IoctlArg::TIOCSCTTY(..) + | IoctlArg::TIOCGPGRP(..) + | IoctlArg::TIOCSPGRP(..) | IoctlArg::TIOCGPTN(..) - | IoctlArg::TIOCGWINSZ(..) => files.run_on_raw_fd( + | IoctlArg::TIOCSPTLCK(..) + | IoctlArg::TIOCGWINSZ(..) + | IoctlArg::TIOCSWINSZ(..) => files.run_on_raw_fd( desc, |fd| { if self.is_stdio(&files.fs, fd)? { @@ -1968,7 +5398,7 @@ impl Task { Errno::ENOTTY })?; if self.global.platform.is_a_tty(stream) { - self.stdio_ioctl(&arg) + self.stdio_ioctl(stream, &arg) } else { Err(Errno::ENOTTY) } @@ -1980,13 +5410,390 @@ impl Task { |_fd| Err(Errno::ENOTTY), |_fd| Err(Errno::ENOTTY), |_fd| Err(Errno::ENOTTY), + |fd| { + let endpoint = self.pty_endpoint(fd).ok_or(Errno::ENOTTY)?; + self.pty_ioctl(&endpoint, &arg) + }, + |_fd| Err(Errno::ENOTTY), + )?, + IoctlArg::FBIOGET_VSCREENINFO(..) + | IoctlArg::FBIOPUT_VSCREENINFO(..) + | IoctlArg::FBIOGET_FSCREENINFO(..) + | IoctlArg::FBIOPAN_DISPLAY(..) + | IoctlArg::FBIOBLANK => files.run_on_raw_fd( + desc, + |fd| -> Result { + if !self.is_fb0(&files.fs, fd)? { + return Err(Errno::ENOTTY); + } + // A framebuffer-typed fd only exists when `default_fs` mounted one (the + // sole source of an fb0 rdev major), so a `None` here would mean an fd + // recognized as fb0 by a filesystem this shim never built -- report + // "not a tty-like device" rather than assume that can't happen. + let Some(fb) = self.global.framebuffer.as_ref() else { + return Err(Errno::ENOTTY); + }; + match &arg { + IoctlArg::FBIOGET_VSCREENINFO(out) => { + out.write_at_offset::(0, fb.var_screeninfo()) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::FBIOPUT_VSCREENINFO(req) => { + let req = req.read_at_offset::(0).ok_or(Errno::EFAULT)?; + fb.put_var_screeninfo(&req); + Ok(0) + } + IoctlArg::FBIOGET_FSCREENINFO(out) => { + out.write_at_offset::(0, fb.fix_screeninfo()) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + IoctlArg::FBIOPAN_DISPLAY(req) => { + let req = req.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if fb.pan_display(req.yoffset) { + Ok(0) + } else { + Err(Errno::EINVAL) + } + } + // litebox has no real display hardware to blank; treat every blank + // level (including an unrecognized one) as a trivially successful + // no-op, matching how fbdevhw.c tolerates a driver that can't blank. + // (`FBIOBLANK` carries no payload, so this is also the only other + // variant the outer match admits here.) + _ => Ok(0), + } + }, + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), |_fd| Err(Errno::ENOTTY), )?, + // The `EVIOC*` family ('E' = 0x45 in the ioctl type byte) arrives undecoded as + // `Raw` -- the variable-length getters (`EVIOCGNAME(len)` etc.) encode the caller's + // buffer length in the command itself, so there's nothing for the static decoder in + // `litebox_common_linux` to name per-command. Dispatched to the input registry when + // the fd is a `/dev/input/event*` device; every other fd falls through to the + // unsupported catch-all below. + IoctlArg::Raw { cmd, arg: raw_arg } if (cmd >> 8) & 0xff == 0x45 => files + .run_on_raw_fd( + desc, + |fd| -> Result { + let Some(minor) = self.input_event_minor(&files.fs, fd) else { + return Err(Errno::EINVAL); + }; + let Some(registry) = self.global.input_registry.as_ref() else { + return Err(Errno::ENODEV); + }; + // The only write-direction commands this registry accepts carry an + // `int`: `EVIOCGRAB`'s flag rides in the argument value itself (the + // kernel never dereferences it), `EVIOCSCLOCKID`'s clockid is in user + // memory. Read the user int only for the latter. + let write_arg = if cmd & 0xff == 0xa0 { + let ptr = litebox_common_linux::user_pointers::UserPtr::::from_usize( + raw_arg.as_usize(), + ); + ptr.read_at_offset::(0).ok_or(Errno::EFAULT)? + } else { + i32::try_from(raw_arg.as_usize() & 0xffff_ffff) + .unwrap_or(i32::MAX) + }; + match registry.evdev_ioctl(minor, cmd, write_arg) { + litebox::fs::devices::EvdevIoctlReply::Copy { data, rc } => { + let dst = litebox_common_linux::user_pointers::UserPtrMut::::from_usize( + raw_arg.as_usize(), + ); + dst.copy_from_slice::(0, &data) + .ok_or(Errno::EFAULT)?; + Ok(rc) + } + litebox::fs::devices::EvdevIoctlReply::Plain { rc } => Ok(rc), + litebox::fs::devices::EvdevIoctlReply::NoEntry => Err(Errno::ENOENT), + litebox::fs::devices::EvdevIoctlReply::Invalid => Err(Errno::EINVAL), + } + }, + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + |_fd| Err(Errno::EINVAL), + )?, + // The VT console family ('V' = 0x56, legacy non-`_IOC`-encoded commands) also + // arrives undecoded as `Raw`. litebox has no virtual terminals to switch between; + // fbdev graphics clients (links2's `-g` fb driver is the archetype) nonetheless + // require `VT_GETMODE`/`VT_SETMODE` to succeed on their controlling tty before + // they will draw, and issue the rest fire-and-forget. Answer as a console whose + // single VT (1) is permanently active -- kernel-shaped for a host with exactly one + // seat and no console switching. + IoctlArg::Raw { cmd, arg: raw_arg } if (cmd >> 8) & 0xff == 0x56 => files + .run_on_raw_fd( + desc, + |fd| -> Result { + if !self.is_stdio(&files.fs, fd)? { + return Err(Errno::ENOTTY); + } + let write_bytes = |bytes: &[u8]| -> Result { + let dst = + litebox_common_linux::user_pointers::UserPtrMut::::from_usize( + raw_arg.as_usize(), + ); + dst.copy_from_slice::(0, bytes) + .ok_or(Errno::EFAULT)?; + Ok(0) + }; + match cmd { + // VT_GETMODE: `struct vt_mode { char mode; char waitv; short + // relsig; short acqsig; short frsig; }` -- VT_AUTO, no signals. + 0x5601 => write_bytes(&[0u8; 8]), + // VT_GETSTATE: `struct vt_stat { u16 v_active; u16 v_signal; + // u16 v_state; }` -- VT 1 active, VTs 0/1 open. + 0x5603 => write_bytes(&[1, 0, 0, 0, 3, 0]), + // VT_SETMODE (accepted and ignored; with no VT switching the + // release/acquire signals it configures can never fire), and + // VT_RELDISP / VT_ACTIVATE / VT_WAITACTIVE (the sole VT is + // always already active). + 0x5602 | 0x5605..=0x5607 => Ok(0), + _ => { + log_unsupported!("VT ioctl {cmd:#x}"); + Err(Errno::EINVAL) + } + } + }, + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + )?, + // Legacy `SIOCGIF*`/`SIOCGIFCONF` socket ioctls ('S' = 0x89 in the ioctl type + // byte). `getifaddrs(3)` reaches interfaces through rtnetlink (see + // `crate::syscalls::netlink`) and never issues these, but tools built directly + // against BSD-style `ifreq`/`ifconf` -- busybox `ifconfig`, `route` -- still do. + // Answered from the same fixed two-interface table netlink synthesises: `lo` + // (127.0.0.1/8) and `eth0` (this guest's real address, from `self.global.net`). + IoctlArg::Raw { cmd, arg: raw_arg } if (cmd >> 8) & 0xff == 0x89 => files + .run_on_raw_fd( + desc, + |_fd| -> Result { Err(Errno::ENOTTY) }, + |_fd| -> Result { + self.sys_ioctl_siocgif(cmd, raw_arg) + }, + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| Err(Errno::ENOTTY), + |_fd| -> Result { + // AF_UNIX sockets answer the same fixed table: real programs only + // ever issue these against an AF_INET socket, but the kernel itself + // does not check the socket's domain for `SIOCGIF*` (it is a global + // ioctl number, not protocol-specific), so neither does this shim. + self.sys_ioctl_siocgif(cmd, raw_arg) + }, + |_fd| -> Result { self.sys_ioctl_siocgif(cmd, raw_arg) }, + )?, + _ => { + log_unsupported!("ioctl with arg {:?}", arg); + Err(Errno::EINVAL) + } + } + } + + /// The `SIOCGIFCONF`/`SIOCGIFFLAGS`/`SIOCGIFADDR`/`SIOCGIFNETMASK`/`SIOCGIFBRDADDR`/ + /// `SIOCGIFHWADDR`/`SIOCGIFMTU`/`SIOCGIFINDEX`/`SIOCGIFTXQLEN` family, against the fixed + /// `lo` + `eth0` interface table (see the call site's doc comment). `raw_arg` points at a + /// `struct ifconf` for `SIOCGIFCONF`, a `struct ifreq` for everything else. + fn sys_ioctl_siocgif(&self, cmd: u32, raw_arg: UserPtrMut) -> Result { + use litebox_common_linux::{ + IFNAMSIZ, IfrFlags, SIOCGIFADDR, SIOCGIFBRDADDR, SIOCGIFCONF, SIOCGIFFLAGS, + SIOCGIFHWADDR, SIOCGIFINDEX, SIOCGIFMTU, SIOCGIFNETMASK, SIOCGIFTXQLEN, + }; + + // `sockaddr_in { sa_family: u16, sin_port: u16, sin_addr: [u8;4], sin_zero: [u8;8] }`, + // padded to `sockaddr`'s 16 bytes -- the shape every `SIOCGIF*` address getter writes + // into `ifr_ifru.ifru_addr`. + const AF_INET: u16 = 2; + fn sockaddr_in(addr: [u8; 4]) -> [u8; 16] { + let mut b = [0u8; 16]; + b[0..2].copy_from_slice(&AF_INET.to_ne_bytes()); + b[4..8].copy_from_slice(&addr); + b + } + fn write_ifreq_name( + dst: UserPtrMut, + name: &[u8], + ) -> Option<()> { + let mut buf = [0u8; IFNAMSIZ]; + buf[..name.len()].copy_from_slice(name); + dst.copy_from_slice::

(0, &buf) + } + const ARPHRD_ETHER: u16 = 1; + const ARPHRD_LOOPBACK: u16 = 772; + const LO_ADDR: [u8; 4] = [127, 0, 0, 1]; + const LO_MASK: [u8; 4] = [255, 0, 0, 0]; + const ETH_MASK: [u8; 4] = [255, 255, 255, 0]; + const ETH_MAC: [u8; 6] = [0x02, 0x00, 0x00, 0x00, 0x00, 0x02]; + const MTU: i32 = 1500; + // `struct ifreq` on musl/aarch64 is 40 bytes total, not the 32 a bare + // `ifr_name[IFNAMSIZ] + sockaddr` sum would suggest: the `ifr_ifru` + // union member at offset 16 is sized 24 bytes (verified live via + // `sizeof(struct ifreq)`/`offsetof` on this target), not 16, to leave + // room for union members not used here (e.g. `sockaddr_in6`-shaped + // ones). `SIOCGIFCONF`'s stride through the caller-provided buffer + // must match this exactly: a 32-byte stride left every entry past + // the first pointing 8 bytes into the *next* real entry's payload, + // so a caller (confirmed live: busybox `ifconfig`, whose own + // `if_readconf` walks `ifc.ifc_req` in `sizeof(struct ifreq)` + // strides) reading the second entry read zeroed padding instead of + // "eth0", producing an empty-named phantom interface and a fatal + // `SIOCGIFFLAGS` probe against it ("error fetching interface + // information: Device not found"). + const IFREQ_SIZE: usize = IFNAMSIZ + 24; + + struct Iface { + name: &'static [u8], + addr: [u8; 4], + netmask: [u8; 4], + flags: IfrFlags, + hw_type: u16, + /// `ifr_ifindex`; the same numbering `crate::syscalls::netlink`'s link dump reports. + index: i32, + /// `ifr_qlen`; a loopback has no transmit queue. + txqueuelen: i32, + } + let eth_addr = self.global.net.lock().interface_ip().octets(); + let ifaces = [ + Iface { + name: b"lo", + addr: LO_ADDR, + netmask: LO_MASK, + flags: IfrFlags::IFF_UP | IfrFlags::IFF_LOOPBACK | IfrFlags::IFF_RUNNING, + hw_type: ARPHRD_LOOPBACK, + index: 1, + txqueuelen: 0, + }, + Iface { + name: b"eth0", + addr: eth_addr, + netmask: ETH_MASK, + flags: IfrFlags::IFF_UP + | IfrFlags::IFF_BROADCAST + | IfrFlags::IFF_RUNNING + | IfrFlags::IFF_MULTICAST, + hw_type: ARPHRD_ETHER, + index: 2, + txqueuelen: 1000, + }, + ]; + + if cmd == SIOCGIFCONF { + // `struct ifconf { int ifc_len; union { char *ifc_buf; struct ifreq *ifc_req; }; }`. + // LP64 pads `ifc_len` to 8 bytes before the pointer. + let header = raw_arg + .to_owned_slice::(16) + .ok_or(Errno::EFAULT)?; + let ifc_len = i32::from_ne_bytes( + <[u8; 4]>::try_from(&header[0..4]).unwrap_or_else(|_| unreachable!()), + ); + let buf_addr = usize::from_ne_bytes( + <[u8; 8]>::try_from(&header[8..16]).unwrap_or_else(|_| unreachable!()), + ); + let dst = UserPtrMut::::from_usize(buf_addr); + + let capacity = usize::try_from(ifc_len.max(0)).unwrap_or(0) / IFREQ_SIZE; + let n = ifaces.len().min(capacity); + for (i, iface) in ifaces.iter().take(n).enumerate() { + let entry = UserPtrMut::::from_usize(dst.as_usize() + i * IFREQ_SIZE); + write_ifreq_name::(entry, iface.name).ok_or(Errno::EFAULT)?; + let addr_entry = UserPtrMut::::from_usize(entry.as_usize() + IFNAMSIZ); + addr_entry + .copy_from_slice::(0, &sockaddr_in(iface.addr)) + .ok_or(Errno::EFAULT)?; + } + let written_len = i32::try_from(n * IFREQ_SIZE).unwrap_or(0); + raw_arg + .copy_from_slice::(0, &written_len.to_ne_bytes()) + .ok_or(Errno::EFAULT)?; + return Ok(0); + } + + // Every other command names one interface in `ifr_name` and gets a reply written + // into the same union slot the name doesn't occupy. + let name_bytes = raw_arg + .to_owned_slice::(IFNAMSIZ) + .ok_or(Errno::EFAULT)?; + let name_len = name_bytes.iter().position(|&b| b == 0).unwrap_or(IFNAMSIZ); + let name = &name_bytes[..name_len]; + let Some(iface) = ifaces.iter().find(|i| i.name == name) else { + return Err(Errno::ENODEV); + }; + let payload = UserPtrMut::::from_usize(raw_arg.as_usize() + IFNAMSIZ); + match cmd { + SIOCGIFFLAGS => { + let mut b = [0u8; 16]; + b[0..2].copy_from_slice(&iface.flags.bits().to_ne_bytes()); + payload + .copy_from_slice::(0, &b) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFADDR => { + payload + .copy_from_slice::(0, &sockaddr_in(iface.addr)) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFNETMASK => { + payload + .copy_from_slice::(0, &sockaddr_in(iface.netmask)) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFBRDADDR => { + let broadcast = [ + iface.addr[0] | !iface.netmask[0], + iface.addr[1] | !iface.netmask[1], + iface.addr[2] | !iface.netmask[2], + iface.addr[3] | !iface.netmask[3], + ]; + payload + .copy_from_slice::(0, &sockaddr_in(broadcast)) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFHWADDR => { + let mut b = [0u8; 16]; + b[0..2].copy_from_slice(&iface.hw_type.to_ne_bytes()); + let mac = if iface.hw_type == ARPHRD_LOOPBACK { + [0u8; 6] + } else { + ETH_MAC + }; + b[2..8].copy_from_slice(&mac); + payload + .copy_from_slice::(0, &b) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFMTU => { + payload + .copy_from_slice::(0, &MTU.to_ne_bytes()) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFINDEX => { + payload + .copy_from_slice::(0, &iface.index.to_ne_bytes()) + .ok_or(Errno::EFAULT)?; + } + SIOCGIFTXQLEN => { + payload + .copy_from_slice::(0, &iface.txqueuelen.to_ne_bytes()) + .ok_or(Errno::EFAULT)?; + } _ => { - log_unsupported!("ioctl with arg {:?}", arg); - Err(Errno::EINVAL) + log_unsupported!("SIOCGIF ioctl {cmd:#x}"); + return Err(Errno::EINVAL); } } + Ok(0) } /// Handle syscall `epoll_create` and `epoll_create1` @@ -2040,20 +5847,26 @@ impl Task { .read() .fd_from_raw_integer::>(epfd as usize) .map_err(|_| Errno::EBADF)?; - let file_descriptor = super::epoll::EpollDescriptor::try_from(&files, fd as usize)?; + let file_descriptor = + super::epoll::EpollDescriptor::try_from(&self.global, &files, fd as usize)?; - let event = if op == litebox_common_linux::EpollOp::EpollCtlDel { - None - } else { - Some(event.read_at_offset::(0).ok_or(Errno::EFAULT)?) - }; let handle = self .global .litebox .descriptor_table() .entry_handle(&epoll_fd) .ok_or(Errno::EBADF)?; - handle.with_entry(|entry| entry.epoll_ctl(&self.global, op, fd, &file_descriptor, event)) + if file_descriptor.is_epoll_identity(handle.identity()) { + return Err(Errno::EINVAL); + } + let event = if op == litebox_common_linux::EpollOp::EpollCtlDel { + None + } else { + Some(event.read_at_offset::(0).ok_or(Errno::EFAULT)?) + }; + handle.with_entry(|entry| { + entry.epoll_ctl(&self.global, &epoll_fd, op, fd, &file_descriptor, event) + }) } /// Handle syscall `epoll_pwait` @@ -2064,11 +5877,18 @@ impl Task { maxevents: u32, timeout: i32, sigmask: Option>, - _sigsetsize: usize, + sigsetsize: usize, ) -> Result { - if sigmask.is_some() { - todo!("sigmask not supported"); - } + // epoll_pwait(2): same temporary-mask contract as `ppoll`/`pselect`. + let sigmask = match sigmask { + Some(sigmask) => { + if sigsetsize != core::mem::size_of::() { + return Err(Errno::EINVAL); + } + Some(sigmask.read_at_offset::(0).ok_or(Errno::EFAULT)?) + } + None => None, + }; let Ok(epfd) = u32::try_from(epfd) else { return Err(Errno::EBADF); }; @@ -2103,7 +5923,7 @@ impl Task { .ok_or(Errno::EBADF)? } }; - handle.with_entry(|epoll_file| { + let wait = || handle.with_entry(|epoll_file| { match epoll_file.wait( &self.global, &self.wait_cx().with_timeout(timeout), @@ -2120,7 +5940,11 @@ impl Task { Err(WaitError::TimedOut) => Ok(0), Err(WaitError::Interrupted) => Err(Errno::EINTR), } - }) + }); + match sigmask { + Some(mask) => self.with_temporary_signal_mask(mask, wait), + None => wait(), + } } /// Handle syscall `ppoll`. @@ -2132,13 +5956,18 @@ impl Task { sigmask: Option>, sigsetsize: usize, ) -> Result { - if sigmask.is_some() { - if sigsetsize != core::mem::size_of::() { - // Expected via ppoll(2) manpage - unimplemented!() + // ppoll(2): the mask, when given, replaces the thread's blocked set for the + // duration of the wait only (see `with_temporary_signal_mask`), exactly like + // `pselect`; a wrong `sigsetsize` is `EINVAL`. + let sigmask = match sigmask { + Some(sigmask) => { + if sigsetsize != core::mem::size_of::() { + return Err(Errno::EINVAL); + } + Some(sigmask.read_at_offset::(0).ok_or(Errno::EFAULT)?) } - unimplemented!("no sigmask support yet"); - } + None => None, + }; let timeout = timeout.read::()?; let nfds_signed = isize::try_from(nfds).map_err(|_| Errno::EINVAL)?; @@ -2152,11 +5981,21 @@ impl Task { set.add_fd(fd.fd, events); } - match set.wait( - &self.global, - &self.wait_cx().with_timeout(timeout), - &self.files.borrow(), - ) { + let wait_result = match sigmask { + Some(mask) => self.with_temporary_signal_mask(mask, || { + set.wait( + &self.global, + &self.wait_cx().with_timeout(timeout), + &self.files.borrow(), + ) + }), + None => set.wait( + &self.global, + &self.wait_cx().with_timeout(timeout), + &self.files.borrow(), + ), + }; + match wait_result { Ok(()) => {} Err(WaitError::Interrupted) => { // TODO: update the remaining time. @@ -2279,12 +6118,19 @@ impl Task { if sigsetpack.size != core::mem::size_of::() { return Err(Errno::EINVAL); } - Some( - sigsetpack - .sigset - .read_at_offset::(0) - .ok_or(Errno::EFAULT)?, - ) + // A null sigset inside a non-null pack means "don't touch the mask" -- exactly how + // the kernel reads it, and exactly what musl's plain `select` always passes + // (`{ss: NULL, ss_len: _NSIG/8}`). + if sigsetpack.sigset.is_null() { + None + } else { + Some( + sigsetpack + .sigset + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?, + ) + } } else { None }; @@ -2382,6 +6228,7 @@ impl Task { let mut dt = task.global.litebox.descriptor_table_mut(); let fd: TypedFd<_> = dt.duplicate(fd).ok_or(DupFdError::BadFd)?; + note_pty_slave_descriptor::(&dt, &fd); if close_on_exec { let old = dt.set_fd_metadata(&fd, FileDescriptorFlags::FD_CLOEXEC); assert!(old.is_none()); @@ -2420,6 +6267,16 @@ impl Task { let close_on_exec = flags.contains(OFlags::CLOEXEC); let files = self.files.borrow(); + // See the matching pre-check in `fork_copy`: inotify fds sit outside `run_on_raw_fd`'s + // hand-enumerated subsystem list, so `dup`/`dup2`/`dup3`/`fcntl(F_DUPFD*)` on one would + // otherwise fail `BadFd` even though the fd is alive and valid. + if let Ok(inotify_fd) = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::>(file) + { + return dup(self, &files, &inotify_fd, close_on_exec, target); + } files .run_on_raw_fd( file, @@ -2429,6 +6286,7 @@ impl Task { |fd| dup(self, &files, fd, close_on_exec, target), |fd| dup(self, &files, fd, close_on_exec, target), |fd| dup(self, &files, fd, close_on_exec, target), + |fd| dup(self, &files, fd, close_on_exec, target), ) .map_err(|_| DupFdError::BadFd)? } @@ -2499,11 +6357,9 @@ enum DupFdError { TargetFdExceedsLimit, } -#[derive(Clone, Copy, Debug, Default)] -struct Diroff(usize); - const DIRENT_STRUCT_BYTES_WITHOUT_NAME: usize = core::mem::offset_of!(litebox_common_linux::LinuxDirent64, __name); +const DIRENT_ALIGNMENT: usize = align_of::(); impl Task { /// Handle syscall `getdents64` @@ -2520,70 +6376,109 @@ impl Task { files.run_on_raw_fd( fd, |file| { - let dir_off: Diroff = self - .global - .litebox - .descriptor_table() - .with_metadata(file, |off: &Diroff| *off) - .unwrap_or_default(); - let mut dir_off = dir_off.0; - let mut nbytes = 0; - - let mut entries = files.fs.read_dir(file)?; + let mut entries = files.fs.read_dir(file).map_err(Errno::from)?; entries.sort_by(|a, b| a.name.cmp(&b.name)); - for entry in entries.iter().skip(dir_off) { - // include null terminator and make it aligned - let len = (DIRENT_STRUCT_BYTES_WITHOUT_NAME + entry.name.len() + 1) - .next_multiple_of(align_of::()); - if nbytes + len > count { - // not enough space - if nbytes == 0 { - // not enough space for even a single entry - return Err(Errno::EINVAL); + files + .fs + .with_dir_position(file, |dir_off| { + let mut nbytes = 0usize; + while let Some(entry) = entries.get(*dir_off) { + let record = (|| -> Result, Errno> { + let next_dir_off = + dir_off.checked_add(1).ok_or(Errno::EOVERFLOW)?; + let unaligned_len = DIRENT_STRUCT_BYTES_WITHOUT_NAME + .checked_add(entry.name.len()) + .and_then(|len| len.checked_add(1)) + .ok_or(Errno::EOVERFLOW)?; + let len = unaligned_len + .checked_next_multiple_of(DIRENT_ALIGNMENT) + .ok_or(Errno::EOVERFLOW)?; + let next_nbytes = + nbytes.checked_add(len).ok_or(Errno::EOVERFLOW)?; + if next_nbytes > count { + return if nbytes == 0 { + Err(Errno::EINVAL) + } else { + Ok(None) + }; + } + + let ino = entry + .ino_info + .as_ref() + .map_or(Ok(0), |node_info| u64::try_from(node_info.ino)) + .map_err(|_| Errno::EOVERFLOW)?; + let continuation = i64::try_from(next_dir_off) + .map_err(|_| Errno::EOVERFLOW)? + .reinterpret_as_unsigned(); + let record_len = + u16::try_from(len).map_err(|_| Errno::EOVERFLOW)?; + let dirent64 = litebox_common_linux::LinuxDirent64 { + ino, + off: continuation, + len: record_len, + typ: litebox_common_linux::DirentType::from( + entry.file_type.clone(), + ) as u8, + __name: [0; 0], + }; + + let header_addr = dirp + .as_usize() + .checked_add(nbytes) + .ok_or(Errno::EFAULT)?; + let name_addr = header_addr + .checked_add(DIRENT_STRUCT_BYTES_WITHOUT_NAME) + .ok_or(Errno::EFAULT)?; + let zeros_addr = name_addr + .checked_add(entry.name.len()) + .ok_or(Errno::EFAULT)?; + let header = UserPtrMut::from_usize(header_addr); + let name = UserPtrMut::from_usize(name_addr); + let zeros = UserPtrMut::from_usize(zeros_addr); + header + .write_at_offset::(0, dirent64) + .ok_or(Errno::EFAULT)?; + name.write_slice_at_offset::(0, entry.name.as_bytes()) + .ok_or(Errno::EFAULT)?; + let zero_count = len + .checked_sub( + DIRENT_STRUCT_BYTES_WITHOUT_NAME + .checked_add(entry.name.len()) + .ok_or(Errno::EOVERFLOW)?, + ) + .ok_or(Errno::EOVERFLOW)?; + let zero_padding = [0u8; DIRENT_ALIGNMENT]; + let zero_padding = zero_padding + .get(..zero_count) + .ok_or(Errno::EOVERFLOW)?; + zeros + .write_slice_at_offset::(0, zero_padding) + .ok_or(Errno::EFAULT)?; + Ok(Some((next_dir_off, next_nbytes))) + })(); + + match record { + Ok(Some((next_dir_off, next_nbytes))) => { + *dir_off = next_dir_off; + nbytes = next_nbytes; + } + Ok(None) => break, + Err(_) if nbytes != 0 => break, + Err(error) => return Err(error), + } } - break; - } - let dirent64 = litebox_common_linux::LinuxDirent64 { - ino: entry.ino_info.as_ref().map_or(0, |node_info| node_info.ino) as u64, - off: dir_off as u64, - len: len.trunc(), - typ: litebox_common_linux::DirentType::from(entry.file_type.clone()) as u8, - __name: [0; 0], - }; - let hdr_ptr = UserPtrMut::from_usize(dirp.as_usize() + nbytes); - hdr_ptr - .write_at_offset::(0, dirent64) - .ok_or(Errno::EFAULT)?; - let name_ptr = UserPtrMut::from_usize( - hdr_ptr.as_usize() + DIRENT_STRUCT_BYTES_WITHOUT_NAME, - ); - name_ptr - .write_slice_at_offset::(0, entry.name.as_bytes()) - .ok_or(Errno::EFAULT)?; - // set the null terminator and padding - let zeros_len = len - (DIRENT_STRUCT_BYTES_WITHOUT_NAME + entry.name.len()); - name_ptr - .write_slice_at_offset::( - isize::try_from(entry.name.len()).unwrap(), - &vec![0; zeros_len], - ) - .ok_or(Errno::EFAULT)?; - nbytes += len; - dir_off += 1; - } - let _old = self - .global - .litebox - .descriptor_table_mut() - .set_fd_metadata(file, Diroff(dir_off)); - Ok(nbytes) + Ok(nbytes) + }) + .map_err(Errno::from)? }, |_fd| Err(Errno::ENOTDIR), |_fd| Err(Errno::ENOTDIR), |_fd| Err(Errno::ENOTDIR), |_fd| Err(Errno::ENOTDIR), |_fd| Err(Errno::ENOTDIR), + |_fd| Err(Errno::ENOTDIR), )? } } @@ -2594,6 +6489,7 @@ mod tests { use alloc::string::String; use core::cell::Cell; use litebox::fs::Mode; + use litebox::platform::StdioProvider as _; extern crate std; @@ -3012,4 +6908,485 @@ mod tests { Errno::ENOENT ); } + + /// Verify `openat`/`newfstatat`/`faccessat` resolve a relative path against a real `dirfd` + /// (as opposed to `AT_FDCWD`), including across a `dup`'d copy of that `dirfd`, and reject a + /// closed or non-directory `dirfd` the way real Linux does. + #[test] + fn dirfd_relative_resolution_via_real_dirfd() { + use litebox_common_linux::{AccessFlags, AtFlags}; + + let task = crate::syscalls::tests::init_platform(None); + + task.sys_mkdirat(litebox_common_linux::AT_FDCWD, "/dirfd_test", 0o777) + .unwrap(); + let dirfd = task + .sys_openat( + litebox_common_linux::AT_FDCWD, + "/dirfd_test", + litebox::fs::OFlags::RDONLY, + Mode::empty(), + ) + .unwrap(); + let dirfd = i32::try_from(dirfd).unwrap(); + + // openat(dirfd, "inner.txt", ...) creates the file inside the directory the dirfd + // refers to, not relative to CWD (which is still "/"). + let file_fd = task + .sys_openat( + dirfd, + "inner.txt", + litebox::fs::OFlags::CREAT | litebox::fs::OFlags::WRONLY, + Mode::RUSR | Mode::WUSR, + ) + .unwrap(); + task.sys_close(i32::try_from(file_fd).unwrap()).unwrap(); + task.sys_stat("/dirfd_test/inner.txt") + .expect("openat(dirfd, relative) should have created the file under /dirfd_test"); + + // newfstatat(dirfd, "inner.txt", ...) resolves the same way. + task.sys_newfstatat(dirfd, "inner.txt", AtFlags::empty()) + .unwrap(); + + // faccessat(dirfd, "inner.txt", ...) resolves the same way. + task.sys_faccessat(dirfd, "inner.txt", AccessFlags::F_OK, AtFlags::empty()) + .unwrap(); + + // A dup'd dirfd resolves relative paths identically, since the recorded path lives on + // the shared open-file-description entry, not the per-descriptor fd metadata. + let dup_dirfd = task.sys_dup(dirfd, None, None).unwrap(); + let dup_dirfd = i32::try_from(dup_dirfd).unwrap(); + task.sys_faccessat(dup_dirfd, "inner.txt", AccessFlags::F_OK, AtFlags::empty()) + .unwrap(); + task.sys_close(dup_dirfd).unwrap(); + + // A non-directory dirfd (a regular file) is rejected by the underlying filesystem's own + // path resolution once "inner.txt" is joined under it, matching real Linux's ENOTDIR. + let non_dir_fd = task + .sys_openat( + dirfd, + "inner.txt", + litebox::fs::OFlags::RDONLY, + Mode::empty(), + ) + .unwrap(); + let non_dir_fd = i32::try_from(non_dir_fd).unwrap(); + assert_eq!( + task.sys_faccessat(non_dir_fd, "x", AccessFlags::F_OK, AtFlags::empty()) + .unwrap_err(), + Errno::ENOTDIR + ); + task.sys_close(non_dir_fd).unwrap(); + + // An unknown/closed dirfd is rejected with EBADF, not treated as AT_FDCWD. + task.sys_close(dirfd).unwrap(); + assert_eq!( + task.sys_faccessat(dirfd, "inner.txt", AccessFlags::F_OK, AtFlags::empty()) + .unwrap_err(), + Errno::EBADF + ); + } + + /// `POLLIN`, matching real Linux's raw `poll(2)`/`ppoll(2)` event-mask bit. + const POLLIN: i16 = 0x0001; + + /// A real-concurrency stress test for `sys_ppoll`'s lost-wakeup window: a writer thread is + /// released (via a barrier) at the same instant the poller calls `ppoll`, hundreds of times + /// in a row, so that across enough iterations the write lands arbitrarily close to whatever + /// internal state transition `ppoll` goes through between its "not ready yet" check and + /// actually blocking. A poller that checked readiness and *then* registered for + /// notification (the classic lost-wakeup ordering bug) would eventually miss a wakeup here + /// and report a spurious timeout instead of the byte that was actually written. + #[test] + fn test_ppoll_does_not_lose_a_concurrent_wakeup() { + const ITERATIONS: usize = 300; + + let task = crate::syscalls::tests::init_platform(None); + let (rfd_u, wfd_u) = task + .sys_pipe2(litebox::fs::OFlags::empty()) + .expect("pipe2 failed"); + let rfd = i32::try_from(rfd_u).unwrap(); + let wfd = i32::try_from(wfd_u).unwrap(); + + let barrier = std::sync::Arc::new(std::sync::Barrier::new(2)); + + for i in 0..ITERATIONS { + let mut pollfd = litebox_common_linux::Pollfd { + fd: rfd, + events: POLLIN, + revents: 0, + }; + + // Vary the writer's timing relative to the poller across iterations (immediate, + // and after a couple of short delays) so both the poller's initial fast-path check + // and its register-then-block path each get real exercise against a genuinely + // concurrent write, rather than one path dominating simply because a raw pipe write + // is fast. + let writer_delay = core::time::Duration::from_micros(match i % 3 { + 0 => 0, + 1 => 500, + _ => 3_000, + }); + let writer = { + let barrier = std::sync::Arc::clone(&barrier); + task.spawn_clone_for_test(move |task| { + barrier.wait(); + if !writer_delay.is_zero() { + std::thread::sleep(writer_delay); + } + task.sys_write(wfd, &[0x42], None).expect("write failed") + }) + }; + + barrier.wait(); + let ready_count = task + .sys_ppoll( + UserPtrMut::from_ptr(&raw mut pollfd), + 1, + TimeParam::Milliseconds(2000), + None, + 0, + ) + .unwrap_or_else(|e| panic!("iteration {i}: ppoll failed: {e:?}")); + + writer.join().expect("writer thread panicked"); + + assert_eq!( + ready_count, 1, + "iteration {i}: ppoll should report exactly one ready fd, not time out -- a 0 \ + here means the wakeup from the concurrent write was lost" + ); + assert_ne!( + pollfd.revents & POLLIN, + 0, + "iteration {i}: the ready fd should be reported as POLLIN" + ); + + // Drain the byte so the next iteration starts from an empty pipe. + let mut buf = [0u8; 1]; + let n = task.sys_read(rfd, &mut buf, None).expect("read failed"); + assert_eq!(n, 1); + assert_eq!(buf, [0x42]); + } + + let _ = task.sys_close(rfd); + let _ = task.sys_close(wfd); + } + + #[test] + fn automatic_pty_acquisition_publishes_foreground_group_once() { + let task = crate::syscalls::tests::init_platform(None); + let original_group = task.sys_getpgid(0).unwrap(); + let allocate_pty = || { + let master = i32::try_from( + task.sys_open("/dev/ptmx", OFlags::RDWR | OFlags::NOCTTY, Mode::empty()) + .unwrap(), + ) + .unwrap(); + let mut number = u32::MAX; + task.sys_ioctl( + master, + IoctlArg::TIOCGPTN(UserPtrMut::from_ptr(&raw mut number)), + ) + .unwrap(); + let unlocked = 0i32; + task.sys_ioctl( + master, + IoctlArg::TIOCSPTLCK(UserPtr::from_ptr(&raw const unlocked)), + ) + .unwrap(); + (master, number) + }; + let (first_master, first_number) = allocate_pty(); + let (second_master, second_number) = allocate_pty(); + + assert_eq!(task.sys_setsid(), Ok(task.pid)); + let session_group = task.sys_getpgid(0).unwrap(); + assert_ne!(session_group, original_group); + let first_slave = i32::try_from( + task.sys_open( + alloc::format!("/dev/pts/{first_number}"), + OFlags::RDWR, + Mode::empty(), + ) + .unwrap(), + ) + .unwrap(); + let foreground_group = |fd| { + let mut group = -1; + task.sys_ioctl( + fd, + IoctlArg::TIOCGPGRP(UserPtrMut::from_ptr(&raw mut group)), + ) + .unwrap(); + group + }; + assert_eq!( + foreground_group(first_slave), + session_group, + "automatic controlling-terminal acquisition must replace the group captured at ptmx open" + ); + + let selected_group = 6767; + task.sys_ioctl( + first_slave, + IoctlArg::TIOCSPGRP(UserPtr::from_ptr(&raw const selected_group)), + ) + .unwrap(); + let reopened_first = i32::try_from( + task.sys_open( + alloc::format!("/dev/pts/{first_number}"), + OFlags::RDWR, + Mode::empty(), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!( + foreground_group(reopened_first), + selected_group, + "reopening an existing controlling terminal must preserve TIOCSPGRP state" + ); + + let second_slave = i32::try_from( + task.sys_open( + alloc::format!("/dev/pts/{second_number}"), + OFlags::RDWR, + Mode::empty(), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!( + foreground_group(second_slave), + original_group, + "failed acquisition of a different controlling terminal must not publish a new foreground group" + ); + + task.sys_close(second_slave).unwrap(); + task.sys_close(reopened_first).unwrap(); + task.sys_close(first_slave).unwrap(); + task.sys_close(second_master).unwrap(); + task.sys_close(first_master).unwrap(); + } + + #[test] + fn unix98_pty_allocates_unlocks_and_transports_both_directions() { + let task = crate::syscalls::tests::init_platform(None); + assert_eq!(task.sys_setsid(), Ok(task.pid)); + assert_eq!( + task.sys_open("/dev/tty", OFlags::RDWR, Mode::empty()), + Err(Errno::ENXIO), + "setsid must leave the process without a controlling terminal" + ); + let master = i32::try_from( + task.sys_open("/dev/ptmx", OFlags::RDWR | OFlags::NOCTTY, Mode::empty()) + .unwrap(), + ) + .unwrap(); + + let mut number = u32::MAX; + assert_eq!( + task.sys_ioctl( + master, + IoctlArg::TIOCGPTN(UserPtrMut::from_ptr(&raw mut number)), + ), + Ok(0) + ); + assert_eq!( + task.sys_open( + alloc::format!("/dev/pts/{number}"), + OFlags::RDWR | OFlags::NOCTTY, + Mode::empty(), + ), + Err(Errno::EIO), + "the slave must remain locked until TIOCSPTLCK" + ); + + let unlocked = 0i32; + assert_eq!( + task.sys_ioctl( + master, + IoctlArg::TIOCSPTLCK(UserPtr::from_ptr(&raw const unlocked)), + ), + Ok(0) + ); + let slave = i32::try_from( + task.sys_open( + alloc::format!("/dev/pts/{number}"), + OFlags::RDWR | OFlags::NOCTTY, + Mode::empty(), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!( + task.sys_open("/dev/tty", OFlags::RDWR, Mode::empty()), + Err(Errno::ENXIO), + "O_NOCTTY must suppress controlling-terminal acquisition" + ); + + assert_eq!(task.sys_write(master, b"input", None), Ok(5)); + let mut input = [0; 5]; + assert_eq!(task.sys_read(slave, &mut input, None), Ok(5)); + assert_eq!(&input, b"input"); + + assert_eq!(task.sys_write(slave, b"output", None), Ok(6)); + let mut output = [0; 6]; + assert_eq!(task.sys_read(master, &mut output, None), Ok(6)); + assert_eq!(&output, b"output"); + + let controlling = i32::try_from( + task.sys_open( + alloc::format!("/dev/pts/{number}"), + OFlags::RDWR, + Mode::empty(), + ) + .unwrap(), + ) + .unwrap(); + assert_eq!(task.sys_ioctl(controlling, IoctlArg::TIOCSCTTY(0)), Ok(0)); + let mut foreground_pgid = -1; + assert_eq!( + task.sys_ioctl( + controlling, + IoctlArg::TIOCGPGRP(UserPtrMut::from_ptr(&raw mut foreground_pgid)), + ), + Ok(0) + ); + let process_group_id = task.sys_getpgid(0).unwrap(); + assert_eq!( + foreground_pgid, process_group_id, + "TIOCSCTTY must publish the caller's process-owned group as the PTY foreground group" + ); + + // Reproduce the desktop race exactly: another process mutates its own process group after + // this PTY has published its foreground group. Neither the first process's identity nor + // the PTY's foreground state may move, and replaying TIOCSCTTY must be idempotent. + let unrelated = task + .global + .clone() + .new_test_task(task.files.borrow().fs.clone()); + assert_eq!(unrelated.sys_setpgid(0, 7777), Ok(())); + assert_eq!(task.sys_getpgid(0), Ok(process_group_id)); + foreground_pgid = -1; + assert_eq!( + task.sys_ioctl( + controlling, + IoctlArg::TIOCGPGRP(UserPtrMut::from_ptr(&raw mut foreground_pgid)), + ), + Ok(0) + ); + assert_eq!(foreground_pgid, process_group_id); + assert_eq!(task.sys_ioctl(controlling, IoctlArg::TIOCSCTTY(0)), Ok(0)); + foreground_pgid = -1; + assert_eq!( + task.sys_ioctl( + controlling, + IoctlArg::TIOCGPGRP(UserPtrMut::from_ptr(&raw mut foreground_pgid)), + ), + Ok(0) + ); + assert_eq!(foreground_pgid, process_group_id); + let tty = i32::try_from( + task.sys_open("/dev/tty", OFlags::RDWR, Mode::empty()) + .unwrap(), + ) + .unwrap(); + assert_eq!(task.sys_write(tty, b"alias", None), Ok(5)); + let mut alias = [0; 5]; + assert_eq!(task.sys_read(master, &mut alias, None), Ok(5)); + assert_eq!(&alias, b"alias"); + + let requested = litebox_common_linux::Winsize { + row: 37, + col: 111, + xpixel: 777, + ypixel: 333, + }; + task.sys_ioctl( + master, + IoctlArg::TIOCSWINSZ(UserPtr::from_ptr(&raw const requested)), + ) + .unwrap(); + let mut observed = litebox_common_linux::Winsize { + row: 0, + col: 0, + xpixel: 0, + ypixel: 0, + }; + task.sys_ioctl( + slave, + IoctlArg::TIOCGWINSZ(UserPtrMut::from_ptr(&raw mut observed)), + ) + .unwrap(); + assert_eq!( + (observed.row, observed.col, observed.xpixel, observed.ypixel), + (37, 111, 777, 333) + ); + + task.sys_close(master).unwrap(); + assert_eq!( + task.sys_open( + alloc::format!("/dev/pts/{number}"), + OFlags::RDWR, + Mode::empty(), + ), + Err(Errno::ENOENT), + "closing the master must retire the slave pathname" + ); + task.sys_close(tty).unwrap(); + task.sys_close(controlling).unwrap(); + task.sys_close(slave).unwrap(); + } + + #[test] + fn tiocgwinsz_never_regresses_to_the_old_hardcoded_20x20() { + // Regression test for a genuine echo-wrapping bug: `TIOCGWINSZ` used to unconditionally + // report a hardcoded 20x20 window, regardless of the real terminal size. Guests' own + // line editors (e.g. `ash`'s `lineedit.c`) query this to decide the column width at + // which to wrap their own echoed-input redisplay, so a fake 20-column width caused + // spurious wraps in the echo of typed input well before the real terminal (which may be + // 80, 120, or wider) would ever need to wrap. + // + // This calls `stdio_ioctl` directly rather than `sys_ioctl`: the latter's stdio path + // additionally gates on `Platform::is_a_tty`, which `cargo test`'s captured stdout + // makes false in the common case, so going through it here would just assert `ENOTTY` + // rather than exercising the fallback logic under test. + let task = crate::syscalls::tests::init_platform(None); + + let mut ws = litebox_common_linux::Winsize { + row: 0xFFFF, + col: 0xFFFF, + xpixel: 0xFFFF, + ypixel: 0xFFFF, + }; + let ws_ptr = UserPtrMut::from_usize((&raw mut ws).expose_provenance()); + assert_eq!( + task.stdio_ioctl(StdioStream::Stdout, &IoctlArg::TIOCGWINSZ(ws_ptr)), + Ok(0) + ); + assert_ne!( + (ws.row, ws.col), + (20, 20), + "must not regress to the old hardcoded 20x20 fake window size" + ); + // The test platform has no real terminal backing its (captured) stdout in the common + // `cargo test` case, so this asserts the 80x24 fallback; on the rare host where stdout + // genuinely is a tty (e.g. an interactive `cargo test -- --nocapture`), it instead + // asserts the handler faithfully reported that real size. + match task.global.platform.tty_window_size() { + None => assert_eq!( + (ws.row, ws.col), + (24, 80), + "must fall back to the traditional 80x24 default when the platform has no real \ + terminal size" + ), + Some(real) => assert_eq!( + (ws.row, ws.col), + real, + "must report the platform's real terminal size when available" + ), + } + } } diff --git a/litebox_shim_linux/src/syscalls/inotify.rs b/litebox_shim_linux/src/syscalls/inotify.rs new file mode 100644 index 0000000000..fdf8f55f0b --- /dev/null +++ b/litebox_shim_linux/src/syscalls/inotify.rs @@ -0,0 +1,187 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! `inotify(7)`: a file-change notification queue. +//! +//! Real Linux `inotify_init1(2)` creates a dedicated fd backed by an event queue and a +//! watch-descriptor table; `inotify_add_watch(2)`/`inotify_rm_watch(2)` manage the watch table, +//! and a `read(2)` on the fd drains queued [`InotifyEvent`] records (variable-length: a fixed +//! header plus an optional NUL-padded name for watches on a directory). +//! +//! This shim implements the fd, watch-descriptor bookkeeping, and queue/read/poll semantics in +//! full -- every part of the ABI a caller can observe through the fd itself -- but does not yet +//! wire any filesystem-change producer that would actually push events onto the queue (no +//! directory-entry-create/delete/rename/etc hook exists in the in-memory or tar-backed +//! filesystem layers today). A watch is accepted, given a real, uniquely-allocated watch +//! descriptor, and remembered; it simply never fires. This is honest, spec-correct behavior for +//! a filesystem that never changes out from under a watch (`read` blocks / reports `EAGAIN` +//! exactly as it would on real Linux with no pending events, `EINVAL` on a bad watch descriptor +//! removal, etc) -- callers that only need `inotify_init1` to succeed and watches to be +//! nominally accepted and removable (e.g. D-Bus's `dbus-daemon`, whose own inotify use is +//! confined to optional config-reload watching and does not block its session-bus readiness +//! path on any watch actually firing) work correctly. A real change-notification producer is a +//! separate, larger undertaking (hooking every mutating filesystem syscall in every backing +//! store) and is intentionally out of scope here absent evidence a caller needs delivered events +//! rather than just a working fd. + +use alloc::{collections::BTreeMap, vec::Vec}; + +use litebox::{ + event::{Events, IOPollable, observer::Observer, polling::Pollee, polling::TryOpError}, + fs::OFlags, +}; +use litebox_common_linux::{InotifyInitFlags, InotifyMask, errno::Errno}; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; + +pub(crate) struct InotifySubsystem(core::marker::PhantomData); +impl FdEnabledSubsystem for InotifySubsystem { + type Entry = InotifyFile; +} +impl FdEnabledSubsystemEntry for InotifyFile {} + +/// One queued, ready-to-read record: the fixed `InotifyEvent` header plus its optional name. +struct QueuedEvent { + wd: i32, + mask: u32, + cookie: u32, + name: Vec, +} + +struct Inner { + /// Next watch descriptor to hand out. Real `inotify_add_watch` returns small increasing + /// integers starting at 1; we do the same rather than reusing a removed wd immediately, so a + /// stale reference to a just-removed watch reliably misses rather than aliasing a new one. + next_wd: i32, + /// Live watches: wd -> mask (path is not retained; nothing yet needs to map a delivered + /// event's source path back to a watch beyond the wd itself, and holding a path here would + /// only matter once a real change-notification producer exists). + watches: BTreeMap, + queue: alloc::collections::VecDeque, +} + +pub(crate) struct InotifyFile { + inner: litebox::sync::Mutex, + pollee: Pollee, + status: core::sync::atomic::AtomicU32, +} + +impl InotifyFile { + fn new(flags: InotifyInitFlags) -> Self { + let mut status = OFlags::RDONLY; + status.set(OFlags::NONBLOCK, flags.contains(InotifyInitFlags::NONBLOCK)); + Self { + inner: litebox::sync::Mutex::new(Inner { + next_wd: 1, + watches: BTreeMap::new(), + queue: alloc::collections::VecDeque::new(), + }), + pollee: Pollee::new(), + status: core::sync::atomic::AtomicU32::new(status.bits()), + } + } + + /// `inotify_add_watch(2)`: accept and remember a watch, returning its descriptor. Combining + /// (`IN_MASK_ADD`) and mask replacement both apply to a watch already registered against the + /// same target; since no path is retained per-watch here (see `Inner::watches`' doc comment), + /// and no caller-visible way exists yet to prove two `add_watch` calls named the same path, + /// every call allocates a fresh watch descriptor -- correct for the common case (watching + /// distinct paths) and only observably different from Linux when a caller re-watches an + /// identical path expecting the same wd back, which no exercised caller in this codebase + /// does today. + pub(crate) fn add_watch(&self, mask: InotifyMask) -> Result { + if mask.intersects(InotifyMask::empty().complement()) && mask.is_empty() { + return Err(Errno::EINVAL); + } + let mut inner = self.inner.lock(); + let wd = inner.next_wd; + inner.next_wd = inner.next_wd.checked_add(1).ok_or(Errno::ENOSPC)?; + inner.watches.insert(wd, mask); + Ok(wd) + } + + /// `inotify_rm_watch(2)`: drop a watch. Real Linux additionally queues a synthetic + /// `IN_IGNORED` event on removal; since no queue consumer here has yet needed to observe + /// that terminal marker (no events are ever delivered for a watch in the first place -- see + /// this module's doc comment), it is not queued. + pub(crate) fn rm_watch(&self, wd: i32) -> Result<(), Errno> { + let mut inner = self.inner.lock(); + if inner.watches.remove(&wd).is_none() { + return Err(Errno::EINVAL); + } + Ok(()) + } + + pub(crate) fn read( + &self, + cx: &litebox::event::wait::WaitContext<'_, Platform>, + buf: &mut [u8], + ) -> Result { + self.pollee + .wait(cx, self.is_nonblocking(), Events::IN, || { + let mut inner = self.inner.lock(); + let Some(front) = inner.queue.front() else { + return Err(TryOpError::::TryAgain); + }; + let name_field_len = if front.name.is_empty() { + 0 + } else { + // Real inotify pads the name field to a multiple of the fixed-header + // alignment (`sizeof(struct inotify_event)` == 16 bytes) with trailing NULs. + front.name.len().div_ceil(16) * 16 + }; + let total = 16 + name_field_len; + if buf.len() < total { + // A too-small buffer leaves the event queued, matching real `EINVAL` from + // `read(2)` on an inotify fd whose buffer cannot hold even one event. + return Err(TryOpError::Other(Errno::EINVAL)); + } + let event = inner.queue.pop_front().unwrap(); + buf[0..4].copy_from_slice(&event.wd.to_ne_bytes()); + buf[4..8].copy_from_slice(&event.mask.to_ne_bytes()); + buf[8..12].copy_from_slice(&event.cookie.to_ne_bytes()); + buf[12..16].copy_from_slice(&(name_field_len as u32).to_ne_bytes()); + if name_field_len > 0 { + buf[16..16 + event.name.len()].copy_from_slice(&event.name); + buf[16 + event.name.len()..total].fill(0); + } + if inner.queue.is_empty() { + drop(inner); + self.pollee.notify_observers(Events::empty()); + } + Ok(total) + }) + .map_err(Errno::from) + } + + super::common_functions_for_file_status!(); + + fn is_nonblocking(&self) -> bool { + self.get_status().contains(OFlags::NONBLOCK) + } +} + +impl IOPollable for InotifyFile { + fn check_io_events(&self) -> Events { + let inner = self.inner.lock(); + if inner.queue.is_empty() { + Events::empty() + } else { + Events::IN + } + } + + fn register_observer(&self, observer: alloc::sync::Weak>, mask: Events) { + self.pollee.register_observer(observer, mask); + } + + fn unregister_observer(&self, observer: alloc::sync::Weak>) { + self.pollee.unregister_observer(observer); + } +} + +impl crate::GlobalState { + pub(crate) fn create_linux_inotify(&self, flags: InotifyInitFlags) -> InotifyFile { + InotifyFile::new(flags) + } +} diff --git a/litebox_shim_linux/src/syscalls/misc.rs b/litebox_shim_linux/src/syscalls/misc.rs index ee546e53eb..8d88502d00 100644 --- a/litebox_shim_linux/src/syscalls/misc.rs +++ b/litebox_shim_linux/src/syscalls/misc.rs @@ -8,7 +8,7 @@ use crate::{ShimFS, ShimPlatform, Task}; use litebox::{platform::Instant as _, utils::TruncateExt as _}; use litebox_common_linux::errno::Errno; -use litebox_common_linux::user_pointers::UserPtrMut; +use litebox_common_linux::user_pointers::{UserPtr, UserPtrMut}; impl Task { /// Handle syscall `getrandom`. @@ -55,12 +55,14 @@ const fn to_fixed_size_array(s: &str) -> [u8; N] { arr } const SYS_INFO: litebox_common_linux::Utsname = litebox_common_linux::Utsname { - sysname: to_fixed_size_array::<65>("LiteBox"), + sysname: to_fixed_size_array::<65>("Linux"), nodename: to_fixed_size_array::<65>("litebox"), release: to_fixed_size_array::<65>("5.11.0"), // libc seems to expect this to be not too old version: to_fixed_size_array::<65>("5.11.0"), #[cfg(target_arch = "x86_64")] machine: to_fixed_size_array::<65>("x86_64"), + #[cfg(target_arch = "aarch64")] + machine: to_fixed_size_array::<65>("aarch64"), domainname: to_fixed_size_array::<65>(""), }; @@ -81,9 +83,14 @@ impl Task { uptime: now.duration_since(&self.global.boot_time).as_secs().trunc(), // TODO: Populate these fields with actual values loads: [0; 3], - #[cfg(target_arch = "x86_64")] - totalram: 4 * 1024 * 1024 * 1024, - freeram: 2 * 1024 * 1024 * 1024, + // Shared with `/proc/meminfo` (`litebox::fs::proc`) so `free` -- which reads + // totalram/freeram from this syscall but Cached/MemAvailable/SReclaimable from + // `/proc/meminfo` -- can't observe the two sources drifting apart. Previously this + // field was `#[cfg(target_arch = "x86_64")]`-only, so `..Default::default()` silently + // left it 0 on aarch64 (this host's own architecture): `free`'s "used" column + // underflowed since `freeram` was nonzero while `totalram` was 0. + totalram: litebox::fs::proc::SYNTHETIC_TOTAL_RAM_BYTES.trunc(), + freeram: litebox::fs::proc::SYNTHETIC_FREE_RAM_BYTES.trunc(), sharedram: 0, // We don't support shared memory bufferram: 0, totalswap: 0, @@ -95,6 +102,24 @@ impl Task { ..Default::default() } } + + /// Handle syscall `getrusage`. + /// + /// LiteBox keeps no per-process CPU or fault accounting to report, so every + /// counter is zero except `ru_maxrss`, which mirrors the synthetic + /// resident-set size `/proc//status` reports (in kilobytes, as + /// `getrusage(2)` specifies on Linux), so the two sources cannot be observed + /// drifting apart. `who` (`RUSAGE_SELF`/`_CHILDREN`/`_THREAD`) makes no + /// difference here: there is one accounting target to report. This is enough + /// for `process.cpuUsage()`/`process.resourceUsage()` to return (zeroed) + /// values instead of throwing `ENOSYS`. + pub(crate) fn sys_getrusage(&self, _who: i32) -> litebox_common_linux::Rusage { + litebox_common_linux::Rusage { + // `/proc//status` reports `VmRSS: 1024 kB`; keep the two in step. + ru_maxrss: 1024, + ..Default::default() + } + } } const _LINUX_CAPABILITY_VERSION_1: u32 = 0x19980330; @@ -159,6 +184,40 @@ impl Task { } } } + + /// Handle syscall `capset`. + /// + /// LiteBox doesn't support capabilities (see `sys_capget`): every process + /// is reported as holding an empty capability set, so this only accepts + /// a request that is consistent with that -- an unversioned/empty + /// request, or an explicit request for the empty set (effective, + /// permitted, and inheritable all zero, which is exactly what a + /// `setpriv --reuid`/`--regid` privilege drop asks for after + /// `PR_SET_KEEPCAPS`: it is not trying to grant itself anything, only + /// making the subsequent uid/gid change not silently clear a + /// permitted set that -- here -- was already empty). A request for any + /// actual capability bit is refused with `EPERM`, matching the real + /// kernel's response to a process trying to set a capability it does + /// not already hold. + pub(crate) fn sys_capset( + &self, + header: UserPtr, + data: Option>, + ) -> Result<(), Errno> { + let hdr = header.read_at_offset::(0).ok_or(Errno::EFAULT)?; + match hdr.version { + _LINUX_CAPABILITY_VERSION_1 | _LINUX_CAPABILITY_VERSION_2 | _LINUX_CAPABILITY_VERSION_3 => {} + _ => return Err(Errno::EINVAL), + } + let Some(data_ptr) = data else { + return Ok(()); + }; + let requested = data_ptr.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if requested.effective != 0 || requested.permitted != 0 || requested.inheritable != 0 { + return Err(Errno::EPERM); + } + Ok(()) + } } #[cfg(test)] diff --git a/litebox_shim_linux/src/syscalls/mm.rs b/litebox_shim_linux/src/syscalls/mm.rs index d62999b85a..65600c428a 100644 --- a/litebox_shim_linux/src/syscalls/mm.rs +++ b/litebox_shim_linux/src/syscalls/mm.rs @@ -6,18 +6,18 @@ use alloc::collections::{BTreeMap, BTreeSet}; use litebox::{ + fd::EntryHandle, mm::linux::{MappingError, PAGE_SIZE, PageRange}, platform::{ PageManagementProvider, RawConstPointer, page_mgmt::{FixedAddressBehavior, MemoryRegionPermissions}, }, }; -use litebox_common_linux::{MRemapFlags, MapFlags, ProtFlags, errno::Errno}; +use litebox_common_linux::{ + MRemapFlags, MapFlags, ProtFlags, errno::Errno, loader::TRAMPOLINE_GUEST_TP_SLOT_OFFSET, +}; -use crate::ShimFS; -use crate::ShimPlatform; -use crate::Task; -use crate::UserPtrMut; +use crate::{ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut}; use litebox::utils::TruncateExt as _; use object::elf::{ET_DYN, FileHeader64, PT_LOAD, ProgramHeader64}; use object::endian::LittleEndian; @@ -25,15 +25,136 @@ use object::endian::LittleEndian; #[cfg(not(target_pointer_width = "64"))] compile_error!("ELF patching code assumes 64-bit pointers (u64 <-> usize is lossless)"); +/// This module publishes the guest thread-pointer offset into the same +/// trampoline word the rewriter's gates read. Mirrors the identical assertion +/// in `crate::loader::elf`, which holds the loader path to the same constant: +/// drift here would make the mmap path write the offset into the middle of an +/// instruction instead of into the slot the gates read. +const _: () = assert!( + TRAMPOLINE_GUEST_TP_SLOT_OFFSET == litebox_syscall_rewriter::TRAMPOLINE_GUEST_TP_SLOT_OFFSET +); + const ENDIAN: LittleEndian = LittleEndian; +/// Makes freshly-written code visible to instruction fetch before it is +/// executed. +/// +/// x86-64 guarantees instruction/data cache coherency in hardware, so this is +/// a no-op there. AArch64 does not: a core that just wrote through the data +/// cache is not guaranteed to see those bytes if it (or another core) fetches +/// the same address as an instruction, until the corresponding cache lines are +/// explicitly cleaned and invalidated. Every write this module makes into +/// guest-executed memory -- the rewriter's patched code, the trampoline stubs, +/// the trap-fallback bytes -- needs this called over the written range before +/// the mapping goes back to executable, or the guest can intermittently +/// execute stale (pre-patch, or partially-written) instructions. +/// +/// This runs the same `dc cvau`/`ic ivau`/barrier sequence +/// `__builtin___clear_cache` generates on AArch64 (see LLVM compiler-rt's +/// `clear_cache.c`), reading the actual cache line sizes from `CTR_EL0` rather +/// than assuming a fixed one. These instructions are permitted from EL0 +/// (unprivileged) code on Linux, which sets `SCTLR_EL1.UCI` for exactly this +/// purpose -- every userspace AArch64 JIT relies on the same permission. +/// +/// Darwin is the exception, and it is not a matter of degree: `SCTLR_EL1.UCI` +/// is set there too, so `dc cvau`/`ic ivau` run fine, but `SCTLR_EL1.UCT` is +/// *not*, so reading `CTR_EL0` raises an illegal-instruction trap. Measured on +/// an Apple M3 Pro: a bare C program doing `mrs x0, ctr_el0` dies with `SIGILL`, +/// while the same program's `dc cvau`/`ic ivau` sequence returns normally. +/// Since this function is the choke point every transition to `PROT_EXEC` +/// passes through, that trap made it impossible to give a guest an executable +/// page at all. Darwin therefore goes through `sys_icache_invalidate`, Apple's +/// own supported entry point for this, which performs the same sequence (plus +/// any chip-specific work) without needing the line sizes in userspace. +#[cfg(all(target_arch = "aarch64", target_vendor = "apple"))] +fn clear_icache_range(start: usize, len: usize) { + if len == 0 { + return; + } + // SAFETY: `start` addresses `len` bytes of the caller's own mapping, which + // is what this call requires; invalidation cannot fault or alter contents. + unsafe { sys_icache_invalidate(start as *mut core::ffi::c_void, len) }; +} + +// Instruction-cache invalidation from Darwin's `libkern/OSCacheControl.h`. +// Declared here rather than reused from the macOS platform crate because that +// crate is a dev-dependency of this one, reachable only from tests. +#[cfg(all(target_arch = "aarch64", target_vendor = "apple"))] +unsafe extern "C" { + fn sys_icache_invalidate(start: *mut core::ffi::c_void, len: usize); +} + +#[cfg(all(target_arch = "aarch64", not(target_vendor = "apple")))] +fn clear_icache_range(start: usize, len: usize) { + if len == 0 { + return; + } + let end = start + len; + + // SAFETY: `ctr_el0` is readable from EL0 on every host reaching this arm; + // the one that traps instead (Darwin) is handled above. + let ctr_el0: u64; + unsafe { + core::arch::asm!("mrs {ctr}, ctr_el0", ctr = out(reg) ctr_el0, options(nomem, nostack, preserves_flags)); + } + // CTR_EL0.DminLine (bits [19:16]) / IminLine (bits [3:0]): log2 of the + // minimum cache line, in words. A line is therefore `4 << field` bytes. + let dcache_line = 4usize << ((ctr_el0 >> 16) & 0xF); + let icache_line = 4usize << (ctr_el0 & 0xF); + + // Clean each dirty D-cache line covering the range to the point of + // unification, so the I-cache fetch below can see the new bytes. + let mut addr = start & !(dcache_line - 1); + while addr < end { + // SAFETY: `addr` is a valid address within the caller's own writable + // mapping (the range just written); `dc cvau` only cleans a cache + // line, it cannot fault or corrupt memory. + unsafe { + core::arch::asm!("dc cvau, {addr}", addr = in(reg) addr, options(nostack, preserves_flags)); + } + addr += dcache_line; + } + // SAFETY: a data synchronization barrier with no other preconditions. + unsafe { + core::arch::asm!("dsb ish", options(nostack, preserves_flags)); + } + + // Invalidate each I-cache line covering the range to the point of + // unification, forcing the next fetch to reload from memory. + let mut addr = start & !(icache_line - 1); + while addr < end { + // SAFETY: as above, for the instruction cache. + unsafe { + core::arch::asm!("ic ivau, {addr}", addr = in(reg) addr, options(nostack, preserves_flags)); + } + addr += icache_line; + } + // SAFETY: a data synchronization barrier followed by an instruction + // synchronization barrier, ensuring the invalidation is complete and any + // speculatively-fetched stale instructions are discarded before this + // function returns. + unsafe { + core::arch::asm!("dsb ish", "isb", options(nostack, preserves_flags)); + } +} + +#[cfg(not(target_arch = "aarch64"))] +fn clear_icache_range(_start: usize, _len: usize) {} + /// Per-fd state for the shim's runtime ELF syscall rewriter. /// /// Tracks base address and trampoline write cursor for each ELF file that /// has executable segments mapped via `do_mmap_file()`. +#[derive(Clone)] pub(crate) struct ElfPatchState { /// Whether this file is already pre-patched (trampoline magic found at file tail). pre_patched: bool, + /// `e_machine` from the ELF header. The runtime rewriter + /// (`patch_code_segment` / `trap_all_syscalls_in_code`) decodes x86-64 + /// instructions only; recording the machine lets the patching path refuse + /// to run that decoder over any other architecture's code instead of + /// silently reinterpreting (and corrupting) it. + machine: u16, /// For pre-patched binaries: file offset and size of the trampoline data. trampoline_file_offset: u64, trampoline_file_size: usize, @@ -62,6 +183,45 @@ pub(crate) struct ElfPatchState { /// Per-process collection of ELF patching state, keyed by fd number. pub(crate) type ElfPatchCache = BTreeMap; +/// A guest ELF image recorded at first-map time, for fault symbolization. +/// +/// Deliberately not part of [`ElfPatchState`]: that cache is keyed by a +/// reusable raw fd and dropped when the fd closes, while a dynamic linker +/// closes each library's fd as soon as its segments are mapped -- long before +/// any fault that needs symbolizing. Entries here live for the process. +pub(crate) struct GuestImage { + /// Lowest mapped guest address covered by the image's PT_LOAD segments. + lo: usize, + /// One past the highest mapped guest address covered by the image. + hi: usize, + /// The load bias: guest address minus ELF vaddr. `addr - base` is the + /// image-relative address `llvm-symbolizer` resolves against the file. + base: usize, + /// The absolute guest path the image was opened with. + path: alloc::string::String, +} + +/// A shared file mapping retained for copy-back and remap bookkeeping. +pub(crate) struct SharedFileMapping { + start: usize, + len: usize, + offset: usize, + writable: bool, + file: EntryHandle, +} + +impl Clone for SharedFileMapping { + fn clone(&self) -> Self { + Self { + start: self.start, + len: self.len, + offset: self.offset, + writable: self.writable, + file: self.file.clone(), + } + } +} + #[inline] fn align_up(addr: usize, align: usize) -> usize { debug_assert!(align.is_power_of_two()); @@ -83,6 +243,7 @@ impl Task { prot: ProtFlags, flags: MapFlags, ensure_space_after: bool, + shared_futex_backing: Option<(litebox::mm::linux::SharedFutexBacking, usize)>, op: impl FnOnce(UserPtrMut) -> Result, ) -> Result, MappingError> { litebox_common_linux::mm::do_mmap( @@ -92,6 +253,7 @@ impl Task { prot, flags, ensure_space_after, + shared_futex_backing, op, ) } @@ -105,7 +267,18 @@ impl Task { flags: MapFlags, ) -> Result, MappingError> { let op = |_| Ok(0); - self.do_mmap(suggested_addr, len, prot, flags, false, op) + let shared_futex_backing = flags + .contains(MapFlags::MAP_SHARED) + .then(|| (litebox::mm::linux::SharedFutexBacking::new(), 0)); + self.do_mmap( + suggested_addr, + len, + prot, + flags, + false, + shared_futex_backing, + op, + ) } fn do_mmap_file( @@ -116,16 +289,31 @@ impl Task { flags: MapFlags, fd: i32, offset: usize, + shared_futex_backing: Option<(litebox::mm::linux::SharedFutexBacking, usize)>, ) -> Result, MappingError> { let is_exec = prot.contains(ProtFlags::PROT_EXEC); // Perform the normal mmap first (CoW or memcpy fallback). - let result = if let Some(cow_result) = - self.try_cow_mmap_file(suggested_addr, len, &prot, &flags, fd, offset) - { + let result = if let Some(cow_result) = self.try_cow_mmap_file( + suggested_addr, + len, + &prot, + &flags, + fd, + offset, + shared_futex_backing, + ) { cow_result? } else { - self.do_mmap_file_memcpy(suggested_addr, len, prot, flags, fd, offset)? + self.do_mmap_file_memcpy( + suggested_addr, + len, + prot, + flags, + fd, + offset, + shared_futex_backing, + )? }; // Runtime syscall rewriting: patch PROT_EXEC segments in-place. @@ -146,7 +334,7 @@ impl Task { self.init_elf_patch_state(fd, result.as_usize(), offset); // Track non-exec file mappings so we can patch them if they later // gain PROT_EXEC via mprotect. - let mut cache = self.global.elf_patch_cache.lock(); + let mut cache = self.process().elf_patch_cache.lock(); if let Some(state) = cache.get_mut(&fd) { let mapping_key = (result.as_usize(), len); // Overlapping entries are safe here: file_mappings is only used @@ -173,7 +361,11 @@ impl Task { flags: &MapFlags, fd: i32, offset: usize, + shared_futex_backing: Option<(litebox::mm::linux::SharedFutexBacking, usize)>, ) -> Option, MappingError>> { + if shared_futex_backing.is_some() { + return None; + } if !len.is_multiple_of(PAGE_SIZE) { return None; } @@ -194,6 +386,7 @@ impl Task { |_| None, |_| None, |_| None, + |_| None, ) .ok()??; @@ -266,6 +459,31 @@ impl Task { } } + /// Reads backing-file bytes for mmap initialization without consulting canonical shared pages. + /// The initialization protocol itself serializes and publishes those pages; routing this read + /// through `sys_read` would recurse into the canonical overlay and wait on its own claim. + fn read_file_for_mmap( + &self, + fd: i32, + buffer: &mut [u8], + offset: usize, + ) -> Result { + let raw_fd = usize::try_from(fd).map_err(|_| Errno::EBADF)?; + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |typed| files.fs.read(typed, buffer, Some(offset)).map_err(Errno::from), + |_| Err(Errno::EINVAL), + |_| Err(Errno::EINVAL), + |_| Err(Errno::EINVAL), + |_| Err(Errno::EINVAL), + |_| Err(Errno::EINVAL), + |_| Err(Errno::EINVAL), + ) + .flatten() + } + /// Fallback mmap implementation using page-by-page memcpy, for files where the CoW attempt /// fails (either due to lack of support on platform, or non-static-backed data, etc.) fn do_mmap_file_memcpy( @@ -276,34 +494,62 @@ impl Task { flags: MapFlags, fd: i32, offset: usize, + shared_futex_backing: Option<(litebox::mm::linux::SharedFutexBacking, usize)>, ) -> Result, MappingError> { let op = |ptr: UserPtrMut| -> Result { // Note a malicious user may unmap ptr while we are reading. // `sys_read` does not handle page faults, so we need to use a // temporary buffer to read the data from fs (without worrying page // faults) and write it to the user buffer with page fault handling. - let mut file_offset = offset; - let mut buffer = [0; PAGE_SIZE]; - let mut copied = 0; - while copied < len { - let size = - self.sys_read(fd, &mut buffer, Some(file_offset)) + let mut initialize = + |relative: core::ops::Range| -> Result<(), MappingError> { + // A canonical backing extent can survive a failed first mmap. Clear the claimed + // initialization gap before reading so bytes beyond EOF cannot retain a partial + // earlier attempt. + let zeroes = [0; PAGE_SIZE]; + let mut cleared = 0; + while cleared < relative.len() { + let size = (relative.len() - cleared).min(PAGE_SIZE); + ptr.copy_from_slice::(relative.start + cleared, &zeroes[..size]) + .ok_or(MappingError::Io(Errno::EFAULT.into()))?; + cleared += size; + } + + let mut file_offset = offset + relative.start; + let mut buffer = [0; PAGE_SIZE]; + let mut copied = 0; + while copied < relative.len() { + let requested = (relative.len() - copied).min(PAGE_SIZE); + let size = self + .read_file_for_mmap(fd, &mut buffer[..requested], file_offset) .map_err(|e| match e { Errno::EBADF => MappingError::BadFD(fd), Errno::EISDIR => MappingError::NotAFile, Errno::EACCES => MappingError::NotForReading, - _ => unimplemented!(), + other => MappingError::Io(other.into()), })?; - if size == 0 { - break; + if size == 0 { + break; + } + ptr.copy_from_slice::(relative.start + copied, &buffer[..size]) + .ok_or(MappingError::Io(Errno::EFAULT.into()))?; + copied += size; + file_offset += size; } - // ptr is a valid pointer returned by do_mmap. - ptr.copy_from_slice::(copied, &buffer[..size]) - .unwrap(); - copied += size; - file_offset += size; + Ok(()) + }; + + if let Some((backing, backing_offset)) = shared_futex_backing { + self.global.platform.initialize_shared_pages( + backing.identity(), + backing_offset, + len, + &mut initialize, + )?; + } else { + initialize(0..len)?; } - Ok(copied) + Ok(len) }; let fixed_addr = flags.intersects(MapFlags::MAP_FIXED | MapFlags::MAP_FIXED_NOREPLACE); self.do_mmap( @@ -314,6 +560,7 @@ impl Task { // Note we need to ensure that the space after the mapping is available // so that we could load trampoline code right after the mapping. offset == 0 && !fixed_addr, + shared_futex_backing, op, ) } @@ -333,18 +580,27 @@ impl Task { return Err(Errno::EINVAL); } - // MAP_SHARED is partially supported: - // - Anonymous shared mappings are fully supported (no backing file concerns). - // Note: since fork is not yet supported, shared anonymous mappings behave - // identically to private ones (no cross-process sharing occurs). - // - File-backed shared mappings are read-only: writable permission is rejected - // upfront and cannot be added later via mprotect, because writes cannot be - // propagated back to the underlying file. - if flags.contains(MapFlags::MAP_SHARED) + // MAP_SHARED file mappings use one stable device/inode backing identity. On platforms with + // canonical shared pages, simultaneous aliases therefore observe the same bytes; writable + // mappings are retained for copy-back on platforms that still use the default allocator. + // `/dev/fb0` remains a specialized live-pixel-store mapping handled separately below. + let writable_shared_file = flags.contains(MapFlags::MAP_SHARED) && prot.contains(ProtFlags::PROT_WRITE) + && !flags.contains(MapFlags::MAP_ANONYMOUS); + let fb0_shared_mapping = writable_shared_file && self.raw_fd_is_fb0(fd); + let shared_file_mapping_handle = if flags.contains(MapFlags::MAP_SHARED) && !flags.contains(MapFlags::MAP_ANONYMOUS) + && !fb0_shared_mapping + { + self.raw_fd_file_handle(fd) + } else { + None + }; + if writable_shared_file + && !fb0_shared_mapping + && shared_file_mapping_handle.is_none() { - todo!("MAP_SHARED with PROT_WRITE on file-backed mappings is not supported"); + return Err(Errno::EINVAL); } if flags.intersects( @@ -357,7 +613,8 @@ impl Task { | MapFlags::MAP_HUGE_2MB | MapFlags::MAP_HUGE_1GB, ) { - todo!("Unsupported flags {:?}", flags); + log_unsupported!("mmap flags {:?}", flags); + return Err(Errno::EINVAL); } let aligned_len = align_up(len, PAGE_SIZE); @@ -369,20 +626,367 @@ impl Task { } let suggested_addr = if addr == 0 { None } else { Some(addr) }; - if flags.contains(MapFlags::MAP_ANONYMOUS) { + let fixed_replace = flags.contains(MapFlags::MAP_FIXED) + && !flags.contains(MapFlags::MAP_FIXED_NOREPLACE); + if fixed_replace { + self.flush_shared_file_mappings(addr, aligned_len)?; + if let Some(fb) = self.global.framebuffer.as_ref() { + fb.clear_guest_mapping_overlapping(addr, aligned_len); + } + } + let shared_file_futex_backing = if flags.contains(MapFlags::MAP_SHARED) + && !flags.contains(MapFlags::MAP_ANONYMOUS) + && !fb0_shared_mapping + { + self.raw_fd_shared_futex_backing(fd) + .map(|backing| (backing, offset)) + } else { + None + }; + let result = if flags.contains(MapFlags::MAP_ANONYMOUS) { self.do_mmap_anonymous(suggested_addr, aligned_len, prot, flags) + } else if fb0_shared_mapping { + self.do_mmap_framebuffer(suggested_addr, aligned_len, prot, flags, offset) } else { - self.do_mmap_file(suggested_addr, aligned_len, prot, flags, fd, offset) + self.do_mmap_file( + suggested_addr, + aligned_len, + prot, + flags, + fd, + offset, + shared_file_futex_backing, + ) + } + .map_err(Errno::from)?; + + self.record_mapped(result.as_usize(), aligned_len); + if let (Some(file), Some((_, _))) = + (shared_file_mapping_handle, shared_file_futex_backing) + { + self.files + .borrow() + .shared_file_mappings + .lock() + .push(SharedFileMapping { + start: result.as_usize(), + len: aligned_len, + offset, + writable: writable_shared_file, + file, + }); + } + Ok(result) + } + + /// Whether raw fd `fd` names `/dev/fb0` (see [`Self::is_fb0`]); `false` for anything that + /// isn't an open fs-backend fd. + fn raw_fd_is_fb0(&self, fd: i32) -> bool { + let Ok(raw_fd) = u32::try_from(fd).and_then(usize::try_from) else { + return false; + }; + let files = self.files.borrow(); + files + .run_on_raw_fd( + raw_fd, + |typed_fd| self.is_fb0(&files.fs, typed_fd).unwrap_or(false), + |_| false, + |_| false, + |_| false, + |_| false, + |_| false, + |_| false, + ) + .unwrap_or(false) + } + + fn raw_fd_shared_futex_backing( + &self, + fd: i32, + ) -> Option { + let raw_fd = usize::try_from(fd).ok()?; + let files = self.files.borrow(); + let typed = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::(raw_fd) + .ok()?; + self.shared_file_backing(&typed, true) + } + + fn raw_fd_file_handle(&self, fd: i32) -> Option> { + let raw_fd = usize::try_from(fd).ok()?; + let files = self.files.borrow(); + let typed = files + .raw_descriptor_store + .read() + .fd_from_raw_integer::(raw_fd) + .ok()?; + self.global.litebox.descriptor_table().entry_handle(&typed) + } + + /// `mmap(MAP_SHARED | PROT_WRITE)` of `/dev/fb0`: allocate ordinary anonymous pages in the + /// (shared shim/guest) address space, then register them with the + /// [`litebox::fs::devices::Framebuffer`] as its + /// live pixel store -- pre-filled with the current contents, adopted until munmap. Guest + /// stores through the mapping are immediately visible to the runner's RFB snapshot with no + /// flush step, which is the coherence contract every fbdev graphics client assumes. + /// + /// `MAP_SHARED` (kept on the anonymous mapping) also keeps `Task::save_address_space` from + /// private-copying the pages out on a fork handoff, so the registration stays valid across + /// guest process switches. + fn do_mmap_framebuffer( + &self, + suggested_addr: Option, + len: usize, + prot: ProtFlags, + flags: MapFlags, + offset: usize, + ) -> Result, MappingError> { + // A nonzero-offset fbdev mmap is legal on Linux but no real client uses it; only the + // offset-0 mapping can become the pixel store. + if offset != 0 { + log_unsupported!("mmap of /dev/fb0 at nonzero offset"); + return Err(MappingError::Io(Errno::EINVAL.into())); + } + // A framebuffer-typed fd only exists when `default_fs` mounted one (the sole source of + // an fb0 rdev major), so `None` would mean an fd recognized as fb0 by a filesystem this + // shim never built. + let Some(fb) = self.global.framebuffer.as_ref() else { + return Err(MappingError::Io(Errno::ENODEV.into())); + }; + // Replace any previous registration first (a client that mmaps fb0 twice): copy-back + // deregistration keeps the old mapping's last-drawn content. + if let Some((old_addr, old_len)) = fb.guest_mapping() { + fb.clear_guest_mapping_overlapping(old_addr, old_len); + } + let ptr = + self.do_mmap_anonymous(suggested_addr, len, prot, flags | MapFlags::MAP_ANONYMOUS)?; + // SAFETY: `ptr` addresses `len` readable+writable bytes in this same address space; + // `sys_munmap`, `sys_mremap`, and the execve bulk-release all clear the registration + // before those pages can go away. + unsafe { fb.set_guest_mapping(ptr.as_usize(), len) }; + Ok(ptr) + } + + /// Copy every byte covered by `range` from a writable guest memfd mapping + /// back into its retained open file description. + fn flush_shared_file_mappings( + &self, + range_start: usize, + range_len: usize, + ) -> Result<(), Errno> { + let range_end = range_start.checked_add(range_len).ok_or(Errno::EINVAL)?; + let mappings = self.files.borrow().shared_file_mappings.lock().clone(); + + for mapping in mappings { + if !mapping.writable { + continue; + } + let mapping_end = mapping.start.saturating_add(mapping.len); + let overlap_start = mapping.start.max(range_start); + let overlap_end = mapping_end.min(range_end); + if overlap_start >= overlap_end { + continue; + } + + let temporary_fd = self + .global + .litebox + .descriptor_table_mut() + .insert_handle(mapping.file.clone()); + let write_result = (|| { + let file_offset = mapping + .offset + .checked_add(overlap_start - mapping.start) + .ok_or(Errno::EOVERFLOW)?; + let files = self.files.borrow(); + let file_size = files + .fs + .fd_file_status(&temporary_fd) + .map_err(Errno::from)? + .size; + let len = (overlap_end - overlap_start).min(file_size.saturating_sub(file_offset)); + if len == 0 { + return Ok(()); + } + let bytes = UserPtr::::from_usize(overlap_start) + .to_owned_slice::(len) + .ok_or(Errno::EFAULT)?; + let mut written = 0; + while written < bytes.len() { + let size = files + .fs + .write( + &temporary_fd, + &bytes[written..], + Some(file_offset + written), + ) + .map_err(Errno::from)?; + if size == 0 || size > bytes.len() - written { + return Err(Errno::EIO); + } + written += size; + } + Ok(()) + })(); + let _ = self + .global + .litebox + .descriptor_table_mut() + .remove(&temporary_fd); + write_result?; + } + Ok(()) + } + + /// Forget the unmapped portions of writable memfd mappings while retaining + /// correctly-offset records for any pages on either side of a partial unmap. + fn clear_shared_file_mappings_for_range(&self, range_start: usize, range_len: usize) { + let range_end = range_start.saturating_add(range_len); + let files = self.files.borrow(); + let mut mappings = files.shared_file_mappings.lock(); + let old = core::mem::take(&mut *mappings); + + for mapping in old { + let mapping_end = mapping.start.saturating_add(mapping.len); + if mapping_end <= range_start || mapping.start >= range_end { + mappings.push(mapping); + continue; + } + if mapping.start < range_start { + mappings.push(SharedFileMapping { + start: mapping.start, + len: range_start - mapping.start, + offset: mapping.offset, + writable: mapping.writable, + file: mapping.file.clone(), + }); + } + if mapping_end > range_end { + mappings.push(SharedFileMapping { + start: range_end, + len: mapping_end - range_end, + offset: mapping.offset + (range_end - mapping.start), + writable: mapping.writable, + file: mapping.file, + }); + } + } + } + + fn remap_shared_file_mapping( + &self, + old_start: usize, + old_len: usize, + new_start: usize, + new_len: usize, + ) { + let old_end = old_start.saturating_add(old_len); + let files = self.files.borrow(); + let mut mappings = files.shared_file_mappings.lock(); + let old = core::mem::take(&mut *mappings); + let mut moved = false; + for mapping in old { + let mapping_end = mapping.start.saturating_add(mapping.len); + if moved || mapping.start > old_start || mapping_end < old_end { + mappings.push(mapping); + continue; + } + let moved_offset = mapping.offset + (old_start - mapping.start); + if mapping.start < old_start { + mappings.push(SharedFileMapping { + start: mapping.start, + len: old_start - mapping.start, + offset: mapping.offset, + writable: mapping.writable, + file: mapping.file.clone(), + }); + } + if mapping_end > old_end { + mappings.push(SharedFileMapping { + start: old_end, + len: mapping_end - old_end, + offset: mapping.offset + (old_end - mapping.start), + writable: mapping.writable, + file: mapping.file.clone(), + }); + } + mappings.push(SharedFileMapping { + start: new_start, + len: new_len, + offset: moved_offset, + writable: mapping.writable, + file: mapping.file, + }); + moved = true; + } + } + + fn set_shared_file_mapping_writable( + &self, + range_start: usize, + range_len: usize, + writable: bool, + ) { + let range_end = range_start.saturating_add(range_len); + let files = self.files.borrow(); + let mut mappings = files.shared_file_mappings.lock(); + let old = core::mem::take(&mut *mappings); + for mapping in old { + let mapping_end = mapping.start.saturating_add(mapping.len); + let start = mapping.start.max(range_start); + let end = mapping_end.min(range_end); + if start >= end { + mappings.push(mapping); + continue; + } + if mapping.start < start { + mappings.push(SharedFileMapping { + start: mapping.start, + len: start - mapping.start, + offset: mapping.offset, + writable: mapping.writable, + file: mapping.file.clone(), + }); + } + mappings.push(SharedFileMapping { + start, + len: end - start, + offset: mapping.offset + (start - mapping.start), + writable, + file: mapping.file.clone(), + }); + if mapping_end > end { + mappings.push(SharedFileMapping { + start: end, + len: mapping_end - end, + offset: mapping.offset + (end - mapping.start), + writable: mapping.writable, + file: mapping.file, + }); + } } - .map_err(Errno::from) } /// Handle syscall `munmap` #[inline] pub(crate) fn sys_munmap(&self, addr: UserPtrMut, len: usize) -> Result<(), Errno> { + let aligned_len = len + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(Errno::EINVAL)?; + self.flush_shared_file_mappings(addr.as_usize(), aligned_len)?; + if let Some(fb) = self.global.framebuffer.as_ref() { + // Copy-back + deregister BEFORE the pages go away. On the (guest-bug) path where + // the munmap itself then fails, this degrades the framebuffer to snapshot mode + // spuriously, which is safe. + fb.clear_guest_mapping_overlapping(addr.as_usize(), aligned_len); + } let result = self.sys_munmap_raw(addr, len); if result.is_ok() { + self.clear_shared_file_mappings_for_range(addr.as_usize(), aligned_len); self.clear_file_mappings_for_range(addr.as_usize(), len); + self.record_unmapped(addr.as_usize(), aligned_len); } result } @@ -399,7 +1003,7 @@ impl Task { /// re-patched instead of skipped. fn clear_file_mappings_for_range(&self, unmap_start: usize, unmap_len: usize) { let unmap_end = unmap_start.saturating_add(unmap_len); - let mut cache = self.global.elf_patch_cache.lock(); + let mut cache = self.process().elf_patch_cache.lock(); for state in cache.values_mut() { state.file_mappings.retain(|&(vaddr, seg_len)| { let seg_end = vaddr.saturating_add(seg_len); @@ -420,6 +1024,13 @@ impl Task { len: usize, prot: ProtFlags, ) -> Result<(), Errno> { + let aligned_len = len + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(Errno::EINVAL)?; + let writable = prot.contains(ProtFlags::PROT_WRITE); + if !writable { + self.flush_shared_file_mappings(addr.as_usize(), aligned_len)?; + } // Intercept transitions to PROT_EXEC: patch unpatched file mappings. if prot.contains(ProtFlags::PROT_EXEC) { let syscall_entry = self.global.platform.get_syscall_entry_point(); @@ -427,11 +1038,23 @@ impl Task { self.maybe_patch_on_mprotect_exec(addr, len, syscall_entry); } } - self.sys_mprotect_raw(addr, len, prot) + let result = self.sys_mprotect_raw(addr, len, prot); + if result.is_ok() { + self.set_shared_file_mapping_writable(addr.as_usize(), aligned_len, writable); + } + result } /// Raw mprotect without exec interception — used internally by the /// patching logic to avoid deadlocks (the patch path holds elf_patch_cache). + /// + /// This is the single choke point every transition to `PROT_EXEC` passes + /// through — the public [`Self::sys_mprotect`] included, via the call at + /// the end of that function — so it is also where instruction-cache + /// maintenance belongs: whatever was just written (loaded segments, the + /// rewriter's patches) has to be flushed to the point where the CPU's + /// instruction fetch path can see it before anything branches into the + /// range. #[inline] fn sys_mprotect_raw( &self, @@ -439,7 +1062,12 @@ impl Task { len: usize, prot: ProtFlags, ) -> Result<(), Errno> { - litebox_common_linux::mm::sys_mprotect(&self.global.pm, addr, len, prot) + let is_exec = prot.contains(ProtFlags::PROT_EXEC); + let result = litebox_common_linux::mm::sys_mprotect(&self.global.pm, addr, len, prot); + if result.is_ok() && is_exec { + clear_icache_range(addr.as_usize(), len); + } + result } #[inline] @@ -451,23 +1079,84 @@ impl Task { flags: MRemapFlags, new_addr: usize, ) -> Result, Errno> { - litebox_common_linux::mm::sys_mremap( + let old_len = old_size + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(Errno::EINVAL)?; + let new_len = new_size + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(Errno::EINVAL)?; + self.flush_shared_file_mappings(old_addr.as_usize(), old_len)?; + if let Some(fb) = self.global.framebuffer.as_ref() { + // A remap can move or shrink the pages backing a live fb0 registration; deregister + // (with copy-back) first rather than track the move -- no fbdev client remaps its + // framebuffer mapping. + fb.clear_guest_mapping_overlapping(old_addr.as_usize(), old_len); + } + let result = litebox_common_linux::mm::sys_mremap( &self.global.pm, old_addr, old_size, new_size, flags, new_addr, - ) + )?; + self.remap_shared_file_mapping( + old_addr.as_usize(), + old_len, + result.as_usize(), + new_len, + ); + self.record_unmapped(old_addr.as_usize(), old_len); + self.record_mapped(result.as_usize(), new_len); + Ok(result) } - /// Handle syscall `brk` - #[inline] + /// Handle syscall `brk`. + /// + /// The page manager is shared by every guest process in this shim but tracks only one program + /// break, so this swaps in the calling process's own break for the duration of the call and + /// takes the updated value back out afterwards, under a lock that keeps two processes from + /// interleaving. See [`crate::syscalls::process::Process::brk`]. pub(crate) fn sys_brk(&self, addr: UserPtrMut) -> Result { - litebox_common_linux::mm::sys_brk(&self.global.pm, addr) + use core::sync::atomic::Ordering; + + let _guard = self.global.brk_lock.lock(); + let process = self.process(); + let old_brk = process.brk.load(Ordering::Relaxed); + let stashed = self.global.pm.swap_brk(old_brk); + debug_assert_eq!(stashed, 0, "the page manager's break is only live in here"); + let result = litebox_common_linux::mm::sys_brk(&self.global.pm, addr); + let new_brk = self.global.pm.swap_brk(0); + // The full swap protocol per call: `stashed` non-zero here means some + // other path left its break live in the manager (a protocol breach + // this per-process model depends on never happening), and + // old->new shows exactly what range a grow/shrink walked -- the + // evidence needed when a break operation touches memory it should + // not (a cross-process brk was one observed way a forked child + // destroyed its suspended parent's heap). + litebox_util_log::trace!( + pid:? = self.pid, requested:? = addr.as_usize(), old_brk:? = old_brk, + stashed:? = stashed, new_brk:? = new_brk; + "brk" + ); + process.brk.store(new_brk, Ordering::Relaxed); + // The break's backing pages are this process's mappings like any other. + let (old_page, new_page) = (align_up(old_brk, PAGE_SIZE), align_up(new_brk, PAGE_SIZE)); + if new_page > old_page { + self.record_mapped(old_page, new_page - old_page); + } else if new_page < old_page { + self.record_unmapped(new_page, old_page - new_page); + } + result } - /// Handle syscall `madvise` + /// Handle syscall `madvise`. + /// + /// Every advice value returns to the guest: what the page manager cannot honour is + /// logged (once per advice, see `log_unsupported!`) and refused with `EINVAL`, and a + /// pure hint is accepted as the no-op it is on Linux. `MADV_WIPEONFORK` marks the range + /// in the shared page manager; the child-side zeroing is `PageManager::wipe_on_fork_child`, + /// run by the fork hand-off after the parent has copied its own image out. #[inline] pub(crate) fn sys_madvise( &self, @@ -475,6 +1164,19 @@ impl Task { len: usize, advice: litebox_common_linux::MadviseBehavior, ) -> Result<(), Errno> { + use litebox_common_linux::mm::{MadviseSupport, madvise_support}; + match madvise_support(&advice) { + MadviseSupport::Implemented => {} + MadviseSupport::AdvisoryNoop => { + litebox_util_log::trace!( + pid:? = self.pid, addr:? = addr.as_usize(), len:? = len, advice:? = advice; + "madvise hint accepted as a no-op" + ); + } + MadviseSupport::Unsupported => { + log_unsupported!("madvise({advice:?}) is not supported; returning EINVAL"); + } + } litebox_common_linux::mm::sys_madvise(&self.global.pm, addr, len, advice) } @@ -491,7 +1193,7 @@ impl Task { // We collect (fd, vaddr, seg_len, file_offset) to avoid holding // the lock while patching. let to_patch: alloc::vec::Vec<(i32, usize, usize)> = { - let cache = self.global.elf_patch_cache.lock(); + let cache = self.process().elf_patch_cache.lock(); let mut result = alloc::vec::Vec::new(); for (&fd, state) in cache.iter() { if state.pre_patched { @@ -552,7 +1254,7 @@ impl Task { /// x86_64 only: assumes 64-bit ELF layout and program header offsets. fn init_elf_patch_state(&self, fd: i32, mapped_addr: usize, file_offset: usize) { // Quick check: skip if already initialized. - if self.global.elf_patch_cache.lock().contains_key(&fd) { + if self.process().elf_patch_cache.lock().contains_key(&fd) { return; } @@ -574,6 +1276,7 @@ impl Task { } let e_type = ehdr.e_type.get(ENDIAN); + let e_machine = ehdr.e_machine.get(ENDIAN); let e_phoff: usize = ehdr.e_phoff.get(ENDIAN).trunc(); let e_phentsize = ehdr.e_phentsize.get(ENDIAN) as usize; let e_phnum = ehdr.e_phnum.get(ENDIAN) as usize; @@ -599,6 +1302,7 @@ impl Task { // Find highest PT_LOAD end (p_vaddr + p_memsz) and compute base_addr // by matching the segment whose p_offset corresponds to file_offset. let mut max_load_end: u64 = 0; + let mut min_load_start: u64 = u64::MAX; let mut base_addr: Option = None; for i in 0..e_phnum { let ph_bytes = &phdrs_buf[i * e_phentsize..][..e_phentsize]; @@ -621,6 +1325,9 @@ impl Task { if end > max_load_end { max_load_end = end; } + if p_vaddr < min_load_start { + min_load_start = p_vaddr; + } // Match segment by page-aligned file offset to derive base address. if base_addr.is_none() && align_down(p_offset, PAGE_SIZE) == align_down(file_offset, PAGE_SIZE) @@ -633,6 +1340,25 @@ impl Task { return; // No PT_LOAD segments } + // Record the image span for fault symbolization. This must happen at + // map time: the dynamic linker closes the fd (dropping the patch-state + // entry below) as soon as the library is mapped, long before any fault + // that needs a `path+offset`. Best-effort -- an fd without a recorded + // path (memfd, inherited fd) simply is not symbolizable later. + let image_base = if e_type == ET_DYN { base_addr } else { Some(0) }; + if let Some(base) = image_base + && let Some(path) = self.fd_abs_path(fd) + { + let lo = base + align_down(min_load_start.trunc(), PAGE_SIZE); + let hi = base + align_up(max_load_end.trunc(), PAGE_SIZE); + self.global.guest_images.lock().push(GuestImage { + lo, + hi, + base, + path: alloc::string::String::from_utf8_lossy(path.as_bytes()).into_owned(), + }); + } + // Check if file is pre-patched by reading the last 32 bytes for magic let (pre_patched, tramp_file_offset, tramp_vaddr, tramp_file_size) = self.check_trampoline_magic(fd); @@ -666,9 +1392,10 @@ impl Task { }; // Insert under lock (re-check for races). - let mut cache = self.global.elf_patch_cache.lock(); + let mut cache = self.process().elf_patch_cache.lock(); cache.entry(fd).or_insert(ElfPatchState { pre_patched, + machine: e_machine, trampoline_file_offset: tramp_file_offset, trampoline_file_size: tramp_file_size.trunc(), trampoline_addr: trampoline_vaddr, @@ -688,7 +1415,14 @@ impl Task { let Ok(stat) = self.sys_fstat(fd) else { return (false, 0, 0, 0); }; + // `st_size` is pointer-width and unsigned in the x86-64 `struct stat`, + // and a signed 64-bit field in the generic layout aarch64 uses. + #[cfg(target_arch = "x86_64")] let file_size = stat.st_size; + #[cfg(target_arch = "aarch64")] + let Ok(file_size) = usize::try_from(stat.st_size) else { + return (false, 0, 0, 0); + }; if file_size < HEADER_SIZE { return (false, 0, 0, 0); } @@ -706,6 +1440,27 @@ impl Task { (true, file_offset, vaddr, trampoline_size) } + /// Write `bytes` at `ptr`, which points into a mapping that is — or is + /// about to become — executable. + /// + /// On hosts with per-thread code write protection (Darwin's `MAP_JIT`), a + /// write into such a mapping faults unless this thread first enables write + /// access, page permissions notwithstanding; see + /// `litebox::platform::PageManagementProvider::jit_write_protect`. This + /// helper brackets the copy accordingly; on every other host the bracket + /// is a no-op, so all code writes in this module go through it + /// unconditionally. + fn write_code_bytes(&self, ptr: UserPtrMut, bytes: &[u8]) -> Option<()> { + // SAFETY: nothing on this thread executes out of a JIT mapping + // between the toggles — the copy below is ordinary host code, and + // guest code is only re-entered long after the closing toggle. + unsafe { self.global.platform.jit_write_protect(false) }; + let result = ptr.copy_from_slice::(0, bytes); + // SAFETY: restores the executable state guest code requires. + unsafe { self.global.platform.jit_write_protect(true) }; + result + } + /// Apply the trap fallback to a mapped code segment: replace all `syscall` /// instructions with traps (`ICEBP;HLT`), then restore RX. /// @@ -740,9 +1495,7 @@ impl Task { ); } assert!( - mapped_addr - .copy_from_slice::(0, &code_buf) - .is_some(), + self.write_code_bytes(mapped_addr, &code_buf).is_some(), "fatal: failed to write trap bytes back to code segment" ); @@ -777,14 +1530,14 @@ impl Task { // Initialize patch state if this is the first mmap for this fd. // Typically the first mapping is at offset 0 (the ELF header), but // some loaders may map an executable segment at a non-zero offset first. - if !self.global.elf_patch_cache.lock().contains_key(&fd) { + if !self.process().elf_patch_cache.lock().contains_key(&fd) { self.init_elf_patch_state(fd, mapped_addr.as_usize(), file_offset.unwrap_or(0)); } // This lock guards the elf_patch_cache and is held for the entire // patching operation. In practice this is fine because the dynamic // linker loads shared libraries sequentially. - let mut cache = self.global.elf_patch_cache.lock(); + let mut cache = self.process().elf_patch_cache.lock(); let Some(state) = cache.get_mut(&fd) else { return true; // No patch state — not an ELF we're tracking }; @@ -833,11 +1586,31 @@ impl Task { tramp_data[..8].copy_from_slice(&syscall_entry.to_le_bytes()); } + // Publish the guest thread-pointer offset the runtime actually + // reserved, mirroring `ElfParsedFile::load_trampoline` in + // `litebox_common_linux::loader`. The loader path only covers + // the main executable and its interpreter; libraries mapped by + // the in-guest dynamic linker arrive here instead, and leaving + // the packager-seeded default in place would make this + // module's gates read the guest TP from a different slot than + // every loader-published module writes it to. Skipped when the + // platform bakes the offset into the gates as an immediate. + if let Some(offset) = self.global.platform.get_guest_tp_slot_offset() { + let end = TRAMPOLINE_GUEST_TP_SLOT_OFFSET + size_of::(); + if tramp_data.len() < end { + // The gates in this image read this word; a trampoline + // too short to hold it means every rewritten syscall + // would compute a garbage thread pointer. Fail the + // mapping rather than continuing silently. + let _ = self.sys_munmap_raw(tramp_ptr, tramp_len); + return false; + } + tramp_data[TRAMPOLINE_GUEST_TP_SLOT_OFFSET..end] + .copy_from_slice(&offset.to_ne_bytes()); + } + // Write to the mapped region. - if tramp_ptr - .copy_from_slice::(0, &tramp_data) - .is_none() - { + if self.write_code_bytes(tramp_ptr, &tramp_data).is_none() { let _ = self.sys_munmap_raw(tramp_ptr, tramp_len); return false; } @@ -863,6 +1636,20 @@ impl Task { // ── Runtime patching path (unpatched binaries) ─────────────── + // The runtime rewriter is an x86-64 instruction decoder. Running it + // (or the trap fallback, which shares that decoder) over another + // architecture's code would reinterpret arbitrary instruction words as + // x86 and corrupt the segment, so refuse and leave the code untouched. + // On such hosts every shipped image is expected to be pre-patched + // (carrying the `LITEBOX0` trailer) and never reaches this arm. + if state.machine != object::elf::EM_X86_64 { + litebox_util_log::warn!( + machine:? = state.machine, addr:? = mapped_addr.as_usize(), len:? = len; + "unpatched non-x86-64 image: runtime syscall patching skipped" + ); + return true; + } + // Allocate the trampoline region if not yet done. let addr_usize = mapped_addr.as_usize(); if !state.trampoline_mapped { @@ -914,8 +1701,8 @@ impl Task { // Write the 8-byte syscall entry point at the start. let entry_ptr = UserPtrMut::::from_usize(actual_addr); - if entry_ptr - .copy_from_slice::(0, &syscall_entry.to_le_bytes()) + if self + .write_code_bytes(entry_ptr, &syscall_entry.to_le_bytes()) .is_none() { litebox_util_log::warn!("failed to write syscall entry point to trampoline"); @@ -1039,10 +1826,7 @@ impl Task { // never target an uninitialized trampoline. let tramp_write_ptr = UserPtrMut::::from_usize(state.trampoline_addr + state.trampoline_cursor); - if tramp_write_ptr - .copy_from_slice::(0, &stubs) - .is_none() - { + if self.write_code_bytes(tramp_write_ptr, &stubs).is_none() { let _ = self.sys_mprotect_raw( mapped_addr, len, @@ -1053,11 +1837,8 @@ impl Task { } // Write patched code back to the mapped region. - if mapped_addr - .copy_from_slice::(0, &code_buf) - .is_none() - { - let _ = mapped_addr.copy_from_slice::(0, &original_code); + if self.write_code_bytes(mapped_addr, &code_buf).is_none() { + let _ = self.write_code_bytes(mapped_addr, &original_code); let _ = self.sys_mprotect_raw( mapped_addr, len, @@ -1074,11 +1855,9 @@ impl Task { // have replaced unpatchable syscalls with trap instructions. // Write back the modified code if it changed. if code_buf != original_code - && mapped_addr - .copy_from_slice::(0, &code_buf) - .is_none() + && self.write_code_bytes(mapped_addr, &code_buf).is_none() { - let _ = mapped_addr.copy_from_slice::(0, &original_code); + let _ = self.write_code_bytes(mapped_addr, &original_code); panic!("fatal: failed to write trap bytes back to code segment"); } // Fall through to restore RX protections below. @@ -1101,12 +1880,26 @@ impl Task { true } + /// Find the guest ELF image containing `addr`, returning its guest path + /// and the image-relative offset (`addr - load bias`) -- the pair + /// `llvm-symbolizer` needs to resolve the address against the guest's own + /// (debug-info-carrying) ELF. Latest mapping wins so an address reused + /// after an image is replaced resolves to the live image. + pub(crate) fn find_guest_image(&self, addr: usize) -> Option<(alloc::string::String, usize)> { + let images = self.global.guest_images.lock(); + images + .iter() + .rev() + .find(|img| (img.lo..img.hi).contains(&addr)) + .map(|img| (img.path.clone(), addr - img.base)) + } + /// Finalize the ELF patching state for `fd`. /// /// Removes the cache entry (preventing stale state if the fd is reused) /// and unmaps any trampoline that was allocated but never used. pub(crate) fn finalize_elf_patch(&self, fd: i32) { - let state = self.global.elf_patch_cache.lock().remove(&fd); + let state = self.process().elf_patch_cache.lock().remove(&fd); if let Some(state) = state && state.trampoline_mapped && !state.pre_patched @@ -1125,17 +1918,27 @@ impl Task { #[cfg(test)] mod tests { - use litebox::{ - fs::{Mode, OFlags}, - platform::PageManagementProvider, - }; - use litebox_common_linux::{MRemapFlags, MapFlags, ProtFlags, errno::Errno}; + use litebox::fs::{Mode, OFlags}; + // Only `test_collision_with_global_allocator` needs these, and it is gated to + // the hosts whose allocator layout it knows. + use litebox::platform::PageManagementProvider; + #[cfg(any(target_os = "linux", target_os = "windows"))] + use litebox_common_linux::MRemapFlags; + use litebox_common_linux::{FcntlArg, MapFlags, ProtFlags, errno::Errno}; use crate::syscalls::tests::TestPlatform as Platform; use crate::{UserPtrMut, syscalls::tests::init_platform}; + /// The host's page size. Sizes and addresses below are written as multiples + /// of this rather than as literals: `mmap`/`mprotect`/`mremap` reject a + /// length that is not a whole number of pages, and Apple Silicon's page is + /// 16 KiB, so a literal `0x1000` is not a page there and every such call + /// fails before reaching the behaviour under test. + use super::PAGE_SIZE; + #[test] fn test_anonymous_mmap() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let addr = task @@ -1156,6 +1959,7 @@ mod tests { #[test] fn test_file_backed_mmap() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let content = b"Hello, world!"; @@ -1186,12 +1990,13 @@ mod tests { #[test] fn test_mremap() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let addr = task .sys_mmap( 0, - 0x2000, + 2 * PAGE_SIZE, ProtFlags::PROT_READ, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE, -1, @@ -1199,11 +2004,13 @@ mod tests { ) .unwrap(); + // Growing the first page in place would run into the second, which this + // same mapping already occupies, so it fails without `MREMAP_MAYMOVE`. assert!(matches!( task.sys_mremap( addr, - 0x1000, - 0x2000, + PAGE_SIZE, + 2 * PAGE_SIZE, litebox_common_linux::MRemapFlags::empty(), 0 ), @@ -1212,26 +2019,32 @@ mod tests { let new_addr = task .sys_mremap( addr, - 0x1000, - 0x2000, + PAGE_SIZE, + 2 * PAGE_SIZE, litebox_common_linux::MRemapFlags::MREMAP_MAYMOVE, 0, ) .unwrap(); - task.sys_munmap(addr, 0x2000).unwrap(); - task.sys_munmap(new_addr, 0x2000).unwrap(); + task.sys_munmap(addr, 2 * PAGE_SIZE).unwrap(); + task.sys_munmap(new_addr, 2 * PAGE_SIZE).unwrap(); } #[test] fn test_mmap_fixed_noreplace() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); // First, create an initial mapping at a specific address away from boundaries - let base_addr = 0x1000_0000usize; // 256 MiB - safe middle ground + // Well clear of the host's lowest mappable address: an arm64 Mach-O + // process reserves the first 4 GiB as `__PAGEZERO`, so a literal low + // address is not mappable there. Test 5 maps one page below this, so + // leave room for that too. + let base_addr = + >::TASK_ADDR_MIN + 0x1000_0000usize; let addr1 = task .sys_mmap( base_addr, - 0x2000, + 2 * PAGE_SIZE, ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, @@ -1248,7 +2061,7 @@ mod tests { let err = task .sys_mmap( addr1.as_usize(), - 0x1000, + PAGE_SIZE, ProtFlags::PROT_READ, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, @@ -1258,11 +2071,11 @@ mod tests { assert_eq!(err, Errno::EEXIST); // Test 2: Partial overlap at end - should fail with EEXIST - // Existing: [addr1, addr1 + 0x2000), New: [addr1 + 0x1000, addr1 + 0x3000) + // Existing: [addr1, addr1 + 2 * PAGE_SIZE), New: [addr1 + PAGE_SIZE, addr1 + 0x3000) let err = task .sys_mmap( - addr1.as_usize() + 0x1000, - 0x2000, + addr1.as_usize() + PAGE_SIZE, + 2 * PAGE_SIZE, ProtFlags::PROT_READ, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, @@ -1272,11 +2085,11 @@ mod tests { assert_eq!(err, Errno::EEXIST); // Test 3: Partial overlap at start - should fail with EEXIST - // Existing: [addr1, addr1 + 0x2000), New: [addr1 - 0x1000, addr1 + 0x1000) + // Existing: [addr1, addr1 + 2 * PAGE_SIZE), New: [addr1 - PAGE_SIZE, addr1 + PAGE_SIZE) let err = task .sys_mmap( - addr1.as_usize() - 0x1000, - 0x2000, + addr1.as_usize() - PAGE_SIZE, + 2 * PAGE_SIZE, ProtFlags::PROT_READ, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, @@ -1288,35 +2101,35 @@ mod tests { // Test 4: Adjacent mapping (right after) - should succeed let addr2 = task .sys_mmap( - addr1.as_usize() + 0x2000, - 0x1000, + addr1.as_usize() + 2 * PAGE_SIZE, + PAGE_SIZE, ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, 0, ) .unwrap(); - assert_eq!(addr2.as_usize(), addr1.as_usize() + 0x2000); + assert_eq!(addr2.as_usize(), addr1.as_usize() + 2 * PAGE_SIZE); // Test 5: Adjacent mapping (right before) - should succeed let addr3 = task .sys_mmap( - addr1.as_usize() - 0x1000, - 0x1000, + addr1.as_usize() - PAGE_SIZE, + PAGE_SIZE, ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, 0, ) .unwrap(); - assert_eq!(addr3.as_usize(), addr1.as_usize() - 0x1000); + assert_eq!(addr3.as_usize(), addr1.as_usize() - PAGE_SIZE); // Test 6: Zero address with MAP_FIXED_NOREPLACE - should fail with EPERM // (matches Linux behavior where vm.mmap_min_addr prevents mapping at address 0) let err = task .sys_mmap( 0, - 0x1000, + PAGE_SIZE, ProtFlags::PROT_READ, MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE | MapFlags::MAP_FIXED_NOREPLACE, -1, @@ -1326,14 +2139,20 @@ mod tests { assert_eq!(err, Errno::EPERM); // Clean up - task.sys_munmap(addr3, 0x1000).unwrap(); - task.sys_munmap(addr1, 0x2000).unwrap(); - task.sys_munmap(addr2, 0x1000).unwrap(); + task.sys_munmap(addr3, PAGE_SIZE).unwrap(); + task.sys_munmap(addr1, 2 * PAGE_SIZE).unwrap(); + task.sys_munmap(addr2, PAGE_SIZE).unwrap(); } + // Not on macOS: `MacOsUserland::GUEST_ADDR_MIN` is 1 TiB (8a65efa), so a + // Darwin host allocation (~4-39 GiB) can never satisfy the in-guest-range + // mmap this loop searches for -- the collision under test is impossible by + // construction and the search spins forever (witnessed as the CI macOS + // job's 420 s slow-timeout SIGKILL). #[cfg(any(target_os = "linux", target_os = "windows"))] #[test] fn test_collision_with_global_allocator() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let platform = task.global.platform; let mut data = alloc::vec::Vec::new(); @@ -1368,6 +2187,33 @@ mod tests { })); addr }; + // Darwin's non-fixed `mmap(NULL, ...)` packs consecutive anonymous + // requests back to back rather than scattering them the way Linux's + // ASLR does, so a bare `mmap(NULL, 0x10_000, ...)` here would make + // `addr - PAGE_SIZE` land inside the previous iteration's (still + // mapped) block every time, and the loop below would never find the + // free page it needs. Map one extra leading page and free just that + // one instead, so `[addr - PAGE_SIZE, addr)` is available by + // construction rather than by chance. + #[cfg(target_os = "macos")] + let addr = { + let base = unsafe { + libc::mmap( + core::ptr::null_mut(), + 0x10_000 + PAGE_SIZE, + libc::PROT_READ | libc::PROT_WRITE, + libc::MAP_PRIVATE | libc::MAP_ANONYMOUS, + -1, + 0, + ) + } as usize; + unsafe { libc::munmap(base as *mut libc::c_void, PAGE_SIZE) }; + let addr = base + PAGE_SIZE; + data.push(alloc::vec::Vec::::from(unsafe { + core::slice::from_raw_parts(addr as *const u8, 0x10_000) + })); + addr + }; let mut included = false; for r in (0).unwrap(); // Anonymous shared mappings allow permission changes including write - task.sys_mprotect(addr, 0x2000, ProtFlags::PROT_READ | ProtFlags::PROT_WRITE) - .unwrap(); + task.sys_mprotect( + addr, + 2 * PAGE_SIZE, + ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, + ) + .unwrap(); addr.write_slice_at_offset::(0, &[0xab; 0x10]) .unwrap(); assert_eq!(addr.read_at_offset::(0).unwrap(), 0xab_u8); // mprotect to read-only or read-exec should also succeed - task.sys_mprotect(addr, 0x2000, ProtFlags::PROT_READ) + task.sys_mprotect(addr, 2 * PAGE_SIZE, ProtFlags::PROT_READ) .unwrap(); - task.sys_mprotect(addr, 0x2000, ProtFlags::PROT_READ_EXEC) + task.sys_mprotect(addr, 2 * PAGE_SIZE, ProtFlags::PROT_READ_EXEC) .unwrap(); - task.sys_munmap(addr, 0x2000).unwrap(); + task.sys_munmap(addr, 2 * PAGE_SIZE).unwrap(); } #[test] fn test_map_shared_anonymous_writable() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); // MAP_SHARED | MAP_ANON with PROT_WRITE should succeed @@ -1486,6 +2341,7 @@ mod tests { #[test] fn test_map_shared_readonly_file() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let content = b"Hello, shared!"; @@ -1497,7 +2353,14 @@ mod tests { // MAP_SHARED with PROT_READ on a file should succeed let addr = task - .sys_mmap(0, 0x1000, ProtFlags::PROT_READ, MapFlags::MAP_SHARED, fd, 0) + .sys_mmap( + 0, + PAGE_SIZE, + ProtFlags::PROT_READ, + MapFlags::MAP_SHARED, + fd, + 0, + ) .unwrap(); // Data should match @@ -1510,16 +2373,61 @@ mod tests { // mprotect to add write permission should fail let err = task - .sys_mprotect(addr, 0x1000, ProtFlags::PROT_READ | ProtFlags::PROT_WRITE) + .sys_mprotect( + addr, + PAGE_SIZE, + ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, + ) .unwrap_err(); assert_eq!(err, Errno::EACCES); - task.sys_munmap(addr, 0x1000).unwrap(); + task.sys_munmap(addr, PAGE_SIZE).unwrap(); task.sys_close(fd).unwrap(); } + #[test] + fn writable_shared_memfd_copies_back_before_descriptor_transfer() { + let _guard = crate::syscalls::tests::address_space_guard(); + let task = init_platform(None); + task.sys_mkdirat(litebox_common_linux::AT_FDCWD, "/tmp", 0o700) + .unwrap(); + let len = 2304; + let fd = i32::try_from(task.sys_memfd_create(c"glycin-texture", 0x3).unwrap()).unwrap(); + task.sys_ftruncate(fd, len).unwrap(); + + let addr = task + .sys_mmap( + 0, + len, + ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, + MapFlags::MAP_SHARED, + fd, + 0, + ) + .unwrap(); + let expected = (0..len) + .map(|index| u8::try_from(index % 251).unwrap()) + .collect::>(); + addr.copy_from_slice::(0, &expected).unwrap(); + task.sys_munmap(addr, len).unwrap(); + assert_eq!(task.sys_fcntl(fd, FcntlArg::GET_SEALS).unwrap(), 0); + assert_eq!(task.sys_fcntl(fd, FcntlArg::ADD_SEALS(0x0e)).unwrap(), 0); + assert_eq!(task.sys_fcntl(fd, FcntlArg::GET_SEALS).unwrap(), 0x0e); + + let transferred = task.transfer_fd(fd).unwrap(); + task.sys_close(fd).unwrap(); + let received_fd = + i32::try_from(task.install_transferred_fd(transferred, false).unwrap()).unwrap(); + let mut actual = alloc::vec![0; len + 1]; + let size = task.sys_read(received_fd, &mut actual, Some(0)).unwrap(); + assert_eq!(size, len, "copy-back must not extend to the page boundary"); + assert_eq!(&actual[..size], expected); + task.sys_close(received_fd).unwrap(); + } + #[test] fn test_madvise() { + let _guard = crate::syscalls::tests::address_space_guard(); let task = init_platform(None); let addr = task @@ -1572,4 +2480,217 @@ mod tests { let result = ptr.read_at_offset::(0); assert!(result.is_none()); } + + /// Regression test: mapping a pre-patched ET_DYN's executable segment must + /// rewrite BOTH runtime-variable trampoline header words -- the syscall + /// entry point (word 0) and, on a platform whose gates read the guest + /// thread-pointer offset from the trampoline, that offset (word 1). + /// + /// The loader path (`ElfParsedFile::load_trampoline`) always published + /// both, but the mmap path used by an in-guest dynamic linker mapping a + /// pre-patched library only published word 0, leaving the packager-seeded + /// default in word 1. On macOS that made every gate in an ld.so-mapped + /// library read the guest TP from the wrong TSD slot, sending node's + /// cross-module `std::call_once` through host-heap garbage to a PC=0 + /// instruction abort. + #[test] + fn test_prepatched_mmap_publishes_trampoline_header() { + use litebox::platform::SystemInfoProvider as _; + + // Values the packager might have seeded; the runtime must replace them. + const SEED_ENTRY: u64 = 0x1111_1111_1111_1111; + const SEED_TP_OFFSET: u64 = 0x2222_2222_2222_2222; + const TRAMP_SIZE: usize = 32; + + let _guard = crate::syscalls::tests::address_space_guard(); + let task = init_platform(None); + + // Synthetic pre-patched ET_DYN: + // [ELF header + one PT_LOAD phdr | pad to PAGE_SIZE] + // [trampoline code (TRAMP_SIZE bytes)] [32-byte LITEBOX0 trailer] + let mut file = alloc::vec![0u8; PAGE_SIZE + TRAMP_SIZE + 32]; + file[0..4].copy_from_slice(b"\x7fELF"); + file[4] = 2; // ELFCLASS64 + file[5] = 1; // ELFDATA2LSB + file[6] = 1; // EV_CURRENT + file[16..18].copy_from_slice(&3u16.to_le_bytes()); // e_type = ET_DYN + #[cfg(target_arch = "x86_64")] + let e_machine: u16 = 62; // EM_X86_64 + #[cfg(target_arch = "aarch64")] + let e_machine: u16 = 183; // EM_AARCH64 + file[18..20].copy_from_slice(&e_machine.to_le_bytes()); + file[20..24].copy_from_slice(&1u32.to_le_bytes()); // e_version + file[32..40].copy_from_slice(&64u64.to_le_bytes()); // e_phoff + file[52..54].copy_from_slice(&64u16.to_le_bytes()); // e_ehsize + file[54..56].copy_from_slice(&56u16.to_le_bytes()); // e_phentsize + file[56..58].copy_from_slice(&1u16.to_le_bytes()); // e_phnum + // PT_LOAD at p_offset 0, p_vaddr 0, R+X, one page. + let ph = 64; + file[ph..ph + 4].copy_from_slice(&1u32.to_le_bytes()); // p_type + file[ph + 4..ph + 8].copy_from_slice(&5u32.to_le_bytes()); // p_flags R|X + file[ph + 32..ph + 40].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_filesz + file[ph + 40..ph + 48].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_memsz + file[ph + 48..ph + 56].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); // p_align + // Trampoline code, seeded like the packager leaves it. + file[PAGE_SIZE..PAGE_SIZE + 8].copy_from_slice(&SEED_ENTRY.to_le_bytes()); + file[PAGE_SIZE + 8..PAGE_SIZE + 16].copy_from_slice(&SEED_TP_OFFSET.to_le_bytes()); + // Trailer: magic, trampoline file offset, vaddr (just past PT_LOAD), size. + let t = PAGE_SIZE + TRAMP_SIZE; + file[t..t + 8].copy_from_slice(b"LITEBOX0"); + file[t + 8..t + 16].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); + file[t + 16..t + 24].copy_from_slice(&(PAGE_SIZE as u64).to_le_bytes()); + file[t + 24..t + 32].copy_from_slice(&(TRAMP_SIZE as u64).to_le_bytes()); + + let fd = task + .sys_open( + "prepatched_test.so", + OFlags::RDWR | OFlags::CREAT, + Mode::RWXU, + ) + .unwrap(); + let fd = i32::try_from(fd).unwrap(); + assert_eq!(task.sys_write(fd, &file, None).unwrap(), file.len()); + + // Map two pages so the trampoline's MAP_FIXED landing zone (page 1, + // per the trailer's vaddr) is this test's own mapping, not whatever + // else the harness put there. + let addr = task + .sys_mmap( + 0, + 2 * PAGE_SIZE, + ProtFlags::PROT_READ | ProtFlags::PROT_EXEC, + MapFlags::MAP_PRIVATE, + fd, + 0, + ) + .unwrap(); + + let tramp = UserPtrMut::::from_usize(addr.as_usize() + PAGE_SIZE); + let header = tramp.to_owned_slice::(16).unwrap(); + let platform = task.global.platform; + + let entry = platform.get_syscall_entry_point(); + assert_ne!(entry, 0, "test platform must expose a syscall entry point"); + assert_eq!( + header[..8], + entry.to_le_bytes(), + "mmap path must publish the runtime syscall entry into trampoline word 0" + ); + + match platform.get_guest_tp_slot_offset() { + Some(offset) => assert_eq!( + header[8..16], + offset.to_ne_bytes(), + "mmap path must publish the runtime guest TP slot offset into trampoline word 1" + ), + None => assert_eq!( + header[8..16], + SEED_TP_OFFSET.to_le_bytes(), + "platforms that bake the TP offset into gates must leave the seeded word alone" + ), + } + + task.sys_munmap(addr, 2 * PAGE_SIZE).unwrap(); + task.sys_close(fd).unwrap(); + } + + /// Regression test: `PageManager::release_memory` must release exactly the + /// address ranges the caller names, never the whole tracked mapping those + /// ranges happen to fall inside. + /// + /// The VMA tree coalesces adjacent ranges with identical properties into + /// one entry, so two separate `mmap`s that abut are reported as a single + /// mapping. That is not an exotic shape here: `Vmem::get_unmmaped_area` + /// hands out the address immediately below an existing range, so a guest's + /// next anonymous `mmap` routinely lands flush against the previous one -- + /// and every guest process in this shim shares one page manager, so the + /// previous one can belong to a *different* process. `execve`'s teardown + /// (`sys_execve`, the `leave_address_space_if_alone` branch) scopes itself + /// to the calling process's `owned_ranges` for exactly that reason; before + /// this, it still released the whole coalesced entry each owned range + /// touched. Observed live as `node -e 'execSync("/bin/sh -c ...")'`: the + /// forked child's post-exec `mmap`s abutted 208 KiB of its suspended + /// parent's musl heap, the child's second `execve` unmapped all of it, and + /// the parent `SIGSEGV`ed on the first libc global it read after taking its + /// address space back. + #[test] + fn release_memory_releases_only_the_named_ranges_of_a_coalesced_mapping() { + use litebox::mm::linux::VmFlags; + + let _guard = crate::syscalls::tests::address_space_guard(); + let task = init_platform(None); + let prot = || ProtFlags::PROT_READ | ProtFlags::PROT_WRITE; + let flags = || MapFlags::MAP_ANON | MapFlags::MAP_PRIVATE; + + // Two *separate* mappings that end up adjacent: take two pages, give + // the second back, then claim it again at that exact address. The + // second `mmap` is a mapping of its own, but carries the same + // properties as the first, so the tree merges them. + let mine = task + .sys_mmap(0, 2 * PAGE_SIZE, prot(), flags(), -1, 0) + .unwrap(); + let neighbour_addr = mine.as_usize() + PAGE_SIZE; + task.sys_munmap(UserPtrMut::from_usize(neighbour_addr), PAGE_SIZE) + .unwrap(); + let neighbour = task + .sys_mmap( + neighbour_addr, + PAGE_SIZE, + prot(), + flags() | MapFlags::MAP_FIXED, + -1, + 0, + ) + .unwrap(); + assert_eq!(neighbour.as_usize(), neighbour_addr); + + let entry = |addr: usize| { + task.global + .pm + .mappings() + .into_iter() + .find(|(r, _)| r.contains(&addr)) + }; + let (merged, _) = entry(mine.as_usize()).expect("the first mapping should be tracked"); + assert!( + merged.contains(&neighbour_addr), + "precondition: the two adjacent mappings should be tracked as one entry \ + ({merged:?} should cover {neighbour_addr:#x}). If they no longer coalesce, the \ + cross-owner teardown this test pins cannot happen -- revisit the test, not the fix." + ); + + // Release only the first page, exactly as `execve` names the calling + // process's own ranges out of a mapping it may share with a sibling. + let mine_range = mine.as_usize()..mine.as_usize() + PAGE_SIZE; + // SAFETY: nothing holds references into the first page; the test does + // not touch it again. + unsafe { + task.global + .pm + .release_memory(|r: core::ops::Range, _: VmFlags| { + let start = r.start.max(mine_range.start); + let end = r.end.min(mine_range.end); + (start < end).then_some(start..end) + }) + } + .unwrap(); + + assert!( + entry(mine.as_usize()).is_none(), + "the named range should have been released" + ); + let (survivor, flags) = entry(neighbour_addr) + .expect("the neighbour's page must survive a release that did not name it"); + assert_eq!(survivor, neighbour_addr..neighbour_addr + PAGE_SIZE); + assert!(flags.contains(VmFlags::VM_READ | VmFlags::VM_WRITE)); + // Still really mapped, not merely still tracked: this is the failure + // that killed the parent process, since `remove_mapping` unmaps at the + // host before it forgets the range. + neighbour + .write_slice_at_offset::(0, &[0xab; 8]) + .expect("the surviving page must still be writable"); + assert_eq!(neighbour.read_at_offset::(0).unwrap(), 0xab); + + task.sys_munmap(neighbour, PAGE_SIZE).unwrap(); + } } diff --git a/litebox_shim_linux/src/syscalls/mod.rs b/litebox_shim_linux/src/syscalls/mod.rs index 53dd561f44..ec48648c0f 100644 --- a/litebox_shim_linux/src/syscalls/mod.rs +++ b/litebox_shim_linux/src/syscalls/mod.rs @@ -6,11 +6,15 @@ pub(crate) mod epoll; pub(crate) mod eventfd; pub mod file; +pub(crate) mod inotify; pub(crate) mod misc; pub(crate) mod mm; pub(crate) mod net; +pub(crate) mod netlink; pub(crate) mod pipe; pub mod process; +#[cfg(target_arch = "aarch64")] +pub(crate) mod ptrace; pub(crate) mod unix; pub(crate) mod signal; diff --git a/litebox_shim_linux/src/syscalls/net.rs b/litebox_shim_linux/src/syscalls/net.rs index 3f01e370b3..f883164884 100644 --- a/litebox_shim_linux/src/syscalls/net.rs +++ b/litebox_shim_linux/src/syscalls/net.rs @@ -4,9 +4,8 @@ //! Socket-related syscalls, e.g., socket, bind, listen, etc. use core::{ - ffi::CStr, mem::{offset_of, size_of}, - net::{Ipv4Addr, SocketAddr, SocketAddrV4}, + net::{Ipv4Addr, Ipv6Addr, SocketAddr, SocketAddrV4, SocketAddrV6}, }; use alloc::string::ToString; @@ -33,7 +32,10 @@ use litebox_common_linux::{ }; use zerocopy::{FromBytes, Immutable, IntoBytes}; -use crate::syscalls::unix::{CSockUnixAddr, UnixSocket, UnixSocketAddr}; +use crate::syscalls::{ + file::TransferredFd, + unix::{CSockUnixAddr, UnixSocket, UnixSocketAddr}, +}; use crate::{GlobalState, ShimFS, ShimPlatform, Task}; use crate::{UserPtr, UserPtrMut, syscalls::signal}; @@ -126,6 +128,41 @@ impl From for CSockInetAddr { } } +#[derive(Clone, Copy, FromBytes, IntoBytes, Immutable)] +#[repr(C, packed)] +struct CSockInet6Addr { + family: i16, + port: u16, + flowinfo: u32, + addr: [u8; 16], + scope_id: u32, +} + +impl From for SocketAddrV6 { + fn from(c_addr: CSockInet6Addr) -> Self { + SocketAddrV6::new( + Ipv6Addr::from(c_addr.addr), + u16::from_be(c_addr.port), + u32::from_be(c_addr.flowinfo), + // scope_id is a local interface index, not a wire quantity, so unlike + // port/flowinfo it is never byte-swapped (see Linux's `struct sockaddr_in6`). + c_addr.scope_id, + ) + } +} + +impl From for CSockInet6Addr { + fn from(addr: SocketAddrV6) -> Self { + CSockInet6Addr { + family: AddressFamily::INET6 as i16, + port: addr.port().to_be(), + flowinfo: addr.flowinfo().to_be(), + addr: addr.ip().octets(), + scope_id: addr.scope_id(), + } + } +} + /// Socket address structure for different address families. /// Currently only supports IPv4 (AF_INET). #[non_exhaustive] @@ -162,6 +199,11 @@ pub(super) struct SocketOptions { pub(super) reuse_address: bool, pub(super) keep_alive: bool, pub(super) broadcast: bool, + /// `SO_PASSCRED`: deliver the sender's `SCM_CREDENTIALS` with every + /// `recvmsg`. Only `AF_UNIX` sockets honour it. + pub(super) pass_cred: bool, + pub(super) receive_ipv4_returned_options: bool, + pub(super) receive_ipv4_ttl: bool, /// Receiving timeout, None (default value) means no timeout pub(super) recv_timeout: Option, /// Sending timeout, None (default value) means no timeout @@ -177,6 +219,11 @@ pub(super) struct SocketOptions { pub(crate) struct SocketOFlags(pub OFlags); pub(crate) struct SocketProxy(pub Arc>); +#[derive(Clone, Copy, Default)] +struct IcmpProxySocket { + peer: Option, +} + impl Clone for SocketProxy { fn clone(&self) -> Self { Self(self.0.clone()) @@ -188,6 +235,47 @@ pub(super) enum SocketOptionValue { U32(u32), } +/// What one receive delivered: the data length plus the ancillary payload a `recvmsg` +/// turns into control messages. +struct Received { + size: usize, + /// Descriptors transferred with the data (`SCM_RIGHTS`). + rights: alloc::vec::Vec>, + /// The sender's credentials (`SCM_CREDENTIALS`), present only when the receiving + /// `AF_UNIX` socket has `SO_PASSCRED` set. + credentials: Option, +} + +/// Appends one control message to `control` the way Linux's `put_cmsg` fills a user buffer +/// of `capacity` bytes: a header that no longer fits is dropped, a payload that no longer +/// fits is cut short (its `cmsg_len` says how much was written), and the cursor advances by +/// `CMSG_SPACE` or to the end of the buffer, whichever comes first. Returns whether anything +/// was truncated (`MSG_CTRUNC`). +fn put_cmsg( + control: &mut alloc::vec::Vec, + capacity: usize, + level: u32, + cmsg_type: u32, + data: &[u8], +) -> bool { + const CMSG_HEADER_LEN: usize = size_of::() + 2 * size_of::(); + let start = control.len(); + let remaining = capacity.saturating_sub(start); + if remaining < CMSG_HEADER_LEN { + return true; + } + let cmsg_len = CMSG_HEADER_LEN + data.len(); + let truncated = remaining < cmsg_len; + let write_len = cmsg_len.min(remaining); + control.extend_from_slice(&write_len.to_ne_bytes()); + control.extend_from_slice(&level.to_ne_bytes()); + control.extend_from_slice(&cmsg_type.to_ne_bytes()); + control.extend_from_slice(&data[..write_len - CMSG_HEADER_LEN]); + let cmsg_space = (cmsg_len + size_of::() - 1) & !(size_of::() - 1); + control.resize(start + cmsg_space.min(remaining), 0); + truncated +} + /// Socket-related implementation. Currently these methods are on `GlobalState` /// so that they can access `net` and the litebox descriptor table. This might /// change if the nature of the litebox descriptor table changes, or if network @@ -224,7 +312,12 @@ impl GlobalState { NetworkProxy::Datagram(proxy) } SockType::Raw => NetworkProxy::Raw, - _ => unimplemented!(), + SockType::SeqPacket => { + unreachable!("AF_UNIX sockets do not use the network proxy") + } + // `SockType` is `#[non_exhaustive]`; all currently declared variants are matched + // above. + _ => unreachable!(), }; // Save the proxy in both the descriptor table and the network subsystem so that the shim layer // can access it without holding the network lock and the network subsystem can access it without @@ -350,9 +443,6 @@ impl GlobalState { } (SocketOption::BROADCAST, SocketOptionValue::U32(val)) => { opt.broadcast = val != 0; - if val == 0 { - todo!("disable SO_BROADCAST"); - } } (SocketOption::KEEPALIVE, SocketOptionValue::U32(val)) => { let keep_alive = val != 0; @@ -378,10 +468,13 @@ impl GlobalState { litebox::net::errors::SetTcpOptionError::InvalidFd => { return Err(Errno::EBADF); } - litebox::net::errors::SetTcpOptionError::NotTcpSocket => { - unimplemented!("SO_KEEPALIVE is not supported for non-TCP sockets") - } - _ => unimplemented!(), + // Linux keeps SO_KEEPALIVE as a generic per-socket flag; only TCP's + // keepalive timer ever reads it, so UDP/raw sockets accept the call + // and it stays inert, same as on real Linux. + litebox::net::errors::SetTcpOptionError::NotTcpSocket => {} + // `SetTcpOptionError` is `#[non_exhaustive]` but only declares these two + // variants, both matched above. + _ => unreachable!(), } } Ok(()) @@ -392,6 +485,24 @@ impl GlobalState { match optname { SocketOptionName::IP(ip) => match ip { + // These IPv4 receive controls are advisory for the in-process stack. The bounded + // ICMP bridge returns only the echo message, and BusyBox ping does not consume + // ancillary TTL or returned-option data, so accepting them preserves Linux's + // socket setup behavior without exposing a raw-packet path. + litebox_common_linux::IpOption::RETOPTS + | litebox_common_linux::IpOption::RECVTTL => { + let enabled = super::read_from_user::(optval, optlen)? != 0; + self.with_socket_options_mut(fd, |options| match ip { + litebox_common_linux::IpOption::RETOPTS => { + options.receive_ipv4_returned_options = enabled; + } + litebox_common_linux::IpOption::RECVTTL => { + options.receive_ipv4_ttl = enabled; + } + litebox_common_linux::IpOption::TOS => unreachable!(), + }); + return Ok(()); + } // IP_TOS is an advisory traffic-class hint. Accept it (we don't // propagate the bit anywhere) instead of returning EOPNOTSUPP, // which Node's Socket.setTypeOfService treats as fatal and @@ -425,6 +536,11 @@ impl GlobalState { SocketOption::TYPE | SocketOption::PEERCRED | SocketOption::ERROR => { return Err(Errno::ENOPROTOOPT); } + // Credentials passing exists only for AF_UNIX sockets here. + SocketOption::PASSCRED => { + log_unsupported!("setsockopt(SO_PASSCRED) on an inet socket"); + return Err(Errno::EOPNOTSUPP); + } }, SocketOptionName::TCP(to) => match to { TcpOption::CONGESTION => { @@ -563,6 +679,12 @@ impl GlobalState { let val: u32 = match optname { SocketOptionName::IP(ipopt) => match ipopt { litebox_common_linux::IpOption::TOS => return Err(Errno::EOPNOTSUPP), + litebox_common_linux::IpOption::RETOPTS => self + .with_socket_options(fd, |options| options.receive_ipv4_returned_options) + .into(), + litebox_common_linux::IpOption::RECVTTL => self + .with_socket_options(fd, |options| options.receive_ipv4_ttl) + .into(), }, SocketOptionName::Socket(sopt) => match sopt { // handled by `getsockopt_common` @@ -590,6 +712,7 @@ impl GlobalState { litebox::net::SOCKET_BUFFER_SIZE.trunc() } SocketOption::PEERCRED => return Err(Errno::ENOPROTOOPT), + SocketOption::PASSCRED => return Err(Errno::EOPNOTSUPP), }, SocketOptionName::TCP(tcpopt) => { match tcpopt { @@ -605,7 +728,9 @@ impl GlobalState { litebox::net::CongestionControl::Reno => "reno", litebox::net::CongestionControl::Cubic => "cubic", litebox::net::CongestionControl::None => "none", - _ => unimplemented!(), + // `CongestionControl` is `#[non_exhaustive]` but only declares these + // three variants, all matched above. + _ => unreachable!(), }; let len = name.len().min(len as usize); optval @@ -655,7 +780,9 @@ impl GlobalState { self.net.lock().accept(fd, peer).map_err(|e| match e { AcceptError::NoConnectionsReady => TryOpError::TryAgain, AcceptError::InvalidFd | AcceptError::NotListening => TryOpError::Other(e.into()), - _ => unimplemented!(), + // `AcceptError` is `#[non_exhaustive]` but only declares these three variants, all + // matched above. + _ => unreachable!(), }) } @@ -678,6 +805,23 @@ impl GlobalState { .map_err(Errno::from) } + fn icmp_proxy_socket(&self, fd: &SocketFd) -> Option { + self.litebox + .descriptor_table() + .with_metadata(fd, |state: &IcmpProxySocket| *state) + .ok() + } + + fn set_icmp_proxy_peer(&self, fd: &SocketFd, peer: Ipv4Addr) -> Result<(), Errno> { + self.litebox + .descriptor_table_mut() + .with_metadata_mut(fd, |state: &mut IcmpProxySocket| state.peer = Some(peer)) + .map_err(|err| match err { + litebox::fd::MetadataError::NoSuchMetadata => Errno::ENOTSOCK, + litebox::fd::MetadataError::ClosedFd => Errno::EBADF, + }) + } + fn bind(&self, fd: &SocketFd, sockaddr: SocketAddr) -> Result<(), Errno> { self.net.lock().bind(fd, &sockaddr).map_err(Errno::from) } @@ -688,9 +832,46 @@ impl GlobalState { fd: &SocketFd, sockaddr: SocketAddr, ) -> Result<(), Errno> { + if self.icmp_proxy_socket(fd).is_some() { + let SocketAddr::V4(peer) = sockaddr else { + return Err(Errno::EAFNOSUPPORT); + }; + if peer.ip().is_unspecified() { + return Err(Errno::EADDRNOTAVAIL); + } + self.set_icmp_proxy_peer(fd, *peer.ip())?; + self.get_proxy(fd)?.set_state(SocketState::Connected); + return Ok(()); + } if sockaddr.port() == 0 || sockaddr.ip().is_unspecified() { return Err(Errno::ECONNREFUSED); } + if let SocketAddr::V4(peer) = sockaddr { + let ip = *peer.ip(); + let socket_type = self.get_socket_type(fd)?; + let so_broadcast = self.with_socket_options(fd, |opt| opt.broadcast); + let net = self.net.lock(); + // Non-unicast destinations. Nothing behind the interface answers a broadcast, + // multicast or link-local SYN (the platform side forwards unicast only), so the + // call would sit out the whole connect timeout. Linux refuses these up front: + // `tcp_v4_connect` reports a broadcast/multicast route as `ENETUNREACH`, and a UDP + // `connect` to a broadcast address needs `SO_BROADCAST` (`EACCES`); UDP multicast + // and link-local are allowed and simply go nowhere here. + let broadcast = ip == Ipv4Addr::BROADCAST || net.is_directed_broadcast(ip); + if broadcast || ip.is_multicast() || ip.is_link_local() { + match socket_type { + SockType::Stream => return Err(Errno::ENETUNREACH), + _ if broadcast && !so_broadcast => return Err(Errno::EACCES), + _ => {} + } + } + // Without an external interface (no `utun` on macOS) a SYN to anything but the + // guest's own addresses is silently dropped, so the connect would only ever end in + // the caller's timeout. Report the missing route immediately, as Linux does. + if !net.is_local_ip(ip) && !net.external_interface_available() { + return Err(Errno::ENETUNREACH); + } + } let mut check_progress = false; cx.wait_on_events::<_, Errno>( self.get_status(fd).contains(OFlags::NONBLOCK), @@ -731,8 +912,60 @@ impl GlobalState { flags: SendFlags, sockaddr: Option, ) -> Result { + let guest_len = buf.len(); + let icmp_state = self.icmp_proxy_socket(fd); + let framed = if let Some(state) = icmp_state { + const MAX_ICMP_ECHO_SIZE: usize = 4096; + if !(8..=MAX_ICMP_ECHO_SIZE).contains(&buf.len()) { + return Err(if buf.len() > MAX_ICMP_ECHO_SIZE { + Errno::EMSGSIZE + } else { + Errno::EINVAL + }); + } + // Only the bounded unprivileged-ping use case is bridged; this is not a general raw + // ICMP tunnel. + if buf[0] != 8 || buf[1] != 0 { + return Err(Errno::EOPNOTSUPP); + } + let target = match sockaddr { + Some(SocketAddr::V4(target)) => *target.ip(), + Some(SocketAddr::V6(_)) => return Err(Errno::EAFNOSUPPORT), + None => state.peer.ok_or(Errno::EDESTADDRREQ)?, + }; + if target.is_unspecified() || target.is_multicast() || target == Ipv4Addr::BROADCAST { + return Err(Errno::EINVAL); + } + let mut frame = alloc::vec::Vec::with_capacity( + litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN + buf.len(), + ); + frame.extend_from_slice(&target.octets()); + frame.extend_from_slice(buf); + Some(frame) + } else { + None + }; + let wire_buf = framed.as_deref().unwrap_or(buf); + let wire_sockaddr = if framed.is_some() { + Some(SocketAddr::V4(litebox::net::ICMP_ECHO_PROXY_ADDR)) + } else { + sockaddr + }; let proxy = self.get_proxy(fd)?; + // A datagram to a broadcast address needs `SO_BROADCAST`, as on Linux (`EACCES`). + if let (NetworkProxy::Datagram(_), Some(SocketAddr::V4(dest))) = + (proxy.as_ref(), wire_sockaddr) + { + let ip = *dest.ip(); + if ip.octets()[3] == 255 + && (ip == Ipv4Addr::BROADCAST || self.net.lock().is_directed_broadcast(ip)) + && !self.with_socket_options(fd, |opt| opt.broadcast) + { + return Err(Errno::EACCES); + } + } + // Auto-bind UDP sockets if not already bound (Linux behavior: sendto() on an unbound // UDP socket implicitly binds it to an ephemeral port before sending). // This is mostly lock-free: we only take the network lock if we need to allocate a port. @@ -754,9 +987,11 @@ impl GlobalState { // Another thread bound it in the meantime - that's fine } litebox::net::errors::BindError::InvalidFd => return Err(Errno::EBADF), + // `BindError` is `#[non_exhaustive]` but only declares these four variants, + // all matched here or above; `_` covers the same two named on the left. litebox::net::errors::BindError::UnsupportedAddress(_) - | litebox::net::errors::BindError::PortAlreadyInUse(_) => unreachable!(), - _ => unimplemented!(), + | litebox::net::errors::BindError::PortAlreadyInUse(_) + | _ => unreachable!(), } } // Get the assigned port @@ -782,25 +1017,52 @@ impl GlobalState { let timeout = self.with_socket_options(fd, |opt| opt.send_timeout); let is_nonblock = self.get_status(fd).contains(OFlags::NONBLOCK) || flags.contains(SendFlags::DONTWAIT); - let is_empty_stream = buf.is_empty() && matches!(proxy.as_ref(), NetworkProxy::Stream(_)); - - cx.with_timeout(timeout) - .wait_on_events( + let is_stream = matches!(proxy.as_ref(), NetworkProxy::Stream(_)); + let is_empty_stream = guest_len == 0 && is_stream; + + // A blocking stream send returns only once every byte is queued (Linux `tcp_sendmsg` + // loops, sleeping on send space): a short count comes back only when `SO_SNDTIMEO` + // or a signal cuts the wait after some bytes went in. Non-blocking and datagram sends + // are a single attempt. + let mut written = 0usize; + loop { + let chunk = &wire_buf[written..]; + let attempt = cx.with_timeout(timeout).wait_on_events( is_nonblock, Events::OUT, |observer, filter| { proxy.register_observer(observer, filter); Ok(()) }, - || match proxy.try_write(buf, new_flags, sockaddr) { - Ok(0) if buf.is_empty() => Ok(0), + || match proxy.try_write(chunk, new_flags, wire_sockaddr) { + Ok(n) if framed.is_some() && n == chunk.len() => Ok(guest_len), + Ok(_) if framed.is_some() => Err(TryOpError::Other(Errno::EIO)), + Ok(0) if guest_len == 0 => Ok(0), Ok(0) => Err(TryOpError::TryAgain), Ok(n) => Ok(n), Err(ChannelWriteError::BufferFull) if is_empty_stream => Ok(0), + // No room in the socket's send buffer: a blocking send waits for the + // network worker to drain it (bounded by `SO_SNDTIMEO`); only `O_NONBLOCK` + // / `MSG_DONTWAIT` turn this into `EAGAIN`, as on Linux. + Err(ChannelWriteError::BufferFull) => Err(TryOpError::TryAgain), Err(e) => Err(TryOpError::Other(Errno::from(e))), }, - ) - .map_err(Errno::from) + ); + match attempt { + Ok(n) => { + written += n; + if written >= wire_buf.len() || is_nonblock || !is_stream || framed.is_some() + { + return Ok(written); + } + } + Err(_) if written > 0 => return Ok(written), + // `SO_SNDTIMEO` ran out with nothing sent: Linux reports `EAGAIN` + // (`sk_stream_wait_memory`), keeping `ETIMEDOUT` for the connection itself. + Err(TryOpError::WaitError(WaitError::TimedOut)) => return Err(Errno::EAGAIN), + Err(e) => return Err(Errno::from(e)), + } + } } /// Receive data via socket channel (lock-free path). @@ -832,17 +1094,73 @@ impl GlobalState { // `MSG_TRUNC` behavior depends on the socket type if flags.contains(ReceiveFlags::TRUNC) { match self.get_socket_type(fd)? { - SockType::Datagram | SockType::Raw => { + SockType::Datagram | SockType::Raw | SockType::SeqPacket => { new_flags.insert(litebox::net::ReceiveFlags::TRUNC); } SockType::Stream => { new_flags.insert(litebox::net::ReceiveFlags::DISCARD); } - _ => unimplemented!(), + // `SockType` is `#[non_exhaustive]`; all currently declared variants are matched + // above. + _ => unreachable!(), } } let proxy = self.get_proxy(fd)?; + if self.icmp_proxy_socket(fd).is_some() { + let frame_len = buf + .len() + .checked_add(litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN) + .ok_or(Errno::EINVAL)?; + let mut frame = alloc::vec![0u8; frame_len]; + return cx + .with_timeout(timeout) + .wait_on_events( + is_nonblock, + Events::IN, + |observer, filter| { + proxy.register_observer(observer, filter); + Ok(()) + }, + || match proxy.try_read(&mut frame, new_flags, None) { + Ok(n) if n < litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN => { + Err(TryOpError::TryAgain) + } + Ok(n) => { + let payload_len = n - litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN; + let available = + n.min(frame.len()) - litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN; + let copied = available.min(buf.len()); + buf[..copied].copy_from_slice( + &frame[litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN + ..litebox::net::ICMP_ECHO_PROXY_PREFIX_LEN + copied], + ); + if let Some(source_addr) = source_addr.as_deref_mut() { + *source_addr = Some(SocketAddr::V4(SocketAddrV4::new( + Ipv4Addr::new(frame[0], frame[1], frame[2], frame[3]), + 0, + ))); + } + Ok(if flags.contains(ReceiveFlags::TRUNC) { + payload_len + } else { + copied + }) + } + Err(ChannelReadError::ReadShutdown) => Ok(0), + Err(ChannelReadError::ConnectionClosed) => { + match proxy.get_async_error(true) { + Some(err) => Err(TryOpError::Other(err.into())), + None => Ok(0), + } + } + Err(ChannelReadError::NotConnected) => { + Err(TryOpError::Other(Errno::ENOTCONN)) + } + }, + ) + .map_err(Errno::from); + } cx.with_timeout(timeout) .wait_on_events( is_nonblock, @@ -862,7 +1180,11 @@ impl GlobalState { Err(ChannelReadError::NotConnected) => Err(TryOpError::Other(Errno::ENOTCONN)), }, ) - .map_err(Errno::from) + .map_err(|err| match err { + // `SO_RCVTIMEO` ran out: Linux reports `EAGAIN` (`sock_rcvtimeo`), not `ETIMEDOUT`. + TryOpError::WaitError(WaitError::TimedOut) => Errno::EAGAIN, + err => Errno::from(err), + }) } fn get_socket_type(&self, fd: &SocketFd) -> Result { @@ -921,7 +1243,9 @@ impl GlobalState { Err(litebox::net::errors::CloseError::InvalidFd) => { Err(TryOpError::Other(Errno::EBADF)) } - Err(_) => unimplemented!(), + // `CloseError` is `#[non_exhaustive]` but only declares these two variants, both + // matched above. + Err(_) => unreachable!(), }, ) { Ok(()) => Ok(()), @@ -975,6 +1299,7 @@ impl Task { log_unsupported!("protocol = {protocol}"); Errno::EPROTONOSUPPORT })?; + let mut icmp_proxy = false; let protocol = match ty { SockType::Stream => { if !matches!(protocol, IPProtocol::Default | IPProtocol::TCP) { @@ -982,17 +1307,39 @@ impl Task { } litebox::net::Protocol::Tcp } - SockType::Datagram => { - if !matches!(protocol, IPProtocol::Default | IPProtocol::UDP) { - return Err(Errno::EINVAL); + SockType::Datagram => match protocol { + IPProtocol::Default | IPProtocol::UDP => litebox::net::Protocol::Udp, + IPProtocol::ICMP => { + icmp_proxy = true; + litebox::net::Protocol::Udp } - litebox::net::Protocol::Udp - } - SockType::Raw => todo!(), - _ => unimplemented!(), + IPProtocol::TCP | IPProtocol::RAW => return Err(Errno::EINVAL), + // `IPProtocol` is non-exhaustive; `try_from` admitted only a declared value. + _ => unreachable!(), + }, + // No raw IP networking of any kind is available: the host has no `tun` + // device without root (see `net_proxy`'s module doc), so there is no path + // to actually emit an ICMP echo or any other raw packet. `EPERM` matches + // what an unprivileged Linux process making this same call without + // `CAP_NET_RAW` sees, so `ping`'s own "Operation not permitted" is exactly + // what a caller would expect in a sandboxed environment -- a clean syscall + // failure instead of a guest-crashing panic. + SockType::Raw => return Err(Errno::EPERM), + SockType::SeqPacket => return Err(Errno::ESOCKTNOSUPPORT), + // `SockType` is `#[non_exhaustive]`; all currently declared variants are + // matched above. + _ => unreachable!(), }; let socket = self.global.net.lock().socket(protocol)?; let _ = self.global.initialize_socket(&socket, ty, flags); + if icmp_proxy { + let old = self + .global + .litebox + .descriptor_table_mut() + .set_entry_metadata(&socket, IcmpProxySocket::default()); + assert!(old.is_none()); + } let Ok(raw_fd) = files.insert_raw_fd(socket) else { unimplemented!() }; @@ -1000,7 +1347,7 @@ impl Task { } AddressFamily::UNIX => { let _ = UnixProtocol::try_from(protocol).map_err(|_| Errno::EPROTONOSUPPORT)?; - let socket = UnixSocket::new(ty, flags).ok_or(Errno::ESOCKTNOSUPPORT)?; + let socket = UnixSocket::new(ty, flags, self).ok_or(Errno::ESOCKTNOSUPPORT)?; let typed = self.global .litebox @@ -1020,8 +1367,38 @@ impl Task { Errno::EMFILE })? } - AddressFamily::INET6 | AddressFamily::NETLINK => return Err(Errno::EAFNOSUPPORT), - _ => unimplemented!(), + AddressFamily::NETLINK => { + // A `NETLINK_ROUTE` socket, just enough for `getifaddrs(3)` / + // `os.networkInterfaces()`. `ty` (SOCK_RAW/SOCK_DGRAM) and `protocol` + // (NETLINK_ROUTE) are accepted without distinction -- no real link or + // address state is ever touched; see `crate::syscalls::netlink`. + let (interface_ip, gateway_ip) = { + let net = self.global.net.lock(); + (net.interface_ip(), net.gateway_ip()) + }; + let socket = crate::syscalls::netlink::NetlinkSocket::new(interface_ip, gateway_ip); + let mut status = OFlags::RDWR; + status.set(OFlags::NONBLOCK, flags.contains(SockFlags::NONBLOCK)); + let mut descriptors = self.global.litebox.descriptor_table_mut(); + let typed = descriptors + .insert::>(socket); + let old = descriptors.set_entry_metadata(&typed, SocketOFlags(status)); + assert!(old.is_none()); + if flags.contains(SockFlags::CLOEXEC) { + let old = descriptors.set_fd_metadata(&typed, FileDescriptorFlags::FD_CLOEXEC); + assert!(old.is_none()); + } + drop(descriptors); + files.insert_raw_fd(typed).map_err(|typed| { + let _ = self.global.litebox.descriptor_table_mut().remove(&typed); + Errno::EMFILE + })? + } + AddressFamily::INET6 => return Err(Errno::EAFNOSUPPORT), + // `AddressFamily` is `#[non_exhaustive]` but only declares these four variants, all + // matched above; `domain` only reaches here via `AddressFamily::try_from`, which + // rejects anything else before construction. + _ => unreachable!(), }; Ok(u32::try_from(file).unwrap()) } @@ -1057,8 +1434,8 @@ impl Task { let (desc1, desc2) = match domain { AddressFamily::UNIX => { let _ = UnixProtocol::try_from(protocol).map_err(|_| Errno::EPROTONOSUPPORT)?; - let (sock1, sock2) = - UnixSocket::new_connected_pair(ty, flags).ok_or(Errno::ESOCKTNOSUPPORT)?; + let (sock1, sock2) = UnixSocket::new_connected_pair(ty, flags, self) + .ok_or(Errno::ESOCKTNOSUPPORT)?; let files = self.files.borrow(); let mut dt = self.global.litebox.descriptor_table_mut(); let typed1 = @@ -1132,12 +1509,22 @@ pub(crate) fn read_sockaddr_from_user( path[1..].to_vec(), ))); } - let s = CStr::from_bytes_until_nul(path).map_err(|_| Errno::EINVAL)?; + // The kernel bounds the pathname by `addrlen` with the NUL optional + // (`unix(7)`: "the terminating null byte is not required"); dbus and X both + // pass exactly `offsetof(sun_path) + strlen(path)`. Requiring an embedded NUL + // here rejected every such bind with EINVAL. + let end = path.iter().position(|&b| b == 0).unwrap_or(path.len()); Ok(SocketAddress::Unix(UnixSocketAddr::Path( - s.to_string_lossy().to_string(), + alloc::string::String::from_utf8_lossy(&path[..end]).to_string(), ))) } - _ => todo!("unsupported family {family:?}"), + // Unlike `do_socket`'s `AddressFamily` match, `INET6`/`NETLINK` really are reachable + // here: this parses a sockaddr the guest supplies as a syscall argument (e.g. to + // `connect`/`bind`), which is independent of whatever family the fd itself was created + // with, so a mismatched or IPv6 sockaddr is a real, guest-triggerable input rather than + // exhaustiveness padding. Report it the same way `do_socket` reports an unsupported + // socket domain instead of aborting on it. + _ => Err(Errno::EAFNOSUPPORT), } } @@ -1199,7 +1586,14 @@ pub(crate) fn write_sockaddr_to_user( } } } - SocketAddress::Inet(SocketAddr::V6(_)) => todo!("copy_sockaddr_to_user for IPv6"), + SocketAddress::Inet(SocketAddr::V6(v6_addr)) => { + let addrlen_val = size_of::().min(addrlen_val as usize); + let c_addr: CSockInet6Addr = v6_addr.into(); + let bytes: &[u8] = c_addr.as_bytes(); + addr.write_slice_at_offset::(0, &bytes[..addrlen_val]) + .ok_or(Errno::EFAULT)?; + size_of::() + } } .trunc(); addrlen @@ -1354,6 +1748,13 @@ impl Task { let Ok(sockfd) = u32::try_from(sockfd) else { return Err(Errno::EBADF); }; + // A `NETLINK_ROUTE` socket is bound to a `sockaddr_nl`, not the inet/unix + // address `read_sockaddr_from_user` understands -- so resolve it before + // parsing the address (which would otherwise reject AF_NETLINK). Accept it + // as a no-op: there is no per-socket netlink group state to register. + if self.netlink_fd(sockfd).is_some() { + return Ok(()); + } let sockaddr = read_sockaddr_from_user::(sockaddr, addrlen)?; self.do_bind(sockfd, sockaddr) } @@ -1372,6 +1773,52 @@ impl Task { ) } + /// If `sockfd` is a `NETLINK_ROUTE` socket, return its typed fd; otherwise + /// `None` (a non-netlink or absent fd falls through to the normal socket path). + fn netlink_fd( + &self, + sockfd: u32, + ) -> Option< + alloc::sync::Arc< + litebox::fd::TypedFd>, + >, + > { + self.files + .borrow() + .raw_descriptor_store + .read() + .fd_from_raw_integer::>( + sockfd as usize, + ) + .ok() + } + + /// `send`/`sendto`/`write` on a netlink socket: enqueue the dump the matching + /// reads will drain. `Some` iff `sockfd` is a netlink socket. + pub(crate) fn netlink_send(&self, sockfd: u32, buf: &[u8]) -> Option> { + let nl = self.netlink_fd(sockfd)?; + Some( + self.global + .litebox + .descriptor_table() + .with_entry(&nl, |sock| sock.handle_send(buf)) + .ok_or(Errno::EBADF), + ) + } + + /// `recv`/`recvfrom`/`read` on a netlink socket: drain pending dump bytes. + /// `Some` iff `sockfd` is a netlink socket. + pub(crate) fn netlink_recv(&self, sockfd: u32, buf: &mut [u8]) -> Option> { + let nl = self.netlink_fd(sockfd)?; + Some( + self.global + .litebox + .descriptor_table() + .with_entry(&nl, |sock| sock.handle_recv(buf)) + .unwrap_or(Err(Errno::EBADF)), + ) + } + /// Handle syscall `listen` pub(crate) fn sys_listen(&self, sockfd: i32, backlog: u16) -> Result<(), Errno> { let Ok(sockfd) = u32::try_from(sockfd) else { @@ -1384,7 +1831,7 @@ impl Task { &self.global, sockfd, |fd| self.global.listen(fd, backlog), - |file| file.listen(backlog, &self.global), + |file| file.listen(backlog, &self.global, self), ) } @@ -1414,6 +1861,9 @@ impl Task { flags: SendFlags, sockaddr: Option, ) -> Result { + if let Some(res) = self.netlink_send(sockfd, buf) { + return res; + } let res = self.files.borrow().with_socket( &self.global, sockfd, @@ -1460,6 +1910,23 @@ impl Task { msg: &litebox_common_linux::UserMsgHdr, flags: SendFlags, ) -> Result { + // A netlink socket takes `sendmsg` exactly like `sendto`: iproute2/busybox `ip`'s + // `rtnl_talk` (`ip route get`) sends with a `sockaddr_nl` name, which the inet + // address parse below rejects as `EAFNOSUPPORT`, so it is answered first. + if self.netlink_fd(sockfd).is_some() { + let data = if msg.msg_iovlen == 0 { + alloc::vec::Vec::new() + } else { + let iovs = msg + .msg_iov + .to_owned_slice::(msg.msg_iovlen) + .ok_or(Errno::EFAULT)?; + copy_iovs_to_vec::(&iovs)? + }; + return self + .netlink_send(sockfd, &data) + .unwrap_or(Err(Errno::EBADF)); + } let msg_name = msg.msg_name; let sock_addr = if msg_name.as_usize() != 0 { Some(read_sockaddr_from_user::( @@ -1469,9 +1936,105 @@ impl Task { } else { None }; - if msg.msg_controllen != 0 { - log_unsupported!("ancillary data is not supported"); - return Err(Errno::EINVAL); + let msg_control = msg.msg_control; + let msg_controllen = msg.msg_controllen; + let mut rights = alloc::vec::Vec::new(); + // An explicit `SCM_CREDENTIALS` the sender attached (at most one), validated below + // the way Linux's `scm_check_creds` does: an unprivileged sender may only claim its + // own process id and one of its real/effective/saved ids. + let mut credentials: Option = None; + if msg_controllen != 0 { + const SOL_SOCKET: u32 = 1; + const SCM_RIGHTS: u32 = 1; + const SCM_CREDENTIALS: u32 = 2; + const SCM_MAX_FD: usize = 253; + const CMSG_HEADER_LEN: usize = size_of::() + 2 * size_of::(); + + let control = msg_control + .to_owned_slice::(msg_controllen) + .ok_or(Errno::EFAULT)?; + let read_usize = |bytes: &[u8], offset: usize| -> Result { + let end = offset + .checked_add(size_of::()) + .ok_or(Errno::EINVAL)?; + let bytes: [u8; size_of::()] = bytes + .get(offset..end) + .ok_or(Errno::EINVAL)? + .try_into() + .map_err(|_| Errno::EINVAL)?; + Ok(usize::from_ne_bytes(bytes)) + }; + let read_u32 = |bytes: &[u8], offset: usize| -> Result { + let end = offset.checked_add(size_of::()).ok_or(Errno::EINVAL)?; + let bytes: [u8; size_of::()] = bytes + .get(offset..end) + .ok_or(Errno::EINVAL)? + .try_into() + .map_err(|_| Errno::EINVAL)?; + Ok(u32::from_ne_bytes(bytes)) + }; + + let mut offset = 0usize; + while offset < control.len() { + let remaining = control.len() - offset; + if remaining < CMSG_HEADER_LEN { + return Err(Errno::EINVAL); + } + let cmsg_len = read_usize(&control, offset)?; + if !(CMSG_HEADER_LEN..=remaining).contains(&cmsg_len) { + return Err(Errno::EINVAL); + } + let level_offset = offset + size_of::(); + let type_offset = level_offset + size_of::(); + let cmsg_level = read_u32(&control, level_offset)?; + let cmsg_type = read_u32(&control, type_offset)?; + let data = &control[offset + CMSG_HEADER_LEN..offset + cmsg_len]; + + match (cmsg_level, cmsg_type) { + (SOL_SOCKET, SCM_RIGHTS) => { + if data.is_empty() || data.len() % size_of::() != 0 { + return Err(Errno::EINVAL); + } + let count = data.len() / size_of::(); + if rights.len().checked_add(count).ok_or(Errno::EINVAL)? > SCM_MAX_FD { + return Err(Errno::EINVAL); + } + for raw_fd in data.chunks_exact(size_of::()) { + let raw_fd = i32::from_ne_bytes(raw_fd.try_into().unwrap()); + rights.push(self.transfer_fd(raw_fd)?); + } + } + (SOL_SOCKET, SCM_CREDENTIALS) => { + if credentials.is_some() + || data.len() != size_of::() + { + return Err(Errno::EINVAL); + } + let supplied = litebox_common_linux::Ucred { + pid: read_u32(data, 0)?, + uid: read_u32(data, size_of::())?, + gid: read_u32(data, 2 * size_of::())?, + }; + let own = self.credentials.borrow(); + let uid_ok = [own.uid, own.euid, own.suid].contains(&supplied.uid); + let gid_ok = [own.gid, own.egid, own.sgid].contains(&supplied.gid); + if supplied.pid != self.pid.cast_unsigned() || !uid_ok || !gid_ok { + return Err(Errno::EPERM); + } + credentials = Some(supplied); + } + _ => return Err(Errno::EINVAL), + } + + let aligned = cmsg_len + .checked_add(size_of::() - 1) + .ok_or(Errno::EINVAL)? + & !(size_of::() - 1); + if aligned >= remaining { + break; + } + offset = offset.checked_add(aligned).ok_or(Errno::EINVAL)?; + } } if msg.msg_iovlen > UIO_MAXIOV { return Err(Errno::EMSGSIZE); @@ -1485,10 +2048,14 @@ impl Task { .ok_or(Errno::EFAULT)?, ) }; + let rights = core::cell::RefCell::new(Some(rights)); let res = self.files.borrow().with_socket( &self.global, sockfd, |fd| { + if credentials.is_some() || !rights.borrow().as_ref().unwrap().is_empty() { + return Err(Errno::EINVAL); + } let sock_addr = sock_addr .clone() .map(|addr| addr.inet().ok_or(Errno::EAFNOSUPPORT)) @@ -1503,7 +2070,14 @@ impl Task { .map(|addr| addr.unix().ok_or(Errno::EAFNOSUPPORT)) .transpose()?; let data = copy_iovs_to_vec::(iovs.as_deref().unwrap_or_default())?; - file.sendto(self, &data, flags, unix_addr) + file.sendmsg( + self, + &data, + flags, + unix_addr, + rights.borrow_mut().take().unwrap(), + credentials, + ) }, ); if let Err(Errno::EPIPE) = res @@ -1625,10 +2199,31 @@ impl Task { flags: ReceiveFlags, source_addr: Option<&mut Option>, ) -> Result { + self.do_recvfrom_with_rights(sockfd, buf, flags, source_addr) + .map(|received| received.size) + } + + /// Receives into `buf` together with the ancillary payload a `recvmsg` would deliver: + /// the transferred descriptors (`SCM_RIGHTS`) and, when the receiving `AF_UNIX` socket + /// has `SO_PASSCRED` set, the sender's `SCM_CREDENTIALS`. + fn do_recvfrom_with_rights( + &self, + sockfd: u32, + buf: &mut [u8], + flags: ReceiveFlags, + source_addr: Option<&mut Option>, + ) -> Result, Errno> { + if let Some(res) = self.netlink_recv(sockfd, buf) { + return res.map(|size| Received { + size, + rights: alloc::vec::Vec::new(), + credentials: None, + }); + } let want_source = source_addr.is_some(); let files = self.files.borrow(); let raw_fd = usize::try_from(sockfd).or(Err(Errno::EBADF))?; - let (size, addr) = { + let (received, addr) = { // We need to do this cell dance because otherwise Rust can't recognize that the two // closures are mutually exclusive. let buf: core::cell::RefCell<&mut [u8]> = core::cell::RefCell::new(buf); @@ -1645,18 +2240,32 @@ impl Task { if want_source { Some(&mut addr) } else { None }, )?; let src_addr = addr.map(SocketAddress::Inet); - Ok((size, src_addr)) + Ok(( + Received { + size, + rights: alloc::vec::Vec::new(), + credentials: None, + }, + src_addr, + )) }, |entry| { let mut addr = None; - let size = entry.recvfrom( + let result = entry.recvmsg( &self.wait_cx(), &mut buf.borrow_mut(), flags, if want_source { Some(&mut addr) } else { None }, )?; let src_addr = addr.map(SocketAddress::Unix); - Ok((size, src_addr)) + Ok(( + Received { + size: result.size, + rights: result.rights, + credentials: result.credentials, + }, + src_addr, + )) }, )? }; @@ -1664,7 +2273,7 @@ impl Task { if let (Some(source_addr), Some(addr)) = (source_addr, addr) { *source_addr = Some(addr); } - Ok(size) + Ok(received) } /// Handle syscall `recvmsg` @@ -1678,7 +2287,8 @@ impl Task { return Err(Errno::EBADF); }; - let supported_flags = ReceiveFlags::DONTWAIT | ReceiveFlags::TRUNC; + let supported_flags = + ReceiveFlags::DONTWAIT | ReceiveFlags::TRUNC | ReceiveFlags::CMSG_CLOEXEC; if flags.intersects(supported_flags.complement()) { log_unsupported!("Unsupported recvmsg flags: {:?}", flags); return Err(Errno::EINVAL); @@ -1692,17 +2302,20 @@ impl Task { msg_ptr: UserPtrMut, flags: ReceiveFlags, ) -> Result { + const SOL_SOCKET: u32 = 1; + const SCM_RIGHTS: u32 = 1; + const SCM_CREDENTIALS: u32 = 2; + const CMSG_HEADER_LEN: usize = size_of::() + 2 * size_of::(); + let msg = msg_ptr.read_at_offset::(0).ok_or(Errno::EFAULT)?; // Copy fields out of the packed struct to avoid unaligned references. let msg_name = msg.msg_name; let msg_iov = msg.msg_iov; let msg_iovlen = msg.msg_iovlen; + let msg_control = msg.msg_control; let msg_controllen = msg.msg_controllen; - if msg_controllen != 0 { - log_unsupported!("ancillary data is not supported"); - } if msg_iovlen > UIO_MAXIOV { return Err(Errno::EMSGSIZE); } @@ -1727,10 +2340,14 @@ impl Task { .map_err(|_| Errno::ENOMEM)?; buffer.resize(total_iov_capacity, 0); let recv_buf = &mut buffer[..]; - let size = self.do_recvfrom( + let Received { + size, + rights, + credentials, + } = self.do_recvfrom_with_rights( sockfd, recv_buf, - flags, + flags.difference(ReceiveFlags::CMSG_CLOEXEC), if want_source { Some(&mut source_addr) } else { @@ -1773,7 +2390,24 @@ impl Task { msg_ptr.as_usize() + core::mem::offset_of!(litebox_common_linux::UserMsgHdr, msg_namelen), ); - if let Some(src_addr) = source_addr { + if self.netlink_fd(sockfd).is_some() { + // A `recvmsg` on a netlink socket reports a `sockaddr_nl` source: + // `{ u16 nl_family = AF_NETLINK, u16 pad, u32 nl_pid = 0 (from the + // kernel), u32 nl_groups = 0 }`. iproute2/busybox `ip` reject a + // reply whose sender length isn't `sizeof(sockaddr_nl)` or whose + // `nl_pid` is nonzero, so this must be present and zero-pid. + let mut nl = [0u8; 12]; + nl[0..2].copy_from_slice(&(AddressFamily::NETLINK as u16).to_ne_bytes()); + let cap = msg.msg_namelen as usize; + let n = nl.len().min(cap); + msg_name + .write_slice_at_offset::(0, &nl[..n]) + .ok_or(Errno::EFAULT)?; + // `sizeof(struct sockaddr_nl)` == 12. + addrlen_ptr + .write_at_offset::(0, 12u32) + .ok_or(Errno::EFAULT)?; + } else if let Some(src_addr) = source_addr { write_sockaddr_to_user::(src_addr, msg_name, addrlen_ptr)?; } else { // No source address (e.g. connected stream socket) — zero out msg_namelen. @@ -1783,21 +2417,119 @@ impl Task { } } - // Ancillary data is not supported, so report that no control bytes were delivered. + // Ancillary data, in the order Linux's `scm_recv` emits it: SCM_CREDENTIALS (only + // when the receiving socket has SO_PASSCRED) first, then SCM_RIGHTS. Each control + // message consumes CMSG_SPACE of the caller's buffer; whatever no longer fits is + // truncated (descriptors that don't fit are closed, not installed) and reported + // through MSG_CTRUNC. A null control pointer offers no space at all. + let control_capacity = if msg_control.as_usize() == 0 { + 0 + } else { + msg_controllen + }; + let mut control = alloc::vec::Vec::new(); + if let Some(ucred) = credentials { + let mut payload = [0u8; 3 * size_of::()]; + payload[..4].copy_from_slice(&ucred.pid.to_ne_bytes()); + payload[4..8].copy_from_slice(&ucred.uid.to_ne_bytes()); + payload[8..].copy_from_slice(&ucred.gid.to_ne_bytes()); + litebox_util_log::trace!( + sockfd, + sender_pid = ucred.pid, + sender_uid = ucred.uid, + sender_gid = ucred.gid, + control_capacity; + "recvmsg: delivering SCM_CREDENTIALS" + ); + if put_cmsg( + &mut control, + control_capacity, + SOL_SOCKET, + SCM_CREDENTIALS, + &payload, + ) { + ret_flags.insert(ReceiveFlags::CTRUNC); + } + } + + let rights_count = rights.len(); + let max_control_fds = control_capacity + .saturating_sub(control.len()) + .checked_sub(CMSG_HEADER_LEN) + .map_or(0, |space| space / size_of::()); + let cloexec = flags.contains(ReceiveFlags::CMSG_CLOEXEC); + let mut installed: alloc::vec::Vec<(usize, i32)> = alloc::vec::Vec::new(); + for transferred in rights.into_iter().take(max_control_fds) { + match self.install_transferred_fd(transferred, cloexec) { + Ok(raw_fd) => { + let Ok(guest_fd) = i32::try_from(raw_fd) else { + let _ = self.do_close(raw_fd); + break; + }; + installed.push((raw_fd, guest_fd)); + } + Err(Errno::EMFILE) => break, + Err(error) => { + for (raw_fd, _) in installed.drain(..) { + let _ = self.do_close(raw_fd); + } + return Err(error); + } + } + } + if installed.len() < rights_count { + ret_flags.insert(ReceiveFlags::CTRUNC); + } + if !installed.is_empty() { + let mut payload = alloc::vec::Vec::with_capacity(installed.len() * size_of::()); + for (_, guest_fd) in &installed { + payload.extend_from_slice(&guest_fd.to_ne_bytes()); + } + if put_cmsg( + &mut control, + control_capacity, + SOL_SOCKET, + SCM_RIGHTS, + &payload, + ) { + ret_flags.insert(ReceiveFlags::CTRUNC); + } + } + + let control_len = control.len(); + if control_len != 0 { + let control_ptr = UserPtrMut::::from_usize(msg_control.as_usize()); + if control_ptr + .write_slice_at_offset::(0, &control) + .is_none() + { + for (raw_fd, _) in installed.drain(..) { + let _ = self.do_close(raw_fd); + } + return Err(Errno::EFAULT); + } + } + let controllen_offset = core::mem::offset_of!(litebox_common_linux::UserMsgHdr, msg_controllen); let controllen_ptr = UserPtrMut::::from_usize(msg_ptr.as_usize() + controllen_offset); - controllen_ptr - .write_at_offset::(0, 0) - .ok_or(Errno::EFAULT)?; - - // Write back msg_flags with any status flags (e.g. MSG_TRUNC). let flags_offset = core::mem::offset_of!(litebox_common_linux::UserMsgHdr, msg_flags); let flags_ptr = UserPtrMut::::from_usize(msg_ptr.as_usize() + flags_offset); - flags_ptr - .write_at_offset::(0, ret_flags) - .ok_or(Errno::EFAULT)?; + let metadata_result = controllen_ptr + .write_at_offset::(0, control_len) + .ok_or(Errno::EFAULT) + .and_then(|()| { + flags_ptr + .write_at_offset::(0, ret_flags) + .ok_or(Errno::EFAULT) + }); + if let Err(error) = metadata_result { + for (raw_fd, _) in installed { + let _ = self.do_close(raw_fd); + } + return Err(error); + } Ok(total_received) } @@ -1811,8 +2543,10 @@ impl Task { flags: ReceiveFlags, timeout: litebox_common_linux::TimeParam, ) -> Result { - let supported_flags = - ReceiveFlags::DONTWAIT | ReceiveFlags::TRUNC | ReceiveFlags::WAITFORONE; + let supported_flags = ReceiveFlags::DONTWAIT + | ReceiveFlags::TRUNC + | ReceiveFlags::WAITFORONE + | ReceiveFlags::CMSG_CLOEXEC; if flags.intersects(supported_flags.complement()) { log_unsupported!("Unsupported recvmmsg flags: {:?}", flags); return Err(Errno::EINVAL); @@ -1961,8 +2695,8 @@ impl Task { return Err(Errno::EBADF); }; let optname = SocketOptionName::try_from(level, optname).ok_or_else(|| { - log_unsupported!("setsockopt(level = {level}, optname = {optname})"); - Errno::EINVAL + log_unsupported!("getsockopt(level = {level}, optname = {optname})"); + Errno::ENOPROTOOPT })?; let len = optlen.read_at_offset::(0).ok_or(Errno::EFAULT)?; if len > i32::MAX as u32 { @@ -2068,10 +2802,19 @@ impl Task { self.files.borrow().with_socket( &self.global, sockfd, - |_fd| { - ShutdownHow::try_from(how).map_err(|_| Errno::EINVAL)?; - log_unsupported!("shutdown on inet socket"); - Err(Errno::EOPNOTSUPP) + |fd| { + let how = ShutdownHow::try_from(how).map_err(|_| Errno::EINVAL)?; + self.global + .net + .lock() + .shutdown(fd, how.is_shutdown_read(), how.is_shutdown_write()) + .map_err(|err| match err { + litebox::net::ShutdownError::InvalidFd => Errno::EBADF, + litebox::net::ShutdownError::NotConnected => Errno::ENOTCONN, + // `ShutdownError` is `#[non_exhaustive]`; both declared variants are + // matched above. + _ => Errno::EINVAL, + }) }, |file| { let how = ShutdownHow::try_from(how).map_err(|_| Errno::EINVAL)?; @@ -2104,7 +2847,7 @@ mod tests { use crate::{ UserPtr, UserPtrMut, syscalls::{ - net::{CSockInetAddr, read_sockaddr_from_user}, + net::{CSockInet6Addr, CSockInetAddr, read_sockaddr_from_user, write_sockaddr_to_user}, tests::init_platform, }, }; @@ -2146,10 +2889,7 @@ mod tests { } fn epoll_add(task: &TestTask, epfd: i32, target_fd: u32, events: litebox::event::Events) { - let ev = litebox_common_linux::EpollEvent { - events: events.bits(), - data: u64::from(target_fd), - }; + let ev = litebox_common_linux::EpollEvent::new(events.bits(), u64::from(target_fd)); let ev_ptr = (&raw const ev).cast::(); let ev_const = UserPtr::from_usize(ev_ptr as usize); task.sys_epoll_ctl( @@ -2234,7 +2974,7 @@ mod tests { if is_nonblocking { // wait on epoll for server to be readable (incoming connection) - let mut events = [litebox_common_linux::EpollEvent { events: 0, data: 0 }; 2]; + let mut events = [litebox_common_linux::EpollEvent::new(0, 0); 2]; let n = epoll_wait(task, epfd, &mut events); assert_eq!(n, 1); for ev in &events[..n] { @@ -2310,7 +3050,7 @@ mod tests { "recvfrom" | "recvmsg" => { if is_nonblocking { epoll_add(task, epfd, client_fd, litebox::event::Events::IN); - let mut events = [litebox_common_linux::EpollEvent { events: 0, data: 0 }; 2]; + let mut events = [litebox_common_linux::EpollEvent::new(0, 0); 2]; let n = epoll_wait(task, epfd, &mut events); for ev in &events[..n] { assert!(ev.events & litebox::event::Events::IN.bits() != 0); @@ -2586,7 +3326,7 @@ mod tests { recv_flags.insert(ReceiveFlags::TRUNC); } if is_nonblocking { - let mut events = [litebox_common_linux::EpollEvent { events: 0, data: 0 }; 2]; + let mut events = [litebox_common_linux::EpollEvent::new(0, 0); 2]; let n = epoll_wait(task, epfd, &mut events); assert_eq!(n, 1); for ev in &events[..n] { @@ -2847,6 +3587,148 @@ mod tests { close_socket(&task, socket_fd); close_socket(&task, socket_fd2); } + + #[test] + fn test_setsockopt_broadcast_disable() { + let task = init_platform(None); + let sockfd = task + .do_socket( + AddressFamily::INET, + SockType::Datagram, + SockFlags::empty(), + 0, + ) + .expect("failed to create socket"); + + let val: u32 = 1; + let optval = UserPtr::from_usize((&raw const val).cast::() as usize); + task.do_setsockopt( + sockfd, + SocketOptionName::Socket(SocketOption::BROADCAST), + optval, + core::mem::size_of::(), + ) + .expect("failed to enable SO_BROADCAST"); + + let val: u32 = 0; + let optval = UserPtr::from_usize((&raw const val).cast::() as usize); + task.do_setsockopt( + sockfd, + SocketOptionName::Socket(SocketOption::BROADCAST), + optval, + core::mem::size_of::(), + ) + .expect("disabling SO_BROADCAST should succeed, not just enabling it"); + + let mut result: u32 = 0xDEAD; + let optval_out = UserPtrMut::from_usize((&raw mut result).cast::() as usize); + let len = task + .do_getsockopt( + sockfd, + SocketOptionName::Socket(SocketOption::BROADCAST), + optval_out, + core::mem::size_of::().trunc(), + ) + .expect("failed to get SO_BROADCAST"); + assert_eq!(len, core::mem::size_of::()); + assert_eq!(result, 0, "SO_BROADCAST should reflect the disabled value"); + + close_socket(&task, sockfd); + } + + #[test] + fn test_setsockopt_keepalive_udp_is_accepted_noop() { + let task = init_platform(None); + let sockfd = task + .do_socket( + AddressFamily::INET, + SockType::Datagram, + SockFlags::empty(), + 0, + ) + .expect("failed to create socket"); + + let val: u32 = 1; + let optval = UserPtr::from_usize((&raw const val).cast::() as usize); + task.do_setsockopt( + sockfd, + SocketOptionName::Socket(SocketOption::KEEPALIVE), + optval, + core::mem::size_of::(), + ) + .expect("SO_KEEPALIVE on a UDP socket should be accepted, like real Linux"); + + let mut result: u32 = 0xDEAD; + let optval_out = UserPtrMut::from_usize((&raw mut result).cast::() as usize); + let len = task + .do_getsockopt( + sockfd, + SocketOptionName::Socket(SocketOption::KEEPALIVE), + optval_out, + core::mem::size_of::().trunc(), + ) + .expect("failed to get SO_KEEPALIVE"); + assert_eq!(len, core::mem::size_of::()); + assert_eq!(result, 1, "the accepted value should still read back"); + + close_socket(&task, sockfd); + } + + #[test] + fn test_write_sockaddr_to_user_ipv6() { + let addr = core::net::SocketAddrV6::new( + core::net::Ipv6Addr::new(0x2001, 0x0db8, 0, 0, 0, 0, 0, 1), + 8080, + 0x1234_5678, + 7, + ); + + let mut buf = [0u8; 32]; + let mut addrlen: u32 = buf.len().trunc(); + write_sockaddr_to_user::( + SocketAddress::Inet(SocketAddr::V6(addr)), + UserPtrMut::from_usize(buf.as_mut_ptr() as usize), + UserPtrMut::from_usize((&raw mut addrlen) as usize), + ) + .expect("write_sockaddr_to_user for IPv6 should succeed"); + + assert_eq!( + addrlen, + u32::try_from(core::mem::size_of::()).unwrap() + ); + assert_eq!( + u16::from_ne_bytes([buf[0], buf[1]]), + AddressFamily::INET6 as u16 + ); + assert_eq!(u16::from_be_bytes([buf[2], buf[3]]), 8080); + assert_eq!( + u32::from_be_bytes([buf[4], buf[5], buf[6], buf[7]]), + 0x1234_5678 + ); + assert_eq!(&buf[8..24], &addr.ip().octets()); + assert_eq!( + u32::from_ne_bytes([buf[24], buf[25], buf[26], buf[27]]), + 7, + "scope_id is a local interface index, not swapped to network byte order" + ); + + // A too-small buffer still succeeds, copies only what fits, and reports the + // true (untruncated) size back through addrlen -- mirroring the pre-existing + // IPv4 truncation behavior in this same function. + let mut small_buf = [0xAAu8; 10]; + let mut small_addrlen: u32 = small_buf.len().trunc(); + write_sockaddr_to_user::( + SocketAddress::Inet(SocketAddr::V6(addr)), + UserPtrMut::from_usize(small_buf.as_mut_ptr() as usize), + UserPtrMut::from_usize((&raw mut small_addrlen) as usize), + ) + .expect("truncated write_sockaddr_to_user for IPv6 should still succeed"); + assert_eq!( + small_addrlen, + u32::try_from(core::mem::size_of::()).unwrap() + ); + assert_eq!(&small_buf[..], &buf[..10]); + } } #[cfg(test)] @@ -2861,9 +3743,10 @@ mod unix_tests { use alloc::{string::ToString, vec::Vec}; use litebox::event::Events; use litebox_common_linux::{ - AddressFamily, AtFlags, ReceiveFlags, SendFlags, SockFlags, SockType, SocketOption, - SocketOptionName, TimeParam, errno::Errno, + AddressFamily, AtFlags, FileDescriptorFlags, ReceiveFlags, SendFlags, SockFlags, SockType, + SocketOption, SocketOptionLevel, SocketOptionName, TimeParam, errno::Errno, }; + use zerocopy::FromZeros as _; use crate::{ UserPtr, UserPtrMut, @@ -2895,6 +3778,215 @@ mod unix_tests { .expect("close socket failed"); } + #[test] + fn unknown_getsockopt_returns_enoprotoopt() { + const SO_PEERPIDFD: u32 = 77; + + let task = init_platform(None); + let socket = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let mut value = 0u32; + let mut len = u32::try_from(core::mem::size_of_val(&value)).unwrap(); + + let error = task + .sys_getsockopt( + socket.try_into().unwrap(), + SocketOptionLevel::SOCKET as u32, + SO_PEERPIDFD, + UserPtrMut::from_usize((&raw mut value).cast::() as usize), + UserPtrMut::from_usize(&raw mut len as usize), + ) + .unwrap_err(); + + assert_eq!(error, Errno::ENOPROTOOPT); + close_socket(&task, socket); + } + + #[test] + fn unix_stream_peercred_uses_credentials_at_listen() { + let task = init_platform(None); + let name = b"unix_peercred_listen_time".to_vec(); + let server = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_bind( + server, + SocketAddress::Unix(UnixSocketAddr::Abstract(name.clone())), + ) + .expect("bind server socket"); + + task.sys_setuid(1000) + .expect("change credentials before listen"); + task.do_listen(server, 1).expect("listen on server socket"); + + let client = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_connect(client, SocketAddress::Unix(UnixSocketAddr::Abstract(name))) + .expect("connect client socket"); + let accepted = task + .do_accept(server, None, SockFlags::empty()) + .expect("accept client socket"); + + let peer_cred = |fd| { + let mut value = litebox_common_linux::Ucred { + pid: u32::MAX, + uid: u32::MAX, + gid: u32::MAX, + }; + let len = task + .do_getsockopt( + fd, + SocketOptionName::Socket(SocketOption::PEERCRED), + UserPtrMut::from_usize((&raw mut value).cast::() as usize), + u32::try_from(core::mem::size_of_val(&value)).unwrap(), + ) + .expect("getsockopt SO_PEERCRED"); + assert_eq!(len, core::mem::size_of_val(&value)); + value + }; + let server_cred = peer_cred(client); + let client_cred = peer_cred(accepted); + let expected_pid = u32::try_from(task.pid).unwrap(); + assert_eq!((server_cred.pid, server_cred.uid), (expected_pid, 1000)); + assert_eq!((client_cred.pid, client_cred.uid), (expected_pid, 1000)); + + close_socket(&task, accepted); + close_socket(&task, client); + close_socket(&task, server); + } + + #[test] + fn scm_rights_transfers_open_description_and_receive_cloexec() { + const SOL_SOCKET: u32 = 1; + const SCM_RIGHTS: u32 = 1; + const CMSG_HEADER_LEN: usize = + core::mem::size_of::() + 2 * core::mem::size_of::(); + const CMSG_LEN_ONE_FD: usize = CMSG_HEADER_LEN + core::mem::size_of::(); + const CMSG_SPACE_ONE_FD: usize = (CMSG_LEN_ONE_FD + core::mem::size_of::() - 1) + & !(core::mem::size_of::() - 1); + + for receive_cloexec in [false, true] { + let task = init_platform(None); + let (bus_sender, bus_receiver) = task + .do_socketpair(AddressFamily::UNIX, SockType::Stream, SockFlags::empty(), 0) + .expect("transport socketpair failed"); + let (payload_sender, payload_peer) = task + .do_socketpair(AddressFamily::UNIX, SockType::Stream, SockFlags::CLOEXEC, 0) + .expect("payload socketpair failed"); + + let data = [b'F']; + let send_iov = [litebox_common_linux::IoVec { + iov_base: UserPtrMut::from_usize(data.as_ptr() as usize), + iov_len: data.len(), + }]; + let mut send_control = [0u8; CMSG_SPACE_ONE_FD]; + send_control[..core::mem::size_of::()] + .copy_from_slice(&CMSG_LEN_ONE_FD.to_ne_bytes()); + let level_offset = core::mem::size_of::(); + let type_offset = level_offset + core::mem::size_of::(); + send_control[level_offset..type_offset].copy_from_slice(&SOL_SOCKET.to_ne_bytes()); + send_control[type_offset..CMSG_HEADER_LEN].copy_from_slice(&SCM_RIGHTS.to_ne_bytes()); + send_control[CMSG_HEADER_LEN..CMSG_LEN_ONE_FD] + .copy_from_slice(&i32::try_from(payload_sender).unwrap().to_ne_bytes()); + let mut send_header = litebox_common_linux::UserMsgHdr::new_zeroed(); + send_header.msg_iov = UserPtr::from_usize(send_iov.as_ptr() as usize); + send_header.msg_iovlen = send_iov.len(); + send_header.msg_control = UserPtr::from_usize(send_control.as_ptr() as usize); + send_header.msg_controllen = send_control.len(); + + task.sys_sendmsg( + i32::try_from(bus_sender).unwrap(), + UserPtr::from_usize(&raw const send_header as usize), + SendFlags::NOSIGNAL, + ) + .expect("SCM_RIGHTS sendmsg failed"); + close_socket(&task, payload_sender); + + let mut received_data = [0u8; 1]; + let recv_iov = [litebox_common_linux::IoVec { + iov_base: UserPtrMut::from_usize(received_data.as_mut_ptr() as usize), + iov_len: received_data.len(), + }]; + let mut recv_control = [0u8; CMSG_SPACE_ONE_FD]; + let mut recv_header = litebox_common_linux::UserMsgHdr::new_zeroed(); + recv_header.msg_iov = UserPtr::from_usize(recv_iov.as_ptr() as usize); + recv_header.msg_iovlen = recv_iov.len(); + recv_header.msg_control = UserPtr::from_usize(recv_control.as_mut_ptr() as usize); + recv_header.msg_controllen = recv_control.len(); + let recv_flags = if receive_cloexec { + ReceiveFlags::CMSG_CLOEXEC + } else { + ReceiveFlags::empty() + }; + + let received = task + .sys_recvmsg( + i32::try_from(bus_receiver).unwrap(), + UserPtrMut::from_usize(&raw mut recv_header as usize), + recv_flags, + ) + .expect("SCM_RIGHTS recvmsg failed"); + assert_eq!(received, 1); + assert_eq!(received_data, data); + let returned_controllen = recv_header.msg_controllen; + let returned_flags = recv_header.msg_flags; + assert_eq!(returned_controllen, CMSG_SPACE_ONE_FD); + assert!(!returned_flags.contains(ReceiveFlags::CTRUNC)); + assert_eq!( + usize::from_ne_bytes( + recv_control[..core::mem::size_of::()] + .try_into() + .unwrap() + ), + CMSG_LEN_ONE_FD + ); + assert_eq!( + u32::from_ne_bytes(recv_control[level_offset..type_offset].try_into().unwrap()), + SOL_SOCKET + ); + assert_eq!( + u32::from_ne_bytes( + recv_control[type_offset..CMSG_HEADER_LEN] + .try_into() + .unwrap() + ), + SCM_RIGHTS + ); + let received_fd = i32::from_ne_bytes( + recv_control[CMSG_HEADER_LEN..CMSG_LEN_ONE_FD] + .try_into() + .unwrap(), + ); + let descriptor_flags = { + let files = task.files.borrow(); + crate::syscalls::file::get_file_descriptor_flags( + usize::try_from(received_fd).unwrap(), + &task.global, + &files, + ) + .expect("F_GETFD failed") + }; + assert_eq!( + descriptor_flags.contains(FileDescriptorFlags::FD_CLOEXEC), + receive_cloexec + ); + + task.do_sendto(payload_peer, b"alive", SendFlags::empty(), None) + .expect("send through payload peer failed"); + let mut payload = [0u8; 5]; + let payload_len = task + .do_recvfrom( + u32::try_from(received_fd).unwrap(), + &mut payload, + ReceiveFlags::empty(), + None, + ) + .expect("received descriptor is unusable"); + assert_eq!(&payload[..payload_len], b"alive"); + + close_socket(&task, u32::try_from(received_fd).unwrap()); + close_socket(&task, payload_peer); + close_socket(&task, bus_sender); + close_socket(&task, bus_receiver); + } + } + fn ppoll(task: &TestTask, fd: u32, events: Events) { let fd = i32::try_from(fd).unwrap(); let mut pollfd = [litebox_common_linux::Pollfd { @@ -3272,6 +4364,240 @@ mod unix_tests { } } + // -- Regression coverage for the previously-missed unix-addr-table gap -- + // + // Before this fix, `task.global.unix_addr_table` was only ever populated + // by `listen()` (stream sockets) and `UnixDatagramInner::bind()` + // (datagram sockets). A plain stream socket that called `bind()` but had + // not (yet) called `listen()` -- the ordinary client-role + // autobind-then-connect pattern -- was never inserted into the table at + // all, so a second, colliding bind was invisible to collision detection + // and could succeed anyway. Separately, `UnixSocketAddr::Abstract`'s + // explicit-bind arm had *zero* collision detection (a bare `TODO`), and + // the read-then-separate-write table access elsewhere was a + // check-then-act race. The tests below exercise exactly those three + // gaps against the real syscall surface (`do_bind`/`do_connect`), not + // against any internal helper. + + #[test] + fn test_unix_stream_abstract_bind_without_listen_blocks_collision() { + let task = init_platform(None); + let name = b"gm_third_attempt_abstract_addr".to_vec(); + + // Socket A explicitly binds an abstract address but never calls + // listen(). This is exactly the gap the prior attempt's own fix + // missed: only listen()/datagram-bind() ever touched the shared + // address table, so a bound-but-not-listening socket was invisible + // to collision detection. + let a_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_bind( + a_fd, + SocketAddress::Unix(UnixSocketAddr::Abstract(name.clone())), + ) + .expect("first abstract bind should succeed"); + + // A second socket colliding on the same abstract address must be + // rejected while A is alive, even though A never listened. The + // pre-existing `Abstract` bind arm had zero collision detection at + // all (a bare `TODO`), so this is also the abstract-collision case. + let b_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let err = task + .do_bind( + b_fd, + SocketAddress::Unix(UnixSocketAddr::Abstract(name.clone())), + ) + .unwrap_err(); + assert_eq!(err, Errno::EADDRINUSE); + + // Once A closes (still never having listened), the address must + // become free again for a fresh bind. + close_socket(&task, a_fd); + task.do_bind(b_fd, SocketAddress::Unix(UnixSocketAddr::Abstract(name))) + .expect("bind should succeed once A releases the address"); + + close_socket(&task, b_fd); + } + + #[test] + fn test_unix_stream_path_bind_without_listen_blocks_collision() { + let task = init_platform(None); + let addr = "/unix_bind_no_listen_path.sock"; + + let a_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_bind( + a_fd, + SocketAddress::Unix(UnixSocketAddr::Path(addr.to_string())), + ) + .expect("first bind should succeed"); + + let b_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let err = task + .do_bind( + b_fd, + SocketAddress::Unix(UnixSocketAddr::Path(addr.to_string())), + ) + .unwrap_err(); + assert_eq!(err, Errno::EADDRINUSE); + + close_socket(&task, a_fd); + task.sys_unlinkat(-1, addr, AtFlags::empty()).unwrap(); + task.do_bind( + b_fd, + SocketAddress::Unix(UnixSocketAddr::Path(addr.to_string())), + ) + .expect("bind should succeed once A is gone and the path is unlinked"); + + close_socket(&task, b_fd); + task.sys_unlinkat(-1, addr, AtFlags::empty()).unwrap(); + } + + #[test] + fn test_unix_bound_client_reservation_survives_into_connected_state() { + let task = init_platform(None); + let server_path = "/unix_client_reservation_server.sock"; + let server_fd = create_unix_server_socket(&task, server_path, SockFlags::empty()).unwrap(); + + let client_name = b"gm_third_attempt_client_addr".to_vec(); + let client_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_bind( + client_fd, + SocketAddress::Unix(UnixSocketAddr::Abstract(client_name.clone())), + ) + .expect("client bind should succeed"); + + // Client connects without ever calling listen() -- the "ordinary + // client-role autobind-then-connect pattern" the bug report names + // directly. + task.do_connect( + client_fd, + SocketAddress::Unix(UnixSocketAddr::Path(server_path.to_string())), + ) + .unwrap(); + + // The client's own bound address must still be reserved while the + // connection is alive -- a second socket must not be able to steal + // it just because the client stopped being in "Init" state. + let other_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let err = task + .do_bind( + other_fd, + SocketAddress::Unix(UnixSocketAddr::Abstract(client_name.clone())), + ) + .unwrap_err(); + assert_eq!(err, Errno::EADDRINUSE); + + // Closing the connected client releases its reservation. + close_socket(&task, client_fd); + task.do_bind( + other_fd, + SocketAddress::Unix(UnixSocketAddr::Abstract(client_name)), + ) + .expect("bind should succeed once the connected client closes"); + + close_socket(&task, other_fd); + let server_conn = task.do_accept(server_fd, None, SockFlags::empty()).unwrap(); + close_socket(&task, server_conn); + close_socket(&task, server_fd); + task.sys_unlinkat(-1, server_path, AtFlags::empty()) + .unwrap(); + } + + #[test] + fn test_unix_concurrent_bind_same_abstract_address_exactly_one_wins() { + let task = init_platform(None); + + for i in 0..20 { + let name = alloc::format!("gm_race_addr_{i}").into_bytes(); + let a_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let b_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let barrier = std::sync::Arc::new(std::sync::Barrier::new(2)); + let barrier_a = barrier.clone(); + let name_a = name.clone(); + + // Both threads race to bind the *same* address, synchronized to + // maximize the chance of hitting the window a check-then-act + // race (read the table, then write, as two separate critical + // sections) would allow -- both observing the address as free + // and both succeeding. + let handle = task.spawn_clone_for_test(move |task| { + barrier_a.wait(); + task.do_bind(a_fd, SocketAddress::Unix(UnixSocketAddr::Abstract(name_a))) + }); + + barrier.wait(); + let result_b = task.do_bind(b_fd, SocketAddress::Unix(UnixSocketAddr::Abstract(name))); + let result_a = handle.join().expect("thread A panicked"); + + let wins = usize::from(result_a.is_ok()) + usize::from(result_b.is_ok()); + assert_eq!( + wins, 1, + "exactly one concurrent bind to the same address must succeed, got a={result_a:?} b={result_b:?}" + ); + + close_socket(&task, a_fd); + close_socket(&task, b_fd); + } + } + + #[test] + fn test_unix_concurrent_bind_different_addresses_both_succeed() { + let task = init_platform(None); + let barrier = std::sync::Arc::new(std::sync::Barrier::new(2)); + let barrier_a = barrier.clone(); + + let a_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let b_fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + let name_a = b"gm_concurrent_addr_a".to_vec(); + let name_b = b"gm_concurrent_addr_b".to_vec(); + let name_a2 = name_a.clone(); + + // Two unrelated binds proceeding concurrently must not spuriously + // collide with each other -- the shared table's write-lock critical + // sections are per-attempt and brief (see `reserve_unix_addr`), not + // one coarse lock serializing every bind against every other. + let handle = task.spawn_clone_for_test(move |task| { + barrier_a.wait(); + task.do_bind(a_fd, SocketAddress::Unix(UnixSocketAddr::Abstract(name_a2))) + }); + + barrier.wait(); + let result_b = task.do_bind(b_fd, SocketAddress::Unix(UnixSocketAddr::Abstract(name_b))); + let result_a = handle.join().expect("thread A panicked"); + + assert!( + result_a.is_ok(), + "bind to address A should succeed: {result_a:?}" + ); + assert!( + result_b.is_ok(), + "bind to address B should succeed: {result_b:?}" + ); + + close_socket(&task, a_fd); + close_socket(&task, b_fd); + } + + #[test] + fn test_unix_stream_autobind_assigns_unique_addresses() { + let task = init_platform(None); + let mut seen = Vec::new(); + for _ in 0..64 { + let fd = create_unix_socket(&task, SockType::Stream, SockFlags::empty()); + task.do_bind(fd, SocketAddress::Unix(UnixSocketAddr::Unnamed)) + .expect("autobind should succeed"); + let addr = task.do_getsockname(fd).unwrap(); + let SocketAddress::Unix(UnixSocketAddr::Abstract(name)) = addr else { + panic!("autobind should assign an abstract address, got {addr:?}"); + }; + assert!( + !seen.contains(&name), + "two autobinds must not collide on the same abstract address" + ); + seen.push(name); + close_socket(&task, fd); + } + } + fn unix_socketpair_bidirectional(ty: SockType, is_nonblocking: bool) { let task = init_platform(None); let mut sv_ptr = alloc::vec![0u32; 2]; @@ -3333,9 +4659,75 @@ mod unix_tests { fn test_unix_socketpair_bidirectional() { unix_socketpair_bidirectional(SockType::Stream, false); unix_socketpair_bidirectional(SockType::Datagram, false); + unix_socketpair_bidirectional(SockType::SeqPacket, false); unix_socketpair_bidirectional(SockType::Stream, true); unix_socketpair_bidirectional(SockType::Datagram, true); + unix_socketpair_bidirectional(SockType::SeqPacket, true); + } + + #[test] + fn test_unix_seqpacket_socketpair_preserves_records_and_reports_type() { + let task = init_platform(None); + let (sock1, sock2) = task + .do_socketpair( + AddressFamily::UNIX, + SockType::SeqPacket, + SockFlags::CLOEXEC, + 0, + ) + .expect("SOCK_SEQPACKET socketpair failed"); + + for fd in [sock1, sock2] { + let mut socket_type = 0u32; + let len = task + .do_getsockopt( + fd, + SocketOptionName::Socket(SocketOption::TYPE), + UserPtrMut::from_usize((&raw mut socket_type).cast::() as usize), + core::mem::size_of::() as u32, + ) + .expect("SO_TYPE failed"); + assert_eq!(len, core::mem::size_of::()); + assert_eq!(socket_type, SockType::SeqPacket as u32); + } + + let first = b"first record"; + let second = b"second record"; + task.do_sendto(sock1, first, SendFlags::empty(), None) + .expect("first send failed"); + task.do_sendto(sock1, second, SendFlags::empty(), None) + .expect("second send failed"); + + let mut buf = [0u8; 64]; + let n = task + .do_recvfrom(sock2, &mut buf, ReceiveFlags::empty(), None) + .expect("first receive failed"); + assert_eq!(n, first.len()); + assert_eq!(&buf[..n], first); + + let n = task + .do_recvfrom(sock2, &mut buf, ReceiveFlags::empty(), None) + .expect("second receive failed"); + assert_eq!(n, second.len()); + assert_eq!(&buf[..n], second); + + let oversized = b"truncate this record"; + task.do_sendto(sock1, oversized, SendFlags::empty(), None) + .expect("oversized send failed"); + let mut short = [0u8; 5]; + let n = task + .do_recvfrom(sock2, &mut short, ReceiveFlags::empty(), None) + .expect("truncated receive failed"); + assert_eq!(n, oversized.len()); + assert_eq!(&short, &oversized[..short.len()]); + + close_socket(&task, sock1); + let n = task + .do_recvfrom(sock2, &mut buf, ReceiveFlags::empty(), None) + .expect("peer close must become EOF"); + assert_eq!(n, 0); + close_socket(&task, sock2); } fn unix_socket_recv_timeout(ty: SockType) { @@ -3372,6 +4764,7 @@ mod unix_tests { fn test_unix_socket_recv_timeout() { unix_socket_recv_timeout(SockType::Stream); unix_socket_recv_timeout(SockType::Datagram); + unix_socket_recv_timeout(SockType::SeqPacket); } #[test] diff --git a/litebox_shim_linux/src/syscalls/netlink.rs b/litebox_shim_linux/src/syscalls/netlink.rs new file mode 100644 index 0000000000..0d442b3305 --- /dev/null +++ b/litebox_shim_linux/src/syscalls/netlink.rs @@ -0,0 +1,325 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! A minimal `AF_NETLINK` / `NETLINK_ROUTE` socket. +//! +//! This exists for exactly one caller: libc's `getifaddrs(3)` (and thus +//! `os.networkInterfaces()` / libuv's `uv_interface_addresses`). musl's +//! `getifaddrs` speaks rtnetlink -- it opens a `NETLINK_ROUTE` socket, `send`s an +//! `RTM_GETLINK` dump request then an `RTM_GETADDR` dump request, and `recv`s the +//! `RTM_NEWLINK`/`RTM_NEWADDR` replies until an `NLMSG_DONE`. There is no +//! `SIOCGIFCONF` fallback, so without this the whole API fails at `socket()` with +//! `EAFNOSUPPORT`. +//! +//! We model a fixed interface table -- loopback (`lo`, 127.0.0.1/8) plus the one +//! synthetic interface LiteBox's `smoltcp` stack answers on (`eth0`, matching the +//! runner's `INTERFACE_IP_ADDR`) -- and synthesise the two dumps as canned +//! netlink messages, plus the `RTM_GETROUTE` dump `ip route` asks for (the +//! default route via the gateway and the connected `/24`). It is +//! request/response only: a `send` records the dump the matching `recv`s will +//! drain. No real link/addr/route state ever changes. + +use alloc::vec::Vec; + +use litebox::{ + event::{Events, IOPollable, observer::Observer, polling::Pollee}, + fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}, + sync::{Mutex, RawSyncPrimitivesProvider}, +}; + +use crate::ShimPlatform; + +pub(crate) struct NetlinkSubsystem(core::marker::PhantomData); +impl FdEnabledSubsystem for NetlinkSubsystem { + type Entry = NetlinkSocket; +} +impl FdEnabledSubsystemEntry for NetlinkSocket {} + +/// An open `NETLINK_ROUTE` socket. `pending` holds bytes produced by `send`s that +/// later `recv`s drain, in order. +pub(crate) struct NetlinkSocket { + pending: Mutex>, + pollee: Pollee, + interface_addr: [u8; 4], + gateway_addr: [u8; 4], +} + +// rtnetlink constants (see `linux/rtnetlink.h`, `linux/netlink.h`, `linux/if.h`). +const NLMSG_DONE: u16 = 3; +const RTM_NEWLINK: u16 = 16; +const RTM_GETLINK: u16 = 18; +const RTM_NEWADDR: u16 = 20; +const RTM_GETADDR: u16 = 22; +const RTM_NEWROUTE: u16 = 24; +const RTM_GETROUTE: u16 = 26; +const NLM_F_MULTI: u16 = 2; + +const AF_UNSPEC: u8 = 0; +const AF_INET: u8 = 2; +const AF_INET6: u8 = 10; + +// Routing-table constants (`linux/rtnetlink.h`). +const RT_TABLE_MAIN: u8 = 254; +const RTPROT_KERNEL: u8 = 2; +const RTPROT_BOOT: u8 = 3; +const RTN_UNICAST: u8 = 1; +const RT_SCOPE_LINK: u8 = 253; +const RTA_DST: u16 = 1; +const RTA_OIF: u16 = 4; +const RTA_GATEWAY: u16 = 5; +const RTA_PREFSRC: u16 = 7; +const RTA_TABLE: u16 = 15; +/// `eth0`'s interface index in the link dump below. +const ETH_IFINDEX: u32 = 2; + +const ARPHRD_ETHER: u16 = 1; +const ARPHRD_LOOPBACK: u16 = 772; + +const IFF_UP: u32 = 0x1; +const IFF_BROADCAST: u32 = 0x2; +const IFF_LOOPBACK: u32 = 0x8; +const IFF_RUNNING: u32 = 0x40; +const IFF_MULTICAST: u32 = 0x1000; + +const IFLA_ADDRESS: u16 = 1; +const IFLA_BROADCAST: u16 = 2; +const IFLA_IFNAME: u16 = 3; + +const IFA_ADDRESS: u16 = 1; +const IFA_LOCAL: u16 = 2; +const IFA_LABEL: u16 = 3; +const IFA_BROADCAST: u16 = 4; + +const IFA_F_PERMANENT: u8 = 0x80; +const RT_SCOPE_UNIVERSE: u8 = 0; +const RT_SCOPE_HOST: u8 = 254; + +// The synthetic loopback address; eth0's address comes from the Network that +// owns this netlink socket. +const LO_ADDR: [u8; 4] = [127, 0, 0, 1]; +const ETH_MAC: [u8; 6] = [0x02, 0x00, 0x00, 0x00, 0x00, 0x02]; + +#[expect( + clippy::cast_possible_truncation, + reason = "attribute payloads here are a handful of bytes; rta_len fits u16 with room to spare" +)] +fn push_attr(body: &mut Vec, atype: u16, payload: &[u8]) { + // `rta_len` counts the 4-byte header plus the (unpadded) payload; the next + // attribute begins at the next 4-byte boundary (`RTA_ALIGN`). + let rta_len = (4 + payload.len()) as u16; + body.extend_from_slice(&rta_len.to_ne_bytes()); + body.extend_from_slice(&atype.to_ne_bytes()); + body.extend_from_slice(payload); + while !body.len().is_multiple_of(4) { + body.push(0); + } +} + +#[expect( + clippy::cast_possible_truncation, + reason = "each synthesised message is well under 256 bytes; nlmsg_len fits u32" +)] +fn push_msg(out: &mut Vec, mtype: u16, seq: u32, body: &[u8]) { + // `nlmsg_len` is the header (16) plus the body; `body` is already 4-byte + // aligned by construction, so the whole message is `NLMSG_ALIGN`ed. + let total = (16 + body.len()) as u32; + out.extend_from_slice(&total.to_ne_bytes()); + out.extend_from_slice(&mtype.to_ne_bytes()); + out.extend_from_slice(&NLM_F_MULTI.to_ne_bytes()); + out.extend_from_slice(&seq.to_ne_bytes()); + out.extend_from_slice(&0u32.to_ne_bytes()); // nlmsg_pid: 0 == from the kernel + out.extend_from_slice(body); + while !out.len().is_multiple_of(4) { + out.push(0); + } +} + +fn ifinfomsg(ty: u16, index: i32, flags: u32) -> Vec { + let mut b = Vec::from([AF_UNSPEC, 0]); // ifi_family, padding + b.extend_from_slice(&ty.to_ne_bytes()); // ifi_type + b.extend_from_slice(&index.to_ne_bytes()); // ifi_index + b.extend_from_slice(&flags.to_ne_bytes()); // ifi_flags + b.extend_from_slice(&0u32.to_ne_bytes()); // ifi_change + b +} + +fn ifaddrmsg(prefixlen: u8, scope: u8, index: u32) -> Vec { + // ifa_family, ifa_prefixlen, ifa_flags, ifa_scope + let mut b = Vec::from([AF_INET, prefixlen, IFA_F_PERMANENT, scope]); + b.extend_from_slice(&index.to_ne_bytes()); // ifa_index + b +} + +/// Build the `RTM_GETLINK` reply: one `RTM_NEWLINK` per interface, then `NLMSG_DONE`. +fn build_link_dump(out: &mut Vec, seq: u32) { + // lo (index 1) + let mut body = ifinfomsg(ARPHRD_LOOPBACK, 1, IFF_UP | IFF_LOOPBACK | IFF_RUNNING); + push_attr(&mut body, IFLA_IFNAME, b"lo\0"); + push_attr(&mut body, IFLA_ADDRESS, &[0u8; 6]); + push_msg(out, RTM_NEWLINK, seq, &body); + + // eth0 (index 2) + let mut body = ifinfomsg( + ARPHRD_ETHER, + 2, + IFF_UP | IFF_RUNNING | IFF_BROADCAST | IFF_MULTICAST, + ); + push_attr(&mut body, IFLA_IFNAME, b"eth0\0"); + push_attr(&mut body, IFLA_ADDRESS, Ð_MAC); + push_attr(&mut body, IFLA_BROADCAST, &[0xffu8; 6]); + push_msg(out, RTM_NEWLINK, seq, &body); + + push_msg(out, NLMSG_DONE, seq, &0i32.to_ne_bytes()); +} + +/// Build the `RTM_GETADDR` reply: one `RTM_NEWADDR` per address, then `NLMSG_DONE`. +fn build_addr_dump(out: &mut Vec, seq: u32, eth_addr: [u8; 4]) { + // lo: 127.0.0.1/8, host scope + let mut body = ifaddrmsg(8, RT_SCOPE_HOST, 1); + push_attr(&mut body, IFA_ADDRESS, &LO_ADDR); + push_attr(&mut body, IFA_LOCAL, &LO_ADDR); + push_attr(&mut body, IFA_LABEL, b"lo\0"); + push_msg(out, RTM_NEWADDR, seq, &body); + + // eth0: configured address with the fixed /24 prefix, universe scope + let broadcast = [eth_addr[0], eth_addr[1], eth_addr[2], 255]; + let mut body = ifaddrmsg(24, RT_SCOPE_UNIVERSE, 2); + push_attr(&mut body, IFA_ADDRESS, ð_addr); + push_attr(&mut body, IFA_LOCAL, ð_addr); + push_attr(&mut body, IFA_BROADCAST, &broadcast); + push_attr(&mut body, IFA_LABEL, b"eth0\0"); + push_msg(out, RTM_NEWADDR, seq, &body); + + push_msg(out, NLMSG_DONE, seq, &0i32.to_ne_bytes()); +} + +fn rtmsg(dst_len: u8, protocol: u8, scope: u8) -> Vec { + // rtm_family, rtm_dst_len, rtm_src_len, rtm_tos, rtm_table, rtm_protocol, rtm_scope, + // rtm_type, rtm_flags + let mut b = Vec::from([ + AF_INET, + dst_len, + 0, + 0, + RT_TABLE_MAIN, + protocol, + scope, + RTN_UNICAST, + ]); + b.extend_from_slice(&0u32.to_ne_bytes()); + b +} + +/// Build the `RTM_GETROUTE` reply: the two IPv4 routes smoltcp actually uses -- the default +/// route via the gateway, and the connected `/24` the interface address lives in -- then +/// `NLMSG_DONE`. Only the main table exists, and only IPv4 (a dump asking for `AF_INET6` gets +/// an empty answer). +fn build_route_dump(out: &mut Vec, seq: u32, eth_addr: [u8; 4], gateway: [u8; 4]) { + // default via dev eth0 + let mut body = rtmsg(0, RTPROT_BOOT, RT_SCOPE_UNIVERSE); + push_attr( + &mut body, + RTA_TABLE, + &u32::from(RT_TABLE_MAIN).to_ne_bytes(), + ); + push_attr(&mut body, RTA_GATEWAY, &gateway); + push_attr(&mut body, RTA_OIF, Ð_IFINDEX.to_ne_bytes()); + push_msg(out, RTM_NEWROUTE, seq, &body); + + // /24 dev eth0 proto kernel scope link src + let subnet = [eth_addr[0], eth_addr[1], eth_addr[2], 0]; + let mut body = rtmsg(24, RTPROT_KERNEL, RT_SCOPE_LINK); + push_attr( + &mut body, + RTA_TABLE, + &u32::from(RT_TABLE_MAIN).to_ne_bytes(), + ); + push_attr(&mut body, RTA_DST, &subnet); + push_attr(&mut body, RTA_OIF, Ð_IFINDEX.to_ne_bytes()); + push_attr(&mut body, RTA_PREFSRC, ð_addr); + push_msg(out, RTM_NEWROUTE, seq, &body); + + push_msg(out, NLMSG_DONE, seq, &0i32.to_ne_bytes()); +} + +impl NetlinkSocket { + pub(crate) fn new(interface_ip: core::net::Ipv4Addr, gateway_ip: core::net::Ipv4Addr) -> Self { + Self { + pending: Mutex::new(Vec::new()), + pollee: Pollee::new(), + interface_addr: interface_ip.octets(), + gateway_addr: gateway_ip.octets(), + } + } + + /// Handle a `send`: parse each request header, enqueue the matching dump. + /// Returns the number of request bytes "sent" (always the whole buffer). + pub(crate) fn handle_send(&self, req: &[u8]) -> usize { + let mut out = self.pending.lock(); + let was_empty = out.is_empty(); + let mut off = 0usize; + while off + 16 <= req.len() { + let nlmsg_len = + u32::from_ne_bytes([req[off], req[off + 1], req[off + 2], req[off + 3]]) as usize; + let nlmsg_type = u16::from_ne_bytes([req[off + 4], req[off + 5]]); + let seq = + u32::from_ne_bytes([req[off + 8], req[off + 9], req[off + 10], req[off + 11]]); + // The request family (`rtgenmsg`/`rtmsg` both start with it) follows the header. + let family = req.get(off + 16).copied().unwrap_or(AF_UNSPEC); + match nlmsg_type { + RTM_GETLINK => build_link_dump(&mut out, seq), + RTM_GETADDR => build_addr_dump(&mut out, seq, self.interface_addr), + RTM_GETROUTE if family != AF_INET6 => { + build_route_dump(&mut out, seq, self.interface_addr, self.gateway_addr); + } + // Any other request type: reply with a bare DONE so the caller's + // dump loop terminates instead of hanging. + _ => push_msg(&mut out, NLMSG_DONE, seq, &0i32.to_ne_bytes()), + } + // Advance by the aligned message length; a malformed/zero length would + // otherwise loop forever. + let step = (nlmsg_len.max(16) + 3) & !3; + off += step; + } + let became_readable = was_empty && !out.is_empty(); + drop(out); + if became_readable { + self.pollee.notify_observers(Events::IN); + } + req.len() + } + + /// Handle a `recv`: copy out (and consume) up to `buf.len()` pending bytes. + /// Empty pending buffer reports `EAGAIN` (getifaddrs uses `MSG_DONTWAIT`). + pub(crate) fn handle_recv( + &self, + buf: &mut [u8], + ) -> Result { + let mut pending = self.pending.lock(); + if pending.is_empty() { + return Err(litebox_common_linux::errno::Errno::EAGAIN); + } + let n = buf.len().min(pending.len()); + buf[..n].copy_from_slice(&pending[..n]); + pending.drain(..n); + Ok(n) + } +} + +impl IOPollable for NetlinkSocket { + fn check_io_events(&self) -> Events { + let mut events = Events::OUT; + if !self.pending.lock().is_empty() { + events |= Events::IN; + } + events + } + + fn register_observer(&self, observer: alloc::sync::Weak>, mask: Events) { + self.pollee.register_observer(observer, mask); + } + + fn unregister_observer(&self, observer: alloc::sync::Weak>) { + self.pollee.unregister_observer(observer); + } +} diff --git a/litebox_shim_linux/src/syscalls/pipe.rs b/litebox_shim_linux/src/syscalls/pipe.rs index 938f3ae9c8..aa1e4981cf 100644 --- a/litebox_shim_linux/src/syscalls/pipe.rs +++ b/litebox_shim_linux/src/syscalls/pipe.rs @@ -10,7 +10,7 @@ use core::num::NonZero; use litebox::{ - event::{IOPollable, wait::WaitContext}, + event::wait::WaitContext, fd::MetadataError, fs::{Mode, OFlags}, pipes::{Flags, HalfPipeType, PipeFd}, @@ -19,7 +19,7 @@ use litebox_common_linux::{FileDescriptorFlags, InodeType, errno::Errno}; use crate::{GlobalState, ShimFS, ShimPlatform}; -const DEFAULT_PIPE_BUF_SIZE: usize = 1024 * 1024; +const DEFAULT_PIPE_BUF_SIZE: usize = 64 * 1024; /// Status flags for Linux pipe file descriptions. /// @@ -51,7 +51,12 @@ impl GlobalState { } pipe_flags.set(Flags::NON_BLOCKING, flags.contains(OFlags::NONBLOCK)); if flags.contains(OFlags::DIRECT) { - todo!("O_DIRECT not supported"); + // Real O_DIRECT pipes are packet-mode: each write is a discrete message and + // reads never span one. The backing ring buffer here is a flat byte stream + // with no message-boundary tracking, so, like `set_linux_pipe_status_flags` + // below does for the same flag via fcntl, we accept and record the bit + // (readable back through fcntl(F_GETFL)) without enforcing packet framing. + log_unsupported!("O_DIRECT (packet-mode) pipe"); } (pipe_flags, flags.contains(OFlags::CLOEXEC)) }; @@ -61,9 +66,11 @@ impl GlobalState { pipe_flags, // See `man 7 pipe` for `PIPE_BUF`. On Linux, this is 4096. NonZero::new(4096), - ); + )?; - let initial_status = OFlags::from(pipe_flags); + // `Flags` (the internal pipe-backend type) only tracks NON_BLOCKING, so DIRECT + // has to be folded back in here to stay visible through fcntl(F_GETFL). + let initial_status = OFlags::from(pipe_flags) | (flags & OFlags::DIRECT); { let mut dt = self.litebox.descriptor_table_mut(); let old = @@ -148,13 +155,6 @@ impl GlobalState { Ok(read_write_mode.bits() | InodeType::NamedPipe as u32) } - pub(crate) fn with_linux_pipe_iopollable( - &self, - fd: &PipeFd, - f: impl FnOnce(&dyn IOPollable) -> R, - ) -> Result { - self.pipes.with_iopollable(fd, f).map_err(Errno::from) - } } fn metadata_to_errno(err: MetadataError) -> Errno { @@ -165,3 +165,39 @@ fn metadata_to_errno(err: MetadataError) -> Errno { } } } + +#[cfg(test)] +mod tests { + use litebox_common_linux::FcntlArg; + + use super::OFlags; + use crate::syscalls::tests::init_platform; + + #[test] + fn test_pipe2_direct_is_accepted_and_stays_usable() { + let task = init_platform(None); + let (read_fd, write_fd) = task + .sys_pipe2(OFlags::DIRECT) + .expect("pipe2(O_DIRECT) should not error"); + let read_fd = i32::try_from(read_fd).unwrap(); + let write_fd = i32::try_from(write_fd).unwrap(); + + for fd in [read_fd, write_fd] { + let flags = OFlags::from_bits_truncate(task.sys_fcntl(fd, FcntlArg::GETFL).unwrap()); + assert!( + flags.contains(OFlags::DIRECT), + "fcntl(F_GETFL) should still report O_DIRECT, as real Linux does" + ); + } + + let msg = b"hello via O_DIRECT pipe"; + let n = task.sys_write(write_fd, msg, None).expect("write failed"); + assert_eq!(n, msg.len()); + let mut buf = [0u8; 64]; + let n = task.sys_read(read_fd, &mut buf, None).expect("read failed"); + assert_eq!(&buf[..n], msg); + + task.sys_close(read_fd).unwrap(); + task.sys_close(write_fd).unwrap(); + } +} diff --git a/litebox_shim_linux/src/syscalls/process.rs b/litebox_shim_linux/src/syscalls/process.rs index 6024e091a4..e174f0ab84 100644 --- a/litebox_shim_linux/src/syscalls/process.rs +++ b/litebox_shim_linux/src/syscalls/process.rs @@ -6,23 +6,26 @@ use crate::{ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut}; use alloc::boxed::Box; use alloc::collections::btree_map::BTreeMap; -use alloc::sync::Arc; +use alloc::sync::{Arc, Weak}; use alloc::vec::Vec; -use core::cell::Cell; +use core::cell::{Cell, RefCell, UnsafeCell}; use core::mem::offset_of; -use core::ops::Range; -use core::sync::atomic::{AtomicBool, Ordering}; +use core::ops::{Deref, DerefMut, Range}; +use core::sync::atomic::{AtomicBool, AtomicI32, AtomicU32, AtomicUsize, Ordering}; use core::time::Duration; use litebox::event::wait::WaitError; -use litebox::mm::linux::VmFlags; +use litebox::mm::linux::{PAGE_SIZE, VmFlags}; use litebox::platform::TimerHandle; use litebox::platform::{ArchSpecificRegister, RawMutex as _}; use litebox::platform::{Instant as _, SystemTime as _, TimeProvider}; -use litebox::sync::Mutex; +use litebox::sync::{ + Mutex, + futex::{FutexKey, FutexManager}, +}; use litebox::utils::TruncateExt as _; use litebox_common_linux::{ ArchPrctlArg, CloneFlags, FutexArgs, IntervalTimer, ItimerVal, PrctlArg, TimeParam, - errno::Errno, + errno::Errno, signal::Signal, }; /// Process-management-related state on [`Task`]. @@ -45,21 +48,136 @@ pub(crate) struct ThreadState { /// of the futex has died. This notification consists of two pieces: the FUTEX_OWNER_DIED bit is set in the futex word, /// and the kernel performs a futex(2) FUTEX_WAKE operation on one of the threads waiting on the futex. robust_list: Cell>>, + /// Signal requested with `PR_SET_PDEATHSIG`, delivered when this process's parent exits. + /// Linux clears it in every freshly cloned task and preserves it across `execve`. + parent_death_signal: Cell>, + /// The program the thread is about to `exec`, staged by [`Task::resolve_shebang`] and + /// consumed by `Task::load_program` once the image is live: the absolute, symlink-resolved + /// path of the image (`/proc//exe`) and the command name (`comm`), which Linux takes + /// from the basename of the filename handed to `execve` -- `sh` for `/bin/sh`, the script's + /// own name for a `#!` script -- not from the image finally mapped. Staged rather than + /// threaded through the ELF loader because the loader keeps its path private and the + /// initial-program path is resolved by the shim's own `load_program` entry point, which + /// never sees a `Task` method. + staged_exec: RefCell>, +} + +/// See `ThreadState::staged_exec`. +struct StagedExec { + exe: alloc::string::String, + comm: Vec, } // TODO: remove once we figure out how to handle Send/Sync for raw pointers. unsafe impl Send for ThreadState {} impl ThreadState { - pub fn new_process(pid: i32) -> Self { + pub fn new_process(pid: i32, process_group_id: i32) -> Self { + Self::new_process_with_shared_futex_manager( + pid, + process_group_id, + Arc::new(FutexManager::new()), + None, + ) + } + + fn new_forked_process( + pid: i32, + process_group_id: i32, + shared_futex_manager: Arc>, + launch: Arc>, + ) -> Self { + Self::new_process_with_shared_futex_manager( + pid, + process_group_id, + shared_futex_manager, + Some(launch), + ) + } + + fn new_vfork_copy_process( + pid: i32, + process_group_id: i32, + shared_futex_manager: Arc>, + completion: Arc>, + launch: Arc>, + ) -> Self { + let futex_namespace = shared_futex_manager.new_private_namespace(); + Self::new_process_with_futex_namespace( + pid, + process_group_id, + shared_futex_manager, + futex_namespace, + Some(completion), + None, + Some(launch), + ) + } + + fn new_vforked_process( + pid: i32, + process_group_id: i32, + parent: &Process, + completion: Arc>, + launch: Arc>, + ) -> Self { + Self::new_process_with_futex_namespace( + pid, + process_group_id, + parent.futex_manager.clone(), + parent.futex_namespace(), + Some(completion), + Some(parent), + Some(launch), + ) + } + + fn new_process_with_shared_futex_manager( + pid: i32, + process_group_id: i32, + shared_futex_manager: Arc>, + launch: Option>>, + ) -> Self { + let futex_namespace = shared_futex_manager.new_private_namespace(); + Self::new_process_with_futex_namespace( + pid, + process_group_id, + shared_futex_manager, + futex_namespace, + None, + None, + launch, + ) + } + + fn new_process_with_futex_namespace( + pid: i32, + process_group_id: i32, + futex_manager: Arc>, + futex_namespace: usize, + vfork_completion: Option>>, + shared_vm_parent: Option<&Process>, + launch: Option>>, + ) -> Self { let remote = Arc::new(ThreadRemote::new()); Self { init_state: Cell::new(ThreadInitState::None), - process: Arc::new(Process::new(pid, remote.clone())), + process: Arc::new(Process::new( + pid, + process_group_id, + remote.clone(), + futex_manager, + futex_namespace, + vfork_completion, + shared_vm_parent, + launch, + )), remote, attached_tid: Cell::new(Some(pid)), clear_child_tid: Cell::new(None), robust_list: Cell::new(None), + parent_death_signal: Cell::new(None), + staged_exec: RefCell::new(None), } } @@ -72,12 +190,22 @@ impl ThreadState { attached_tid: Cell::new(Some(tid)), clear_child_tid: Cell::new(None), robust_list: Cell::new(None), + parent_death_signal: Cell::new(None), + staged_exec: RefCell::new(None), }) } - fn detach_from_process(&self) { + /// Detaches this thread from its process. + /// + /// Returns `true` if this was the last thread of the process to detach (i.e., the whole + /// process is now gone), `false` otherwise -- including when this thread was already + /// detached (so callers relying on this to run exactly-once cleanup, like closing every fd + /// on process exit, don't double-run it if `Drop` invokes this a second time). + fn detach_from_process(&self) -> bool { if let Some(tid) = self.attached_tid.take() { - self.process.detach_thread(tid); + self.process.detach_thread(tid) + } else { + false } } } @@ -89,12 +217,58 @@ impl Drop for ThreadState { } /// Thread state that can be accessed from a remote thread. -struct ThreadRemote { +/// Closed bit of [`Process::fork_gate`]'s word; the low 31 bits count parked +/// threads. +const FORK_GATE_CLOSED: u32 = 1 << 31; + +/// Reopens a [`Process::fork_gate`] closed by +/// [`Task::park_sibling_threads_for_fork`] when dropped, releasing every +/// parked sibling. Held across the whole of `do_fork`'s remaining body -- +/// including the parent's suspension for the child's address-space turn -- so +/// the gate reopens on success, on any error return, and on panic alike. +struct ForkGateGuard<'a, Platform: ShimPlatform> { + process: &'a Process, +} + +impl Drop for ForkGateGuard<'_, Platform> { + fn drop(&mut self) { + self.process + .fork_gate + .underlying_atomic() + .fetch_and(!FORK_GATE_CLOSED, Ordering::AcqRel); + self.process.fork_gate.wake_all(); + } +} + +pub(crate) struct ThreadRemote { /// Always set under the process `inner` lock, but can be read without /// locking. is_exiting: AtomicBool, /// Handle to interrupt waits on this thread. handle: once_cell::race::OnceBox>, + /// Signals directed at this specific thread by a remote `tkill`/`tgkill` (as opposed to a + /// process-directed `kill`, which uses [`Process`]-wide `shared_pending` instead). The + /// owning task's own `signals.pending` is a bare `RefCell` and therefore neither `Send` nor + /// `Sync` -- it can only ever be touched by the thread it belongs to -- so a sender on a + /// different thread has nowhere else to hand off a specifically-targeted signal. Drained + /// into that `RefCell` by the owning thread itself in `Task::process_signals`/ + /// `Task::has_pending_signals`, the same way `Process::shared_pending` already is. + remote_pending: Mutex, + /// `ptrace` attach/stop state for this thread. See [`super::ptrace::PtraceState`]. + /// + /// AArch64-only: `NT_PRSTATUS`/`NT_ARM_TLS` wire layouts and the `PTRACE_*` request numbers + /// this builds on live in `litebox_common_linux::ptrace`, gated the same way. + #[cfg(target_arch = "aarch64")] + pub(crate) ptrace: super::ptrace::PtraceState, + /// This thread's command name, as `/proc//task//comm` reports it. The owning + /// task's `comm` is a `Cell` only its own thread may read, so the value is mirrored here for + /// `/proc` readers on other threads (see `Task::set_task_comm`). + comm: Mutex, + /// This thread's nice value (`-20..=19`), the `setpriority(PRIO_PROCESS, tid)` / + /// `getpriority` state. Per thread, as on Linux, where every thread is its own scheduling + /// entity; lives here so a sibling thread's `getpriority(tid)` can read it. Purely + /// bookkeeping -- the host scheduler is never told. + nice: core::sync::atomic::AtomicI32, } impl ThreadRemote { @@ -102,556 +276,2693 @@ impl ThreadRemote { Self { is_exiting: AtomicBool::new(false), handle: once_cell::race::OnceBox::new(), + remote_pending: Mutex::new(super::signal::PendingSignals::new()), + #[cfg(target_arch = "aarch64")] + ptrace: super::ptrace::PtraceState::new(), + comm: Mutex::new([0; litebox_common_linux::TASK_COMM_LEN]), + nice: core::sync::atomic::AtomicI32::new(0), } } - fn interrupt(&self) { + /// The thread's nice value; see [`Self::nice`]. + pub(crate) fn nice(&self) -> i32 { + self.nice.load(Ordering::Relaxed) + } + + pub(crate) fn set_nice(&self, nice: i32) { + self.nice.store(nice, Ordering::Relaxed); + } + + /// Mirror the owning task's command name for `/proc` readers. + pub(crate) fn set_comm(&self, comm: &[u8; litebox_common_linux::TASK_COMM_LEN]) { + *self.comm.lock() = *comm; + } + + /// The command name, trimmed of trailing NULs. + fn comm(&self) -> Vec { + let comm = *self.comm.lock(); + let end = comm.iter().position(|&b| b == 0).unwrap_or(comm.len()); + comm[..end].to_vec() + } + + /// Interrupts a wait or, under HVF, kicks the vCPU lane this thread may currently be running + /// on -- see [`litebox::event::wait::ThreadHandle::interrupt`]. `pub(crate)` (rather than + /// only `super`-visible) so `syscalls::ptrace`'s `PTRACE_ATTACH` can reach a tracee that may + /// be deep inside a blocking syscall or `hv_vcpu_run`, the same way `tkill`/process-directed + /// signals already do. + pub(crate) fn interrupt(&self) { if let Some(handle) = self.handle.get() { handle.interrupt(); } } + + /// Queues `signal` for specifically this thread (a `tkill`/`tgkill` target) and wakes it out + /// of any interruptible wait so it notices next time it checks for pending signals. Safe to + /// call from any thread: `remote_pending` is the one piece of this thread's signal state + /// that is `Send`/`Sync`, precisely so a sender elsewhere never has to touch the owning + /// thread's local, non-`Send` `SignalState`. + pub(crate) fn deliver_remote_signal( + &self, + rlimits: &ResourceLimits, + signal: litebox_common_linux::signal::Signal, + siginfo: litebox_common_linux::signal::Siginfo, + ) { + self.remote_pending.lock().push(rlimits, signal, siginfo); + self.interrupt(); + } + + /// Drains any signals queued for specifically this thread (see `remote_pending`) into + /// `local`, the owning task's own thread-local pending set. Called only by the thread this + /// `ThreadRemote` belongs to. + pub(crate) fn drain_remote_signals_into(&self, local: &mut super::signal::PendingSignals) { + let mut remote = self.remote_pending.lock(); + remote.drain_into(local); + } } +/// Sentinel used by [`Process::controlling_pty`]. PTY numbers are allocated upward from zero and +/// never use this value. +const NO_CONTROLLING_PTY: u32 = u32::MAX; + /// A Linux process, which may have multiple threads. pub(crate) struct Process { /// Number of threads in this process. Always updated under the `inner` /// mutex lock. nr_threads: ::RawMutex, - inner: Mutex>, + /// Stop-the-world gate for `fork` from a multithreaded process. + /// + /// The delayed-address-space-handoff fork model (see [`SharedAddressSpace`]) + /// requires that no sibling thread touches guest memory during the child's + /// turn: the parent's private memory is snapshotted at `fork` and restored + /// when the turn comes back, so a sibling that kept running would have its + /// writes silently rolled back. Rather than refusing `fork` outright for + /// multithreaded guests (which breaks every libuv/Node `spawn`, whose + /// child does nothing but the classic dup2/close/execve dance), the + /// forking thread closes this gate: every sibling parks here -- woken out + /// of any interruptible wait by [`ThreadRemote::interrupt`] and caught at + /// the `CheckForInterrupt::check_for_interrupt`/ + /// [`Task::prepare_to_run_guest`] choke points before it can touch guest + /// memory again -- until the parent's turn resumes and the gate reopens. + /// + /// Word layout: bit 31 = closed; low 31 bits = number of currently-parked + /// threads. Mirrors how `nr_threads` uses its `RawMutex` word purely as a + /// blockable atomic. + fork_gate: ::RawMutex, + inner: Arc>>, + /// The swappable identity of the Linux `mm_struct`-like bookkeeping this process uses. + /// A vfork child gets its own slot pointing at the parent's identity, then swaps only its slot + /// to a fresh identity on successful exec. + vm: Arc>, + /// Futex wait queues inherited by every process in one fork family. + /// + /// Each independent VM identity has a distinct private futex namespace; MAP_SHARED + /// non-private keys instead use the manager's reserved shared namespace zero. + futex_manager: Arc>, + vfork_completion: Mutex>>>, + /// Whether this vfork child points at its parent's live VM identity, independently of whether + /// its parent waits for exec/exit. Lone `CLONE_VFORK` waits but owns a copied VM. + shares_parent_vm: AtomicBool, + launch: Option>>, /// Resource limits for this process. - pub(crate) limits: ResourceLimits, + pub(crate) limits: Arc, /// Process-wide alarm timer. pub(crate) alarm_timer: Mutex>, + /// The address ranges this process (as opposed to some other guest process sharing the same + /// host address space) had mapped. + /// + /// Needed because `fork` has to be able to save and restore *this* process's memory without + /// touching a sibling's -- see [`Task::save_address_space`]. The page manager's own view is + /// process-blind: it is one flat map of every guest mapping in the shim. + pub(crate) owned_ranges: SharedVmLockedField, + /// Runtime ELF rewriting state for this process's VM identity. + pub(crate) elf_patch_cache: SharedVmLockedField, + /// This process's program break. + /// + /// Every guest process shares one [`litebox::mm::PageManager`] (they live at disjoint + /// addresses in the one host address space), and that manager tracks a single break, so the + /// authoritative per-process value has to live here and be swapped into the manager around + /// each break operation. See `Task::sys_brk`. + pub(crate) brk: SharedVmAtomicUsize, + /// Total host CPU time (nanoseconds) consumed by every thread of this process so far. + /// + /// Each thread adds its own [`litebox::platform::TimeProvider::thread_cpu_time`] reading + /// here as it exits (see + /// `Task::prepare_for_exit`), since that clock is only readable by the thread it measures. + /// Reported to a `wait4(..., &rusage)` caller as `ru_utime` once the whole process is a + /// zombie -- see `Task::sys_wait4`. + pub(crate) cpu_time_nanos: core::sync::atomic::AtomicU64, + /// Session inherited across `fork` and replaced by `setsid`. + session_id: AtomicI32, + /// Process-group identity inherited across `fork` and shared by every thread in this process. + /// The process table keeps only a weak reference to this atomic, so remote parent operations do + /// not make the whole (platform-specific and potentially non-`Send`) process object global. + #[expect( + clippy::struct_field_names, + reason = "the full POSIX term distinguishes it from session identity" + )] + process_group_id: Arc, + /// Unix98 PTY number serving as this process's controlling terminal, or + /// [`NO_CONTROLLING_PTY`] when it has none. + controlling_pty: AtomicU32, + /// This process's place in a [`SharedAddressSpace`], `None` when its memory is its own. + /// Process-wide (every thread runs on the same memory and takes turns as one member), see + /// [`AddressSpaceMembership`]. + pub(crate) address_space: Mutex>>>, + /// `prctl(PR_SET_DUMPABLE)` state. Linux keeps this on the `mm` (so it is process-wide, + /// inherited by `fork` and reset by `execve`: to 1 for an ordinary exec, to the + /// `suid_dumpable` sysctl's default 0 for a set-uid/set-gid one). Only the flag itself is + /// modelled -- LiteBox writes no core dumps and has no `ptrace` access check that consults + /// it -- so that a launcher like Chromium's `chrome-sandbox`, which clears it before + /// dropping root and `CHECK`s that it read back 0, sees Linux's answers. + dumpable: AtomicBool, } -pub(crate) struct Alarm { - /// Handle for the alarm timer. - pub(crate) handle: Option<::TimerHandle>, - /// The deadline for the alarm. - pub(crate) deadline: Option<::Instant>, -} - -impl Alarm { - /// Returns the time remaining until [`Self::deadline`], or zero if the - /// alarm is not armed or its deadline has already passed. - pub(crate) fn remaining( - &self, - now: ::Instant, - ) -> Duration { - self.deadline - .as_ref() - .and_then(|d| d.checked_duration_since(&now)) - .unwrap_or(Duration::ZERO) - } -} - -/// The locked portion of the process state. -struct ProcessInner { - /// If true, the whole process is exiting. - group_exit: bool, - /// If true, one thread is waiting for other threads to exit. - is_killing_other_threads: bool, - /// The exit code of the last exited thread in the process. Not updated once - /// `group_exit` is set. - exit_status: ExitStatus, - /// The thread list for the process, mapped by thread ID. - threads: BTreeMap>>, +/// What `/proc//{status,stat,cmdline,exe}` describe about a process, kept where a reader +/// on another thread can see it (the owning task's credentials and `comm` are thread-local +/// `Cell`s/`RefCell`s). Refreshed by the owning task on every change it makes (`execve`, +/// `prctl(PR_SET_NAME)`) and ahead of each of its own `/proc` lookups, so a reader sees at worst +/// the state as of the target's last publish -- a live `setuid` by a process that never looks at +/// `/proc` afterwards is the one thing that can lag. +#[derive(Clone, Default)] +pub(crate) struct ProcIdentity { + ppid: i32, + uid: u32, + gid: u32, + /// The thread-group leader's command name, trimmed of trailing NULs. + comm: Vec, + /// NUL-separated, NUL-terminated `argv` of the current image. + cmdline: Vec, + /// Absolute, symlink-resolved path of the current image (`/proc//exe`). + exe: Option, } -#[derive(Clone, Copy, Debug)] -pub(crate) enum ExitStatus { - Exit(i8), - Signal(litebox_common_linux::signal::Signal), +/// A set of address ranges, kept sorted and non-overlapping. +/// +/// Small and linear on purpose: it holds one entry per live mapping of a single guest process, +/// which is a handful for the programs this shim runs, and it is only walked when that process +/// `fork`s. +#[derive(Clone, Default)] +pub(crate) struct OwnedRanges { + ranges: Vec>, } -impl Process { - /// Creates a new process with the given initial thread. - fn new(pid: i32, remote: Arc>) -> Self { - let nr_threads = ::RawMutex::INIT; - nr_threads.underlying_atomic().store(1, Ordering::Relaxed); - Self { - nr_threads, - inner: Mutex::new(ProcessInner { - exit_status: ExitStatus::Exit(0), - group_exit: false, - is_killing_other_threads: false, - threads: BTreeMap::from_iter([(pid, remote)]), - }), - limits: ResourceLimits::default(), - alarm_timer: Mutex::new(Alarm { - handle: None, - deadline: None, - }), +impl OwnedRanges { + /// Adds `range`, replacing anything it overlaps. + pub(crate) fn insert(&mut self, range: Range) { + if range.is_empty() { + return; } + self.remove(range.clone()); + let at = self.ranges.partition_point(|r| r.start < range.start); + self.ranges.insert(at, range); } - /// Returns the current number of threads in this process. - pub fn nr_threads(&self) -> u32 { - self.nr_threads.underlying_atomic().load(Ordering::Relaxed) + /// The parts of this set that `other` does not cover. + fn difference(&self, other: &OwnedRanges) -> OwnedRanges { + let mut out = OwnedRanges::default(); + for range in &self.ranges { + let mut cursor = range.start; + for covered in other.intersect(range) { + if cursor < covered.start { + out.ranges.push(cursor..covered.start); + } + cursor = cursor.max(covered.end); + } + if cursor < range.end { + out.ranges.push(cursor..range.end); + } + } + out } - /// Waits for all threads in this process to exit, returning the exit code. - pub fn wait_for_exit(&self) -> ExitStatus { - loop { - let n = self.nr_threads.underlying_atomic().load(Ordering::Acquire); - if n == 0 { - break; + /// Adds every range of `other`, merging with whatever it overlaps (a true union, unlike + /// [`Self::insert`], which replaces). + fn union_with(&mut self, other: &OwnedRanges) { + for range in &other.ranges { + let mut lo = range.start; + let mut hi = range.end; + for existing in &self.ranges { + if existing.start < hi && lo < existing.end { + lo = lo.min(existing.start); + hi = hi.max(existing.end); + } } - let _ = self.nr_threads.block(n); + self.insert(lo..hi); } - self.inner.lock().exit_status } - /// Attaches a new thread to this process, returning a new remote state for - /// the thread. - fn attach_thread(&self, tid: i32) -> Option>> { - // Allocate outside the lock. - let remote = Arc::new(ThreadRemote::new()); - let mut inner = self.inner.lock(); - if inner.group_exit || inner.is_killing_other_threads { - return None; + fn insert_bounded(&mut self, range: Range, max_ranges: usize) { + self.insert(range); + if self.ranges.len() > max_ranges { + let start = self.ranges.first().unwrap().start; + let end = self.ranges.last().unwrap().end; + self.ranges.clear(); + self.ranges.push(start..end); } - let old_thread = inner.threads.insert(tid, remote.clone()); - assert!(old_thread.is_none(), "thread ID {tid} already exists"); - let nr_threads = self.nr_threads.underlying_atomic(); - nr_threads.store(nr_threads.load(Ordering::Relaxed) + 1, Ordering::Release); - Some(remote) } - /// Detaches a thread from this process. - /// - /// # Panics - /// Panics if the thread ID does not exist in this process. - fn detach_thread(&self, tid: i32) { - let data; - let notify = { - let mut inner = self.inner.lock(); - data = inner.threads.remove(&tid); - assert!(data.is_some()); - - let nr_threads = self.nr_threads.underlying_atomic(); - let n = nr_threads.load(Ordering::Relaxed); - let new_count = n.checked_sub(1).expect("decrementing from zero threads"); - nr_threads.store(new_count, Ordering::Release); - if new_count == 0 { - assert!(inner.threads.is_empty()); - // The last thread exited. Prevent new threads. - inner.group_exit = true; + /// Removes `range`, splitting any entry that only partially overlaps it. + pub(crate) fn remove(&mut self, range: Range) { + if range.is_empty() { + return; + } + let mut out = Vec::with_capacity(self.ranges.len() + 1); + for r in self.ranges.drain(..) { + if r.end <= range.start || r.start >= range.end { + out.push(r); + continue; + } + if r.start < range.start { + out.push(r.start..range.start); + } + if r.end > range.end { + out.push(range.end..r.end); } - - // Notify waiters if this is the last thread of the process - // (`wait_for_exit`) or if this is the last thread being killed - // during an exec (`kill_other_threads`). - new_count == 0 || (new_count == 1 && inner.is_killing_other_threads) - }; - if notify { - self.nr_threads.wake_all(); } + self.ranges = out; } -} -impl Task { - /// Updates the process exit status for a thread exit. - fn exit_thread(&self, code: i8) { - let mut inner = self.thread.process.inner.lock(); - if self.is_exiting() { - return; - } - inner.exit_status = ExitStatus::Exit(code); - self.thread.remote.is_exiting.store(true, Ordering::Relaxed); + pub(crate) fn clear(&mut self) { + self.ranges.clear(); } - /// Updates the process exit status for a group exit and signals all threads - /// to exit. - pub(crate) fn exit_group(&self, status: ExitStatus) { - let mut inner = self.thread.process.inner.lock(); - if self.is_exiting() { - return; - } - assert!(!inner.group_exit); - inner.exit_status = status; - inner.group_exit = true; - for thread in inner.threads.values() { - thread.is_exiting.store(true, Ordering::Relaxed); - thread.interrupt(); - } + /// The parts of `range` that this set covers. + fn intersect(&self, range: &Range) -> impl Iterator> + '_ { + let range = range.clone(); + self.ranges.iter().filter_map(move |r| { + let start = r.start.max(range.start); + let end = r.end.min(range.end); + (start < end).then_some(start..end) + }) } +} - /// Kills all other threads in the process, waiting for them to exit. - /// - /// Returns false if this thread is already exiting. - #[must_use] - fn kill_other_threads(&self) -> bool { - { - let mut inner = self.thread.process.inner.lock(); - if self.is_exiting() { - return false; - } - for (&tid, thread) in &inner.threads { - if tid == self.tid { - continue; - } - thread.is_exiting.store(true, Ordering::Relaxed); - thread.interrupt(); - } - assert!(!inner.is_killing_other_threads); - inner.is_killing_other_threads = true; +struct VmLockedValue { + state: ::RawMutex, + value: UnsafeCell, +} + +impl VmLockedValue { + fn new(value: T) -> Self { + Self { + state: ::RawMutex::INIT, + value: UnsafeCell::new(value), } - // Wait for other threads to exit. + } + + fn lock(&self) { loop { - let n = self - .thread - .process - .nr_threads + if self + .state .underlying_atomic() - .load(Ordering::Acquire); - if n == 1 { - break; + .compare_exchange(0, 1, Ordering::Acquire, Ordering::Relaxed) + .is_ok() + { + return; } - let _ = self.thread.process.nr_threads.block(n); + let _ = self.state.block(1); } - self.thread.process.inner.lock().is_killing_other_threads = false; - true } - /// Returns true if the task is exiting and should not continue running - /// guest code. - pub fn is_exiting(&self) -> bool { - self.thread.remote.is_exiting.load(Ordering::Relaxed) + unsafe fn unlock(&self) { + self.state + .underlying_atomic() + .store(0, Ordering::Release); + self.state.wake_all(); } } -#[derive(Default)] -enum ThreadInitState { - #[default] - None, - NewProcess(crate::loader::elf::ElfLoadInfo), - NewThread { - stack: Option, - tls: Option, - set_child_tid: Option>, - }, +unsafe impl Send for VmLockedValue {} +unsafe impl Sync for VmLockedValue {} + +struct VmBookkeeping { + owned_ranges: VmLockedValue, + elf_patch_cache: VmLockedValue, + brk: AtomicUsize, + futex_namespace: usize, + /// DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): identity of + /// this process's current `SharedAddressSpace` family, as `Arc::as_ptr(&membership.shared) + /// as usize` (0 = not currently a family member). Lives here, rather than on `Process` + /// directly, so [`ProcessTable::overlaps_another_process`] can read it through the same + /// narrow `Weak>` already kept for `owned_ranges` -- see that field's + /// own doc comment for why a `Weak>` is not used. Set in `Task::join_address_space`, + /// cleared wherever a membership is dropped (`Task::release_address_space`'s single-threaded + /// branch, `Task::leave_address_space_if_alone`). Distinguishes an overlap that is *expected* + /// (two members of the SAME family, which by this platform's design believe they own the + /// same addresses -- see `SharedAddressSpace`'s own doc comment) from one between processes + /// that should have disjoint memory, which is the actual corruption candidate. + family_id: AtomicUsize, } -/// Credentials of a process -#[derive(Clone)] -pub(crate) struct Credentials { - pub uid: u32, - pub euid: u32, - pub gid: u32, - pub egid: u32, +impl VmBookkeeping { + fn new(futex_namespace: usize) -> Self { + Self { + owned_ranges: VmLockedValue::new(OwnedRanges::default()), + elf_patch_cache: VmLockedValue::new(BTreeMap::new()), + brk: AtomicUsize::new(0), + futex_namespace, + family_id: AtomicUsize::new(0), + } + } } -impl Task { - pub(crate) fn process(&self) -> &Arc> { - &self.thread.process - } +struct VmBookkeepingSlot { + current: Mutex>>, +} - /// Set the current task's command name. - pub(crate) fn set_task_comm(&self, comm: &[u8]) { - let mut new_comm = [0u8; litebox_common_linux::TASK_COMM_LEN]; - let comm = &comm[..comm.len().min(litebox_common_linux::TASK_COMM_LEN - 1)]; - new_comm[..comm.len()].copy_from_slice(comm); - self.comm.set(new_comm); +impl VmBookkeepingSlot { + fn new(futex_namespace: usize) -> Self { + Self { + current: Mutex::new(Arc::new(VmBookkeeping::new(futex_namespace))), + } } - /// Handle syscall `prctl`. - pub(crate) fn sys_prctl(&self, arg: PrctlArg) -> Result { - match arg { - PrctlArg::GetName(name) => name - .write_slice_at_offset::(0, &self.comm.get()) - .ok_or(Errno::EFAULT) - .map(|()| 0), - PrctlArg::SetName(name) => { - let mut name_buf = [0u8; litebox_common_linux::TASK_COMM_LEN - 1]; - // strncpy - for (i, byte) in name_buf.iter_mut().enumerate() { - let b = name - .read_at_offset::(isize::try_from(i).unwrap()) - .ok_or(Errno::EFAULT)?; - if b == 0 { - break; - } - *byte = b; - } - self.set_task_comm(&name_buf); - Ok(0) - } - PrctlArg::CapBSetRead(cap) => { - // Return 1 if the capability specified in cap is in the calling - // thread's capability bounding set, or 0 if it is not. - if cap - > litebox_common_linux::CapSet::LAST_CAP - .bits() - .trailing_zeros() as usize - { - return Err(Errno::EINVAL); - } - // Note we don't support capabilities in LiteBox, so we always return 0. - Ok(0) - } - _ => unimplemented!(), + fn shared_with(other: &Self) -> Self { + Self { + current: Mutex::new(other.current.lock().clone()), } } - /// Handle syscall `arch_prctl`. - pub(crate) fn sys_arch_prctl(&self, arg: ArchPrctlArg) -> Result<(), Errno> { - match arg { - #[cfg(target_arch = "x86_64")] - ArchPrctlArg::SetFs(addr) => self - .global - .platform - .set_arch_specific_register(&ArchSpecificRegister::FsBase, addr) - .map_err(Errno::from), - #[cfg(target_arch = "x86_64")] - ArchPrctlArg::GetFs(addr) => { - let fsbase = self - .global - .platform - .get_arch_specific_register(&ArchSpecificRegister::FsBase)?; - addr.write_at_offset::(0, fsbase) - .ok_or(Errno::EFAULT)?; - Ok(()) - } - ArchPrctlArg::CETStatus | ArchPrctlArg::CETDisable | ArchPrctlArg::CETLock => { - Err(Errno::EINVAL) - } - _ => unimplemented!(), - } + fn current(&self) -> Arc> { + self.current.lock().clone() + } + + fn detach(&self, futex_namespace: usize) { + *self.current.lock() = Arc::new(VmBookkeeping::new(futex_namespace)); } } -const ROBUST_LIST_LIMIT: isize = 2048; +pub(crate) struct SharedVmLockedField { + vm: Arc>, + field: fn(&VmBookkeeping) -> &VmLockedValue, +} + +impl SharedVmLockedField { + fn new( + vm: Arc>, + field: fn(&VmBookkeeping) -> &VmLockedValue, + ) -> Self { + Self { vm, field } + } -/* - * Process a futex-list entry, check whether it's owned by the - * dying task, and do notification if so: - */ -fn handle_futex_death(futex_addr: UserPtr, _pi: bool, _pending_op: bool) -> Result<(), Errno> { - if !futex_addr.as_usize().is_multiple_of(4) { - return Err(Errno::EINVAL); + pub(crate) fn lock(&self) -> SharedVmLockedFieldGuard { + let vm = self.vm.current(); + (self.field)(&vm).lock(); + SharedVmLockedFieldGuard { + vm, + field: self.field, + } } +} - todo!("handle_futex_death is not implemented yet"); +pub(crate) struct SharedVmLockedFieldGuard { + vm: Arc>, + field: fn(&VmBookkeeping) -> &VmLockedValue, } -fn fetch_robust_entry( - head: UserPtr, -) -> (UserPtr, bool) { - let next = head.as_usize(); - (UserPtr::from_usize(next & !1), next & 1 != 0) +impl SharedVmLockedFieldGuard { + fn value(&self) -> &VmLockedValue { + (self.field)(&self.vm) + } } -fn wake_robust_list( - head: UserPtr, -) -> Result<(), Errno> { - let mut limit = ROBUST_LIST_LIMIT; - let head_ptr = head.as_usize(); - let head = head.read_at_offset::(0).ok_or(Errno::EFAULT)?; - let (mut entry, mut pi) = fetch_robust_entry(UserPtr::from_usize(head.list.next)); - let (pending, ppi) = fetch_robust_entry(UserPtr::from_usize(head.list_op_pending)); - let futex_offset = head.futex_offset; - let entry_head = head_ptr + offset_of!(litebox_common_linux::RobustListHead, list); - while entry.as_usize() != entry_head && limit > 0 { - let nxt = entry - .read_at_offset::(0) - .map(|e| fetch_robust_entry(UserPtr::from_usize(e.next))); - if entry.as_usize() != pending.as_usize() { - handle_futex_death( - UserPtr::from_usize(entry.as_usize() + futex_offset), - pi, - false, - )?; - } - let Some((next_entry, next_pi)) = nxt else { - return Err(Errno::EFAULT); - }; +impl Deref for SharedVmLockedFieldGuard { + type Target = T; - entry = next_entry; - pi = next_pi; - limit -= 1; + fn deref(&self) -> &Self::Target { + unsafe { &*self.value().value.get() } } +} - if pending.as_usize() != 0 { - let _ = handle_futex_death( - UserPtr::from_usize(pending.as_usize() + futex_offset), - ppi, - true, - ); +impl DerefMut for SharedVmLockedFieldGuard { + fn deref_mut(&mut self) -> &mut Self::Target { + unsafe { &mut *self.value().value.get() } } - Ok(()) } -impl Task { - /// Called when the task is exiting. - pub(crate) fn prepare_for_exit(&mut self) { - self.thread.detach_from_process(); - - if let Some(clear_child_tid) = self.thread.clear_child_tid.take() { - // Clear the child TID if requested - // TODO: if we are the last thread, we don't need to clear it - let _ = clear_child_tid.write_at_offset::(0, 0); - // Cast from *i32 to *u32 - let clear_child_tid = UserPtrMut::from_usize(clear_child_tid.as_usize()); - let _ = self.sys_futex(litebox_common_linux::FutexArgs::Wake { - addr: clear_child_tid, - flags: litebox_common_linux::FutexFlags::PRIVATE, - count: 1, - }); - } - if let Some(robust_list) = self.thread.robust_list.take() { - let _ = wake_robust_list::(robust_list); - } +impl Drop for SharedVmLockedFieldGuard { + fn drop(&mut self) { + unsafe { self.value().unlock() }; } +} - pub(crate) fn sys_exit(&self, status: i32) { - // The `Task` will be dropped on the way out of the shim, which will - // call `self.prepare_for_exit()`. - self.exit_thread(status.trunc()); +pub(crate) struct SharedVmAtomicUsize { + vm: Arc>, + field: fn(&VmBookkeeping) -> &AtomicUsize, +} + +impl SharedVmAtomicUsize { + fn new( + vm: Arc>, + field: fn(&VmBookkeeping) -> &AtomicUsize, + ) -> Self { + Self { vm, field } } - pub(crate) fn sys_exit_group(&self, status: i32) { - // Tear down occurs similarly to `sys_exit`. - self.exit_group(ExitStatus::Exit(status.trunc())); + pub(crate) fn load(&self, order: Ordering) -> usize { + (self.field)(&self.vm.current()).load(order) + } + + pub(crate) fn store(&self, value: usize, order: Ordering) { + (self.field)(&self.vm.current()).store(value, order); } } -/// A descriptor for thread-local storage (TLS). +/// One guest address space, shared by a `fork`ed child and its parent, plus the hand-off that +/// keeps exactly one of them running on it at a time. /// -/// On `x86_64`, this is represented as a `*mut u8`. The TLS pointer can point to -/// an arbitrary-sized memory region. -#[cfg(target_arch = "x86_64")] -type ThreadLocalDescriptor = UserPtrMut; - -struct NewThreadArgs { - /// Task struct that maintains all per-thread data - task: Task, +/// LiteBox executes guest code natively, so a guest virtual address *is* a host virtual address +/// (see `litebox::mm::linux::Vmem::insert_mapping`, which passes the guest's own range straight to +/// the platform allocator). One host address space therefore cannot hold two guest processes that +/// both believe they own the same addresses, which is exactly what a copying `fork` would have to +/// produce. So the child runs in the parent's address space, on the parent's stack. +/// +/// What this type adds is that the parent does not have to stay suspended for the child's whole +/// lifetime. The address space is a *token*. Its holder is the one member whose memory is +/// currently live in it; every other member is parked, holding a host-memory copy of its own view +/// (see [`AddressSpaceMembership::parked`]). A member gives the token up whenever it is about to +/// block -- `litebox::event::wait::CheckForInterrupt::yield_while_blocking`, which fires for every +/// interruptible wait in the shim -- and takes it back before it looks at guest memory again. +/// Since a member only ever reads or writes guest memory while it holds the token, and taking the +/// token restores that member's own copy, each member sees exactly the memory `fork(2)` promises +/// it. +/// +/// That is what makes a `fork`ed child that never `execve`s -- a shell builtin on the left of a +/// pipeline, a background subshell -- able to run concurrently with its parent: when it blocks on +/// a full pipe, the parent gets the address space back and can fork the stage that drains it. +/// +/// A member leaves for good when it `execve`s (the new image is loaded at addresses no other +/// member owns, so it no longer needs the token) or when it exits. +/// +/// Known limits, all of them "it hangs", never "it silently returns the wrong bytes": +/// +/// * A member that never blocks and never exits starves the others. The token is only ever +/// yielded voluntarily; there is no preemption, because memory cannot be taken away from a +/// thread that is in the middle of executing guest instructions on it. +/// * Membership is per *process*: every thread of a member runs on the memory while the process +/// holds the token. A single-threaded member gives the token up whenever it blocks, as above. +/// A multithreaded member gives it up only when another member is waiting for it (a waiter +/// kicks the holder's threads, see [`Self::acquire`]): the first thread to notice at a safe +/// point closes the process's fork gate so every sibling parks off the memory +/// ([`Task::quiesce_and_hand_off`]), copies the image out, releases, and waits for the token +/// to come back before reopening the gate. That is what lets a `fork`ed child that never +/// `exec`s -- a Chromium zygote's renderer -- create threads. The cost is that every hand-off +/// copies the member's whole shared image, and a member busy in guest code yields only at its +/// next syscall or vCPU kick. +pub(crate) struct SharedAddressSpace { + /// [`ADDRESS_SPACE_FREE`], or the pid of the member holding the token. Used directly as the + /// word members block on while waiting to acquire. + holder: ::RawMutex, + /// Members currently blocked in [`Self::acquire`]. A multithreaded holder yields only while + /// this is non-zero (see [`Task::yield_address_space_to_waiters`]). + waiters: AtomicUsize, + /// The holder's thread table, so a waiter can kick its threads to a safe point. Cleared on + /// release. + holder_threads: Mutex>>>>, } -impl litebox::shim::InitThread for NewThreadArgs { - type ExecutionContext = litebox_common_linux::PtRegs; - - fn init( - self: alloc::boxed::Box, - ) -> alloc::boxed::Box> - { - let Self { task } = *self; +const PROCESS_LAUNCH_PENDING: u32 = 0; +const PROCESS_LAUNCH_COMMITTED: u32 = 1; +const PROCESS_LAUNCH_ABORTED: u32 = 2; - Box::new(crate::LinuxShimEntrypoints { - task, - _not_send: core::marker::PhantomData, - }) - } +struct ProcessLaunch { + state: ::RawMutex, } -impl Task { - pub(crate) fn sys_clone( - &self, - ctx: &litebox_common_linux::PtRegs, - args: &litebox_common_linux::CloneArgs, - ) -> Result { - self.do_clone(ctx, args, false) +impl ProcessLaunch { + fn new() -> Self { + Self { + state: ::RawMutex::INIT, + } } - pub(crate) fn sys_clone3( - &self, - ctx: &litebox_common_linux::PtRegs, - args: UserPtr, - ) -> Result { - let args = args.read_at_offset::(0).ok_or(Errno::EFAULT)?; - self.do_clone(ctx, &args, true) + fn commit(&self) { + self.state + .underlying_atomic() + .store(PROCESS_LAUNCH_COMMITTED, Ordering::Release); + self.state.wake_all(); } - /// Creates a new thread or process. - /// - /// Note we currently only support creating threads with the VM, FS, and FILES flags set. - fn do_clone( - &self, - ctx: &litebox_common_linux::PtRegs, - args: &litebox_common_linux::CloneArgs, - clone3: bool, - ) -> Result { - const MAX_SIGNAL_NUMBER: u64 = 64; - - let litebox_common_linux::CloneArgs { - mut flags, - pidfd: _, - child_tid, - parent_tid, - exit_signal, - stack, - stack_size, - tls, - set_tid, - set_tid_size, - cgroup, - } = *args; + fn abort(&self) { + self.state + .underlying_atomic() + .store(PROCESS_LAUNCH_ABORTED, Ordering::Release); + self.state.wake_all(); + } - // `CLONE_DETACHED` is ignored but has been reserved for reuse with - // `clone3` or in combination with `CLONE_PIDFD`. - if !clone3 && !flags.contains(CloneFlags::PIDFD) { - flags.remove(CloneFlags::DETACHED); + fn wait(&self) -> bool { + loop { + let state = self.state.underlying_atomic().load(Ordering::Acquire); + match state { + PROCESS_LAUNCH_COMMITTED => return true, + PROCESS_LAUNCH_ABORTED => return false, + PROCESS_LAUNCH_PENDING => { + let _ = self.state.block(PROCESS_LAUNCH_PENDING); + } + _ => unreachable!("invalid process launch state"), + } } + } - let required_clone_flags = - CloneFlags::VM | CloneFlags::THREAD | CloneFlags::SIGHAND | CloneFlags::FILES; + fn is_committed(&self) -> bool { + self.state.underlying_atomic().load(Ordering::Acquire) == PROCESS_LAUNCH_COMMITTED + } +} - let supported_clone_flags = CloneFlags::VM - | CloneFlags::FS - | CloneFlags::FILES - | CloneFlags::SIGHAND - | CloneFlags::PARENT - | CloneFlags::THREAD - | CloneFlags::SETTLS - | CloneFlags::PARENT_SETTID - | CloneFlags::CHILD_CLEARTID - | CloneFlags::CHILD_SETTID - // Ignored since we don't support sysv semaphores anyway. - | CloneFlags::SYSVSEM; +const VFORK_ACTIVE: u32 = 0; +const VFORK_COMPLETE: u32 = 1; + +pub(crate) struct VforkCompletion { + state: ::RawMutex, + /// Whether the vfork parent was already a [`SharedAddressSpace`] member when it vforked. A + /// child on the parent's memory then cannot fork (see `do_process_clone`): the membership it + /// would hand back below has nowhere to go. + parent_shares_address_space: bool, + /// The membership a `CLONE_VM` child that forked while still on its parent's memory hands to + /// that parent as it exits or execs. See [`Task::hand_address_space_to_vfork_parent`]. + inherited_membership: Mutex>>>, +} - if flags.intersects(!supported_clone_flags) { - log_unsupported!( - "clone with unsupported flags: {:?}", - flags & !supported_clone_flags - ); - return Err(Errno::EINVAL); +impl VforkCompletion { + fn new(parent_shares_address_space: bool) -> Self { + let state = ::RawMutex::INIT; + state + .underlying_atomic() + .store(VFORK_ACTIVE, Ordering::Relaxed); + Self { + state, + parent_shares_address_space, + inherited_membership: Mutex::new(None), } - if !flags.contains(required_clone_flags) { - log_unsupported!( - "clone with missing required flags: {:?}", - required_clone_flags & !flags - ); - return Err(Errno::EINVAL); + } + + fn complete(&self) { + if self + .state + .underlying_atomic() + .swap(VFORK_COMPLETE, Ordering::Release) + == VFORK_ACTIVE + { + self.state.wake_all(); } + } - if cgroup != 0 { - log_unsupported!("clone with cgroup"); - return Err(Errno::EINVAL); + fn wait(&self) { + loop { + let state = self.state.underlying_atomic().load(Ordering::Acquire); + if state == VFORK_COMPLETE { + return; + } + let _ = self.state.block(state); } + } +} - if set_tid != 0 || set_tid_size != 0 { - log_unsupported!("clone with set_tid"); - return Err(Errno::EINVAL); +/// One piece of a guest process's view of the memory it shares with other members: the page +/// protection it had, and -- for a writable piece -- its contents. +struct SavedRange { + start: usize, + end: usize, + /// The mapping's protection (`VM_READ`/`VM_WRITE`/`VM_EXEC`) at save time. + flags: VmFlags, + /// The bytes, for a writable piece. A non-writable piece (a `PROT_NONE` allocator + /// reservation, a read-only segment) carries only its protection, which the restore + /// re-applies -- and, for `PROT_NONE`, clears whatever another member left there, since + /// Linux hands a process zero pages when it commits such a range. + bytes: Option>, +} + +/// A copy of a guest process's view of its shared memory; see [`SavedRange`]. +type MemoryImage = Vec; + +/// The `mprotect` protection a page-manager mapping's flags describe. +fn prot_of(flags: VmFlags) -> litebox_common_linux::ProtFlags { + use litebox_common_linux::ProtFlags; + let mut prot = ProtFlags::PROT_NONE; + prot.set(ProtFlags::PROT_READ, flags.contains(VmFlags::VM_READ)); + prot.set(ProtFlags::PROT_WRITE, flags.contains(VmFlags::VM_WRITE)); + prot.set(ProtFlags::PROT_EXEC, flags.contains(VmFlags::VM_EXEC)); + prot +} + +/// The protection bits of a mapping's flags, for comparing two members' views of one range. +fn access_bits(flags: VmFlags) -> VmFlags { + flags & (VmFlags::VM_READ | VmFlags::VM_WRITE | VmFlags::VM_EXEC) +} + +/// Above this many disjoint below-SP ABI ranges, preserve their one bounding interval instead. +const MAX_PRESERVED_STACK_RANGES: usize = 64; + +/// One member process's place in a [`SharedAddressSpace`], shared by all of its threads (hence +/// `Arc` in [`Process::address_space`] and interior synchronization throughout). +pub(crate) struct AddressSpaceMembership { + shared: Arc>, + /// Whether this member currently holds the token. + holding: AtomicBool, + /// This member's copy of its own private memory, taken when it gave the token up. `Some` + /// exactly while [`Self::holding`] is false and the member still intends to come back. + parked: Mutex>, + /// Small guest ranges whose contents remain semantically live even when they sit below the + /// current stack pointer. Clone child-TID words are the canonical case. + preserved_stack_ranges: Mutex, + /// The ranges this member has ever shared with another member: its owned ranges at every + /// `fork` it took part in (for a child, the parent's at that moment). Only these need + /// copying out and back on a hand-off -- memory a member mapped afterwards is at addresses + /// no other member owns, so nobody else can disturb it. For a renderer that grows to + /// hundreds of megabytes after being forked from a small zygote, this is the difference + /// between copying the zygote's image and copying everything. + shared_ranges: Mutex, + /// Set by the one thread quiescing this multithreaded member for a hand-off; see + /// [`Task::quiesce_and_hand_off`]. + quiescing: AtomicBool, + /// When this member last took the token, as the platform's monotonic clock. A + /// multithreaded member keeps the token for at least [`Self::QUANTUM`] before yielding to + /// a waiter: every hand-off copies its whole shared image out and back, so yielding at the + /// first syscall after each acquisition -- with several runnable members, every few + /// microseconds -- spent everything on copying and nothing on the guest (live-measured: + /// 22,606 hand-offs in 90 s, none of the members getting anywhere). + acquired_at: Mutex>, +} + +impl AddressSpaceMembership { + fn new( + shared: Arc>, + preserved_stack_ranges: OwnedRanges, + shared_ranges: OwnedRanges, + ) -> Self { + Self { + shared, + holding: AtomicBool::new(true), + parked: Mutex::new(None), + preserved_stack_ranges: Mutex::new(preserved_stack_ranges), + shared_ranges: Mutex::new(shared_ranges), + quiescing: AtomicBool::new(false), + acquired_at: Mutex::new(None), } + } - // Note `exit_signal` is ignored because we don't support `fork` yet; we just validate it. - if exit_signal > MAX_SIGNAL_NUMBER { - return Err(Errno::EINVAL); + /// The least a multithreaded member runs between hand-offs; see [`Self::acquired_at`]. + const QUANTUM: Duration = Duration::from_millis(40); + + fn holding(&self) -> bool { + self.holding.load(Ordering::Acquire) + } + + fn mark_acquired(&self, now: Platform::Instant) { + *self.acquired_at.lock() = Some(now); + } + + /// Whether this member has had the token for at least [`Self::QUANTUM`]. + fn quantum_elapsed(&self, now: Platform::Instant) -> bool { + self.acquired_at + .lock() + .is_none_or(|since| now.duration_since(&since) >= Self::QUANTUM) + } +} + +/// The value of [`SharedAddressSpace::holder`] when no member holds the token. No tid is ever +/// zero, so this cannot collide with one. +const ADDRESS_SPACE_FREE: u32 = 0; + +/// Encodes a tid as a [`SharedAddressSpace::holder`] value. +fn tid_as_holder(tid: i32) -> u32 { + let raw = tid.cast_unsigned(); + assert_ne!(raw, ADDRESS_SPACE_FREE, "tid 0 cannot own an address space"); + raw +} + +impl SharedAddressSpace { + /// How long to block before re-checking whether this task is being torn down, and between + /// kicks of a multithreaded holder's threads. The token is handed over explicitly, so in the + /// single-threaded case this only bounds how long a *dying* task waits for a holder that + /// will never release; it is not a polling interval in the normal case. + const ABANDON_CHECK_INTERVAL: Duration = Duration::from_millis(20); + + fn new( + initial_holder: i32, + holder_inner: &Arc>>, + ) -> Self { + let holder = ::RawMutex::INIT; + holder + .underlying_atomic() + .store(tid_as_holder(initial_holder), Ordering::Relaxed); + Self { + holder, + waiters: AtomicUsize::new(0), + holder_threads: Mutex::new(Some(Arc::downgrade(holder_inner))), } + } - let tls = if flags.contains(CloneFlags::SETTLS) { - let addr = tls.trunc(); - #[cfg(target_arch = "x86_64")] - { - // Validate the user-controlled TLS base before spawning the thread. - if !litebox_common_linux::arch::is_valid_user_fs_base(addr) { - return Err(Errno::EPERM); - } + fn waiters(&self) -> usize { + self.waiters.load(Ordering::Acquire) + } + + /// Interrupts every thread of the current holder so each reaches a safe point and, if the + /// holder is multithreaded, notices the waiter (see [`Task::yield_address_space_to_waiters`]). + fn kick_holder(&self) { + let holder = self.holder_threads.lock().clone(); + if let Some(inner) = holder.and_then(|weak| weak.upgrade()) { + for thread in inner.lock().threads.values() { + thread.interrupt(); } - #[cfg(target_arch = "x86_64")] - let desc = UserPtrMut::from_usize(addr); - Some(desc) - } else { + } + } + + /// Blocks until the token is free and takes it for the process `pid`, whose thread table is + /// `inner`. + /// + /// Returns `true` once the process holds the token -- taken here, or (`already_held`) by a + /// sibling thread of the same process in the meantime. Returns `false` if `abandon` became + /// true first, which only happens when the caller is being torn down and will never run + /// guest code again. + fn acquire( + &self, + pid: i32, + already_held: impl Fn() -> bool, + mut abandon: impl FnMut() -> bool, + inner: &Arc>>, + ) -> bool { + let me = tid_as_holder(pid); + let mut waiting = false; + let outcome = loop { + if already_held() { + break true; + } + match self.holder.underlying_atomic().compare_exchange( + ADDRESS_SPACE_FREE, + me, + Ordering::Acquire, + Ordering::Relaxed, + ) { + Ok(_) => { + *self.holder_threads.lock() = Some(Arc::downgrade(inner)); + // Siblings blocked below on the holder word re-check `already_held`. + self.holder.wake_all(); + break true; + } + Err(current) => { + if abandon() { + break false; + } + if !waiting { + waiting = true; + self.waiters.fetch_add(1, Ordering::AcqRel); + } + self.kick_holder(); + let _ = self + .holder + .block_or_timeout(current, Self::ABANDON_CHECK_INTERVAL); + } + } + }; + if waiting { + self.waiters.fetch_sub(1, Ordering::AcqRel); + } + outcome + } + + /// Gives the token up, waking anything waiting for it. + fn release(&self) { + *self.holder_threads.lock() = None; + self.holder + .underlying_atomic() + .store(ADDRESS_SPACE_FREE, Ordering::Release); + self.holder.wake_all(); + } + + /// After a [`Self::release`] made on a waiter's behalf: blocks until some waiter has taken + /// the token (or every waiter has given up), so the releasing member -- still on-CPU and + /// about to re-acquire -- cannot snatch it straight back. Without this a multithreaded + /// member quiesced, released and re-acquired thousands of times a second while the waiter + /// it had yielded for never once won the race (live-measured: 23,444 hand-offs with + /// `away_us=0` in 90 s). + fn wait_until_taken(&self, mut abandon: impl FnMut() -> bool) { + loop { + if self.waiters() == 0 || abandon() { + return; + } + let current = self.holder.underlying_atomic().load(Ordering::Acquire); + if current != ADDRESS_SPACE_FREE { + return; + } + let _ = self + .holder + .block_or_timeout(ADDRESS_SPACE_FREE, Self::ABANDON_CHECK_INTERVAL); + } + } + + /// Passes the token straight to process `pid` (thread table `inner`) without ever making it + /// free. + /// + /// Used by `fork`: the child is not running yet and so cannot [`Self::acquire`] for itself, + /// and a free window here would let some other member take the address space out from under + /// it before its first instruction. + fn hand_off_to(&self, pid: i32, inner: &Arc>>) { + *self.holder_threads.lock() = Some(Arc::downgrade(inner)); + self.holder + .underlying_atomic() + .store(tid_as_holder(pid), Ordering::Release); + } +} + +/// Parent/child relationships and exit statuses of every guest process in the shim. +/// +/// This is the bookkeeping `wait4` reaps from. It is deliberately separate from [`Process`], +/// which models a *thread group*: a zombie has to outlive its `Process` (the parent may not call +/// `wait4` until long after the child's last thread is gone), and a waiting parent has to be able +/// to name a child it holds no reference to. +pub(crate) struct ProcessTable { + inner: Mutex>, +} + +struct ProcessTableInner { + /// Every live or zombie child, keyed by its pid. + children: BTreeMap, + /// Parents currently blocked in `wait4`, as (parent pid, registration token, waker). + waiters: Vec<(i32, u64, litebox::event::wait::Waker)>, + next_waiter_token: u64, + /// Every live guest process, so that a signal can be posted to one of them from another. + live: BTreeMap>, +} + +/// The handle needed to post a process-directed signal to another guest process. +/// +/// Deliberately not a `Task`: the sender runs on a different host thread, and a `Task` is +/// full of `Cell`s and `RefCell`s that only its own thread may touch. Everything here is +/// `Sync`. +struct LiveProcess { + /// The target's process-group identity. Kept separate from `Process` because platform timer + /// handles inside that object are not required to be `Send`, while this table is shim-global. + process_group_id: Weak, + /// The target's process-wide pending queue -- the same one its own threads drain from. + /// Survives `execve` (which replaces the handler table, not this). + signals: crate::syscalls::signal::RemoteSignalTarget, + /// The target state needed to wake every thread after posting a signal. + process_inner: Weak>>, + /// The target's live resource limits, used when queueing user-originated signals. + limits: Weak, + /// The target's VM identity slot, so a cross-lineage guest-address-range overlap can be + /// detected against its `owned_ranges` (see [`ProcessTable::overlaps_another_process`]). + /// `Weak>` rather than `Weak>`: the latter drags in + /// `Process::alarm_timer`'s platform `TimerHandle`, which is not `Send`, and this table must + /// stay `Sync` (see the other fields' narrow `Weak`s, chosen for the same reason). + /// + /// DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): added to + /// directly test the hypothesis in `litebox-chromium-zygote-fork-corruption.md` -- that the + /// one flat, process-blind host address space (see `Process::owned_ranges`'s own doc + /// comment: "they live at disjoint addresses in the one host address space") ever actually + /// fails to keep two unrelated guest processes' address ranges disjoint, which is the + /// precondition for one process's fork/save/restore machinery to ever touch another's live + /// memory. + vm: Weak>, +} + +/// One selected recipient of a process-directed signal: its pending queue, the thread list to +/// wake once the signal is posted, and the limits a user-originated signal is queued against. +type SignalTarget = ( + crate::syscalls::signal::RemoteSignalTarget, + Arc>>, + Arc, +); + +/// The initial guest process's pid; `GlobalState::next_thread_id` starts at 2 to leave it free. +const INIT_PID: i32 = 1; + +struct ChildRecord { + ppid: i32, + /// Signal the child asked to receive if `ppid` exits; `None` means disabled. + parent_death_signal: Option, + /// `None` while the child is still running; `Some` once it is a zombie awaiting `wait4`. + status: Option, + /// Total host CPU time (nanoseconds) the child consumed, set alongside `status`. See + /// `Process::cpu_time_nanos`. + cpu_time_nanos: u64, +} + +impl ProcessTable { + pub(crate) fn new() -> Self { + Self { + inner: Mutex::new(ProcessTableInner { + children: BTreeMap::new(), + waiters: Vec::new(), + next_waiter_token: 0, + live: BTreeMap::new(), + }), + } + } + + /// Records a newly `fork`ed child of `parent`. + fn add_child(&self, child: i32, parent: i32) { + let old = self.inner.lock().children.insert( + child, + ChildRecord { + ppid: parent, + parent_death_signal: None, + status: None, + cpu_time_nanos: 0, + }, + ); + assert!(old.is_none(), "pid {child} is already live"); + } + + /// Registers a live guest process so signals can be posted to it. + fn register_process( + &self, + pid: i32, + signals: crate::syscalls::signal::RemoteSignalTarget, + process: &Arc>, + ) { + self.inner.lock().live.insert( + pid, + LiveProcess { + process_group_id: Arc::downgrade(&process.process_group_id), + signals, + process_inner: Arc::downgrade(&process.inner), + limits: Arc::downgrade(&process.limits), + vm: Arc::downgrade(&process.owned_ranges.vm), + }, + ); + } + + /// DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): every OTHER + /// live process (by pid) whose `owned_ranges` currently overlaps `range` AND whose + /// `family_id` differs from `self_family_id` (0 = not in any family) -- i.e. excludes the + /// EXPECTED overlap between two members of the SAME `SharedAddressSpace` family (which, by + /// this platform's design, believe they own the same addresses; see that type's own doc + /// comment), surfacing only overlap between processes that are supposed to have disjoint + /// memory. Each hit also carries the other process's own `family_id`, so a hit can still be + /// told apart from "family-id tracking itself missed a relationship" (e.g. vfork, which does + /// not go through `Task::join_address_space`) during triage. An empty result under every + /// real trial is direct evidence against the "flat shared address space lets unrelated + /// lineages collide" hypothesis; any non-empty result is the smoking gun the investigation is + /// looking for -- see call sites in `Task::save_address_space` / `Task::restore_address_space`. + fn overlaps_another_process( + &self, + self_pid: i32, + self_family_id: usize, + range: &Range, + ) -> Vec<(i32, usize, Range)> { + if range.start >= range.end { + return Vec::new(); + } + let others: Vec<(i32, Arc>)> = { + let inner = self.inner.lock(); + inner + .live + .iter() + .filter(|&(&pid, _)| pid != self_pid) + .filter_map(|(&pid, live)| Some((pid, live.vm.upgrade()?))) + .collect() + }; + let mut hits = Vec::new(); + for (other_pid, vm) in others { + let bookkeeping = vm.current(); + let other_family_id = bookkeeping.family_id.load(Ordering::Acquire); + if self_family_id != 0 && self_family_id == other_family_id { + continue; + } + bookkeeping.owned_ranges.lock(); + // SAFETY: `owned_ranges.lock()` above establishes exclusive access to the cell until + // `unlock()` below, mirroring `SharedVmLockedFieldGuard`'s own Deref. + let overlaps: Vec> = + unsafe { &*bookkeeping.owned_ranges.value.get() } + .intersect(range) + .collect(); + unsafe { bookkeeping.owned_ranges.unlock() }; + for overlap in overlaps { + hits.push((other_pid, other_family_id, overlap)); + } + } + hits + } + + /// Every registered live pid, ascending -- what `/proc` lists. + pub(crate) fn live_pids(&self) -> Vec { + self.inner.lock().live.keys().copied().collect() + } + + /// Thread `tid` of live process `tgid`, with that process's resource limits, for a + /// thread-directed signal from another process (`tgkill`/`rt_tgsigqueueinfo`). `None` when + /// either is not live, which is Linux's `ESRCH`. + pub(crate) fn remote_thread( + &self, + tgid: i32, + tid: i32, + ) -> Option<(Arc>, Arc)> { + let (process_inner, limits) = { + let inner = self.inner.lock(); + let live = inner.live.get(&tgid)?; + (live.process_inner.upgrade()?, live.limits.upgrade()?) + }; + let thread = process_inner.lock().threads.get(&tid).cloned()?; + Some((thread, limits)) + } + + /// The live threads of process `pid` (empty when it is not live), plus its process group + /// and real uid, for `getpriority`/`setpriority` over another process. Table lock first, + /// then the process lock, the order `send_process_signal` uses. + pub(crate) fn priority_targets( + &self, + pid: i32, + ) -> Option<(Vec>>, i32, u32)> { + let (process_inner, pgid) = { + let inner = self.inner.lock(); + let live = inner.live.get(&pid)?; + ( + live.process_inner.upgrade()?, + live.process_group_id.upgrade()?.load(Ordering::Relaxed), + ) + }; + let inner = process_inner.lock(); + Some(( + inner.threads.values().cloned().collect(), + pgid, + inner.identity.uid, + )) + } + + /// The `/proc/` view of registered live process `pid`, or `None` when no such process + /// is registered (or it has already torn down its `Process`). + pub(crate) fn proc_task_info(&self, pid: i32) -> Option { + // Take the process handle out from under the table lock before locking the process + // itself, the same order `send_process_signal` uses. + let process_inner = self.inner.lock().live.get(&pid)?.process_inner.upgrade()?; + let inner = process_inner.lock(); + Some(proc_task_info(pid, &inner)) + } + + pub(crate) fn send_process_signal( + &self, + pid: i32, + signal: litebox_common_linux::signal::Signal, + siginfo: litebox_common_linux::signal::Siginfo, + ) -> bool { + let Some((signals, process_inner, limits)) = ({ + let inner = self.inner.lock(); + inner.live.get(&pid).and_then(|live| { + Some(( + live.signals.clone(), + live.process_inner.upgrade()?, + live.limits.upgrade()?, + )) + }) + }) else { + return false; + }; + signals.post_from_user(&limits, signal, siginfo); + let inner = process_inner.lock(); + for thread in inner.threads.values() { + thread.interrupt(); + } + true + } + + /// Selects every live process `select` accepts. + /// + /// Target discovery and weak-reference upgrades happen under one process-table lock, so an + /// exit cannot leave a selected target half-upgraded. Signal delivery and thread wakeups happen + /// after releasing that lock, in [`Self::post_to_targets`]. + fn select_signal_targets( + &self, + mut select: impl FnMut(i32, &LiveProcess) -> bool, + ) -> Vec> { + let inner = self.inner.lock(); + inner + .live + .iter() + .filter_map(|(&pid, live)| { + select(pid, live).then(|| { + Some(( + live.signals.clone(), + live.process_inner.upgrade()?, + live.limits.upgrade()?, + )) + })? + }) + .collect() + } + + /// Posts a user-originated `signal` to each of `targets` and wakes every one of their threads. + /// Returns how many processes were signalled. + fn post_to_targets( + targets: &[SignalTarget], + signal: litebox_common_linux::signal::Signal, + siginfo: &litebox_common_linux::signal::Siginfo, + ) -> usize { + for (signals, process_inner, limits) in targets { + signals.post_from_user(limits, signal, siginfo.clone()); + for thread in process_inner.lock().threads.values() { + thread.interrupt(); + } + } + targets.len() + } + + fn in_process_group(live: &LiveProcess, process_group_id: i32) -> bool { + live.process_group_id + .upgrade() + .is_some_and(|group| group.load(Ordering::Acquire) == process_group_id) + } + + /// Posts a process-directed signal to every live process in `process_group_id` except + /// `excluded_pid`. Returns how many processes were signalled. + pub(crate) fn send_process_group_signal( + &self, + process_group_id: i32, + excluded_pid: i32, + signal: litebox_common_linux::signal::Signal, + siginfo: litebox_common_linux::signal::Siginfo, + ) -> usize { + let targets = self.select_signal_targets(|pid, live| { + pid != excluded_pid && Self::in_process_group(live, process_group_id) + }); + Self::post_to_targets(&targets, signal, &siginfo) + } + + /// Posts a process-directed signal to every live process except `excluded_pid` and the + /// initial process, which `kill(-1, sig)` spares exactly as Linux spares init. Returns how + /// many processes were signalled. + pub(crate) fn send_signal_to_all_processes( + &self, + excluded_pid: i32, + signal: litebox_common_linux::signal::Signal, + siginfo: litebox_common_linux::signal::Siginfo, + ) -> usize { + let targets = self.select_signal_targets(|pid, _| pid != excluded_pid && pid != INIT_PID); + Self::post_to_targets(&targets, signal, &siginfo) + } + + /// Whether some live process other than `excluded_pid` is in `process_group_id` -- the + /// existence test behind `kill(-pgid, 0)`. + pub(crate) fn has_process_group_member( + &self, + process_group_id: i32, + excluded_pid: i32, + ) -> bool { + self.inner.lock().live.iter().any(|(&pid, live)| { + pid != excluded_pid && Self::in_process_group(live, process_group_id) + }) + } + + /// Whether [`Self::send_signal_to_all_processes`] from `excluded_pid` would reach anything -- + /// the existence test behind `kill(-1, 0)`. + pub(crate) fn has_other_live_process(&self, excluded_pid: i32) -> bool { + self.inner + .lock() + .live + .keys() + .any(|&pid| pid != excluded_pid && pid != INIT_PID) + } + + /// Returns whether `pid` currently names a live guest process. + pub(crate) fn is_live(&self, pid: i32) -> bool { + self.inner.lock().live.contains_key(&pid) + } + + /// Returns the process-group identity for `child` only when it is a live child of `parent`. + /// + /// Parentage and liveness are observed under one table lock, so exit cannot interleave between + /// validating the relationship and upgrading the live process's weak group reference. + fn live_child_process_group_id(&self, parent: i32, child: i32) -> Option> { + let inner = self.inner.lock(); + if inner.children.get(&child)?.ppid != parent { + return None; + } + inner.live.get(&child)?.process_group_id.upgrade() + } + + /// Removes a process that has exited from the live set. + fn unregister_process(&self, pid: i32) { + self.inner.lock().live.remove(&pid); + } + + /// Turns `child` into a zombie carrying `status`, wakes its parent if one is waiting, and + /// posts `SIGCHLD` to that parent. + /// + /// Does nothing for a pid with no recorded parent (the initial process, or a child whose + /// parent already exited and dropped it). + fn record_exit(&self, child: i32, status: ExitStatus, cpu_time_nanos: u64) { + let mut inner = self.inner.lock(); + let Some(record) = inner.children.get_mut(&child) else { + return; + }; + record.status = Some(status); + record.cpu_time_nanos = cpu_time_nanos; + let parent = record.ppid; + let wakers: Vec<_> = inner + .waiters + .iter() + .filter(|(waiting, _, _)| *waiting == parent) + .map(|(_, _, waker)| waker.clone()) + .collect(); + // Queue the parent's `SIGCHLD` before the wakeups, so that whichever of its threads wakes + // first already finds the signal pending. + // + // Without this, a guest that blocks waiting for `SIGCHLD` -- which is exactly how + // busybox's `ash` implements a blocking `wait`, via `sigsuspend` -- never wakes up. The + // signal is discarded harmlessly by a parent that has no `SIGCHLD` handler; see + // [`Task::has_pending_signals`]. + let parent_inner = inner.live.get(&parent).and_then(|live| { + live.signals.post( + litebox_common_linux::signal::Signal::SIGCHLD, + crate::syscalls::signal::siginfo_child_exited(child, status), + ); + live.process_inner.upgrade() + }); + drop(inner); + for waker in wakers { + waker.wake(); + } + // The wakers above only cover a parent registered in `wait4`/`rt_sigsuspend`. One blocked + // anywhere else interruptible -- `ppoll`, `read`, `nanosleep`, `epoll_pwait` -- has to be + // kicked the way `send_process_signal` kicks it, or it runs its `SIGCHLD` handler (or gets + // its `EINTR`) only once something unrelated wakes it: `sudo` and xterm's close path both + // hang exactly there. Deliverability stays the parent's own call, in + // `check_for_interrupt`; a parent ignoring `SIGCHLD` simply goes back to sleep. + if let Some(parent_inner) = parent_inner { + for thread in parent_inner.lock().threads.values() { + thread.interrupt(); + } + } + } + + fn set_parent_death_signal(&self, child: i32, signal: Option) { + if let Some(record) = self.inner.lock().children.get_mut(&child) { + record.parent_death_signal = signal; + } + } + + /// Drops every record naming `parent` as a parent, delivering each live child's configured + /// parent-death signal first. + /// + /// Real Linux reparents orphans to init, which then reaps them; this shim has no init, and a + /// record nobody can ever wait on is just a leak, so they are discarded instead. + fn signal_and_discard_children_of(&self, parent: i32) { + let targets = { + let mut inner = self.inner.lock(); + let targets: Vec<_> = inner + .children + .iter() + .filter_map(|(&child, record)| { + (record.ppid == parent && record.status.is_none()) + .then_some(record.parent_death_signal.map(|signal| (child, signal))) + .flatten() + }) + .collect(); + inner.children.retain(|_, record| record.ppid != parent); + targets + }; + for (child, signal) in targets { + self.send_process_signal( + child, + signal, + crate::syscalls::signal::siginfo_parent_death(signal), + ); + } + } + + /// Whether `parent` has any child matching `filter`, zombie or not. + fn has_child(&self, parent: i32, filter: WaitFilter) -> bool { + self.inner + .lock() + .children + .iter() + .any(|(&child, r)| r.ppid == parent && filter.matches(child)) + } + + /// Reaps one zombie child of `parent` matching `filter`, removing it from the table. + fn reap(&self, parent: i32, filter: WaitFilter) -> Option<(i32, ExitStatus, u64)> { + let mut inner = self.inner.lock(); + let (child, status, cpu_time_nanos) = inner.children.iter().find_map(|(&child, r)| { + (r.ppid == parent && filter.matches(child)) + .then_some(r.status) + .flatten() + .map(|status| (child, status, r.cpu_time_nanos)) + })?; + inner.children.remove(&child); + Some((child, status, cpu_time_nanos)) + } + + /// Whether [`Self::reap`] would find something right now, without consuming it. + fn reap_ready(&self, parent: i32, filter: WaitFilter) -> bool { + self.inner + .lock() + .children + .iter() + .any(|(&child, r)| r.ppid == parent && filter.matches(child) && r.status.is_some()) + } + + /// Removes every process-table trace of a child whose host thread could not be spawned. + fn forget_failed_spawn(&self, child: i32) { + let mut inner = self.inner.lock(); + inner.children.remove(&child); + inner.live.remove(&child); + } + + pub(crate) fn register_waiter( + &self, + parent: i32, + waker: litebox::event::wait::Waker, + ) -> u64 { + let mut inner = self.inner.lock(); + let token = inner.next_waiter_token; + inner.next_waiter_token += 1; + inner.waiters.push((parent, token, waker)); + token + } + + pub(crate) fn unregister_waiter(&self, token: u64) { + self.inner.lock().waiters.retain(|(_, t, _)| *t != token); + } +} + +/// The guest stack pointer recorded in a saved register context. +pub(crate) fn guest_stack_pointer(ctx: &litebox_common_linux::PtRegs) -> usize { + #[cfg(target_arch = "x86_64")] + { + ctx.rsp + } + #[cfg(target_arch = "aarch64")] + { + ctx.sp + } +} + +/// Packs an exit status into the `int` layout `wait4`'s `wstatus` uses, as decoded by libc's +/// `WIFEXITED`/`WEXITSTATUS`/`WTERMSIG` macros: a normal exit puts the code in bits 8..16 and +/// leaves the low seven bits (the terminating signal) zero, while a signal death puts the signal +/// number in those low bits. +fn encode_wait_status(status: ExitStatus) -> i32 { + match status { + ExitStatus::Exit(code) => (i32::from(code) & 0xff) << 8, + ExitStatus::Signal(signal) => signal.as_i32() & 0x7f, + } +} + +/// Which children a `wait4` call is willing to reap. +#[derive(Clone, Copy)] +enum WaitFilter { + /// `pid < -1` and `pid == 0` are process-group filters on Linux. LiteBox now tracks + /// per-process groups, but `wait4` group filtering is not implemented yet, so both remain the + /// same conservative "any child" approximation as `pid == -1`. + Any, + Pid(i32), +} + +impl WaitFilter { + fn matches(self, pid: i32) -> bool { + match self { + WaitFilter::Any => true, + WaitFilter::Pid(p) => p == pid, + } + } +} + +pub(crate) struct Alarm { + /// Handle for the alarm timer. + pub(crate) handle: Option<::TimerHandle>, + /// The deadline for the alarm. + pub(crate) deadline: Option<::Instant>, +} + +impl Alarm { + /// Returns the time remaining until [`Self::deadline`], or zero if the + /// alarm is not armed or its deadline has already passed. + pub(crate) fn remaining( + &self, + now: ::Instant, + ) -> Duration { + self.deadline + .as_ref() + .and_then(|d| d.checked_duration_since(&now)) + .unwrap_or(Duration::ZERO) + } +} + +/// The locked portion of the process state. +struct ProcessInner { + /// If true, the whole process is exiting. + group_exit: bool, + /// If true, one thread is waiting for other threads to exit. + is_killing_other_threads: bool, + /// The exit code of the last exited thread in the process. Not updated once + /// `group_exit` is set. + exit_status: ExitStatus, + /// The thread list for the process, mapped by thread ID. + threads: BTreeMap>>, + /// See [`ProcIdentity`]. + identity: ProcIdentity, +} + +/// [`ProcessInner`] as `/proc/` describes it: the published identity plus every live +/// thread's id and command name, ascending by tid. +fn proc_task_info( + pid: i32, + inner: &ProcessInner, +) -> litebox::fs::proc::ProcTaskInfo { + let identity = &inner.identity; + litebox::fs::proc::ProcTaskInfo { + pid, + ppid: identity.ppid, + uid: identity.uid, + gid: identity.gid, + comm: identity.comm.clone(), + cmdline: identity.cmdline.clone(), + exe: identity.exe.clone(), + threads: inner + .threads + .iter() + .map(|(&tid, remote)| litebox::fs::proc::ProcThreadInfo { + tid, + comm: remote.comm(), + }) + .collect(), + } +} + +#[derive(Clone, Copy, Debug)] +pub(crate) enum ExitStatus { + Exit(i8), + Signal(litebox_common_linux::signal::Signal), +} + +impl Process { + /// Creates a new process with the given initial thread. + fn new( + pid: i32, + process_group_id: i32, + remote: Arc>, + futex_manager: Arc>, + futex_namespace: usize, + vfork_completion: Option>>, + shared_vm_parent: Option<&Process>, + launch: Option>>, + ) -> Self { + let nr_threads = ::RawMutex::INIT; + nr_threads.underlying_atomic().store(1, Ordering::Relaxed); + let fork_gate = ::RawMutex::INIT; + fork_gate.underlying_atomic().store(0, Ordering::Relaxed); + let shares_parent_vm = shared_vm_parent.is_some(); + let vm = Arc::new(match shared_vm_parent { + Some(parent) => VmBookkeepingSlot::shared_with(&parent.vm), + None => VmBookkeepingSlot::new(futex_namespace), + }); + Self { + nr_threads, + fork_gate, + inner: Arc::new(Mutex::new(ProcessInner { + exit_status: ExitStatus::Exit(0), + group_exit: false, + is_killing_other_threads: false, + threads: BTreeMap::from_iter([(pid, remote)]), + identity: ProcIdentity::default(), + })), + vm: vm.clone(), + futex_manager, + vfork_completion: Mutex::new(vfork_completion), + shares_parent_vm: AtomicBool::new(shares_parent_vm), + launch, + limits: Arc::new(ResourceLimits::default()), + alarm_timer: Mutex::new(Alarm { + handle: None, + deadline: None, + }), + brk: SharedVmAtomicUsize::new(vm.clone(), |vm| &vm.brk), + owned_ranges: SharedVmLockedField::new(vm.clone(), |vm| &vm.owned_ranges), + elf_patch_cache: SharedVmLockedField::new(vm, |vm| &vm.elf_patch_cache), + cpu_time_nanos: core::sync::atomic::AtomicU64::new(0), + session_id: AtomicI32::new(pid), + process_group_id: Arc::new(AtomicI32::new(process_group_id)), + controlling_pty: AtomicU32::new(NO_CONTROLLING_PTY), + dumpable: AtomicBool::new(true), + address_space: Mutex::new(None), + } + } + + /// `PR_GET_DUMPABLE`. + pub(crate) fn dumpable(&self) -> bool { + self.dumpable.load(Ordering::Relaxed) + } + + /// `PR_SET_DUMPABLE`, and the `execve`/`fork` resets described on the field. + pub(crate) fn set_dumpable(&self, dumpable: bool) { + self.dumpable.store(dumpable, Ordering::Relaxed); + } + + /// `/proc/` view of this process: see [`ProcIdentity`]. + pub(crate) fn proc_task_info(&self, pid: i32) -> litebox::fs::proc::ProcTaskInfo { + proc_task_info(pid, &self.inner.lock()) + } + + /// Publish the credential half of [`ProcIdentity`]. + fn set_proc_credentials(&self, ppid: i32, uid: u32, gid: u32) { + let mut inner = self.inner.lock(); + inner.identity.ppid = ppid; + inner.identity.uid = uid; + inner.identity.gid = gid; + } + + /// Publish the thread-group leader's command name (`/proc//comm`). + fn set_proc_comm(&self, comm: &[u8]) { + let end = comm.iter().position(|&b| b == 0).unwrap_or(comm.len()); + self.inner.lock().identity.comm = comm[..end].to_vec(); + } + + /// Publish the image half of [`ProcIdentity`] after a successful `execve`. + fn set_proc_image(&self, cmdline: Vec, exe: Option) { + let mut inner = self.inner.lock(); + inner.identity.cmdline = cmdline; + inner.identity.exe = exe; + } + + /// A `fork` child starts out describing the same image as its parent (a new pid, and a + /// parent of its own, but the same `argv`/`exe`/`comm` until it `exec`s). + fn inherit_proc_identity(&self, parent: &Process, ppid: i32) { + let mut identity = parent.inner.lock().identity.clone(); + identity.ppid = ppid; + let mut inner = self.inner.lock(); + inner.identity = identity; + self.dumpable.store(parent.dumpable(), Ordering::Relaxed); + drop(inner); + } + + fn futex_manager(&self) -> &FutexManager { + self.futex_manager.as_ref() + } + + fn futex_namespace(&self) -> usize { + self.vm.current().futex_namespace + } + + fn detach_vfork_vm(&self) { + let futex_namespace = self.futex_manager.new_private_namespace(); + self.vm.detach(futex_namespace); + self.shares_parent_vm.store(false, Ordering::Release); + } + + fn shares_parent_vm(&self) -> bool { + self.shares_parent_vm.load(Ordering::Acquire) + } + + fn complete_vfork(&self) { + if let Some(completion) = self.vfork_completion.lock().take() { + completion.complete(); + } + } + + /// Whether the parent this vfork child shares its memory with is itself a + /// [`SharedAddressSpace`] member. `false` once the vfork has completed. + fn vfork_parent_shares_address_space(&self) -> bool { + self.vfork_completion + .lock() + .as_ref() + .is_some_and(|completion| completion.parent_shares_address_space) + } + + fn await_launch(&self) -> bool { + self.launch.as_ref().is_none_or(|launch| launch.wait()) + } + + fn exit_is_publishable(&self) -> bool { + self.launch + .as_ref() + .is_none_or(|launch| launch.is_committed()) + } + + /// Returns this process's process-group ID. + pub(crate) fn process_group_id(&self) -> i32 { + self.process_group_id.load(Ordering::Acquire) + } + + /// Returns this process's controlling Unix98 PTY number, if one is assigned. + pub(crate) fn controlling_pty(&self) -> Option { + let number = self.controlling_pty.load(Ordering::Acquire); + (number != NO_CONTROLLING_PTY).then_some(number) + } + + /// Assigns `number` as the controlling terminal when the caller is a session leader. + /// + /// Returns `true` when this call makes the assignment and `false` when the same terminal was + /// already assigned. Stealing a different terminal is rejected. + pub(crate) fn acquire_controlling_pty(&self, pid: i32, number: u32) -> Result { + if self.session_id.load(Ordering::Acquire) != pid { + return Err(Errno::EPERM); + } + match self.controlling_pty.compare_exchange( + NO_CONTROLLING_PTY, + number, + Ordering::AcqRel, + Ordering::Acquire, + ) { + Ok(_) => Ok(true), + Err(current) if current == number => Ok(false), + Err(_) => Err(Errno::EPERM), + } + } + + /// Returns the current number of threads in this process. + pub fn nr_threads(&self) -> u32 { + self.nr_threads.underlying_atomic().load(Ordering::Relaxed) + } + + /// Returns the remote handle for thread `tid` of this process, if it is currently attached + /// (i.e. running or blocked, not yet exited). Used by `tkill`/`tgkill` to deliver a + /// specifically-targeted signal without reaching into another thread's non-`Send` local + /// state -- see [`ThreadRemote::remote_pending`]. + pub(crate) fn thread_remote(&self, tid: i32) -> Option>> { + self.inner.lock().threads.get(&tid).cloned() + } + + /// Parks the calling thread while this process's fork gate is closed. + /// + /// The fast path -- gate open, the only case any thread sees outside a + /// concurrent multithreaded `fork` -- is a single atomic load. A parked + /// thread blocks on the raw gate word (never an interruptible wait: this + /// is called from `CheckForInterrupt::check_for_interrupt`, whose + /// contract forbids interruptible waiting) and resumes when + /// [`ForkGateGuard`] reopens the gate. An exiting thread never parks -- + /// it proceeds to detach, and `detach_thread` wakes the gate so the + /// forker re-evaluates how many siblings it is still waiting for. + fn park_while_fork_gate_closed(&self, is_exiting: bool) { + let word = self.fork_gate.underlying_atomic(); + if word.load(Ordering::Acquire) & FORK_GATE_CLOSED == 0 { + return; + } + if is_exiting { + return; + } + word.fetch_add(1, Ordering::AcqRel); + // The forker blocks on this same word until enough siblings are + // parked; every increment must wake it to re-count. + self.fork_gate.wake_all(); + loop { + let cur = word.load(Ordering::Acquire); + if cur & FORK_GATE_CLOSED == 0 { + break; + } + let _ = self.fork_gate.block(cur); + } + word.fetch_sub(1, Ordering::AcqRel); + } + + /// Waits for all threads in this process to exit, returning the exit code. + pub fn wait_for_exit(&self) -> ExitStatus { + loop { + let n = self.nr_threads.underlying_atomic().load(Ordering::Acquire); + if n == 0 { + break; + } + let _ = self.nr_threads.block(n); + } + self.inner.lock().exit_status + } + + /// Attaches a new thread to this process, returning a new remote state for + /// the thread. + fn attach_thread(&self, tid: i32) -> Option>> { + // Allocate outside the lock. + let remote = Arc::new(ThreadRemote::new()); + let mut inner = self.inner.lock(); + if inner.group_exit || inner.is_killing_other_threads { + return None; + } + let old_thread = inner.threads.insert(tid, remote.clone()); + assert!(old_thread.is_none(), "thread ID {tid} already exists"); + let nr_threads = self.nr_threads.underlying_atomic(); + nr_threads.store(nr_threads.load(Ordering::Relaxed) + 1, Ordering::Release); + Some(remote) + } + + /// Detaches a thread from this process. + /// + /// Returns `true` if this was the last thread in the process (i.e., the process as a whole + /// is now exiting), `false` if other threads remain. + /// + /// # Panics + /// Panics if the thread ID does not exist in this process. + fn detach_thread(&self, tid: i32) -> bool { + let data; + let (notify, is_last_thread) = { + let mut inner = self.inner.lock(); + data = inner.threads.remove(&tid); + assert!(data.is_some()); + + let nr_threads = self.nr_threads.underlying_atomic(); + let n = nr_threads.load(Ordering::Relaxed); + let new_count = n.checked_sub(1).expect("decrementing from zero threads"); + nr_threads.store(new_count, Ordering::Release); + let is_last_thread = new_count == 0; + if is_last_thread { + assert!(inner.threads.is_empty()); + // The last thread exited. Prevent new threads. + inner.group_exit = true; + } + + // Notify waiters if this is the last thread of the process + // (`wait_for_exit`) or if this is the last thread being killed + // during an exec (`kill_other_threads`). + ( + is_last_thread || (new_count == 1 && inner.is_killing_other_threads), + is_last_thread, + ) + }; + if notify { + self.nr_threads.wake_all(); + } + // A forker blocked in `park_sibling_threads_for_fork` counts parked + // siblings against `nr_threads`; an exiting sibling shrinks the + // latter without ever parking, so the forker must recount. + if self.fork_gate.underlying_atomic().load(Ordering::Acquire) & FORK_GATE_CLOSED != 0 { + self.fork_gate.wake_all(); + } + // Release any attached tracer rather than leaving it blocked forever on a rendezvous + // that can now never happen: this thread is gone, so a stop it may still owe its tracer + // will never arrive. A tracer's next `ptrace` request against this `tid` finds no + // `ThreadRemote` at all (already removed from `threads` above) and gets `ESRCH`, exactly + // like `tkill` against an exited thread -- PID/TID-reuse-safe by the same construction. + #[cfg(target_arch = "aarch64")] + if let Some(remote) = &data { + remote.ptrace.on_thread_exit(); + } + is_last_thread + } +} + +impl Task { + /// Updates the process exit status for a thread exit. + fn exit_thread(&self, code: i8) { + let mut inner = self.thread.process.inner.lock(); + if self.is_exiting() { + return; + } + inner.exit_status = ExitStatus::Exit(code); + self.thread.remote.is_exiting.store(true, Ordering::Relaxed); + } + + /// Updates the process exit status for a group exit and signals all threads + /// to exit. + pub(crate) fn exit_group(&self, status: ExitStatus) { + let mut inner = self.thread.process.inner.lock(); + if self.is_exiting() { + return; + } + assert!(!inner.group_exit); + inner.exit_status = status; + inner.group_exit = true; + for thread in inner.threads.values() { + thread.is_exiting.store(true, Ordering::Relaxed); + thread.interrupt(); + } + } + + /// Closes the process's fork gate and waits until every sibling thread is + /// parked at it, so the caller can take the address-space turn (see + /// [`SharedAddressSpace`]) with the same guarantees a single-threaded + /// process has: nothing else will touch guest memory until the returned + /// guard reopens the gate. + /// + /// Modeled on [`Self::kill_other_threads`]'s stop-the-world shape: + /// interrupt every sibling ([`ThreadRemote::interrupt`] wakes a thread + /// blocked in any interruptible shim wait, and yanks one executing guest + /// code back into the shim via the platform interrupt), then block on the + /// gate word until the parked count accounts for every sibling. A sibling + /// mid-syscall parks at the next + /// `CheckForInterrupt::check_for_interrupt` or + /// [`Task::prepare_to_run_guest`] point -- in particular, one mid-copy + /// into guest memory finishes that copy *before* parking, so the snapshot + /// taken after this returns cannot lose an in-flight write. Siblings that + /// exit instead of parking are handled by `detach_thread` waking the gate + /// so the count converges either way. + /// [`Process::park_while_fork_gate_closed`] for this task; see + /// `Process::fork_gate`. Called from the two guest-memory choke points in + /// `crate::wait`. + pub(crate) fn park_while_fork_gate_closed(&self) { + self.process() + .park_while_fork_gate_closed(self.is_exiting()); + } + + fn park_sibling_threads_for_fork(&self) -> ForkGateGuard<'_, Platform> { + let process = self.process(); + let word = process.fork_gate.underlying_atomic(); + // If another thread is mid-fork, park like any other sibling until + // its gate reopens, then take our own turn. + loop { + let prev = word.fetch_or(FORK_GATE_CLOSED, Ordering::AcqRel); + if prev & FORK_GATE_CLOSED == 0 { + break; + } + process.park_while_fork_gate_closed(self.is_exiting()); + } + let guard = ForkGateGuard { process }; + { + let inner = process.inner.lock(); + for (&tid, thread) in &inner.threads { + if tid != self.tid { + thread.interrupt(); + } + } + } + loop { + let cur = word.load(Ordering::Acquire); + let parked = cur & !FORK_GATE_CLOSED; + let others = process.nr_threads().saturating_sub(1); + if parked >= others { + break; + } + let _ = process.fork_gate.block(cur); + } + guard + } + + /// Kills all other threads in the process, waiting for them to exit. + /// + /// Returns false if this thread is already exiting. + #[must_use] + fn kill_other_threads(&self) -> bool { + { + let mut inner = self.thread.process.inner.lock(); + if self.is_exiting() { + return false; + } + for (&tid, thread) in &inner.threads { + if tid == self.tid { + continue; + } + thread.is_exiting.store(true, Ordering::Relaxed); + thread.interrupt(); + } + assert!(!inner.is_killing_other_threads); + inner.is_killing_other_threads = true; + } + // Wait for other threads to exit. + loop { + let n = self + .thread + .process + .nr_threads + .underlying_atomic() + .load(Ordering::Acquire); + if n == 1 { + break; + } + let _ = self.thread.process.nr_threads.block(n); + } + self.thread.process.inner.lock().is_killing_other_threads = false; + true + } + + /// Returns true if the task is exiting and should not continue running + /// guest code. + pub fn is_exiting(&self) -> bool { + self.thread.remote.is_exiting.load(Ordering::Relaxed) + } +} + +#[derive(Default)] +enum ThreadInitState { + #[default] + None, + NewProcess(crate::loader::elf::ElfLoadInfo), + NewThread { + stack: Option, + tls: Option, + set_child_tid: Option>, + /// The parent's FPSIMD register file at the moment of `clone`/`fork`, + /// captured on the parent thread (where [`ThreadProvider::get_fp_state`] + /// reads the *calling* thread's state) since the new host OS thread's + /// own per-thread FP shadow otherwise starts zeroed -- Linux's + /// `copy_thread` copies the parent's FPSIMD state into the child task, + /// so a cloned/forked guest thread must observe the same vector/FPCR/ + /// FPSR values the parent had at the syscall, not a cleared file (that + /// reset is correct only for `execve`, via `NewProcess`). + /// + /// [`ThreadProvider::get_fp_state`]: litebox::platform::ThreadProvider::get_fp_state + #[cfg(target_arch = "aarch64")] + fp: litebox::platform::FpSimdState64, + }, +} + +#[derive(Clone, Default)] +struct SupplementaryGroups(Box<[u32]>); + +impl SupplementaryGroups { + const MAX: usize = 65_536; + + fn from_user(size: usize, list: UserPtr) -> Result { + if size > Self::MAX { + return Err(Errno::EINVAL); + } + let mut groups = if size == 0 { + Box::default() + } else { + list.to_owned_slice::(size).ok_or(Errno::EFAULT)? + }; + groups.sort_unstable(); + Ok(Self(groups)) + } + + fn as_slice(&self) -> &[u32] { + &self.0 + } +} + +/// Credentials of a process +#[derive(Clone)] +pub(crate) struct Credentials { + pub uid: u32, + pub euid: u32, + pub suid: u32, + pub gid: u32, + pub egid: u32, + pub sgid: u32, + supplementary_groups: SupplementaryGroups, + no_new_privs: bool, + /// `PR_SET_KEEPCAPS` state. LiteBox does not model capabilities (see + /// `PrctlArg::SetKeepCaps`'s doc comment), so this is stored only so + /// `PR_GET_KEEPCAPS` reads back whatever was last set -- there is no + /// actual capability set for it to gate. + keep_caps: bool, +} + +impl Credentials { + #[expect( + clippy::similar_names, + reason = "uid, euid, gid, and egid are the POSIX credential names" + )] + pub(crate) fn new(uid: u32, euid: u32, gid: u32, egid: u32) -> Self { + Self { + uid, + euid, + suid: euid, + gid, + egid, + sgid: egid, + supplementary_groups: SupplementaryGroups::default(), + no_new_privs: false, + keep_caps: false, + } + } + + pub(crate) fn supplementary_groups(&self) -> &[u32] { + self.supplementary_groups.as_slice() + } + + pub(crate) fn no_new_privs(&self) -> bool { + self.no_new_privs + } + + pub(crate) fn keep_caps(&self) -> bool { + self.keep_caps + } +} + +impl Task { + pub(crate) fn process(&self) -> &Arc> { + &self.thread.process + } + + /// This task's own remote handle -- the piece of its signal/wait state a `tkill`/`tgkill` + /// from another thread can safely touch. See [`ThreadRemote::remote_pending`]. + pub(crate) fn thread_remote(&self) -> &Arc> { + &self.thread.remote + } + + /// Set the current task's command name. + pub(crate) fn set_task_comm(&self, comm: &[u8]) { + let mut new_comm = [0u8; litebox_common_linux::TASK_COMM_LEN]; + let comm = &comm[..comm.len().min(litebox_common_linux::TASK_COMM_LEN - 1)]; + new_comm[..comm.len()].copy_from_slice(comm); + self.comm.set(new_comm); + + // Publish to `/proc//task//comm` (every thread) and `/proc//comm` (the + // leader), alongside the credentials `/proc//status` reports -- see `ProcIdentity`. + self.thread.remote.set_comm(&new_comm); + if self.tid == self.pid { + self.process().set_proc_comm(&new_comm); + } + self.publish_proc_credentials(); + } + + /// Refresh the credential half of this process's [`ProcIdentity`] from this task's live + /// credentials. + pub(crate) fn publish_proc_credentials(&self) { + let credentials = self.credentials.borrow(); + self.process() + .set_proc_credentials(self.ppid, credentials.uid, credentials.gid); + } + + /// This task's own `/proc/` view, refreshed from its live credentials first; what the + /// shim publishes as `/proc/self` ahead of each lookup (see `syscalls::file`'s + /// `publish_proc_view`). + pub(crate) fn proc_task_info(&self) -> litebox::fs::proc::ProcTaskInfo { + self.publish_proc_credentials(); + self.process().proc_task_info(self.pid) + } + + /// Handle syscall `seccomp` (and `prctl(PR_SET_SECCOMP)`, which decodes to it). + /// + /// The shim has no BPF filter engine, so this answers as a kernel built without + /// `CONFIG_SECCOMP`: `ENOSYS` for both mode-setting operations, never a fake success. A `0` + /// here would make a sandboxed program believe its filter is enforced when nothing is -- + /// Chromium's renderer would then run with a "layer-2 sandbox" that filters nothing. + /// Chromium's probes (`sandbox/linux/seccomp-bpf/sandbox_bpf.cc`, + /// `KernelSupportsSeccompBPF`/`KernelSupportsSeccompFlags`) take only `EFAULT` as "supported" + /// and `DCHECK` that anything else is `ENOSYS` or `EINVAL`, so `seccomp_bpf_supported_` + /// stays false, `StartSeccompBPF` returns without a promise and `CheckForBrokenPromises` + /// has nothing to `CHECK`; the setuid/namespace layer-1 sandbox is unaffected. The two + /// query operations answer as a kernel without `CONFIG_SECCOMP_FILTER` does + /// (`EOPNOTSUPP`); an unknown operation is `EINVAL`, as in Linux's `do_seccomp`. + pub(crate) fn sys_seccomp( + &self, + operation: u32, + flags: u32, + args: UserPtr, + ) -> Result { + const SECCOMP_SET_MODE_STRICT: u32 = 0; + const SECCOMP_SET_MODE_FILTER: u32 = 1; + const SECCOMP_GET_ACTION_AVAIL: u32 = 2; + const SECCOMP_GET_NOTIF_SIZES: u32 = 3; + // One line per (operation, flags), not per call: the per-call `args` pointer is in + // the trace-level `syscall req=` record. + let _ = args; + match operation { + SECCOMP_SET_MODE_STRICT | SECCOMP_SET_MODE_FILTER => { + log_unsupported!( + "seccomp(operation = {operation}, flags = {flags:#x}): no BPF filtering -> ENOSYS" + ); + Err(Errno::ENOSYS) + } + SECCOMP_GET_ACTION_AVAIL | SECCOMP_GET_NOTIF_SIZES => { + log_unsupported!("seccomp(operation = {operation}) -> EOPNOTSUPP"); + Err(Errno::EOPNOTSUPP) + } + _ => Err(Errno::EINVAL), + } + } + + /// Handle syscall `prctl`. + pub(crate) fn sys_prctl(&self, arg: PrctlArg) -> Result { + match arg { + PrctlArg::SetPDeathSig(signal) => { + self.thread.parent_death_signal.set(signal); + self.global + .processes + .set_parent_death_signal(self.pid, signal); + Ok(0) + } + PrctlArg::GetPDeathSig(signal_ptr) => { + let signal = self.thread.parent_death_signal.get(); + signal_ptr + .write_at_offset::(0, signal.map_or(0, |signal| signal.as_i32())) + .ok_or(Errno::EFAULT) + .map(|()| 0) + } + PrctlArg::GetName(name) => name + .write_slice_at_offset::(0, &self.comm.get()) + .ok_or(Errno::EFAULT) + .map(|()| 0), + PrctlArg::SetName(name) => { + let mut name_buf = [0u8; litebox_common_linux::TASK_COMM_LEN - 1]; + // strncpy + for (i, byte) in name_buf.iter_mut().enumerate() { + let b = name + .read_at_offset::(isize::try_from(i).unwrap()) + .ok_or(Errno::EFAULT)?; + if b == 0 { + break; + } + *byte = b; + } + self.set_task_comm(&name_buf); + Ok(0) + } + PrctlArg::CapBSetRead(cap) => { + // Return 1 if the capability specified in cap is in the calling + // thread's capability bounding set, or 0 if it is not. + if cap + > litebox_common_linux::CapSet::LAST_CAP + .bits() + .trailing_zeros() as usize + { + return Err(Errno::EINVAL); + } + // Note we don't support capabilities in LiteBox, so we always return 0. + Ok(0) + } + PrctlArg::GetDumpable => Ok(usize::from(self.process().dumpable())), + // Only `SUID_DUMP_DISABLE` (0) and `SUID_DUMP_USER` (1) may be set; Linux refuses + // `SUID_DUMP_ROOT` (2) and anything else with `EINVAL`. + PrctlArg::SetDumpable(value) => match value { + 0 | 1 => { + self.process().set_dumpable(value == 1); + Ok(0) + } + _ => Err(Errno::EINVAL), + }, + PrctlArg::SetNoNewPrivs => { + let mut credentials = self.credentials.borrow().as_ref().clone(); + credentials.no_new_privs = true; + *self.credentials.borrow_mut() = Arc::new(credentials); + Ok(0) + } + PrctlArg::GetNoNewPrivs => Ok(usize::from( + self.credentials.borrow().no_new_privs(), + )), + PrctlArg::SetKeepCaps(keep) => { + let mut credentials = self.credentials.borrow().as_ref().clone(); + credentials.keep_caps = keep; + *self.credentials.borrow_mut() = Arc::new(credentials); + Ok(0) + } + PrctlArg::GetKeepCaps => Ok(usize::from(self.credentials.borrow().keep_caps())), + // `PrctlArg` is `#[non_exhaustive]`; the syscall decoder rejects every option not + // represented above with `EINVAL` before constructing one. + _ => unreachable!(), + } + } + + /// Handle syscall `arch_prctl`. + pub(crate) fn sys_arch_prctl(&self, arg: ArchPrctlArg) -> Result<(), Errno> { + match arg { + #[cfg(target_arch = "x86_64")] + ArchPrctlArg::SetFs(addr) => self + .global + .platform + .set_arch_specific_register(&ArchSpecificRegister::FsBase, addr) + .map_err(Errno::from), + #[cfg(target_arch = "x86_64")] + ArchPrctlArg::GetFs(addr) => { + let fsbase = self + .global + .platform + .get_arch_specific_register(&ArchSpecificRegister::FsBase)?; + addr.write_at_offset::(0, fsbase) + .ok_or(Errno::EFAULT)?; + Ok(()) + } + ArchPrctlArg::CETStatus | ArchPrctlArg::CETDisable | ArchPrctlArg::CETLock => { + Err(Errno::EINVAL) + } + // `ArchPrctlArg` is `#[non_exhaustive]`, but on every target it declares (`SetFs`/ + // `GetFs` exist only under x86_64) every variant is matched above, and the syscall + // decoder itself only runs `#[cfg(target_arch = "x86_64")]`, so on other targets this + // is never even reachable via a real `arch_prctl` syscall. + _ => unreachable!(), + } + } +} + +const ROBUST_LIST_LIMIT: isize = 2048; + +/// Bit set in a robust futex word's low bits by the kernel (here, the shim) when the thread +/// that held the lock dies without releasing it, so the next owner can detect the previous +/// holder died mid-critical-section. Matches Linux's `FUTEX_OWNER_DIED`. +const FUTEX_OWNER_DIED: u32 = 0x4000_0000; +/// Bit set in a robust futex word's low bits when at least one thread is (or might be) sleeping +/// in `FUTEX_WAIT` on it, so the unlocker knows to `FUTEX_WAKE`. Matches Linux's `FUTEX_WAITERS`. +const FUTEX_WAITERS: u32 = 0x8000_0000; +/// Mask isolating the TID stored in a robust futex word's low bits. Matches Linux's +/// `FUTEX_TID_MASK`. +const FUTEX_TID_MASK: u32 = 0x3fff_ffff; + +impl Task { + /// Processes a single robust-futex-list entry belonging to a dying thread: if the futex word + /// still records this thread as the owner, marks it dead (setting [`FUTEX_OWNER_DIED`] and + /// clearing the TID) and, if a waiter may be present, wakes one -- mirroring Linux's + /// `handle_futex_death` (`kernel/futex/core.c`). Without this, a thread that dies while + /// still holding a robust `pthread_mutex_t` would leave every future waiter on that lock + /// blocked forever, since its owner can never call `FUTEX_WAKE` again. + fn handle_futex_death(&self, futex_addr: UserPtr, pending_op: bool) -> Result<(), Errno> { + if !futex_addr.as_usize().is_multiple_of(4) { + return Err(Errno::EINVAL); + } + let futex_addr = UserPtrMut::from_usize(futex_addr.as_usize()); + + let Some(mut word) = futex_addr.read_at_offset::(0) else { + return Err(Errno::EFAULT); + }; + + loop { + // Only touch the word if it's still (nominally) owned by this dying thread -- a lock + // that was already unlocked and re-acquired by someone else, or never actually locked + // by us despite being linked into our robust list, must be left alone. + #[expect( + clippy::cast_sign_loss, + reason = "tid is always non-negative; only ever compared against another tid read \ + back from a futex word, never used arithmetically" + )] + if (word & FUTEX_TID_MASK) != self.tid as u32 { + return Ok(()); + } + + let had_waiters = word & FUTEX_WAITERS != 0; + let new_word = (word & FUTEX_WAITERS) | FUTEX_OWNER_DIED; + match futex_addr.compare_exchange::(word, new_word) { + None => return Err(Errno::EFAULT), + Some(Err(actual)) => { + word = actual; + continue; + } + Some(Ok(_)) => { + if had_waiters || pending_op { + let _ = self.sys_futex(FutexArgs::Wake { + addr: futex_addr, + flags: litebox_common_linux::FutexFlags::empty(), + count: 1, + }); + } + return Ok(()); + } + } + } + } +} + +fn fetch_robust_entry( + head: UserPtr, +) -> (UserPtr, bool) { + let next = head.as_usize(); + (UserPtr::from_usize(next & !1), next & 1 != 0) +} + +impl Task { + fn wake_robust_list( + &self, + head: UserPtr, + ) -> Result<(), Errno> { + let mut limit = ROBUST_LIST_LIMIT; + let head_ptr = head.as_usize(); + let head = head.read_at_offset::(0).ok_or(Errno::EFAULT)?; + let (mut entry, _pi) = fetch_robust_entry(UserPtr::from_usize(head.list.next)); + let (pending, _ppi) = fetch_robust_entry(UserPtr::from_usize(head.list_op_pending)); + let futex_offset = head.futex_offset; + let entry_head = head_ptr + offset_of!(litebox_common_linux::RobustListHead, list); + while entry.as_usize() != entry_head && limit > 0 { + let nxt = entry + .read_at_offset::(0) + .map(|e| fetch_robust_entry(UserPtr::from_usize(e.next))); + if entry.as_usize() != pending.as_usize() { + self.handle_futex_death( + UserPtr::from_usize(entry.as_usize().wrapping_add_signed(futex_offset)), + false, + )?; + } + let Some((next_entry, _next_pi)) = nxt else { + return Err(Errno::EFAULT); + }; + + entry = next_entry; + limit -= 1; + } + + if pending.as_usize() != 0 { + let _ = self.handle_futex_death( + UserPtr::from_usize(pending.as_usize().wrapping_add_signed(futex_offset)), + true, + ); + } + Ok(()) + } +} + +impl Task { + /// Called when the task is exiting. + pub(crate) fn prepare_for_exit(&mut self) { + // `CLOCK_THREAD_CPUTIME_ID` only ever reads the calling thread's own clock, so this has + // to happen here, on the exiting thread itself, rather than later from whichever thread + // ends up reaping it. Accumulated into the process (rather than overwritten) so that a + // multithreaded process's rusage reflects every thread that has exited so far, not just + // the last one. + self.thread.process.cpu_time_nanos.fetch_add( + self.global + .platform + .thread_cpu_time() + .as_nanos() + .try_into() + .unwrap_or(u64::MAX), + Ordering::Relaxed, + ); + + let exit_is_publishable = self.process().exit_is_publishable(); + let clear_child_tid_on_exit = + self.process().nr_threads() > 1 || self.process().shares_parent_vm(); + let can_touch_guest_memory = self + .membership() + .is_none_or(|membership| membership.holding()); + if exit_is_publishable && can_touch_guest_memory { + if let Some(clear_child_tid) = self.thread.clear_child_tid.take() + && clear_child_tid_on_exit + { + let _ = clear_child_tid.write_at_offset::(0, 0); + // Cast from *i32 to *u32. + let clear_child_tid = UserPtrMut::from_usize(clear_child_tid.as_usize()); + let _ = self.sys_futex(litebox_common_linux::FutexArgs::Wake { + addr: clear_child_tid, + flags: litebox_common_linux::FutexFlags::empty(), + count: 1, + }); + } + if let Some(robust_list) = self.thread.robust_list.take() { + let _ = self.wake_robust_list(robust_list); + } + } else { + // An aborted `ProcessLaunch` never entered guest userspace, and a parked copying-fork + // task does not currently own the bytes at its guest addresses. Neither may clear a + // child-TID word or walk a robust list in whichever process image is actually live. + self.thread.clear_child_tid.take(); + self.thread.robust_list.take(); + } + + // Keep this thread counted until every exit-time access to guest memory is complete. An + // execing sibling waits for `nr_threads == 1` before tearing the old address space down. + let is_last_thread = self.thread.detach_from_process(); + + // Every write to guest memory above is done, so a task that shares its address space can + // now hand it back -- which it must, or the members still alive would wait for it + // forever. An aborted launch never received the fork token in the first place, despite its + // not-yet-running membership being constructed as the future holder, so it must only drop + // that membership rather than release the actual parent's token. See [`SharedAddressSpace`]. + // The ranges this process shared with its family, captured before the membership is + // given up below: what is *not* in them is this process's alone to release. + let shared_ranges = self + .membership() + .map(|membership| membership.shared_ranges.lock().clone()); + // Membership is the process's, so only its last thread settles it; a sibling still + // running needs the token exactly as before. + let alone_in_address_space = if !is_last_thread { + false + } else if !exit_is_publishable { + self.process().address_space.lock().take(); + false + } else if self.process().shares_parent_vm() { + // The memory is the vfork parent's, and so is any family this task forked into. + self.hand_address_space_to_vfork_parent(); + false + } else { + self.leave_address_space() + }; + + // `FilesState` is shared (via `Arc`) across every `CLONE_FILES` thread of the process, + // and closing an fd is only ever done explicitly (via `do_close`, which routes through + // `Descriptors::remove` and the resource's own `Drop` impl -- e.g. a pipe write-end's + // `Drop` firing its `HUP` notification). Just letting `FilesState`/`RawDescriptorStorage` + // fall out of scope does NOT do this: `OwnedFd::drop` is a no-op for any fd that was + // never explicitly closed, so any fd still open when the process exits would otherwise + // leak forever at the descriptor-table level -- e.g. hanging a reader elsewhere in the + // process that is blocked in `read()` waiting for a pipe write-end's `EOF`, regardless of + // whether something else (like an epoll registration, see `epoll.rs`) also still + // references the fd. Real Linux closes every fd of a process as part of process exit, so + // mirror that here -- but only once, when the *last* thread sharing this file table is + // the one exiting, matching `CLONE_FILES` semantics (a single thread of a still-running + // multithreaded process exiting must NOT close fds out from under its siblings). + if is_last_thread { + // Descriptors go first: anything at the other end of one of them (an X server seeing + // its client vanish, a pipe reader getting EOF) then learns of the exit before the + // slower memory teardown below, and nothing in a close path needs the mappings -- + // shared-file copy-back holds its own entry handles, not fds, and runs on `munmap`. + self.close_all_fds_on_exit(); + // Linux tears a process's address space down when its last thread exits. Here that + // is only safe when no other guest process can be using the bytes at this process's + // addresses: a fork-family member that is not alone has a parked sibling whose live + // image occupies exactly these ranges (the child took the parent's image over at + // the same addresses), and a vfork-shared child's `owned_ranges` *are* its parent's. + // Both of those keep their memory for the survivor, exactly as `execve` decides via + // `leave_address_space_if_alone`/`detach_vfork_vm`. Everything else -- every + // ordinary exec'd process -- releases what it owns, so short-lived processes stop + // leaking their whole image and stack into the process-blind page manager forever. + if alone_in_address_space && exit_is_publishable && !self.process().shares_parent_vm() { + self.release_owned_memory_on_exit(); + } else if exit_is_publishable + && !self.process().shares_parent_vm() + && let Some(shared_ranges) = shared_ranges + { + // Still a family member's sibling: the shared ranges stay (the others own them + // too), but what this process mapped for itself after the fork is nobody else's + // and would otherwise leak for the family's lifetime -- a zygote spawns many + // short-lived children. + self.release_private_memory_on_exit(&shared_ranges); + } + // The process is gone: become a zombie its parent can `wait4`, and let go of any + // children of our own (nothing can ever reap them now). + let status = self.thread.process.inner.lock().exit_status; + let cpu_time_nanos = self.thread.process.cpu_time_nanos.load(Ordering::Relaxed); + self.global.processes.unregister_process(self.pid); + if exit_is_publishable { + self.global + .processes + .record_exit(self.pid, status, cpu_time_nanos); + self.global + .processes + .signal_and_discard_children_of(self.pid); + } + self.process().complete_vfork(); + } + } + + pub(crate) fn sys_exit(&self, status: i32) { + // The `Task` will be dropped on the way out of the shim, which will + // call `self.prepare_for_exit()`. + self.exit_thread(status.trunc()); + } + + pub(crate) fn sys_exit_group(&self, status: i32) { + // Tear down occurs similarly to `sys_exit`. + self.exit_group(ExitStatus::Exit(status.trunc())); + } +} + +/// A descriptor for thread-local storage (TLS). +/// +/// On both `x86_64` and `aarch64` this is a `*mut u8` pointing at an +/// arbitrarily sized memory region: the value `clone(CLONE_SETTLS)` supplies +/// becomes `FS.base` on x86-64 and `TPIDR_EL0` on aarch64. +type ThreadLocalDescriptor = UserPtrMut; + +/// The architecture register holding the guest's thread pointer. +/// +/// The platform owns the hardware register in both cases and virtualizes the +/// guest's view of it, so the shim always goes through [`ArchSpecificRegister`] +/// rather than touching it directly. +#[cfg(target_arch = "x86_64")] +const GUEST_TLS_REGISTER: ArchSpecificRegister = ArchSpecificRegister::FsBase; +#[cfg(target_arch = "aarch64")] +const GUEST_TLS_REGISTER: ArchSpecificRegister = ArchSpecificRegister::TpidrEl0; + +struct NewThreadArgs { + /// Task struct that maintains all per-thread data + task: Task, +} + +#[derive(Clone, Copy)] +enum ProcessCloneKind { + Fork, + VforkCopy, + VforkShared, +} + +impl ProcessCloneKind { + fn copies_vm(self) -> bool { + matches!(self, Self::Fork | Self::VforkCopy) + } + + fn waits_for_exec_or_exit(self) -> bool { + matches!(self, Self::VforkCopy | Self::VforkShared) + } + + fn shares_parent_vm(self) -> bool { + matches!(self, Self::VforkShared) + } +} + +impl litebox::shim::InitThread for NewThreadArgs { + type ExecutionContext = litebox_common_linux::PtRegs; + + fn init( + self: alloc::boxed::Box, + ) -> alloc::boxed::Box> + { + let Self { task } = *self; + + Box::new(crate::LinuxShimEntrypoints { + task, + _not_send: core::marker::PhantomData, + }) + } +} + +impl Task { + pub(crate) fn sys_clone( + &self, + ctx: &litebox_common_linux::PtRegs, + args: &litebox_common_linux::CloneArgs, + ) -> Result { + self.do_clone(ctx, args, false) + } + + pub(crate) fn sys_clone3( + &self, + ctx: &litebox_common_linux::PtRegs, + args: UserPtr, + ) -> Result { + let args = args.read_at_offset::(0).ok_or(Errno::EFAULT)?; + self.do_clone(ctx, &args, true) + } + + pub(crate) fn sys_unshare(&self, flags: CloneFlags) -> Result { + if flags.is_empty() { + return Ok(0); + } + let namespace_flags = CloneFlags::NEWNS + | CloneFlags::NEWCGROUP + | CloneFlags::NEWUTS + | CloneFlags::NEWIPC + | CloneFlags::NEWUSER + | CloneFlags::NEWPID + | CloneFlags::NEWNET + | CloneFlags::NEWTIME; + if flags.intersects(namespace_flags) { + // LiteBox deliberately exposes no namespace-creation capability. `EPERM` matches a + // kernel where the caller lacks that capability and lets sandbox probes fail closed. + return Err(Errno::EPERM); + } + log_unsupported!("unshare with unsupported flags: {flags:?}"); + Err(Errno::EINVAL) + } + + /// Creates a new thread or process. + /// + /// Note we currently only support creating threads with the VM, FS, and FILES flags set. + fn do_clone( + &self, + ctx: &litebox_common_linux::PtRegs, + args: &litebox_common_linux::CloneArgs, + clone3: bool, + ) -> Result { + const MAX_SIGNAL_NUMBER: u64 = 64; + + let litebox_common_linux::CloneArgs { + mut flags, + pidfd: _, + child_tid, + parent_tid, + exit_signal, + stack, + stack_size, + tls, + set_tid, + set_tid_size, + cgroup, + } = *args; + + // `CLONE_DETACHED` is ignored but has been reserved for reuse with + // `clone3` or in combination with `CLONE_PIDFD`. + if !clone3 && !flags.contains(CloneFlags::PIDFD) { + flags.remove(CloneFlags::DETACHED); + } + + let process_kind_flags = CloneFlags::VM | CloneFlags::VFORK; + let process_tid_flags = + CloneFlags::PARENT_SETTID | CloneFlags::CHILD_SETTID | CloneFlags::CHILD_CLEARTID; + // `CLONE_FS` on a new *process* shares the cwd/umask/root with the parent, the way + // Chromium's `chrome-sandbox` spawns its chroot helper (`clone(CLONE_FS | SIGCHLD)`) + // so that the helper's `chroot` lands on the sandboxed process. + let supported_process_flags = process_kind_flags | process_tid_flags | CloneFlags::FS; + if !flags.intersects(!supported_process_flags) { + match flags & process_kind_flags { + kind_flags if kind_flags.is_empty() => { + return self.do_process_clone(ctx, args, flags, ProcessCloneKind::Fork); + } + kind_flags if kind_flags.bits() == CloneFlags::VFORK.bits() => { + return self.do_process_clone(ctx, args, flags, ProcessCloneKind::VforkCopy); + } + kind_flags if kind_flags.bits() == process_kind_flags.bits() => { + return self.do_process_clone(ctx, args, flags, ProcessCloneKind::VforkShared); + } + _ => {} + } + } + + let required_clone_flags = + CloneFlags::VM | CloneFlags::THREAD | CloneFlags::SIGHAND | CloneFlags::FILES; + + let supported_clone_flags = CloneFlags::VM + | CloneFlags::FS + | CloneFlags::FILES + | CloneFlags::SIGHAND + | CloneFlags::PARENT + | CloneFlags::THREAD + | CloneFlags::SETTLS + | CloneFlags::PARENT_SETTID + | CloneFlags::CHILD_CLEARTID + | CloneFlags::CHILD_SETTID + // Ignored since we don't support sysv semaphores anyway. + | CloneFlags::SYSVSEM; + + if flags.intersects(!supported_clone_flags) { + log_unsupported!( + "clone with unsupported flags: {:?}", + flags & !supported_clone_flags + ); + return Err(Errno::EINVAL); + } + if !flags.contains(required_clone_flags) { + log_unsupported!( + "clone with missing required flags: {:?}", + required_clone_flags & !flags + ); + return Err(Errno::EINVAL); + } + + if cgroup != 0 { + log_unsupported!("clone with cgroup"); + return Err(Errno::EINVAL); + } + + if set_tid != 0 || set_tid_size != 0 { + log_unsupported!("clone with set_tid"); + return Err(Errno::EINVAL); + } + + // `exit_signal` names the signal to send the parent when this task dies. Only its range + // is checked: this shim always sends `SIGCHLD` (see `ProcessTable::record_exit`), and a + // parent that asked for something else would learn of its children through `wait4` + // anyway (see `Task::sys_wait4`). + if exit_signal > MAX_SIGNAL_NUMBER { + return Err(Errno::EINVAL); + } + + // A new thread shares its process's place in a `SharedAddressSpace` (membership is + // per process), so a forked child that never `exec`s may go multithreaded; see + // `Task::quiesce_and_hand_off` for how such a process takes its turns. This used to + // opportunistically drop the membership here when the process had gone solo in its + // family (every forked child since exited or exec'ed) -- but a process is not done + // forking just because its most recent child already exited (a fork server's ordinary + // loop), and doing this on the hottest possible path (`pthread_create`, called before + // every new thread) churned through disjoint families constantly, implicated live + // 2026-09-07 in guest-level `struct pthread` corruption after a fork (see + // `Task::release_address_space`'s doc comment and memory + // litebox-chromium-zygote-fork-corruption.md). A solo-but-still-membered process costs + // nothing to leave as is: `leave_address_space` (execve, or this process's last thread + // exiting) retires it for real when the process is actually done with it. + + let tls = if flags.contains(CloneFlags::SETTLS) { + let addr = tls.trunc(); + #[cfg(target_arch = "x86_64")] + { + // Validate the user-controlled TLS base before spawning the + // thread: `wrfsbase` faults on a non-canonical address, so an + // unchecked value would take down the host, not the guest. + // aarch64 needs no equivalent check -- the guest thread pointer + // is virtualized into a memory slot rather than written to the + // hardware register, so any value is inert until the guest + // dereferences it. Linux's `copy_thread` likewise stores the + // aarch64 value unvalidated. + if !litebox_common_linux::arch::is_valid_user_fs_base(addr) { + return Err(Errno::EPERM); + } + } + Some(ThreadLocalDescriptor::from_usize(addr)) + } else { None }; @@ -665,75 +2976,1404 @@ impl Task { } else { None }; - let clear_child_tid = if flags.contains(CloneFlags::CHILD_CLEARTID) { - child_tid + let clear_child_tid = if flags.contains(CloneFlags::CHILD_CLEARTID) { + child_tid + } else { + None + }; + let set_parent_tid = if flags.contains(CloneFlags::PARENT_SETTID) && parent_tid != 0 { + Some(UserPtrMut::from_usize(parent_tid.trunc())) + } else { + None + }; + + let fs = if flags.contains(CloneFlags::FS) { + self.fs.borrow().clone() + } else { + alloc::sync::Arc::new((**self.fs.borrow()).clone()) + }; + + let child_tid = self.global.next_thread_id.fetch_add(1, Ordering::Relaxed); + if let Some(parent_tid_ptr) = set_parent_tid { + let _ = parent_tid_ptr.write_at_offset::(0, child_tid); + } + + if (stack == 0 && stack_size != 0) || (stack != 0 && clone3 && stack_size == 0) { + return Err(Errno::EINVAL); + } + let sp = if stack != 0 { + let stack: usize = stack.trunc(); + Some(stack.wrapping_add(stack_size.trunc())) + } else { + None + }; + + let thread = self.thread.new_thread(child_tid).ok_or(Errno::EBUSY)?; + thread.remote.set_comm(&self.comm.get()); + thread.init_state.set(ThreadInitState::NewThread { + stack: sp, + tls, + set_child_tid, + // Captured on this (the parent/calling) thread: `get_fp_state` + // reads whichever thread it is called on, so the child's FPSIMD + // file must be read here, before the new host OS thread (with its + // own zeroed FP shadow) starts running. + #[cfg(target_arch = "aarch64")] + fp: self.global.platform.get_fp_state(), + }); + thread.clear_child_tid.set(clear_child_tid); + + let r = unsafe { + self.global.platform.spawn_thread( + ctx, + Box::new(NewThreadArgs { + task: Task { + global: self.global.clone(), + wait_state: crate::wait::WaitState::new(self.global.platform), + thread, + pid: self.pid, + tid: child_tid, + ppid: self.ppid, + credentials: RefCell::new(self.credentials.borrow().clone()), + comm: self.comm.clone(), + fs: fs.into(), + files: self.files.clone(), // TODO: !CLONE_FILES support + signals: self.signals.clone_for_new_task(), + guest_sp: Cell::new(0), + }, + }), + ) + }; + if let Err(err) = r { + litebox_util_log::error!(err:% = err; "failed to spawn thread"); + // Treat all spawn errors as `ENOMEM`. `EAGAIN` and other errors are + // for conditions the user can control (such as "in-shim" rlimit + // violations). + return Err(Errno::ENOMEM); + } + + Ok(usize::try_from(child_tid).unwrap()) + } + + fn do_process_clone( + &self, + ctx: &litebox_common_linux::PtRegs, + args: &litebox_common_linux::CloneArgs, + flags: CloneFlags, + kind: ProcessCloneKind, + ) -> Result { + const MAX_SIGNAL_NUMBER: u64 = 64; + if args.exit_signal > MAX_SIGNAL_NUMBER { + return Err(Errno::EINVAL); + } + if args.set_tid != 0 || args.set_tid_size != 0 || args.cgroup != 0 { + log_unsupported!("fork with set_tid or cgroup"); + return Err(Errno::EINVAL); + } + // A stack of the child's own is honoured only when the child runs on the parent's live + // memory (`CLONE_VM|CLONE_VFORK`): musl's `posix_spawn` is `clone(CLONE_VM|CLONE_VFORK| + // SIGCHLD, stack, ...)` with the child function's frame on a small buffer in the parent, + // and the suspended parent is exactly why that memory stays valid. A copying fork snapshots + // and restores memory around the parent's own `sp`, which a foreign stack would sit outside + // of, so it is still refused. Legacy `clone` passes the initial `sp` itself (`stack_size` + // 0); `clone3` passes the base and size, the way `do_clone` already reads them. + let child_sp = if args.stack != 0 || args.stack_size != 0 { + if !kind.shares_parent_vm() || args.stack == 0 { + log_unsupported!("fork with a stack"); + return Err(Errno::EINVAL); + } + let stack: usize = args.stack.trunc(); + Some(stack.wrapping_add(args.stack_size.trunc())) + } else { + None + }; + let child_tid_ptr = + (args.child_tid != 0).then(|| UserPtrMut::from_usize(args.child_tid.trunc())); + let set_child_tid = flags + .contains(CloneFlags::CHILD_SETTID) + .then_some(child_tid_ptr) + .flatten(); + let clear_child_tid = flags + .contains(CloneFlags::CHILD_CLEARTID) + .then_some(child_tid_ptr) + .flatten(); + let set_parent_tid = if flags.contains(CloneFlags::PARENT_SETTID) { + Some(UserPtrMut::from_usize(args.parent_tid.trunc())) } else { None }; - let set_parent_tid = if flags.contains(CloneFlags::PARENT_SETTID) && parent_tid != 0 { - Some(UserPtrMut::from_usize(parent_tid.trunc())) - } else { - None + if matches!(kind, ProcessCloneKind::VforkCopy) && self.process().nr_threads() > 1 { + // A copied VM still uses the one native host mapping. Keeping only the calling parent + // task suspended while sibling threads continue on the parent's copy needs a + // process-wide address-space membership, which this backend does not yet have. + log_unsupported!("CLONE_VFORK without CLONE_VM from a multithreaded process"); + return Err(Errno::ENOSYS); + } + // A vfork child that has not exec'd yet runs on its *parent's* memory, so the family + // this fork creates is really the parent's: the child stands in for it until it exits or + // execs and hands the membership over (`hand_address_space_to_vfork_parent`). That has + // nowhere to go when the parent is already a member of another family, so refuse rather + // than run two token protocols over one memory. + if kind.copies_vm() + && self.process().shares_parent_vm() + && self.process().vfork_parent_shares_address_space() + { + log_unsupported!("fork from a vfork child whose parent has itself forked"); + return Err(Errno::ENOSYS); + } + let _fork_gate_guard = match kind { + ProcessCloneKind::Fork if self.process().nr_threads() > 1 => { + Some(self.park_sibling_threads_for_fork()) + } + ProcessCloneKind::Fork + | ProcessCloneKind::VforkCopy + | ProcessCloneKind::VforkShared => None, }; + let parent_tid_is_shared = set_parent_tid.is_some_and(|ptr| { + let start = ptr.as_usize(); + let end = start.saturating_add(core::mem::size_of::()); + self.global.pm.mappings().into_iter().any(|(range, flags)| { + range.start <= start && end <= range.end && flags.contains(VmFlags::VM_SHARED) + }) + }); + // A vfork child shares every byte with its parent. An ordinary fork child sees a + // CLONE_PARENT_SETTID store only when the pointer itself names a shared mapping; private + // parent memory is updated after the parent reacquires its own image below. + let set_parent_tid_before_child = kind.shares_parent_vm() || parent_tid_is_shared; + let child_pid = self.global.next_thread_id.fetch_add(1, Ordering::Relaxed); + let files = self.files.borrow().fork_copy(self)?; let fs = if flags.contains(CloneFlags::FS) { self.fs.borrow().clone() } else { alloc::sync::Arc::new((**self.fs.borrow()).clone()) }; - let child_tid = self.global.next_thread_id.fetch_add(1, Ordering::Relaxed); - if let Some(parent_tid_ptr) = set_parent_tid { - let _ = parent_tid_ptr.write_at_offset::(0, child_tid); + // The guest's thread pointer lives in a per-host-thread slot, so the new host thread has + // to be told the value the parent is running with -- the libc data it points at is in the + // address space the child is about to share. + let tls = self + .global + .platform + .get_arch_specific_register(&GUEST_TLS_REGISTER) + .ok() + .filter(|tls| *tls != 0) + .map(ThreadLocalDescriptor::from_usize); + + let created_parent_address_space = kind.copies_vm() && !self.shares_address_space(); + let fork_sp = guest_stack_pointer(ctx); + let child_tid_range = (set_child_tid.is_some() || clear_child_tid.is_some()) + .then(|| { + let start = child_tid_ptr.unwrap().as_usize(); + start..start.saturating_add(core::mem::size_of::()) + }) + .and_then(|tid_range| { + self.global + .pm + .mappings() + .into_iter() + .any(|(stack, flags)| { + flags.contains(VmFlags::VM_WRITE) + && !flags.contains(VmFlags::VM_SHARED) + && stack.start < fork_sp + && fork_sp <= stack.end + && stack.start <= tid_range.start + && tid_range.end <= stack.end + }) + .then_some(tid_range) + }); + let (shared, preserved_stack_ranges, shared_ranges) = if kind.copies_vm() { + let membership = self.join_address_space(); + let mut ranges = self.preserved_address_space_ranges(); + if let Some(range) = child_tid_range.as_ref() { + ranges.insert_bounded(range.clone(), MAX_PRESERVED_STACK_RANGES); + } + // Everything the parent owns right now is what the child inherits, and so what the + // two of them must copy out and back for each other. + let shared_ranges = self.process().owned_ranges.lock().clone(); + (Some(membership.shared.clone()), ranges, shared_ranges) + } else { + (None, OwnedRanges::default(), OwnedRanges::default()) + }; + let vfork_completion = kind + .waits_for_exec_or_exit() + .then(|| Arc::new(VforkCompletion::new(self.shares_address_space()))); + let launch = Arc::new(ProcessLaunch::new()); + + let thread = match kind { + ProcessCloneKind::Fork => ThreadState::new_forked_process( + child_pid, + self.process().process_group_id(), + self.process().futex_manager.clone(), + launch.clone(), + ), + ProcessCloneKind::VforkCopy => ThreadState::new_vfork_copy_process( + child_pid, + self.process().process_group_id(), + self.process().futex_manager.clone(), + vfork_completion.as_ref().unwrap().clone(), + launch.clone(), + ), + ProcessCloneKind::VforkShared => ThreadState::new_vforked_process( + child_pid, + self.process().process_group_id(), + self.process(), + vfork_completion.as_ref().unwrap().clone(), + launch.clone(), + ), + }; + thread.init_state.set(ThreadInitState::NewThread { + // Usually no stack of its own: it runs on the parent's, below the parent's `sp`. + stack: child_sp, + tls, + set_child_tid, + // See the `do_clone` call site's identical capture: read on the + // parent thread, before the child's own OS thread (and its + // separately zeroed FP shadow) starts running. + #[cfg(target_arch = "aarch64")] + fp: self.global.platform.get_fp_state(), + }); + thread.clear_child_tid.set(clear_child_tid); + + let child = Task { + global: self.global.clone(), + wait_state: crate::wait::WaitState::new(self.global.platform), + thread, + pid: child_pid, + tid: child_pid, + ppid: self.pid, + credentials: RefCell::new(self.credentials.borrow().clone()), + comm: self.comm.clone(), + fs: fs.into(), + files: Arc::new(files).into(), + signals: self.signals.clone_for_new_process(), + guest_sp: Cell::new(fork_sp), + }; + if let Some(shared) = shared.as_ref() { + let membership = Arc::new(AddressSpaceMembership::new( + shared.clone(), + preserved_stack_ranges.clone(), + shared_ranges, + )); + membership.mark_acquired(self.global.platform.now()); + *child.process().address_space.lock() = Some(membership); + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. This is the CHILD's side of joining the family -- + // `Task::join_address_space` (which sets `family_id` on the parent) only ever runs + // on the forking (parent) task, so without this the child's own `family_id` would + // stay at its default 0 despite genuinely, correctly sharing `shared` with the + // parent, producing a false-positive "cross-lineage" reading for every legitimate + // parent/child overlap. + child + .process() + .vm + .current() + .family_id + .store(Arc::as_ptr(shared) as usize, Ordering::Release); + } + let child_inner = child.process().inner.clone(); + child.process().limits.inherit_from(&self.process().limits); + child.process().inherit_proc_identity(self.process(), self.pid); + child.thread.remote.set_comm(&self.comm.get()); + child.process().session_id.store( + self.process().session_id.load(Ordering::Acquire), + Ordering::Release, + ); + child.process().controlling_pty.store( + self.process().controlling_pty.load(Ordering::Acquire), + Ordering::Release, + ); + if kind.copies_vm() { + child.process().brk.store( + self.process().brk.load(Ordering::Relaxed), + Ordering::Relaxed, + ); + // A forked child initially owns a snapshot of the same ranges and patch state. A + // vfork child already resolves these fields through the parent's live VM identity. + *child.process().owned_ranges.lock() = self.process().owned_ranges.lock().clone(); + *child.process().elf_patch_cache.lock() = + self.process().elf_patch_cache.lock().clone(); } + self.global.processes.add_child(child_pid, self.pid); + // Registered before the child can run, so a child that exits immediately still finds its + // parent (this one) in the live set and can post it a `SIGCHLD`. + self.register_for_remote_signals(); + self.global.processes.register_process( + child_pid, + child.remote_signal_target(), + child.process(), + ); - if (stack == 0 && stack_size != 0) || (stack != 0 && clone3 && stack_size == 0) { - return Err(Errno::EINVAL); + let r = unsafe { + self.global + .platform + .spawn_thread(ctx, Box::new(NewThreadArgs { task: child })) + }; + if let Err(err) = r { + litebox_util_log::error!(err:% = err; "failed to spawn child process"); + launch.abort(); + self.global.processes.forget_failed_spawn(child_pid); + if created_parent_address_space { + // `join_address_space` made this membership solely for the child that failed to + // launch. The parent still holds the token, so discard the bookkeeping without + // releasing it. + self.process().address_space.lock().take(); + } + return Err(Errno::ENOMEM); } - let sp = if stack != 0 { - let stack: usize = stack.trunc(); - Some(stack.wrapping_add(stack_size.trunc())) + if kind.copies_vm() + && let Some(range) = child_tid_range + { + self.preserve_address_space_range(range); + } + // The child host thread exists but remains blocked behind `ProcessLaunch`. A shared-memory + // parent-TID store must be visible before it starts; a private ordinary-fork store is + // deliberately deferred until the parent's snapshotted image is live again. + if set_parent_tid_before_child && let Some(parent_tid_ptr) = set_parent_tid { + let _ = parent_tid_ptr.write_at_offset::(0, child_pid); + } + // Keep the new host thread behind its launch gate until spawn has succeeded. This makes + // registration and exit publication transactional: an error cannot leave a zombie or + // SIGCHLD for a child that never ran. For a copying fork, hand off the snapshotted address + // space only now: if host-thread creation failed, the parent still owns the token and can + // return ENOMEM instead of waiting forever for a child that does not exist. + if kind.copies_vm() { + self.park_and_hand_off(fork_sp, child_pid, &child_inner); + } + launch.commit(); + + let parent_image_is_live = match kind { + ProcessCloneKind::Fork => self.acquire_address_space(), + ProcessCloneKind::VforkCopy => { + vfork_completion.as_ref().unwrap().wait(); + self.acquire_address_space() + } + ProcessCloneKind::VforkShared => { + let completion = vfork_completion.as_ref().unwrap(); + completion.wait(); + self.inherit_address_space_from_vfork_child(completion) + } + }; + if parent_image_is_live + && !set_parent_tid_before_child + && let Some(parent_tid_ptr) = set_parent_tid + { + let _ = parent_tid_ptr.write_at_offset::(0, child_pid); + } + Ok(usize::try_from(child_pid).unwrap()) + } + + /// Returns this process's membership in a shared address space, creating one (with this + /// process as its first member and current holder) if it is not in one yet, and records the + /// ranges this process owns right now as shared: a child forked now inherits all of them. + fn join_address_space(&self) -> Arc> { + let owned = self.process().owned_ranges.lock().clone(); + let mut slot = self.process().address_space.lock(); + if let Some(membership) = slot.as_ref() { + debug_assert!( + membership.holding(), + "forking without holding the address space" + ); + membership.shared_ranges.lock().union_with(&owned); + return membership.clone(); + } + // DIAGNOSTIC (musl-fork-atfork-parent-corruption-20260905, temporary, additive-only): + // logs every time a fork starts a *brand new* SharedAddressSpace rather than reusing an + // existing one for this process. Fires legitimately on a process's first-ever fork; if it + // also fires on a LATER fork by a process with a live fork-family history, that would mean + // `leave_address_space_if_alone` (called from every plain `do_clone`/pthread_create) raced + // this fork and dropped the prior membership out from under it, silently starting a second, + // disconnected token/SharedAddressSpace over the same physical guest memory. + litebox_util_log::debug!( + pid:? = self.pid, tid:? = self.tid; + "diag: join_address_space creating a brand-new SharedAddressSpace (no prior membership)" + ); + let shared = Arc::new(SharedAddressSpace::new(self.pid, &self.process().inner)); + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + self.process() + .vm + .current() + .family_id + .store(Arc::as_ptr(&shared) as usize, Ordering::Release); + let membership = Arc::new(AddressSpaceMembership::new( + shared, + OwnedRanges::default(), + owned, + )); + *slot = Some(membership.clone()); + membership + } + + fn preserve_address_space_range(&self, range: Range) { + let membership = self.membership().expect("preserving a range before fork join"); + membership + .preserved_stack_ranges + .lock() + .insert_bounded(range, MAX_PRESERVED_STACK_RANGES); + } + + /// This process's membership, if its memory is shared with another guest process. + fn membership(&self) -> Option>> { + self.process().address_space.lock().clone() + } + + fn preserved_address_space_ranges(&self) -> OwnedRanges { + let membership = self.membership().expect("copying ranges before fork join"); + let ranges = membership.preserved_stack_ranges.lock().clone(); + ranges + } + + /// Copies this process's memory out and passes the address space directly to the child + /// process `pid` (thread table `inner`). + /// + /// # Panics + /// + /// Panics if this process is not currently a member holding the token; `fork` is the only + /// caller and it has just made sure of both. + fn park_and_hand_off( + &self, + sp: usize, + pid: i32, + inner: &Arc>>, + ) { + let membership = self.membership().expect("forking outside an address space"); + assert!(membership.holding()); + let saved = { + let preserved = membership.preserved_stack_ranges.lock(); + let shared_ranges = membership.shared_ranges.lock(); + self.save_address_space(sp, &preserved, &shared_ranges) + }; + // Linux `dup_mmap`: `MADV_WIPEONFORK` ranges are zero-filled in the child. The parent's + // copy has just been saved, so wiping the live pages now (before the child runs on them) + // is exactly what the child sees and nothing the parent cannot restore. Wipes are + // clamped to this process's own ranges: the manager's entries may cover a neighbour. + let owned = self.process().owned_ranges.lock(); + let wiped = unsafe { + self.global + .pm + .wipe_on_fork_child(|r, _| owned.intersect(&r).collect::>()) + }; + drop(owned); + if wiped != 0 { + litebox_util_log::debug!(pid:? = self.pid, tid:? = pid, wiped; "fork: wiped MADV_WIPEONFORK ranges for the child"); + } + *membership.parked.lock() = Some(saved); + membership.holding.store(false, Ordering::Release); + membership.shared.hand_off_to(pid, inner); + } + + /// Gives the address space up for as long as this task is blocked, so that another member can + /// run on it. Paired with [`Task::acquire_address_space`]. + /// + /// Does nothing if this process is not sharing an address space, or is already parked. If it + /// is the *only* remaining member, membership is dropped instead of parked: nobody can take + /// the token, so copying memory out and back would be pure cost. (A new member can only + /// appear via `fork`, which requires holding the token, so no member can turn up while this + /// runs.) + /// + /// A single-threaded member parks eagerly, every time it blocks: it has nothing else to run, + /// and a shell's forked child must be able to run the moment the shell waits on it. A + /// multithreaded member parks only when another member is actually waiting -- a sibling + /// thread may well have work to do, and a hand-off means quiescing all of them -- see + /// [`Task::quiesce_and_hand_off`]. + pub(crate) fn release_address_space(&self) { + let Some(membership) = self.membership() else { + return; + }; + if !membership.holding() { + return; + } + // `strong_count == 1` ("nobody else is left in this family") is only a safe signal to + // tear the membership down outright for a single-threaded member: it is about to park + // anyway (below), so there is nothing else useful the fact could gate. For a + // multithreaded member -- a fork server / zygote is exactly this shape -- it is not: + // the most recently forked child dying does not mean THIS process will not fork again + // shortly, and destroying the family here just forces the next fork's `join_address_space` + // to build a brand-new one from scratch instead of reusing this one. Diagnosed live + // 2026-09-07 (musl-fork-atfork-parent-corruption-20260905): a Chromium zygote observed + // creating 23 independent, disjoint `SharedAddressSpace` families in a single ~7 s run + // this way, immediately before/around guest-level `struct pthread` corruption in forked + // children (see memory litebox-chromium-zygote-fork-corruption.md). The multithreaded + // branch below already no-ops correctly when genuinely alone (`waiters()` is 0 with + // nobody left to wait), so skipping this shortcut there costs nothing but leaving an + // otherwise-idle membership allocated a little longer, until `leave_address_space` + // (execve or last-thread-exit) retires it for real. + if self.process().nr_threads() > 1 { + if membership.shared.waiters() > 0 + && membership.quantum_elapsed(self.global.platform.now()) + { + self.quiesce_and_hand_off(&membership); + } + return; + } + if Arc::strong_count(&membership.shared) == 1 { + *self.process().address_space.lock() = None; + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + self.process().vm.current().family_id.store(0, Ordering::Release); + return; + } + let saved = { + let preserved = membership.preserved_stack_ranges.lock(); + let shared_ranges = membership.shared_ranges.lock(); + self.save_address_space(self.guest_sp.get(), &preserved, &shared_ranges) + }; + *membership.parked.lock() = Some(saved); + membership.holding.store(false, Ordering::Release); + membership.shared.release(); + } + + /// Hands the address space to a waiting member and takes it back, on behalf of this whole + /// multithreaded process. Called at every safe point -- before blocking, when kicked out of + /// a wait, and before re-entering guest code -- once a waiter exists. + /// + /// A single-threaded member does not go through here: it yields when it blocks, and only + /// then (no preemption), which is the behaviour every shell-shaped guest was built against. + pub(crate) fn yield_address_space_to_waiters(&self) { + let Some(membership) = self.membership() else { + return; + }; + if !membership.holding() || membership.shared.waiters() == 0 || self.is_exiting() { + return; + } + if self.process().nr_threads() <= 1 { + return; + } + if !membership.quantum_elapsed(self.global.platform.now()) { + return; + } + self.quiesce_and_hand_off(&membership); + } + + /// The multithreaded hand-off. The first thread here becomes the initiator: it closes the + /// fork gate so every sibling parks at its next safe point (blocked siblings are kicked + /// there; one mid-copy into guest memory finishes first), copies the process's shared image + /// out, releases the token, waits for it to come back, restores the image and reopens the + /// gate. Any later thread just parks at that gate like a sibling. Nothing but the initiator + /// touches guest memory between the copy out and the copy back. + fn quiesce_and_hand_off(&self, membership: &Arc>) { + if membership + .quiescing + .compare_exchange(false, true, Ordering::AcqRel, Ordering::Acquire) + .is_err() + { + self.park_while_fork_gate_closed(); + return; + } + let started = self.global.platform.now(); + let guard = self.park_sibling_threads_for_fork(); + let quiesced = self.global.platform.now(); + // Re-checked with every sibling parked: the token may have changed hands while this + // thread was closing the gate (a sibling won the race in `acquire_address_space`). + if membership.holding() && membership.shared.waiters() > 0 { + let saved = { + let preserved = membership.preserved_stack_ranges.lock(); + let shared_ranges = membership.shared_ranges.lock(); + self.save_address_space(self.guest_sp.get(), &preserved, &shared_ranges) + }; + *membership.parked.lock() = Some(saved); + membership.holding.store(false, Ordering::Release); + membership.shared.release(); + membership.shared.wait_until_taken(|| self.is_exiting()); + let released = self.global.platform.now(); + let got = membership.shared.acquire( + self.pid, + || membership.holding(), + || self.is_exiting(), + &self.process().inner, + ); + let back = self.global.platform.now(); + if got && !membership.holding() { + if let Some(saved) = membership.parked.lock().take() { + self.restore_address_space(saved); + } + membership.holding.store(true, Ordering::Release); + } + if got { + membership.mark_acquired(self.global.platform.now()); + } + litebox_util_log::debug!( + pid:? = self.pid, tid:? = self.tid, + quiesce_us:? = quiesced.duration_since(&started).as_micros(), + save_us:? = released.duration_since(&quiesced).as_micros(), + away_us:? = back.duration_since(&released).as_micros(), + got; + "address space: multithreaded hand-off complete" + ); + } + membership.quiescing.store(false, Ordering::Release); + drop(guard); + } + + /// Takes the address space back and restores this process's memory into it, blocking until + /// the current holder gives it up. + /// + /// Does nothing if this process is not sharing an address space, or already holds it. Returns + /// whether this process's own memory image is live when the call completes. + pub(crate) fn acquire_address_space(&self) -> bool { + let Some(membership) = self.membership() else { + return true; + }; + if membership.holding() { + return true; + } + if !membership.shared.acquire( + self.pid, + || membership.holding(), + || self.is_exiting(), + &self.process().inner, + ) { + // Preserve the non-holding membership until exit cleanup. Dropping it here would make + // `prepare_for_exit` mistake whichever sibling's image is live for this task's own and + // dereference stale robust-list/child-TID pointers into that sibling. + return false; + } + if !membership.holding() { + if let Some(saved) = membership.parked.lock().take() { + self.restore_address_space(saved); + } + membership.holding.store(true, Ordering::Release); + membership.mark_acquired(self.global.platform.now()); + } + true + } + + /// Leaves the shared address space for good, waking anything waiting for it. + /// + /// Returns `true` if, afterwards, this process's guest memory is its alone -- either it was + /// never shared, or this was the last member -- and so is safe to tear down. Called from + /// `execve`, whose new image lives at addresses no other member owns, and from the exit of + /// a process's last thread. + fn leave_address_space(&self) -> bool { + let Some(membership) = self.process().address_space.lock().take() else { + return true; + }; + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. This process is leaving its family for good either way. + self.process().vm.current().family_id.store(0, Ordering::Release); + if membership.holding() { + membership.shared.release(); + } + // The only other strong reference is this one, so no other member is left to care about + // the memory. A member can only be created by `fork`, which needs the token, and this + // task held it until the line above. + Arc::strong_count(&membership.shared) == 1 + } + + /// Whether this process's guest memory is shared with another guest process. + fn shares_address_space(&self) -> bool { + self.process().address_space.lock().is_some() + } + + /// Passes this vfork child's [`AddressSpaceMembership`] -- token, parked image and all -- to + /// the parent whose memory it has been running on, which + /// [`Task::inherit_address_space_from_vfork_child`] installs once the vfork completes. + /// + /// The family was created by a `fork` this child issued on the parent's memory (see + /// `do_process_clone`), so its other members' images alias the *parent's* mappings; the + /// parent has to keep taking turns with them, exactly as it would had it forked itself. A + /// membership is only ever handed over before [`Process::complete_vfork`] wakes the parent. + /// Dropped nothing: a child that never forked has no membership to pass. + fn hand_address_space_to_vfork_parent(&self) { + let Some(membership) = self.process().address_space.lock().take() else { + return; + }; + match self.process().vfork_completion.lock().as_ref() { + Some(completion) => *completion.inherited_membership.lock() = Some(membership), + // The vfork already completed (an exec'd child exiting), so this membership is the + // child's own and leaves the family the ordinary way. + None => { + if membership.holding() { + membership.shared.release(); + } + } + } + } + + /// Takes over the family a vfork child forked into while on this task's memory, and returns + /// whether this task's own image is live afterwards. + /// + /// The child may have died parked (killed while waiting for the token), in which case the + /// token is taken and the child's last image put back before this task runs another guest + /// instruction on it. + fn inherit_address_space_from_vfork_child( + &self, + completion: &VforkCompletion, + ) -> bool { + let Some(membership) = completion.inherited_membership.lock().take() else { + return false; + }; + debug_assert!( + self.process().address_space.lock().is_none(), + "vfork parent was refused a family of its own (see do_process_clone)" + ); + // The membership now belongs to this process; its threads are the ones to kick. + if membership.holding() { + membership + .shared + .hand_off_to(self.pid, &self.process().inner); + } + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + self.process() + .vm + .current() + .family_id + .store(Arc::as_ptr(&membership.shared) as usize, Ordering::Release); + *self.process().address_space.lock() = Some(membership); + self.acquire_address_space() + } + + /// Leaves the shared address space if this task is its only remaining member, and reports + /// whether this task's guest memory is now unshared. + /// + /// `execve` uses this to decide whether the old mappings are its to tear down. A sole member + /// still holds the token, so no new member can appear while this runs. + /// Releases every mapping this process owns, at process exit, with the same + /// intersection discipline `execve` uses when it discards an old image: only + /// `owned_ranges ∩ mapping`, never a whole coalesced page-manager entry that + /// merely overlaps (an adjacent sibling's memory shares entries with ours), + /// reserved mappings (empty `VmFlags`) untouched, and a live `/dev/fb0` + /// guest mapping deregistered before its pages go away. Callers must have + /// established that no other guest process can be using these bytes. + /// Releases, at the exit of a process that is still sharing an address space with others, + /// only the ranges no other member can own: `owned_ranges` minus the ranges shared with the + /// family (see [`AddressSpaceMembership::shared_ranges`]). Same intersection discipline as + /// [`Self::release_owned_memory_on_exit`]. + fn release_private_memory_on_exit(&self, shared_ranges: &OwnedRanges) { + let mut owned = self.process().owned_ranges.lock(); + let private = owned.difference(shared_ranges); + if private.ranges.is_empty() { + return; + } + litebox_util_log::debug!( + pid:? = self.pid, count:? = private.ranges.len(); + "exit: releasing the process's private mappings, shared ones stay with the family" + ); + if let Some(fb) = self.global.framebuffer.as_ref() + && let Some((fb_addr, fb_len)) = fb.guest_mapping() + && private + .intersect(&(fb_addr..fb_addr.saturating_add(fb_len))) + .next() + .is_some() + { + fb.clear_guest_mapping_overlapping(fb_addr, fb_len); + } + let release = |r: Range, vm: VmFlags| { + if vm.is_empty() { + Vec::new() + } else { + private.intersect(&r).collect::>() + } + }; + // SAFETY: only ranges this process mapped after it joined the family are released -- + // no other member's `owned_ranges` can include them -- and its last thread is exiting. + if let Err(error) = unsafe { self.global.pm.release_memory(release) } { + litebox_util_log::error!(error:? = error; "exit: failed to release the process's private mappings"); + } + let remaining = owned.difference(&private); + *owned = remaining; + } + + fn release_owned_memory_on_exit(&self) { + let owned = self.process().owned_ranges.lock(); + if let Some(fb) = self.global.framebuffer.as_ref() + && let Some((fb_addr, fb_len)) = fb.guest_mapping() + && owned + .intersect(&(fb_addr..fb_addr.saturating_add(fb_len))) + .next() + .is_some() + { + fb.clear_guest_mapping_overlapping(fb_addr, fb_len); + } + let release = |r: Range, vm: VmFlags| { + if vm.is_empty() { + Vec::new() + } else { + owned.intersect(&r).collect::>() + } + }; + // SAFETY: the caller established that this process is the sole user of its owned + // ranges (alone in its address-space family and not vfork-sharing a parent), and its + // last thread is exiting, so no guest code can touch them again. + if let Err(error) = unsafe { self.global.pm.release_memory(release) } { + litebox_util_log::error!(error:? = error; "exit: failed to release the process's mappings"); + } + drop(owned); + self.process().owned_ranges.lock().clear(); + } + + fn leave_address_space_if_alone(&self) -> bool { + let mut slot = self.process().address_space.lock(); + let Some(membership) = slot.as_ref() else { + return true; + }; + if Arc::strong_count(&membership.shared) != 1 { + return false; + } + // DIAGNOSTIC (musl-fork-atfork-parent-corruption-20260905, temporary, additive-only): + // logs every time a plain do_clone (pthread_create) drops this process's + // AddressSpaceMembership because it observed strong_count==1. See the matching + // diagnostic in join_address_space: if that one also fires soon after for the SAME pid, + // the drop-then-recreate pair is the suspected TOCTOU. + litebox_util_log::debug!( + pid:? = self.pid, tid:? = self.tid; + "diag: leave_address_space_if_alone dropping membership (strong_count==1)" + ); + debug_assert!(membership.holding()); + *slot = None; + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + self.process().vm.current().family_id.store(0, Ordering::Release); + true + } + + /// Records the guest stack pointer for the current trip through the shim. + pub(crate) fn record_guest_sp(&self, sp: usize) { + self.guest_sp.set(sp); + } + + /// GUARD (litebox-ordinary-syscall-cross-process-clobber): whether `range` currently belongs, + /// in whole or in part, to a live process outside this one's own family. + /// + /// `overlaps_another_process` (used by `save_address_space`/`restore_address_space` above) is + /// otherwise the *only* place in this codebase that checks "a process may only ever touch its + /// own memory" before acting -- confirmed live during this investigation: an ordinary guest + /// `munmap`/`mprotect`/`mmap(MAP_FIXED)` reaches `litebox_common_linux::mm`'s handlers, which + /// operate directly on the guest-supplied address with no such check, because on real Linux a + /// process's own address space makes touching another process's memory this way physically + /// impossible -- an invariant this platform's one flat, shared, permission-mirrored host + /// address space (`hvf_backend.rs`'s own doc comment) does not itself provide. The *only* thing + /// that stopped an ordinary mutating call from ever landing on a stranger's memory was `Vmem`'s + /// own bookkeeping happening to stay perfectly consistent -- true for a placement decision + /// (already covered elsewhere: `Vmem::reserve_external`), but not a defense against a wrong or + /// stale *explicit* address a caller already has in hand for any other reason. This is called + /// from the ordinary-syscall dispatch path in `lib.rs` (not `syscalls/mm.rs`, which stays + /// unmodified) for exactly the operations that can make another process's memory disappear or + /// change permissions out from under it: `munmap`, `mprotect`, and a `MAP_FIXED` `mmap`. + pub(crate) fn touches_another_process(&self, addr: usize, length: usize) -> bool { + let Some(end) = addr.checked_add(length) else { + return false; + }; + let self_family_id = self.process().vm.current().family_id.load(Ordering::Acquire); + let hits = self.global.processes.overlaps_another_process( + self.pid, + self_family_id, + &(addr..end), + ); + if !hits.is_empty() { + // Permanent diagnostic aid, not a temporary probe: firing here is rare (a real + // cross-process overlap on an ordinary syscall) and worth a permanent record when it + // does. Confirmed 2026-09-08: across repeated live reproductions of the concurrent + // multi-`node`-process SIGSEGV this guard was added for, it never fired once, ruling + // out an ordinary explicit-address `munmap`/`mprotect`/`mmap(MAP_FIXED)` collision as + // that crash's mechanism -- the guard stays in as a real, independently-justified + // safety net regardless (see this function's own doc comment), not because it was + // shown to fix that specific bug. + litebox_util_log::error!( + pid:? = self.pid, addr:? = addr, end:? = end, hits:? = hits; + "diag: touches_another_process fired on an ordinary syscall" + ); + } + !hits.is_empty() + } + + /// Copies this process's view of the memory it shares with other members out into host + /// memory: the protection of every shared piece, plus the contents of the writable ones. + /// + /// This is what makes a shared-address-space `fork` behave like a real one for a guest that + /// was built for a real one. The child necessarily runs on the parent's memory (see + /// [`SharedAddressSpace`]), and `vfork(2)`'s contract -- "the child may only `exec` or + /// `_exit`" -- is one a `fork(2)`-using program has no reason to honour. busybox's shell, for + /// instance, *returns* out of the function that called `fork`, overwriting the frames its + /// parent is parked in, and its `forkchild` frees the parent's job list on the shared heap. + /// But only the token holder ever runs on that memory, so none of that has to be visible to + /// anyone else: this copies the memory out, and [`Task::restore_address_space`] puts it back + /// before the task executes another guest instruction. Each member then sees exactly what + /// `fork(2)` promises -- its own memory, untouched -- while the child saw a faithful copy of + /// the parent's, because it *was* it. + /// + /// Protections are part of the view. An allocator (PartitionAlloc in every Chromium + /// process) reserves gigabytes `PROT_NONE` and commits pieces with `mprotect` as it goes; a + /// member that commits and writes into a piece its sibling still has reserved must not leave + /// the sibling reading its data once the sibling commits the same piece expecting zero pages. + /// So every shared piece is recorded with its protection, and the restore re-establishes it + /// (see [`SavedRange`]). + /// + /// Deliberate limits on what is saved: + /// + /// * Only ranges this process owns *and* has shared with another member (see + /// [`AddressSpaceMembership::shared_ranges`]), so that a sibling guest process running + /// concurrently at other addresses is never rolled back, and memory a member mapped for + /// itself after the fork is never copied. + /// * Of the mapping holding the stack pointer, only the ABI-live suffix is copied: `[sp, end)` + /// on AArch64 and the 128-byte red zone plus that suffix on x86-64. Explicit kernel ABI + /// pointers below that boundary, such as clone child-TID words, are copied separately through + /// `preserved_stack_ranges`; this avoids copying the whole 8 MiB stack on every handoff. + /// * Read-only pieces (the image's text and rodata) carry no contents: nothing correct writes + /// to them, and they are the bulk of the image. + fn save_address_space( + &self, + sp: usize, + preserved_stack_ranges: &OwnedRanges, + shared_ranges: &OwnedRanges, + ) -> MemoryImage { + let started = self.global.platform.now(); + // Cloned, not held: the diagnostic `overlaps_another_process` call inside `save` below + // (musl-fork-struct-pthread-corruption, temporary, additive-only) locks OTHER processes' + // `owned_ranges` while running, and this process is quiesced for the whole duration of a + // hand-off (see `Task::quiesce_and_hand_off`/the single-threaded caller in + // `release_address_space`), so nothing mutates `owned_ranges` underneath a snapshot here + // -- but holding this process's OWN lock while acquiring another's would be a lock-order + // inversion against a concurrent, unrelated process doing the same in the other + // direction (deadlock, live-reproduced during this investigation). + let owned = self.process().owned_ranges.lock().clone(); + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + let self_family_id = self.process().vm.current().family_id.load(Ordering::Acquire); + let mut saved = Vec::new(); + let mut save = |start: usize, end: usize, flags: VmFlags, with_bytes: bool| { + if start >= end { + return; + } + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `ProcessTable::overlaps_another_process`. This range is believed to be exclusively + // this process's own (per `owned_ranges`); if it also belongs to another live + // process OUTSIDE this process's own family right now, saving it is reading memory + // this process does not own. + let cross = self + .global + .processes + .overlaps_another_process(self.pid, self_family_id, &(start..end)); + if !cross.is_empty() { + litebox_util_log::error!( + pid:? = self.pid, tid:? = self.tid, self_family_id:? = self_family_id, + start:? = start, end:? = end, cross:? = cross; + "diag: CROSS-LINEAGE OVERLAP -- save_address_space range also owned by another live process" + ); + } + let bytes = if with_bytes { + match UserPtr::::from_usize(start).to_owned_slice::(end - start) { + Some(bytes) => Some(bytes), + // Only reachable if a mapping this process owns is no longer readable, which no + // correct program arranges. Loud, because the consequence is that this task's + // own writes to that range are silently lost the next time another member runs. + None => { + litebox_util_log::error!( + pid:? = self.pid, start:? = start, end:? = end; + "could not copy a mapping out before giving up the address space; this \ + process's data in it will be whatever the next process to run leaves there" + ); + return; + } + } + } else { + None + }; + saved.push(SavedRange { + start, + end, + flags: access_bits(flags), + bytes, + }); + }; + #[cfg(target_arch = "x86_64")] + let live_stack_start = sp.saturating_sub(128); + #[cfg(target_arch = "aarch64")] + let live_stack_start = sp; + + // The stack-suffix shortcut is only sound when this thread's stack is the only live one + // in its mapping: musl thread stacks are separate mmaps that the page manager coalesces + // with their neighbours, so in a multithreaded process the mapping holding `sp` may also + // hold sibling threads' stacks and TCBs, below `sp`. Save it whole then. + let suffix_only = self.process().nr_threads() <= 1; + for (range, flags) in self.global.pm.mappings() { + if flags.contains(VmFlags::VM_SHARED) { + continue; + } + let writable = flags.contains(VmFlags::VM_WRITE); + let stack = suffix_only && range.start < sp && sp <= range.end; + for owned_part in owned.intersect(&range) { + // Only what another member may also own (see + // `AddressSpaceMembership::shared_ranges`); the rest is this process's alone. + for part in shared_ranges.intersect(&owned_part) { + if !writable { + save(part.start, part.end, flags, false); + continue; + } + let start = if stack { + part.start.max(live_stack_start) + } else { + part.start + }; + save(start, part.end, flags, true); + if stack { + for extra in preserved_stack_ranges.intersect(&part) { + // The main suffix already includes anything at or above `start`. + save(extra.start, extra.end.min(start), flags, true); + } + } + } + } + } + let bytes: usize = saved.iter().map(|p| p.bytes.as_ref().map_or(0, |b| b.len())).sum(); + let elapsed = self.global.platform.now().duration_since(&started); + litebox_util_log::debug!( + pid:? = self.pid, tid:? = self.tid, pieces:? = saved.len(), bytes, + elapsed_us:? = elapsed.as_micros(); + "address space: saved this process's view" + ); + // GUARD (litebox-fork-family-allocator-reuse): this process's addresses are about to + // vanish from `vmas` (a sibling keeps running, and may `execve`, tearing down and + // rebuilding the very ranges this snapshot remembers -- `leave_address_space`'s own + // doc comment assumes "the new image lives at addresses no other member owns," which + // nothing previously enforced). Reserve every saved range so a fresh, flexible placement + // is steered elsewhere for as long as this snapshot is outstanding; released by + // `restore_address_space` below. + for piece in &saved { + self.global.pm.reserve_external(piece.start..piece.end); + } + saved + } + + /// Puts back what [`Task::save_address_space`] took, undoing everything another member did + /// to this process's view of the shared memory: protections first (a piece another member + /// committed while this process had it reserved is dropped back to zero pages and + /// `PROT_NONE`; one it decommitted is made writable again), then contents. + fn restore_address_space(&self, saved: MemoryImage) { + use litebox_common_linux::{MadviseBehavior, MapFlags, ProtFlags}; + // GUARD (litebox-fork-family-allocator-reuse): releases what `save_address_space` + // reserved. From here on this snapshot is being actively replayed back (or, per the + // existing cross-family guard below, refused piece by piece) rather than merely + // remembered, so a fresh placement colliding with it is once again this process's own + // problem to detect the ordinary way, not something the allocator needs to steer around. + for piece in &saved { + self.global.pm.release_external(piece.start..piece.end); + } + let started = self.global.platform.now(); + // DIAGNOSTIC (musl-fork-struct-pthread-corruption, temporary, additive-only): see + // `VmBookkeeping::family_id`. + let self_family_id = self.process().vm.current().family_id.load(Ordering::Acquire); + let (mut pieces, mut bytes_copied, mut protects, mut drops, mut remaps, mut protected_from_clobber) = + (0usize, 0usize, 0usize, 0usize, 0usize, 0usize); + for piece in saved { + pieces += 1; + // GUARD (litebox-restore-stale-mappings-snapshot): re-queried per piece, not once for + // the whole restore. A single up-front snapshot went stale the moment any earlier + // piece in this same loop actually mutated the address space below + // (`sys_mmap`/`sys_mprotect`/`sys_madvise`/`copy_from_slice`) -- and adjacent pieces + // from the very same original mapping are routine, not an edge case: the stack-suffix + // split above (`save_address_space`'s `preserved_stack_ranges.intersect`) deliberately + // emits two directly-touching `SavedRange`s from one VMA. A later piece reconciling + // against a stale "what's mapped right now" view can misjudge a range an earlier piece + // just re-created or re-protected -- e.g. treating a gap the earlier piece already + // filled as still needing a fresh `MAP_FIXED` remap, which zero-fills over content that + // piece just wrote. Live-observed downstream symptom: a translation fault reading a + // musl heap chunk header 4 bytes before an otherwise-valid pointer, on a page with no + // mapping at all, matching exactly what a wrongly-re-mapped-over-real-content gap would + // produce for a neighboring allocation. + let mappings = self.global.pm.mappings(); + let SavedRange { + start, + end, + flags: wanted, + bytes, + } = piece; + // GUARD (musl-fork-struct-pthread-corruption): confirmed live during this + // investigation -- `Task::restore_address_space` had no check that a saved piece's + // addresses still belong to this process's own family before touching them. When a + // piece is currently mapped but read-only (this process remembers it writable), the + // code just below would `mprotect` it to PROT_READ|PROT_WRITE and then blindly + // `copy_from_slice` this process's stale saved bytes into it; when a piece is + // currently unmapped, the code just below would `MAP_FIXED`-remap it. Neither check + // considered WHO else might legitimately, currently own that address: this platform + // has no per-process hardware isolation (`hvf_backend.rs`'s own doc comment -- one + // flat, permission-mirrored host address space, guest VA == host VA), so a family's + // remembered range that has since been reused by a completely unrelated, live guest + // process (e.g. its own fresh `execve`'s interpreter landing at the same top-down + // "highest free slot" while this family's member was parked) is real, currently-live + // memory belonging to someone else. Live-reproduced: a Chromium browser-process + // thread's restore repeatedly `mprotect`+overwrote a 16 KiB slice of an unrelated, + // concurrently-running process's just-loaded `ld-musl-aarch64.so.1` mapping this way, + // immediately preceding a real guest SIGSEGV inside musl. `overlaps_another_process` + // (added this investigation) is the one place in this codebase that checks the + // invariant "a process may only ever touch its own memory" before acting -- refusing + // this piece entirely (not just the offending sub-range) trades this process's own, + // now on its own head, incomplete restore for never writing into a byte that belongs + // to someone else: the same "honest, contained failure beats silent cross-process + // corruption" preference this whole platform is built on elsewhere. + let cross = self + .global + .processes + .overlaps_another_process(self.pid, self_family_id, &(start..end)); + if !cross.is_empty() { + protected_from_clobber += 1; + litebox_util_log::error!( + pid:? = self.pid, tid:? = self.tid, self_family_id:? = self_family_id, + start:? = start, end:? = end, cross:? = cross, has_bytes:? = bytes.is_some(); + "restore_address_space: range now owned by another live process outside this \ + family -- refusing to touch it (would silently corrupt that process's live \ + memory); this process's own restore for this piece is incomplete" + ); + continue; + } + // What is at these addresses right now, piece by piece; a gap means another member + // unmapped it. + let mut cursor = start; + let mut current: Vec<(Range, VmFlags)> = Vec::new(); + for (range, flags) in &mappings { + if range.end <= start || range.start >= end { + continue; + } + let lo = range.start.max(start); + let hi = range.end.min(end); + if cursor < lo { + current.push((cursor..lo, VmFlags::empty())); + } + current.push((lo..hi, *flags)); + cursor = hi; + } + if cursor < end { + current.push((cursor..end, VmFlags::empty())); + } + for (range, flags) in current { + let mapped = !flags.is_empty() || { + // `mappings()` reports reserved (empty-flag) entries too; a truly unmapped + // gap is what `VmFlags::empty()` with no entry means here. + mappings.iter().any(|(m, _)| m.start <= range.start && range.end <= m.end) + }; + let prot_now = access_bits(flags); + if !mapped { + // Re-create the mapping another member removed, with this process's + // protection; contents (if any) follow below. + let initial = if bytes.is_some() { + ProtFlags::PROT_READ | ProtFlags::PROT_WRITE + } else { + prot_of(wanted) + }; + remaps += 1; + if let Err(error) = self.sys_mmap( + range.start, + range.end - range.start, + initial, + MapFlags::MAP_PRIVATE | MapFlags::MAP_ANONYMOUS | MapFlags::MAP_FIXED, + -1, + 0, + ) { + litebox_util_log::error!( + pid:? = self.pid, start:? = range.start, end:? = range.end, error:?; + "failed to re-map a range another member unmapped" + ); + continue; + } + } else if bytes.is_some() { + if !flags.contains(VmFlags::VM_WRITE) { + protects += 1; + } + if !flags.contains(VmFlags::VM_WRITE) + && let Err(error) = self.sys_mprotect( + UserPtrMut::from_usize(range.start), + range.end - range.start, + ProtFlags::PROT_READ | ProtFlags::PROT_WRITE, + ) + { + litebox_util_log::error!( + pid:? = self.pid, start:? = range.start, end:? = range.end, error:?; + "failed to make a range writable when taking the address space back" + ); + continue; + } + } else if prot_now != wanted { + protects += 1; + if !wanted.contains(VmFlags::VM_READ) { + // Reserved in this process's view, committed and written by another + // member: what Linux hands out on a later commit is zero pages. + drops += 1; + if let Err(error) = self.sys_madvise( + UserPtrMut::from_usize(range.start), + range.end - range.start, + MadviseBehavior::DontNeed, + ) { + litebox_util_log::error!( + pid:? = self.pid, start:? = range.start, end:? = range.end, error:?; + "failed to drop another member's pages from a reserved range" + ); + } + } + if let Err(error) = self.sys_mprotect( + UserPtrMut::from_usize(range.start), + range.end - range.start, + prot_of(wanted), + ) { + litebox_util_log::error!( + pid:? = self.pid, start:? = range.start, end:? = range.end, error:?; + "failed to put a range's protection back when taking the address space back" + ); + } + } + } + if let Some(bytes) = bytes { + bytes_copied += bytes.len(); + if UserPtrMut::::from_usize(start) + .copy_from_slice::(0, &bytes) + .is_none() + { + let pm_view = self + .global + .pm + .mappings() + .into_iter() + .find(|(range, _)| range.contains(&start)) + .map(|(range, flags)| (range.start, range.end, flags)); + litebox_util_log::error!( + pid:? = self.pid, start:? = start, len:? = bytes.len(), pm_view:? = pm_view; + "failed to restore a mapping when taking the address space back" + ); + continue; + } + let prot = prot_of(wanted); + if prot != (ProtFlags::PROT_READ | ProtFlags::PROT_WRITE) + && let Err(error) = self.sys_mprotect( + UserPtrMut::from_usize(start), + end - start, + prot, + ) + { + litebox_util_log::error!( + pid:? = self.pid, start:? = start, end:? = end, error:?; + "failed to put a restored range's protection back" + ); + } + } + } + let elapsed = self.global.platform.now().duration_since(&started); + litebox_util_log::debug!( + pid:? = self.pid, tid:? = self.tid, pieces, bytes_copied, protects, drops, remaps, + protected_from_clobber, elapsed_us:? = elapsed.as_micros(); + "address space: restored this process's view" + ); + } + + /// Publishes this process in the shim's live-process set so other processes can post signals + /// to it. Idempotent: re-registering simply replaces the entry with an identical one. + /// + /// Done at `fork` rather than at task construction because that is the first moment a process + /// can acquire a child, and a process with no children has nothing to receive. + pub(crate) fn register_for_remote_signals(&self) { + self.global.processes.register_process( + self.pid, + self.remote_signal_target(), + self.process(), + ); + } + + /// Handle syscall `wait4`. + pub(crate) fn sys_wait4( + &self, + pid: i32, + wstatus: Option>, + options: i32, + rusage: usize, + ) -> Result { + /// `WNOHANG`: return immediately if no child has exited. + const WNOHANG: u32 = 0x1; + /// `WUNTRACED`/`WCONTINUED`: accepted and then never acted on, because this shim has no + /// way to stop or continue a process in the first place, so a wait for either event + /// simply never has one to report. + const WUNTRACED: u32 = 0x2; + const WCONTINUED: u32 = 0x8; + /// `__WNOTHREAD`/`__WALL`/`__WCLONE`: which *kinds* of child to consider. Every child + /// here is an ordinary one belonging to the caller alone, so all three are no-ops. + const WNOTHREAD: u32 = 0x2000_0000; + const WALL: u32 = 0x4000_0000; + const WCLONE: u32 = 0x8000_0000; + /// Deliberately absent: `WNOWAIT` (leave the child reapable), which this cannot honour + /// -- the reap below is destructive -- and `WEXITED`/`WSTOPPED`, which are `waitid`'s, + /// not `wait4`'s. + const SUPPORTED: u32 = WNOHANG | WUNTRACED | WCONTINUED | WNOTHREAD | WALL | WCLONE; + + let options = options.cast_unsigned(); + if options & !SUPPORTED != 0 { + log_unsupported!("wait4 with options {options:#x}"); + return Err(Errno::EINVAL); + } + let rusage = + (rusage != 0).then(|| UserPtrMut::::from_usize(rusage)); + + let filter = if pid > 0 { + WaitFilter::Pid(pid) } else { - None + // `-1` means any child. Linux applies process-group filters for `0` and `< -1`; that + // filtering is not implemented yet, so both currently use the same any-child path. + WaitFilter::Any }; + let table = &self.global.processes; - let thread = self.thread.new_thread(child_tid).ok_or(Errno::EBUSY)?; - thread.init_state.set(ThreadInitState::NewThread { - stack: sp, - tls, - set_child_tid, - }); - thread.clear_child_tid.set(clear_child_tid); + // Registered before the first check so that a child exiting in the gap between the check + // and the block cannot be missed. + let token = table.register_waiter(self.pid, self.wait_cx().waker().clone()); + let _unregister = litebox::utils::defer(|| table.unregister_waiter(token)); - let r = unsafe { - self.global.platform.spawn_thread( - ctx, - Box::new(NewThreadArgs { - task: Task { - global: self.global.clone(), - wait_state: crate::wait::WaitState::new(self.global.platform), - thread, - pid: self.pid, - tid: child_tid, - ppid: self.ppid, - credentials: self.credentials.clone(), - comm: self.comm.clone(), - fs: fs.into(), - files: self.files.clone(), // TODO: !CLONE_FILES support - signals: self.signals.clone_for_new_task(), - }, - }), - ) - }; - if let Err(err) = r { - litebox_util_log::error!(err:% = err; "failed to spawn thread"); - // Treat all spawn errors as `ENOMEM`. `EAGAIN` and other errors are - // for conditions the user can control (such as "in-shim" rlimit - // violations). - return Err(Errno::ENOMEM); + loop { + if let Some((pid, status, cpu_time_nanos)) = table.reap(self.pid, filter) { + if let Some(wstatus) = wstatus { + wstatus + .write_at_offset::(0, encode_wait_status(status)) + .ok_or(Errno::EFAULT)?; + } + if let Some(rusage) = rusage { + // `ru_utime` is the one field real scripts actually consume (`busybox time` + // among them) and the one this shim can measure honestly: real, host-metered + // CPU time summed across every thread the child ever ran (see + // `Process::cpu_time_nanos`). `ru_stime` is left at zero rather than + // fabricated -- guest syscalls run as ordinary host user-mode Rust, so this + // shim has no meaningful "kernel time" of its own to attribute, and reporting + // a fake nonzero value would be worse than reporting none. Every other field + // (`ru_maxrss` etc.) is zeroed for the same reason. This is still a strict + // improvement over leaving the caller's buffer untouched: reading uninitialized + // guest memory back as a `struct rusage` is both a correctness bug (nonsensical + // output, as seen from `busybox time`) and an information disclosure. + let value = litebox_common_linux::Rusage { + ru_utime: core::time::Duration::from_nanos(cpu_time_nanos).into(), + ..Default::default() + }; + rusage + .write_at_offset::(0, value) + .ok_or(Errno::EFAULT)?; + } + return Ok(pid); + } + if !table.has_child(self.pid, filter) { + return Err(Errno::ECHILD); + } + if options & WNOHANG != 0 { + return Ok(0); + } + self.wait_cx() + .wait_until(|| table.reap_ready(self.pid, filter)) + .map_err(|_| Errno::EINTR)?; } + } - Ok(usize::try_from(child_tid).unwrap()) + /// Records `range` as mapped by this process. See [`Process::owned_ranges`]. + pub(crate) fn record_mapped(&self, start: usize, len: usize) { + if len != 0 { + self.process() + .owned_ranges + .lock() + .insert(start..start.saturating_add(len)); + } + } + + /// Records `range` as no longer mapped by this process. + pub(crate) fn record_unmapped(&self, start: usize, len: usize) { + if len != 0 { + self.process() + .owned_ranges + .lock() + .remove(start..start.saturating_add(len)); + } } /// Handle syscall `set_tid_address`. @@ -749,8 +4389,19 @@ impl Task { } // TODO: enforce the following limits: -pub(crate) const RLIMIT_NOFILE_CUR: usize = 1024 * 1024; -const RLIMIT_NOFILE_MAX: usize = 1024 * 1024; +// +// The soft (`cur`) default is deliberately much lower than the hard ceiling, matching real Linux +// distros (e.g. systemd's `DefaultLimitNOFILE=1024:524288`-style split): a process that wants +// more can still raise it via `setrlimit`/`prlimit` up to `RLIMIT_NOFILE_MAX`. This split isn't +// just convention -- it's load-bearing here. Startup code across many real daemons (observed +// live via `dbus-daemon`, whose `bus/main.c` closes every fd up to the reported soft +// `RLIMIT_NOFILE` with one `fcntl(fd, F_GETFD)` syscall per candidate fd) scales the length of +// that loop directly off this value. With a 1,048,576 soft default that trace showed the guest +// still counting up through fd 298000+ after 6 real seconds, so dbus-daemon never became ready +// within any reasonable readiness-probe window; a low soft default keeps that a sub-millisecond, +// unnoticeable loop, exactly as it is on a real Linux host. +pub(crate) const RLIMIT_NOFILE_SOFT_DEFAULT: usize = 1024; +pub(crate) const RLIMIT_NOFILE_MAX: usize = 1024 * 1024; struct AtomicRlimit { cur: core::sync::atomic::AtomicUsize, @@ -770,26 +4421,56 @@ pub(crate) struct ResourceLimits { limits: [AtomicRlimit; litebox_common_linux::RlimitResource::RLIM_NLIMITS], } +/// `RLIMIT_NPROC` and `RLIMIT_SIGPENDING` default. Linux fills both in at boot from +/// `max_threads / 2`, which lands around here on an ordinary machine; neither is enforced. +const RLIMIT_NPROC_DEFAULT: usize = 16384; +/// `RLIMIT_MEMLOCK` default: Linux's `MLOCK_LIMIT`, 8 MiB. +const RLIMIT_MEMLOCK_DEFAULT: usize = 8 * 1024 * 1024; +/// `RLIMIT_MSGQUEUE` default: Linux's `MQ_BYTES_MAX`. +const RLIMIT_MSGQUEUE_DEFAULT: usize = 819_200; +const RLIM_INFINITY: usize = litebox_common_linux::rlim_t::MAX; + impl ResourceLimits { + /// Linux's `INIT_RLIMITS`, plus the boot-time `NPROC`/`SIGPENDING` fill-in. Every resource + /// is stored and reported; only `NOFILE` (descriptor table) and `SIGPENDING` (signal queue) + /// are actually charged anywhere. const fn default() -> Self { + use litebox_common_linux::RlimitResource as R; seq_macro::seq!(N in 0..16 { let mut limits = [ #( - AtomicRlimit::new(0, 0), + AtomicRlimit::new(RLIM_INFINITY, RLIM_INFINITY), )* ]; }); - limits[litebox_common_linux::RlimitResource::NOFILE as usize] = AtomicRlimit { - cur: core::sync::atomic::AtomicUsize::new(RLIMIT_NOFILE_CUR), - max: core::sync::atomic::AtomicUsize::new(RLIMIT_NOFILE_MAX), - }; - limits[litebox_common_linux::RlimitResource::STACK as usize] = AtomicRlimit { - cur: core::sync::atomic::AtomicUsize::new(crate::loader::DEFAULT_STACK_SIZE), - max: core::sync::atomic::AtomicUsize::new(litebox_common_linux::rlim_t::MAX), - }; + limits[R::STACK as usize] = + AtomicRlimit::new(crate::loader::DEFAULT_STACK_SIZE, RLIM_INFINITY); + limits[R::CORE as usize] = AtomicRlimit::new(0, RLIM_INFINITY); + limits[R::NPROC as usize] = AtomicRlimit::new(RLIMIT_NPROC_DEFAULT, RLIMIT_NPROC_DEFAULT); + limits[R::NOFILE as usize] = + AtomicRlimit::new(RLIMIT_NOFILE_SOFT_DEFAULT, RLIMIT_NOFILE_MAX); + limits[R::MEMLOCK as usize] = + AtomicRlimit::new(RLIMIT_MEMLOCK_DEFAULT, RLIMIT_MEMLOCK_DEFAULT); + limits[R::SIGPENDING as usize] = + AtomicRlimit::new(RLIMIT_NPROC_DEFAULT, RLIMIT_NPROC_DEFAULT); + limits[R::MSGQUEUE as usize] = + AtomicRlimit::new(RLIMIT_MSGQUEUE_DEFAULT, RLIMIT_MSGQUEUE_DEFAULT); + limits[R::NICE as usize] = AtomicRlimit::new(0, 0); + limits[R::RTPRIO as usize] = AtomicRlimit::new(0, 0); Self { limits } } + /// Copies every limit from `parent`: `fork` inherits them, and a shell's `ulimit` is only + /// ever observed by the children it then spawns. + fn inherit_from(&self, parent: &Self) { + for (mine, theirs) in self.limits.iter().zip(&parent.limits) { + mine.cur + .store(theirs.cur.load(Ordering::Relaxed), Ordering::Relaxed); + mine.max + .store(theirs.max.load(Ordering::Relaxed), Ordering::Relaxed); + } + } + pub(crate) fn get_rlimit( &self, resource: litebox_common_linux::RlimitResource, @@ -824,16 +4505,8 @@ impl Task { resource: litebox_common_linux::RlimitResource, new_limit: Option, ) -> Result { - let old_rlimit = match resource { - litebox_common_linux::RlimitResource::NOFILE - | litebox_common_linux::RlimitResource::STACK => { - self.thread.process.limits.get_rlimit(resource) - } - _ => { - log_unsupported!("Unsupported resource for get_rlimit: {:?}", resource); - return Err(Errno::EINVAL); - } - }; + let limits = &self.thread.process.limits; + let old_rlimit = limits.get_rlimit(resource); if let Some(new_limit) = new_limit { if new_limit.rlim_cur > new_limit.rlim_max { return Err(Errno::EINVAL); @@ -848,13 +4521,10 @@ impl Task { if new_limit.rlim_max > old_rlimit.rlim_max { return Err(Errno::EPERM); } - match resource { - litebox_common_linux::RlimitResource::NOFILE => { - let new_max_fd = new_limit.rlim_cur.saturating_sub(1); - self.thread.process.limits.set_rlimit(resource, new_limit); - self.files.borrow().set_max_fd(new_max_fd); - } - _ => unimplemented!("Unsupported resource for set_rlimit: {:?}", resource), + let new_max_fd = new_limit.rlim_cur.saturating_sub(1); + limits.set_rlimit(resource, new_limit); + if let litebox_common_linux::RlimitResource::NOFILE = resource { + self.files.borrow().set_max_fd(new_max_fd); } } Ok(old_rlimit) @@ -871,12 +4541,12 @@ impl Task { new_rlim: Option>, old_rlim: Option>, ) -> Result<(), Errno> { - if pid != 0 { + if pid != 0 && pid != self.pid { unimplemented!("prlimit for a specific PID is not supported yet"); } let new_limit = match new_rlim { Some(rlim) => { - let rlim = rlim.read_at_offset::(0).ok_or(Errno::EINVAL)?; + let rlim = rlim.read_at_offset::(0).ok_or(Errno::EFAULT)?; Some(litebox_common_linux::rlimit64_to_rlimit(rlim)) } None => None, @@ -886,7 +4556,7 @@ impl Task { if let Some(old_rlim) = old_rlim { old_rlim .write_at_offset::(0, old_limit) - .ok_or(Errno::EINVAL)?; + .ok_or(Errno::EFAULT)?; } Ok(()) } @@ -899,7 +4569,7 @@ impl Task { ) -> Result<(), Errno> { let old_limit = self.do_prlimit(resource, None)?; rlim.write_at_offset::(0, old_limit) - .ok_or(Errno::EINVAL) + .ok_or(Errno::EFAULT) } /// Handle syscall `setrlimit`. @@ -925,7 +4595,7 @@ impl Task { pid: Option, head_ptr: UserPtrMut, ) -> Result<(), Errno> { - if pid.is_some() { + if pid.is_some_and(|pid| pid != self.tid) { unimplemented!("Getting robust list for a specific PID is not supported yet"); } let head = self @@ -938,7 +4608,7 @@ impl Task { .ok_or(Errno::EFAULT) } - fn real_time_as_duration_since_epoch(&self) -> core::time::Duration { + pub(crate) fn real_time_as_duration_since_epoch(&self) -> core::time::Duration { let now = self.global.platform.current_time(); let unix_epoch = ::SystemTime::UNIX_EPOCH; now.duration_since(&unix_epoch) @@ -964,22 +4634,40 @@ impl Task { // CLOCK_REALTIME self.real_time_as_duration_since_epoch() } - litebox_common_linux::ClockId::Monotonic => { - // CLOCK_MONOTONIC - self.global - .platform - .now() - .duration_since(&self.global.boot_time) + litebox_common_linux::ClockId::RealTimeCoarse => { + // CLOCK_REALTIME_COARSE - a faster, lower-resolution CLOCK_REALTIME. + // Simplification: we have no cheaper coarse clock source, so we reuse the exact + // same (full-precision) value as CLOCK_REALTIME; see `sys_clock_getres` for the + // (still coarse) resolution we report for this clock. + self.real_time_as_duration_since_epoch() } - litebox_common_linux::ClockId::MonotonicCoarse => { - // CLOCK_MONOTONIC_COARSE - provides faster but less precise monotonic time - // For simplicity, we can reuse the same monotonic time as CLOCK_MONOTONIC - // In a real implementation, this would typically have lower resolution + litebox_common_linux::ClockId::Monotonic + | litebox_common_linux::ClockId::MonotonicCoarse + | litebox_common_linux::ClockId::MonotonicRaw + | litebox_common_linux::ClockId::Boottime => { + // CLOCK_MONOTONIC / CLOCK_MONOTONIC_COARSE / CLOCK_MONOTONIC_RAW / + // CLOCK_BOOTTIME. + // + // Simplification: LiteBox tracks only a single monotonic clock, so all four map + // onto it. This is exact for CLOCK_MONOTONIC; for the others it elides real + // Linux's distinctions (COARSE trades precision for speed; RAW excludes NTP + // slewing; BOOTTIME additionally counts suspend time) -- see the `ClockId` + // variant docs for why each is a legitimate simplification here. self.global .platform .now() .duration_since(&self.global.boot_time) } + litebox_common_linux::ClockId::ProcessCpuTime => { + // CLOCK_PROCESS_CPUTIME_ID - genuine per-process CPU-time accounting, sourced + // from the host (not wall-clock time). + self.global.platform.process_cpu_time() + } + litebox_common_linux::ClockId::ThreadCpuTime => { + // CLOCK_THREAD_CPUTIME_ID - genuine per-thread CPU-time accounting, sourced from + // the host (not wall-clock time). + self.global.platform.thread_cpu_time() + } _ => { log_unsupported!("gettime for {clockid:?}"); return Err(Errno::EINVAL); @@ -1001,7 +4689,9 @@ impl Task { ) -> Result::Instant>, Errno> { match clock_id { litebox_common_linux::ClockId::Monotonic - | litebox_common_linux::ClockId::MonotonicCoarse => { + | litebox_common_linux::ClockId::MonotonicCoarse + | litebox_common_linux::ClockId::MonotonicRaw + | litebox_common_linux::ClockId::Boottime => { // No need to compute the current time since the offset from the // request to `Instant` is known. Ok(self.global.boot_time.checked_add(duration)) @@ -1027,16 +4717,27 @@ impl Task { ) -> Result<(), Errno> { // Return the resolution of the clock let resolution = match clockid { - litebox_common_linux::ClockId::MonotonicCoarse => { - // Coarse clocks typically have lower resolution (e.g., 4 millisecond) + litebox_common_linux::ClockId::MonotonicCoarse + | litebox_common_linux::ClockId::RealTimeCoarse => { + // Coarse clocks typically have lower resolution (e.g., 4 millisecond). We report + // this even though we actually source these from the full-precision clock (see + // `gettime_as_duration`), matching the resolution real coarse clocks advertise. Duration::from_millis(4) } - litebox_common_linux::ClockId::RealTime | litebox_common_linux::ClockId::Monotonic => { + litebox_common_linux::ClockId::RealTime + | litebox_common_linux::ClockId::Monotonic + | litebox_common_linux::ClockId::MonotonicRaw + | litebox_common_linux::ClockId::Boottime + | litebox_common_linux::ClockId::ProcessCpuTime + | litebox_common_linux::ClockId::ThreadCpuTime => { // For most modern systems, the resolution is typically 1 nanosecond // This is a reasonable default for high-resolution timers Duration::from_nanos(1) } - _ => unimplemented!(), + // `ClockId` is `#[non_exhaustive]` but only declares the variants matched above; + // `clockid` only reaches here via `ClockId::try_from`, which rejects anything else + // with `EINVAL` before construction. + _ => unreachable!(), }; res.write::(resolution) @@ -1050,6 +4751,16 @@ impl Task { request: TimeParam, remain: TimeParam, ) -> Result<(), Errno> { + if matches!( + clockid, + litebox_common_linux::ClockId::ProcessCpuTime + | litebox_common_linux::ClockId::ThreadCpuTime + ) { + // Real Linux rejects sleeping against a CPU-time clock: a blocked (not-running) + // thread cannot accumulate CPU time, so waiting for one of these clocks to reach a + // given value could never wake up. + return Err(Errno::EINVAL); + } let request = request.read::()?.ok_or(Errno::EFAULT)?; if flags.intersects(litebox_common_linux::TimerFlags::ABSTIME.complement()) { return Err(Errno::EINVAL); @@ -1147,7 +4858,7 @@ impl Task { let new_deadline = if delay.is_zero() { None } else { - Some(now.checked_add(delay).ok_or(Errno::EINVAL)?) + Some(now.checked_add(delay).ok_or(Errno::EFAULT)?) }; if alarm.handle.is_none() { match self @@ -1157,7 +4868,9 @@ impl Task { { Ok(handle) => alarm.handle = Some(handle), Err(litebox::platform::TimerCreationError::Unsupported) => {} - Err(_) => unimplemented!(), + // `TimerCreationError` is `#[non_exhaustive]` but only declares this one + // variant, already matched above. + Err(_) => unreachable!(), } } if let Some(handle) = &alarm.handle { @@ -1247,24 +4960,294 @@ impl Task { self.ppid } + /// Resolves `pid`, as passed to `setpgid`/`getpgid`, to the process-group identity of the + /// calling process or one of its live children. + fn pgid_target(&self, pid: i32) -> Result, Errno> { + if pid == 0 || pid == self.pid { + return Ok(self.process().process_group_id.clone()); + } + let Some(target) = self + .global + .processes + .live_child_process_group_id(self.pid, pid) + else { + log_unsupported!("setpgid/getpgid for a pid that is not a live child"); + return Err(Errno::ESRCH); + }; + Ok(target) + } + + /// Handle syscall `setsid`. + /// + /// A process-group leader cannot create a new session. On success the caller becomes both the + /// session and process-group leader and loses any controlling terminal, matching Linux. + pub(crate) fn sys_setsid(&self) -> Result { + let process = self.process(); + if process.process_group_id() == self.pid { + return Err(Errno::EPERM); + } + process.session_id.store(self.pid, Ordering::Release); + process + .controlling_pty + .store(NO_CONTROLLING_PTY, Ordering::Release); + process.process_group_id.store(self.pid, Ordering::Release); + Ok(self.pid) + } + + /// Handle syscall `getpgid`. + /// + /// `pid == 0` means "the calling process". A parent may also query one of its live children. + pub(crate) fn sys_getpgid(&self, pid: i32) -> Result { + Ok(self.pgid_target(pid)?.load(Ordering::Acquire)) + } + + /// Handle syscall `setpgid`. + /// + /// Real Linux additionally restricts this to processes in the same session and forbids + /// retargeting a child that has already called `execve` (`EACCES`). LiteBox does not yet track + /// an exec generation per child, so only the live self/child identity check is enforced. + /// `pgid == 0` means "use the target's own pid", matching Linux. + #[allow(clippy::similar_names)] + pub(crate) fn sys_setpgid(&self, pid: i32, pgid: i32) -> Result<(), Errno> { + if pgid < 0 { + return Err(Errno::EINVAL); + } + let target_pid = if pid == 0 { self.pid } else { pid }; + let target = self.pgid_target(pid)?; + let new_pgid = if pgid == 0 { target_pid } else { pgid }; + target.store(new_pgid, Ordering::Release); + Ok(()) + } + /// Handle syscall `getuid`. pub(crate) fn sys_getuid(&self) -> u32 { - self.credentials.uid + self.credentials.borrow().uid } /// Handle syscall `geteuid`. pub(crate) fn sys_geteuid(&self) -> u32 { - self.credentials.euid + self.credentials.borrow().euid } /// Handle syscall `getgid`. pub(crate) fn sys_getgid(&self) -> u32 { - self.credentials.gid + self.credentials.borrow().gid } /// Handle syscall `getegid`. pub(crate) fn sys_getegid(&self) -> u32 { - self.credentials.egid + self.credentials.borrow().egid + } + + /// Whether this task may change its uid/gid to an arbitrary value. + /// + /// LiteBox models no capability set, so `CAP_SETUID`/`CAP_SETGID` have + /// nothing to check. An effective uid of 0 is used as the stand-in, + /// mirroring the classic pre-capabilities Unix kernel, which gated the + /// same operations on `suser()` (effective uid 0) alone. + /// Whether this task may perform a privileged identity change (`setuid` + /// family, `setgroups`). + /// + /// Real Linux gates these on holding `CAP_SETUID`/`CAP_SETGID` in the + /// effective capability set, not literally on `euid == 0` -- the two + /// usually coincide (root normally holds every capability), but they + /// diverge exactly when a process has called `PR_SET_KEEPCAPS` before + /// dropping its uid away from 0: without that flag the kernel would + /// clear the permitted set on the uid change, but with it the + /// capabilities survive, so a later `setresgid`/`setresuid` from the + /// now-unprivileged-looking euid still succeeds. This is the standard + /// sequence `setpriv --reuid --regid` uses (`PR_SET_KEEPCAPS(1)` -> + /// `capset` -> `setresuid` -> `setresgid`), and LiteBox does not model + /// individual capability bits at all (`CapBSetRead`/`capget` always + /// report an empty set) -- the same coarse stance extended here: once + /// `keep_caps` is set, treat this task as retaining root's implicit + /// authority for these calls, mirroring what a real kernel would do for + /// a process that actually held (and kept) `CAP_SETUID`/`CAP_SETGID`. + fn is_privileged(&self) -> bool { + let credentials = self.credentials.borrow(); + credentials.euid == 0 || credentials.keep_caps() + } + + /// Install `new` as this task's credentials with the side effects Linux's `commit_creds` + /// (`kernel/cred.c`) attaches to a change of *effective* identity: the process becomes + /// non-dumpable (`suid_dumpable`'s default 0) and loses its parent-death signal. The kernel's + /// test is on `euid`/`egid`/`fsuid`/`fsgid` and the capability sets -- a change of only the + /// real or saved ids leaves both alone -- and LiteBox models neither `fsuid` nor capability + /// bits, so the effective ids are the whole test. This is what makes a `setuid` helper that + /// drops root (`doas`, `chrome-sandbox`) read back `PR_GET_DUMPABLE == 0` afterwards, as on + /// Linux, instead of the `1` an unrelated earlier exec left behind. + fn commit_credentials(&self, new: Credentials) { + let (old_euid, old_egid) = { + let old = self.credentials.borrow(); + (old.euid, old.egid) + }; + let identity_changed = old_euid != new.euid || old_egid != new.egid; + let (new_euid, new_egid) = (new.euid, new.egid); + *self.credentials.borrow_mut() = Arc::new(new); + if identity_changed { + litebox_util_log::debug!( + pid:? = self.pid, old_euid:? = old_euid, new_euid:? = new_euid, + old_egid:? = old_egid, new_egid:? = new_egid; + "effective identity changed: process is no longer dumpable, parent-death signal cleared" + ); + self.process().set_dumpable(false); + self.thread.parent_death_signal.set(None); + self.global.processes.set_parent_death_signal(self.pid, None); + } + } + + /// Handle syscall `setuid`. + /// + /// A privileged task sets its real, effective, and saved user IDs together. An + /// unprivileged task may only select its real or saved ID as the new effective ID. + pub(crate) fn sys_setuid(&self, uid: u32) -> Result<(), Errno> { + let old = self.credentials.borrow().clone(); + let mut new = old.as_ref().clone(); + if old.euid == 0 { + new.uid = uid; + new.euid = uid; + new.suid = uid; + } else if uid == old.uid || uid == old.suid { + new.euid = uid; + } else { + return Err(Errno::EPERM); + } + self.commit_credentials(new); + Ok(()) + } + + /// Handle syscall `setgid`; see [`Self::sys_setuid`] for the analogous user-ID rules. + pub(crate) fn sys_setgid(&self, gid: u32) -> Result<(), Errno> { + let old = self.credentials.borrow().clone(); + let mut new = old.as_ref().clone(); + if old.euid == 0 { + new.gid = gid; + new.egid = gid; + new.sgid = gid; + } else if gid == old.gid || gid == old.sgid { + new.egid = gid; + } else { + return Err(Errno::EPERM); + } + self.commit_credentials(new); + Ok(()) + } + + /// Handle syscall `setresuid`. `u32::MAX` leaves the corresponding field unchanged. + pub(crate) fn sys_setresuid(&self, ruid: u32, euid: u32, suid: u32) -> Result<(), Errno> { + let old = self.credentials.borrow().clone(); + let privileged = self.is_privileged(); + let allowed = |value: u32| { + privileged || value == old.uid || value == old.euid || value == old.suid + }; + for value in [ruid, euid, suid] { + if value != u32::MAX && !allowed(value) { + return Err(Errno::EPERM); + } + } + + let mut new = old.as_ref().clone(); + if ruid != u32::MAX { + new.uid = ruid; + } + if euid != u32::MAX { + new.euid = euid; + } + if suid != u32::MAX { + new.suid = suid; + } + self.commit_credentials(new); + Ok(()) + } + + /// Handle syscall `setresgid`; see [`Self::sys_setresuid`], with group IDs. + pub(crate) fn sys_setresgid(&self, rgid: u32, egid: u32, sgid: u32) -> Result<(), Errno> { + let old = self.credentials.borrow().clone(); + let privileged = self.is_privileged(); + let allowed = |value: u32| { + privileged || value == old.gid || value == old.egid || value == old.sgid + }; + for value in [rgid, egid, sgid] { + if value != u32::MAX && !allowed(value) { + return Err(Errno::EPERM); + } + } + + let mut new = old.as_ref().clone(); + if rgid != u32::MAX { + new.gid = rgid; + } + if egid != u32::MAX { + new.egid = egid; + } + if sgid != u32::MAX { + new.sgid = sgid; + } + self.commit_credentials(new); + Ok(()) + } + + /// Handle syscall `getresuid`. + pub(crate) fn sys_getresuid( + &self, + ruid: UserPtrMut, + euid: UserPtrMut, + suid: UserPtrMut, + ) -> Result<(), Errno> { + let credentials = self.credentials.borrow(); + ruid.write_at_offset::(0, credentials.uid) + .ok_or(Errno::EFAULT)?; + euid.write_at_offset::(0, credentials.euid) + .ok_or(Errno::EFAULT)?; + suid.write_at_offset::(0, credentials.suid) + .ok_or(Errno::EFAULT)?; + Ok(()) + } + + /// Handle syscall `getresgid`; see [`Self::sys_getresuid`], with group IDs. + pub(crate) fn sys_getresgid( + &self, + rgid: UserPtrMut, + egid: UserPtrMut, + sgid: UserPtrMut, + ) -> Result<(), Errno> { + let credentials = self.credentials.borrow(); + rgid.write_at_offset::(0, credentials.gid) + .ok_or(Errno::EFAULT)?; + egid.write_at_offset::(0, credentials.egid) + .ok_or(Errno::EFAULT)?; + sgid.write_at_offset::(0, credentials.sgid) + .ok_or(Errno::EFAULT)?; + Ok(()) + } + + pub(crate) fn sys_getgroups(&self, size: i32, list: UserPtrMut) -> Result { + if size < 0 { + return Err(Errno::EINVAL); + } + let size = usize::try_from(size).map_err(|_| Errno::EINVAL)?; + let credentials = self.credentials.borrow(); + let groups = credentials.supplementary_groups.as_slice(); + if size == 0 { + return Ok(groups.len()); + } + if size < groups.len() { + return Err(Errno::EINVAL); + } + list.write_slice_at_offset::(0, groups) + .ok_or(Errno::EFAULT)?; + Ok(groups.len()) + } + + pub(crate) fn sys_setgroups(&self, size: usize, list: UserPtr) -> Result<(), Errno> { + if !self.is_privileged() { + return Err(Errno::EPERM); + } + let supplementary_groups = SupplementaryGroups::from_user::(size, list)?; + let mut new = self.credentials.borrow().as_ref().clone(); + new.supplementary_groups = supplementary_groups; + *self.credentials.borrow_mut() = Arc::new(new); + Ok(()) } } @@ -1285,6 +5268,130 @@ impl CpuSet { } impl Task { + /// Resolves `which`/`who` of `getpriority`/`setpriority` to the threads it names, each with + /// the real uid of its process (what `set_one_prio_perm` compares against). `PRIO_PROCESS` + /// names one thread (`who == 0` is the caller; a tid selects that thread, in this or any + /// live process); `PRIO_PGRP` every thread of every process in the group (`who == 0`: the + /// caller's group); `PRIO_USER` every thread of every process with that real uid (`who == + /// 0`: the caller's). Returns `EINVAL` for an unknown `which`, `ESRCH` when nothing matched. + fn priority_targets( + &self, + which: i32, + who: i32, + ) -> Result>, u32)>, Errno> { + const PRIO_PROCESS: i32 = 0; + const PRIO_PGRP: i32 = 1; + const PRIO_USER: i32 = 2; + let own_uid = self.credentials.borrow().uid; + let mut out = Vec::new(); + match which { + PRIO_PROCESS => { + let who = if who == 0 { self.tid } else { who }; + if who == self.tid || self.process().thread_remote(who).is_some() { + let remote = self.process().thread_remote(who).unwrap_or_else(|| self.thread_remote().clone()); + out.push((remote, own_uid)); + } else if let Some((threads, _, uid)) = self.global.processes.priority_targets(who) { + // A tid in another process: Linux resolves `who` as a tid (`find_task_by_vpid`), + // and the table is keyed by pid = leader tid, so this selects that leader. + if let Some(leader) = threads.into_iter().next() { + out.push((leader, uid)); + } + } + } + PRIO_PGRP => { + let group = if who == 0 { self.process().process_group_id() } else { who }; + for pid in self.global.processes.live_pids() { + if let Some((threads, pgid, uid)) = self.global.processes.priority_targets(pid) + && pgid == group + { + out.extend(threads.into_iter().map(|t| (t, uid))); + } + } + } + PRIO_USER => { + let target = if who == 0 { own_uid } else { who.cast_unsigned() }; + for pid in self.global.processes.live_pids() { + if let Some((threads, _, uid)) = self.global.processes.priority_targets(pid) + && uid == target + { + out.extend(threads.into_iter().map(|t| (t, uid))); + } + } + } + _ => return Err(Errno::EINVAL), + } + if out.is_empty() { + return Err(Errno::ESRCH); + } + Ok(out) + } + + /// Handle syscall `getpriority`. + /// + /// Returns the *highest* priority (lowest nice) among the matched threads, encoded as Linux's + /// syscall does -- `20 - nice`, so `1..=40` -- for the libc wrapper to turn back into a + /// nice value. + pub(crate) fn sys_getpriority(&self, which: i32, who: i32) -> Result { + let targets = self.priority_targets(which, who)?; + let lowest_nice = targets + .iter() + .map(|(thread, _)| thread.nice()) + .min() + .unwrap_or(0); + Ok((20 - lowest_nice).cast_unsigned() as usize) + } + + /// Handle syscall `setpriority`. + /// + /// `niceval` is clamped to `-20..=19`. Linux's `set_one_prio` rules: a target owned by a + /// different real uid needs the caller to be privileged (`EPERM`); lowering a nice value + /// (raising priority) needs `CAP_SYS_NICE` or an `RLIMIT_NICE` that admits it (`EACCES`); + /// raising nice is always allowed. One matched thread in error does not stop the others + /// (the last error is reported after all are tried), as in the kernel's loop. + pub(crate) fn sys_setpriority(&self, which: i32, who: i32, niceval: i32) -> Result<(), Errno> { + let niceval = niceval.clamp(-20, 19); + let targets = self.priority_targets(which, who)?; + let privileged = self.is_privileged(); + let own_euid = self.credentials.borrow().euid; + let own_uid = self.credentials.borrow().uid; + // `nice_to_rlimit`: a nice of `n` needs `RLIMIT_NICE >= 20 - n`. + let nice_rlim = (20 - niceval).cast_unsigned() as usize; + let rlimit_nice = self + .process() + .limits + .get_rlimit_cur(litebox_common_linux::RlimitResource::NICE); + let mut error = None; + for (thread, target_uid) in targets { + if !privileged && target_uid != own_uid && target_uid != own_euid { + error = Some(Errno::EPERM); + continue; + } + if niceval < thread.nice() && !privileged && nice_rlim > rlimit_nice { + error = Some(Errno::EACCES); + continue; + } + thread.set_nice(niceval); + } + error.map_or(Ok(()), Err) + } + + /// Handle syscall `membarrier`. + /// + /// Answers as a kernel built with `CONFIG_MEMBARRIER=n` does: `ENOSYS` for every command, + /// `MEMBARRIER_CMD_QUERY` included. A membarrier's guarantee is that every *other* thread + /// of the process has executed a full barrier before the call returns; the host (no + /// `membarrier(2)` on macOS, guest threads running on vCPU lanes) has no primitive for that + /// yet, and a command that returned `0` without providing it would be a lie a lock-free + /// algorithm could act on. Callers must already handle `ENOSYS` (pre-4.3 kernels, or this + /// config). Decoded rather than left to the unknown-syscall path so the log names the + /// command the guest asked for. + pub(crate) fn sys_membarrier(&self, cmd: i32, flags: u32, cpu_id: i32) -> Result { + log_unsupported!( + "membarrier(cmd = {cmd:#x}, flags = {flags:#x}, cpu_id = {cpu_id}): no cross-thread barrier primitive -> ENOSYS" + ); + Err(Errno::ENOSYS) + } + /// Handle syscall `sched_getaffinity`. /// /// Note this is a dummy implementation that always returns the same CPU set @@ -1293,30 +5400,164 @@ impl Task { cpuset.iter_mut().for_each(|mut b| *b = true); CpuSet { bits: cpuset } } + + /// Returns whether `pid`, as passed to one of the `sched_*` syscalls below, refers to the + /// calling thread. `pid == 0` (as with all four `sched_*` syscalls per their man pages) means + /// "the calling thread"; `sched_*` operates at thread (not process) granularity on Linux, so + /// this compares against `self.tid`, not a process-wide id. + fn sched_target_is_self(&self, pid: Option) -> bool { + pid.is_none_or(|pid| pid == self.tid) + } + + /// Handle syscall `sched_getparam`. + /// + /// LiteBox's process model has no real scheduling-class enforcement to expose, so every + /// thread is always reported as `SCHED_OTHER` with priority 0 -- the same default every + /// unprivileged Linux thread starts with, and the only priority `SCHED_OTHER` ever accepts. + pub(crate) fn sys_sched_getparam( + &self, + pid: Option, + param: UserPtrMut, + ) -> Result { + if !self.sched_target_is_self(pid) { + log_unsupported!("sched_getparam for a remote pid"); + return Err(Errno::ESRCH); + } + param + .write_at_offset::(0, litebox_common_linux::SchedParam { sched_priority: 0 }) + .ok_or(Errno::EFAULT)?; + Ok(0) + } + + /// Handle syscall `sched_setparam`. + /// + /// Since every thread is always `SCHED_OTHER` (see [`Self::sys_sched_getparam`]), and + /// `SCHED_OTHER`'s only valid priority is 0, this accepts a priority-0 request as a no-op and + /// rejects anything else with `EINVAL`, matching what real Linux would do to a process that + /// never leaves `SCHED_OTHER`. + pub(crate) fn sys_sched_setparam( + &self, + pid: Option, + param: UserPtr, + ) -> Result { + if !self.sched_target_is_self(pid) { + log_unsupported!("sched_setparam for a remote pid"); + return Err(Errno::ESRCH); + } + let param = param.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if param.sched_priority != 0 { + return Err(Errno::EINVAL); + } + Ok(0) + } + + /// Handle syscall `sched_getscheduler`. + pub(crate) fn sys_sched_getscheduler(&self, pid: Option) -> Result { + if !self.sched_target_is_self(pid) { + log_unsupported!("sched_getscheduler for a remote pid"); + return Err(Errno::ESRCH); + } + // The return value of `sched_getscheduler` IS the policy (unlike most syscalls, it is + // not a separate out-parameter), so no bitwise cast/sign issues arise turning a small + // non-negative `i32` constant into a `usize` success value. + Ok(usize::try_from(litebox_common_linux::sched_policy::SCHED_OTHER).unwrap()) + } + + /// Handle syscall `sched_setscheduler`. + /// + /// Non-real-time policies (`SCHED_OTHER`/`SCHED_BATCH`/`SCHED_IDLE`) are accepted as no-ops, + /// same as a real unprivileged Linux process switching between them would experience. + /// Real-time policies (`SCHED_FIFO`/`SCHED_RR`/`SCHED_DEADLINE`) are rejected with `EPERM`, + /// matching real Linux's behavior for a process without `CAP_SYS_NICE` -- a real, accurate + /// constraint here, since LiteBox guests never have that capability, not a shortcut. + pub(crate) fn sys_sched_setscheduler( + &self, + pid: Option, + policy: i32, + param: UserPtr, + ) -> Result { + use litebox_common_linux::sched_policy::{ + SCHED_BATCH, SCHED_DEADLINE, SCHED_FIFO, SCHED_IDLE, SCHED_OTHER, SCHED_RESET_ON_FORK, + SCHED_RR, + }; + + if !self.sched_target_is_self(pid) { + log_unsupported!("sched_setscheduler for a remote pid"); + return Err(Errno::ESRCH); + } + match policy & !SCHED_RESET_ON_FORK { + SCHED_OTHER | SCHED_BATCH | SCHED_IDLE => {} + SCHED_FIFO | SCHED_RR | SCHED_DEADLINE => { + log_unsupported!( + "sched_setscheduler(policy = {policy}): real-time scheduling is never available to a LiteBox guest" + ); + return Err(Errno::EPERM); + } + _ => return Err(Errno::EINVAL), + } + let param = param.read_at_offset::(0).ok_or(Errno::EFAULT)?; + if param.sched_priority != 0 { + return Err(Errno::EINVAL); + } + Ok(0) + } } impl Task { - /// Handle syscall `futex` - pub(crate) fn sys_futex(&self, arg: litebox_common_linux::FutexArgs) -> Result { - /// Note our mutex implementation assumes futexes are private as we don't support shared memory yet. - /// It should be fine to treat shared futexes as private for now. - macro_rules! warn_shared_futex { - ($flag:ident) => { - if !$flag.contains(litebox_common_linux::FutexFlags::PRIVATE) { - log_unsupported!("shared futex"); - } - }; + fn futex_key( + &self, + mappings: &litebox::mm::MappingReadGuard<'_, Platform, PAGE_SIZE>, + addr: UserPtrMut, + flags: &litebox_common_linux::FutexFlags, + ) -> Result { + if !addr.as_usize().is_multiple_of(align_of::()) { + return Err(Errno::EINVAL); } + let key = if flags.contains(litebox_common_linux::FutexFlags::PRIVATE) { + FutexKey::new(self.process().futex_namespace(), addr.as_usize()) + } else { + let mapping = mappings.flags_at(addr.as_usize()).ok_or(Errno::EFAULT)?; + if mapping.contains(VmFlags::VM_SHARED) { + let (backing, offset) = mappings + .shared_futex_key_at(addr.as_usize()) + .ok_or(Errno::EFAULT)?; + FutexKey::new_shared(backing, offset) + } else { + FutexKey::new(self.process().futex_namespace(), addr.as_usize()) + } + }; + Ok(key) + } + /// Handle syscall `futex` + pub(crate) fn sys_futex(&self, arg: litebox_common_linux::FutexArgs) -> Result { let res = match arg { FutexArgs::Wake { addr, flags, count } => { - warn_shared_futex!(flags); - let Some(count) = core::num::NonZeroU32::new(count) else { - return Ok(0); - }; - self.global - .futex_manager - .wake(addr.to_platform_ptr::(), count, None)? as usize + // Linux's traditional FUTEX_WAKE takes a signed `int`. Its queue loop wakes one + // waiter before testing whether the count has been reached, so zero and negative + // raw values both mean one wake rather than zero or an enormous unsigned quota. + let count = if (count as i32) <= 0 { 1 } else { count }; + let count = core::num::NonZeroU32::new(count).unwrap(); + let mappings = self.global.pm.lock_mappings(); + let key = self.futex_key(&mappings, addr, &flags)?; + self.process() + .futex_manager() + .wake_keyed(key, count, None)? as usize + } + FutexArgs::WakeBitset { + addr, + flags, + count, + bitmask, + } => { + let count = if (count as i32) <= 0 { 1 } else { count }; + let count = core::num::NonZeroU32::new(count).unwrap(); + let bitmask = core::num::NonZeroU32::new(bitmask).ok_or(Errno::EFAULT)?; + let mappings = self.global.pm.lock_mappings(); + let key = self.futex_key(&mappings, addr, &flags)?; + self.process() + .futex_manager() + .wake_keyed(key, count, Some(bitmask))? as usize } FutexArgs::Wait { addr, @@ -1324,13 +5565,16 @@ impl Task { val, timeout, } => { - warn_shared_futex!(flags); let timeout = timeout.read::()?; - self.global.futex_manager.wait( + let mappings = self.global.pm.lock_mappings(); + let key = self.futex_key(&mappings, addr, &flags)?; + self.process().futex_manager().wait_keyed( &self.wait_cx().with_timeout(timeout), + key, addr.to_platform_ptr::(), val, None, + || drop(mappings), )?; 0 } @@ -1341,7 +5585,7 @@ impl Task { timeout, bitmask, } => { - warn_shared_futex!(flags); + let bitmask = core::num::NonZeroU32::new(bitmask).ok_or(Errno::EFAULT)?; let deadline = if let Some(timeout) = timeout.read::()? { let clock_id = if flags.contains(litebox_common_linux::FutexFlags::CLOCK_REALTIME) { @@ -1353,15 +5597,61 @@ impl Task { } else { None }; - self.global.futex_manager.wait( + let mappings = self.global.pm.lock_mappings(); + let key = self.futex_key(&mappings, addr, &flags)?; + self.process().futex_manager().wait_keyed( &self.wait_cx().with_deadline(deadline), + key, + addr.to_platform_ptr::(), + val, + Some(bitmask), + || drop(mappings), + )?; + 0 + } + litebox_common_linux::FutexArgs::Requeue { + addr, + flags, + num_to_wake, + num_to_requeue, + addr2, + } => { + let mappings = self.global.pm.lock_mappings(); + let key1 = self.futex_key(&mappings, addr, &flags)?; + let key2 = self.futex_key(&mappings, addr2, &flags)?; + self.process().futex_manager().requeue_keyed( + key1, + key2, + addr.to_platform_ptr::(), + num_to_wake, + num_to_requeue, + None, + )? as usize + } + litebox_common_linux::FutexArgs::CmpRequeue { + addr, + flags, + num_to_wake, + num_to_requeue, + addr2, + expected_value, + } => { + let mappings = self.global.pm.lock_mappings(); + let key1 = self.futex_key(&mappings, addr, &flags)?; + let key2 = self.futex_key(&mappings, addr2, &flags)?; + self.process().futex_manager().requeue_keyed( + key1, + key2, addr.to_platform_ptr::(), - val, - core::num::NonZeroU32::new(bitmask), - )?; - 0 + num_to_wake, + num_to_requeue, + Some(expected_value), + )? as usize + } + _ => { + log_unsupported!("futex operation {:?}", arg); + return Err(Errno::ENOSYS); } - _ => unimplemented!("Unsupported futex operation"), }; Ok(res) } @@ -1371,7 +5661,7 @@ const MAX_VEC: usize = 4096; // limit count const MAX_TOTAL_BYTES: usize = 256 * 1024; // size cap /// Maximum shebang (#!) recursion depth (from Linux's `exec_binprm`) -const SHEBANG_MAX_RECURSION: u32 = 6; +const SHEBANG_MAX_RECURSION: u32 = 4; /// Maximum length of a shebang line that we inspect. Matches Linux `BINPRM_BUF_SIZE`. const SHEBANG_MAX_LINE: usize = 256; @@ -1405,33 +5695,39 @@ fn parse_shebang(buf: &[u8]) -> Option<(&str, Option<&str>)> { } impl Task { - /// Resolve shebang (`#!`) chains for the given path and argv if the file starts with a shebang line. - /// Otherwise, returns the original path and argv. + /// Resolve shebang (`#!`) chains for the given path and argv. + /// + /// Every probe follows symlinks. A script still contributes the spelling used to reach it to + /// the interpreter's argv, while the returned non-script path is the final followed target the + /// ELF loader must open. pub(crate) fn resolve_shebang( &self, mut path: alloc::string::String, mut argv: alloc::vec::Vec, ) -> Result<(alloc::string::String, alloc::vec::Vec), Errno> { - for _ in 0..SHEBANG_MAX_RECURSION { + let mut recursion = 0; + let comm = path + .rsplit('/') + .next() + .unwrap_or("unknown") + .as_bytes() + .to_vec(); + loop { let full_path = self.resolve_path(&path)?; - let file = self.do_open( - full_path, - litebox::fs::OFlags::RDONLY, - litebox::fs::Mode::empty(), - )?; + let full_path = self.follow_open_path(full_path, litebox::fs::OFlags::RDONLY)?; let mut header = [0u8; SHEBANG_MAX_LINE]; - let files = self.files.borrow(); - let n = match files.fs.read(&file, &mut header, Some(0)) { - Ok(n) => n, - Err(e) => { - let _ = files.fs.close(&file); - return Err(Errno::from(e)); - } - }; - let _ = files.fs.close(&file); + let n = crate::loader::elf::read_executable_header( + self, + full_path.clone(), + &mut header, + )?; match parse_shebang(&header[..n]) { Some((interp, opt_arg)) => { + if recursion == SHEBANG_MAX_RECURSION { + return Err(Errno::ELOOP); + } + recursion += 1; let mut new_argv = alloc::vec::Vec::new(); new_argv.push(alloc::ffi::CString::new(interp).map_err(|_| Errno::EINVAL)?); if let Some(arg) = opt_arg { @@ -1445,28 +5741,65 @@ impl Task { path = alloc::string::String::from(interp); argv = new_argv; } - None => return Ok((path, argv)), + None => { + let path = full_path.into_string().map_err(|_| Errno::EINVAL)?; + // Linux's `/proc//exe` names the image actually mapped: for a `#!` + // script that is the interpreter, which is what `path` is by now. + *self.thread.staged_exec.borrow_mut() = Some(StagedExec { + exe: path.clone(), + comm, + }); + return Ok((path, argv)); + } + } + } + } + + fn credentials_for_exec( + &self, + status: &litebox::fs::FileStatus, + ) -> (Arc, bool) { + let old = self.credentials.borrow().clone(); + let mut candidate = old.as_ref().clone(); + if !old.no_new_privs() { + if status.mode.contains(litebox::fs::Mode::SUID) { + candidate.euid = u32::from(status.owner.user); + } + if status + .mode + .contains(litebox::fs::Mode::SGID | litebox::fs::Mode::XGRP) + { + candidate.egid = u32::from(status.owner.group); } } - Err(Errno::ELOOP) + candidate.suid = candidate.euid; + candidate.sgid = candidate.egid; + let transitioned = candidate.euid != old.euid || candidate.egid != old.egid; + let secure = transitioned + || candidate.euid != candidate.uid + || candidate.egid != candidate.gid; + (Arc::new(candidate), secure) } /// Handle syscall `execve`. + // `c_char` rather than a fixed `i8`: it is signed on x86-64 and on Apple's + // AArch64 ABI but unsigned on AArch64 Linux, and `SyscallRequest::Execve` + // hands these over as `UserPtr`. pub(crate) fn sys_execve( &self, - pathname: UserPtr, - argv: UserPtr>, - envp: UserPtr>, + pathname: UserPtr, + argv: UserPtr>, + envp: UserPtr>, ctx: &mut litebox_common_linux::PtRegs, ) -> Result { fn copy_vector( - mut base: UserPtr>, + mut base: UserPtr>, _which: &str, ) -> Result, Errno> { let mut out = alloc::vec::Vec::new(); let mut total = 0usize; for _ in 0..MAX_VEC { - let p: UserPtr = { + let p: UserPtr = { // read pointer-sized entries match base.read_at_offset::(0) { Some(ptr) => ptr, @@ -1511,6 +5844,7 @@ impl Task { let (path, argv_vec) = self.resolve_shebang(alloc::string::String::from(path), argv_vec)?; let loader = crate::loader::elf::ElfLoader::new(self, &path)?; + let (exec_credentials, secure_exec) = self.credentials_for_exec(loader.main_status()); // After this point, the old program is torn down and failures must terminate the process. @@ -1526,40 +5860,206 @@ impl Task { // unmmap all memory mappings and reset brk if let Some(robust_list) = self.thread.robust_list.take() { - let _ = wake_robust_list::(robust_list); + let _ = self.wake_robust_list(robust_list); + } + let shares_parent_vm = self.process().shares_parent_vm(); + if shares_parent_vm { + // Linux's mm_release clears and wakes this address on vfork exec while the old VM is + // still shared with the suspended parent. Do it before detaching the child's VM slot; + // retaining the pointer into the old image would instead corrupt parent memory later. + if let Some(clear_child_tid) = self.thread.clear_child_tid.take() { + let _ = clear_child_tid.write_at_offset::(0, 0); + let _ = self.sys_futex(litebox_common_linux::FutexArgs::Wake { + addr: UserPtrMut::from_usize(clear_child_tid.as_usize()), + flags: litebox_common_linux::FutexFlags::empty(), + count: 1, + }); + } + } else { + self.thread.clear_child_tid.set(None); } - self.thread.clear_child_tid.set(None); self.signals.reset_for_exec(); - // Don't release reserved mappings. - let release = |_r: Range, vm: VmFlags| !vm.is_empty(); - unsafe { self.global.pm.release_memory(release) } - .expect("failed to release memory mappings"); + if shares_parent_vm { + // The child has so far operated on the parent's live VM identity. Swap only this + // process's slot to a fresh identity before building the new image; the suspended + // parent keeps the original identity, including every pre-exec mapping and brk change. + self.process().detach_vfork_vm(); + } else if self.leave_address_space_if_alone() { + // Release only the mappings this process owns, not everything the + // (process-blind) page manager tracks. "Alone in the shared + // address space" -- or never having shared at all -- does not + // mean alone in the page manager: a forked child that already + // completed one exec has left the shared space, yet its suspended + // parent's entire live memory is still in the manager, and a + // release-everything here destroys it. Observed live as Node's + // `execSync("/bin/sh -c ...")`: fork, exec /bin/sh (first exec + // keeps the parent's memory via the branch below), sh execs the + // command (second exec took this branch and unmapped the + // suspended parent wholesale -- every one of its subsequent + // address-space restores failed and it died on the first libc + // global it touched). `owned_ranges` exists precisely to name + // which mappings are this process's, and the fork/exec paths + // maintain it for every mapping source (mmap, mremap, brk, the + // loader's stack); reserved mappings carry empty `VmFlags` and + // are skipped as before. + // + // What is released is the *intersection* with `owned_ranges`, never a whole tracked + // mapping that merely overlaps it. The page manager coalesces adjacent ranges with + // identical properties into a single entry (see `PageManager::mappings`), and + // adjacency between this process's memory and a suspended sibling's is not a + // coincidence here: `Vmem::get_unmmaped_area` hands out the address immediately below + // an existing range, so a forked child's very first anonymous `mmap` lands flush + // against whatever its parent had there. Observed live, exactly so: the + // `execSync("/bin/sh -c ...")` child `mmap`ed 16 KiB that abutted 48 KiB of its + // parent's musl heap, and (via `mprotect`) another 16 KiB that abutted 160 KiB more + // of it -- four of the parent's ranges the manager had silently merged into two of + // this process's -- and a whole-entry release then unmapped all 208 KiB of the + // parent's, which died on the first libc global it touched after taking its address + // space back. + let owned = self.process().owned_ranges.lock(); + // A live `/dev/fb0` guest mapping (see `do_mmap_framebuffer`) whose pages this + // release is about to free must be deregistered first -- the framebuffer would + // otherwise keep reading freed memory. A sibling's registration is not in this + // process's `owned_ranges` and is left alone. + if let Some(fb) = self.global.framebuffer.as_ref() + && let Some((fb_addr, fb_len)) = fb.guest_mapping() + && owned + .intersect(&(fb_addr..fb_addr.saturating_add(fb_len))) + .next() + .is_some() + { + fb.clear_guest_mapping_overlapping(fb_addr, fb_len); + } + let release = |r: Range, vm: VmFlags| { + if vm.is_empty() { + Vec::new() + } else { + owned.intersect(&r).collect::>() + } + }; + if let Err(error) = unsafe { self.global.pm.release_memory(release) } { + litebox_util_log::error!(error:? = error; "execve: failed to release old mappings"); + self.exit_group(ExitStatus::Signal( + litebox_common_linux::signal::Signal::SIGKILL, + )); + return Err(error.into()); + } + } + + // Either the old mappings are gone or (for a `fork`ed child) they were never this + // process's to begin with. `load_program` re-populates this as it maps the new image. + self.process().owned_ranges.lock().clear(); + self.process().elf_patch_cache.lock().clear(); - self.global + if let Err(error) = self + .global .platform - .set_arch_specific_register(&ArchSpecificRegister::FsBase, 0) - .expect("failed to clear guest TLS on execve"); + .set_arch_specific_register(&GUEST_TLS_REGISTER, 0) + { + litebox_util_log::error!(error:? = error; "execve: failed to clear guest TLS"); + self.exit_group(ExitStatus::Signal( + litebox_common_linux::signal::Signal::SIGKILL, + )); + return Err(Errno::EIO); + } - self.load_program(loader, argv_vec, envp_vec) - .expect("TODO: terminate the process cleanly"); + if let Err(error) = self.load_program_with_credentials( + loader, + argv_vec, + envp_vec, + exec_credentials, + secure_exec, + ) { + self.exit_group(ExitStatus::Signal( + litebox_common_linux::signal::Signal::SIGKILL, + )); + return Err(error.into()); + } self.init_thread_context(ctx); + // The new image is fully built, at addresses no other member of the old address space + // owns, so this task no longer needs the shared one. Handing it back here rather than + // earlier means no other member ever observes a half-built image. A vfork child that + // forked while on its parent's memory hands its place to the parent it is about to wake + // instead: the family is the parent's memory, and the parent runs on it next. + if shares_parent_vm { + self.hand_address_space_to_vfork_parent(); + } else { + let _ = self.leave_address_space(); + } + self.process().complete_vfork(); Ok(0) } /// Loads the specified program into the process's address space and prepares the thread /// to start executing it. pub(crate) fn load_program( + &self, + loader: crate::loader::elf::ElfLoader<'_, Platform, FS>, + argv: Vec, + envp: Vec, + ) -> Result<(), crate::loader::elf::ElfLoaderError> { + let (credentials, secure) = self.credentials_for_exec(loader.main_status()); + self.load_program_with_credentials(loader, argv, envp, credentials, secure) + } + + fn load_program_with_credentials( &self, mut loader: crate::loader::elf::ElfLoader<'_, Platform, FS>, argv: Vec, envp: Vec, + credentials: Arc, + secure: bool, ) -> Result<(), crate::loader::elf::ElfLoaderError> { - let load_info = loader.load(argv, envp, self.init_auxv())?; + let mut proc_cmdline = Vec::new(); + for arg in &argv { + proc_cmdline.extend_from_slice(arg.as_bytes()); + proc_cmdline.push(0); + } + + // The loader publishes the new image's initial break through the (single, shared) page + // manager; take it back out into this process's own slot, restoring the manager's + // "no break set" sentinel, so that a sibling process's break is unaffected. See + // `Process::brk`. + let load_info = { + let _guard = self.global.brk_lock.lock(); + let auxv = self.init_auxv(credentials.as_ref(), secure); + let load_info = loader.load(argv, envp, auxv); + // Take the break back out even when the load failed part-way: a loader that already + // published one and then bailed would otherwise leave it in the manager for the + // next image (any process) to inherit. + let initial_brk = self.global.pm.swap_brk(0); + let load_info = load_info?; + if initial_brk == 0 { + // The loader did not publish a break for this image; the first `brk` this + // process makes will fail (see `PageManager::brk`'s zero-break refusal) and + // its libc will fall back to mmap. Loud, because it means a loader path + // skipped `set_initial_brk` -- the root cause worth fixing. + litebox_util_log::warn!(pid:? = self.pid; "execve: loader left no initial brk"); + } + self.process().brk.store(initial_brk, Ordering::Relaxed); + load_info + }; - self.set_task_comm(loader.comm()); + // Commit the candidate credentials only after every fallible image-building step succeeded. + *self.credentials.borrow_mut() = credentials; + // Linux: `setup_new_exec` makes an ordinary exec dumpable again and a secure (set-uid/ + // set-gid) one not, per the default `suid_dumpable` of 0. + self.process().set_dumpable(!secure); + if secure { + // Linux `begin_new_exec`: "Make sure parent cannot signal privileged process." + self.thread.parent_death_signal.set(None); + self.global.processes.set_parent_death_signal(self.pid, None); + } + let staged = self.thread.staged_exec.borrow_mut().take(); + let (exe, comm) = staged.map_or((None, None), |staged| (Some(staged.exe), Some(staged.comm))); + self.process().set_proc_image(proc_cmdline, exe); + self.set_task_comm(comm.as_deref().unwrap_or_else(|| loader.comm())); + // Every process with an image is reachable through the live table from here on, so + // `/proc/` and `kill(pid)` work for it whether or not it ever forks. + self.register_for_remote_signals(); self.thread .init_state @@ -1568,6 +6068,10 @@ impl Task { } pub(crate) fn handle_init_request(&self, ctx: &mut litebox_common_linux::PtRegs) { + if !self.process().await_launch() { + self.thread.remote.is_exiting.store(true, Ordering::Release); + return; + } self.init_thread_context(ctx); // Attach the thread handle so that the thread can be interrupted. self.thread @@ -1609,11 +6113,31 @@ impl Task { ss: 0x2b, // __USER_DS }; } + #[cfg(target_arch = "aarch64")] + { + // A fresh aarch64 process starts with every general-purpose + // register cleared, `sp` at the top of the initial stack and + // `pc` at the entry point. `pstate` starts at 0, which is + // EL0t/AArch64 with no flags set and nothing masked -- + // exactly what `SAFE_USER_PSTATE` permits. + *ctx = litebox_common_linux::PtRegs { + regs: [0; litebox_common_linux::AARCH64_GENERAL_REGISTER_COUNT], + sp: load_info.user_stack_top, + pc: load_info.entry_point, + pstate: 0, + orig_x0: 0, + // No syscall is in flight on entry. + syscallno: -1, + unused2: 0, + }; + } } ThreadInitState::NewThread { tls, stack, set_child_tid, + #[cfg(target_arch = "aarch64")] + fp, } => { // Set the stack and the return value from clone(). #[cfg(target_arch = "x86_64")] @@ -1623,16 +6147,32 @@ impl Task { } ctx.rax = 0; } + #[cfg(target_arch = "aarch64")] + { + if let Some(stack) = stack { + ctx.sp = stack; + } + // `clone` returns 0 in the child, in x0. + ctx.regs[0] = 0; + } // Set the TLS for the new thread. if let Some(tls) = tls { - #[cfg(target_arch = "x86_64")] - { - self.sys_arch_prctl(ArchPrctlArg::SetFs(tls.as_usize())) - .unwrap(); - } + self.global + .platform + .set_arch_specific_register(&GUEST_TLS_REGISTER, tls.as_usize()) + .expect("failed to set guest TLS for new thread"); } + // Linux's `copy_thread` copies the parent's FPSIMD register + // file into the child task at clone/fork time; this new host + // OS thread's own per-thread FP shadow otherwise starts + // zeroed (correct only for `execve`, see `NewProcess` above), + // so seed it here with the snapshot taken on the parent + // thread at the `clone`/`fork` syscall itself. + #[cfg(target_arch = "aarch64")] + self.global.platform.set_fp_state(&fp); + if let Some(child_tid_ptr) = set_child_tid { // Set the child TID if requested. let _ = child_tid_ptr.write_at_offset::(0, self.tid); @@ -1645,6 +6185,7 @@ impl Task { #[cfg(test)] mod tests { use crate::{UserPtr, UserPtrMut}; + use core::time::Duration; extern crate std; @@ -1675,25 +6216,548 @@ mod tests { .expect("Failed to get FS base"); assert_eq!(current_fs_base, new_fs_base.as_ptr() as usize); - // Restore old FS base - let ptr: UserPtrMut = UserPtrMut::from_usize(old_fs_base); - task.sys_arch_prctl(ArchPrctlArg::SetFs(ptr.as_usize())) - .expect("Failed to restore FS base"); + // Restore old FS base + let ptr: UserPtrMut = UserPtrMut::from_usize(old_fs_base); + task.sys_arch_prctl(ArchPrctlArg::SetFs(ptr.as_usize())) + .expect("Failed to restore FS base"); + } + + #[test] + fn test_sched_getaffinity() { + let task = crate::syscalls::tests::init_platform(None); + + let cpuset = task.sys_sched_getaffinity(None); + assert_eq!(cpuset.bits.len(), super::NR_CPUS); + cpuset.bits.iter().for_each(|b| assert!(*b)); + let ones: usize = cpuset + .as_bytes() + .iter() + .map(|b| b.count_ones() as usize) + .sum(); + assert_eq!(ones, super::NR_CPUS); + } + + /// Reproduces the V8-startup-abort scenario this row was filed for: V8's own startup code + /// aborts the whole process if `clock_gettime` returns an error for any of these clock IDs. + /// Before this change, `ClockId::try_from` rejected everything but `RealTime`/`Monotonic`/ + /// `MonotonicCoarse`, so a real guest binary probing any of the other five clocks at startup + /// (as V8 does) would see `clock_gettime` fail and abort. Verifies every clock ID Linux + /// actually defines round-trips successfully through the real syscall path (`sys_clock_gettime` + /// on `MacOsUserland`/`LinuxUserland`/`WindowsUserland`, backed by real host clocks -- not a + /// mock), and returns a plausible (non-negative) value. + #[test] + fn test_clock_gettime_and_getres_succeed_for_every_clock_id() { + use litebox_common_linux::{ClockId, TimeParam, Timespec}; + + let task = crate::syscalls::tests::init_platform(None); + + for clock_id in [ + ClockId::RealTime, + ClockId::Monotonic, + ClockId::ProcessCpuTime, + ClockId::ThreadCpuTime, + ClockId::MonotonicRaw, + ClockId::RealTimeCoarse, + ClockId::MonotonicCoarse, + ClockId::Boottime, + ] { + let mut ts = Timespec { + tv_sec: -1, + tv_nsec: 0, + }; + let ptr = UserPtrMut::from_ptr(&raw mut ts); + task.sys_clock_gettime(clock_id, TimeParam::Timespec64(ptr)) + .unwrap_or_else(|e| { + panic!( + "clock_gettime({clock_id:?}) unexpectedly failed with {e:?} -- this is \ + exactly the error that makes V8 abort at startup" + ) + }); + assert!( + ts.tv_sec >= 0, + "clock_gettime({clock_id:?}) returned a nonsensical negative tv_sec: {}", + ts.tv_sec + ); + assert!( + ts.tv_nsec < 1_000_000_000, + "clock_gettime({clock_id:?}) returned an out-of-range tv_nsec: {}", + ts.tv_nsec + ); + + let mut res = Timespec { + tv_sec: -1, + tv_nsec: 0, + }; + let res_ptr = UserPtrMut::from_ptr(&raw mut res); + task.sys_clock_getres(clock_id, TimeParam::Timespec64(res_ptr)) + .unwrap_or_else(|e| { + panic!("clock_getres({clock_id:?}) unexpectedly failed: {e:?}") + }); + assert!( + res.tv_sec > 0 || res.tv_nsec > 0, + "clock_getres({clock_id:?}) reported a zero resolution" + ); + } + } + + /// The newly added monotonic-family clocks (`CLOCK_MONOTONIC_RAW`, `CLOCK_BOOTTIME`) must + /// behave like real monotonic clocks: never go backwards, and actually advance across real + /// elapsed wall-clock time. + #[test] + fn test_clock_gettime_monotonic_raw_and_boottime_are_monotonic() { + use litebox_common_linux::{ClockId, TimeParam, Timespec}; + + let task = crate::syscalls::tests::init_platform(None); + + let read = |clock_id: ClockId| -> Duration { + let mut ts = Timespec { + tv_sec: 0, + tv_nsec: 0, + }; + let ptr = UserPtrMut::from_ptr(&raw mut ts); + task.sys_clock_gettime(clock_id, TimeParam::Timespec64(ptr)) + .unwrap_or_else(|e| panic!("clock_gettime({clock_id:?}) failed: {e:?}")); + Duration::try_from(ts).expect("valid timespec") + }; + + for clock_id in [ClockId::MonotonicRaw, ClockId::Boottime] { + let before = read(clock_id); + std::thread::sleep(Duration::from_millis(50)); + let after = read(clock_id); + assert!( + after > before, + "{clock_id:?} did not advance across a real 50ms sleep: before={before:?} after={after:?}" + ); + } + } + + /// Real, host-sourced CPU-time accounting: `CLOCK_THREAD_CPUTIME_ID` must genuinely advance + /// while the thread burns real CPU, and must *not* advance (by anywhere close to the same + /// amount) while the thread is merely sleeping -- proving this isn't wall-clock time + /// silently mislabeled as CPU time. + #[test] + fn test_clock_gettime_thread_cpu_time_tracks_real_cpu_usage_not_wall_clock() { + use litebox_common_linux::{ClockId, TimeParam, Timespec}; + + let task = crate::syscalls::tests::init_platform(None); + + let read_thread_cpu_time = || -> Duration { + let mut ts = Timespec { + tv_sec: 0, + tv_nsec: 0, + }; + let ptr = UserPtrMut::from_ptr(&raw mut ts); + task.sys_clock_gettime(ClockId::ThreadCpuTime, TimeParam::Timespec64(ptr)) + .expect("clock_gettime(CLOCK_THREAD_CPUTIME_ID) failed"); + Duration::try_from(ts).expect("valid timespec") + }; + + let before_busy = read_thread_cpu_time(); + + // Burn real CPU on this thread. `std::hint::black_box` keeps the optimizer from + // eliminating the loop. + let mut acc: u64 = 0; + for i in 0..300_000_000u64 { + acc = std::hint::black_box(acc.wrapping_add(std::hint::black_box(i))); + } + std::hint::black_box(acc); + + let after_busy = read_thread_cpu_time(); + assert!( + after_busy > before_busy, + "thread CPU time did not increase after a real busy loop: before={before_busy:?} \ + after={after_busy:?}" + ); + let consumed_by_busy_loop = after_busy.saturating_sub(before_busy); + assert!( + consumed_by_busy_loop > Duration::from_millis(1), + "expected a meaningful amount of CPU time consumed by the busy loop, got \ + {consumed_by_busy_loop:?}" + ); + + // Sleep for much longer than the busy loop took, without doing any CPU work, and + // confirm thread CPU time barely moves. + std::thread::sleep(Duration::from_millis(300)); + let after_sleep = read_thread_cpu_time(); + let consumed_by_sleep = after_sleep.saturating_sub(after_busy); + assert!( + consumed_by_sleep < Duration::from_millis(100), + "thread CPU time advanced by {consumed_by_sleep:?} across a 300ms *sleep* (no CPU \ + work performed) -- real CPU-time accounting should barely move here, this looks \ + like wall-clock time mislabeled as CPU time" + ); + } + + /// `CLOCK_PROCESS_CPUTIME_ID` sums CPU time across the whole process; it must at least + /// reflect the real CPU work done by the calling thread (the only thread in this test). + #[test] + fn test_clock_gettime_process_cpu_time_tracks_real_cpu_usage() { + use litebox_common_linux::{ClockId, TimeParam, Timespec}; + + let task = crate::syscalls::tests::init_platform(None); + + let read_process_cpu_time = || -> Duration { + let mut ts = Timespec { + tv_sec: 0, + tv_nsec: 0, + }; + let ptr = UserPtrMut::from_ptr(&raw mut ts); + task.sys_clock_gettime(ClockId::ProcessCpuTime, TimeParam::Timespec64(ptr)) + .expect("clock_gettime(CLOCK_PROCESS_CPUTIME_ID) failed"); + Duration::try_from(ts).expect("valid timespec") + }; + + let before = read_process_cpu_time(); + let mut acc: u64 = 0; + for i in 0..300_000_000u64 { + acc = std::hint::black_box(acc.wrapping_add(std::hint::black_box(i))); + } + std::hint::black_box(acc); + let after = read_process_cpu_time(); + + assert!( + after > before, + "process CPU time did not increase after a real busy loop: before={before:?} \ + after={after:?}" + ); + } + + /// `clock_nanosleep` against a CPU-time clock can never wake up (a blocked thread cannot + /// accumulate CPU time), so real Linux rejects it outright; confirm LiteBox does too now that + /// these clock IDs are otherwise recognized. + #[test] + fn test_clock_nanosleep_rejects_cpu_time_clocks() { + use litebox_common_linux::{ClockId, TimeParam, Timespec}; + + let task = crate::syscalls::tests::init_platform(None); + + for clock_id in [ClockId::ProcessCpuTime, ClockId::ThreadCpuTime] { + let mut request = Timespec { + tv_sec: 0, + tv_nsec: 1, + }; + let result = task.sys_clock_nanosleep( + clock_id, + litebox_common_linux::TimerFlags::empty(), + TimeParam::Timespec64(UserPtrMut::from_ptr(&raw mut request)), + TimeParam::None, + ); + assert_eq!( + result, + Err(litebox_common_linux::errno::Errno::EINVAL), + "clock_nanosleep({clock_id:?}) should be rejected with EINVAL" + ); + } + } + + /// `sched_getscheduler`/`sched_setscheduler` round-trip: every thread is always reported as + /// (and can always be, as a no-op, "set" to) `SCHED_OTHER`, matching what any real guest + /// program checking "did the syscall succeed, and is the policy the plain default" would + /// see. + #[test] + fn test_sched_getscheduler_and_setscheduler_round_trip() { + use litebox_common_linux::sched_policy::SCHED_OTHER; + + let task = crate::syscalls::tests::init_platform(None); + + assert_eq!( + task.sys_sched_getscheduler(None), + Ok(usize::try_from(SCHED_OTHER).unwrap()) + ); + + let param = litebox_common_linux::SchedParam { sched_priority: 0 }; + let param_ptr = UserPtr::from_ptr(&raw const param); + assert_eq!( + task.sys_sched_setscheduler(None, SCHED_OTHER, param_ptr), + Ok(0) + ); + + // Also works when explicitly targeting our own tid (pid == 0 and pid == self.tid are + // both "self", matching real Linux semantics for these thread-granularity syscalls). + assert_eq!( + task.sys_sched_getscheduler(Some(task.sys_gettid())), + Ok(usize::try_from(SCHED_OTHER).unwrap()) + ); + } + + /// Real, unprivileged-process-accurate rejection: LiteBox guests never have `CAP_SYS_NICE`, + /// so real-time policies must be rejected with `EPERM`, exactly as they would be on a real + /// unprivileged Linux process. Also checks the ordinary `EINVAL` cases (unknown policy, + /// out-of-range priority for `SCHED_OTHER`). + #[test] + fn test_sched_setscheduler_rejects_real_time_policies_and_bad_priority() { + use litebox_common_linux::errno::Errno; + use litebox_common_linux::sched_policy::{ + SCHED_DEADLINE, SCHED_FIFO, SCHED_OTHER, SCHED_RR, + }; + + let task = crate::syscalls::tests::init_platform(None); + + let param_zero = litebox_common_linux::SchedParam { sched_priority: 0 }; + let param_zero_ptr = UserPtr::from_ptr(&raw const param_zero); + + for policy in [SCHED_FIFO, SCHED_RR, SCHED_DEADLINE] { + assert_eq!( + task.sys_sched_setscheduler(None, policy, param_zero_ptr), + Err(Errno::EPERM), + "real-time policy {policy} should be rejected with EPERM (no CAP_SYS_NICE)" + ); + } + + // An unrecognized policy value is EINVAL, not EPERM. + assert_eq!( + task.sys_sched_setscheduler(None, 0x1234, param_zero_ptr), + Err(Errno::EINVAL) + ); + + // SCHED_OTHER only accepts priority 0. + let param_nonzero = litebox_common_linux::SchedParam { sched_priority: 5 }; + let param_nonzero_ptr = UserPtr::from_ptr(&raw const param_nonzero); + assert_eq!( + task.sys_sched_setscheduler(None, SCHED_OTHER, param_nonzero_ptr), + Err(Errno::EINVAL) + ); + } + + /// `sched_getparam`/`sched_setparam` round-trip. + #[test] + fn test_sched_getparam_setparam_round_trip() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + + let mut got = litebox_common_linux::SchedParam { sched_priority: -1 }; + let got_ptr = UserPtrMut::from_ptr(&raw mut got); + assert_eq!(task.sys_sched_getparam(None, got_ptr), Ok(0)); + assert_eq!(got.sched_priority, 0); + + let set = litebox_common_linux::SchedParam { sched_priority: 0 }; + let set_ptr = UserPtr::from_ptr(&raw const set); + assert_eq!(task.sys_sched_setparam(None, set_ptr), Ok(0)); + + let bad = litebox_common_linux::SchedParam { sched_priority: 1 }; + let bad_ptr = UserPtr::from_ptr(&raw const bad); + assert_eq!(task.sys_sched_setparam(None, bad_ptr), Err(Errno::EINVAL)); + } + + /// None of the four `sched_*` syscalls can honestly answer for a thread other than the + /// caller (LiteBox tracks no state for one), so a pid that isn't "self" must fail with + /// `ESRCH`, matching what real Linux would do for a genuinely nonexistent target thread. + #[test] + fn test_sched_calls_reject_a_remote_pid() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + let remote_pid = task.sys_gettid().wrapping_add(999_999); + + assert_eq!( + task.sys_sched_getscheduler(Some(remote_pid)), + Err(Errno::ESRCH) + ); + + let mut param = litebox_common_linux::SchedParam { sched_priority: 0 }; + let param_ptr = UserPtrMut::from_ptr(&raw mut param); + assert_eq!( + task.sys_sched_getparam(Some(remote_pid), param_ptr), + Err(Errno::ESRCH) + ); + + let set_param = litebox_common_linux::SchedParam { sched_priority: 0 }; + let set_param_ptr = UserPtr::from_ptr(&raw const set_param); + assert_eq!( + task.sys_sched_setparam(Some(remote_pid), set_param_ptr), + Err(Errno::ESRCH) + ); + assert_eq!( + task.sys_sched_setscheduler( + Some(remote_pid), + litebox_common_linux::sched_policy::SCHED_OTHER, + set_param_ptr + ), + Err(Errno::ESRCH) + ); + } + + /// `setpgid(0, N)` followed by `getpgid` (both `pid == 0` and the caller's own pid) must + /// observe `N`. + #[test] + fn test_setpgid_getpgid_self_round_trip() { + let task = crate::syscalls::tests::init_platform(None); + + assert_eq!(task.sys_setpgid(0, 4242), Ok(())); + assert_eq!(task.sys_getpgid(0), Ok(4242)); + assert_eq!(task.sys_getpgid(task.pid), Ok(4242)); + assert_eq!(task.sys_setpgid(0, 4242), Ok(())); + assert_eq!(task.sys_getpgid(0), Ok(4242)); + } + + /// `setpgid(pid, 0)` means "make `pid` its own group leader" -- real `setpgid`'s + /// well-known zero-pgid convention, used by busybox `ash` to start a new job. + #[test] + fn test_setpgid_zero_pgid_targets_own_pid() { + let task = crate::syscalls::tests::init_platform(None); + + assert_eq!(task.sys_setpgid(0, 4242), Ok(())); + assert_eq!(task.sys_setpgid(0, 0), Ok(())); + assert_eq!(task.sys_getpgid(0), Ok(task.pid)); + } + + #[test] + fn test_setpgid_rejects_negative_pgid() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + + assert_eq!(task.sys_setpgid(0, -1), Err(Errno::EINVAL)); + } + + /// A pid this shim cannot vouch for (neither the caller nor a recorded child) is `ESRCH` for + /// both syscalls, matching real Linux's response to a genuinely nonexistent target. + #[test] + fn test_setpgid_getpgid_reject_unrelated_pid() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + let unrelated_pid = task.pid.wrapping_add(999_999); + + assert_eq!(task.sys_getpgid(unrelated_pid), Err(Errno::ESRCH)); + assert_eq!(task.sys_setpgid(unrelated_pid, 4242), Err(Errno::ESRCH)); + } + + /// A live child (registered exactly as `do_fork` registers it) is a permitted + /// `setpgid`/`getpgid` target. Updating it must not alter the parent's own process group. + #[test] + fn test_setpgid_getpgid_accept_a_live_child() { + let parent = crate::syscalls::tests::init_platform(None); + let child = parent + .global + .clone() + .new_test_task(parent.files.borrow().fs.clone()); + parent.global.processes.add_child(child.pid, parent.pid); + parent.global.processes.register_process( + child.pid, + child.remote_signal_target(), + child.process(), + ); + + assert_eq!(parent.sys_setpgid(0, 3131), Ok(())); + assert_eq!(parent.sys_setpgid(child.pid, 4242), Ok(())); + assert_eq!(parent.sys_getpgid(child.pid), Ok(4242)); + assert_eq!(child.sys_getpgid(0), Ok(4242)); + assert_eq!(parent.sys_getpgid(0), Ok(3131)); + } + + /// Threads of one process share one group identity. The channels impose an explicit + /// happens-before order between writes, so both threads and the original task must observe the + /// second write after first observing the first. + #[test] + fn test_setpgid_threads_share_happens_before_order() { + let task = crate::syscalls::tests::init_platform(None); + let first = task.clone_for_test().expect("clone first test thread"); + let second = task.clone_for_test().expect("clone second test thread"); + let (first_done_tx, first_done_rx) = std::sync::mpsc::channel(); + let (second_done_tx, second_done_rx) = std::sync::mpsc::channel(); + + let first_handle = std::thread::spawn(move || { + let first_write = first.sys_setpgid(0, 3131); + first_done_tx.send(()).expect("publish first write"); + second_done_rx.recv().expect("await second write"); + (first_write, first.sys_getpgid(0)) + }); + let second_handle = std::thread::spawn(move || { + first_done_rx.recv().expect("await first write"); + let after_first = second.sys_getpgid(0); + let second_write = second.sys_setpgid(0, 4242); + second_done_tx.send(()).expect("publish second write"); + (after_first, second_write) + }); + + assert_eq!( + second_handle.join().expect("second test thread"), + (Ok(3131), Ok(())) + ); + assert_eq!( + first_handle.join().expect("first test thread"), + (Ok(()), Ok(4242)) + ); + assert_eq!(task.sys_getpgid(0), Ok(4242)); + } + + /// Process-group identity belongs to a process, not the shim. Each barrier round puts writes + /// from two independent processes in the same concurrency window, then makes both reads happen + /// only after both writes. A shim-global group would therefore deterministically make at least + /// one observation wrong in every round. + #[test] + fn test_setpgid_is_isolated_between_concurrent_independent_processes() { + const ROUNDS: i32 = 64; + + let first = crate::syscalls::tests::init_platform(None); + let second = first + .global + .clone() + .new_test_task(first.files.borrow().fs.clone()); + let barrier = std::sync::Arc::new(std::sync::Barrier::new(2)); + let first_barrier = barrier.clone(); + + let first_handle = std::thread::spawn(move || { + let mut observations = std::vec::Vec::new(); + for round in 0..ROUNDS { + first_barrier.wait(); + let expected = 10_000 + round; + let write = first.sys_setpgid(0, expected); + first_barrier.wait(); + observations.push((expected, write, first.sys_getpgid(0))); + first_barrier.wait(); + } + observations + }); + let second_handle = std::thread::spawn(move || { + let mut observations = std::vec::Vec::new(); + for round in 0..ROUNDS { + barrier.wait(); + let expected = 20_000 + round; + let write = second.sys_setpgid(0, expected); + barrier.wait(); + observations.push((expected, write, second.sys_getpgid(0))); + barrier.wait(); + } + observations + }); + + for (expected, write, observed) in first_handle.join().expect("first test process") { + assert_eq!(write, Ok(())); + assert_eq!(observed, Ok(expected)); + } + for (expected, write, observed) in second_handle.join().expect("second test process") { + assert_eq!(write, Ok(())); + assert_eq!(observed, Ok(expected)); + } } #[test] - fn test_sched_getaffinity() { - let task = crate::syscalls::tests::init_platform(None); + fn test_prctl_set_get_parent_death_signal() { + use litebox_common_linux::PrctlArg; + use litebox_common_linux::signal::Signal; - let cpuset = task.sys_sched_getaffinity(None); - assert_eq!(cpuset.bits.len(), super::NR_CPUS); - cpuset.bits.iter().for_each(|b| assert!(*b)); - let ones: usize = cpuset - .as_bytes() - .iter() - .map(|b| b.count_ones() as usize) - .sum(); - assert_eq!(ones, super::NR_CPUS); + let task = crate::syscalls::tests::init_platform(None); + let mut value = -1i32; + let value_ptr = UserPtrMut::from_ptr(&raw mut value); + + task.sys_prctl(PrctlArg::GetPDeathSig(value_ptr)) + .expect("initial PR_GET_PDEATHSIG failed"); + assert_eq!(value, 0); + + task.sys_prctl(PrctlArg::SetPDeathSig(Some(Signal::SIGKILL))) + .expect("PR_SET_PDEATHSIG failed"); + task.sys_prctl(PrctlArg::GetPDeathSig(value_ptr)) + .expect("PR_GET_PDEATHSIG failed"); + assert_eq!(value, Signal::SIGKILL.as_i32()); + + task.sys_prctl(PrctlArg::SetPDeathSig(None)) + .expect("clearing PR_SET_PDEATHSIG failed"); + task.sys_prctl(PrctlArg::GetPDeathSig(value_ptr)) + .expect("PR_GET_PDEATHSIG after clear failed"); + assert_eq!(value, 0); } #[test] @@ -1823,6 +6887,7 @@ mod tests { use litebox::platform::{Instant as _, TimeProvider}; use litebox_common_linux::{ClockId, TimerFlags, Timespec}; + let _guard = crate::syscalls::tests::async_signal_guard(); let task = crate::syscalls::tests::init_platform(None); ::run_test_thread(|| { let platform = task.global.platform; @@ -1857,16 +6922,20 @@ mod tests { "nanosleep should have been interrupted" ); let millis = remain.tv_sec.cast_unsigned() * 1000 + remain.tv_nsec / 1_000_000; - // Allow tolerance for timer imprecision (especially on Windows). + // The upper bound guards against the alarm firing early; the lower + // bound only bounds scheduler lateness, which loaded CI runners + // stretch past 100 ms (witnessed: 1888 on the CI macOS runner). assert!( - (1900..=2100).contains(&millis), + (1500..=2100).contains(&millis), "expected ~2s remaining, got {millis:?}" ); let elapsed_ms = elapsed.as_millis(); std::println!("Alarm fired after {elapsed_ms} ms"); + // The lower bound guards against the alarm firing early; the + // upper bound only bounds scheduler lateness on loaded runners. assert!( - (900..=1100).contains(&elapsed_ms), + (900..=1500).contains(&elapsed_ms), "expected alarm after ~1000 ms, got {elapsed_ms} ms" ); @@ -1882,6 +6951,7 @@ mod tests { fn test_alarm_cancel_prevents_signal() { use litebox_common_linux::{ClockId, TimerFlags, Timespec}; + let _guard = crate::syscalls::tests::async_signal_guard(); let task = crate::syscalls::tests::init_platform(None); ::run_test_thread(|| { assert_eq!(task.sys_alarm(1).unwrap(), 0); @@ -1918,6 +6988,7 @@ mod tests { signal::{SigSet, SigmaskHow, Signal}, }; + let _guard = crate::syscalls::tests::async_signal_guard(); let task = crate::syscalls::tests::init_platform(None); ::run_test_thread(|| { let block_set = SigSet::empty().with(Signal::SIGUSR1); @@ -1965,6 +7036,7 @@ mod tests { use litebox_common_linux::signal::{SIG_IGN, SaFlags, SigAction, SigSet, Signal}; use litebox_common_linux::{ClockId, TimerFlags, Timespec}; + let _guard = crate::syscalls::tests::async_signal_guard(); let task = crate::syscalls::tests::init_platform(None); ::run_test_thread(|| { // Install SIG_IGN for SIGALRM. @@ -2021,6 +7093,7 @@ mod tests { use litebox_common_linux::signal::Signal; use litebox_common_linux::{ClockId, TimerFlags, Timespec}; + let _guard = crate::syscalls::tests::async_signal_guard(); let task = crate::syscalls::tests::init_platform(None); ::run_test_thread(|| { let platform = task.global.platform; @@ -2114,4 +7187,894 @@ mod tests { Some(("/usr/bin/env", Some("python3"))) ); } + + #[test] + fn test_setuid_privileged_sets_uid_and_euid() { + let task = crate::syscalls::tests::init_platform(None); + assert_eq!(task.sys_getuid(), 0); + + task.sys_setuid(1000) + .expect("privileged setuid to an arbitrary uid should succeed"); + assert_eq!(task.sys_getuid(), 1000); + assert_eq!(task.sys_geteuid(), 1000); + } + + #[test] + fn test_setuid_unprivileged_restricted_to_current_ids() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + task.sys_setuid(1000) + .expect("privileged setuid should succeed"); + + // No longer privileged: switching to its own uid is a no-op success... + task.sys_setuid(1000) + .expect("setuid to the caller's own uid should succeed"); + // ...but becoming any other uid is not. + let err = task.sys_setuid(0).unwrap_err(); + assert_eq!(err, Errno::EPERM); + assert_eq!(task.sys_getuid(), 1000); + } + + #[test] + fn test_setgid_privileged_sets_gid_and_egid() { + let task = crate::syscalls::tests::init_platform(None); + assert_eq!(task.sys_getgid(), 0); + + task.sys_setgid(1000) + .expect("privileged setgid to an arbitrary gid should succeed"); + assert_eq!(task.sys_getgid(), 1000); + assert_eq!(task.sys_getegid(), 1000); + } + + #[test] + fn test_setgid_unprivileged_restricted_to_current_ids() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + // The privilege check keys off euid, not gid, so pick a gid while + // still privileged, then drop uid to make the calls below run + // unprivileged and confirm the gid check isn't secretly keying off uid. + task.sys_setgid(2000) + .expect("privileged setgid should succeed"); + task.sys_setuid(1000) + .expect("privileged setuid should succeed"); + + task.sys_setgid(2000) + .expect("setgid to the caller's own gid should succeed"); + let err = task.sys_setgid(0).unwrap_err(); + assert_eq!(err, Errno::EPERM); + assert_eq!(task.sys_getgid(), 2000); + } + + #[test] + fn test_setuid_does_not_affect_sibling_thread_credentials() { + let task = crate::syscalls::tests::init_platform(None); + let sibling = task + .clone_for_test() + .expect("clone_for_test should succeed"); + + task.sys_setuid(1000).expect("setuid should succeed"); + + assert_eq!(task.sys_getuid(), 1000); + assert_eq!(sibling.sys_getuid(), 0); + } + + #[test] + fn test_setgroups_getgroups_round_trip_boundaries_and_copy_on_write() { + use litebox_common_linux::errno::Errno; + + let task = crate::syscalls::tests::init_platform(None); + let null_list = UserPtr::from_usize(0); + let null_out = UserPtrMut::from_usize(0); + + assert_eq!(task.sys_getgroups(0, null_out), Ok(0)); + assert_eq!(task.sys_getgroups(-1, null_out), Err(Errno::EINVAL)); + + let input = [41u32, 7, 41]; + task.sys_setgroups(input.len(), UserPtr::from_ptr(input.as_ptr())) + .expect("root setgroups should succeed"); + assert_eq!(task.sys_getgroups(0, null_out), Ok(input.len())); + + let mut short = [u32::MAX; 2]; + assert_eq!( + task.sys_getgroups(2, UserPtrMut::from_ptr(short.as_mut_ptr())), + Err(Errno::EINVAL) + ); + assert_eq!(short, [u32::MAX; 2]); + + let mut output = [0u32; 3]; + assert_eq!( + task.sys_getgroups(3, UserPtrMut::from_ptr(output.as_mut_ptr())), + Ok(3) + ); + assert_eq!(output, [7, 41, 41]); + + let sibling = task + .clone_for_test() + .expect("clone_for_test should succeed"); + let sibling_input = [9u32]; + sibling + .sys_setgroups( + sibling_input.len(), + UserPtr::from_ptr(sibling_input.as_ptr()), + ) + .expect("sibling setgroups should succeed"); + let mut sibling_output = [0u32; 1]; + assert_eq!( + sibling.sys_getgroups(1, UserPtrMut::from_ptr(sibling_output.as_mut_ptr()),), + Ok(1) + ); + assert_eq!(sibling_output, sibling_input); + assert_eq!( + task.sys_getgroups(3, UserPtrMut::from_ptr(output.as_mut_ptr())), + Ok(3) + ); + assert_eq!(output, [7, 41, 41]); + + assert_eq!(task.sys_setgroups(1, null_list), Err(Errno::EFAULT)); + assert_eq!( + task.sys_setgroups(super::SupplementaryGroups::MAX + 1, null_list), + Err(Errno::EINVAL) + ); + assert_eq!( + task.sys_getgroups(3, UserPtrMut::from_ptr(output.as_mut_ptr())), + Ok(3) + ); + assert_eq!(output, [7, 41, 41]); + + task.sys_setuid(1000).expect("setuid should succeed"); + let denied_input = [11u32]; + assert_eq!( + task.sys_setgroups(denied_input.len(), UserPtr::from_ptr(denied_input.as_ptr()),), + Err(Errno::EPERM) + ); + assert_eq!( + task.sys_getgroups(3, UserPtrMut::from_ptr(output.as_mut_ptr())), + Ok(3) + ); + assert_eq!(output, [7, 41, 41]); + + sibling + .sys_setgroups(0, null_list) + .expect("setgroups with size zero should clear the set"); + assert_eq!(sibling.sys_getgroups(0, null_out), Ok(0)); + } + + #[test] + fn test_prlimit_own_pid_is_self() { + let task = crate::syscalls::tests::init_platform(None); + + task.sys_prlimit( + task.pid, + litebox_common_linux::RlimitResource::NOFILE, + None, + None, + ) + .expect("own pid should be treated the same as pid 0"); + task.sys_prlimit(0, litebox_common_linux::RlimitResource::NOFILE, None, None) + .expect("pid 0 should still mean self"); + } + + #[test] + fn test_get_robust_list_own_tid_is_self() { + let task = crate::syscalls::tests::init_platform(None); + + let mut head_via_tid: usize = 0; + task.sys_get_robust_list(Some(task.tid), UserPtrMut::from_ptr(&raw mut head_via_tid)) + .expect("own tid should be treated the same as pid None"); + + let mut head_via_none: usize = 0; + task.sys_get_robust_list(None, UserPtrMut::from_ptr(&raw mut head_via_none)) + .expect("None should still mean self"); + + assert_eq!(head_via_tid, head_via_none); + } + + /// Real threads, real `sys_futex` syscalls: `FUTEX_REQUEUE` must wake exactly + /// `num_to_wake` waiters directly and *move* the rest onto the second futex word's own wait + /// queue without waking them -- provable only by observing that the requeued waiters stay + /// blocked until a separate, later `FUTEX_WAKE` on the new address, not merely that every + /// thread eventually finishes. + #[test] + fn test_futex_requeue_across_real_threads() { + use litebox_common_linux::{FutexArgs, FutexFlags, TimeParam}; + use std::sync::Barrier; + use std::sync::atomic::{AtomicUsize, Ordering}; + + const N: usize = 4; + const NUM_TO_WAKE: u32 = 1; + + let task = crate::syscalls::tests::init_platform(None); + + // Real, shared guest-visible memory for both futex words; each spawned thread reaches it + // via the raw address (a `Send` `usize`), reconstructing the pointer on its own thread, + // exactly as translated syscall arguments would be. + let mut futex1: u32 = 0; + let mut futex2: u32 = 0; + let futex1_addr = core::ptr::from_mut(&mut futex1) as usize; + let futex2_addr = core::ptr::from_mut(&mut futex2) as usize; + + let completed = std::sync::Arc::new(AtomicUsize::new(0)); + let ready = std::sync::Arc::new(Barrier::new(N + 1)); + + let waiters: std::vec::Vec<_> = (0..N) + .map(|_| { + let completed = std::sync::Arc::clone(&completed); + let ready = std::sync::Arc::clone(&ready); + task.spawn_clone_for_test(move |task| { + ready.wait(); + let result = task.sys_futex(FutexArgs::Wait { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + val: 0, + timeout: TimeParam::Milliseconds(10_000), + }); + completed.fetch_add(1, Ordering::SeqCst); + result + }) + }) + .collect(); + + ready.wait(); // release all N waiters together + std::thread::sleep(core::time::Duration::from_millis(100)); // let them genuinely block + + let woken = task + .sys_futex(FutexArgs::Requeue { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + num_to_wake: NUM_TO_WAKE, + num_to_requeue: u32::try_from(N).unwrap() - NUM_TO_WAKE, + addr2: UserPtrMut::from_usize(futex2_addr), + }) + .expect("futex requeue failed"); + assert_eq!( + usize::try_from(NUM_TO_WAKE).unwrap(), + woken, + "futex(FUTEX_REQUEUE) returns the wake count, not the requeue count" + ); + + // Give the directly-woken waiter(s) ample time to actually return, and any + // incorrectly-also-woken requeued waiters a real chance to (wrongly) return too. + std::thread::sleep(core::time::Duration::from_millis(150)); + assert_eq!( + completed.load(Ordering::SeqCst), + usize::try_from(NUM_TO_WAKE).unwrap(), + "only the directly-woken waiter(s) should have returned -- the requeued ones must \ + still be genuinely blocked, now waiting on futex2, not woken early by the requeue \ + call itself" + ); + + // A stale wake on the *original* address must find nobody left there. + let woken_on_stale_addr = task + .sys_futex(FutexArgs::Wake { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + count: u32::MAX, + }) + .expect("wake on stale addr failed"); + assert_eq!( + woken_on_stale_addr, 0, + "the requeued waiters must have genuinely moved off futex1's wait queue" + ); + + // Now wake the requeued waiters via their new address. + let woken_on_addr2 = task + .sys_futex(FutexArgs::Wake { + addr: UserPtrMut::from_usize(futex2_addr), + flags: FutexFlags::PRIVATE, + count: u32::MAX, + }) + .expect("wake on addr2 failed"); + assert_eq!( + woken_on_addr2, + N - usize::try_from(NUM_TO_WAKE).unwrap(), + "every requeued waiter must be discoverable, and wakeable, via the new address" + ); + + for waiter in waiters { + waiter + .join() + .expect("waiter thread panicked") + .expect("sys_futex(Wait) should not have errored"); + } + assert_eq!(completed.load(Ordering::SeqCst), N); + } + + /// Real threads, real `sys_futex` syscalls: `FUTEX_CMP_REQUEUE` must actually check the + /// futex word before requeuing and fail with `EAGAIN` (never wake or move anyone) once it no + /// longer matches -- the documented race-closing behavior that plain `FUTEX_REQUEUE` does + /// not perform. + #[test] + fn test_futex_cmp_requeue_rejects_stale_value_across_real_threads() { + use litebox_common_linux::errno::Errno; + use litebox_common_linux::{FutexArgs, FutexFlags, TimeParam}; + + let task = crate::syscalls::tests::init_platform(None); + + let mut futex1: u32 = 5; + let mut futex2: u32 = 0; + let futex1_addr = core::ptr::from_mut(&mut futex1) as usize; + let futex2_addr = core::ptr::from_mut(&mut futex2) as usize; + + let waiter = task.spawn_clone_for_test(move |task| { + task.sys_futex(FutexArgs::Wait { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + val: 5, + timeout: TimeParam::Milliseconds(10_000), + }) + }); + + std::thread::sleep(core::time::Duration::from_millis(100)); // let it genuinely block + + let err = task + .sys_futex(FutexArgs::CmpRequeue { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + num_to_wake: 1, + num_to_requeue: 0, + addr2: UserPtrMut::from_usize(futex2_addr), + expected_value: 999, // stale on purpose: the real word is still 5 + }) + .expect_err("a value-mismatched CMP_REQUEUE must fail, not silently requeue"); + assert_eq!(err, Errno::EAGAIN); + + // The waiter must still be genuinely blocked on the original address. + let woken = task + .sys_futex(FutexArgs::Wake { + addr: UserPtrMut::from_usize(futex1_addr), + flags: FutexFlags::PRIVATE, + count: 1, + }) + .expect("wake on futex1 failed"); + assert_eq!( + woken, 1, + "the waiter must still be on futex1's own wait queue -- a mismatched CMP_REQUEUE \ + must not have moved it" + ); + + waiter + .join() + .expect("waiter thread panicked") + .expect("sys_futex(Wait) should not have errored"); + } + + /// Regression test for a thread that dies while still recorded as the owner of a robust + /// futex: [`Task::handle_futex_death`] must set [`FUTEX_OWNER_DIED`] on the futex word and + /// wake a waiter -- mirroring Linux's `handle_futex_death`/`exit_robust_list` + /// (`kernel/futex/core.c`). Before this fix, `handle_futex_death` was `todo!()`, so any + /// dying thread whose robust list was non-empty would panic mid-teardown instead of + /// notifying waiters, permanently stranding a sibling thread blocked in `FUTEX_WAIT` on that + /// lock. + /// + /// This drives `Task::handle_futex_death` directly (rather than round-tripping through a + /// hand-built `RobustListHead`/`RobustList` guest-memory layout, which is real guest-ABI + /// plumbing already covered by `wake_robust_list`'s straightforward list-walking logic) to + /// isolate exactly the piece that was unimplemented: does processing one owned, waited-on + /// futex entry correctly mark it dead and wake the waiter, without panicking. + #[test] + fn test_handle_futex_death_wakes_waiter_and_sets_owner_died() { + use litebox_common_linux::{FutexArgs, FutexFlags, TimeParam}; + use std::sync::Barrier; + use std::sync::atomic::{AtomicU32, Ordering}; + + let task = crate::syscalls::tests::init_platform(None); + + let mut futex_word: u32 = 0; + let futex_addr = core::ptr::from_mut(&mut futex_word) as usize; + let barrier = std::sync::Arc::new(Barrier::new(2)); + + let bg = { + let barrier = std::sync::Arc::clone(&barrier); + task.spawn_clone_for_test(move |bg_task| { + // Simulate this (cloned) thread having locked a robust mutex: the futex word + // records this thread as owner, with the waiters bit set since the main thread + // is about to block on it. + #[expect(clippy::cast_sign_loss, reason = "tid is always non-negative")] + let owner_word = (bg_task.tid as u32) | super::FUTEX_WAITERS; + let futex_atomic = unsafe { &*(futex_addr as *const AtomicU32) }; + futex_atomic.store(owner_word, Ordering::SeqCst); + + barrier.wait(); + // Give the main thread time to actually park in FUTEX_WAIT before "dying" -- + // otherwise this would trivially pass even with the pre-fix `todo!()` never + // running (there would be nothing parked to prove got woken). + std::thread::sleep(core::time::Duration::from_millis(100)); + + bg_task + .handle_futex_death(UserPtr::from_usize(futex_addr), false) + .expect("handle_futex_death should not error for a well-formed entry"); + }) + }; + + barrier.wait(); + let owner_word = { + let futex_atomic = unsafe { &*(futex_addr as *const AtomicU32) }; + futex_atomic.load(Ordering::SeqCst) + }; + let result = task.sys_futex(FutexArgs::Wait { + addr: UserPtrMut::from_usize(futex_addr), + flags: FutexFlags::PRIVATE, + val: owner_word, + timeout: TimeParam::Milliseconds(10_000), + }); + assert_eq!( + result, + Ok(0), + "main thread's FUTEX_WAIT on the robust futex should be woken once \ + handle_futex_death runs for its dying owner, not hang forever" + ); + + let final_word = { + let futex_atomic = unsafe { &*(futex_addr as *const AtomicU32) }; + futex_atomic.load(Ordering::SeqCst) + }; + assert_eq!( + final_word & super::FUTEX_OWNER_DIED, + super::FUTEX_OWNER_DIED, + "the futex word should have FUTEX_OWNER_DIED set once its owner dies without \ + unlocking" + ); + + bg.join().expect("background thread panicked"); + } + + /// Real process-exit teardown (`prepare_for_exit`), a real pipe, and a real epoll + /// registration on its write end: proves a still-open write-end fd left behind when the + /// *last* thread of a process exits -- with no explicit `close()` from the guest, exactly + /// how a real Linux program that just calls `_exit()` (or crashes) behaves, relying on the + /// kernel to close its fds -- is unconditionally closed, so a reader elsewhere gets `EOF` + /// instead of hanging forever, regardless of the epoll registration. + #[test] + fn test_process_exit_closes_pipe_write_end_even_with_epoll_registered() { + use litebox::fd::TypedFd; + use litebox::fs::OFlags; + use litebox::pipes::Pipes; + use litebox_common_linux::{EpollCreateFlags, EpollEvent, EpollOp}; + + let writer_task = crate::syscalls::tests::init_platform(None); + let fs = writer_task.files.borrow().fs.clone(); + // A second, wholly independent process -- its own `Process` and its own `FilesState` -- + // sharing only the same underlying `GlobalState`/`litebox` object, exactly as two real + // OS processes sharing one machine would. This is what makes "the reader is unaffected + // by the writer's own fd-table teardown" a meaningful, non-tautological claim: the + // reader's fd table is not the one `prepare_for_exit` walks. + let reader_task = writer_task.global.clone().new_test_task(fs); + + let (read_fd, write_fd) = writer_task + .sys_pipe2(OFlags::empty()) + .expect("pipe2 failed"); + let write_fd_i32 = i32::try_from(write_fd).unwrap(); + + // Register the write end with an epoll instance the writer also owns -- the exact + // scenario under investigation: an epoll registration must not keep the write end alive + // past the writer's exit. + let epfd = writer_task + .sys_epoll_create(EpollCreateFlags::empty()) + .expect("epoll_create failed"); + let event = EpollEvent::new(litebox::event::Events::OUT.bits(), 0); + writer_task + .sys_epoll_ctl( + i32::try_from(epfd).unwrap(), + EpollOp::EpollCtlAdd, + write_fd_i32, + UserPtr::from_ptr(&raw const event), + ) + .expect("epoll_ctl(ADD) on the write end failed"); + + // Hand the *read* end to the independent reader process, mirroring what real fd + // inheritance (fork, or SCM_RIGHTS over a Unix socket) would produce: a second, + // independent owning reference to the same underlying pipe object, reachable through a + // completely different process's fd table. + let dup_read_fd = { + let writer_files = writer_task.files.borrow(); + let rds = writer_files.raw_descriptor_store.read(); + let original: alloc::sync::Arc>> = + rds.fd_from_raw_integer(read_fd as usize).unwrap(); + drop(rds); + writer_task + .global + .litebox + .descriptor_table_mut() + .duplicate(&original) + .expect("duplicating the read end should succeed") + }; + let reader_raw_fd = { + let reader_files = reader_task.files.borrow(); + let mut rds = reader_files.raw_descriptor_store.write(); + rds.fd_into_raw_integer(dup_read_fd) + }; + let reader_raw_fd = i32::try_from(reader_raw_fd).unwrap(); + + // The reader blocks in a real `read()` on its own, independent fd, waiting for EOF. + let reader = reader_task.spawn_clone_for_test(move |task| { + let mut buf = [0u8; 1]; + task.sys_read(reader_raw_fd, &mut buf, None) + }); + + std::thread::sleep(core::time::Duration::from_millis(100)); // let it genuinely block + assert!( + !reader.is_finished(), + "the reader should still be blocked: the write end is still open" + ); + + // The writer "process" exits -- its last (only) thread -- *without* explicitly closing + // either the pipe write end or the epoll fd. + drop(writer_task); + + let result = reader + .join() + .expect("reader thread panicked") + .expect("read() should not have errored"); + assert_eq!( + result, 0, + "the reader should observe EOF (a 0-byte read) once the writer's process exits, not \ + hang forever" + ); + } + + /// [`super::OwnedRanges`] has to be a real set -- inserting over, and removing out of the + /// middle of, an existing range must split rather than drop or duplicate it -- because a + /// stale entry would let `fork`'s snapshot roll back memory that by then belongs to a + /// different guest process. + #[test] + fn owned_ranges_splits_on_partial_overlap() { + let mut ranges = super::OwnedRanges::default(); + ranges.insert(0x1000..0x5000); + + // A hole punched out of the middle leaves the two ends. + ranges.remove(0x2000..0x3000); + assert_eq!( + ranges + .intersect(&(0..0x10000)) + .collect::>(), + std::vec![0x1000..0x2000, 0x3000..0x5000] + ); + + // Re-inserting across the hole coalesces back into one entry, replacing what it overlaps + // rather than duplicating it. + ranges.insert(0x1000..0x5000); + assert_eq!( + ranges + .intersect(&(0..0x10000)) + .collect::>(), + std::vec![0x1000..0x5000] + ); + + // `intersect` clips to the queried range, since callers use it to pick the owned parts of + // a mapping that may extend past them. + assert_eq!( + ranges + .intersect(&(0x4000..0x9000)) + .collect::>(), + std::vec![0x4000..0x5000] + ); + + ranges.remove(0..usize::MAX); + assert_eq!(ranges.intersect(&(0..0x10000)).count(), 0); + } + + /// The `wstatus` word `wait4` writes is what libc's `WIFEXITED`/`WEXITSTATUS`/`WTERMSIG` + /// decode, so the packing has to match theirs exactly -- a shell reports `$?` straight out of + /// it. + #[test] + fn wait_status_matches_the_libc_macros() { + use litebox_common_linux::signal::Signal; + + let exited = super::encode_wait_status(super::ExitStatus::Exit(42)); + assert_eq!(exited & 0x7f, 0, "WIFEXITED: low seven bits clear"); + assert_eq!((exited >> 8) & 0xff, 42, "WEXITSTATUS"); + + let zero = super::encode_wait_status(super::ExitStatus::Exit(0)); + assert_eq!(zero, 0); + + // An exit code is truncated to 8 bits by the kernel, so `exit(-1)` reads back as 255. + assert_eq!( + (super::encode_wait_status(super::ExitStatus::Exit(-1)) >> 8) & 0xff, + 255 + ); + + let killed = super::encode_wait_status(super::ExitStatus::Signal(Signal::SIGSEGV)); + assert_eq!(killed & 0x7f, Signal::SIGSEGV.as_i32(), "WTERMSIG"); + assert_ne!( + killed & 0x7f, + 0, + "WIFEXITED must be false for a signal death" + ); + } + + /// `wait4` has to distinguish "no children at all" (`ECHILD`) from "children, none finished" + /// (block, or return 0 under `WNOHANG`), and must reap exactly once. + #[test] + fn wait4_reports_no_children_children_running_and_a_finished_child() { + use litebox_common_linux::errno::Errno; + const WNOHANG: i32 = 1; + let task = crate::syscalls::tests::init_platform(None); + let table = &task.global.processes; + + assert_eq!( + task.sys_wait4(-1, None, 0, 0).unwrap_err(), + Errno::ECHILD, + "a task with no children cannot wait for one" + ); + + let child = 0x4242; + table.add_child(child, task.pid); + assert_eq!( + task.sys_wait4(-1, None, WNOHANG, 0).unwrap(), + 0, + "a running child is not reapable, and WNOHANG must not block for it" + ); + assert_eq!( + task.sys_wait4(child + 1, None, WNOHANG, 0).unwrap_err(), + Errno::ECHILD, + "waiting for a pid that is not our child is ECHILD even though we have one" + ); + + table.record_exit(child, super::ExitStatus::Exit(7), 0); + let mut status = 0i32; + let status_ptr = UserPtrMut::from_ptr(&raw mut status); + assert_eq!(task.sys_wait4(-1, Some(status_ptr), 0, 0).unwrap(), child); + assert_eq!((status >> 8) & 0xff, 7); + + assert_eq!( + task.sys_wait4(-1, None, 0, 0).unwrap_err(), + Errno::ECHILD, + "a reaped child is gone: waiting again is ECHILD, not a second reap" + ); + } + + /// Regression test for a `wait4(..., &rusage)` bug: the buffer used to be left completely + /// untouched whenever a caller passed one, so a reader like `busybox time` printed whatever + /// was already sitting in that guest memory -- observed in practice as `sys 2367004162h 16m + /// 32s`. `sys_wait4` must now populate it for real, using each thread's host-measured CPU + /// time (see `Process::cpu_time_nanos`), and must not leave any field -- including the ones + /// this shim cannot measure -- as leftover uninitialized memory. + #[test] + fn wait4_populates_real_rusage_instead_of_leaving_it_uninitialized() { + use litebox_common_linux::{Rusage, TimeVal}; + use zerocopy::{FromBytes as _, IntoBytes as _}; + + let task = crate::syscalls::tests::init_platform(None); + let table = &task.global.processes; + + let child = 0x4343; + table.add_child(child, task.pid); + // As if the child had genuinely consumed 2.5s of host CPU time across its threads. + let cpu_time = Duration::from_millis(2500); + table.record_exit( + child, + super::ExitStatus::Exit(0), + u64::try_from(cpu_time.as_nanos()).unwrap(), + ); + + // A sentinel fill: if `sys_wait4` ever again leaves the buffer untouched, this pattern + // survives every assertion below rather than silently reading back as zero. + let mut buf = [0xAAu8; core::mem::size_of::()]; + let rusage_ptr = UserPtrMut::::from_ptr(buf.as_mut_ptr().cast()); + + assert_eq!( + task.sys_wait4(-1, None, 0, rusage_ptr.as_usize()).unwrap(), + child + ); + + let rusage = Rusage::read_from_bytes(&buf).unwrap(); + assert_eq!( + rusage.ru_utime.as_bytes(), + TimeVal::from(cpu_time).as_bytes(), + "ru_utime must be the real, host-measured CPU time, not the sentinel or garbage" + ); + assert_eq!( + rusage.ru_stime.as_bytes(), + TimeVal::default().as_bytes(), + "ru_stime is honestly zero (this shim has no meaningful kernel time of its own to \ + attribute), not the sentinel" + ); + assert_eq!( + rusage.ru_maxrss, 0, + "unmeasured fields are zeroed, not sentinel garbage" + ); + } + + /// A `fork`ed child gets its own descriptor *table* over the same open file *descriptions*. + /// The shell relies on both halves: it rearranges fds 0/1/2 for the command it is about to + /// `exec` (which must not reach back into the shell), and it expects the descriptions + /// themselves -- offsets, pipe ends -- to be shared with what it forked from. + #[test] + fn fork_copies_the_descriptor_table_but_shares_the_descriptions() { + let _guard = crate::syscalls::tests::address_space_guard(); + let task = crate::syscalls::tests::init_platform(None); + + let (read_fd, write_fd) = task.sys_pipe2(litebox::fs::OFlags::empty()).unwrap(); + let (read_fd, write_fd) = ( + i32::try_from(read_fd).unwrap(), + i32::try_from(write_fd).unwrap(), + ); + + let child_files = task.files.borrow().fork_copy(&task).unwrap(); + let child_fds: std::vec::Vec = child_files + .raw_descriptor_store + .read() + .iter_alive() + .collect(); + let parent_fds: std::vec::Vec = task + .files + .borrow() + .raw_descriptor_store + .read() + .iter_alive() + .collect(); + assert_eq!( + child_fds, parent_fds, + "every descriptor is duplicated at the same number" + ); + + // Closing in the child's table leaves the parent's number alive... + let parent_files = task.files.replace(alloc::sync::Arc::new(child_files)); + task.sys_close(write_fd).unwrap(); + let child_files = task.files.replace(parent_files); + assert!( + !child_files + .raw_descriptor_store + .read() + .iter_alive() + .any(|fd| fd == usize::try_from(write_fd).unwrap()) + ); + assert!( + task.files + .borrow() + .raw_descriptor_store + .read() + .iter_alive() + .any(|fd| fd == usize::try_from(write_fd).unwrap()), + "the parent's write end must survive the child closing its own" + ); + + // ...and the shared description is still open, so the read end has not seen EOF: a write + // through the parent's still-open write end is readable. + assert_eq!(task.sys_write(write_fd, b"hi", None).unwrap(), 2); + let mut buf = [0u8; 2]; + assert_eq!(task.sys_read(read_fd, &mut buf, None).unwrap(), 2); + assert_eq!(&buf, b"hi"); + + task.sys_close(read_fd).unwrap(); + task.sys_close(write_fd).unwrap(); + } + + /// The address-space token is a strict hand-off: only one member holds it at a time, a + /// waiter takes it the moment it is released, and `hand_off_to` never lets it go free (which + /// is what stops a third member from stealing a freshly `fork`ed child's memory before its + /// first instruction). + #[test] + fn address_space_token_is_held_by_exactly_one_member() { + use super::{ADDRESS_SPACE_FREE, Ordering, SharedAddressSpace}; + use litebox::platform::RawMutex as _; + + let shared: SharedAddressSpace = + SharedAddressSpace::new(1000); + let word = || shared.holder.underlying_atomic().load(Ordering::Relaxed); + assert_eq!(word(), 1000); + + // Acquiring while another member holds it must not succeed; `abandon` is the only way + // out, and it must not have taken the token. + assert!(!shared.acquire(1001, || true)); + assert_eq!(word(), 1000); + + // A direct hand-off never passes through the free state. + shared.hand_off_to(1001); + assert_eq!(word(), 1001); + + shared.release(); + assert_eq!(word(), ADDRESS_SPACE_FREE); + assert!(shared.acquire(1002, || panic!("should not have had to block"))); + assert_eq!(word(), 1002); + } + + /// A child becoming a zombie posts `SIGCHLD` to its parent. + /// + /// Without this, busybox `ash`'s blocking `wait` -- which is a `sigsuspend` loop waiting for + /// its `SIGCHLD` handler to set a flag -- spins forever. + #[test] + fn child_exit_posts_sigchld_to_the_parent() { + use litebox_common_linux::signal::{SaFlags, SigAction, SigSet, Signal}; + + let task = crate::syscalls::tests::init_platform(None); + let table = &task.global.processes; + let child = task.pid + 1; + table.register_process(task.pid, task.remote_signal_target(), task.process()); + table.add_child(child, task.pid); + + // With the default disposition (ignore), the signal must not make blocking syscalls + // return `EINTR`, exactly as on Linux, where an ignored signal is never queued at all. + table.record_exit(child, super::ExitStatus::Exit(0), 0); + assert!( + !task.has_pending_signals(), + "an ignored SIGCHLD must not count as deliverable" + ); + + // With a handler installed it must be deliverable. + let act = SigAction { + sigaction: 0x1234, + flags: SaFlags::empty(), + #[cfg(target_pointer_width = "64")] + __pad: 0, + restorer: 0, + mask: SigSet::empty(), + }; + task.sys_rt_sigaction( + Signal::SIGCHLD, + Some(UserPtr::from_ptr(&raw const act)), + None, + core::mem::size_of::(), + ) + .expect("rt_sigaction failed"); + assert!( + task.has_pending_signals(), + "a handled SIGCHLD must be deliverable" + ); + assert!(task.pending_signal_set().contains(Signal::SIGCHLD)); + } + + /// `rt_sigsuspend` always fails with `EINTR`, and leaves the caller's original mask to be put + /// back by the return-to-guest path rather than restoring it itself -- restoring it early + /// would re-block the signal whose handler the caller is waiting to run. + #[test] + fn rt_sigsuspend_defers_restoring_the_callers_mask() { + use litebox_common_linux::{ + errno::Errno, + signal::{SigSet, SigmaskHow, Signal}, + }; + + let _guard = crate::syscalls::tests::async_signal_guard(); + let task = crate::syscalls::tests::init_platform(None); + ::run_test_thread( + || { + // Block everything, as busybox's `waitproc` does before it suspends. + let everything = !SigSet::empty(); + task.sys_rt_sigprocmask( + SigmaskHow::SIG_SETMASK, + Some(UserPtr::from_ptr(&raw const everything)), + None, + core::mem::size_of::(), + ) + .expect("block everything failed"); + + // Suspend under a mask that leaves everything through, and let the alarm end it. + let allow_everything = SigSet::empty(); + assert_eq!(task.sys_alarm(1).unwrap(), 0); + assert_eq!( + task.sys_rt_sigsuspend( + Some(UserPtr::from_ptr(&raw const allow_everything)), + core::mem::size_of::() + ), + Err(Errno::EINTR) + ); + task.sys_alarm(0).unwrap(); + + // Still under the temporary mask, so the signal that ended the wait is still + // deliverable and its handler would run with SIGALRM unblocked. + assert!( + task.pending_signal_set().contains(Signal::SIGALRM), + "the suspending mask must still be in effect on return" + ); + + // The return-to-guest path puts the caller's mask back. + task.restore_saved_signal_mask(); + let mut current = SigSet::empty(); + task.sys_rt_sigprocmask( + SigmaskHow::SIG_BLOCK, + None, + Some(UserPtrMut::from_ptr(&raw mut current)), + core::mem::size_of::(), + ) + .expect("read mask failed"); + assert_eq!( + current.as_u64(), + everything.as_u64(), + "the mask in force before rt_sigsuspend must be restored afterwards" + ); + }, + ); + } } diff --git a/litebox_shim_linux/src/syscalls/ptrace.rs b/litebox_shim_linux/src/syscalls/ptrace.rs new file mode 100644 index 0000000000..632e02d4bb --- /dev/null +++ b/litebox_shim_linux/src/syscalls/ptrace.rs @@ -0,0 +1,458 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! `ptrace(2)`: attach/seize, genuine stop rendezvous, `NT_PRSTATUS`/`NT_ARM_TLS` +//! `GETREGSET`/`SETREGSET`, `PTRACE_CONT`, `PTRACE_DETACH`. +//! +//! # Scope +//! +//! Same-process only: a tracer targets a sibling thread of its own guest process by `tid`, +//! looked up through [`super::process::Process::thread_remote`] -- the same PID/TID-reuse-safe +//! mechanism `tkill`/`tgkill` already use (a `tid` that has exited is simply not found; a `tid` +//! later reused by an unrelated new thread gets a distinct `ThreadRemote`, so a tracer holding a +//! reference to the old one can never observe or mutate the new thread's state). There is +//! currently no cross-process credential/namespace infrastructure in this shim (`sys_kill`'s own +//! `pid != self.pid` path is explicitly unimplemented, `log_unsupported!("sys_{{t|tg}}kill with +//! remote pid")`), so a genuinely cross-process, cross-uid `PTRACE_ATTACH` is out of scope for +//! this pass; the permission check below is the same-process, same-thread-group counterpart of +//! Linux's `ptrace_may_access` (a thread may always trace another thread of its own process +//! sharing its credentials). +//! +//! # Stop rendezvous +//! +//! A "ptrace stop" is requested by setting [`PtraceState`]'s word to +//! [`STATE_STOP_REQUESTED`] and kicking the target thread's [`super::process::ThreadRemote`] +//! handle (the exact mechanism `tkill`/interrupt already use to reach a thread that may be deep +//! inside `hv_vcpu_run`). The target only actually parks -- and only then is its register state +//! read into [`PtraceState`] for the tracer to observe -- from +//! [`crate::wait::Task::prepare_to_run_guest`], which runs after every `syscall`/`exception`/ +//! `interrupt` return and strictly before the thread re-enters guest code. At that point the +//! HVF backend has already released the thread's vCPU lane back to the pool (see +//! `HvfBackend::run_thread`: `release_lane` happens before `dispatch`, which is what eventually +//! calls `shim.syscall`/`shim.interrupt`/`shim.exception` and, through those, `prepare_to_run_guest`) +//! and captured its architectural state out of the vCPU into `ctx: &mut PtRegs` -- so the thread +//! is provably not mid-`hv_vcpu_run`, and `ctx` is the authoritative, complete, host-boundary-safe +//! snapshot of its logical Linux-guest state. This is exactly the "safe rendezvous point" pattern +//! `ThreadHandle::interrupt`/`HvfThreadSlot::kick` already establish for interrupts, reused rather +//! than reinvented: no new HVF-backend primitive is needed, and no Darwin host register state is +//! ever exposed (only `ctx`, the shim's own logical `PtRegs`, plus `TPIDR_EL0` read through the +//! existing [`litebox::platform::ArchSpecificProvider`] accessor while still running as the +//! tracee's own thread). +//! +//! # What this does not implement +//! +//! `PTRACE_SEIZE`'s only real difference from `PTRACE_ATTACH` here is that it does not force an +//! immediate stop (matching Linux). `PTRACE_INTERRUPT` (seize-only stop-on-demand) and hardware +//! single-step/breakpoint regsets (`NT_ARM_HW_BREAK`/`NT_ARM_HW_WATCH`) are not implemented -- +//! `PTRACE_SETOPTIONS`/`PTRACE_PEEKTEXT`/`PTRACE_POKETEXT`/`PTRACE_SINGLESTEP` and any other +//! request return `ENOSYS`. `NT_PRFPREG`/`NT_ARM_VFP` (FPSIMD) and every hardware-debug regset +//! return `ENODEV` from `GETREGSET`/`SETREGSET`, explicitly, rather than silently returning +//! zeroed or partial data. `waitpid`-visible stop/continue status transitions (`WUNTRACED`) are +//! not wired into `wait4`'s existing exit-only `ChildRecord`/reap machinery in this pass; a +//! tracer observes stop/continue directly through this module's own blocking primitives +//! ([`Task::sys_ptrace`]'s `PTRACE_ATTACH` and `PTRACE_CONT` handling), not through `waitpid`. + +use crate::{ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut}; +use litebox::platform::{ArchSpecificRegister, RawMutex as _}; +use litebox::sync::Mutex; +use litebox_common_linux::PtRegs; +use litebox_common_linux::errno::Errno; +use litebox_common_linux::ptrace::{ + NT_ARM_TLS, NT_PRSTATUS, PTRACE_ATTACH, PTRACE_CONT, PTRACE_DETACH, PTRACE_GETREGSET, + PTRACE_SEIZE, PTRACE_SETREGSET, UserPtRegs, +}; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +/// The real Linux `struct iovec` layout (`{ void *iov_base; size_t iov_len; }`), read/written +/// generically as two machine words. `ptrace`'s `data` argument for `GETREGSET`/`SETREGSET` +/// points at one of these; unlike [`litebox_common_linux::IoReadVec`]/`IoWriteVec` (each fixed to +/// one access direction), this same iovec is both read from (`SETREGSET`) and written to +/// (`GETREGSET`), so a single direction-neutral raw layout is the correct fit rather than either +/// existing typed alias. +#[derive(Clone, Copy, FromBytes, IntoBytes, Immutable)] +#[repr(C)] +struct RawIovec { + iov_base: usize, + iov_len: usize, +} + +/// No tracer attached. +const STATE_DETACHED: u32 = 0; +/// A tracer is attached and the tracee runs normally. +const STATE_RUNNING: u32 = 1; +/// A stop was requested; the tracee has not yet reached the rendezvous point in +/// `prepare_to_run_guest`. +const STATE_STOP_REQUESTED: u32 = 2; +/// The tracee has parked at the rendezvous point; `registers`/`tpidr_el0` are a valid, stable +/// snapshot the tracer may read, and (before the tracee resumes) mutate. +const STATE_STOPPED: u32 = 3; + +/// `ptrace` attach/stop state carried on every thread's [`super::process::ThreadRemote`]. +/// +/// The state word is a [`litebox::platform::RawMutex`] used purely as a blockable atomic (the +/// same idiom `Process::fork_gate`/`ProcessLaunch`/`VforkCompletion` already use for cross-thread, +/// non-interruptible rendezvous): every transition is a `compare_exchange` on the word followed by +/// `wake_all`, and every wait is a loop of `load` -> `block(observed)`. `tracer_tid`/`registers`/ +/// `tpidr_el0` are guarded by their own mutex separately from the word so that a tracer reading a +/// stopped snapshot never has to hold the word's own (raw, non-reentrant) lock. +pub(crate) struct PtraceState { + word: ::RawMutex, + /// The attached tracer's `tid`, valid whenever `word != STATE_DETACHED`. Only this tid may + /// `GETREGSET`/`SETREGSET`/`PTRACE_CONT`/`PTRACE_DETACH` -- Linux's own "only the tracer may + /// act on its tracee" rule. + tracer_tid: core::sync::atomic::AtomicI32, + /// Valid only while `word == STATE_STOPPED`. + snapshot: Mutex, +} + +#[derive(Clone)] +struct StoppedSnapshot { + registers: PtRegs, + tpidr_el0: u64, +} + +impl PtraceState { + pub(crate) fn new() -> Self { + Self { + word: ::RawMutex::INIT, + tracer_tid: core::sync::atomic::AtomicI32::new(0), + snapshot: Mutex::new(StoppedSnapshot { + registers: PtRegs::default(), + tpidr_el0: 0, + }), + } + } + + fn state(&self) -> u32 { + self.word + .underlying_atomic() + .load(core::sync::atomic::Ordering::Acquire) + } + + /// `PTRACE_ATTACH`/`PTRACE_SEIZE`: claims this (detached) tracee for `tracer_tid`. + /// + /// Returns `false` if a tracer is already attached (Linux: `EPERM`). + fn attach(&self, tracer_tid: i32) -> bool { + let attached = self + .word + .underlying_atomic() + .compare_exchange( + STATE_DETACHED, + STATE_RUNNING, + core::sync::atomic::Ordering::AcqRel, + core::sync::atomic::Ordering::Acquire, + ) + .is_ok(); + if attached { + self.tracer_tid + .store(tracer_tid, core::sync::atomic::Ordering::Release); + } + attached + } + + /// Whether `tid` is this tracee's currently-attached tracer. + fn is_tracer(&self, tid: i32) -> bool { + self.state() != STATE_DETACHED + && self.tracer_tid.load(core::sync::atomic::Ordering::Acquire) == tid + } + + /// Requests a stop and blocks until the tracee has genuinely parked (or detaches/exits + /// first). Returns `false` if the tracee is no longer attached at all by the time this + /// observes a terminal state (e.g. raced with a concurrent detach) -- callers other than + /// `attach` itself do not currently hit this path, but it is handled rather than assumed + /// away. + fn request_stop_and_wait(&self) -> bool { + loop { + let observed = self.state(); + match observed { + STATE_STOPPED => return true, + STATE_DETACHED => return false, + STATE_RUNNING => { + let _ = self.word.underlying_atomic().compare_exchange( + STATE_RUNNING, + STATE_STOP_REQUESTED, + core::sync::atomic::Ordering::AcqRel, + core::sync::atomic::Ordering::Acquire, + ); + // Whether this call or a racing one made the transition, the loop's next + // iteration re-checks the (now current) state regardless. + } + STATE_STOP_REQUESTED => { + let _ = self.word.block(observed); + } + _ => unreachable!("invalid ptrace state"), + } + } + } + + /// Called only by the tracee's own thread, from + /// [`crate::wait::Task::prepare_to_run_guest`] -- the safe rendezvous point where the vCPU + /// lane has already been released and `ctx`/`tpidr_el0` are the authoritative, complete + /// logical guest state. Parks (genuine host blocking, not a spin) while a stop is requested + /// or in effect, capturing the snapshot on entry and re-applying any tracer mutation on exit. + fn rendezvous(&self, ctx: &mut PtRegs, tpidr_el0: u64) -> u64 { + if self.state() != STATE_STOP_REQUESTED { + return tpidr_el0; + } + *self.snapshot.lock() = StoppedSnapshot { + registers: ctx.clone(), + tpidr_el0, + }; + self.word.underlying_atomic().store( + STATE_STOPPED, + core::sync::atomic::Ordering::Release, + ); + self.word.wake_all(); + loop { + let observed = self.state(); + match observed { + STATE_STOPPED => { + let _ = self.word.block(observed); + } + STATE_RUNNING | STATE_DETACHED => break, + _ => unreachable!("invalid ptrace state"), + } + } + // The tracer may have mutated `snapshot` (`PTRACE_SETREGSET`) any time before this + // resumed the tracee; apply it now, on the tracee's own thread, before guest re-entry. + let snapshot = self.snapshot.lock().clone(); + *ctx = snapshot.registers; + snapshot.tpidr_el0 + } + + /// `PTRACE_CONT`: resumes a stopped tracee. Returns `false` if it was not stopped. + fn resume(&self) -> bool { + let resumed = self + .word + .underlying_atomic() + .compare_exchange( + STATE_STOPPED, + STATE_RUNNING, + core::sync::atomic::Ordering::AcqRel, + core::sync::atomic::Ordering::Acquire, + ) + .is_ok(); + if resumed { + self.word.wake_all(); + } + resumed + } + + /// `PTRACE_DETACH`: releases the tracee unconditionally (stopped or running) and resumes it + /// if it was stopped. + fn detach(&self) { + self.word + .underlying_atomic() + .store(STATE_DETACHED, core::sync::atomic::Ordering::Release); + self.tracer_tid + .store(0, core::sync::atomic::Ordering::Release); + self.word.wake_all(); + } + + /// Reads the stopped snapshot's `NT_PRSTATUS` view. Caller must have already confirmed + /// `word == STATE_STOPPED` under the tracer's own serialized use of this state (ptrace + /// requests from one tracer are not concurrent with each other by construction: they are + /// ordinary syscalls on the tracer's single thread). + fn read_prstatus(&self) -> UserPtRegs { + UserPtRegs::from(&self.snapshot.lock().registers) + } + + fn write_prstatus(&self, regs: &UserPtRegs) { + regs.write_into(&mut self.snapshot.lock().registers); + } + + fn read_tls(&self) -> u64 { + self.snapshot.lock().tpidr_el0 + } + + fn write_tls(&self, value: u64) { + self.snapshot.lock().tpidr_el0 = value; + } + + fn is_stopped(&self) -> bool { + self.state() == STATE_STOPPED + } + + /// Called when the tracee thread detaches from its process (exits): unconditionally + /// releases any attached tracer rather than leaving it blocked forever on a rendezvous that + /// can now never happen. A tracer's next request against this `tid` finds no `ThreadRemote` + /// (`Process::thread_remote` returns `None`, since `detach_thread` has already removed it) + /// and fails `ESRCH`, exactly like `tkill` against an exited thread. + pub(crate) fn on_thread_exit(&self) { + self.word + .underlying_atomic() + .store(STATE_DETACHED, core::sync::atomic::Ordering::Release); + self.word.wake_all(); + } +} + +impl Task { + /// Handle syscall `ptrace`. + pub(crate) fn sys_ptrace( + &self, + request: i64, + pid: i32, + addr: usize, + data: usize, + ) -> Result { + // Same-process scope only -- see this module's doc comment. `pid` here is a Linux `tid` + // (ptrace addresses individual threads, not thread groups). + if pid == self.tid { + // A thread may not trace itself: Linux's own `ptrace_attach` rejects + // `task == current`. + return Err(Errno::EPERM); + } + let Some(remote) = self.process().thread_remote(pid) else { + // Not found (exited, or never existed in this process): PID-reuse-safe by + // construction, since `thread_remote` looks up the live `threads` map, not a + // reusable slot index. + return Err(Errno::ESRCH); + }; + + match request { + PTRACE_ATTACH | PTRACE_SEIZE => { + if !remote.ptrace.attach(self.tid) { + return Err(Errno::EPERM); + } + if request == PTRACE_ATTACH { + // PTRACE_ATTACH stops the tracee immediately (Linux delivers a synthetic + // group-stop the tracer observes via `waitpid`); this shim's tracer instead + // observes it by the stop having genuinely completed before this call + // returns. + remote.interrupt(); + if !remote.ptrace.request_stop_and_wait() { + return Err(Errno::ESRCH); + } + } + // PTRACE_SEIZE attaches without forcing a stop; the tracee keeps running until a + // later PTRACE_ATTACH-style stop request. `PTRACE_INTERRUPT` (seize-only + // stop-on-demand) is not implemented. + Ok(0) + } + PTRACE_GETREGSET | PTRACE_SETREGSET => { + if !remote.ptrace.is_tracer(self.tid) { + return Err(Errno::ESRCH); + } + if !remote.ptrace.is_stopped() { + return Err(Errno::ESRCH); + } + // `addr` carries the small `NT_*` type identifier here (never a real address, + // per the real ptrace ABI for GETREGSET/SETREGSET); a value with no `i32` + // representation cannot match any known `NT_*` constant and correctly falls + // through to the explicit `ENODEV` arm below, so truncation is the intended + // matching behavior, not a truncation bug. + #[allow(clippy::cast_possible_truncation, clippy::cast_possible_wrap)] + let nt_type = addr as i32; + // `data` is `struct iovec *`: `{ void *iov_base; size_t iov_len; }`. Read + // generically as two machine words -- unlike `IoReadVec`/`IoWriteVec`, ptrace's + // iovec is read *and* written back through the same pointer direction, so + // neither of those (each fixed to one direction) applies cleanly here. + let iovec = UserPtr::::from_usize(data) + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + let write_len = |actual: usize| { + // Linux writes the regset's actual byte length back into `iov_len` on a + // successful GETREGSET. Best-effort: a tracer that only reads `iov_base`'s + // contents (the common case) is unaffected if this padding write races + // something odd. + let len_field = UserPtrMut::::from_usize( + data + core::mem::offset_of!(RawIovec, iov_len), + ); + let _ = len_field.write_at_offset::(0, actual); + }; + match nt_type { + NT_PRSTATUS => { + if iovec.iov_len < UserPtRegs::SIZE { + return Err(Errno::EINVAL); + } + if request == PTRACE_GETREGSET { + let regs = remote.ptrace.read_prstatus(); + UserPtrMut::::from_usize(iovec.iov_base) + .write_at_offset::(0, regs) + .ok_or(Errno::EFAULT)?; + write_len(UserPtRegs::SIZE); + } else { + let regs = UserPtr::::from_usize(iovec.iov_base) + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + remote.ptrace.write_prstatus(®s); + } + Ok(0) + } + NT_ARM_TLS => { + if iovec.iov_len < core::mem::size_of::() { + return Err(Errno::EINVAL); + } + if request == PTRACE_GETREGSET { + let value = remote.ptrace.read_tls(); + UserPtrMut::::from_usize(iovec.iov_base) + .write_at_offset::(0, value) + .ok_or(Errno::EFAULT)?; + write_len(core::mem::size_of::()); + } else { + let value = UserPtr::::from_usize(iovec.iov_base) + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + remote.ptrace.write_tls(value); + } + Ok(0) + } + // Real, distinct Linux regset types this shim does not populate: FPSIMD + // (`NT_PRFPREG`/`NT_ARM_VFP`) and hardware debug/watch state + // (`NT_ARM_HW_BREAK`/`NT_ARM_HW_WATCH`), among others. Fail explicitly rather + // than returning zeroed or partial data. + _ => Err(Errno::ENODEV), + } + } + PTRACE_CONT => { + if !remote.ptrace.is_tracer(self.tid) { + return Err(Errno::ESRCH); + } + // `data` as a pending signal to deliver on resume is not implemented; only 0 + // (no signal) is accepted, matching every other unsupported nonzero-argument + // case in this shim. + if data != 0 { + return Err(Errno::EINVAL); + } + if !remote.ptrace.resume() { + return Err(Errno::ESRCH); + } + Ok(0) + } + PTRACE_DETACH => { + if !remote.ptrace.is_tracer(self.tid) { + return Err(Errno::ESRCH); + } + remote.ptrace.detach(); + Ok(0) + } + _ => { + let _ = addr; + Err(Errno::ENOSYS) + } + } + } + + /// Called from [`crate::wait::Task::prepare_to_run_guest`]: parks this thread at the ptrace + /// stop rendezvous if a tracer has requested one, applying any tracer register mutation on + /// resume. No-op (a single atomic load) when untraced or not currently stop-requested. + pub(crate) fn ptrace_rendezvous(&self, ctx: &mut litebox_common_linux::PtRegs) { + let tpidr_el0 = self + .global + .platform + .get_arch_specific_register(&ArchSpecificRegister::TpidrEl0) + .unwrap_or(0) as u64; + let new_tpidr = self.thread_remote().ptrace.rendezvous(ctx, tpidr_el0); + if new_tpidr != tpidr_el0 { + // `TpidrEl0` is aarch64-only (this whole module is arch-gated), where `usize` is + // 64-bit and this cast is exact; no fallible conversion is warranted for a target + // this code never runs on. + #[allow(clippy::cast_possible_truncation)] + let value = new_tpidr as usize; + let _ = self + .global + .platform + .set_arch_specific_register(&ArchSpecificRegister::TpidrEl0, value); + } + } +} diff --git a/litebox_shim_linux/src/syscalls/signal/aarch64.rs b/litebox_shim_linux/src/syscalls/signal/aarch64.rs new file mode 100644 index 0000000000..03295a0d3f --- /dev/null +++ b/litebox_shim_linux/src/syscalls/signal/aarch64.rs @@ -0,0 +1,312 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! aarch64 signal-frame construction and teardown. +//! +//! The aarch64 frame differs from the x86-64 one in three ways that matter +//! here, all dictated by `arch/arm64/kernel/signal.c`: +//! +//! * `siginfo` comes *before* `ucontext` in `struct rt_sigframe`, and the frame +//! sits exactly at `sp` -- there is no return address pushed below it, so +//! `rt_sigreturn` reads the frame straight off `sp`. +//! * The trampoline is handed to the handler in `x30` rather than pushed, and a +//! `frame_record` (a saved `x29`/`x30` pair) is written just above the frame +//! so an unwinder can chain out of the handler. +//! * AAPCS64 has no red zone, so nothing below `sp` needs to be stepped over. + +use crate::ShimPlatform; +use crate::UserPtrMut; +use crate::syscalls::signal::{DeliverFault, SignalState}; +use core::mem::offset_of; +use litebox::shim::{Exception, ExceptionInfo}; +use litebox::utils::{ReinterpretUnsignedExt as _, TruncateExt as _}; +use litebox_common_linux::{ + AARCH64_GENERAL_REGISTER_COUNT, PtRegs, + signal::{SaFlags, SigAction, SigSet, Siginfo, Signal, Ucontext, aarch64::Sigcontext}, +}; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +/// `pt_regs::syscallno` value meaning "no syscall is in flight", matching the +/// kernel's `NO_SYSCALL`. +const NO_SYSCALL: i32 = -1; + +/// The kernel's `struct _aarch64_ctx`: the common header on every extension +/// record in `sigcontext.__reserved`'s chain, terminated by a record with +/// `magic == 0`. +#[repr(C)] +#[derive(Clone, Copy, FromBytes, IntoBytes, Immutable)] +struct Aarch64CtxHeader { + magic: u32, + size: u32, +} + +/// The kernel's `struct fpsimd_context` +/// (`arch/arm64/include/uapi/asm/sigcontext.h`): `fpsr`/`fpcr` come *before* +/// `vregs` -- verified against the kernel header directly (fetched from +/// `torvalds/linux` `master`), not assumed from Darwin's own +/// `__darwin_arm_neon_state64` putting them in the opposite order. +#[repr(C)] +#[derive(Clone, Copy, FromBytes, IntoBytes, Immutable)] +struct FpsimdContext { + head: Aarch64CtxHeader, + fpsr: u32, + fpcr: u32, + vregs: [u128; 32], +} + +/// `FPSIMD_MAGIC` from `arch/arm64/include/uapi/asm/sigcontext.h`. +const FPSIMD_MAGIC: u32 = 0x4650_8001; + +const _: () = assert!(size_of::() == 528); +const _: () = assert!(size_of::() <= 4096); + +/// Builds `sigcontext.__reserved`'s context-record chain holding `fp`'s state +/// as a single `fpsimd_context` record, followed by nothing but zero bytes -- +/// a well-formed zero-`magic` terminator immediately after it, identical in +/// spirit to leaving the whole area zeroed (a well-formed *empty* chain). +fn fpsimd_reserved(fp: &litebox::platform::FpSimdState64) -> [u8; 4096] { + let record = FpsimdContext { + head: Aarch64CtxHeader { + magic: FPSIMD_MAGIC, + size: u32::try_from(size_of::()) + .expect("FpsimdContext's fixed 528-byte size fits comfortably in a u32"), + }, + fpsr: fp.fpsr, + fpcr: fp.fpcr, + vregs: fp.v, + }; + let mut reserved = [0u8; 4096]; + reserved[..size_of::()].copy_from_slice(record.as_bytes()); + reserved +} + +/// Reads an `fpsimd_context` record back out of `sigcontext.__reserved`, if +/// the chain's first record is genuinely one (checked by `magic`, matching +/// how a real kernel walks this chain). A guest that never touched this area, +/// or a handler that reconstructed its own frame without one, leaves no valid +/// record here -- reporting `None` rather than faulting keeps `rt_sigreturn` +/// tolerant of both, exactly as leaving the guest's FP state alone (neither +/// restored nor cleared) would be if this platform had never modelled FP +/// state at all. +fn parse_fpsimd_reserved(reserved: &[u8; 4096]) -> Option { + let header = + Aarch64CtxHeader::read_from_bytes(&reserved[..size_of::()]).ok()?; + if header.magic != FPSIMD_MAGIC || (header.size as usize) < size_of::() { + return None; + } + let record = FpsimdContext::read_from_bytes(&reserved[..size_of::()]).ok()?; + Some(litebox::platform::FpSimdState64 { + v: record.vregs, + fpsr: record.fpsr, + fpcr: record.fpcr, + }) +} + +/// The kernel's `struct rt_sigframe` for aarch64. +#[repr(C)] +#[derive(Clone, FromBytes, IntoBytes)] +struct SignalFrame { + siginfo: Siginfo, + ucontext: Ucontext, +} + +/// The kernel's `struct frame_record`: the saved frame pointer and link +/// register an unwinder follows to step from the handler back into the +/// interrupted frame. +#[repr(C)] +#[derive(Clone, FromBytes, IntoBytes)] +struct FrameRecord { + fp: usize, + lr: usize, +} + +/// The frame is placed at a 16-byte-aligned `sp` and the frame record sits +/// immediately above it, so the frame's own size has to preserve that +/// alignment. `sys_rt_sigreturn` rejects a misaligned `sp` outright. +const _: () = assert!(size_of::().is_multiple_of(16)); +const _: () = assert!(size_of::().is_multiple_of(16)); + +/// State recorded for a thread that has taken no exception yet. +pub(super) const NO_EXCEPTION: ExceptionInfo = ExceptionInfo { + exception: Exception(0), + fault_address: 0, + esr: 0, + kernel_mode: false, +}; + +/// Maps an aarch64 exception class to the signal Linux raises for it, together +/// with the address reported in the accompanying `si_addr`. +pub(super) fn exception_signal(info: &ExceptionInfo) -> (Signal, usize) { + let signal = match info.exception { + // A `BRK` or a hardware breakpoint is a debug trap. + Exception::BRK64 | Exception::BREAKPOINT_LOWER_EL | Exception::BREAKPOINT_CURRENT_EL => { + Signal::SIGTRAP + } + // Class 0 is "unknown reason", which is what an undefined instruction + // raises; the kernel's `do_el0_undef` turns it into SIGILL. A trapped + // system-register access lands in the same place. + Exception::UNKNOWN | Exception::SYSTEM_REGISTER_TRAP => Signal::SIGILL, + Exception::FP_EXCEPTION_A64 => Signal::SIGFPE, + // Aborts and anything unclassified become SIGSEGV, mirroring how the + // x86-64 path treats page faults and unknown vectors. There may be a + // more appropriate signal in some cases (e.g., SIGBUS for an alignment + // fault), which needs the abort's fault-status code to distinguish. + _ => Signal::SIGSEGV, + }; + let fault_address = match info.exception { + Exception::DATA_ABORT_LOWER_EL + | Exception::DATA_ABORT_CURRENT_EL + | Exception::INSTRUCTION_ABORT_LOWER_EL + | Exception::INSTRUCTION_ABORT_CURRENT_EL => info.fault_address, + _ => 0, + }; + (signal, fault_address) +} + +pub(super) fn uctx_addr(ctx: &PtRegs) -> usize { + // `sp` points at the whole frame, whose first member is the `siginfo`. + ctx.sp.wrapping_add(offset_of!(SignalFrame, ucontext)) +} + +pub(super) fn sp(ctx: &PtRegs) -> usize { + ctx.sp +} + +pub(super) fn get_signal_frame(sp: usize, _action: &SigAction) -> usize { + // Reserve the frame record at the top, then the frame below it. Both sizes + // are 16-byte multiples (asserted above), so a 16-aligned result stays + // aligned all the way down. + let next_frame = sp.wrapping_sub(size_of::()) & !15; + next_frame.wrapping_sub(size_of::()) +} + +/// Address of the frame record belonging to the frame at `frame_addr`. +fn frame_record_addr(frame_addr: usize) -> usize { + frame_addr.wrapping_add(size_of::()) +} + +impl SignalState { + pub(super) fn write_signal_frame( + &self, + platform: &Platform, + frame_addr: usize, + siginfo: &Siginfo, + action: &SigAction, + ctx: &mut PtRegs, + ) -> Result<(), DeliverFault> { + // The kernel falls back to the vDSO's `sigtramp` when the guest + // supplies no `sa_restorer`. LiteBox exposes no vDSO to the guest, but + // a platform can provide its own equivalent trampoline (see + // `SystemInfoProvider::get_sigreturn_trampoline_address`'s doc + // comment) -- fall back to that, and only refuse delivery (better + // than entering the handler with a wild `x30`, matching the x86-64 + // path) when the platform has no trampoline to offer either. + let restorer = if action.flags.contains(SaFlags::RESTORER) { + action.restorer + } else { + platform + .get_sigreturn_trampoline_address() + .ok_or(DeliverFault)? + }; + + let mut regs = [0u64; AARCH64_GENERAL_REGISTER_COUNT]; + for (slot, value) in regs.iter_mut().zip(ctx.regs.iter()) { + *slot = *value as u64; + } + + let last_exception = self.last_exception.get(); + let frame = SignalFrame { + siginfo: siginfo.clone(), + ucontext: Ucontext { + flags: 0, + link: 0, // core::ptr::null_mut() + stack: self.altstack.get(), + sigmask: self.blocked.get(), + __unused: [0; 1024 / 8 - size_of::()], + __align_pad: [0; 8], + mcontext: Sigcontext { + fault_address: last_exception.fault_address as u64, + regs, + sp: ctx.sp as u64, + pc: ctx.pc as u64, + pstate: ctx.pstate, + __reserved_pad: [0; 8], + // A real `fpsimd_context` record holding the guest's + // current vector state, followed by a well-formed + // zero-`magic` terminator (see `fpsimd_reserved`). + __reserved: fpsimd_reserved(&platform.get_fp_state()), + }, + }, + }; + + let frame_ptr = UserPtrMut::from_usize(frame_addr); + frame_ptr + .write_at_offset::(0, frame) + .ok_or(DeliverFault)?; + + let record_addr = frame_record_addr(frame_addr); + let record_ptr = UserPtrMut::::from_usize(record_addr); + record_ptr + .write_at_offset::( + 0, + FrameRecord { + fp: ctx.regs[29], + lr: ctx.regs[30], + }, + ) + .ok_or(DeliverFault)?; + + ctx.sp = frame_addr; + ctx.pc = action.sigaction; + ctx.regs[0] = siginfo.signo.reinterpret_as_unsigned() as usize; + if action.flags.contains(SaFlags::SIGINFO) { + ctx.regs[1] = frame_addr.wrapping_add(offset_of!(SignalFrame, siginfo)); + ctx.regs[2] = frame_addr.wrapping_add(offset_of!(SignalFrame, ucontext)); + } + ctx.regs[29] = record_addr; + ctx.regs[30] = restorer; + Ok(()) + } +} + +pub(super) fn restore_sigcontext( + platform: &Platform, + ctx: &mut PtRegs, + sigctx: &Sigcontext, +) -> usize { + let Sigcontext { + fault_address: _, + ref regs, + sp, + pc, + pstate, + __reserved_pad: _, + ref __reserved, + } = *sigctx; + + // A handler may have inspected or modified its frame's vector state + // before calling `sigreturn` (e.g. fixing up an FP exception); restore + // whatever is genuinely there. A frame with no valid `fpsimd_context` + // record (never written by `write_signal_frame`, or a handler that built + // its own frame from scratch) leaves the guest's FP state untouched, + // matching this platform's behavior before any of this was modelled. + if let Some(fp) = parse_fpsimd_reserved(__reserved) { + platform.set_fp_state(&fp); + } + + for (slot, value) in ctx.regs.iter_mut().zip(regs.iter()) { + *slot = (*value).trunc(); + } + ctx.sp = sp.trunc(); + ctx.pc = pc.trunc(); + // Keep only the PSTATE bits a guest is allowed to own. Everything else -- + // exception level, execution state, mask bits, illegal-state, single-step -- + // is imposed by the ABI, which is what the kernel's `valid_user_regs` check + // enforces on this path. + ctx.pstate = pstate & litebox_common_linux::arch::SAFE_USER_PSTATE; + // Returning from a handler leaves no syscall in flight, so no restart logic + // should re-issue the interrupted call. + ctx.syscallno = NO_SYSCALL; + + ctx.regs[0] +} diff --git a/litebox_shim_linux/src/syscalls/signal/mod.rs b/litebox_shim_linux/src/syscalls/signal/mod.rs index b793a35fd6..d0a59782ef 100644 --- a/litebox_shim_linux/src/syscalls/signal/mod.rs +++ b/litebox_shim_linux/src/syscalls/signal/mod.rs @@ -3,9 +3,13 @@ //! Signal handling syscalls and support. +#[cfg(target_arch = "aarch64")] +mod aarch64; #[cfg(target_arch = "x86_64")] mod x86_64; +#[cfg(target_arch = "aarch64")] +use aarch64 as arch; use litebox_common_linux::signal::SignalDisposition; #[cfg(target_arch = "x86_64")] use x86_64 as arch; @@ -16,11 +20,13 @@ use crate::{ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut}; use alloc::collections::vec_deque::VecDeque; use alloc::sync::Arc; use core::cell::{Cell, RefCell}; -use litebox::{shim::Exception, sync::Mutex, utils::ReinterpretUnsignedExt as _}; +use litebox::{sync::Mutex, utils::ReinterpretUnsignedExt as _}; use litebox_common_linux::signal::{ - MINSIGSTKSZ, NSIG, SI_KERNEL, SI_USER, SIG_DFL, SIG_IGN, SaFlags, SigAction, SigAltStack, + MINSIGSTKSZ, NSIG, SI_KERNEL, SI_TKILL, SI_USER, SIG_DFL, SIG_IGN, SaFlags, SigAction, + SigAltStack, SigSet, Siginfo, SiginfoData, SigmaskHow, Signal, SsFlags, Ucontext, }; +use litebox::event::wait::WaitError; use litebox_common_linux::{PtRegs, errno::Errno}; pub(crate) struct SignalState { @@ -36,6 +42,24 @@ pub(crate) struct SignalState { altstack: Cell, /// The last exception info recorded for signal delivery. last_exception: Cell, + /// The signal mask to put back once the signal that ended an `rt_sigsuspend` has been + /// delivered. + /// + /// `rt_sigsuspend(2)` installs a temporary mask, blocks, and must run the handler that woke + /// it *under that temporary mask* -- restoring the caller's mask any earlier would re-block + /// the very signal the caller was waiting for, and the guest would spin calling + /// `rt_sigsuspend` forever. Linux solves this with `saved_sigmask` plus + /// `TIF_RESTORE_SIGMASK`; this is that saved mask, and + /// [`Task::restore_saved_signal_mask`] is the restore, run once signals have been processed + /// on the way back to guest code. + saved_blocked: Cell>, + /// The set an in-progress `rt_sigtimedwait` is waiting for: Linux's `real_blocked`. + /// + /// While the wait lasts these signals count as deliverable for wakeup purposes even though + /// they stay in `blocked` (so a `SIG_IGN`'d member is queued rather than discarded, exactly + /// as `sig_ignored()` consults `real_blocked`), and the waiter dequeues them itself instead + /// of `process_signals`. Empty outside such a wait. + sigwait_set: Cell, } impl SignalState { @@ -49,15 +73,12 @@ impl SignalState { sp: 0, flags: SsFlags::DISABLE, size: 0, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }), - last_exception: Cell::new(litebox::shim::ExceptionInfo { - exception: litebox::shim::Exception(0), - error_code: 0, - cr2: 0, - kernel_mode: false, - }), + last_exception: Cell::new(arch::NO_EXCEPTION), + saved_blocked: Cell::new(None), + sigwait_set: Cell::new(SigSet::empty()), } } @@ -76,12 +97,41 @@ impl SignalState { flags: SsFlags::DISABLE, sp: 0, size: 0, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, } .into(), // Preserve last exception last_exception: self.last_exception.clone(), + saved_blocked: Cell::new(None), + sigwait_set: Cell::new(SigSet::empty()), + } + } + + /// Returns the signal state a `fork`ed child starts with. + /// + /// Unlike [`Self::clone_for_new_task`], which models `CLONE_THREAD` and therefore keeps the + /// process-wide parts shared, a new process gets private copies: its own pending queues (a + /// child does not inherit pending signals) and its own handler table (so a later + /// `rt_sigaction` in either process cannot be seen by the other). The blocked mask *is* + /// inherited, as `fork(2)` specifies. + pub fn clone_for_new_process(&self) -> Self { + Self { + pending: RefCell::new(PendingSignals::new()), + shared_pending: Arc::new(Mutex::new(PendingSignals::new())), + blocked: Cell::new(self.blocked.get()), + handlers: RefCell::new(Arc::new((**self.handlers.borrow()).clone())), + altstack: SigAltStack { + flags: SsFlags::DISABLE, + sp: 0, + size: 0, + #[cfg(target_pointer_width = "64")] + __pad: 0, + } + .into(), + last_exception: Cell::new(arch::NO_EXCEPTION), + saved_blocked: Cell::new(None), + sigwait_set: Cell::new(SigSet::empty()), } } @@ -101,7 +151,7 @@ impl SignalState { restorer: 0, flags: SaFlags::empty(), mask: SigSet::empty(), - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }; } @@ -109,6 +159,51 @@ impl SignalState { } } +/// A handle for posting a process-directed signal to a *different* guest process. +/// +/// The sending thread cannot touch the target's [`SignalState`] -- that is full of `Cell`s owned +/// by the target's own host thread -- but the process-wide pending queue behind it is an +/// ordinary `Arc>` and is safe to push into from anywhere. Whether the signal is +/// actually deliverable is decided by the target, on its own thread, in +/// [`Task::process_signals`] and [`Task::has_pending_signals`], because only it can read its live +/// handler table. +pub(crate) struct RemoteSignalTarget { + shared_pending: Arc>, +} + +impl Clone for RemoteSignalTarget { + fn clone(&self) -> Self { + Self { + shared_pending: self.shared_pending.clone(), + } + } +} + +impl RemoteSignalTarget { + /// Queues a shim-generated `siginfo` on the target process. + /// + /// # Panics + /// + /// Panics unless `signal` is a standard (non-realtime) signal with a kernel-originated + /// `si_code`. Those are the ones Linux exempts from `RLIMIT_SIGPENDING` + /// (`__send_signal_locked`'s `override_rlimit`), and exempting them is what lets this bypass + /// the target's rlimits -- which the sender cannot read anyway, since they live in the + /// target's `Process`. + pub(crate) fn post(&self, signal: Signal, siginfo: Siginfo) { + assert!(!signal.is_rt_signal() && siginfo.code >= 0); + self.shared_pending.lock().push_from_kernel(signal, siginfo); + } + + pub(crate) fn post_from_user( + &self, + limits: &super::process::ResourceLimits, + signal: Signal, + siginfo: Siginfo, + ) { + self.shared_pending.lock().push(limits, signal, siginfo); + } +} + struct SignalHandlers { inner: Mutex, } @@ -156,7 +251,7 @@ impl SignalHandlers { restorer: 0, flags: SaFlags::empty(), mask: SigSet::empty(), - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }, immutable: i == SignalHandlersInner::sig_index(Signal::SIGKILL) @@ -175,7 +270,7 @@ impl Clone for SignalHandlers { } } -struct PendingSignals { +pub(crate) struct PendingSignals { /// The set of pending signals. pending: SigSet, /// The queue of pending siginfo structures. @@ -183,7 +278,7 @@ struct PendingSignals { } impl PendingSignals { - fn new() -> Self { + pub(crate) fn new() -> Self { Self { pending: SigSet::empty(), queue: VecDeque::new(), @@ -231,7 +326,27 @@ impl PendingSignals { self.queue.remove(pos).unwrap() } - fn push(&mut self, rlimits: &super::process::ResourceLimits, signal: Signal, siginfo: Siginfo) { + /// Queues a standard signal generated by the shim itself, with no `RLIMIT_SIGPENDING` check. + /// + /// Linux applies that limit only to signals a *user* queued (`si_code < 0`, e.g. `SI_QUEUE`) + /// and to realtime signals; a kernel-generated `SIGCHLD` is never dropped for it. The + /// standard-signal dedup below means at most one such entry can be outstanding anyway. + fn push_from_kernel(&mut self, signal: Signal, siginfo: Siginfo) { + assert_eq!(signal.as_i32(), siginfo.signo); + assert!(!signal.is_rt_signal()); + if self.pending.contains(signal) { + return; + } + self.queue.push_back(siginfo); + self.pending.add(signal); + } + + pub(crate) fn push( + &mut self, + rlimits: &super::process::ResourceLimits, + signal: Signal, + siginfo: Siginfo, + ) { assert_eq!(signal.as_i32(), siginfo.signo); // Don't queue duplicates for standard signals. @@ -250,6 +365,23 @@ impl PendingSignals { self.queue.push_back(siginfo); self.pending.add(signal); } + + /// Moves every entry queued here into `dest`, preserving standard-signal dedup (an entry + /// whose signal is already pending in `dest` is dropped, matching `push`'s own dedup rule -- + /// this can only apply to standard signals, since `push`'s realtime-signal branch never + /// hits `pending.contains` at all). Does not re-apply an `RLIMIT_SIGPENDING` check: the + /// limit was already enforced when each entry was queued here. + pub(crate) fn drain_into(&mut self, dest: &mut Self) { + for siginfo in self.queue.drain(..) { + let signal = Signal::try_from(siginfo.signo).expect("queued an invalid signal"); + if !signal.is_rt_signal() && dest.pending.contains(signal) { + continue; + } + dest.queue.push_back(siginfo); + dest.pending.add(signal); + } + self.pending = SigSet::empty(); + } } /// Returns whether `sp` is within the given signal stack. @@ -268,7 +400,7 @@ fn siginfo_exception(signal: Signal, fault_address: usize) -> Siginfo { signo: signal.as_i32(), errno: 0, code: SI_KERNEL, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, data: SiginfoData::new_addr(fault_address), } @@ -281,12 +413,43 @@ pub(crate) fn siginfo_kill(signal: Signal) -> Siginfo { signo: signal.as_i32(), errno: 0, code: SI_USER, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] + __pad: 0, + data: SiginfoData::new_zeroed(), + } +} + +/// Creates the kernel-originated signal requested through `PR_SET_PDEATHSIG`. +pub(crate) fn siginfo_parent_death(signal: Signal) -> Siginfo { + Siginfo { + signo: signal.as_i32(), + errno: 0, + code: SI_KERNEL, + #[cfg(target_pointer_width = "64")] __pad: 0, data: SiginfoData::new_zeroed(), } } +/// Creates the `SIGCHLD` a parent gets when one of its children becomes a zombie. +pub(crate) fn siginfo_child_exited(child: i32, status: ExitStatus) -> Siginfo { + let (code, status) = match status { + ExitStatus::Exit(code) => ( + litebox_common_linux::signal::CLD_EXITED, + i32::from(code) & 0xff, + ), + ExitStatus::Signal(signal) => (litebox_common_linux::signal::CLD_KILLED, signal.as_i32()), + }; + Siginfo { + signo: Signal::SIGCHLD.as_i32(), + errno: 0, + code, + #[cfg(target_pointer_width = "64")] + __pad: 0, + data: SiginfoData::new_child(child, 0, status), + } +} + impl SignalState { /// Updates the blocked signal mask. fn set_signal_mask(&self, mask: SigSet) { @@ -313,7 +476,7 @@ impl SignalState { sp: ss.sp, flags: ss.flags & SsFlags::AUTODISARM, size: ss.size, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }); Ok(()) @@ -326,13 +489,14 @@ impl SignalState { sp: 0, flags: SsFlags::DISABLE, size: 0, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }); } fn deliver_signal( &self, + platform: &Platform, signal: Signal, siginfo: &Siginfo, action: &SigAction, @@ -356,7 +520,7 @@ impl SignalState { return Err(DeliverFault); } - self.write_signal_frame(frame_addr, siginfo, action, ctx)?; + self.write_signal_frame(platform, frame_addr, siginfo, action, ctx)?; let mut mask = self.blocked.get() | action.mask; if !action.flags.contains(SaFlags::NODEFER) { @@ -425,6 +589,71 @@ impl Task { Ok(0) } + /// Handle syscall `rt_sigsuspend`. + /// + /// Installs `mask_ptr` as the blocked set, blocks until a signal that is *not* in it becomes + /// deliverable, and always fails with `EINTR` -- `rt_sigsuspend(2)` has no success return. + /// + /// The caller's original mask is not put back here. It is stashed in + /// [`SignalState::saved_blocked`] and restored by [`Task::restore_saved_signal_mask`] after + /// the return path has delivered the signal that ended the wait, so that the handler runs + /// under the temporary mask exactly as Linux specifies. Restoring it here instead would + /// re-block the awaited signal before its handler could observe it, which is precisely the + /// livelock busybox's `ash` hits: its `waitproc` loops + /// `while (!got_sigchld && !pending_sig) sigsuspend(&mask);`, and `got_sigchld` is only ever + /// set by the `SIGCHLD` handler. + pub(crate) fn sys_rt_sigsuspend( + &self, + mask_ptr: Option>, + sigsetsize: usize, + ) -> Result { + if sigsetsize != core::mem::size_of::() { + return Err(Errno::EINVAL); + } + let mask = mask_ptr + .ok_or(Errno::EFAULT)? + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + // `SIGKILL` and `SIGSTOP` cannot be blocked, here or anywhere else. + let mask = { + let mut mask = mask; + mask.remove(Signal::SIGKILL); + mask.remove(Signal::SIGSTOP); + mask + }; + + let previous = self.signals.blocked.get(); + // A nested `rt_sigsuspend` (only reachable from a signal handler) must not lose the + // outermost caller's mask, so keep the first one stashed. + if self.signals.saved_blocked.get().is_none() { + self.signals.saved_blocked.set(Some(previous)); + } + self.signals.set_signal_mask(mask); + + // A `SIGCHLD` posted by an exiting child in another host thread reaches this task's + // pending set directly, but nothing would nudge *this* thread out of its wait. Registering + // here is what turns a child's exit into a wakeup; it is the same list `wait4` uses. + let table = &self.global.processes; + let token = table.register_waiter(self.pid, self.wait_cx().waker().clone()); + let _unregister = litebox::utils::defer(|| table.unregister_waiter(token)); + + // `wait_cx` interrupts on any deliverable signal or on task teardown, which is exactly + // the set of reasons `rt_sigsuspend` returns. The condition is never true on its own. + let _ = self.wait_cx().wait_until(|| false); + Err(Errno::EINTR) + } + + /// Puts back the mask an `rt_sigsuspend` replaced, if one is outstanding. + /// + /// Called from the return-to-guest path *after* `process_signals`, so the handler frame that + /// signal delivery just built captured the temporary mask. See + /// [`SignalState::saved_blocked`]. + pub(crate) fn restore_saved_signal_mask(&self) { + if let Some(previous) = self.signals.saved_blocked.take() { + self.signals.set_signal_mask(previous); + } + } + pub(crate) fn sys_sigaltstack( &self, ss_ptr: Option>, @@ -464,7 +693,11 @@ impl Task { self.signals.set_signal_mask(uctx.sigmask); - Ok(arch::restore_sigcontext(ctx, &uctx.mcontext)) + Ok(arch::restore_sigcontext( + self.global.platform, + ctx, + &uctx.mcontext, + )) } pub(crate) fn sys_rt_sigaction( @@ -509,8 +742,60 @@ impl Task { Ok(0) } + /// Handle syscall `kill`, with Linux's reading of `pid`: `> 0` names one process, `0` the + /// caller's process group, `-1` every process but the caller (and init), and `< -1` the + /// process group `-pid`. A group or broadcast target reports `ESRCH` only when nothing at all + /// matched. pub(crate) fn sys_kill(&self, pid: i32, signal: i32) -> Result { - self.do_kill(Some(pid), None, signal) + let processes = &self.global.processes; + let own_group = self.process().process_group_id(); + // `pid` is `i32::MIN` only for a group nothing can be in. + let target_group = (pid < -1).then(|| pid.checked_neg()).flatten(); + if signal == 0 { + let exists = match pid { + 0 => true, + -1 => processes.has_other_live_process(self.pid), + pid if pid < -1 => target_group.is_some_and(|group| { + own_group == group || processes.has_process_group_member(group, self.pid) + }), + pid => pid == self.pid || processes.is_live(pid), + }; + return exists.then_some(0).ok_or(Errno::ESRCH); + } + let signal = Signal::try_from(signal)?; + let delivered = match pid { + 0 => { + self.send_shared_signal(signal, siginfo_kill(signal)); + processes.send_process_group_signal( + own_group, + self.pid, + signal, + siginfo_kill(signal), + ) + 1 + } + -1 => processes.send_signal_to_all_processes(self.pid, signal, siginfo_kill(signal)), + pid if pid < -1 => { + let Some(group) = target_group else { + return Err(Errno::ESRCH); + }; + let mut delivered = processes.send_process_group_signal( + group, + self.pid, + signal, + siginfo_kill(signal), + ); + if own_group == group { + self.send_shared_signal(signal, siginfo_kill(signal)); + delivered += 1; + } + delivered + } + pid if pid != self.pid => { + usize::from(processes.send_process_signal(pid, signal, siginfo_kill(signal))) + } + pid => return self.do_kill(Some(pid), None, signal.as_i32()), + }; + (delivered > 0).then_some(0).ok_or(Errno::ESRCH) } pub(crate) fn sys_tkill(&self, tid: i32, signal: i32) -> Result { @@ -521,26 +806,265 @@ impl Task { self.do_kill(Some(pid), Some(tid), signal) } + /// Handle syscall `rt_sigtimedwait`. + /// + /// Linux's `do_sigtimedwait`: dequeue a pending signal in `set` (thread-directed first, then + /// process-directed; the caller's block mask is irrelevant to the dequeue), and if none is + /// pending and `timeout` allows, sleep with `set` treated as unblocked until one arrives + /// (`EAGAIN` when the timeout runs out first, `EINTR` when a signal outside `set` wakes the + /// sleep instead). `SIGKILL`/`SIGSTOP` cannot be waited for. On success the signal number + /// is returned and its `siginfo` is copied out to `info`. + pub(crate) fn sys_rt_sigtimedwait( + &self, + set: Option>, + info: Option>, + timeout: litebox_common_linux::TimeParam, + sigsetsize: usize, + ) -> Result { + if sigsetsize != core::mem::size_of::() { + return Err(Errno::EINVAL); + } + let mut which = set + .ok_or(Errno::EFAULT)? + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + which.remove(Signal::SIGKILL); + which.remove(Signal::SIGSTOP); + // Read the timeout up front: a bad pointer is `EFAULT` even when a signal is ready. + let timeout = timeout.read::()?; + let wait_allowed = match timeout { + None => true, + Some(t) => !t.is_zero(), + }; + + if let Some((signal, siginfo)) = self.dequeue_signal_in(which) { + self.copy_siginfo_out(info, &siginfo)?; + return Ok(signal.as_i32().cast_unsigned() as usize); + } + if !wait_allowed { + return Err(Errno::EAGAIN); + } + + // Sleep with `which` acting as unblocked (see `SignalState::sigwait_set`): the arrival + // of any member interrupts the wait through `check_for_interrupt`, as does any other + // deliverable signal or task teardown. + self.signals.sigwait_set.set(which); + let _restore = litebox::utils::defer(|| self.signals.sigwait_set.set(SigSet::empty())); + let wait_cx = self.wait_cx(); + let wait_cx = match timeout { + Some(t) => wait_cx.with_timeout(t), + None => wait_cx, + }; + let outcome = wait_cx.sleep(); + if let Some((signal, siginfo)) = self.dequeue_signal_in(which) { + self.copy_siginfo_out(info, &siginfo)?; + return Ok(signal.as_i32().cast_unsigned() as usize); + } + match outcome { + WaitError::TimedOut => Err(Errno::EAGAIN), + WaitError::Interrupted => Err(Errno::EINTR), + } + } + + /// Dequeues the first pending signal that is a member of `which`, thread-directed queue + /// first (a remote `tkill` is drained in), then the process-wide queue -- the order + /// `process_signals` uses. Ignores the block mask, like Linux's `dequeue_signal` with the + /// caller's mask. + fn dequeue_signal_in(&self, which: SigSet) -> Option<(Signal, Siginfo)> { + self.thread_remote() + .drain_remote_signals_into(&mut self.signals.pending.borrow_mut()); + let not_wanted = !which; + { + let mut pending = self.signals.pending.borrow_mut(); + if let Some(signal) = pending.next(not_wanted) { + let siginfo = pending.remove(signal); + return Some((signal, siginfo)); + } + } + let mut shared = self.signals.shared_pending.lock(); + let signal = shared.next(not_wanted)?; + let siginfo = shared.remove(signal); + Some((signal, siginfo)) + } + + fn copy_siginfo_out( + &self, + info: Option>, + siginfo: &Siginfo, + ) -> Result<(), Errno> { + if let Some(info) = info { + info.write_at_offset::(0, siginfo.clone()) + .ok_or(Errno::EFAULT)?; + } + Ok(()) + } + + /// Reads and validates the caller-supplied `siginfo` of `rt_sigqueueinfo`/ + /// `rt_tgsigqueueinfo`: `si_signo` is forced to `sig`, and a kernel-looking `si_code` + /// (`>= 0`, or `SI_TKILL`) may only be sent to one's own process (`EPERM`), as Linux's + /// `do_rt_sigqueueinfo` checks. + fn read_queued_siginfo( + &self, + info: Option>, + sig: i32, + target_pid: i32, + ) -> Result { + let mut siginfo = info + .ok_or(Errno::EFAULT)? + .read_at_offset::(0) + .ok_or(Errno::EFAULT)?; + if (siginfo.code >= 0 || siginfo.code == SI_TKILL) && target_pid != self.pid { + return Err(Errno::EPERM); + } + siginfo.signo = sig; + Ok(siginfo) + } + + /// Handle syscall `rt_sigqueueinfo`: `kill(pid, sig)` carrying the caller's `siginfo`. + pub(crate) fn sys_rt_sigqueueinfo( + &self, + pid: i32, + sig: i32, + info: Option>, + ) -> Result { + let siginfo = self.read_queued_siginfo(info, sig, pid)?; + if sig == 0 { + // Existence/permission probe only, exactly as `kill(pid, 0)`. + return self.sys_kill(pid, 0); + } + let signal = Signal::try_from(sig)?; + if pid == self.pid { + self.send_shared_signal(signal, siginfo); + return Ok(0); + } + if pid <= 0 { + log_unsupported!("rt_sigqueueinfo to a process group"); + return Err(Errno::EPERM); + } + self.global + .processes + .send_process_signal(pid, signal, siginfo) + .then_some(0) + .ok_or(Errno::ESRCH) + } + + /// Handle syscall `rt_tgsigqueueinfo`: `tgkill(tgid, tid, sig)` carrying the caller's + /// `siginfo` -- how crashpad's handler re-raises a crash signal with its original + /// `siginfo` intact. + pub(crate) fn sys_rt_tgsigqueueinfo( + &self, + tgid: i32, + tid: i32, + sig: i32, + info: Option>, + ) -> Result { + if tgid <= 0 || tid <= 0 { + return Err(Errno::EINVAL); + } + let siginfo = self.read_queued_siginfo(info, sig, tgid)?; + if sig == 0 { + return self.do_kill(Some(tgid), Some(tid), 0).or_else(|err| { + // `do_kill` rejects signal 0 as `EINVAL`; the probe form only needs existence. + if err == Errno::EINVAL { + self.tgkill_target_exists(tgid, tid) + .then_some(0) + .ok_or(Errno::ESRCH) + } else { + Err(err) + } + }); + } + let signal = Signal::try_from(sig)?; + if tgid != self.pid { + let Some((remote, limits)) = self.global.processes.remote_thread(tgid, tid) else { + return Err(Errno::ESRCH); + }; + remote.deliver_remote_signal(&limits, signal, siginfo); + return Ok(0); + } + if tid == self.tid { + self.send_signal(signal, siginfo); + return Ok(0); + } + let Some(remote) = self.process().thread_remote(tid) else { + return Err(Errno::ESRCH); + }; + remote.deliver_remote_signal(&self.process().limits, signal, siginfo); + Ok(0) + } + + fn tgkill_target_exists(&self, tgid: i32, tid: i32) -> bool { + if tgid == self.pid { + tid == self.tid || self.process().thread_remote(tid).is_some() + } else { + self.global.processes.remote_thread(tgid, tid).is_some() + } + } + fn do_kill(&self, pid: Option, tid: Option, signal: i32) -> Result { let signal = Signal::try_from(signal)?; - if pid.is_none_or(|pid| pid == self.pid) && tid.is_none_or(|tid| tid == self.tid) { + if let Some(pid) = pid + && pid != self.pid + { + // `tgkill` at another process's thread: crashpad's handler does this to the crashed + // client's threads. Delivered thread-directed through that thread's remote queue, + // exactly as a sibling's is below. + let tid = tid.unwrap_or(pid); + let Some((remote, limits)) = self.global.processes.remote_thread(pid, tid) else { + return Err(Errno::ESRCH); + }; + remote.deliver_remote_signal(&limits, signal, siginfo_kill(signal)); + return Ok(0); + } + let Some(tid) = tid else { self.send_signal(signal, siginfo_kill(signal)); - Ok(0) - } else { - log_unsupported!("sys_{{t|tg}}kill with remote pid/tid"); - Err(Errno::ESRCH) + return Ok(0); + }; + if tid == self.tid { + self.send_signal(signal, siginfo_kill(signal)); + return Ok(0); } + // A sibling thread's `SignalState.pending` is a bare `RefCell`, not `Send`/`Sync` -- + // only reachable from the thread it belongs to -- so delivery goes through + // `ThreadRemote::remote_pending`, the one piece of a thread's signal state built to be + // touched remotely. `thread_remote` returns `None` once the target has detached + // (exited), which is exactly when a real `tkill` would also report ESRCH: Linux never + // resurrects a reaped tid. + let Some(remote) = self.process().thread_remote(tid) else { + return Err(Errno::ESRCH); + }; + remote.deliver_remote_signal(&self.process().limits, signal, siginfo_kill(signal)); + Ok(0) } /// Returns whether there are any pending signals that can be delivered. + /// + /// A signal whose disposition is "ignore" does not count. It is pending only in the sense + /// that [`Task::process_signals`] has not got round to discarding it yet, and treating it as + /// deliverable would make it interrupt waits (`check_for_interrupt`) and hand the guest a + /// spurious `EINTR` from a syscall that nothing actually interrupted. Linux never queues such + /// a signal in the first place; this is where that is enforced, rather than at the sending + /// end, because a sender in another guest process cannot see the target's live handler table. pub(crate) fn has_pending_signals(&self) -> bool { - let blocked = self.signals.blocked.get(); + self.thread_remote() + .drain_remote_signals_into(&mut self.signals.pending.borrow_mut()); + // A signal an `rt_sigtimedwait` is waiting for must wake the wait even while blocked. + let blocked = self.signals.blocked.get() & !self.signals.sigwait_set.get(); let thread_pending = self.signals.pending.borrow().pending & !blocked; - if !thread_pending.is_empty() { - return true; - } let shared_pending = self.signals.shared_pending.lock().pending & !blocked; - !shared_pending.is_empty() + let pending = thread_pending | shared_pending; + if pending.is_empty() { + return false; + } + let handlers = self.signals.handlers.borrow(); + let inner = handlers.inner.lock(); + pending + .into_iter() + .any(|signal| match inner[signal].action.sigaction { + SIG_IGN => false, + SIG_DFL => !matches!(signal.default_disposition(), SignalDisposition::Ignore), + _ => true, + }) } /// Returns the set of all pending (deliverable) signals. @@ -554,6 +1078,8 @@ impl Task { /// Deliver any pending signals. pub(crate) fn process_signals(&self, ctx: &mut PtRegs) { + self.thread_remote() + .drain_remote_signals_into(&mut self.signals.pending.borrow_mut()); loop { let blocked = self.signals.blocked.get(); let (signal, siginfo) = { @@ -602,9 +1128,13 @@ impl Task { } SIG_IGN => {} _ => { - if let Err(DeliverFault) = - self.signals.deliver_signal(signal, &siginfo, &action, ctx) - { + if let Err(DeliverFault) = self.signals.deliver_signal( + self.global.platform, + signal, + &siginfo, + &action, + ctx, + ) { // Failed to deliver signal. Inject a SIGSEGV // (terminating the process if we were trying to deliver // a SIGSEGV). @@ -658,8 +1188,9 @@ impl Task { return false; } // Blocked signals are never ignored, since the signal handler may - // change by the time it is unblocked. - if self.signals.blocked.get().contains(signal) { + // change by the time it is unblocked. Nor is a signal an `rt_sigtimedwait` is + // waiting for (Linux checks `real_blocked` here too). + if (self.signals.blocked.get() | self.signals.sigwait_set.get()).contains(signal) { return false; } let handlers = self.signals.handlers.borrow(); @@ -671,6 +1202,13 @@ impl Task { } } + /// Returns a handle other guest processes can use to post a signal to this one. + pub(crate) fn remote_signal_target(&self) -> RemoteSignalTarget { + RemoteSignalTarget { + shared_pending: self.signals.shared_pending.clone(), + } + } + /// Only supports sending signals to self for now. pub(crate) fn send_signal(&self, signal: Signal, siginfo: Siginfo) { if self.is_signal_ignored(signal) { @@ -699,7 +1237,7 @@ impl Task { signo: signal.as_i32(), errno: 0, code: SI_KERNEL, - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, data: SiginfoData::new_zeroed(), }; @@ -707,7 +1245,18 @@ impl Task { } fn force_signal_with_info(&self, signal: Signal, force_exit: bool, siginfo: Siginfo) { - assert!(matches!(signal, Signal::SIGKILL | Signal::SIGSEGV)); + // This function resets the handler to `SIG_DFL` when forcing delivery, + // so the signal must be fatal by default; otherwise the guest would + // never actually see it acted on. `handle_exception_request` reaches + // this with any signal `arch::exception_signal` can decode a hardware + // exception into -- not just `SIGSEGV` (e.g. `SIGILL` for an + // undefined instruction, `SIGTRAP` for a breakpoint, `SIGFPE` for a + // floating-point exception) -- so the check has to match on + // disposition rather than enumerate specific signals. + assert!(matches!( + signal.default_disposition(), + SignalDisposition::Core | SignalDisposition::Terminate + )); self.signals .pending @@ -730,7 +1279,7 @@ impl Task { restorer: 0, flags: SaFlags::empty(), mask: SigSet::empty(), - #[cfg(target_arch = "x86_64")] + #[cfg(target_pointer_width = "64")] __pad: 0, }; // Don't allow further changes to this action. @@ -739,21 +1288,54 @@ impl Task { } pub(crate) fn handle_exception_request(&self, info: &litebox::shim::ExceptionInfo) { - let signal = match info.exception { - Exception::DIVIDE_ERROR => Signal::SIGFPE, - Exception::BREAKPOINT => Signal::SIGTRAP, - Exception::INVALID_OPCODE => Signal::SIGILL, - // Page faults and unknown exceptions map to SIGSEGV. There may be - // more appropriate signals in some other cases (e.g., SIGBUS). - _ => Signal::SIGSEGV, - }; - // For page faults, provide the faulting address. - let fault_address = if info.exception == Exception::PAGE_FAULT { - info.cr2 - } else { - 0 - }; + // Decoding an exception vector into a signal is entirely architectural, + // so it lives alongside the rest of the per-architecture frame handling. + let (signal, fault_address) = arch::exception_signal(info); + litebox_util_log::error!( + info:? = info, + signal:? = signal, + fault_address:? = fault_address, + pid:% = self.pid, + tid:% = self.tid; + "guest hardware exception" + ); self.signals.last_exception.set(*info); self.force_signal_with_info(signal, false, siginfo_exception(signal, fault_address)); } } + +#[cfg(test)] +mod tests { + use super::*; + + extern crate std; + + #[test] + fn kill_zero_signals_only_the_callers_process_group() { + let caller = crate::syscalls::tests::init_platform(None); + let peer = caller + .global + .clone() + .new_test_task(caller.files.borrow().fs.clone()); + let outsider = caller + .global + .clone() + .new_test_task(caller.files.borrow().fs.clone()); + + caller.register_for_remote_signals(); + peer.register_for_remote_signals(); + outsider.register_for_remote_signals(); + assert_eq!(caller.sys_setpgid(0, 3131), Ok(())); + assert_eq!(peer.sys_setpgid(0, 3131), Ok(())); + assert_eq!(outsider.sys_setpgid(0, 4242), Ok(())); + assert_eq!(caller.sys_kill(0, 0), Ok(0)); + assert!(caller.pending_signal_set().is_empty()); + assert!(peer.pending_signal_set().is_empty()); + assert!(outsider.pending_signal_set().is_empty()); + + assert_eq!(caller.sys_kill(0, Signal::SIGUSR1.as_i32()), Ok(0)); + assert!(caller.pending_signal_set().contains(Signal::SIGUSR1)); + assert!(peer.pending_signal_set().contains(Signal::SIGUSR1)); + assert!(!outsider.pending_signal_set().contains(Signal::SIGUSR1)); + } +} diff --git a/litebox_shim_linux/src/syscalls/signal/x86_64.rs b/litebox_shim_linux/src/syscalls/signal/x86_64.rs index 692d2267c1..535c674241 100644 --- a/litebox_shim_linux/src/syscalls/signal/x86_64.rs +++ b/litebox_shim_linux/src/syscalls/signal/x86_64.rs @@ -5,10 +5,11 @@ use crate::ShimPlatform; use crate::UserPtrMut; use crate::syscalls::signal::{DeliverFault, SignalState}; use core::mem::offset_of; +use litebox::shim::{Exception, ExceptionInfo}; use litebox::utils::{ReinterpretUnsignedExt as _, TruncateExt as _}; use litebox_common_linux::{ PtRegs, - signal::{SaFlags, SigAction, Siginfo, Ucontext, x86_64::Sigcontext}, + signal::{SaFlags, SigAction, Siginfo, Signal, Ucontext, x86_64::Sigcontext}, }; use zerocopy::{FromBytes, IntoBytes}; @@ -20,6 +21,34 @@ struct SignalFrame { siginfo: Siginfo, } +/// State recorded for a thread that has taken no exception yet. +pub(super) const NO_EXCEPTION: ExceptionInfo = ExceptionInfo { + exception: Exception(0), + error_code: 0, + cr2: 0, + kernel_mode: false, +}; + +/// Maps an x86 exception vector to the signal Linux raises for it, together +/// with the address reported in the accompanying `si_addr`. +pub(super) fn exception_signal(info: &ExceptionInfo) -> (Signal, usize) { + let signal = match info.exception { + Exception::DIVIDE_ERROR => Signal::SIGFPE, + Exception::BREAKPOINT => Signal::SIGTRAP, + Exception::INVALID_OPCODE => Signal::SIGILL, + // Page faults and unknown exceptions map to SIGSEGV. There may be + // more appropriate signals in some other cases (e.g., SIGBUS). + _ => Signal::SIGSEGV, + }; + // Only a page fault carries a faulting address. + let fault_address = if info.exception == Exception::PAGE_FAULT { + info.cr2 + } else { + 0 + }; + (signal, fault_address) +} + pub(super) fn uctx_addr(ctx: &PtRegs) -> usize { ctx.rsp } @@ -45,8 +74,13 @@ pub(super) fn get_signal_frame(sp: usize, _action: &SigAction) -> usize { } impl SignalState { + /// `_platform` matches aarch64's signature so `mod.rs`'s single generic + /// call site works for both; unused here -- x86-64's `SA_RESTORER`-less + /// and FP/SIMD-state gaps (`fpstate: 0` below) are unrelated, unverified- + /// on-this-hardware gaps this pass deliberately leaves alone. pub(super) fn write_signal_frame( &self, + _platform: &Platform, frame_addr: usize, siginfo: &Siginfo, action: &SigAction, @@ -114,7 +148,10 @@ impl SignalState { } } -pub(super) fn restore_sigcontext( +/// `_platform`/`Platform` match aarch64's signature so `mod.rs`'s single +/// generic call site works for both; unused here -- see `write_signal_frame`. +pub(super) fn restore_sigcontext( + _platform: &Platform, ctx: &mut PtRegs, sigctx: &litebox_common_linux::signal::x86_64::Sigcontext, ) -> usize { diff --git a/litebox_shim_linux/src/syscalls/tests.rs b/litebox_shim_linux/src/syscalls/tests.rs index 5c13f1ea52..e3d6eacbf8 100644 --- a/litebox_shim_linux/src/syscalls/tests.rs +++ b/litebox_shim_linux/src/syscalls/tests.rs @@ -2,7 +2,10 @@ // Licensed under the MIT license. use litebox::fs::{FileSystem as _, Mode, OFlags}; -use litebox_common_linux::{AtFlags, EfdFlags, FcntlArg, FileDescriptorFlags, errno::Errno}; +use litebox_common_linux::{ + AtFlags, EfdFlags, FcntlArg, FileDescriptorFlags, FlockOperation, Timespec, UTIME_NOW, + UTIME_OMIT, errno::Errno, +}; use zerocopy::FromBytes as _; use crate::UserPtrMut; @@ -18,6 +21,8 @@ const TEST_TAR_FILE: &[u8] = include_bytes!("../../../litebox/src/fs/test.tar"); /// hard-wired to one. #[cfg(target_os = "linux")] pub(crate) use litebox_platform_linux_userland::LinuxUserland as TestPlatform; +#[cfg(target_os = "macos")] +pub(crate) use litebox_platform_macos_userland::MacOsUserland as TestPlatform; #[cfg(target_os = "windows")] pub(crate) use litebox_platform_windows_userland::WindowsUserland as TestPlatform; @@ -30,6 +35,10 @@ pub(crate) fn test_platform(tun_device_name: Option<&str>) -> &'static TestPlatf { TestPlatform::new(tun_device_name) } + #[cfg(target_os = "macos")] + { + TestPlatform::new(tun_device_name) + } #[cfg(target_os = "windows")] { let _ = tun_device_name; @@ -38,6 +47,52 @@ pub(crate) fn test_platform(tun_device_name: Option<&str>) -> &'static TestPlatf }) } +/// Serializes tests that map guest memory. +/// +/// Each test builds its own task with its own virtual-memory manager, but every +/// task in this binary maps into the one host address space, and a VMM models +/// only its own mappings. Two tests running at once therefore pick addresses +/// without seeing each other's, and the loser gets a collision. Holding this for +/// the duration of a mapping test makes the placement search meaningful again. +/// +/// This is only reliably visible on a host whose guest range overlaps the host's +/// own: arm64 macOS puts both above the 4 GiB `__PAGEZERO` floor, so collisions +/// are routine there and rare elsewhere. +static ADDRESS_SPACE: std::sync::Mutex<()> = std::sync::Mutex::new(()); + +/// Take the guest-address-space lock for the rest of the current test. A +/// poisoned lock is not a failure here: it only means some earlier test panicked +/// while holding it, and the address space is no less usable for that. +pub(crate) fn address_space_guard() -> std::sync::MutexGuard<'static, ()> { + ADDRESS_SPACE + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) +} + +/// Serializes tests that exercise real asynchronous-signal delivery: alarms, +/// timers, and anything that ends up checking or draining pending signals. +/// +/// Each test builds its own task, but every task in this binary shares the one +/// `TestPlatform`, and every `TestPlatform` installs the same real host signal +/// handlers into this one process -- `SIGINT`/`SIGALRM` (and, on macOS, the +/// timer-thread wakeup signal) land regardless of which test's task "owns" +/// them. `litebox_platform_macos_userland`'s pending-signal bitmap is now +/// per-thread rather than process-wide, so the specific race this guard was +/// first added for (two tasks racing to drain one shared bitmap) no longer +/// applies there; this mutex still serializes the coarser hazard of two tests' +/// real host signals landing on whichever test happens to be blocked in a +/// syscall at the time, which per-thread bitmap state does not by itself +/// prevent. +static ASYNC_SIGNAL: std::sync::Mutex<()> = std::sync::Mutex::new(()); + +/// Take the async-signal lock for the rest of the current test. A poisoned +/// lock is not a failure here, matching [`address_space_guard`]. +pub(crate) fn async_signal_guard() -> std::sync::MutexGuard<'static, ()> { + ASYNC_SIGNAL + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) +} + #[must_use] pub(crate) fn init_platform( tun_device_name: Option<&str>, @@ -103,15 +158,30 @@ fn test_fcntl() { let write_fd = i32::try_from(write_fd).unwrap(); check(write_fd, OFlags::WRONLY | OFlags::NONBLOCK, OFlags::WRONLY); - // Test eventfd - let eventfd = task + // Eventfd works without a broker via the local fallback backend (it used + // to fail with EIO here, which aborted Node at uv_loop_init). + let event_fd = task .sys_eventfd2( 0, EfdFlags::CLOEXEC | EfdFlags::SEMAPHORE | EfdFlags::NONBLOCK, ) - .expect("Failed to create eventfd"); - let eventfd = i32::try_from(eventfd).unwrap(); - check(eventfd, OFlags::RDWR | OFlags::NONBLOCK, OFlags::RDWR); + .expect("brokerless eventfd must fall back to the local backend"); + task.sys_close(i32::try_from(event_fd).unwrap()) + .expect("closing the eventfd"); + + // Regular (non-stdio) files carry no `StdioStatusFlags` metadata; SETFL on one must be a + // real-Linux-matching no-op rather than panicking. + let regular_fd = task + .sys_open( + "/fcntl_setfl_regular_file.txt", + OFlags::CREAT | OFlags::RDWR, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create regular file for SETFL no-op check"); + let regular_fd = i32::try_from(regular_fd).unwrap(); + task.sys_fcntl(regular_fd, FcntlArg::SETFL(OFlags::NONBLOCK)) + .expect("SETFL on a regular file should be a no-op, not panic"); + let _ = task.sys_close(regular_fd); // Test fcntl with DUPFD let fd = task @@ -264,6 +334,8 @@ fn test_getdent64() { "bar", "dev", "foo", + // `/proc`, mounted alongside `/dev` by `default_fs` (see `litebox::fs::proc`). + "proc", "test_file1.txt", "test_file2.txt" ] @@ -430,6 +502,7 @@ fn test_getdent64() { "bar", "dev", "foo", + "proc", "test_file1.txt", "test_file2.txt" ] @@ -633,6 +706,428 @@ fn test_unlinkat() { ); } +#[test] +fn test_chmod_fchmod_fchmodat_round_trip() { + let task = init_platform(None); + + let file_path = "/chmod_test_file.txt"; + let fd = task + .sys_open( + file_path, + OFlags::CREAT | OFlags::WRONLY, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create test file"); + let fd = i32::try_from(fd).unwrap(); + + // `chmod` via path. `chmod` has no wrapper of its own (see `sys_fchmodat`'s doc comment); the + // syscall dispatcher reaches it by constructing an `Fchmodat` request with `dirfd` forced to + // `AT_FDCWD`, so exercise that exact shape here. + task.sys_fchmodat( + litebox_common_linux::AT_FDCWD, + file_path, + 0o640, + AtFlags::empty(), + ) + .expect("chmod (via fchmodat + AT_FDCWD) should succeed"); + let stat = task.sys_stat(file_path).expect("stat should succeed"); + assert_eq!( + stat.st_mode & 0o7777, + 0o640, + "chmod should have set the new mode, read back via stat" + ); + + // `fchmod` via the still-open fd. + task.sys_fchmod(fd, 0o600).expect("fchmod should succeed"); + let stat = task.sys_fstat(fd).expect("fstat should succeed"); + assert_eq!( + stat.st_mode & 0o7777, + 0o600, + "fchmod should have set the new mode, read back via fstat" + ); + + // `fchmodat` with `AT_FDCWD` + a relative path. (This shim does not yet resolve a real, + // non-`AT_FDCWD` dirfd against a relative path -- see `resolve_path_at`'s `FsPath::FdRelative` + // arm -- a pre-existing limitation shared by every `*at` syscall, not something specific to + // this change.) + task.sys_chdir("/").unwrap(); + task.sys_fchmodat( + litebox_common_linux::AT_FDCWD, + "chmod_test_file.txt", + 0o755, + AtFlags::empty(), + ) + .expect("fchmodat should succeed"); + let stat = task.sys_stat(file_path).expect("stat should succeed"); + assert_eq!( + stat.st_mode & 0o7777, + 0o755, + "fchmodat should have set the new mode, read back via stat" + ); + + // An unrecognized flag is rejected. + assert_eq!( + task.sys_fchmodat( + litebox_common_linux::AT_FDCWD, + "chmod_test_file.txt", + 0o755, + AtFlags::AT_EMPTY_PATH + ), + Err(Errno::EINVAL) + ); + + // `fchmod` on a closed fd fails with `EBADF`. + task.sys_close(fd).unwrap(); + assert_eq!(task.sys_fchmod(fd, 0o600), Err(Errno::EBADF)); +} + +#[test] +fn test_utimensat_futimens_round_trip() { + let task = init_platform(None); + + let file_path = "/utime_test_file.txt"; + let fd = task + .sys_open( + file_path, + OFlags::CREAT | OFlags::WRONLY, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create test file"); + let fd = i32::try_from(fd).unwrap(); + + // Explicit atime/mtime via `utimensat`. + let atime = Timespec { + tv_sec: 1_000_000, + tv_nsec: 123, + }; + let mtime = Timespec { + tv_sec: 2_000_000, + tv_nsec: 456, + }; + task.sys_utimensat( + litebox_common_linux::AT_FDCWD, + file_path, + Some([atime, mtime]), + AtFlags::empty(), + ) + .expect("utimensat should succeed"); + let stat = task.sys_stat(file_path).unwrap(); + assert_eq!((stat.st_atime, stat.st_atime_nsec), (1_000_000, 123)); + assert_eq!((stat.st_mtime, stat.st_mtime_nsec), (2_000_000, 456)); + + // `UTIME_OMIT` on `atime` leaves it unchanged; an explicit `mtime` still applies. + let omit = Timespec { + tv_sec: 0, + tv_nsec: UTIME_OMIT, + }; + let mtime2 = Timespec { + tv_sec: 3_000_000, + tv_nsec: 789, + }; + task.sys_utimensat( + litebox_common_linux::AT_FDCWD, + file_path, + Some([omit, mtime2]), + AtFlags::empty(), + ) + .expect("utimensat with UTIME_OMIT should succeed"); + let stat = task.sys_stat(file_path).unwrap(); + assert_eq!( + (stat.st_atime, stat.st_atime_nsec), + (1_000_000, 123), + "UTIME_OMIT must leave atime unchanged" + ); + assert_eq!((stat.st_mtime, stat.st_mtime_nsec), (3_000_000, 789)); + + // `UTIME_NOW` (explicit, both fields) resolves against wall-clock time. + let before = task.real_time_as_duration_since_epoch(); + let now_ts = Timespec { + tv_sec: 0, + tv_nsec: UTIME_NOW, + }; + task.sys_utimensat( + litebox_common_linux::AT_FDCWD, + file_path, + Some([now_ts, now_ts]), + AtFlags::empty(), + ) + .expect("utimensat with UTIME_NOW should succeed"); + let after = task.real_time_as_duration_since_epoch(); + let stat = task.sys_stat(file_path).unwrap(); + let atime_secs = u64::try_from(stat.st_atime).unwrap(); + assert!( + atime_secs >= before.as_secs() && atime_secs <= after.as_secs(), + "UTIME_NOW should resolve to the current wall-clock time, got {atime_secs} \ + outside [{}, {}]", + before.as_secs(), + after.as_secs() + ); + // Copied to locals before comparing: `assert_eq!` takes references to its arguments, and a + // reference straight into a packed struct's field is unaligned (UB) even if never + // dereferenced -- see the tuple-literal comparisons above, which sidestep this by + // constructing a new, properly-aligned tuple value instead of referencing the field in place. + let (atime, mtime) = (stat.st_atime, stat.st_mtime); + assert_eq!( + atime, mtime, + "UTIME_NOW applied to both fields should produce matching timestamps" + ); + + // A `NULL` `times` pointer (`None`) also means "both UTIME_NOW". + task.sys_utimensat( + litebox_common_linux::AT_FDCWD, + file_path, + None, + AtFlags::empty(), + ) + .expect("utimensat with NULL times should succeed"); + + // `futimens`, reached (per the shim's syscall dispatcher) via a `NULL` pathname, operates on + // the fd directly rather than re-resolving a path. + let atime3 = Timespec { + tv_sec: 5_000_000, + tv_nsec: 111, + }; + let mtime3 = Timespec { + tv_sec: 6_000_000, + tv_nsec: 222, + }; + task.sys_futimens(fd, Some([atime3, mtime3])) + .expect("futimens should succeed"); + let stat = task.sys_fstat(fd).unwrap(); + assert_eq!((stat.st_atime, stat.st_atime_nsec), (5_000_000, 111)); + assert_eq!((stat.st_mtime, stat.st_mtime_nsec), (6_000_000, 222)); + + // An invalid (out-of-range, non-sentinel) `tv_nsec` is rejected. + let bad = Timespec { + tv_sec: 0, + tv_nsec: 1_000_000_000, + }; + assert_eq!( + task.sys_utimensat( + litebox_common_linux::AT_FDCWD, + file_path, + Some([bad, bad]), + AtFlags::empty() + ), + Err(Errno::EINVAL) + ); + + // `futimens` on a closed fd fails with `EBADF`. + task.sys_close(fd).unwrap(); + assert_eq!( + task.sys_futimens(fd, Some([atime3, mtime3])), + Err(Errno::EBADF) + ); +} + +#[test] +fn test_flock_shared_exclusive_contention() { + let task = init_platform(None); + + let file_path = "/flock_test_file.txt"; + let fd1 = task + .sys_open( + file_path, + OFlags::CREAT | OFlags::RDWR, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create test file"); + let fd1 = i32::try_from(fd1).unwrap(); + // An independent second open of the same file: real `flock` treats these as independent + // holders that can contend with each other, which is exactly what's being exercised here. + let fd2 = task + .sys_open(file_path, OFlags::RDWR, Mode::empty()) + .unwrap(); + let fd2 = i32::try_from(fd2).unwrap(); + + // Exclusive lock via fd1 succeeds uncontended. + task.sys_flock(fd1, FlockOperation::LOCK_EX) + .expect("LOCK_EX should succeed uncontended"); + + // A non-blocking exclusive attempt via fd2 fails: fd1 holds it exclusively. + assert_eq!( + task.sys_flock(fd2, FlockOperation::LOCK_EX | FlockOperation::LOCK_NB), + Err(Errno::EWOULDBLOCK) + ); + // A non-blocking shared attempt via fd2 fails too, for the same reason. + assert_eq!( + task.sys_flock(fd2, FlockOperation::LOCK_SH | FlockOperation::LOCK_NB), + Err(Errno::EWOULDBLOCK) + ); + + // Re-locking (converting) from the SAME holder never blocks on itself. + task.sys_flock(fd1, FlockOperation::LOCK_EX | FlockOperation::LOCK_NB) + .expect("re-affirming the exclusive lock we already hold must not block"); + task.sys_flock(fd1, FlockOperation::LOCK_SH | FlockOperation::LOCK_NB) + .expect("downgrading the exclusive lock we hold must not block"); + + // Now that fd1 only holds a shared lock, a second shared lock via fd2 succeeds concurrently. + task.sys_flock(fd2, FlockOperation::LOCK_SH | FlockOperation::LOCK_NB) + .expect("two shared holders should be able to coexist"); + + // But fd2 cannot upgrade to exclusive while fd1 still holds a shared lock too. + assert_eq!( + task.sys_flock(fd2, FlockOperation::LOCK_EX | FlockOperation::LOCK_NB), + Err(Errno::EWOULDBLOCK) + ); + + // Unlocking fd1 lets fd2 upgrade. + task.sys_flock(fd1, FlockOperation::LOCK_UN).unwrap(); + task.sys_flock(fd2, FlockOperation::LOCK_EX | FlockOperation::LOCK_NB) + .expect("fd2 should now be able to acquire exclusively"); + + // `LOCK_UN` on an fd that isn't (or is no longer) a holder is a harmless no-op. + task.sys_flock(fd1, FlockOperation::LOCK_UN).unwrap(); + + // An unrecognized operation is rejected. + assert_eq!( + task.sys_flock(fd1, FlockOperation::empty()), + Err(Errno::EINVAL) + ); + + task.sys_flock(fd2, FlockOperation::LOCK_UN).unwrap(); + task.sys_close(fd1).unwrap(); + task.sys_close(fd2).unwrap(); +} + +#[test] +fn test_flock_blocks_across_real_threads_and_wakes_on_unlock() { + fn join_with_timeout( + handle: std::thread::JoinHandle, + timeout: std::time::Duration, + thread_name: &str, + ) -> T { + let start = std::time::Instant::now(); + while !handle.is_finished() { + assert!( + start.elapsed() < timeout, + "{thread_name} timed out after {timeout:?}" + ); + std::thread::sleep(std::time::Duration::from_millis(1)); + } + handle.join().expect("{thread_name} panicked") + } + + let task = init_platform(None); + let file_path = "/flock_blocking_test_file.txt"; + let fd1 = task + .sys_open( + file_path, + OFlags::CREAT | OFlags::RDWR, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create test file"); + let fd1 = i32::try_from(fd1).unwrap(); + let fd2 = task + .sys_open(file_path, OFlags::RDWR, Mode::empty()) + .unwrap(); + let fd2 = i32::try_from(fd2).unwrap(); + + // The main "thread" (guest thread) holds an exclusive lock. + task.sys_flock(fd1, FlockOperation::LOCK_EX).unwrap(); + + // A second real guest thread, sharing the same fd table (via a real host thread), blocks + // trying to acquire the same file exclusively too. + let blocked = + task.spawn_clone_for_test(move |task| task.sys_flock(fd2, FlockOperation::LOCK_EX)); + + // Give the second thread a real chance to actually block before we check it hasn't finished. + std::thread::sleep(std::time::Duration::from_millis(50)); + assert!( + !blocked.is_finished(), + "the second thread should still be blocked on the lock fd1 holds" + ); + + // Releasing the lock must wake the blocked waiter. + task.sys_flock(fd1, FlockOperation::LOCK_UN).unwrap(); + let result = join_with_timeout( + blocked, + std::time::Duration::from_secs(5), + "blocked flock waiter", + ); + assert_eq!( + result, + Ok(()), + "the blocked LOCK_EX call should succeed once fd1 releases the lock" + ); + + task.sys_flock(fd2, FlockOperation::LOCK_UN).unwrap(); + task.sys_close(fd1).unwrap(); + task.sys_close(fd2).unwrap(); +} + +/// A holder converting its own exclusive lock down to shared can unblock a different, real thread +/// blocked wanting a shared lock -- without that holder ever calling `LOCK_UN`. Regression test +/// for a real bug caught during development: only `unlock` used to wake waiters, so this exact +/// scenario hung until the blocking party's *next* unrelated wake. +#[test] +fn test_flock_downgrade_wakes_a_different_blocked_waiter() { + fn join_with_timeout( + handle: std::thread::JoinHandle, + timeout: std::time::Duration, + thread_name: &str, + ) -> T { + let start = std::time::Instant::now(); + while !handle.is_finished() { + assert!( + start.elapsed() < timeout, + "{thread_name} timed out after {timeout:?}" + ); + std::thread::sleep(std::time::Duration::from_millis(1)); + } + handle.join().expect("{thread_name} panicked") + } + + let task = init_platform(None); + let file_path = "/flock_downgrade_test_file.txt"; + let fd1 = task + .sys_open( + file_path, + OFlags::CREAT | OFlags::RDWR, + Mode::RUSR | Mode::WUSR, + ) + .expect("Failed to create test file"); + let fd1 = i32::try_from(fd1).unwrap(); + let fd2 = task + .sys_open(file_path, OFlags::RDWR, Mode::empty()) + .unwrap(); + let fd2 = i32::try_from(fd2).unwrap(); + + // fd1 holds an exclusive lock. + task.sys_flock(fd1, FlockOperation::LOCK_EX).unwrap(); + + // A second real guest thread blocks wanting a *shared* lock via fd2. + let blocked = + task.spawn_clone_for_test(move |task| task.sys_flock(fd2, FlockOperation::LOCK_SH)); + + std::thread::sleep(std::time::Duration::from_millis(50)); + assert!( + !blocked.is_finished(), + "fd2 should still be blocked while fd1 holds the lock exclusively" + ); + + // fd1 downgrades its own lock to shared -- never calls LOCK_UN. Real flock(2) treats this as + // an in-place conversion, and it should immediately make room for fd2's shared request. + task.sys_flock(fd1, FlockOperation::LOCK_SH) + .expect("downgrading the lock we hold must not block"); + + let result = join_with_timeout( + blocked, + std::time::Duration::from_secs(5), + "blocked LOCK_SH waiter", + ); + assert_eq!( + result, + Ok(()), + "fd1's downgrade to shared should have woken fd2's blocked LOCK_SH" + ); + + task.sys_flock(fd1, FlockOperation::LOCK_UN).unwrap(); + task.sys_flock(fd2, FlockOperation::LOCK_UN).unwrap(); + task.sys_close(fd1).unwrap(); + task.sys_close(fd2).unwrap(); +} + /// Regression test for a bug where readers can be permanently starved on /// platforms where `wake_one` does not report whether it actually woke a thread /// (e.g. Windows with `WakeByAddressSingle`). diff --git a/litebox_shim_linux/src/syscalls/unix.rs b/litebox_shim_linux/src/syscalls/unix.rs index ca45f122a8..0c40c8a93d 100644 --- a/litebox_shim_linux/src/syscalls/unix.rs +++ b/litebox_shim_linux/src/syscalls/unix.rs @@ -9,6 +9,7 @@ use core::{ }; use alloc::{ + boxed::Box, collections::{btree_map::BTreeMap, vec_deque::VecDeque}, string::String, sync::{Arc, Weak}, @@ -21,19 +22,22 @@ use litebox::{ wait::WaitContext, }, fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}, - fs::{Mode, OFlags, errors::OpenError}, + fs::{AccessCredentials, Mode, OFlags, errors::OpenError}, sync::{Mutex, RwLock}, utils::TruncateExt as _, }; use litebox_common_linux::{ - IpOption, ReceiveFlags, SendFlags, ShutdownHow, SockFlags, SockType, SocketOption, - SocketOptionName, errno::Errno, + ReceiveFlags, SendFlags, ShutdownHow, SockFlags, SockType, SocketOption, SocketOptionName, + Ucred, errno::Errno, }; use crate::{ FileFd, GlobalState, ShimFS, ShimPlatform, Task, UserPtr, UserPtrMut, channel::{Channel, ReadEnd, WriteEnd}, - syscalls::net::{SocketOptionValue, SocketOptions}, + syscalls::{ + file::TransferredFd, + net::{SocketOptionValue, SocketOptions}, + }, }; pub(crate) struct UnixSocketSubsystem( @@ -80,69 +84,287 @@ enum UnixBoundSocketAddr { /// /// This is used internally to track which addresses are currently bound /// by listening sockets. -#[derive(PartialEq, Eq, Hash, Debug, Ord, PartialOrd)] +#[derive(PartialEq, Eq, Hash, Debug, Ord, PartialOrd, Clone)] pub(crate) enum UnixSocketAddrKey { // TODO: add inode reference once the file system supports it. Path(String), Abstract(Vec), } +/// Marker used purely for `Arc::ptr_eq` identity of a placeholder reserved +/// in the shared Unix address table. Carries no data of its own. +struct ReservationToken; + +/// Next candidate abstract name handed out by autobind, masked to the same +/// 20-bit range Linux draws `sun_path[1..]` from. The mask alone can't give +/// collision-freedom (it wraps every 2^20 binds), so the picking loop in +/// `UnixSocketAddr::bind_and_reserve` retries through `reserve_unix_addr` +/// against the live table on every draw. +static AUTOBIND_COUNTER: AtomicU32 = AtomicU32::new(0); + +/// The `SO_PEERCRED` view of a task: its process id and *effective* ids, as +/// Linux's `cred_to_ucred` reports them. +fn task_ucred(task: &Task) -> Ucred { + let credentials = task.credentials.borrow(); + Ucred { + pid: task.pid.cast_unsigned(), + uid: credentials.euid, + gid: credentials.egid, + } +} + +/// The `SCM_CREDENTIALS` a task implicitly attaches to what it sends: its +/// process id and *real* ids, as Linux's `maybe_add_creds` captures them +/// (`task_tgid(current)` + `current_uid_gid`). +fn task_send_ucred(task: &Task) -> Ucred { + let credentials = task.credentials.borrow(); + Ucred { + pid: task.pid.cast_unsigned(), + uid: credentials.uid, + gid: credentials.gid, + } +} + +/// What a `SO_PASSCRED` receiver sees when no sender credentials are +/// attached to what it read (EOF, for instance): pid 0 and the overflow +/// ids, matching Linux's `from_kuid_munged(INVALID_UID)`. +const UNSET_UCRED: Ucred = Ucred { + pid: 0, + uid: 65534, + gid: 65534, +}; + +/// An exclusive claim on one key in the shared Unix address table, held +/// from `bind()` time until the socket either upgrades it into a real +/// listening/datagram entry (`upgrade`) or is dropped without ever doing +/// so, at which point `Drop` releases the placeholder. +/// +/// This is what makes address-collision detection atomic and total: every +/// real bind path (autobind, explicit path, explicit abstract) claims +/// through the same `reserve_unix_addr`, under one write-lock critical +/// section covering both the check and the insert, so a bound-but-not-yet- +/// listening socket is exactly as visible to a colliding bind as a fully +/// listening one -- and two concurrent binds to the same address can never +/// both observe it as free. +struct UnixAddrReservation { + key: UnixSocketAddrKey, + token: Arc, + global: Arc>, +} + +impl UnixAddrReservation { + /// Atomically replaces this reservation's placeholder in the shared + /// table with the finished entry, consuming the reservation. Nothing + /// else can observe or touch a `Reserved` slot except through this same + /// token (see `reserve_unix_addr` and this type's `Drop`), so the slot + /// is guaranteed to still hold our own placeholder here. + fn upgrade(self, entry: UnixEntryInner) { + let mut table = self.global.unix_addr_table.write(); + if let Some(slot) = table.get_mut(&self.key) { + debug_assert!( + matches!(&slot.0, UnixEntryInner::Reserved(current) if Arc::ptr_eq(current, &self.token)), + "unix_addr_table slot changed out from under an unconsumed reservation" + ); + slot.0 = entry; + } else { + debug_assert!(false, "unix_addr_table reservation missing at upgrade time"); + } + // `table`'s write lock is released here, before `self` (and its own + // `Drop`) runs at the end of this function -- that `Drop` will find + // the slot no longer `Reserved` and no-op. + } +} + +impl Drop for UnixAddrReservation { + fn drop(&mut self) { + let mut table = self.global.unix_addr_table.write(); + if let Some(UnixEntry(UnixEntryInner::Reserved(current))) = table.get(&self.key) + && Arc::ptr_eq(current, &self.token) + { + table.remove(&self.key); + } + } +} + +/// A bound address together with its (optional) shared-table reservation. +/// `None` for path addresses -- see `UnixSocketAddr::bind_and_reserve`. +type BoundUnixAddr = ( + UnixBoundSocketAddr, + Option>, +); + +/// Atomically checks and claims `key` in the shared Unix address table: a +/// single write-lock acquisition covers both the presence check and the +/// insert, so no other bind can observe the key as free in between (the +/// check-then-act race a read-then-separate-write split would allow). +/// Released quickly -- callers must not hold this call's result across +/// unrelated I/O (e.g. filesystem access) any longer than necessary, so +/// unrelated binds to other keys are never blocked behind it. +fn reserve_unix_addr( + global: &Arc>, + key: UnixSocketAddrKey, +) -> Result, Errno> { + let mut table = global.unix_addr_table.write(); + if table.contains_key(&key) { + return Err(Errno::EADDRINUSE); + } + let token = Arc::new(ReservationToken); + table.insert( + key.clone(), + UnixEntry(UnixEntryInner::Reserved(token.clone())), + ); + drop(table); + Ok(UnixAddrReservation { + key, + token, + global: global.clone(), + }) +} + +/// Mode bits used when creating (or merely reopening) a Unix socket path +/// file. Mirrors the permissions Linux itself grants a freshly bound socket +/// inode. +fn unix_socket_file_mode() -> Mode { + Mode::RWXU | Mode::RGRP | Mode::XGRP | Mode::ROTH | Mode::XOTH +} + impl UnixSocketAddr { /// Returns true if this is an unnamed socket address. fn is_unnamed(&self) -> bool { matches!(self, UnixSocketAddr::Unnamed) } - /// Binds this address to the filesystem or abstract namespace. + /// Validates that `self` is reachable as a `connect()` target, mirroring + /// Linux's own permission/existence check on the peer's path. Performs + /// no reservation and never touches the shared address table -- the + /// peer's own `bind` already owns whatever entry exists there, and this + /// is only a read-only check on our end. /// - /// # Arguments + /// # Errors /// - /// * `task` - The current task context - /// * `is_server` - Whether this is a server socket (creates the file if true) + /// Returns an error if the address cannot be reached (e.g., file + /// doesn't exist, permission denied). + fn check_reachable( + self, + task: &Task, + ) -> Result, Errno> { + match self { + UnixSocketAddr::Path(path) => { + // TODO: extend fs to support creating sock file (i.e., with type `InodeType::Socket`) + let credential_snapshot = task.credentials.borrow().clone(); + let credentials = AccessCredentials::new( + credential_snapshot.euid, + credential_snapshot.egid, + credential_snapshot.supplementary_groups(), + ); + let fs = task.files.borrow().fs.clone(); + let file = fs + .open_as( + credentials, + path.as_str(), + OFlags::RDWR, + unix_socket_file_mode(), + ) + .map_err(|err| match err { + OpenError::AlreadyExists => Errno::EADDRINUSE, + other => Errno::from(other), + })?; + Ok(UnixBoundSocketAddr::Path((path, file, fs))) + } + UnixSocketAddr::Abstract(data) => Ok(UnixBoundSocketAddr::Abstract(data)), + // Nothing legitimately connects to an unnamed address. + UnixSocketAddr::Unnamed => Err(Errno::EINVAL), + } + } + + /// Claims `self` exclusively. This is the real `bind(2)` path, shared by + /// stream `bind` and datagram `bind` -- every real bind goes through + /// this single function, so every real bind is subject to the same + /// collision check. + /// + /// For abstract (and autobound) addresses, which have no filesystem + /// backing, the returned `Some(reservation)` atomically reserves the + /// address in the shared table (see `reserve_unix_addr`); the caller + /// must upgrade it into a real entry (`UnixAddrReservation::upgrade`) + /// once the bind is otherwise complete. + /// + /// For path addresses the return is `None`: collision detection for + /// paths is the filesystem's own `O_EXCL` create, keyed on the path + /// *string* resolving to an inode, not on any copy of that string held + /// elsewhere -- the same reason two listening sockets can legitimately + /// share one path string over time (bind, unlink, bind again) while the + /// first is still alive, elsewhere. The shared table has no inode + /// identity to key on yet (see `UnixSocketAddrKey`'s own `TODO`), so + /// routing it through `reserve_unix_addr` would reject that legitimate + /// unlink-and-rebind sequence just because the *string* is still + /// present from the first, now-unreachable-by-path bind. The caller + /// inserts unconditionally for this case, exactly as it did before this + /// reservation scheme existed. /// /// # Errors /// - /// Returns an error if the address cannot be bound (e.g., file doesn't exist, - /// permission denied). - fn bind( + /// Returns an error if the address is already in use, or (for path + /// addresses) cannot be created. + fn bind_and_reserve( self, task: &Task, - is_server: bool, - ) -> Result, Errno> { + ) -> Result, Errno> { match self { UnixSocketAddr::Path(path) => { - let flags = if is_server { - // create the socket file if not exists; - // use O_EXCL to ensure exclusive creation - OFlags::CREAT | OFlags::EXCL | OFlags::RDWR - } else { - OFlags::RDWR - }; // TODO: extend fs to support creating sock file (i.e., with type `InodeType::Socket`) - let file = task - .files - .borrow() - .fs - .open( + let credential_snapshot = task.credentials.borrow().clone(); + let credentials = AccessCredentials::new( + credential_snapshot.euid, + credential_snapshot.egid, + credential_snapshot.supplementary_groups(), + ); + let fs = task.files.borrow().fs.clone(); + let file = fs + .open_as( + credentials, path.as_str(), - flags, - Mode::RWXU | Mode::RGRP | Mode::XGRP | Mode::ROTH | Mode::XOTH, + OFlags::CREAT | OFlags::EXCL | OFlags::RDWR, + unix_socket_file_mode(), ) .map_err(|err| match err { OpenError::AlreadyExists => Errno::EADDRINUSE, other => Errno::from(other), })?; - Ok(UnixBoundSocketAddr::Path(( - path, - file, - task.files.borrow().fs.clone(), - ))) + Ok((UnixBoundSocketAddr::Path((path, file, fs)), None)) } UnixSocketAddr::Abstract(data) => { - // TODO: check if the abstract address is already in use - Ok(UnixBoundSocketAddr::Abstract(data)) + let reservation = + reserve_unix_addr(&task.global, UnixSocketAddrKey::Abstract(data.clone()))?; + Ok((UnixBoundSocketAddr::Abstract(data), Some(reservation))) + } + UnixSocketAddr::Unnamed => { + // Autobind: draw a fresh candidate from Linux's 20-bit + // abstract namespace and atomically reserve it, retrying + // past a collision. Each candidate's check-and-reserve is + // its own short critical section (see `reserve_unix_addr`) + // -- the retry loop deliberately never holds the table lock + // across attempts, so unrelated binds are never blocked + // behind an autobind search. + for _ in 0..=0xFFFFFu32 { + let name = AUTOBIND_COUNTER.fetch_add(1, Ordering::Relaxed) & 0xFFFFF; + let candidate = alloc::format!("{name:05x}").into_bytes(); + match reserve_unix_addr( + &task.global, + UnixSocketAddrKey::Abstract(candidate.clone()), + ) { + Ok(reservation) => { + return Ok(( + UnixBoundSocketAddr::Abstract(candidate), + Some(reservation), + )); + } + // Try the next candidate. + Err(Errno::EADDRINUSE) => {} + Err(other) => return Err(other), + } + } + Err(Errno::ENOSPC) } - UnixSocketAddr::Unnamed => todo!("autobind for unnamed unix socket"), } } @@ -188,13 +410,24 @@ impl From<&UnixBoundSocketAddr> for UnixSocketAddr { } } +/// A rejected `UnixInitStream`, handed back to the caller alongside the +/// `Errno` that rejected it (so the caller can keep using the socket, e.g. +/// on a failed `listen()` or `connect()`). Boxed because `UnixInitStream` +/// carries its own `BoundUnixAddr`, which makes the pair large enough that +/// `Result`'s error variant would otherwise dominate the type's size. +type InitRejection = Box<(UnixInitStream, Errno)>; + /// Represents a Unix stream socket in its initial state. /// /// This is the state immediately after socket creation, before the socket /// has been connected, or put into listening mode. struct UnixInitStream { - /// Optional bound address for this socket - addr: Option>, + /// The bound address and its table reservation (if any -- path + /// addresses have none, see `UnixSocketAddr::bind_and_reserve`), set + /// together by `bind` and released together -- kept as one field + /// (rather than two separate top-level `Option`s) so the two can never + /// go out of sync with each other. + bound: Option>, pollee: Pollee, read_shutdown: AtomicBool, write_shutdown: AtomicBool, @@ -203,7 +436,7 @@ struct UnixInitStream { impl UnixInitStream { fn new() -> Self { Self { - addr: None, + bound: None, pollee: Pollee::new(), read_shutdown: AtomicBool::new(false), write_shutdown: AtomicBool::new(false), @@ -221,12 +454,11 @@ impl UnixInitStream { /// Binds this socket to the given address. fn bind(&mut self, task: &Task, addr: UnixSocketAddr) -> Result<(), Errno> { - if self.addr.is_some() && !addr.is_unnamed() { + if self.bound.is_some() && !addr.is_unnamed() { return Err(Errno::EINVAL); } - if self.addr.is_none() { - let bound_addr = addr.bind(task, true)?; - self.addr = Some(bound_addr); + if self.bound.is_none() { + self.bound = Some(addr.bind_and_reserve(task)?); } Ok(()) } @@ -240,16 +472,27 @@ impl UnixInitStream { self, backlog: u16, global: &Arc>, - ) -> Result, (Self, Errno)> { - let Some(addr) = self.addr else { - return Err((self, Errno::EINVAL)); + listener_cred: Ucred, + ) -> Result, InitRejection> { + let Some((addr, reservation)) = self.bound else { + return Err(Box::new((self, Errno::EINVAL))); }; let key = addr.to_key(); - let backlog = Arc::new(Backlog::new(addr, backlog, self.pollee)); - global - .unix_addr_table - .write() - .insert(key, UnixEntry(UnixEntryInner::Stream(backlog.clone()))); + let backlog = Arc::new(Backlog::new(addr, backlog, self.pollee, listener_cred)); + if let Some(reservation) = reservation { + // Upgrade the existing reservation in place instead of + // inserting a fresh table entry -- the slot has been ours, + // exclusively, since `bind` reserved it. + reservation.upgrade(UnixEntryInner::Stream(backlog.clone())); + } else { + // Path addresses were never reserved through the table at bind + // time (see `bind_and_reserve`) -- insert unconditionally, + // exactly as this did before the reservation scheme existed. + global + .unix_addr_table + .write() + .insert(key, UnixEntry(UnixEntryInner::Stream(backlog.clone()))); + } Ok(UnixListenStream { backlog, global: global.clone(), @@ -259,23 +502,38 @@ impl UnixInitStream { /// Converts this initial socket into a connected stream pair. fn into_connected( self, + self_cred: Ucred, peer_addr: Arc>, + peer_cred: Ucred, ) -> ( UnixConnectedStream, UnixConnectedStream, ) { let UnixInitStream { - addr, + bound, pollee, read_shutdown, write_shutdown, } = self; + let (addr, reservation) = match bound { + Some((addr, reservation)) => (Some(Arc::new(addr)), reservation), + None => (None, None), + }; + // The reservation (if this socket explicitly bound before + // connecting -- the client-role autobind-then-connect pattern) + // carries into the connected stream rather than being dropped here, + // so the bound address stays claimed for the connection's whole + // lifetime, matching real Unix domain socket semantics. UnixConnectedStream::new_pair( - addr.map(Arc::new), + addr, + self_cred, Some(Arc::new(pollee)), Some(peer_addr), + peer_cred, + reservation, read_shutdown.load(Ordering::Acquire), write_shutdown.load(Ordering::Acquire), + false, ) } } @@ -286,6 +544,7 @@ impl UnixInitStream { struct Backlog { /// The address this socket is listening on addr: Arc>, + listener_cred: Ucred, state: Mutex>, pollee: Pollee, } @@ -298,9 +557,15 @@ struct BacklogState { } impl Backlog { - fn new(addr: UnixBoundSocketAddr, backlog: u16, pollee: Pollee) -> Self { + fn new( + addr: UnixBoundSocketAddr, + backlog: u16, + pollee: Pollee, + listener_cred: Ucred, + ) -> Self { Self { addr: Arc::new(addr), + listener_cred, state: litebox::sync::Mutex::new(BacklogState { sockets: VecDeque::new(), limit: backlog, @@ -319,17 +584,19 @@ impl Backlog { fn try_connect( &self, init: UnixInitStream, - ) -> Result, (UnixInitStream, Errno)> { + client_cred: Ucred, + ) -> Result, InitRejection> { let mut state = self.state.lock(); if state.is_shutdown { - return Err((init, Errno::ECONNREFUSED)); + return Err(Box::new((init, Errno::ECONNREFUSED))); } if state.sockets.len() >= state.limit as usize { - return Err((init, Errno::EAGAIN)); + return Err(Box::new((init, Errno::EAGAIN))); } - let (client, server) = init.into_connected(self.addr.clone()); + let (client, server) = + init.into_connected(client_cred, self.addr.clone(), self.listener_cred); state.sockets.push_back(server); self.pollee.notify_observers(Events::IN); @@ -453,20 +720,41 @@ impl AddrView { } /// A message sent over a Unix socket. -struct Message { +struct Message { data: Vec, - // TODO: add control messages - // cmsgs: Option>, + rights: Vec>, + /// The sender's credentials, captured when the message was queued + /// (explicit `SCM_CREDENTIALS` if the sender supplied one, else its + /// process id and real ids). + credentials: Ucred, +} + +pub(super) struct RecvResult { + pub(super) size: usize, + pub(super) rights: Vec>, + /// `Some` only when the receiving socket has `SO_PASSCRED` set: the + /// `SCM_CREDENTIALS` payload to deliver (the first consumed message's + /// sender credentials). + pub(super) credentials: Option, } /// Represents a connected Unix stream socket. struct UnixConnectedStream { addr: AddrView, + peer_cred: Ucred, /// The read end of the local socket's channel for receiving messages. - recv_channel: crate::channel::ReadEnd, + recv_channel: crate::channel::ReadEnd>, /// The write end of the connected peer socket for sending messages. - connected_send_channel: crate::channel::WriteEnd, + connected_send_channel: crate::channel::WriteEnd>, pollee: Arc>, + preserve_message_boundaries: bool, + /// Kept alive only for its `Drop` side effect: releases this stream's + /// own bound-address reservation (if it explicitly bound before + /// connecting) once the connection itself closes, not merely once it + /// stops listening -- matching real Unix domain socket semantics, where + /// a bound client address stays claimed for as long as the socket + /// exists. + _reservation: Option>, } const UNIX_BUF_SIZE: usize = 65536; @@ -475,13 +763,22 @@ impl UnixConnectedStream { /// /// `read_shutdown` and `write_shutdown` half-close the corresponding sides of the /// *first* returned socket only (used to carry pre-connect shutdown flags from - /// `UnixInitStream` across `connect(2)` into the connected state). + /// `UnixInitStream` across `connect(2)` into the connected state). `reservation` + /// (if any) belongs to the *first* returned socket only, matching `addr`. + #[expect( + clippy::too_many_arguments, + reason = "the arguments initialize distinct state for both socket endpoints" + )] fn new_pair( addr: Option>>, + self_cred: Ucred, pollee: Option>>, peer: Option>>, + peer_cred: Ucred, + reservation: Option>, read_shutdown: bool, write_shutdown: bool, + preserve_message_boundaries: bool, ) -> (Self, Self) { let (addr1, addr2) = AddrView::new_pair(addr, peer); let pollee1 = pollee.unwrap_or(Arc::new(Pollee::new())); @@ -492,15 +789,21 @@ impl UnixConnectedStream { crate::channel::Channel::new(UNIX_BUF_SIZE, pollee1.clone(), pollee2.clone()).split(); let first = UnixConnectedStream { addr: addr1, + peer_cred, recv_channel, connected_send_channel: send_channel_peer, pollee: pollee1, + preserve_message_boundaries, + _reservation: reservation, }; let second = UnixConnectedStream { addr: addr2, + peer_cred: self_cred, recv_channel: recv_channel_peer, connected_send_channel: send_channel, pollee: pollee2, + preserve_message_boundaries, + _reservation: None, }; if read_shutdown { first.recv_channel.shutdown(); @@ -525,39 +828,104 @@ impl UnixConnectedStream { } } - fn try_sendto(&self, msg: Message) -> Result<(), (Message, Errno)> { + fn try_sendto(&self, msg: Message) -> Result<(), (Message, Errno)> { // TODO: write partial data? self.connected_send_channel.try_write_one(msg) } - fn try_recvfrom(&self, mut buf: &mut [u8]) -> Result> { - let mut total_read = 0; + /// `want_credentials` is the receiving socket's `SO_PASSCRED`: when set, the + /// result carries the first consumed message's sender credentials and a + /// stream read stops at a message whose sender credentials differ from + /// them, as Linux's `unix_stream_read_generic` does (`unix_skb_scm_eq`), + /// so one `SCM_CREDENTIALS` always describes every byte returned. + fn try_recvfrom( + &self, + mut buf: &mut [u8], + want_credentials: bool, + ) -> Result, TryOpError> { + if buf.is_empty() { + return Ok(RecvResult { + size: 0, + rights: Vec::new(), + credentials: want_credentials.then_some(UNSET_UCRED), + }); + } + if self.preserve_message_boundaries { + return self + .recv_channel + .peek_and_consume_one(|msg| { + let message_len = msg.data.len(); + let copy_len = buf.len().min(message_len); + buf[..copy_len].copy_from_slice(&msg.data[..copy_len]); + Ok(( + true, + RecvResult { + size: message_len, + rights: core::mem::take(&mut msg.rights), + credentials: want_credentials.then_some(msg.credentials), + }, + )) + }) + .map_err(|e| match e { + Errno::EAGAIN => TryOpError::TryAgain, + other => TryOpError::Other(other), + }); + } + + let mut result = RecvResult { + size: 0, + rights: Vec::new(), + credentials: None, + }; while !buf.is_empty() { - let n = match self.recv_channel.peek_and_consume_one(|msg| { - if buf.len() >= msg.data.len() { - buf[..msg.data.len()].copy_from_slice(&msg.data); - Ok((true, msg.data.len())) - } else { - buf.copy_from_slice(&msg.data[..buf.len()]); - msg.data = msg.data.split_off(buf.len()); - Ok((false, buf.len())) - } - }) { - Ok(n) => n, - Err(e) => { - if total_read > 0 { - break; + let (n, mut rights, credentials, ancillary_barrier) = + match self.recv_channel.peek_and_consume_one(|msg| { + if want_credentials + && let Some(first) = result.credentials + && first != msg.credentials + { + // Credentials boundary: leave this message queued for + // the next read. + return Err(Errno::EAGAIN); } - return match e { - Errno::EAGAIN => Err(TryOpError::TryAgain), - other => Err(TryOpError::Other(other)), + let ancillary_barrier = !msg.rights.is_empty(); + let copy_len = buf.len().min(msg.data.len()); + buf[..copy_len].copy_from_slice(&msg.data[..copy_len]); + let rights = if copy_len == 0 { + Vec::new() + } else { + core::mem::take(&mut msg.rights) }; - } - }; - total_read += n; + let credentials = msg.credentials; + if copy_len == msg.data.len() { + Ok((true, (copy_len, rights, credentials, ancillary_barrier))) + } else { + msg.data = msg.data.split_off(copy_len); + Ok((false, (copy_len, rights, credentials, ancillary_barrier))) + } + }) { + Ok(value) => value, + Err(e) => { + if result.size > 0 { + break; + } + return match e { + Errno::EAGAIN => Err(TryOpError::TryAgain), + other => Err(TryOpError::Other(other)), + }; + } + }; + result.size += n; + result.rights.append(&mut rights); + if want_credentials && result.credentials.is_none() { + result.credentials = Some(credentials); + } buf = &mut buf[n..]; + if ancillary_barrier { + break; + } } - Ok(total_read) + Ok(result) } fn check_io_events(&self) -> Events { @@ -592,6 +960,13 @@ impl UnixConnectedStream { } } +impl Drop for UnixConnectedStream { + fn drop(&mut self) { + self.recv_channel.shutdown(); + self.connected_send_channel.shutdown(); + } +} + enum UnixStreamState { Init(UnixInitStream), Listen(UnixListenStream), @@ -664,13 +1039,21 @@ impl UnixStream { }) } - fn listen(&self, backlog: u16, global: &Arc>) -> Result<(), Errno> { + fn listen( + &self, + backlog: u16, + global: &Arc>, + listener_cred: Ucred, + ) -> Result<(), Errno> { self.with_state(|state| { let ret = match state { UnixStreamState::Init(init) => { - return match init.listen(backlog, global) { + return match init.listen(backlog, global, listener_cred) { Ok(listen) => (UnixStreamState::Listen(listen), Ok(())), - Err((init, err)) => (UnixStreamState::Init(init), Err(err)), + Err(boxed) => { + let (init, err) = *boxed; + (UnixStreamState::Init(init), Err(err)) + } }; } UnixStreamState::Listen(ref listen) => { @@ -698,13 +1081,23 @@ impl UnixStream { match &entry.0 { UnixEntryInner::Stream(backlog) => Ok(backlog.clone()), UnixEntryInner::Datagram(_) => Err(Errno::EPROTOTYPE), + // Bound but not (yet) listening: nothing is there to connect to, + // exactly like Linux's ECONNREFUSED for a non-listening peer. + UnixEntryInner::Reserved(_) => Err(Errno::ECONNREFUSED), } } - fn try_connect(&self, backlog: &Backlog) -> Result<(), TryOpError> { + fn try_connect( + &self, + backlog: &Backlog, + client_cred: Ucred, + ) -> Result<(), TryOpError> { self.with_state(|state| match state { - UnixStreamState::Init(init) => match backlog.try_connect(init) { + UnixStreamState::Init(init) => match backlog.try_connect(init, client_cred) { Ok(connected) => (UnixStreamState::Connected(connected), Ok(())), - Err((init, err)) => (UnixStreamState::Init(init), Err(err)), + Err(boxed) => { + let (init, err) = *boxed; + (UnixStreamState::Init(init), Err(err)) + } }, UnixStreamState::Listen(s) => (UnixStreamState::Listen(s), Err(Errno::EINVAL)), UnixStreamState::Connected(s) => (UnixStreamState::Connected(s), Err(Errno::EISCONN)), @@ -721,8 +1114,9 @@ impl UnixStream { is_nonblocking: bool, ) -> Result<(), Errno> { let backlog = self.lookup(task, &addr)?; - // check if we can bind to the address - let _ = addr.bind(task, false)?; + // check if we can reach the address + let _ = addr.check_reachable(task)?; + let client_cred = task_ucred(task); task.wait_cx() .wait_on_events( is_nonblocking, @@ -731,7 +1125,7 @@ impl UnixStream { backlog.pollee.register_observer(observer, mask); Ok(()) }, - || self.try_connect(&backlog), + || self.try_connect(&backlog, client_cred), ) .map_err(Errno::from) } @@ -779,11 +1173,15 @@ impl UnixStream { &self, cx: &WaitContext<'_, Platform>, timeout: Option, - buf: &[u8], + msg: Message, is_nonblocking: bool, addr: Option, ) -> Result { - let mut msg = Some(Message { data: buf.to_vec() }); + let len = msg.data.len(); + if len == 0 { + return Ok(0); + } + let mut msg = Some(msg); cx.with_timeout(timeout) .wait_on_events( is_nonblocking, @@ -804,7 +1202,7 @@ impl UnixStream { return Err(TryOpError::Other(Errno::EISCONN)); } match conn.try_sendto(msg.take().unwrap()) { - Ok(()) => Ok(buf.len()), + Ok(()) => Ok(len), Err((m, Errno::EAGAIN)) => { let _ = msg.replace(m); Err(TryOpError::TryAgain) @@ -823,8 +1221,9 @@ impl UnixStream { timeout: Option, buf: &mut [u8], is_nonblocking: bool, + want_credentials: bool, mut source_addr: Option<&mut Option>, - ) -> Result { + ) -> Result, Errno> { let res = cx .with_timeout(timeout) .wait_on_events( @@ -842,7 +1241,7 @@ impl UnixStream { let conn = state .connected() .ok_or(TryOpError::Other(Errno::ENOTCONN))?; - let n = conn.try_recvfrom(buf)?; + let n = conn.try_recvfrom(buf, want_credentials)?; // For connected stream sockets, no need to return the source address if let Some(source_addr) = source_addr.as_deref_mut() { *source_addr = None; @@ -862,9 +1261,11 @@ impl UnixStream { fn get_local_addr(&self) -> UnixSocketAddr { self.with_state_ref(|state| match state { UnixStreamState::Init(init) => init - .addr + .bound .as_ref() - .map_or(UnixSocketAddr::Unnamed, UnixSocketAddr::from), + .map_or(UnixSocketAddr::Unnamed, |(addr, _)| { + UnixSocketAddr::from(addr) + }), UnixStreamState::Listen(listen) => UnixSocketAddr::from(listen.get_local_addr()), UnixStreamState::Connected(connect) => connect.get_local_addr(), }) @@ -876,6 +1277,13 @@ impl UnixStream { }) } + fn get_peer_cred(&self) -> Option { + self.with_state_ref(|state| match state { + UnixStreamState::Init(_) | UnixStreamState::Listen(_) => None, + UnixStreamState::Connected(connect) => Some(connect.peer_cred), + }) + } + fn register_observer( &self, observer: Weak>, @@ -889,6 +1297,19 @@ impl UnixStream { } }); } + + fn unregister_observer(&self, observer: Weak>) { + self.with_state_ref(|state| match state { + UnixStreamState::Init(init) => init.pollee.unregister_observer(observer), + UnixStreamState::Listen(listen) => { + listen.backlog.pollee.unregister_observer(observer); + } + UnixStreamState::Connected(connect) => { + connect.pollee.unregister_observer(observer); + } + }); + } + fn check_io_events(&self) -> Events { self.with_state_ref(|state| match state { UnixStreamState::Init(init) => { @@ -924,9 +1345,11 @@ impl UnixStream { #[derive(Clone)] struct DatagramMessage { data: Vec, - // TODO: add control messages - // cmsgs: Option>, + // TODO: add SCM_RIGHTS source: UnixSocketAddr, + /// The sender's credentials, captured when the datagram was queued (see + /// `Message::credentials`). + credentials: Ucred, } impl WriteEnd { @@ -966,12 +1389,13 @@ impl ReadEnd { /// /// Reads exactly one message, preserving message boundaries. If the buffer /// is smaller than the message, the excess data is discarded (truncated). - /// Returns the original message size (which may exceed `buf.len()`). + /// Returns the original message size (which may exceed `buf.len()`) and + /// the sender's credentials. fn try_read( &self, buf: &mut [u8], mut source_addr: Option<&mut Option>, - ) -> Result> { + ) -> Result<(usize, Ucred), TryOpError> { let is_self_shutdown = self.is_shutdown(); self.peek_and_consume_one(|msg| { let copy_len = buf.len().min(msg.data.len()); @@ -980,7 +1404,7 @@ impl ReadEnd { *source_addr = Some(msg.source.clone()); } // Always consume the entire message to preserve boundaries. - Ok((true, msg.data.len())) + Ok((true, (msg.data.len(), msg.credentials))) }) .map_err(|e| match e { Errno::EAGAIN => TryOpError::TryAgain, @@ -1043,17 +1467,26 @@ impl UnixDatagramInner { return Err(Errno::EINVAL); } - let bound_addr = addr.bind(task, true)?; - let key = bound_addr.to_key(); + let (bound_addr, reservation) = addr.bind_and_reserve(task)?; // Registers the write end of the socket in the global address table so it // can receive messages sent to this address. let (send_channel, recv_channel) = Channel::new(UNIX_BUF_SIZE, Arc::new(Pollee::new()), self.pollee.clone()).split(); - let _ = task - .global - .unix_addr_table - .write() - .insert(key, UnixEntry(UnixEntryInner::Datagram(send_channel))); + if let Some(reservation) = reservation { + // Upgrade the reservation `bind_and_reserve` already atomically + // claimed, rather than inserting a fresh entry that could race + // a colliding bind. + reservation.upgrade(UnixEntryInner::Datagram(send_channel)); + } else { + // Path addresses were never reserved through the table at bind + // time (see `bind_and_reserve`) -- insert unconditionally, + // exactly as this did before the reservation scheme existed. + let key = bound_addr.to_key(); + task.global + .unix_addr_table + .write() + .insert(key, UnixEntry(UnixEntryInner::Datagram(send_channel))); + } self.addr = Some((bound_addr, task.global.clone())); if self.read_shutdown { recv_channel.shutdown(); @@ -1146,11 +1579,14 @@ impl UnixDatagram { let Some(entry) = guard.get(&key) else { return Err(Errno::ECONNREFUSED); }; - // check if we can bind to the address - let _ = addr.bind(task, false)?; + // check if we can reach the address + let _ = addr.check_reachable(task)?; match &entry.0 { UnixEntryInner::Stream(_) => Err(Errno::EPROTOTYPE), UnixEntryInner::Datagram(send_channel) => Ok(send_channel.clone()), + // Bound but not (yet) actually receiving: nothing is there to + // send to, matching Linux's ECONNREFUSED for an unreachable peer. + UnixEntryInner::Reserved(_) => Err(Errno::ECONNREFUSED), } } @@ -1174,7 +1610,7 @@ impl UnixDatagram { buf: &mut [u8], is_nonblocking: bool, mut source_addr: Option<&mut Option>, - ) -> Result { + ) -> Result<(usize, Ucred), Errno> { let res = cx .with_timeout(timeout) .wait_on_events( @@ -1213,6 +1649,7 @@ impl UnixDatagram { task: &Task, timeout: Option, buf: &[u8], + credentials: Ucred, is_nonblocking: bool, addr: Option, ) -> Result { @@ -1241,6 +1678,7 @@ impl UnixDatagram { DatagramMessage { data: buf.to_vec(), source, + credentials, }, is_nonblocking, )?; @@ -1306,22 +1744,32 @@ enum UnixSocketInner { } pub(crate) struct UnixSocket { inner: UnixSocketInner, + sock_type: SockType, status: AtomicU32, options: Mutex, } impl UnixSocket { - fn new_with_inner(inner: UnixSocketInner, flags: SockFlags) -> Self { + fn new_with_inner( + inner: UnixSocketInner, + sock_type: SockType, + flags: SockFlags, + ) -> Self { let mut status = OFlags::RDWR; status.set(OFlags::NONBLOCK, flags.contains(SockFlags::NONBLOCK)); Self { inner, + sock_type, status: AtomicU32::new(status.bits()), options: litebox::sync::Mutex::new(SocketOptions::default()), } } - pub(super) fn new(sock_type: SockType, flags: SockFlags) -> Option { + pub(super) fn new( + sock_type: SockType, + flags: SockFlags, + _task: &Task, + ) -> Option { let inner = match sock_type { SockType::Stream => UnixSocketInner::Stream(UnixStream::new(UnixStreamState::Init( UnixInitStream::new(), @@ -1332,7 +1780,7 @@ impl UnixSocket { return None; } }; - Some(Self::new_with_inner(inner, flags)) + Some(Self::new_with_inner(inner, sock_type, flags)) } pub(super) fn bind( @@ -1350,9 +1798,10 @@ impl UnixSocket { &self, backlog: u16, global: &Arc>, + task: &Task, ) -> Result<(), Errno> { match &self.inner { - UnixSocketInner::Stream(stream) => stream.listen(backlog, global), + UnixSocketInner::Stream(stream) => stream.listen(backlog, global, task_ucred(task)), UnixSocketInner::Datagram(_) => Err(Errno::EOPNOTSUPP), } } @@ -1384,7 +1833,7 @@ impl UnixSocket { self.get_status().contains(OFlags::NONBLOCK) | flags.contains(SockFlags::NONBLOCK), )?; - Ok(UnixSocket::new_with_inner(accepted, flags)) + Ok(UnixSocket::new_with_inner(accepted, self.sock_type, flags)) } UnixSocketInner::Datagram(_) => Err(Errno::EOPNOTSUPP), } @@ -1396,21 +1845,49 @@ impl UnixSocket { buf: &[u8], flags: SendFlags, addr: Option, + ) -> Result { + self.sendmsg(task, buf, flags, addr, Vec::new(), None) + } + + /// `credentials` is an explicit, already-validated `SCM_CREDENTIALS` the + /// sender attached; without one the sender's own process id and real ids + /// travel with the data, so a `SO_PASSCRED` receiver always learns who + /// sent what (Linux `maybe_add_creds`). + pub(super) fn sendmsg( + &self, + task: &Task, + buf: &[u8], + flags: SendFlags, + addr: Option, + rights: Vec>, + credentials: Option, ) -> Result { let supported_flags = SendFlags::DONTWAIT | SendFlags::NOSIGNAL; if flags.intersects(supported_flags.complement()) { - log_unsupported!("Unsupported sendto flags: {:?}", flags); + log_unsupported!("Unsupported sendmsg flags: {:?}", flags); return Err(Errno::EINVAL); } let is_nonblocking = flags.contains(SendFlags::DONTWAIT) || self.get_status().contains(OFlags::NONBLOCK); let timeout = self.options.lock().send_timeout; + let credentials = credentials.unwrap_or_else(|| task_send_ucred(task)); match &self.inner { - UnixSocketInner::Stream(stream) => { - stream.sendto(&task.wait_cx(), timeout, buf, is_nonblocking, addr) - } + UnixSocketInner::Stream(stream) => stream.sendto( + &task.wait_cx(), + timeout, + Message { + data: buf.to_vec(), + rights, + credentials, + }, + is_nonblocking, + addr, + ), UnixSocketInner::Datagram(datagram) => { - datagram.sendto(task, timeout, buf, is_nonblocking, addr) + if !rights.is_empty() { + return Err(Errno::EOPNOTSUPP); + } + datagram.sendto(task, timeout, buf, credentials, is_nonblocking, addr) } } } @@ -1422,24 +1899,48 @@ impl UnixSocket { flags: ReceiveFlags, source_addr: Option<&mut Option>, ) -> Result { + self.recvmsg(cx, buf, flags, source_addr) + .map(|result| result.size) + } + + pub(super) fn recvmsg( + &self, + cx: &WaitContext<'_, Platform>, + buf: &mut [u8], + flags: ReceiveFlags, + source_addr: Option<&mut Option>, + ) -> Result, Errno> { let supported_flags = ReceiveFlags::DONTWAIT | ReceiveFlags::TRUNC; if flags.intersects(supported_flags.complement()) { - log_unsupported!("Unsupported recvfrom flags: {:?}", flags); + log_unsupported!("Unsupported recvmsg flags: {:?}", flags); return Err(Errno::EINVAL); } let is_nonblocking = flags.contains(ReceiveFlags::DONTWAIT) || self.get_status().contains(OFlags::NONBLOCK); - let timeout = self.options.lock().recv_timeout; + let (timeout, pass_cred) = { + let options = self.options.lock(); + (options.recv_timeout, options.pass_cred) + }; let ret = match &self.inner { UnixSocketInner::Stream(stream) => { - stream.recvfrom(cx, timeout, buf, is_nonblocking, source_addr) - } - UnixSocketInner::Datagram(datagram) => { - datagram.recvfrom(cx, timeout, buf, is_nonblocking, source_addr) + stream.recvfrom(cx, timeout, buf, is_nonblocking, pass_cred, source_addr) } + UnixSocketInner::Datagram(datagram) => datagram + .recvfrom(cx, timeout, buf, is_nonblocking, source_addr) + .map(|(size, credentials)| RecvResult { + size, + rights: Vec::new(), + credentials: pass_cred.then_some(credentials), + }), }; match ret { - Err(Errno::ESHUTDOWN) => Ok(0), + // EOF: Linux still runs `scm_recv`, so a `SO_PASSCRED` receiver gets an + // `SCM_CREDENTIALS` naming nobody (pid 0, overflow ids). + Err(Errno::ESHUTDOWN) => Ok(RecvResult { + size: 0, + rights: Vec::new(), + credentials: pass_cred.then_some(UNSET_UCRED), + }), other => other, } } @@ -1460,17 +1961,31 @@ impl UnixSocket { pub(super) fn new_connected_pair( ty: SockType, flags: SockFlags, + task: &Task, ) -> Option<(UnixSocket, UnixSocket)> { match ty { - SockType::Stream => { - let (conn1, conn2) = UnixConnectedStream::new_pair(None, None, None, false, false); + SockType::Stream | SockType::SeqPacket => { + let cred = task_ucred(task); + let (conn1, conn2) = UnixConnectedStream::new_pair( + None, + cred, + None, + None, + cred, + None, + false, + false, + matches!(ty, SockType::SeqPacket), + ); Some(( UnixSocket::new_with_inner( UnixSocketInner::Stream(UnixStream::new(UnixStreamState::Connected(conn1))), + ty, flags, ), UnixSocket::new_with_inner( UnixSocketInner::Stream(UnixStream::new(UnixStreamState::Connected(conn2))), + ty, flags, ), )) @@ -1478,8 +1993,8 @@ impl UnixSocket { SockType::Datagram => { let (datagram1, datagram2) = UnixDatagram::new_pair(); Some(( - UnixSocket::new_with_inner(UnixSocketInner::Datagram(datagram1), flags), - UnixSocket::new_with_inner(UnixSocketInner::Datagram(datagram2), flags), + UnixSocket::new_with_inner(UnixSocketInner::Datagram(datagram1), ty, flags), + UnixSocket::new_with_inner(UnixSocketInner::Datagram(datagram2), ty, flags), )) } _ => None, @@ -1522,9 +2037,7 @@ impl UnixSocket { } match optname { - SocketOptionName::IP(ip) => match ip { - IpOption::TOS => Err(Errno::EOPNOTSUPP), - }, + SocketOptionName::IP(_) => Err(Errno::EOPNOTSUPP), SocketOptionName::Socket(so) => match so { // handled by `setsockopt_common` SocketOption::RCVTIMEO @@ -1539,6 +2052,13 @@ impl UnixSocket { SocketOption::TYPE | SocketOption::PEERCRED | SocketOption::ERROR => { Err(Errno::ENOPROTOOPT) } + // SO_PASSCRED: from now on every recvmsg on this socket carries an + // SCM_CREDENTIALS control message naming the sender. + SocketOption::PASSCRED => { + let enabled = super::read_from_user::(optval, optlen)? != 0; + self.options.lock().pass_cred = enabled; + Ok(()) + } // SO_RCVBUF / SO_SNDBUF are advisory hints. Accept them and keep // the fixed internal buffer size, instead of returning EOPNOTSUPP. // Log at debug so the accepted-but-ignored option stays visible. @@ -1579,9 +2099,7 @@ impl UnixSocket { } let val: u32 = match optname { - SocketOptionName::IP(ip) => match ip { - IpOption::TOS => return Err(Errno::EOPNOTSUPP), - }, + SocketOptionName::IP(_) => return Err(Errno::EOPNOTSUPP), SocketOptionName::Socket(so) => match so { // handled by `getsockopt_common` SocketOption::RCVTIMEO @@ -1594,24 +2112,16 @@ impl UnixSocket { } // Unix sockets don't track async errors SocketOption::ERROR => 0, - SocketOption::TYPE => match self.inner { - UnixSocketInner::Stream(_) => SockType::Stream as u32, - UnixSocketInner::Datagram(_) => SockType::Datagram as u32, - }, + SocketOption::TYPE => self.sock_type as u32, + SocketOption::PASSCRED => u32::from(self.options.lock().pass_cred), SocketOption::RCVBUF | SocketOption::SNDBUF => UNIX_BUF_SIZE.trunc(), SocketOption::PEERCRED => match &self.inner { UnixSocketInner::Stream(stream) => { - let ucred = stream.with_state_ref(|state| match state { - UnixStreamState::Connected(_) => { - log_unsupported!("get PEERCRED for unix socket"); - Err(Errno::EOPNOTSUPP) - } - _ => Ok(litebox_common_linux::Ucred { - pid: 0, - uid: u32::MAX, - gid: u32::MAX, - }), - })?; + let ucred = stream.get_peer_cred().unwrap_or(Ucred { + pid: 0, + uid: u32::MAX, + gid: u32::MAX, + }); return super::write_to_user::<_, Platform>(ucred, optval, len); } UnixSocketInner::Datagram(_) => { @@ -1655,6 +2165,17 @@ impl IOPollable for UnixSocket } } + fn unregister_observer(&self, observer: Weak>) { + match &self.inner { + UnixSocketInner::Stream(stream) => { + stream.unregister_observer(observer); + } + UnixSocketInner::Datagram(datagram) => { + datagram.inner.read().pollee.unregister_observer(observer); + } + } + } + fn check_io_events(&self) -> Events { match &self.inner { UnixSocketInner::Stream(stream) => stream.check_io_events(), @@ -1667,6 +2188,13 @@ pub(crate) struct UnixEntry(UnixEntryInner

{ Stream(Arc>), Datagram(WriteEnd), + /// A placeholder claimed by `bind()` (autobind, explicit path, or + /// explicit abstract) before the socket has gone on to `listen()` or + /// (for datagram sockets) finished its atomic bind. Nothing can lookup + /// or connect through a `Reserved` slot -- it exists purely to make the + /// address collide with any other bind attempt, exactly as a live + /// `Stream`/`Datagram` entry would. + Reserved(Arc), } /// Type alias for the global Unix socket address table. diff --git a/litebox_shim_linux/src/vsock_transport.rs b/litebox_shim_linux/src/vsock_transport.rs new file mode 100644 index 0000000000..627e1ca4d3 --- /dev/null +++ b/litebox_shim_linux/src/vsock_transport.rs @@ -0,0 +1,191 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Transport for a point-to-point, non-IP byte channel (e.g. a vsock-style hypercall channel), +//! generic over whatever actually backs it. +//! +//! [`ShimTransport`](crate::transport::ShimTransport) is TCP-specific: it goes through +//! [`litebox::net::Network`], a full smoltcp IP stack. A vsock-style channel is not IP traffic +//! and has no address, port, or routing -- it is architecturally a peer to the IP stack, not a +//! mode of it (see `litebox_runner_snp`'s boot-channel design notes). [`PointToPointTransport`] +//! is the transport-agnostic counterpart: it implements the same +//! [`litebox::fs::nine_p::transport::Read`]/`Write` contract `ShimTransport` does, but over any +//! [`ByteChannel`], so the 9P client above it needs no changes at all to run over a real +//! vsock-style channel once one exists. +//! +//! No platform in this repo backs [`ByteChannel`] with a real vsock-style hypercall yet -- see +//! `docs/vsock-boot-channel.md` for the host-side contract a future SEV-SNP implementation would +//! need. This module exists so that day's patch is "implement `ByteChannel` for the real +//! hypercall and switch the call site," not "invent this whole abstraction under pressure." + +use litebox::fs::nine_p::transport; + +/// A point-to-point, non-IP byte channel: something that can move bytes to and from a single +/// fixed peer, with no addressing of its own. +/// +/// Implementations are non-blocking: `try_read`/`try_write` return `Ok(0)` (not an error) when +/// no progress can be made right now, exactly like [`litebox::net::socket_channel::NetworkProxy`]'s +/// `try_read`, so [`PointToPointTransport`] can spin-poll them the same way +/// [`ShimTransport`](crate::transport::ShimTransport) spin-polls its `NetworkProxy`. +pub trait ByteChannel { + /// Reads up to `buf.len()` bytes. `Ok(0)` means no data is available right now, not EOF -- + /// this channel has no end-of-stream concept, matching a real vsock-style channel's lifetime + /// being tied to the VM itself, not to a stream close. + fn try_read(&mut self, buf: &mut [u8]) -> Result; + + /// Writes up to `buf.len()` bytes. `Ok(0)` means the channel is temporarily full, not an + /// error -- the caller should retry. + fn try_write(&mut self, buf: &[u8]) -> Result; +} + +/// Opaque channel failure. [`ByteChannel`] implementations do not need Linux `errno` semantics +/// (there is no guest-visible fd behind this channel -- see [`PointToPointTransport`]'s doc +/// comment), so this carries no payload; callers only need to know the channel is no longer +/// usable. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ChannelError; + +/// A [`litebox::fs::nine_p::transport::Read`]/`Write` implementation over any [`ByteChannel`]. +/// +/// This is the transport-agnostic sibling of +/// [`ShimTransport`](crate::transport::ShimTransport): where that type is hardwired to a raw TCP +/// `SocketFd`, this type is generic over the channel, so the same spin-poll `Read`/`Write` glue +/// works for a TCP-backed channel, a test mock, or (once implemented) a real vsock-style +/// hypercall channel, without duplicating this code three times. +pub struct PointToPointTransport { + channel: C, +} + +impl PointToPointTransport { + pub fn new(channel: C) -> Self { + Self { channel } + } +} + +impl transport::Read for PointToPointTransport { + fn read(&mut self, buf: &mut [u8]) -> Result { + loop { + match self.channel.try_read(buf) { + Ok(0) => core::hint::spin_loop(), + Ok(n) => return Ok(n), + Err(ChannelError) => return Err(transport::ReadError), + } + } + } +} + +impl transport::Write for PointToPointTransport { + fn write(&mut self, buf: &[u8]) -> Result { + loop { + match self.channel.try_write(buf) { + Ok(0) => core::hint::spin_loop(), + Ok(n) => return Ok(n), + Err(ChannelError) => return Err(transport::WriteError), + } + } + } +} + +#[cfg(test)] +mod tests { + extern crate std; + + use std::sync::mpsc; + + use litebox::fs::nine_p::transport::{Read as _, Write as _}; + + use super::*; + + /// An in-process mock [`ByteChannel`], standing in for a real vsock-style hypercall channel: + /// two byte queues, one per direction, so a pair of these forms a full-duplex pipe between + /// "guest" and "host" ends -- close enough to a real point-to-point channel's contract + /// (no addressing, no stream EOF, `Ok(0)` for "nothing right now") to exercise + /// [`PointToPointTransport`]'s spin-poll logic honestly. + struct MockChannel { + outgoing: mpsc::Sender, + incoming: mpsc::Receiver, + } + + fn mock_pair() -> (MockChannel, MockChannel) { + let (a_to_b_tx, a_to_b_rx) = mpsc::channel(); + let (b_to_a_tx, b_to_a_rx) = mpsc::channel(); + ( + MockChannel { + outgoing: a_to_b_tx, + incoming: b_to_a_rx, + }, + MockChannel { + outgoing: b_to_a_tx, + incoming: a_to_b_rx, + }, + ) + } + + impl ByteChannel for MockChannel { + fn try_read(&mut self, buf: &mut [u8]) -> Result { + let mut n = 0; + while n < buf.len() { + match self.incoming.try_recv() { + Ok(byte) => { + buf[n] = byte; + n += 1; + } + Err(mpsc::TryRecvError::Empty) => break, + Err(mpsc::TryRecvError::Disconnected) => return Err(ChannelError), + } + } + Ok(n) + } + + fn try_write(&mut self, buf: &[u8]) -> Result { + for &byte in buf { + self.outgoing.send(byte).map_err(|_| ChannelError)?; + } + Ok(buf.len()) + } + } + + #[test] + fn round_trips_bytes_in_both_directions() { + let (a, b) = mock_pair(); + let mut guest = PointToPointTransport::new(a); + let mut host = PointToPointTransport::new(b); + + guest.write_all(b"ping").unwrap(); + let mut buf = [0u8; 4]; + host.read_exact(&mut buf).unwrap(); + assert_eq!(&buf, b"ping"); + + host.write_all(b"pong!").unwrap(); + let mut buf = [0u8; 5]; + guest.read_exact(&mut buf).unwrap(); + assert_eq!(&buf, b"pong!"); + } + + #[test] + fn read_spins_rather_than_erroring_when_nothing_is_available_yet() { + // No writer has sent anything: try_read must return Ok(0), not an error -- `read_exact` + // must not give up after one empty poll, which is exactly the "spin until the byte a + // concurrent writer sends a moment later arrives" behavior the boot channel depends on. + let (a, b) = mock_pair(); + let mut guest_side = a; + std::thread::spawn(move || { + std::thread::sleep(std::time::Duration::from_millis(20)); + guest_side.try_write(b"late").unwrap(); + }); + + let mut host = PointToPointTransport::new(b); + let mut buf = [0u8; 4]; + host.read_exact(&mut buf).unwrap(); + assert_eq!(&buf, b"late"); + } + + #[test] + fn disconnected_channel_reports_a_transport_error_not_a_hang() { + let (a, b) = mock_pair(); + drop(a); + let mut host = PointToPointTransport::new(b); + let mut buf = [0u8; 1]; + assert!(host.read(&mut buf).is_err()); + } +} diff --git a/litebox_shim_linux/src/wait.rs b/litebox_shim_linux/src/wait.rs index c2eedb6596..f1fc991558 100644 --- a/litebox_shim_linux/src/wait.rs +++ b/litebox_shim_linux/src/wait.rs @@ -36,6 +36,23 @@ impl Task { /// exit instead. #[must_use] pub(crate) fn prepare_to_run_guest(&self, ctx: &mut litebox_common_linux::PtRegs) -> bool { + // The safe `ptrace` stop rendezvous: this is the one point every guest thread reaches, + // after every syscall/exception/interrupt return and strictly before guest re-entry, + // with its vCPU lane already released and `ctx` holding its complete, authoritative + // logical register state -- see `syscalls::ptrace`'s module documentation. A no-op + // (single atomic load) when untraced or not currently stop-requested; ahead of the fork + // gate below so a tracer observing a stop never has to reason about a concurrent `fork` + // interleaving with it. + #[cfg(target_arch = "aarch64")] + self.ptrace_rendezvous(ctx); + // A sibling `fork` in flight must not see this thread touch guest + // memory (see `Process::fork_gate`); park here, before re-entering + // guest code, until the forker's turn completes. No-op single load + // when no fork is in flight. + self.park_while_fork_gate_closed(); + // A member of a shared address space whose turn another member is waiting for hands + // it over here, before touching guest memory again (see `Task::quiesce_and_hand_off`). + self.yield_address_space_to_waiters(); self.wait_state.0.prepare_to_run_guest(|| { self.global.platform.take_pending_signals(|signal| { self.queue_signals(signal); @@ -43,6 +60,9 @@ impl Task { #[cfg(feature = "alarm_fallback")] self.check_alarm_deadline(); self.process_signals(ctx); + // After delivery, so that an `rt_sigsuspend` handler frame captured the temporary + // mask rather than the one being put back here. + self.restore_saved_signal_mask(); !self.is_exiting() }) } @@ -52,6 +72,13 @@ impl litebox::event::wait::CheckForInterrupt for Task { fn check_for_interrupt(&self) -> bool { + // See `Process::fork_gate`: a woken waiter passes through here before + // re-blocking, which is what lets a forking sibling park a thread that + // was asleep in a futex/epoll/read wait. Parking blocks on a raw + // (non-interruptible) word, satisfying this hook's no-interruptible- + // wait contract. + self.park_while_fork_gate_closed(); + self.yield_address_space_to_waiters(); self.global.platform.take_pending_signals(|sig| { self.queue_signals(sig); }); @@ -59,4 +86,19 @@ impl litebox::event::wait::CheckForInterrupt self.check_alarm_deadline(); self.is_exiting() || self.has_pending_signals() } + + /// Hands a shared guest address space to whichever other guest process wants it, for as long + /// as this task is asleep. + /// + /// This is the hook that lets a `fork`ed child and its parent make progress in turn instead + /// of the parent being suspended for the child's whole lifetime; see + /// `syscalls::process::SharedAddressSpace`. It is a no-op -- a single predictable branch -- + /// for the overwhelmingly common case of a task that has never `fork`ed. + fn yield_while_blocking(&self) { + self.release_address_space(); + } + + fn resume_after_blocking(&self) { + self.acquire_address_space(); + } } diff --git a/litebox_shim_optee/src/lib.rs b/litebox_shim_optee/src/lib.rs index ce89efbd80..c230e7ce5d 100644 --- a/litebox_shim_optee/src/lib.rs +++ b/litebox_shim_optee/src/lib.rs @@ -319,7 +319,10 @@ impl OpteeShim { /// The caller must ensure that no references to the released memory regions /// are held after this call. pub unsafe fn release_user_mappings(&self) { - let release = |_r: core::ops::Range, _vm: litebox::mm::linux::VmFlags| true; + // This shim instance owns every mapping the manager tracks, so each tracked range is + // released whole. See `PageManager::release_memory` for why the callback names ranges + // rather than answering yes/no. + let release = |r: core::ops::Range, _vm: litebox::mm::linux::VmFlags| Some(r); unsafe { let _ = self.page_manager().release_memory(release); } diff --git a/litebox_shim_optee/src/loader/elf.rs b/litebox_shim_optee/src/loader/elf.rs index 18c490874f..968e3f15b5 100644 --- a/litebox_shim_optee/src/loader/elf.rs +++ b/litebox_shim_optee/src/loader/elf.rs @@ -208,7 +208,11 @@ impl<'a> FileAndParsed<'a> { let file = ElfFileInMemory::new(task, elf_buf); let mut parsed = litebox_common_linux::loader::ElfParsedFile::parse(&mut &file) .map_err(ElfLoaderError::ParseError)?; - match parsed.parse_trampoline(&mut &file, task.global.platform.get_syscall_entry_point()) { + match parsed.parse_trampoline( + &mut &file, + task.global.platform.get_syscall_entry_point(), + task.global.platform.get_guest_tp_slot_offset(), + ) { Ok(()) | Err(ElfParseError::UnpatchedBinary) => {} Err(e) => return Err(e.into()), } diff --git a/litebox_shim_optee/src/syscalls/mm.rs b/litebox_shim_optee/src/syscalls/mm.rs index 2263e68c6b..ac284073a2 100644 --- a/litebox_shim_optee/src/syscalls/mm.rs +++ b/litebox_shim_optee/src/syscalls/mm.rs @@ -31,6 +31,7 @@ impl Task { prot, flags, false, + None, op, ) .map(UserPtrMut::to_platform_ptr::) diff --git a/litebox_shim_windows/Cargo.toml b/litebox_shim_windows/Cargo.toml new file mode 100644 index 0000000000..3a1c96d41e --- /dev/null +++ b/litebox_shim_windows/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "litebox_shim_windows" +version = "0.1.0" +edition = "2024" + +[dependencies] +bitflags = { version = "2.9.0", default-features = false } +int-enum = "1.2.0" +litebox = { path = "../litebox/", version = "0.1.0" } +litebox_common_linux = { path = "../litebox_common_linux/", version = "0.1.0" } +litebox_common_windows = { path = "../litebox_common_windows/", version = "0.1.0" } +litebox_util_log = { path = "../litebox_util_log", version = "0.1.0" } +rangemap = { version = "1.5.1", features = ["const_fn"] } +thiserror = { version = "2.0.6", default-features = false } +zerocopy = { version = "0.8", default-features = false, features = ["derive"] } + +[target.'cfg(target_os = "linux")'.dev-dependencies] +litebox_platform_linux_userland = { path = "../litebox_platform_linux_userland/", version = "0.1.0" } + +[target.'cfg(target_os = "windows")'.dev-dependencies] +litebox_platform_windows_userland = { path = "../litebox_platform_windows_userland/", version = "0.1.0" } + +[lints] +workspace = true diff --git a/litebox_shim_windows/src/lib.rs b/litebox_shim_windows/src/lib.rs new file mode 100644 index 0000000000..ee1c9d0aed --- /dev/null +++ b/litebox_shim_windows/src/lib.rs @@ -0,0 +1,2474 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! A placeholder Windows NT shim for LiteBox. +//! +//! This crate intentionally only exposes the runner-facing skeleton for now. +//! The actual NT syscall, PE loading, and Windows process environment support +//! will be filled in piece by piece. + +// The NT guest ABI this shim implements is the x86-64 one throughout: the PE +// loader, the syscall dispatch and the exception plumbing all speak it, and the +// syscall rewriter's aarch64 support covers Linux ELF images only. Every runner +// above this crate is already gated the same way. +#![cfg(target_arch = "x86_64")] +#![no_std] + +extern crate alloc; + +use alloc::borrow::Cow; +use alloc::collections::BTreeMap; +use alloc::sync::Arc; +use alloc::vec::Vec; +use core::marker::PhantomData; +use core::sync::atomic::{AtomicI32, AtomicU32, Ordering}; +use litebox_common_windows::nt_status::NtStatus; + +use litebox::LiteBox; +use litebox::mm::PageManager; +use litebox::platform::{ + ArchSpecificProvider, ArchSpecificRegister, CrngProvider, PageManagementProvider, + RawConstPointer as _, RawMutPointer as _, RawPointerProvider, StdioProvider, + SystemInfoProvider, TimeProvider, +}; +use litebox::shim::{ContinueOperation, EnterShim, ExceptionInfo}; +use litebox::sync::RawSyncPrimitivesProvider; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::NtSysno; +use litebox_common_windows::loader::{MappingInfo, PAGE_SIZE}; + +use crate::syscalls::event::{EventHandleObject, EventSubsystem}; +use crate::syscalls::file::{FileObject, FileObjectSubsystem}; +use crate::syscalls::iocp::{IoCompletionHandleObject, IoCompletionSubsystem}; +use crate::syscalls::lpc::{LpcPortHandleObject, LpcPortSubsystem}; +use crate::syscalls::object_manager::{ + DirectoryHandleObject, DirectoryObjectSubsystem, ObjectManager, +}; +use crate::syscalls::registry::{RegistryKeyObject, RegistryKeySubsystem}; +use crate::syscalls::section::{ + MapViewOfSectionParameters, SectionHandleObject, SectionObject, SectionSubsystem, +}; +use crate::syscalls::symlink::{SymbolicLinkHandleObject, SymbolicLinkSubsystem}; +use crate::syscalls::timer::{TimerCreateParameters, TimerHandleObject, TimerSubsystem}; +use crate::syscalls::token::{TokenHandleObject, TokenObject, TokenSubsystem}; +use crate::syscalls::wait_completion_packet::{ + WaitCompletionPacketAssociateParameters, WaitCompletionPacketHandleObject, + WaitCompletionPacketSubsystem, +}; +use crate::syscalls::worker_factory::{ + WorkerFactoryCreateParameters, WorkerFactoryHandleObject, WorkerFactorySubsystem, +}; +use crate::syscalls::{SyscallRequest, mm}; + +mod loader; +mod nt_types; +mod syscalls; + +#[cfg(test)] +mod tests; + +const DEFAULT_PROCESS_EXIT_CODE: i32 = 1; + +/// A LiteBox platform with the services required by the Windows shim. +pub trait ShimPlatform: + RawSyncPrimitivesProvider + + RawPointerProvider + + PageManagementProvider + + ArchSpecificProvider + + SystemInfoProvider + + TimeProvider + + 'static +{ +} + +impl ShimPlatform for T where + T: RawSyncPrimitivesProvider + + RawPointerProvider + + PageManagementProvider + + ArchSpecificProvider + + SystemInfoProvider + + TimeProvider + + 'static +{ +} + +pub(crate) type ConstPtr = + ::RawConstPointer; +pub(crate) type MutPtr = + ::RawMutPointer; +pub(crate) type WindowsPageManager = PageManager; +pub(crate) type WindowsHandleStore = + litebox::sync::RwLock; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct DuplicateOptions: u32 { + const CLOSE_SOURCE = 0x0000_0001; + const SAME_ACCESS = 0x0000_0002; + const SAME_ATTRIBUTES = 0x0000_0004; + + const _ = !0; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] + struct HandleAttributes: u32 { + const PROTECT_FROM_CLOSE = 0x0000_0001; + const INHERIT = 0x0000_0002; + const AUDIT_OBJECT_CLOSE = 0x0000_0004; + const HANDLE_BEHAVIOR_ATTRIBUTES = Self::PROTECT_FROM_CLOSE.bits() + | Self::INHERIT.bits() + | Self::AUDIT_OBJECT_CLOSE.bits(); + + const _ = !0; + } +} + +impl HandleAttributes { + fn from_token_open_attributes(attributes: u32) -> Option { + const OBJ_EXCLUSIVE: u32 = 0x20; + const OBJ_OPENLINK: u32 = 0x100; + + // NtOpenProcessTokenEx accepts and ignores unrelated object attributes, but native + // Windows rejects the two attributes that cannot apply to opening an existing token. + if attributes & (OBJ_EXCLUSIVE | OBJ_OPENLINK) != 0 { + return None; + } + Some(Self::from_bits_retain( + attributes & Self::HANDLE_BEHAVIOR_ATTRIBUTES.bits(), + )) + } + + fn from_duplicate_attributes(attributes: u32) -> Option { + const OBJ_EXCLUSIVE: u32 = 0x20; + + // Native NtDuplicateObject accepts and ignores unrelated object attributes, but an + // exclusive duplicate is invalid because the object already has an open handle. + if attributes & OBJ_EXCLUSIVE != 0 { + return None; + } + Some(Self::from_bits_retain( + attributes & Self::HANDLE_BEHAVIOR_ATTRIBUTES.bits(), + )) + } +} + +#[derive(Clone, Copy, Default)] +struct WindowsHandleMetadata { + granted_access: u32, + attributes: HandleAttributes, +} + +pub(crate) trait WindowsHandleSubsystem: litebox::fd::FdEnabledSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32; + + fn resolve_duplicate_access( + _entry: &Self::Entry, + desired_access: u32, + ) -> Result { + let maximum_allowed = desired_access & nt_types::AccessMask::MAXIMUM_ALLOWED.bits() != 0; + let explicit_access = desired_access & !nt_types::AccessMask::MAXIMUM_ALLOWED.bits(); + let normalized = Self::normalize_desired_access(explicit_access); + Ok(if maximum_allowed { + // TODO(dacl-access-check): Derive this grant from the caller's token and the + // object's security descriptor instead of assuming a single trust context. + normalized | Self::normalize_desired_access(nt_types::AccessMask::GENERIC_ALL.bits()) + } else { + normalized + }) + } +} +pub(crate) type WindowsNlsSectionMappings = + litebox::sync::RwLock>; +pub(crate) type WindowsVirtualAllocations = + litebox::sync::RwLock>; +pub(crate) type WindowsSectionViews = + litebox::sync::RwLock>>; +pub(crate) type WindowsObjectManager = ObjectManager; + +#[derive(Clone, Debug, Eq, PartialEq)] +pub(crate) struct WindowsVirtualAllocation { + pub(crate) base: usize, + pub(crate) size: usize, + pub(crate) allocation_protect: syscalls::mm::PageProtection, + pub(crate) type_: syscalls::mm::MemoryType, + pub(crate) pages: rangemap::RangeMap, +} + +pub(crate) struct WindowsSectionView { + pub(crate) size: usize, + pub(crate) section_offset: usize, + pub(crate) section: Option>>, +} + +impl Clone for WindowsSectionView { + fn clone(&self) -> Self { + Self { + size: self.size, + section_offset: self.section_offset, + section: self.section.clone(), + } + } +} + +pub type DefaultFS = WindowsFS; + +pub type WindowsFS = litebox::fs::layered::FileSystem< + Platform, + litebox::fs::in_mem::FileSystem, + litebox::fs::layered::FileSystem< + Platform, + litebox::fs::resolver::Resolver, + litebox::fs::resolver::Resolver, + >, +>; + +/// A trait required for file systems to be used by the Windows shim. +pub trait ShimFS: litebox::fs::FileSystem + Send + Sync + 'static {} +impl ShimFS for T {} + +fn write_value(address: usize, value: T) -> Option<()> +where + Platform: RawPointerProvider, + T: zerocopy::FromBytes + zerocopy::IntoBytes, +{ + let ptr = ::RawMutPointer::::from_usize( + address, + ); + ptr.write_at_offset(0, value) +} + +fn read_field_at_offset(base: usize, field_offset: usize) -> Option +where + Platform: RawPointerProvider, + Field: zerocopy::FromBytes, +{ + let address = base.checked_add(field_offset)?; + let ptr = ConstPtr::::from_usize(address); + ptr.read_at_offset(0) +} + +fn write_field_at_offset( + base: usize, + field_offset: usize, + value: Field, +) -> Option<()> +where + Platform: RawPointerProvider, + Field: zerocopy::FromBytes + zerocopy::IntoBytes, +{ + let address = base.checked_add(field_offset)?; + let ptr = MutPtr::::from_usize(address); + ptr.write_at_offset(0, value) +} + +fn write_slice(address: usize, values: &[T]) -> Option<()> +where + Platform: RawPointerProvider, + T: Copy + zerocopy::FromBytes + zerocopy::IntoBytes, +{ + let ptr = ::RawMutPointer::::from_usize( + address, + ); + for (index, value) in values.iter().copied().enumerate() { + ptr.write_at_offset(index.try_into().ok()?, value)?; + } + Some(()) +} + +pub(crate) fn probe_guest_output_preserving_value( + ptr: MutPtr, +) -> Result<(), NtStatus> +where + Platform: RawPointerProvider, + T: zerocopy::FromBytes + zerocopy::IntoBytes, +{ + let value = ptr.read_at_offset(0).ok_or(NtStatus::ACCESS_VIOLATION)?; + ptr.write_at_offset(0, value) + .ok_or(NtStatus::ACCESS_VIOLATION) +} + +pub(crate) fn probe_guest_output_buffer( + buffer: MutPtr, + buffer_length: usize, +) -> Result<(), NtStatus> +where + Platform: RawPointerProvider, +{ + if buffer_length == 0 { + return Ok(()); + } + probe_guest_output_preserving_value::(buffer)?; + let last_offset = isize::try_from(buffer_length - 1).map_err(|_| NtStatus::ACCESS_VIOLATION)?; + let value = buffer + .read_at_offset(last_offset) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + buffer + .write_at_offset(last_offset, value) + .ok_or(NtStatus::ACCESS_VIOLATION) +} + +fn set_guest_teb(platform: &Platform, teb_address: usize) -> bool { + if let Err(error) = + platform.set_arch_specific_register(&ArchSpecificRegister::FsBase, teb_address) + { + litebox_util_log::warn!(error:? = error, teb:% = format_args!("{teb_address:#x}"); "Failed to set Windows TEB base"); + return false; + } + + true +} + +fn insert_raw_handle( + litebox: &LiteBox, + handles: &WindowsHandleStore, + typed: litebox::fd::TypedFd, + cleanup_entry: impl FnOnce(Subsystem::Entry), +) -> Result +where + Platform: RawSyncPrimitivesProvider, +{ + let mut handles = handles.write(); + let raw_fd = handles.fd_into_raw_integer(typed); + let Some(handle) = syscalls::Handle::from_raw_fd(raw_fd) else { + let typed = handles.fd_consume_raw_integer::(raw_fd).ok(); + drop(handles); + let entry = typed.and_then(|typed| { + let mut descriptor_table = litebox.descriptor_table_mut(); + descriptor_table.remove(&typed) + }); + if let Some(entry) = entry { + cleanup_entry(entry); + } + return Err(NtStatus::QUOTA_EXCEEDED); + }; + Ok(handle) +} + +pub(crate) fn raw_handle_entry( + litebox: &LiteBox, + handles: &WindowsHandleStore, + handle: syscalls::Handle, +) -> Option> +where + Platform: RawSyncPrimitivesProvider, +{ + let raw_fd = handle.raw_fd()?; + let typed = { + let handles = handles.read(); + handles.fd_from_raw_integer::(raw_fd).ok() + }?; + litebox.descriptor_table().entry_handle(&typed) +} + +fn remove_raw_handle( + litebox: &LiteBox, + handles: &WindowsHandleStore, + handle: syscalls::Handle, + cleanup_entry: impl FnOnce(Subsystem::Entry), +) where + Platform: RawSyncPrimitivesProvider, +{ + let Some(raw_fd) = handle.raw_fd() else { + return; + }; + let _ = + remove_raw_handle_by_raw_fd::(litebox, handles, raw_fd, cleanup_entry); +} + +fn remove_raw_handle_by_raw_fd( + litebox: &LiteBox, + handles: &WindowsHandleStore, + raw_fd: usize, + cleanup_entry: impl FnOnce(Subsystem::Entry), +) -> bool +where + Platform: RawSyncPrimitivesProvider, +{ + let typed = { + let mut handles = handles.write(); + handles.fd_consume_raw_integer::(raw_fd).ok() + }; + let Some(typed) = typed else { + return false; + }; + let entry = { + let mut descriptor_table = litebox.descriptor_table_mut(); + descriptor_table.remove(&typed) + }; + if let Some(entry) = entry { + cleanup_entry(entry); + } + true +} + +/// Builds a Windows NT shim instance. +pub struct WindowsShimBuilder { + platform: &'static Platform, + litebox: LiteBox, +} + +impl WindowsShimBuilder { + #[must_use] + pub fn new(platform: &'static Platform) -> Self { + Self { + platform, + litebox: LiteBox::new(platform), + } + } + + #[must_use] + pub fn litebox(&self) -> &LiteBox { + &self.litebox + } + + /// Build a default layered file system with the given in-memory and tar read-only layers. + #[must_use] + pub fn default_fs( + &self, + in_mem_fs: litebox::fs::in_mem::FileSystem, + tar_data: Cow<'static, [u8]>, + ) -> DefaultFS + where + Platform: CrngProvider + StdioProvider, + { + default_fs(&self.litebox, in_mem_fs, tar_data) + } + + #[must_use] + pub fn build(self) -> WindowsShim { + let global = Arc::new(GlobalState { + platform: self.platform, + page_manager: PageManager::new(&self.litebox), + registry: syscalls::registry::RegistryStore::new(&self.litebox), + wnf_states: syscalls::wnf::WnfStateStore::new( + syscalls::wnf::WnfStateStoreData::default(), + ), + qpc_boot_instant: TimeProvider::now(self.platform), + litebox: self.litebox, + _fs: PhantomData, + }); + WindowsShim(global) + } +} + +/// Wine and ReactOS model KUSER_SHARED_DATA as a fixed user page at +/// 0x7FFE0000. Native Windows hosts already provide that page; Non-Windows hosts +/// need LiteBox to create it before guest ntdll reads it during startup. +#[cfg(not(target_os = "windows"))] +const WINDOWS_USER_SHARED_DATA_BASE: usize = 0x7FFE_0000; + +#[cfg(not(target_os = "windows"))] +fn map_windows_user_shared_data( + page_manager: &crate::WindowsPageManager, +) -> Option { + use litebox::mm::linux::{CreatePagesFlags, MappingError, NonZeroAddress, NonZeroPageSize}; + use zerocopy::IntoBytes as _; + let address = NonZeroAddress::new(WINDOWS_USER_SHARED_DATA_BASE)?; + let length = + NonZeroPageSize::new(size_of::().next_multiple_of(PAGE_SIZE))?; + let shared_data = windows_user_shared_data(); + let shared_data_bytes = shared_data.as_bytes(); + // SAFETY: `NOREPLACE` makes the fixed mapping fail instead of replacing any + // existing host or guest mapping at the shared-data address. + unsafe { + page_manager.create_readable_pages( + Some(address), + length, + CreatePagesFlags::FIXED_ADDR | CreatePagesFlags::NOREPLACE, + |ptr| { + ptr.copy_from_slice(0, shared_data_bytes) + .ok_or(MappingError::OutOfMemory)?; + Ok(0) + }, + ) + } + .map(|ptr| ptr.as_usize()) + .ok() +} + +// TODO: This is a temporary placeholder for the Windows shared data page. +// Once we have a proper shared mapping implementation, we can remove this +// and instead map the shared data page from the host into the guest. +#[cfg(not(target_os = "windows"))] +fn windows_user_shared_data() -> nt_types::KUserSharedData { + use zerocopy::FromZeros as _; + let mut shared_data = nt_types::KUserSharedData::new_zeroed(); + shared_data.nt_build_number = u32::from(syscalls::sysinfo::WINDOWS_OS_BUILD_NUMBER); + shared_data.nt_product_type = syscalls::sysinfo::WINDOWS_NT_PRODUCT_WORKSTATION; + shared_data.product_type_is_valid = 1; + shared_data.nt_major_version = u32::from(syscalls::sysinfo::WINDOWS_OS_MAJOR_VERSION); + shared_data.nt_minor_version = u32::from(syscalls::sysinfo::WINDOWS_OS_MINOR_VERSION); + for (index, code_unit) in r"C:\Windows".encode_utf16().enumerate() { + shared_data.nt_system_root[index] = code_unit; + } + + shared_data +} + +pub struct WindowsShim(Arc>); + +impl WindowsShim { + /// Loads the program at `path` as the shim's initial task. + pub fn load_program( + &self, + fs: Arc, + path: &str, + argv: Vec, + envp: Vec, + ) -> Result, loader::WindowsLoadError> { + // TODO: refactor the shared mapping + #[cfg(not(target_os = "windows"))] + let _ = map_windows_user_shared_data::(&self.0.page_manager) + .ok_or(loader::WindowsLoadError::MapSharedMemory)?; + let load_info = loader::PeLoader::new(self.0.platform, fs.clone(), &self.0.page_manager) + .load(path, &argv, &envp)?; + // TODO: shared section should be only created once and shared across all processes, not created per-process. + let windows_shared_section = crate::syscalls::section::load_time_windows_shared_section( + load_info.environment.windows_shared_section, + ); + let mut process = + Process::default(Some(load_info.virtual_allocations), windows_shared_section); + process.ntdll_mapping = load_info.ntdll_mapping; + process.peb_address = load_info.environment.peb; + let process = Arc::new(process); + Ok(LoadedProgram { + entrypoints: WindowsShimEntrypoints { + task: Task { + global: self.0.clone(), + process: process.clone(), + fs, + entry_point: load_info.entry_point, + stack_top: load_info.stack_top, + teb_address: load_info.environment.teb, + context: load_info.environment.context, + }, + _not_send: PhantomData, + }, + process, + }) + } +} + +/// Global shim state shared by all Windows tasks loaded by this shim. +struct GlobalState { + platform: &'static Platform, + page_manager: WindowsPageManager, + registry: syscalls::registry::RegistryStore, + wnf_states: syscalls::wnf::WnfStateStore, + qpc_boot_instant: ::Instant, + litebox: LiteBox, + _fs: PhantomData, +} + +/// Per-process Windows state shared by every thread in the process. +pub struct Process { + ntdll_mapping: Option, + peb_address: usize, + handles: WindowsHandleStore, + token: Arc, + condrv_console: syscalls::condrv::CondrvConsole, + object_manager: WindowsObjectManager, + section_views: WindowsSectionViews, + // TODO: move this into `GlobalState` once we have a proper shared mapping implementation. + #[expect( + dead_code, + reason = "keeps alive the section registered weakly in the object namespace" + )] + windows_shared_section: Arc>, + nls_section_mappings: WindowsNlsSectionMappings, + virtual_allocations: WindowsVirtualAllocations, + system_lcid: AtomicU32, + user_lcid: AtomicU32, + user_ui_language: AtomicU32, + default_hard_error_mode: AtomicU32, + cookie: u32, + exit_code: AtomicI32, +} + +impl Process { + /// Wait for the process to exit, returning its exit code. + /// + /// Currently a placeholder that returns a fixed exit code immediately. + /// Once NT process lifecycle exists, this will actually block. + #[must_use] + pub fn wait(&self) -> i32 { + // TODO: Wait for the NT process object once process lifecycle exists. + self.exit_code.load(Ordering::Relaxed) + } + + fn default( + virtual_allocations: Option>, + windows_shared_section: Arc>, + ) -> Self { + let object_manager = syscalls::object_manager::seed_object_manager(); + let status = object_manager.create_section( + syscalls::section::WINDOWS_SESSION_SHARED_SECTION_OBJECT, + &windows_shared_section, + ); + assert!( + status == NtStatus::SUCCESS, + "seeded Windows shared section must have seeded ancestors: {status:?}" + ); + Process { + ntdll_mapping: None, + peb_address: 0, + handles: WindowsHandleStore::::new(litebox::fd::RawDescriptorStorage::new()), + token: Arc::new(TokenObject::primary()), + // TODO(condrv-shared-console): move console ownership to shared state or a broker when + // LiteBox supports AttachConsole/IOCTL_CONDRV_BIND_PID across guest processes. + condrv_console: syscalls::condrv::CondrvConsole::new(), + object_manager, + windows_shared_section, + section_views: WindowsSectionViews::::new(BTreeMap::new()), + nls_section_mappings: WindowsNlsSectionMappings::::new(BTreeMap::new()), + virtual_allocations: virtual_allocations + .unwrap_or_else(|| WindowsVirtualAllocations::::new(BTreeMap::new())), + system_lcid: AtomicU32::new(syscalls::nls::DEFAULT_LOCALE_ID), + user_lcid: AtomicU32::new(syscalls::nls::DEFAULT_LOCALE_ID), + user_ui_language: AtomicU32::new(syscalls::nls::DEFAULT_LOCALE_ID), + default_hard_error_mode: AtomicU32::new(0), + cookie: syscalls::process::default_process_cookie(), + exit_code: AtomicI32::new(DEFAULT_PROCESS_EXIT_CODE), + } + } +} + +struct Task { + global: Arc>, + process: Arc>, + fs: Arc, + entry_point: usize, + stack_top: usize, + context: usize, + teb_address: usize, +} + +impl Task { + fn init(&self, ctx: &mut litebox_common_linux::PtRegs) -> ContinueOperation { + if !set_guest_teb(self.global.platform, self.teb_address) { + return ContinueOperation::Terminate; + } + + ctx.rip = self.entry_point; + debug_assert_eq!(self.stack_top % 16, core::mem::size_of::()); + ctx.rsp = self.stack_top; + ctx.eflags = 0x202; + ctx.rcx = self.context; + ctx.rdx = self + .process + .ntdll_mapping + .as_ref() + .map_or(0, |mapping| mapping.base_addr); + litebox_util_log::debug!( + entry_point:% = format_args!("{:#x}", self.entry_point), + stack_top:% = format_args!("{:#x}", self.stack_top); + "Starting initial Windows guest thread" + ); + + ContinueOperation::Resume + } + + fn typed_handle_entry( + &self, + handle: syscalls::Handle, + ) -> Result, NtStatus> + where + Subsystem: litebox::fd::FdEnabledSubsystem, + { + let typed = self.typed_handle::(handle)?; + self.global + .litebox + .descriptor_table() + .entry_handle(&typed) + .ok_or(NtStatus::INVALID_HANDLE) + } + + fn typed_handle_entry_with_access( + &self, + handle: syscalls::Handle, + required_access: u32, + ) -> Result, NtStatus> + where + Subsystem: WindowsHandleSubsystem, + { + let typed = self.typed_handle::(handle)?; + self.require_typed_handle_access(&typed, required_access)?; + self.global + .litebox + .descriptor_table() + .entry_handle(&typed) + .ok_or(NtStatus::INVALID_HANDLE) + } + + fn typed_handle( + &self, + handle: syscalls::Handle, + ) -> Result>, NtStatus> + where + Subsystem: litebox::fd::FdEnabledSubsystem, + { + let Some(raw_fd) = handle.raw_fd() else { + return Err(NtStatus::INVALID_HANDLE); + }; + let handles = self.process.handles.read(); + match handles.fd_from_raw_integer::(raw_fd) { + Ok(typed) => Ok(typed), + Err(litebox::fd::ErrRawIntFd::NotFound) => Err(NtStatus::INVALID_HANDLE), + Err(litebox::fd::ErrRawIntFd::InvalidSubsystem) => Err(NtStatus::OBJECT_TYPE_MISMATCH), + } + } + + fn typed_handle_metadata( + &self, + typed: &litebox::fd::TypedFd, + ) -> Result + where + Subsystem: WindowsHandleSubsystem, + { + self.global + .litebox + .descriptor_table() + .with_metadata::(typed, |metadata| *metadata) + .map_err(|_| NtStatus::INVALID_HANDLE) + } + + pub(crate) fn require_handle_access( + &self, + handle: syscalls::Handle, + required_access: u32, + ) -> Result<(), NtStatus> + where + Subsystem: WindowsHandleSubsystem, + { + let typed = self.typed_handle::(handle)?; + self.require_typed_handle_access(&typed, required_access) + } + + pub(crate) fn require_typed_handle_access( + &self, + typed: &litebox::fd::TypedFd, + required_access: u32, + ) -> Result<(), NtStatus> + where + Subsystem: WindowsHandleSubsystem, + { + if self.typed_handle_metadata(typed)?.granted_access & required_access == required_access { + Ok(()) + } else { + Err(NtStatus::ACCESS_DENIED) + } + } + + fn insert_typed_handle( + &self, + entry: Subsystem::Entry, + granted_access: u32, + cleanup_entry: impl FnOnce(Subsystem::Entry), + ) -> Result + where + Subsystem: WindowsHandleSubsystem, + { + self.insert_typed_handle_with_attributes::( + entry, + granted_access, + HandleAttributes::empty(), + cleanup_entry, + ) + } + + fn insert_typed_handle_with_attributes( + &self, + entry: Subsystem::Entry, + granted_access: u32, + attributes: HandleAttributes, + cleanup_entry: impl FnOnce(Subsystem::Entry), + ) -> Result + where + Subsystem: WindowsHandleSubsystem, + { + let typed = { + let mut descriptors = self.global.litebox.descriptor_table_mut(); + let typed = descriptors.insert::(entry); + let old = descriptors.set_fd_metadata( + &typed, + WindowsHandleMetadata { + granted_access, + attributes, + }, + ); + debug_assert!(old.is_none()); + typed + }; + insert_raw_handle::( + &self.global.litebox, + &self.process.handles, + typed, + cleanup_entry, + ) + } + + fn close_typed_handle( + &self, + handle: syscalls::Handle, + cleanup_entry: impl FnOnce(Subsystem::Entry), + ) where + Subsystem: litebox::fd::FdEnabledSubsystem, + { + remove_raw_handle::( + &self.global.litebox, + &self.process.handles, + handle, + cleanup_entry, + ); + } + + fn handle_syscall_request(&self, ctx: &mut litebox_common_linux::PtRegs) -> ContinueOperation { + let Some(req) = SyscallRequest::::try_from_raw(ctx) else { + let caller = ConstPtr::::from_usize(ctx.rsp) + .read_at_offset(0) + .unwrap_or_default(); + litebox_util_log::error!( + syscall:? = NtSysno::from_raw(ctx.orig_rax), + rip:% = format_args!("{:#x}", ctx.rip), + caller:% = format_args!("{caller:#x}"), + arg0:% = format_args!("{:#x}", ctx.r10), + arg1:% = format_args!("{:#x}", ctx.rdx), + arg2:% = format_args!("{:#x}", ctx.r8), + arg3:% = format_args!("{:#x}", ctx.r9); + "Unsupported Windows syscall; terminating Windows guest" + ); + return ContinueOperation::Terminate; + }; + litebox_util_log::debug!( + syscall:? = NtSysno::from_raw(ctx.orig_rax); + "Handling Windows syscall" + ); + let (result, op) = match req { + SyscallRequest::NtClose { handle } => { + let status = self.sys_nt_close(handle); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtDuplicateObject { + source_process_handle, + source_handle, + target_process_handle, + target_handle, + desired_access, + handle_attributes, + options, + } => { + let status = self.sys_nt_duplicate_object( + source_process_handle, + source_handle, + target_process_handle, + target_handle, + desired_access, + handle_attributes, + options, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateEvent { + event_handle, + desired_access, + object_attributes, + event_type, + initial_state, + } => { + let status = self.sys_nt_create_event( + event_handle, + desired_access, + object_attributes, + event_type, + initial_state, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateDirectoryObject { + directory_handle, + desired_access, + object_attributes, + } => { + let status = self.sys_nt_create_directory_object( + directory_handle, + desired_access, + object_attributes, + syscalls::Handle::default(), + 0, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateDirectoryObjectEx { + directory_handle, + desired_access, + object_attributes, + shadow_directory_handle, + flags, + } => { + let status = self.sys_nt_create_directory_object( + directory_handle, + desired_access, + object_attributes, + shadow_directory_handle, + flags, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenDirectoryObject { + directory_handle, + desired_access, + object_attributes, + } => { + let status = self.sys_nt_open_directory_object( + directory_handle, + desired_access, + object_attributes, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenSection { + section_handle, + desired_access, + object_attributes, + } => { + let status = + self.sys_nt_open_section(section_handle, desired_access, object_attributes); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryDirectoryObject { + directory_handle, + buffer, + buffer_length, + return_single_entry, + restart_scan, + context, + return_length, + } => { + let status = self.sys_nt_query_directory_object( + syscalls::object_manager::DirectoryQueryParameters { + directory_handle, + buffer, + buffer_length, + return_single_entry, + restart_scan, + context, + return_length, + }, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateSymbolicLinkObject { + link_handle, + desired_access, + object_attributes, + link_target, + } => { + let status = self.sys_nt_create_symbolic_link_object( + link_handle, + desired_access, + object_attributes, + link_target, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenSymbolicLinkObject { + link_handle, + desired_access, + object_attributes, + } => { + let status = self.sys_nt_open_symbolic_link_object( + link_handle, + desired_access, + object_attributes, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQuerySymbolicLinkObject { + link_handle, + link_target, + returned_length, + } => { + let status = self.sys_nt_query_symbolic_link_object( + link_handle, + link_target, + returned_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateIoCompletion { + io_completion_handle, + desired_access, + object_attributes, + number_of_concurrent_threads, + } => { + let status = self.sys_nt_create_io_completion( + io_completion_handle, + desired_access, + object_attributes, + number_of_concurrent_threads, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtConnectPort { + port_handle, + port_name, + security_qos, + client_view, + server_view, + max_message_length, + connection_information, + connection_information_length, + } => { + let status = self.sys_nt_connect_port(syscalls::lpc::ConnectPortParameters { + port_handle, + port_name, + security_qos, + client_view, + server_view, + max_message_length, + connection_information, + connection_information_length, + }); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSecureConnectPort => { + litebox_util_log::debug!( + "Rejected NtSecureConnectPort; only the CSR NtConnectPort subset is modeled" + ); + (NtStatus::NOT_SUPPORTED, ContinueOperation::Resume) + } + SyscallRequest::NtCreateSection { + section_handle, + desired_access, + object_attributes, + maximum_size, + section_page_protection, + allocation_attributes, + file_handle, + } => { + let status = self.sys_nt_create_section( + section_handle, + desired_access, + object_attributes, + maximum_size, + section_page_protection, + allocation_attributes, + file_handle, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateSectionEx { + section_handle, + desired_access, + object_attributes, + maximum_size, + section_page_protection, + allocation_attributes, + file_handle, + extended_parameters, + extended_parameter_count, + } => { + let status = self.sys_nt_create_section_ex( + section_handle, + desired_access, + object_attributes, + maximum_size, + section_page_protection, + allocation_attributes, + file_handle, + extended_parameters, + extended_parameter_count, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateWaitCompletionPacket { + wait_completion_packet_handle, + desired_access, + object_attributes, + } => { + let status = self.sys_nt_create_wait_completion_packet( + wait_completion_packet_handle, + desired_access, + object_attributes, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtAssociateWaitCompletionPacket { + wait_completion_packet_handle, + io_completion_handle, + target_object_handle, + key_context, + apc_context, + io_status, + io_status_information, + already_signaled, + } => { + let status = self.sys_nt_associate_wait_completion_packet( + WaitCompletionPacketAssociateParameters { + wait_completion_packet_handle, + io_completion_handle, + target_object_handle, + key_context, + apc_context, + io_status, + io_status_information, + already_signaled, + }, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCancelWaitCompletionPacket { + wait_completion_packet_handle, + remove_signaled_packet, + } => { + let status = self.sys_nt_cancel_wait_completion_packet( + wait_completion_packet_handle, + remove_signaled_packet, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateWorkerFactory { + worker_factory_handle, + desired_access, + object_attributes, + completion_port_handle, + worker_process_handle, + start_routine, + start_parameter, + max_thread_count, + stack_reserve, + stack_commit, + } => { + let status = self.sys_nt_create_worker_factory(WorkerFactoryCreateParameters { + worker_factory_handle, + desired_access, + object_attributes, + completion_port_handle, + worker_process_handle, + start_routine, + start_parameter, + max_thread_count, + stack_reserve, + stack_commit, + }); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetInformationWorkerFactory { + worker_factory_handle, + worker_factory_information_class, + worker_factory_information, + worker_factory_information_length, + } => { + let status = self.sys_nt_set_information_worker_factory( + worker_factory_handle, + worker_factory_information_class, + worker_factory_information, + worker_factory_information_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtShutdownWorkerFactory { + worker_factory_handle, + pending_worker_count, + } => { + let status = self + .sys_nt_shutdown_worker_factory(worker_factory_handle, pending_worker_count); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateTimer2 { + timer_handle, + timer_id, + object_attributes, + attributes, + desired_access, + } => { + let status = self.sys_nt_create_timer2(TimerCreateParameters { + timer_handle, + timer_id, + object_attributes, + attributes, + desired_access, + }); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetTimer2 { + timer_handle, + due_time, + period, + parameters, + } => { + let status = self.sys_nt_set_timer2(timer_handle, due_time, period, parameters); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenEvent { + event_handle, + desired_access, + object_attributes, + } => { + let status = + self.sys_nt_open_event(event_handle, desired_access, object_attributes); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetEvent { + event_handle, + previous_state, + } => { + let status = self.sys_nt_set_event(event_handle, previous_state); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtResetEvent { + event_handle, + previous_state, + } => { + let status = self.sys_nt_reset_event(event_handle, previous_state); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtClearEvent { event_handle } => { + let status = self.sys_nt_clear_event(event_handle); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtPulseEvent { + event_handle, + previous_state, + } => { + let status = self.sys_nt_pulse_event(event_handle, previous_state); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryEvent { + event_handle, + event_information_class, + event_information, + event_information_length, + return_length, + } => { + let status = self.sys_nt_query_event( + event_handle, + event_information_class, + event_information, + event_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetEventBoostPriority { event_handle } => { + let status = self.sys_nt_set_event_boost_priority(event_handle); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenFile { + file_handle, + desired_access, + object_attributes, + io_status_block, + share_access, + open_options, + } => { + let status = self.sys_nt_open_file( + file_handle, + desired_access, + object_attributes, + io_status_block, + share_access, + open_options, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateFile { + file_handle, + desired_access, + object_attributes, + io_status_block, + allocation_size, + file_attributes, + share_access, + create_disposition, + create_options, + ea_buffer, + ea_length, + } => { + let status = self.sys_nt_create_file( + file_handle, + desired_access, + object_attributes, + io_status_block, + allocation_size, + file_attributes, + share_access, + create_disposition, + create_options, + ea_buffer, + ea_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtWriteFile { + file_handle, + event, + apc_routine, + apc_context, + io_status_block, + buffer, + length, + byte_offset, + key, + } => { + let status = self.sys_nt_write_file( + file_handle, + event, + apc_routine, + apc_context, + io_status_block, + buffer, + length, + byte_offset, + key, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryVolumeInformationFile { + file_handle, + io_status_block, + fs_information, + length, + fs_information_class, + } => { + let status = self.sys_nt_query_volume_information_file( + file_handle, + io_status_block, + fs_information, + length, + fs_information_class, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtDeviceIoControlFile { + file_handle, + event, + apc_routine, + apc_context, + io_status_block, + io_control_code, + input_buffer, + input_buffer_length, + output_buffer, + output_buffer_length, + } => { + let status = self.sys_nt_device_io_control_file( + file_handle, + event, + apc_routine, + apc_context, + io_status_block, + io_control_code, + input_buffer, + input_buffer_length, + output_buffer, + output_buffer_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtApphelpCacheControl { + service_class, + service_data, + } => { + let status = syscalls::apphelp::sys_nt_apphelp_cache_control::( + service_class, + service_data, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenKey { + key_handle, + desired_access, + object_attributes, + } => { + let status = self.sys_nt_open_key(key_handle, desired_access, object_attributes); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryValueKey { + key_handle, + value_name, + key_value_information_class, + key_value_information, + length, + result_length, + } => { + let status = self.sys_nt_query_value_key( + key_handle, + value_name, + key_value_information_class, + key_value_information, + length, + result_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtGetNlsSectionPtr { + section_type, + section_data, + context_data, + section_pointer, + section_size, + } => { + let status = self.sys_nt_get_nls_section_ptr( + section_type, + section_data, + context_data, + section_pointer, + section_size, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtInitializeNlsFiles { + base_address, + default_locale_id, + default_casing_table_size, + } => { + let status = self.sys_nt_initialize_nls_files( + base_address, + default_locale_id, + default_casing_table_size, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryDefaultLocale { + user_profile, + default_locale_id, + } => { + let status = self.sys_nt_query_default_locale(user_profile, default_locale_id); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetDefaultLocale { + user_profile, + default_locale_id, + } => { + let status = self.sys_nt_set_default_locale(user_profile, default_locale_id); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryDefaultUILanguage { + default_ui_language, + } => { + let status = self.sys_nt_query_default_ui_language(default_ui_language); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetDefaultUILanguage { + default_ui_language, + } => { + let status = self.sys_nt_set_default_ui_language(default_ui_language); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryInstallUILanguage { + install_ui_language, + } => { + let status = self.sys_nt_query_install_ui_language(install_ui_language); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryPerformanceCounter { + performance_counter, + performance_frequency, + } => { + let status = self + .sys_nt_query_performance_counter(performance_counter, performance_frequency); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQuerySystemInformation { + system_information_class, + system_information, + system_information_length, + return_length, + } => { + let status = Self::sys_nt_query_system_information( + system_information_class, + system_information, + system_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQuerySystemInformationEx { + system_information_class, + input_buffer, + input_buffer_length, + system_information, + system_information_length, + return_length, + } => { + let status = Self::sys_nt_query_system_information_ex( + system_information_class, + input_buffer, + input_buffer_length, + system_information, + system_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryWnfStateData { + state_name, + type_id, + explicit_scope, + change_stamp, + buffer, + buffer_size, + } => { + let status = self.sys_nt_query_wnf_state_data( + state_name, + type_id, + explicit_scope, + change_stamp, + buffer, + buffer_size, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtCreateWnfStateName { + state_name, + name_lifetime, + data_scope, + persist_data, + type_id, + maximum_state_size, + security_descriptor, + } => { + let status = self.sys_nt_create_wnf_state_name( + syscalls::wnf::WnfCreateStateNameParameters { + state_name, + name_lifetime, + data_scope, + persist_data, + type_id, + maximum_state_size, + security_descriptor, + }, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtUpdateWnfStateData { + state_name, + buffer, + buffer_size, + type_id, + explicit_scope, + matching_change_stamp, + check_stamp, + } => { + let status = self.sys_nt_update_wnf_state_data( + syscalls::wnf::WnfUpdateStateDataParameters { + state_name, + buffer, + buffer_size, + type_id, + explicit_scope, + matching_change_stamp, + check_stamp, + }, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtDeleteWnfStateData { + state_name, + explicit_scope, + } => { + let status = self.sys_nt_delete_wnf_state_data(state_name, explicit_scope); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtDeleteWnfStateName { state_name } => { + let status = self.sys_nt_delete_wnf_state_name(state_name); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryWnfStateNameInformation { + state_name, + name_information_class, + explicit_scope, + buffer, + buffer_size, + } => { + let status = self.sys_nt_query_wnf_state_name_information( + state_name, + name_information_class, + explicit_scope, + buffer, + buffer_size, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQuerySection { + section_handle, + section_information_class, + section_information, + section_information_length, + return_length, + } => { + let status = self.sys_nt_query_section( + section_handle, + section_information_class, + section_information, + section_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryInformationProcess { + process_handle, + process_information_class, + process_information, + process_information_length, + return_length, + } => { + let status = self.sys_nt_query_information_process( + process_handle, + process_information_class, + process_information, + process_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetInformationProcess { + process_handle, + process_information_class, + process_information, + process_information_length, + } => { + let status = self.sys_nt_set_information_process( + process_handle, + process_information_class, + process_information, + process_information_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtSetInformationThread { + thread_handle, + thread_information_class, + thread_information, + thread_information_length, + } => { + let status = Self::sys_nt_set_information_thread( + thread_handle, + thread_information_class, + thread_information, + thread_information_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenThreadToken { + thread_handle, + desired_access, + open_as_self, + token_handle, + } => { + let status = Self::sys_nt_open_thread_token( + thread_handle, + desired_access, + open_as_self, + token_handle, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenThreadTokenEx { + thread_handle, + desired_access, + open_as_self, + handle_attributes, + token_handle, + } => { + let status = Self::sys_nt_open_thread_token_ex( + thread_handle, + desired_access, + open_as_self, + handle_attributes, + token_handle, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenProcessToken { + process_handle, + desired_access, + token_handle, + } => { + let status = + self.sys_nt_open_process_token(process_handle, desired_access, token_handle); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtOpenProcessTokenEx { + process_handle, + desired_access, + handle_attributes, + token_handle, + } => { + let status = self.sys_nt_open_process_token_ex( + process_handle, + desired_access, + handle_attributes, + token_handle, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryInformationToken { + token_handle, + token_information_class, + token_information, + token_information_length, + return_length, + } => { + let status = self.sys_nt_query_information_token( + token_handle, + token_information_class, + token_information, + token_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQuerySecurityAttributesToken { + token_handle, + attributes, + number_of_attributes, + buffer, + length, + return_length, + } => { + let status = self.sys_nt_query_security_attributes_token( + token_handle, + attributes, + number_of_attributes, + buffer, + length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtConvertBetweenAuxiliaryCounterAndPerformanceCounter { + flag, + source, + destination, + conversion_error, + } => { + let status = Self::sys_nt_convert_between_auxiliary_counter_and_performance_counter( + flag, + source, + destination, + conversion_error, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtAllocateVirtualMemory { + process_handle, + base_address, + zero_bits, + region_size, + allocation_type, + protect, + } => { + let status = self.sys_nt_allocate_virtual_memory( + process_handle, + base_address, + zero_bits, + region_size, + allocation_type, + protect, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtAllocateVirtualMemoryEx { + process_handle, + base_address, + region_size, + allocation_type, + protect, + extended_parameters, + extended_parameter_count, + } => { + let status = self.sys_nt_allocate_virtual_memory_ex( + process_handle, + base_address, + region_size, + allocation_type, + protect, + mm::MemoryExtendedParameters { + parameters: extended_parameters, + count: extended_parameter_count, + }, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtFreeVirtualMemory { + process_handle, + base_address, + region_size, + free_type, + } => { + let status = self.sys_nt_free_virtual_memory( + process_handle, + base_address, + region_size, + free_type, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtProtectVirtualMemory { + process_handle, + base_address, + region_size, + new_protect, + old_protect, + } => { + let status = self.sys_nt_protect_virtual_memory( + process_handle, + base_address, + region_size, + new_protect, + old_protect, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtQueryVirtualMemory { + process_handle, + base_address, + memory_information_class, + memory_information, + memory_information_length, + return_length, + } => { + let status = self.sys_nt_query_virtual_memory( + process_handle, + base_address, + memory_information_class, + memory_information, + memory_information_length, + return_length, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtMapViewOfSection { + section_handle, + process_handle, + base_address, + zero_bits, + commit_size, + section_offset, + view_size, + inherit_disposition, + allocation_type, + page_protection, + } => { + let status = self.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle, + process_handle, + base_address, + zero_bits, + commit_size, + section_offset, + view_size, + inherit_disposition, + allocation_type, + page_protection, + }); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtMapViewOfSectionEx { + section_handle, + process_handle, + base_address, + zero_bits, + commit_size, + section_offset, + view_size, + inherit_disposition, + allocation_type, + page_protection, + extended_parameters, + extended_parameter_count, + } => { + let status = self.sys_nt_map_view_of_section_ex( + MapViewOfSectionParameters { + section_handle, + process_handle, + base_address, + zero_bits, + commit_size, + section_offset, + view_size, + inherit_disposition, + allocation_type, + page_protection, + }, + extended_parameters, + extended_parameter_count, + ); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtUnmapViewOfSection { + process_handle, + base_address, + } => { + let status = self.sys_nt_unmap_view_of_section(process_handle, base_address); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtUnmapViewOfSectionEx { + process_handle, + base_address, + flags, + } => { + let status = + self.sys_nt_unmap_view_of_section_ex(process_handle, base_address, flags); + (status, ContinueOperation::Resume) + } + SyscallRequest::NtContinue { + context, + test_alert, + } => match Self::sys_nt_continue(ctx, context, test_alert) { + Ok(()) => return ContinueOperation::Resume, + Err(status) => (status, ContinueOperation::Resume), + }, + SyscallRequest::NtTerminateProcess { + process_handle, + exit_status, + } => { + if !process_handle.is_null() && !process_handle.is_current() { + // TODO: allow terminating other processes + litebox_util_log::error!("Terminating other processes is not yet supported"); + (NtStatus::INVALID_HANDLE, ContinueOperation::Resume) + } else { + // TODO: Terminate all threads except the calling one if process_handle is zero. + self.process.exit_code.store(exit_status, Ordering::Relaxed); + (NtStatus::SUCCESS, ContinueOperation::Terminate) + } + } + SyscallRequest::NtTestAlert => { + Self::test_alert(); + (NtStatus::SUCCESS, ContinueOperation::Resume) + } + SyscallRequest::NtManageHotPatch => { + (NtStatus::NOT_IMPLEMENTED, ContinueOperation::Resume) + } + }; + + ctx.rax = result.as_raw().cast_unsigned() as usize; + op + } + + fn sys_nt_continue( + ctx: &mut litebox_common_linux::PtRegs, + context: ConstPtr, + test_alert: bool, + ) -> Result<(), NtStatus> { + if context.as_usize() == 0 { + return Err(NtStatus::ACCESS_VIOLATION); + } + let context = context + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + + if test_alert { + Self::test_alert(); + } + + let context_flags = nt_types::ContextFlags::from_bits_retain(context.context_flags); + + if context_flags.contains(nt_types::ContextFlags::CONTROL) { + ctx.rip = context.rip.trunc(); + ctx.rsp = context.rsp.trunc(); + ctx.eflags = context.e_flags as usize; + ctx.cs = context.seg_cs as usize; + ctx.ss = context.seg_ss as usize; + } + + if context_flags.contains(nt_types::ContextFlags::INTEGER) { + ctx.rax = context.rax.trunc(); + ctx.rbx = context.rbx.trunc(); + ctx.rcx = context.rcx.trunc(); + ctx.rdx = context.rdx.trunc(); + ctx.rsi = context.rsi.trunc(); + ctx.rdi = context.rdi.trunc(); + ctx.rbp = context.rbp.trunc(); + ctx.r8 = context.r8.trunc(); + ctx.r9 = context.r9.trunc(); + ctx.r10 = context.r10.trunc(); + ctx.r11 = context.r11.trunc(); + ctx.r12 = context.r12.trunc(); + ctx.r13 = context.r13.trunc(); + ctx.r14 = context.r14.trunc(); + ctx.r15 = context.r15.trunc(); + } + + // TODO(context-model): Restore floating-point, extended, and debug-register state. + if context_flags.contains(nt_types::ContextFlags::FLOATING_POINT) { + litebox_util_log::warn!( + "NtContinue requested floating-point state, which is not yet restored" + ); + } + if context_flags.contains(nt_types::ContextFlags::XSTATE) { + litebox_util_log::warn!( + "NtContinue requested extended state, which is not yet restored" + ); + } + if context_flags.contains(nt_types::ContextFlags::DEBUG_REGISTERS) { + litebox_util_log::warn!( + "NtContinue requested debug-register state, which is not yet restored" + ); + } + + Ok(()) + } + + fn test_alert() { + // TODO(apc-model): Deliver queued user-mode APCs once thread alert and APC state + // are modeled. + litebox_util_log::debug!( + "NtTestAlert is a no-op; user-mode APC delivery is not yet modeled" + ); + } + + pub(crate) fn sys_nt_close(&self, handle: syscalls::Handle) -> NtStatus { + self.close_handle(handle, CloseRawHandleVisitor { task: self }) + } + + #[expect( + clippy::too_many_arguments, + reason = "matches the native NtDuplicateObject contract" + )] + pub(crate) fn sys_nt_duplicate_object( + &self, + source_process_handle: syscalls::ProcessHandle, + source_handle: syscalls::Handle, + target_process_handle: syscalls::ProcessHandle, + target_handle: Option>, + desired_access: u32, + handle_attributes: u32, + options: u32, + ) -> NtStatus { + let options = DuplicateOptions::from_bits_retain(options); + if let Some(target_handle) = target_handle + && target_handle + .write_at_offset(0, syscalls::Handle::default()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if !source_process_handle.is_current() { + // TODO(duplicate-object-cross-process): resolve process handles once LiteBox supports + // multiple guest processes and per-process handle tables. + return NtStatus::INVALID_HANDLE; + } + + let status = self.duplicate_object( + source_handle, + target_process_handle, + target_handle, + desired_access, + handle_attributes, + options, + ); + + if options.contains(DuplicateOptions::CLOSE_SOURCE) { + let _ = self.sys_nt_close(source_handle); + } + status + } + + fn duplicate_object( + &self, + source_handle: syscalls::Handle, + target_process_handle: syscalls::ProcessHandle, + target_handle: Option>, + desired_access: u32, + handle_attributes: u32, + options: DuplicateOptions, + ) -> NtStatus { + if target_process_handle.is_null() { + return if options.contains(DuplicateOptions::CLOSE_SOURCE) { + NtStatus::SUCCESS + } else { + NtStatus::INVALID_PARAMETER + }; + } + if !target_process_handle.is_current() { + // TODO(duplicate-object-cross-process): insert into the target process handle table. + return NtStatus::INVALID_HANDLE; + } + let duplicate = match self.duplicate_handle( + source_handle, + desired_access, + handle_attributes, + options, + ) { + Ok(handle) => handle, + Err(status) => return status, + }; + if let Some(target_handle) = target_handle + && target_handle.write_at_offset(0, duplicate).is_none() + { + let _ = self.remove_handle(duplicate, CloseRawHandleVisitor { task: self }, false); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + fn duplicate_handle( + &self, + source_handle: syscalls::Handle, + desired_access: u32, + handle_attributes: u32, + options: DuplicateOptions, + ) -> Result { + macro_rules! try_duplicate { + ($subsystem:ty) => { + if let Some(result) = self.try_duplicate_handle::<$subsystem>( + source_handle, + desired_access, + handle_attributes, + options, + ) { + return result; + } + }; + } + + try_duplicate!(FileObjectSubsystem); + try_duplicate!(RegistryKeySubsystem); + try_duplicate!(EventSubsystem); + try_duplicate!(DirectoryObjectSubsystem); + try_duplicate!(SymbolicLinkSubsystem); + try_duplicate!(IoCompletionSubsystem); + try_duplicate!(LpcPortSubsystem); + try_duplicate!(TimerSubsystem); + try_duplicate!(WaitCompletionPacketSubsystem); + try_duplicate!(WorkerFactorySubsystem); + try_duplicate!(SectionSubsystem); + try_duplicate!(TokenSubsystem); + + Err(NtStatus::INVALID_HANDLE) + } + + fn try_duplicate_handle( + &self, + source_handle: syscalls::Handle, + desired_access: u32, + handle_attributes: u32, + options: DuplicateOptions, + ) -> Option> + where + Subsystem: WindowsHandleSubsystem, + { + let typed = match self.typed_handle::(source_handle) { + Ok(typed) => typed, + Err(NtStatus::OBJECT_TYPE_MISMATCH) => return None, + Err(status) => return Some(Err(status)), + }; + let source_metadata = match self.typed_handle_metadata(&typed) { + Ok(metadata) => metadata, + Err(status) => return Some(Err(status)), + }; + + let source_access = source_metadata.granted_access; + let duplicate_access = if options.contains(DuplicateOptions::SAME_ACCESS) { + source_access + } else { + let descriptors = self.global.litebox.descriptor_table(); + match descriptors.with_entry(&typed, |entry| { + Subsystem::resolve_duplicate_access(entry, desired_access) + }) { + Some(Ok(access)) => access, + Some(Err(status)) => return Some(Err(status)), + None => return Some(Err(NtStatus::INVALID_HANDLE)), + } + }; + let duplicate_attributes = if options.contains(DuplicateOptions::SAME_ATTRIBUTES) { + source_metadata.attributes + } else { + let Some(attributes) = HandleAttributes::from_duplicate_attributes(handle_attributes) + else { + return Some(Err(NtStatus::INVALID_PARAMETER)); + }; + attributes + }; + + let duplicate = { + let mut descriptors = self.global.litebox.descriptor_table_mut(); + let Some(duplicate) = descriptors.duplicate(&typed) else { + return Some(Err(NtStatus::INVALID_HANDLE)); + }; + let old = descriptors.set_fd_metadata( + &duplicate, + WindowsHandleMetadata { + granted_access: duplicate_access, + attributes: duplicate_attributes, + }, + ); + debug_assert!(old.is_none()); + duplicate + }; + Some(insert_raw_handle::( + &self.global.litebox, + &self.process.handles, + duplicate, + drop, + )) + } + + fn close_handle( + &self, + handle: syscalls::Handle, + visitor: impl RawHandleVisitor, + ) -> NtStatus { + self.remove_handle(handle, visitor, true) + } + + fn remove_handle( + &self, + handle: syscalls::Handle, + visitor: impl RawHandleVisitor, + enforce_protect_from_close: bool, + ) -> NtStatus { + macro_rules! try_close { + ($subsystem:ty, $visit:ident) => { + if let Some(status) = self.try_remove_handle::<$subsystem>( + handle, + enforce_protect_from_close, + |entry| visitor.$visit(entry), + ) { + return status; + } + }; + } + + try_close!(FileObjectSubsystem, file); + try_close!(RegistryKeySubsystem, registry_key); + try_close!(EventSubsystem, event); + try_close!(DirectoryObjectSubsystem, directory); + try_close!(SymbolicLinkSubsystem, symbolic_link); + try_close!(IoCompletionSubsystem, io_completion); + try_close!(LpcPortSubsystem, lpc_port); + try_close!(TimerSubsystem, timer); + try_close!( + WaitCompletionPacketSubsystem, + wait_completion_packet + ); + try_close!(WorkerFactorySubsystem, worker_factory); + try_close!(SectionSubsystem, section); + try_close!(TokenSubsystem, token); + + NtStatus::INVALID_HANDLE + } + + fn try_remove_handle( + &self, + handle: syscalls::Handle, + enforce_protect_from_close: bool, + cleanup_entry: impl FnOnce(Subsystem::Entry), + ) -> Option + where + Subsystem: WindowsHandleSubsystem, + { + let typed = match self.typed_handle::(handle) { + Ok(typed) => typed, + Err(NtStatus::OBJECT_TYPE_MISMATCH) => return None, + Err(status) => return Some(status), + }; + let metadata = match self.typed_handle_metadata(&typed) { + Ok(metadata) => metadata, + Err(status) => return Some(status), + }; + if enforce_protect_from_close + && metadata + .attributes + .contains(HandleAttributes::PROTECT_FROM_CLOSE) + { + return Some(NtStatus::HANDLE_NOT_CLOSABLE); + } + + remove_raw_handle::( + &self.global.litebox, + &self.process.handles, + handle, + cleanup_entry, + ); + Some(NtStatus::SUCCESS) + } + + fn handle_interrupt_request( + &self, + _ctx: &mut litebox_common_linux::PtRegs, + ) -> ContinueOperation { + litebox_util_log::debug!( + stack_top:% = format_args!("{:#x}", self.stack_top); + "Windows guest interrupt" + ); + ContinueOperation::Resume + } +} + +trait RawHandleVisitor { + fn file(&self, file: FileObject); + + fn registry_key(&self, key: RegistryKeyObject); + + fn event(&self, event: EventHandleObject); + + fn directory(&self, directory: DirectoryHandleObject); + + fn symbolic_link(&self, link: SymbolicLinkHandleObject); + + fn io_completion(&self, io_completion: IoCompletionHandleObject); + + fn lpc_port(&self, lpc_port: LpcPortHandleObject); + + fn timer(&self, timer: TimerHandleObject); + + fn wait_completion_packet( + &self, + wait_completion_packet: WaitCompletionPacketHandleObject, + ); + + fn worker_factory(&self, worker_factory: WorkerFactoryHandleObject); + + fn section(&self, section: SectionHandleObject); + + fn token(&self, token: TokenHandleObject); +} + +struct CloseRawHandleVisitor<'task, Platform: ShimPlatform, FS: ShimFS> { + task: &'task Task, +} + +impl RawHandleVisitor + for CloseRawHandleVisitor<'_, Platform, FS> +{ + fn file(&self, file: FileObject) { + self.task.close_file(file); + } + + fn registry_key(&self, key: RegistryKeyObject) { + self.task.close_registry_key(key); + } + + fn event(&self, event: EventHandleObject) { + Task::::close_event(event); + } + + fn directory(&self, directory: DirectoryHandleObject) { + Task::::close_directory(directory); + } + + fn symbolic_link(&self, link: SymbolicLinkHandleObject) { + Task::::close_symbolic_link(link); + } + + fn io_completion(&self, io_completion: IoCompletionHandleObject) { + Task::::close_io_completion(io_completion); + } + + fn lpc_port(&self, lpc_port: LpcPortHandleObject) { + Task::::close_lpc_port(lpc_port); + } + + fn timer(&self, timer: TimerHandleObject) { + Task::::close_timer(timer); + } + + fn wait_completion_packet( + &self, + wait_completion_packet: WaitCompletionPacketHandleObject, + ) { + Task::::close_wait_completion_packet(wait_completion_packet); + } + + fn worker_factory(&self, worker_factory: WorkerFactoryHandleObject) { + Task::::close_worker_factory(worker_factory); + } + + fn section(&self, section: SectionHandleObject) { + Task::::close_section(section); + } + + fn token(&self, token: TokenHandleObject) { + Task::::close_token(token); + } +} + +/// The shim entrypoint object passed to the platform. +pub struct WindowsShimEntrypoints { + task: Task, + _not_send: PhantomData<*const ()>, +} + +impl EnterShim for WindowsShimEntrypoints { + type ExecutionContext = litebox_common_linux::PtRegs; + + fn init(&self, ctx: &mut Self::ExecutionContext) -> ContinueOperation { + self.task.init(ctx) + } + + fn syscall(&self, ctx: &mut Self::ExecutionContext) -> ContinueOperation { + self.task.handle_syscall_request(ctx) + } + + fn exception( + &self, + ctx: &mut Self::ExecutionContext, + info: &ExceptionInfo, + ) -> ContinueOperation { + litebox_util_log::debug!( + exception:? = info.exception, + rip:% = format_args!("{:#x}", ctx.rip), + cr2:% = format_args!("{:#x}", info.cr2); + "Windows guest exception" + ); + // TODO: Translate hardware exceptions into Windows SEH where appropriate. + ContinueOperation::Terminate + } + + fn interrupt(&self, ctx: &mut Self::ExecutionContext) -> ContinueOperation { + self.task.handle_interrupt_request(ctx) + } +} + +/// A loaded Windows program and the process handle used to wait for it. +pub struct LoadedProgram { + /// The initial-thread entrypoint state passed to the platform's `run_thread`. + pub entrypoints: WindowsShimEntrypoints, + /// Handle used to wait for the loaded program to exit. + pub process: Arc>, +} + +fn default_fs( + litebox: &LiteBox, + in_mem_fs: litebox::fs::in_mem::FileSystem, + tar_data: Cow<'static, [u8]>, +) -> WindowsFS +where + Platform: ShimPlatform + CrngProvider + StdioProvider, +{ + let devices = litebox::fs::resolver::Resolver::new( + litebox, + litebox::fs::composer::Composer::builder() + .mount("/dev", |allocator| { + litebox::fs::devices::Devices::new(litebox, allocator) + }) + .build() + .unwrap(), + ); + let tar_ro = litebox::fs::resolver::Resolver::new( + litebox, + litebox::fs::composer::Composer::builder() + .mount("/", |allocator| { + litebox::fs::tar_ro::TarRo::new(tar_data, allocator) + }) + .build() + .unwrap(), + ); + litebox::fs::layered::FileSystem::new( + litebox, + in_mem_fs, + litebox::fs::layered::FileSystem::new( + litebox, + devices, + tar_ro, + litebox::fs::layered::LayeringSemantics::LowerLayerReadOnly, + ), + litebox::fs::layered::LayeringSemantics::LowerLayerWritableFiles, + ) +} diff --git a/litebox_shim_windows/src/loader/mod.rs b/litebox_shim_windows/src/loader/mod.rs new file mode 100644 index 0000000000..7922fdcdb8 --- /dev/null +++ b/litebox_shim_windows/src/loader/mod.rs @@ -0,0 +1,7 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +mod pe; + +pub(super) use pe::{PeLoader, WindowsLoadError}; +pub(crate) use pe::{image_section_metadata, load_image_section}; diff --git a/litebox_shim_windows/src/loader/pe.rs b/litebox_shim_windows/src/loader/pe.rs new file mode 100644 index 0000000000..ea85684cd1 --- /dev/null +++ b/litebox_shim_windows/src/loader/pe.rs @@ -0,0 +1,2856 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::collections::btree_map::BTreeMap; +use alloc::{ffi::CString, string::String, sync::Arc, vec::Vec}; +use core::{ + marker::PhantomData, + mem::{align_of, size_of}, +}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt as _; +use litebox::{ + fs::{Mode, OFlags}, + mm::linux::{ + CreatePagesFlags, MappingError, NonZeroAddress, NonZeroPageSize, VmemProtectError, + }, + platform::RawPointerProvider, +}; +use litebox_common_windows::loader::{ + AccessMemory, Fault, KiUserInvertedFunctionTableEntry, KiUserInvertedFunctionTableHeader, + MAXIMUM_INVERTED_FUNCTION_TABLE_SIZE, MapMemory, MappingInfo, PAGE_SIZE, PeExportError, + PeLoadError, PeParseError, PeParsedFile, Protection, ReadAt, build_api_set_namespace, + page_align_down, +}; +use rangemap::RangeMap; +use thiserror::Error; +use zerocopy::{FromBytes, FromZeros, Immutable, IntoBytes, KnownLayout}; + +use crate::nt_types::{ + ClientId, Luid, PebBitField, ProcessEnvironmentBlock, RtlUserProcFlags, + RtlUserProcessParameters, ThreadEnvironmentBlock, UnicodeString, X64Context, +}; +use crate::syscalls::mm::{MemoryType, PageProtection}; +use crate::syscalls::process::{INITIAL_PROCESS_ID, INITIAL_THREAD_ID}; +use crate::{MutPtr, ShimFS}; + +const NTDLL_WRITABLE_SECTIONS: &[&[u8]] = &[b".mrdata"]; +const NTDLL_PATH: &str = "/Windows/System32/ntdll.dll"; +const RUNTIME_FUNCTION_ENTRY_SIZE: usize = 12; +const ZERO_CHUNK: [u8; PAGE_SIZE] = [0; PAGE_SIZE]; +const FILE_CHUNK_BYTES: usize = 64 * 1024; +const INITIAL_STACK_SIZE: usize = 1024 * 1024; +const WINDOWS_SHARED_SECTION_SIZE: usize = 0x1_0000; +const CSR_SERVER_DLL_MAX: usize = 4; +const BASESRV_SERVERDLL_INDEX: usize = 1; +const WINDOWS_CRITICAL_SECTION_TIMEOUT_100NS: i64 = -150 * 10_000_000; +const WINDOWS_HEAP_SEGMENT_RESERVE: u64 = 1024 * 1024; +const WINDOWS_HEAP_SEGMENT_COMMIT: u64 = 2 * PAGE_SIZE as u64; +const WINDOWS_HEAP_DECOMMIT_TOTAL_FREE_THRESHOLD: u64 = 64 * 1024; +const WINDOWS_HEAP_DECOMMIT_FREE_BLOCK_THRESHOLD: u64 = PAGE_SIZE as u64; +const WINDOWS_NT_TIB_VERSION: usize = 30 << 8; + +macro_rules! write_static_server_data_field { + ($platform:ty, $base:expr, $field:ident, $value:expr $(,)?) => { + write_guest_field_at_offset::<$platform, _, _>( + $base, + core::mem::offset_of!(BaseStaticServerData, $field), + $value, + ) + }; +} + +pub(crate) struct WindowsProcessEnvironment { + pub(crate) peb: usize, + pub(crate) teb: usize, + pub(crate) context: usize, + pub(crate) windows_shared_section: usize, +} + +pub(crate) struct PeLoadInfo { + pub(crate) entry_point: usize, + pub(crate) stack_top: usize, + pub(crate) ntdll_mapping: Option, + pub(crate) virtual_allocations: crate::WindowsVirtualAllocations, + pub(crate) environment: WindowsProcessEnvironment, +} + +struct ProcessEnvironmentInput<'a> { + image: &'a PeParsedFile, + image_base_address: usize, + image_path: &'a str, + argv: &'a [CString], + envp: &'a [CString], + stack_base: usize, + stack_allocation_top: usize, +} + +pub(crate) struct PeLoader<'a, Platform: crate::ShimPlatform, FS: ShimFS> { + platform: &'static Platform, + fs: Arc, + page_manager: &'a crate::WindowsPageManager, +} + +impl<'a, Platform: crate::ShimPlatform, FS: ShimFS> PeLoader<'a, Platform, FS> { + pub(crate) fn new( + platform: &'static Platform, + fs: Arc, + page_manager: &'a crate::WindowsPageManager, + ) -> Self { + Self { + platform, + fs, + page_manager, + } + } + + pub(crate) fn load( + &self, + path: &str, + argv: &[CString], + envp: &[CString], + ) -> Result, WindowsLoadError> { + let image = load_image(self.platform, self.fs.clone(), path, self.page_manager)?; + let application_entry_point = image.mapping.entry_point; + let ntdll = load_ntdll(self.platform, self.fs.clone(), self.page_manager)?; + + let entry_point = if let Some(ntdll) = &ntdll { + if !ntdll.image.parsed.has_trampoline() { + return Err(WindowsLoadError::UnrewrittenNtDll); + } + Self::initialize_ki_user_inverted_function_table(&image, ntdll)?; + ntdll.exports.ldr_initialize_thunk + } else { + application_entry_point + }; + + let length = + NonZeroPageSize::new(INITIAL_STACK_SIZE).ok_or(PeImageAccessError::AddressOverflow)?; + // SAFETY: `suggested_address` is `None` and `CreatePagesFlags::empty()` does not set + // `fixed_addr`, so the page manager picks an unused region and cannot replace a mapping. + let stack_base = unsafe { + self.page_manager + .create_stack_pages(None, length, CreatePagesFlags::empty()) + } + .map_err(PeImageAccessError::Mapping)?; + let stack_allocation_top = stack_base + .as_usize() + .checked_add(INITIAL_STACK_SIZE) + .ok_or(PeImageAccessError::AddressOverflow)?; + let stack_top = if stack_allocation_top.is_multiple_of(16) { + stack_allocation_top - core::mem::size_of::() + } else { + stack_allocation_top + }; + + let environment = self.create_process_environment(ProcessEnvironmentInput { + image: &image.parsed, + image_base_address: image.mapping.base_addr, + image_path: path, + argv, + envp, + stack_base: stack_base.as_usize(), + stack_allocation_top, + })?; + if let Some(ntdll) = &ntdll { + let context = X64Context::initial_thread_context( + ntdll.exports.rtl_user_thread_start, + application_entry_point, + stack_top, + environment.peb, + ); + write_guest_slice::(environment.context, context.as_bytes())?; + } + + let virtual_allocations = + crate::WindowsVirtualAllocations::::new(BTreeMap::new()); + register_image_virtual_allocation(&virtual_allocations, image.mapping, image.pages); + let ntdll_mapping = if let Some(ntdll) = ntdll { + let mapping = ntdll.image.mapping; + register_image_virtual_allocation(&virtual_allocations, mapping, ntdll.image.pages); + Some(mapping) + } else { + None + }; + + Ok(PeLoadInfo { + entry_point, + stack_top, + ntdll_mapping, + virtual_allocations, + environment, + }) + } + + fn initialize_ki_user_inverted_function_table( + application: &LoadedImage, + ntdll: &LoadedNtDll, + ) -> Result<(), WindowsLoadError> { + let table_address = ntdll.exports.ki_user_inverted_function_table; + + let mut entries = Vec::new(); + for image in [&ntdll.image, application] { + if let Some(entry) = image.inverted_function_table_entry()? { + entries.push(entry); + } + } + + let header = KiUserInvertedFunctionTableHeader { + current_size: entries.len().trunc(), + maximum_size: MAXIMUM_INVERTED_FUNCTION_TABLE_SIZE, + epoch: 0, + overflow: 0, + padding_0: [0; 3], + }; + + // `KI_USER_INVERTED_FUNCTION_TABLE` lives in ntdll's writable `.mrdata` section. + write_guest_value::(table_address, header)?; + let entries_address = table_address + .checked_add(core::mem::size_of::()) + .ok_or(PeImageAccessError::AddressOverflow)?; + write_guest_slice::(entries_address, &entries)?; + + litebox_util_log::debug!( + table:% = format_args!("{table_address:#x}"); + "Initialized ntdll!KiUserInvertedFunctionTable" + ); + + Ok(()) + } + + fn create_process_environment( + &self, + input: ProcessEnvironmentInput<'_>, + ) -> Result { + let create_pages = |size: usize| -> Result { + let aligned_length = size.next_multiple_of(PAGE_SIZE); + let length = + NonZeroPageSize::new(aligned_length).ok_or(PeImageAccessError::AddressOverflow)?; + // SAFETY: `suggested_address` is `None` and `CreatePagesFlags::empty()` leaves address + // selection to the page manager, so this cannot replace an existing mapping. + let ptr = unsafe { + self.page_manager.create_writable_pages( + None, + length, + CreatePagesFlags::empty(), + |_| Ok(0), + ) + }?; + Ok(ptr.as_usize()) + }; + let teb_ptr = create_pages(size_of::())?; + let peb_ptr = create_pages(size_of::())?; + let api_set_map = build_api_set_namespace(API_SET_MAPPINGS) + .map_err(|_| PeImageAccessError::AddressOverflow)?; + let api_set_map_ptr = create_pages(api_set_map.len())?; + write_guest_slice::(api_set_map_ptr, &api_set_map)?; + let ctx_ptr = create_pages(size_of::())?; + + let win32_image_path = win32_image_path(input.image_path); + let dos_image_path = dos_image_path(input.image_path); + let current_directory_path = Utf16StringBuffer::new(r"C:\")?; + let dll_path = Utf16StringBuffer::new(r"C:\Windows\System32;C:\")?; + let image_path_name = Utf16StringBuffer::new(&dos_image_path)?; + let command_line = + Utf16StringBuffer::new(&windows_command_line(&win32_image_path, input.argv))?; + let window_title = Utf16StringBuffer::new(&dos_image_path)?; + let desktop_info = Utf16StringBuffer::new("")?; + let shell_info = Utf16StringBuffer::new("")?; + let runtime_data = Utf16StringBuffer::new("")?; + let redirection_dll_name = Utf16StringBuffer::new("")?; + let environment_block = windows_environment_block(input.envp); + let environment_size = checked_mul(environment_block.len(), size_of::())?; + let environment_ptr = create_pages(environment_size)?; + write_guest_slice::(environment_ptr, &environment_block)?; + let process_parameter_strings = [ + ¤t_directory_path, + &dll_path, + &image_path_name, + &command_line, + &window_title, + &desktop_info, + &shell_info, + &runtime_data, + &redirection_dll_name, + ]; + let process_parameters_length = process_parameter_strings.iter().try_fold( + size_of::(), + |length, string| { + length + .checked_add(usize::from(string.maximum_length)) + .ok_or(PeImageAccessError::AddressOverflow) + }, + )?; + let process_parameters_allocation_length = + process_parameters_length.next_multiple_of(PAGE_SIZE); + let process_parameters_ptr = create_pages(process_parameters_length)?; + + let mut process_parameters = RtlUserProcessParameters::new_zeroed(); + process_parameters.maximum_length = to_u32(process_parameters_allocation_length)?; + process_parameters.length = to_u32(process_parameters_length)?; + process_parameters.flags = RtlUserProcFlags::NORMALIZED.bits(); + process_parameters.environment = environment_ptr; + process_parameters.environment_size = + u64::try_from(environment_size).map_err(|_| PeImageAccessError::AddressOverflow)?; + let mut process_parameters_allocation = + GuestMemoryAllocator::new(process_parameters_ptr, process_parameters_length)?; + let guest_process_parameters = + process_parameters_allocation.allocate::()?; + process_parameters.current_directory.dos_path = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + ¤t_directory_path, + )?; + process_parameters.dll_path = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &dll_path, + )?; + process_parameters.image_path_name = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &image_path_name, + )?; + process_parameters.command_line = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &command_line, + )?; + process_parameters.window_title = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &window_title, + )?; + process_parameters.desktop_info = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &desktop_info, + )?; + process_parameters.shell_info = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &shell_info, + )?; + process_parameters.runtime_data = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &runtime_data, + )?; + process_parameters.redirection_dll_name = allocate_guest_unicode_string::( + &mut process_parameters_allocation, + &redirection_dll_name, + )?; + guest_process_parameters + .write_at_offset(0, process_parameters) + .ok_or(PeImageAccessError::MemoryAccess)?; + + let read_only_shared_memory_base = create_pages(WINDOWS_SHARED_SECTION_SIZE)?; + let mut shared_heap = + GuestMemoryAllocator::new(read_only_shared_memory_base, WINDOWS_SHARED_SECTION_SIZE)?; + let read_only_static_server_data = + initialize_windows_static_server_data::(&mut shared_heap)?; + let mut peb = ProcessEnvironmentBlock::new_zeroed(); + peb.image_base_address = input.image_base_address; + if input.image_base_address != input.image.image_base() || input.image.has_dynamic_base() { + peb.bit_field = PebBitField::IS_IMAGE_DYNAMICALLY_RELOCATED.bits(); + } + let process_heaps = initial_process_heaps_array(peb_ptr)?; + let fast_peb_lock = create_pages(size_of::())?; + write_guest_value::(fast_peb_lock, RtlCriticalSection::initialized(0))?; + let loader_lock = create_pages(size_of::())?; + write_guest_value::(loader_lock, RtlCriticalSection::initialized(0))?; + + peb.api_set_map = api_set_map_ptr; + peb.process_parameters = process_parameters_ptr; + peb.fast_peb_lock = fast_peb_lock; + peb.shared_data = read_only_shared_memory_base; + peb.number_of_processors = 1; + peb.critical_section_timeout = WINDOWS_CRITICAL_SECTION_TIMEOUT_100NS; + peb.heap_segment_reserve = WINDOWS_HEAP_SEGMENT_RESERVE; + peb.heap_segment_commit = WINDOWS_HEAP_SEGMENT_COMMIT; + peb.heap_de_commit_total_free_threshold = WINDOWS_HEAP_DECOMMIT_TOTAL_FREE_THRESHOLD; + peb.heap_de_commit_free_block_threshold = WINDOWS_HEAP_DECOMMIT_FREE_BLOCK_THRESHOLD; + peb.maximum_number_of_heaps = process_heaps.maximum_number_of_heaps; + peb.process_heaps = process_heaps.address; + peb.loader_lock = loader_lock; + peb.active_process_affinity_mask = 1; + peb.os_major_version = u32::from(crate::syscalls::sysinfo::WINDOWS_OS_MAJOR_VERSION); + peb.os_minor_version = u32::from(crate::syscalls::sysinfo::WINDOWS_OS_MINOR_VERSION); + peb.os_build_number = crate::syscalls::sysinfo::WINDOWS_OS_BUILD_NUMBER; + peb.os_platform_id = crate::syscalls::sysinfo::WINDOWS_OS_PLATFORM_WIN32_NT; + peb.image_subsystem = u32::from(input.image.subsystem()); + peb.image_subsystem_major_version = u32::from(input.image.major_subsystem_version()); + peb.image_subsystem_minor_version = u32::from(input.image.minor_subsystem_version()); + peb.read_only_shared_memory_base = read_only_shared_memory_base; + peb.read_only_static_server_data = read_only_static_server_data; + // TODO(csr-shared-section): model shared backing with distinct client and CSRSS + // virtual addresses instead of aliasing both PEB bases to this single mapping. + peb.csr_server_read_only_shared_memory_base = read_only_shared_memory_base as u64; + + write_guest_value::(peb_ptr, peb)?; + + let mut teb = ThreadEnvironmentBlock::new_zeroed(); + teb.nt_tib.exception_list = 0; + teb.nt_tib.stack_base = input.stack_allocation_top; + teb.nt_tib.stack_limit = input.stack_base; + teb.nt_tib.fiber_data_or_version = WINDOWS_NT_TIB_VERSION; + teb.nt_tib.self_pointer = teb_ptr; + // TODO: set real ID + teb.client_id = ClientId { + unique_process: INITIAL_PROCESS_ID, + unique_thread: INITIAL_THREAD_ID, + }; + teb.thread_local_storage_pointer = + teb_ptr + core::mem::offset_of!(ThreadEnvironmentBlock, tls_slots); + teb.process_environment_block = peb_ptr; + teb.real_client_id = teb.client_id; + teb.activation_context_stack_pointer = + teb_ptr + core::mem::offset_of!(ThreadEnvironmentBlock, activation_stack); + teb.static_unicode_string = + initial_teb_static_unicode_string(teb_ptr, &teb.static_unicode_buffer)?; + teb.deallocation_stack = input.stack_base; + write_guest_value::(teb_ptr, teb)?; + Ok(WindowsProcessEnvironment { + peb: peb_ptr, + teb: teb_ptr, + context: ctx_ptr, + windows_shared_section: read_only_shared_memory_base, + }) + } +} + +fn initialize_windows_static_server_data( + shared_heap: &mut GuestMemoryAllocator, +) -> Result { + let read_only_static_server_data = + shared_heap.allocate_array::(CSR_SERVER_DLL_MAX)?; + // TODO(csr-server-dlls): populate CSRSRV (0), CONSRV (2), and USERSRV (3) + // when their shared static server data is modeled. + let client_base_static_server_data = + shared_heap.allocate::()?; + initialize_static_server_data::(shared_heap, client_base_static_server_data)?; + + read_only_static_server_data + .write_at_offset( + BASESRV_SERVERDLL_INDEX.cast_signed(), + client_base_static_server_data.as_usize(), + ) + .ok_or(PeImageAccessError::MemoryAccess)?; + Ok(read_only_static_server_data.as_usize()) +} + +fn initialize_static_server_data( + shared_heap: &mut GuestMemoryAllocator, + base_static_server_data: MutPtr, +) -> Result<(), PeImageAccessError> { + let windows_directory = allocate_guest_unicode_string_from_str::( + shared_heap, + crate::syscalls::sysinfo::WINDOWS_DIRECTORY, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + windows_directory, + windows_directory, + )?; + let windows_system_directory = allocate_guest_unicode_string_from_str::( + shared_heap, + crate::syscalls::sysinfo::WINDOWS_SYSTEM_DIRECTORY, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + windows_system_directory, + windows_system_directory, + )?; + let named_object_directory = allocate_guest_unicode_string_from_str::( + shared_heap, + crate::syscalls::sysinfo::WINDOWS_NAMED_OBJECT_DIRECTORY, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + named_object_directory, + named_object_directory, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + windows_major_version, + crate::syscalls::sysinfo::WINDOWS_OS_MAJOR_VERSION, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + windows_minor_version, + crate::syscalls::sysinfo::WINDOWS_OS_MINOR_VERSION, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + build_number, + crate::syscalls::sysinfo::WINDOWS_OS_BUILD_NUMBER, + )?; + + let ini_file_mapping = shared_heap + .allocate::()? + .as_usize(); + write_static_server_data_field!( + Platform, + base_static_server_data, + ini_file_mapping, + ini_file_mapping, + )?; + write_static_server_data_field!( + Platform, + base_static_server_data, + termsrv_client_time_zone_id, + crate::syscalls::sysinfo::WINDOWS_TIME_ZONE_ID_INVALID, + )?; + Ok(()) +} + +fn register_image_virtual_allocation( + virtual_allocations: &crate::WindowsVirtualAllocations, + mapping: MappingInfo, + pages: RangeMap, +) { + virtual_allocations.write().insert( + mapping.base_addr, + crate::WindowsVirtualAllocation { + base: mapping.base_addr, + size: mapping.mapping_size, + allocation_protect: PageProtection::PAGE_EXECUTE_WRITECOPY, + type_: MemoryType::MEM_IMAGE, + pages, + }, + ); +} + +struct LoadedImage { + mapping: MappingInfo, + pages: RangeMap, + parsed: PeParsedFile, +} + +impl LoadedImage { + fn inverted_function_table_entry( + &self, + ) -> Result, WindowsLoadError> { + let Some(exception_directory) = self.parsed.exception_directory() else { + return Ok(None); + }; + if !exception_directory + .size + .is_multiple_of(RUNTIME_FUNCTION_ENTRY_SIZE) + { + return Err(WindowsLoadError::InvalidNtDllExceptionDirectory); + } + + let exception_directory_address = self + .mapping + .base_addr + .checked_add(exception_directory.rva) + .ok_or(PeImageAccessError::AddressOverflow)?; + let size_of_image = u32::try_from(self.parsed.image_size()) + .map_err(|_| PeImageAccessError::AddressOverflow)?; + + Ok(Some(KiUserInvertedFunctionTableEntry { + exception_directory_address, + image_base: self.mapping.base_addr, + image_size: size_of_image, + size_of_table: u32::try_from(exception_directory.size) + .map_err(|_| PeImageAccessError::AddressOverflow)?, + })) + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, FromBytes, Immutable, IntoBytes, KnownLayout)] +struct RtlCriticalSection { + debug_info: usize, + lock_count: i32, + recursion_count: u32, + owning_thread: usize, + lock_semaphore: usize, + spin_count: usize, +} + +impl RtlCriticalSection { + const fn initialized(spin_count: usize) -> Self { + Self { + debug_info: usize::MAX, + lock_count: -1, + recursion_count: 0, + owning_thread: 0, + lock_semaphore: 0, + spin_count, + } + } +} + +struct GuestMemoryAllocator { + cursor: usize, + end: usize, +} + +impl GuestMemoryAllocator { + fn new(base: usize, size: usize) -> Result { + let cursor = base; + let end = checked_add(base, size)?; + if cursor > end { + return Err(PeImageAccessError::AddressOverflow); + } + Ok(Self { cursor, end }) + } + + fn allocate(&mut self) -> Result, PeImageAccessError> + where + Platform: RawPointerProvider, + T: FromBytes + IntoBytes, + { + self.allocate_array::(1) + } + + fn allocate_array( + &mut self, + count: usize, + ) -> Result, PeImageAccessError> + where + Platform: RawPointerProvider, + T: FromBytes + IntoBytes, + { + let address = self.allocate_bytes(checked_mul(size_of::(), count)?, align_of::())?; + Ok(MutPtr::::from_usize(address)) + } + + fn allocate_bytes( + &mut self, + size: usize, + alignment: usize, + ) -> Result { + debug_assert!(alignment.is_power_of_two()); + let address = self + .cursor + .checked_next_multiple_of(alignment) + .ok_or(PeImageAccessError::AddressOverflow)?; + let cursor = checked_add(address, size)?; + if cursor > self.end { + return Err(PeImageAccessError::AddressOverflow); + } + self.cursor = cursor; + Ok(address) + } +} + +// Reference layout from ReactOS `sdk/include/reactos/subsys/win/base.h`. +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct BaseStaticServerData { + windows_directory: UnicodeString, + windows_system_directory: UnicodeString, + named_object_directory: UnicodeString, + windows_major_version: u16, + windows_minor_version: u16, + build_number: u16, + csd_number: u16, + rc_number: u16, + csd_version: [u16; 128], + padding_0: [u8; 6], + sys_info: SystemBasicInformation, + time_of_day: SystemTimeOfDayInformation, + ini_file_mapping: usize, + nls_user_info: NlsUserInfo, + default_separate_vdm: u8, + is_wow_task_ready: u8, + padding_1: [u8; 6], + windows_sys32_x86_directory: UnicodeString, + f_termsrv_app_install_mode: u8, + padding_2: [u8; 3], + tzi_termsrv_client_time_zone: TimeZoneInformation, + kt_termsrv_client_bias: crate::nt_types::KSystemTime, + termsrv_client_time_zone_id: u32, + luid_device_maps_enabled: u8, + padding_3: [u8; 3], + termsrv_client_time_zone_change_num: u32, +} + +#[allow(clippy::struct_field_names)] +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct IniFileMapping { + file_names: usize, + default_file_name_mapping: usize, + win_ini_file_mapping: usize, + reserved: u32, + padding: [u8; 4], +} + +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct SystemBasicInformation { + reserved: u32, + timer_resolution: u32, + page_size: u32, + number_of_physical_pages: u32, + lowest_physical_page_number: u32, + highest_physical_page_number: u32, + allocation_granularity: u32, + padding_0: [u8; 4], + minimum_user_mode_address: usize, + maximum_user_mode_address: usize, + active_processors_affinity_mask: usize, + number_of_processors: u8, + padding_1: [u8; 7], +} + +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct SystemTimeOfDayInformation { + boot_time: i64, + current_time: i64, + time_zone_bias: i64, + time_zone_id: u32, + reserved: u32, + boot_time_bias: u64, + sleep_time_bias: u64, +} + +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct NlsUserInfo { + s_language: [u16; 80], + i_country: [u16; 80], + s_country: [u16; 80], + s_list: [u16; 80], + i_measure: [u16; 80], + i_paper_size: [u16; 80], + s_decimal: [u16; 80], + s_thousand: [u16; 80], + s_grouping: [u16; 80], + i_digits: [u16; 80], + i_l_zero: [u16; 80], + i_neg_number: [u16; 80], + s_native_digits: [u16; 80], + num_shape: [u16; 80], + s_currency: [u16; 80], + s_mon_dec_sep: [u16; 80], + s_mon_thou_sep: [u16; 80], + s_mon_grouping: [u16; 80], + i_curr_digits: [u16; 80], + i_currency: [u16; 80], + i_neg_curr: [u16; 80], + s_positive_sign: [u16; 80], + s_negative_sign: [u16; 80], + s_time_format: [u16; 80], + s_time: [u16; 80], + i_time: [u16; 80], + i_tl_zero: [u16; 80], + i_time_prefix: [u16; 80], + s_1159: [u16; 80], + s_2359: [u16; 80], + s_short_date: [u16; 80], + s_date: [u16; 80], + i_date: [u16; 80], + s_year_month: [u16; 80], + s_long_date: [u16; 80], + i_cal_type: [u16; 80], + i_first_day_of_week: [u16; 80], + i_first_week_of_year: [u16; 80], + locale: [u16; 80], + user_locale_id: u32, + interactive_user_luid: Luid, + ul_cache_update_count: u32, +} + +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct TimeZoneInformation { + bias: i32, + standard_name: [u16; 32], + standard_date: SystemTime, + standard_bias: i32, + daylight_name: [u16; 32], + daylight_date: SystemTime, + daylight_bias: i32, +} + +#[repr(C)] +#[derive(FromBytes, IntoBytes)] +struct SystemTime { + year: u16, + month: u16, + day_of_week: u16, + day: u16, + hour: u16, + minute: u16, + second: u16, + milliseconds: u16, +} + +const API_SET_MAPPINGS: &[(&str, &str)] = &[ + ("api-ms-win-core-apiquery-l1-1-0", "ntdll.dll"), + ("api-ms-win-core-apiquery-l1-1-2", "ntdll.dll"), + ("api-ms-win-core-apiquery-l2-1-1", "kernelbase.dll"), + ("api-ms-win-core-appcompat-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-appcompat-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-appinit-l1-1-0", "kernel32.dll"), + ("api-ms-win-core-atoms-l1-1-0", "kernel32.dll"), + ("api-ms-win-core-backgroundtask-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-calendar-l1-1-0", "kernel32.dll"), + ("api-ms-win-core-comm-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-comm-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-commandlinetoargv-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-console-ansi-l2-1-0", "kernel32.dll"), + ("api-ms-win-core-console-internal-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-console-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-console-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-console-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-console-l1-2-2", "kernelbase.dll"), + ("api-ms-win-core-console-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-console-l2-2-0", "kernelbase.dll"), + ("api-ms-win-core-console-l3-1-0", "kernelbase.dll"), + ("api-ms-win-core-console-l3-2-0", "kernelbase.dll"), + ("api-ms-win-core-crt-l1-1-0", "ntdll.dll"), + ("api-ms-win-core-crt-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-datetime-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-datetime-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-datetime-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-debug-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-debug-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-debug-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-delayload-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-delayload-l1-1-1", "kernelbase.dll"), + ("api-ms-win-downlevel-shlwapi-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-errorhandling-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-errorhandling-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-errorhandling-l1-1-3", "kernelbase.dll"), + ("api-ms-win-core-fibers-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-fibers-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-fibers-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-fibers-l2-1-1", "kernelbase.dll"), + ("api-ms-win-core-file-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-file-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-file-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-file-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-file-l1-2-2", "kernelbase.dll"), + ("api-ms-win-core-file-l1-2-3", "kernelbase.dll"), + ("api-ms-win-core-file-l1-2-5", "kernelbase.dll"), + ("api-ms-win-core-file-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-file-l2-1-1", "kernelbase.dll"), + ("api-ms-win-core-file-l2-1-2", "kernelbase.dll"), + ("api-ms-win-core-file-l2-1-3", "kernelbase.dll"), + ("api-ms-win-core-file-l2-1-4", "kernelbase.dll"), + ("api-ms-win-core-handle-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-heap-obsolete-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-heap-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-heap-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-heap-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-interlocked-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-io-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-io-l1-1-1", "kernel32.dll"), + ("api-ms-win-core-job-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-largeinteger-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l1-2-2", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l1-2-3", "kernelbase.dll"), + ("api-ms-win-core-libraryloader-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-localization-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-localization-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-localization-l1-2-4", "kernelbase.dll"), + ("api-ms-win-core-localization-l2-1-0", "kernelbase.dll"), + ( + "api-ms-win-core-localization-private-l1-1-0", + "kernelbase.dll", + ), + ("api-ms-win-core-localregistry-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-memory-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-memory-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-memory-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-memory-l1-1-9", "kernelbase.dll"), + ("api-ms-win-core-misc-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-namedpipe-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-namedpipe-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-namedpipe-l1-2-2", "kernelbase.dll"), + ("api-ms-win-core-namespace-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-normalization-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-path-l1-1-0", "kernelbase.dll"), + ( + "api-ms-win-core-processenvironment-l1-1-0", + "kernelbase.dll", + ), + ( + "api-ms-win-core-processenvironment-l1-1-1", + "kernelbase.dll", + ), + ( + "api-ms-win-core-processenvironment-l1-2-0", + "kernelbase.dll", + ), + ("api-ms-win-core-processsnapshot-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-processthreads-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-processthreads-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-processthreads-l1-1-2", "kernelbase.dll"), + ("api-ms-win-core-processthreads-l1-1-3", "kernelbase.dll"), + ("api-ms-win-core-processthreads-l1-1-8", "kernel32.dll"), + ("api-ms-win-core-processtopology-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-profile-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-pcw-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-psapi-ansi-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-psapi-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-realtime-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-registry-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-rtlsupport-l1-1-0", "ntdll.dll"), + ("api-ms-win-core-rtlsupport-l1-1-1", "ntdll.dll"), + ("api-ms-win-core-rtlsupport-l1-2-2", "ntdll.dll"), + ("api-ms-win-core-sidebyside-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-string-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-string-l2-1-1", "kernelbase.dll"), + ("api-ms-win-core-synch-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-synch-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-synch-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-synch-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-2-0", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-2-1", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-2-3", "kernelbase.dll"), + ("api-ms-win-core-sysinfo-l1-2-8", "kernelbase.dll"), + ("api-ms-win-core-systemtopology-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-systemtopology-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-threadpool-legacy-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-threadpool-l1-2-0", "kernelbase.dll"), + ( + "api-ms-win-core-threadpool-private-l1-1-0", + "kernelbase.dll", + ), + ("api-ms-win-core-timezone-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-util-l1-1-0", "kernelbase.dll"), + ( + "api-ms-win-core-windowserrorreporting-l1-1-0", + "kernelbase.dll", + ), + ( + "api-ms-win-core-windowserrorreporting-l1-1-1", + "kernelbase.dll", + ), + ( + "api-ms-win-core-windowserrorreporting-l1-1-2", + "kernelbase.dll", + ), + ( + "api-ms-win-core-windowserrorreporting-l1-1-3", + "kernelbase.dll", + ), + ("api-ms-win-core-wow64-l1-1-0", "kernelbase.dll"), + ("api-ms-win-core-wow64-l1-1-1", "kernelbase.dll"), + ("api-ms-win-core-wow64-l1-1-3", "kernelbase.dll"), + ("api-ms-win-core-xstate-l2-1-0", "kernelbase.dll"), + ("api-ms-win-core-xstate-l2-1-1", "kernelbase.dll"), + ("api-ms-win-core-xstate-l2-1-2", "kernelbase.dll"), + ("api-ms-win-eventing-consumer-l1-1-0", "sechost.dll"), + ("api-ms-win-eventing-consumer-l1-1-1", "sechost.dll"), + ("api-ms-win-eventing-controller-l1-1-0", "sechost.dll"), + ("api-ms-win-eventing-provider-l1-1-0", "kernelbase.dll"), + ("api-ms-win-security-audit-l1-1-0", "sechost.dll"), + ("api-ms-win-security-audit-l1-1-1", "sechost.dll"), + ("api-ms-win-security-appcontainer-l1-1-0", "kernelbase.dll"), + ("api-ms-win-security-base-l1-1-0", "kernelbase.dll"), + ("api-ms-win-security-base-l1-2-0", "kernelbase.dll"), + ("api-ms-win-security-base-private-l1-1-0", "kernelbase.dll"), + ("api-ms-win-security-lsalookup-l1-1-0", "sechost.dll"), + ("api-ms-win-security-sddl-l1-1-0", "sechost.dll"), + ("api-ms-win-service-core-l1-1-0", "sechost.dll"), + ("api-ms-win-service-core-l1-1-1", "sechost.dll"), + ("api-ms-win-service-core-l1-1-2", "sechost.dll"), + ("api-ms-win-service-management-l1-1-0", "sechost.dll"), + ("api-ms-win-service-management-l2-1-0", "sechost.dll"), + ("api-ms-win-service-private-l1-1-0", "sechost.dll"), + ("api-ms-win-service-private-l1-1-2", "sechost.dll"), + ("api-ms-win-service-private-l1-1-3", "sechost.dll"), + ("api-ms-win-service-winsvc-l1-1-0", "sechost.dll"), + ("ext-ms-win-appcompat-apphelp-l1-1-2", "apphelp.dll"), + ("ext-ms-win-authz-context-l1-1-0", "authz.dll"), + ("ext-ms-win-core-winrt-remote-l1-1-0", ""), + ("ext-ms-win-oobe-query-l1-1-0", ""), + ( + "ext-ms-win-packagevirtualizationcontext-l1-1-0", + "daxexec.dll", + ), + ("ext-ms-win-rpc-ssl-l1-1-0", "rpcrtremote.dll"), +]; + +fn checked_add(left: usize, right: usize) -> Result { + left.checked_add(right) + .ok_or(PeImageAccessError::AddressOverflow) +} + +fn checked_mul(left: usize, right: usize) -> Result { + left.checked_mul(right) + .ok_or(PeImageAccessError::AddressOverflow) +} + +fn to_u32(value: usize) -> Result { + u32::try_from(value).map_err(|_| PeImageAccessError::AddressOverflow) +} + +fn write_guest_value(address: usize, value: T) -> Result<(), PeImageAccessError> +where + Platform: RawPointerProvider, + T: FromBytes + IntoBytes, +{ + crate::write_value::(address, value).ok_or(PeImageAccessError::MemoryAccess) +} + +fn write_guest_field_at_offset( + base: MutPtr, + field_offset: usize, + value: Field, +) -> Result<(), PeImageAccessError> +where + Platform: RawPointerProvider, + Struct: FromBytes + IntoBytes, + Field: FromBytes + IntoBytes, +{ + crate::write_field_at_offset::(base.as_usize(), field_offset, value) + .ok_or(PeImageAccessError::MemoryAccess) +} + +fn write_guest_slice(address: usize, values: &[T]) -> Result<(), PeImageAccessError> +where + Platform: RawPointerProvider, + T: Copy + FromBytes + IntoBytes, +{ + crate::write_slice::(address, values).ok_or(PeImageAccessError::MemoryAccess) +} + +struct LoadedNtDll { + image: LoadedImage, + exports: NtDllExports, +} + +#[derive(Clone, Copy, Debug)] +struct NtDllExports { + /// `LdrInitializeThunk` + ldr_initialize_thunk: usize, + /// `RtlUserThreadStart` + rtl_user_thread_start: usize, + /// `KiUserInvertedFunctionTable` + ki_user_inverted_function_table: usize, +} + +fn load_ntdll( + platform: &'static Platform, + fs: Arc, + page_manager: &crate::WindowsPageManager, +) -> Result, WindowsLoadError> { + match load_image_with_writable_sections( + fs, + NTDLL_PATH, + platform, + page_manager, + NTDLL_WRITABLE_SECTIONS, + ) { + Ok(image) => { + let exports = ntdll_exports::(&image)?; + litebox_util_log::debug!(path:% = NTDLL_PATH; "Loaded guest ntdll.dll"); + Ok(Some(LoadedNtDll { image, exports })) + } + Err(error) if is_missing_file_error(&error) => { + litebox_util_log::debug!("Guest ntdll.dll was not found in the initial filesystem"); + Ok(None) + } + Err(error) => Err(error), + } +} + +fn load_image( + platform: &'static Platform, + fs: Arc, + path: &str, + page_manager: &crate::WindowsPageManager, +) -> Result { + load_image_with_writable_sections(fs, path, platform, page_manager, &[]) +} + +pub(crate) fn load_image_section( + platform: &'static Platform, + fs: Arc, + path: &str, + page_manager: &crate::WindowsPageManager, + virtual_allocations: &crate::WindowsVirtualAllocations, +) -> Result { + let image = load_image(platform, fs, path, page_manager)?; + let mapping = image.mapping; + register_image_virtual_allocation(virtual_allocations, mapping, image.pages); + Ok(mapping) +} + +pub(crate) struct ImageSectionMetadata { + pub(crate) transfer_address: usize, + pub(crate) file_size: u32, + pub(crate) subsystem: u32, + pub(crate) subsystem_major_version: u16, + pub(crate) subsystem_minor_version: u16, + pub(crate) image_characteristics: u16, + pub(crate) dll_characteristics: u16, + pub(crate) machine: u16, +} + +pub(crate) fn image_section_metadata( + fs: Arc, + path: &str, +) -> Result { + let file = PeImageFile::open(fs, path)?; + let parsed = PeParsedFile::parse(&mut &file).map_err(WindowsLoadError::Parse)?; + let file_size = file + .fs + .fd_file_status(&file.fd) + .map_err(PeImageAccessError::FileStatus)? + .size + .try_into() + .map_err(|_| PeImageAccessError::AddressOverflow)?; + Ok(ImageSectionMetadata { + transfer_address: parsed + .image_base() + .checked_add(parsed.entry_point_rva()) + .ok_or(PeImageAccessError::AddressOverflow)?, + file_size, + subsystem: u32::from(parsed.subsystem()), + subsystem_major_version: parsed.major_subsystem_version(), + subsystem_minor_version: parsed.minor_subsystem_version(), + image_characteristics: parsed.characteristics(), + dll_characteristics: parsed.dll_characteristics(), + machine: parsed.machine(), + }) +} + +fn load_image_with_writable_sections( + fs: Arc, + path: &str, + platform: &'static Platform, + page_manager: &crate::WindowsPageManager, + writable_section_names: &[&[u8]], +) -> Result { + let file = PeImageFile::open(fs, path)?; + let mut parsed = PeParsedFile::parse(&mut &file).map_err(WindowsLoadError::Parse)?; + parsed + .parse_trampoline(&mut &file, platform.get_syscall_entry_point()) + .map_err(WindowsLoadError::Parse)?; + let mut mapper = PeImageMapper { + file: &file, + page_manager, + chunk: alloc::vec![0u8; FILE_CHUNK_BYTES], + pages: RangeMap::new(), + }; + let mut memory = PeImageMemory::(PhantomData); + let mapping = parsed + .load_with_writable_sections(&mut mapper, &mut memory, writable_section_names) + .map_err(WindowsLoadError::Load)?; + Ok(LoadedImage { + mapping, + pages: mapper.pages, + parsed, + }) +} + +fn ntdll_exports( + image: &LoadedImage, +) -> Result { + let export_names = [ + "LdrInitializeThunk", + "RtlUserThreadStart", + "KiUserInvertedFunctionTable", + ]; + let mut memory = PeImageMemory::(PhantomData); + let addresses = image + .parsed + .find_export_addresses(image.mapping.base_addr, &mut memory, &export_names) + .map_err(WindowsLoadError::Export)?; + let [ + ldr_initialize_thunk, + rtl_user_thread_start, + ki_user_inverted_function_table, + ]: [Option; 3] = addresses + .try_into() + .map_err(|_| WindowsLoadError::MissingNtDllInvertedFunctionTable)?; + + let ldr_initialize_thunk = + ldr_initialize_thunk.ok_or(WindowsLoadError::MissingNtDllLoaderEntrypoint)?; + let rtl_user_thread_start = + rtl_user_thread_start.ok_or(WindowsLoadError::MissingNtDllThreadEntrypoint)?; + let ki_user_inverted_function_table = ki_user_inverted_function_table + .ok_or(WindowsLoadError::MissingNtDllInvertedFunctionTable)?; + + Ok(NtDllExports { + ldr_initialize_thunk, + rtl_user_thread_start, + ki_user_inverted_function_table, + }) +} + +/// Errors that can occur while opening, parsing, and mapping a Windows PE image. +#[derive(Debug, Error)] +pub enum WindowsLoadError { + #[error("failed to parse PE image")] + Parse(#[source] PeParseError), + #[error("failed to load PE image")] + Load(#[source] PeLoadError), + #[error("failed to parse PE export table")] + Export(#[source] PeExportError), + /// Accessing the PE backing file or its mapped memory failed. + #[error(transparent)] + Access(#[from] PeImageAccessError), + /// Guest ntdll.dll does not export LdrInitializeThunk. + #[error("guest ntdll.dll does not export LdrInitializeThunk")] + MissingNtDllLoaderEntrypoint, + /// Guest ntdll.dll does not export RtlUserThreadStart. + #[error("guest ntdll.dll does not export RtlUserThreadStart")] + MissingNtDllThreadEntrypoint, + /// Guest ntdll.dll does not export KiUserInvertedFunctionTable. + #[error("guest ntdll.dll does not export KiUserInvertedFunctionTable")] + MissingNtDllInvertedFunctionTable, + /// Guest ntdll.dll has an invalid exception directory. + #[error("guest ntdll.dll has an invalid exception directory")] + InvalidNtDllExceptionDirectory, + /// Guest ntdll.dll has not been rewritten for LiteBox syscall/GS handling. + #[error("guest ntdll.dll must be rewritten for LiteBox before entering its loader")] + UnrewrittenNtDll, + #[error("failed to map shared memory")] + MapSharedMemory, + #[error("memory access failed")] + MemoryAccess, +} + +fn is_missing_file_error(error: &WindowsLoadError) -> bool { + let WindowsLoadError::Access(PeImageAccessError::Open(error)) = error else { + return false; + }; + + matches!( + error, + litebox::fs::errors::OpenError::PathError( + litebox::fs::errors::PathError::NoSuchFileOrDirectory + | litebox::fs::errors::PathError::MissingComponent + ) + ) +} + +struct PeImageFile { + fs: Arc, + fd: litebox::fd::TypedFd, +} + +impl PeImageFile { + fn open(fs: Arc, path: &str) -> Result { + let fd = fs.open(path, OFlags::RDONLY, Mode::empty())?; + Ok(Self { fs, fd }) + } + + fn read_exact_at( + &self, + mut offset: usize, + mut buf: &mut [u8], + ) -> Result<(), PeImageAccessError> { + while !buf.is_empty() { + let bytes_read = self.fs.read(&self.fd, buf, Some(offset))?; + if bytes_read == 0 { + return Err(PeImageAccessError::ShortRead); + } + offset = offset + .checked_add(bytes_read) + .ok_or(PeImageAccessError::AddressOverflow)?; + buf = &mut buf[bytes_read..]; + } + Ok(()) + } +} + +impl Drop for PeImageFile { + fn drop(&mut self) { + if let Err(e) = self.fs.close(&self.fd) { + litebox_util_log::warn!(error:? = e; "failed to close PE image file"); + } + } +} + +impl ReadAt for &'_ PeImageFile { + type Error = PeImageAccessError; + + fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), Self::Error> { + self.read_exact_at( + offset + .try_into() + .map_err(|_| PeImageAccessError::AddressOverflow)?, + buf, + ) + } + + fn size(&mut self) -> Result { + self.fs + .fd_file_status(&self.fd)? + .size + .try_into() + .map_err(|_| PeImageAccessError::AddressOverflow) + } +} + +struct PeImageMapper<'a, Platform: crate::ShimPlatform, FS: ShimFS> { + file: &'a PeImageFile, + page_manager: &'a crate::WindowsPageManager, + /// Reusable per-call I/O staging buffer for [`MapMemory::map_file`]. + chunk: Vec, + pages: RangeMap, +} + +impl PeImageMapper<'_, Platform, FS> { + fn record_pages( + &mut self, + address: usize, + len: usize, + protect: PageProtection, + ) -> Result<(), PeImageAccessError> { + let (start, len) = page_range(address, len)?; + if len == 0 { + return Ok(()); + } + let end = start + .checked_add(len) + .ok_or(PeImageAccessError::AddressOverflow)?; + self.pages.insert(start..end, protect); + Ok(()) + } + + fn protect_and_record_pages( + &mut self, + address: usize, + len: usize, + prot: Protection, + ) -> Result<(), PeImageAccessError> { + protect_pages(self.page_manager, address, len, prot)?; + self.record_pages(address, len, page_protection_from_loader_protection(prot)) + } +} + +impl MapMemory for PeImageMapper<'_, Platform, FS> { + type Error = PeImageAccessError; + + fn reserve( + &mut self, + preferred_base: usize, + len: usize, + _align: usize, + ) -> Result { + let length = NonZeroPageSize::new(len).ok_or(PeImageAccessError::AddressOverflow)?; + let suggested_address = if preferred_base == 0 { + None + } else { + Some(NonZeroAddress::new(preferred_base).ok_or(PeImageAccessError::AddressOverflow)?) + }; + + // SAFETY: `CreatePagesFlags::empty()` does not set `fixed_addr`, so the kernel + // treats `suggested_address` as a hint and never silently unmaps an existing + // mapping; the documented overlap precondition therefore does not apply. + let ptr = unsafe { + self.page_manager.create_inaccessible_pages( + suggested_address, + length, + CreatePagesFlags::empty(), + |_| Ok(0), + )? + }; + let base = ptr.as_usize(); + self.record_pages(base, len, PageProtection::PAGE_NOACCESS)?; + Ok(base) + } + + fn map_zero( + &mut self, + address: usize, + len: usize, + prot: &Protection, + ) -> Result<(), Self::Error> { + make_pages_writable(self.page_manager, address, len)?; + let ptr = ::RawMutPointer::::from_usize(address); + let mut written = 0; + while written < len { + let chunk = (len - written).min(ZERO_CHUNK.len()); + ptr.copy_from_slice(written, &ZERO_CHUNK[..chunk]) + .ok_or(PeImageAccessError::MemoryAccess)?; + written += chunk; + } + self.protect_and_record_pages(address, len, *prot) + } + + fn map_file( + &mut self, + address: usize, + len: usize, + offset: u64, + prot: &Protection, + ) -> Result<(), Self::Error> { + make_pages_writable(self.page_manager, address, len)?; + let ptr = ::RawMutPointer::::from_usize(address); + let file_offset: usize = offset + .try_into() + .map_err(|_| PeImageAccessError::AddressOverflow)?; + let mut read = 0; + while read < len { + let remaining = len - read; + let n = remaining.min(self.chunk.len()); + self.file.read_exact_at( + file_offset + .checked_add(read) + .ok_or(PeImageAccessError::AddressOverflow)?, + &mut self.chunk[..n], + )?; + ptr.copy_from_slice(read, &self.chunk[..n]) + .ok_or(PeImageAccessError::MemoryAccess)?; + read += n; + } + self.protect_and_record_pages(address, len, *prot) + } + + fn protect( + &mut self, + address: usize, + len: usize, + prot: &Protection, + ) -> Result<(), Self::Error> { + self.protect_and_record_pages(address, len, *prot) + } +} + +fn page_protection_from_loader_protection(protect: Protection) -> PageProtection { + match (protect.read, protect.write, protect.execute) { + (_, true, true) => PageProtection::PAGE_EXECUTE_READWRITE, + (_, true, false) => PageProtection::PAGE_READWRITE, + (true, false, true) => PageProtection::PAGE_EXECUTE_READ, + (false, false, true) => PageProtection::PAGE_EXECUTE, + (true, false, false) => PageProtection::PAGE_READONLY, + (false, false, false) => PageProtection::PAGE_NOACCESS, + } +} + +/// Errors from the shim-side PE image backing file and memory mapper. +#[derive(Debug, Error)] +pub enum PeImageAccessError { + #[error("failed to open PE image")] + Open(#[from] litebox::fs::errors::OpenError), + #[error("failed to read PE image")] + Read(#[from] litebox::fs::errors::ReadError), + #[error("failed to read PE image metadata")] + FileStatus(#[from] litebox::fs::errors::FileStatusError), + /// The backing file ended before the requested range was filled. + #[error("short read from PE image")] + ShortRead, + /// A PE file offset or image address overflowed the host's `usize`. + #[error("PE image address overflow")] + AddressOverflow, + #[error(transparent)] + Mapping(#[from] MappingError), + #[error(transparent)] + Protect(#[from] VmemProtectError), + #[error("mapped PE image memory access failed")] + MemoryAccess, +} + +struct PeImageMemory(PhantomData); + +impl AccessMemory for PeImageMemory { + fn read(&mut self, address: usize, buf: &mut [u8]) -> Result<(), Fault> { + let ptr = ::RawConstPointer::::from_usize(address); + buf.copy_from_slice(&ptr.to_owned_slice(buf.len()).ok_or(Fault)?); + Ok(()) + } + + fn write(&mut self, address: usize, data: &[u8]) -> Result<(), Fault> { + let ptr = ::RawMutPointer::::from_usize(address); + ptr.copy_from_slice(0, data).ok_or(Fault) + } +} + +fn make_pages_writable( + page_manager: &crate::WindowsPageManager, + address: usize, + len: usize, +) -> Result<(), PeImageAccessError> { + let (start, len) = page_range(address, len)?; + if len == 0 { + return Ok(()); + } + let ptr = ::RawMutPointer::::from_usize(start); + // SAFETY: Loading happens before the initial guest thread is allowed to execute. + unsafe { page_manager.make_pages_writable(ptr, len)? }; + Ok(()) +} + +fn protect_pages( + page_manager: &crate::WindowsPageManager, + address: usize, + len: usize, + prot: Protection, +) -> Result<(), PeImageAccessError> { + let (start, len) = page_range(address, len)?; + if len == 0 { + return Ok(()); + } + let ptr = ::RawMutPointer::::from_usize(start); + // SAFETY: All `make_pages_*` calls happen during PE load, before the initial + // guest thread starts, so there is no concurrent read/write/execute on these + // pages. The RWX arm is only reached when a section's COFF characteristics + // demand WRITE|EXECUTE; the bytes copied into the section come from the + // attacker-controlled PE file and are not executed until protections are set, + // so this is no looser than running the same PE under the real Windows loader. + match (prot.read, prot.write, prot.execute) { + (_, true, true) => unsafe { page_manager.make_pages_rwx(ptr, len)? }, + (_, true, false) => unsafe { page_manager.make_pages_writable(ptr, len)? }, + (_, false, true) => unsafe { page_manager.make_pages_executable(ptr, len)? }, + (true, false, false) => unsafe { page_manager.make_pages_readable(ptr, len)? }, + (false, false, false) => unsafe { page_manager.make_pages_inaccessible(ptr, len)? }, + } + Ok(()) +} + +fn page_range(address: usize, len: usize) -> Result<(usize, usize), PeImageAccessError> { + if len == 0 { + return Ok((address, 0)); + } + let start = page_align_down(address); + let end = address + .checked_add(len) + .and_then(|v| v.checked_next_multiple_of(PAGE_SIZE)) + .ok_or(PeImageAccessError::AddressOverflow)?; + Ok((start, end - start)) +} + +fn win32_image_path(path: &str) -> String { + let mut win32_path = String::from("C:"); + if !path.starts_with('/') && !path.starts_with('\\') { + win32_path.push('\\'); + } + for ch in path.chars() { + win32_path.push(if ch == '/' { '\\' } else { ch }); + } + win32_path +} + +fn dos_image_path(path: &str) -> String { + let mut dos_path = String::from(r"\??\"); + dos_path.push_str(&win32_image_path(path)); + dos_path +} + +fn windows_command_line(image_path: &str, argv: &[CString]) -> String { + let mut command_line = String::new(); + if let Some(arg0) = argv.first() { + push_windows_quoted_arg(&mut command_line, &cstring_to_string(arg0)); + } else { + push_windows_quoted_arg(&mut command_line, image_path); + } + for arg in argv.iter().skip(1) { + command_line.push(' '); + push_windows_quoted_arg(&mut command_line, &cstring_to_string(arg)); + } + command_line +} + +fn push_windows_quoted_arg(command_line: &mut String, arg: &str) { + if !arg.is_empty() && !arg.contains([' ', '\t', '"']) { + command_line.push_str(arg); + return; + } + + command_line.push('"'); + let mut backslashes = 0; + for ch in arg.chars() { + if ch == '\\' { + backslashes += 1; + } else if ch == '"' { + for _ in 0..=backslashes * 2 { + command_line.push('\\'); + } + command_line.push('"'); + backslashes = 0; + } else { + for _ in 0..backslashes { + command_line.push('\\'); + } + command_line.push(ch); + backslashes = 0; + } + } + for _ in 0..backslashes * 2 { + command_line.push('\\'); + } + command_line.push('"'); +} + +fn windows_environment_block(envp: &[CString]) -> Vec { + let mut variables = envp.iter().map(cstring_to_string).collect::>(); + variables.sort_by(|left, right| { + left.bytes() + .map(|byte| byte.to_ascii_uppercase()) + .cmp(right.bytes().map(|byte| byte.to_ascii_uppercase())) + }); + + let mut block = Vec::new(); + for variable in variables { + block.extend(variable.encode_utf16()); + block.push(0); + } + block.push(0); + if envp.is_empty() { + block.push(0); + } + block +} + +fn cstring_to_string(value: &CString) -> String { + match core::str::from_utf8(value.as_bytes()) { + Ok(value) => String::from(value), + Err(_) => String::from_utf8_lossy(value.as_bytes()).into_owned(), + } +} + +struct InitialProcessHeaps { + address: usize, + maximum_number_of_heaps: u32, +} + +fn initial_process_heaps_array(peb_ptr: usize) -> Result { + let peb_size = core::mem::size_of::(); + let address = peb_ptr + .checked_add(peb_size) + .ok_or(PeImageAccessError::AddressOverflow)?; + let maximum_number_of_heaps = + (peb_size.next_multiple_of(PAGE_SIZE) - peb_size) / core::mem::size_of::(); + Ok(InitialProcessHeaps { + address, + maximum_number_of_heaps: maximum_number_of_heaps.trunc(), + }) +} + +fn allocate_guest_unicode_string_from_str( + shared_heap: &mut GuestMemoryAllocator, + value: &str, +) -> Result { + let string = Utf16StringBuffer::new(value)?; + allocate_guest_unicode_string::(shared_heap, &string) +} + +fn allocate_guest_unicode_string( + allocation: &mut GuestMemoryAllocator, + string: &Utf16StringBuffer, +) -> Result { + let buffer = allocation.allocate_array::(string.units.len())?; + buffer + .write_slice_at_offset(0, &string.units) + .ok_or(PeImageAccessError::MemoryAccess)?; + Ok(UnicodeString { + length: string.length, + maximum_length: string.maximum_length, + padding_0: [0; 4], + buffer: buffer.as_usize(), + }) +} + +fn initial_teb_static_unicode_string( + teb_ptr: usize, + static_unicode_buffer: &[u16], +) -> Result { + let buffer = teb_ptr + .checked_add(core::mem::offset_of!( + ThreadEnvironmentBlock, + static_unicode_buffer + )) + .ok_or(PeImageAccessError::AddressOverflow)?; + Ok(UnicodeString { + length: 0, + maximum_length: u16::try_from(core::mem::size_of_val(static_unicode_buffer)) + .map_err(|_| PeImageAccessError::AddressOverflow)?, + padding_0: [0; 4], + buffer, + }) +} + +struct Utf16StringBuffer { + length: u16, + maximum_length: u16, + units: Vec, +} + +impl Utf16StringBuffer { + fn new(value: &str) -> Result { + let mut units: Vec = value.encode_utf16().collect(); + let length = utf16_byte_len(units.len())?; + units.push(0); + let maximum_length = utf16_byte_len(units.len())?; + Ok(Self { + length, + maximum_length, + units, + }) + } +} + +fn utf16_byte_len(units: usize) -> Result { + units + .checked_mul(core::mem::size_of::()) + .and_then(|bytes| u16::try_from(bytes).ok()) + .ok_or(PeImageAccessError::AddressOverflow) +} + +#[cfg(all(test, target_os = "windows", target_arch = "x86_64"))] +mod tests { + extern crate std; + + use alloc::{string::String, vec, vec::Vec}; + use litebox::platform::RawPointerProvider; + use litebox_common_windows::loader::{ + ApiSetHashEntry, ApiSetNamespace, ApiSetNamespaceEntry, ApiSetValueEntry, + MAX_API_SET_NAMESPACE_SIZE, api_set_hash_prefix, + }; + + use super::*; + use crate::nt_types::{ProcessEnvironmentBlock, ThreadEnvironmentBlock, UnicodeString}; + + const TEST_STACK_BASE: usize = 0x7000_0000; + const TEST_STACK_TOP: usize = TEST_STACK_BASE + 0x100000; + + #[allow(non_snake_case)] + #[link(name = "kernel32")] + unsafe extern "system" { + fn GetCurrentProcessId() -> u32; + fn GetCurrentThreadId() -> u32; + fn GetModuleHandleW(lp_module_name: *const u16) -> *mut core::ffi::c_void; + fn GetProcAddress( + h_module: *mut core::ffi::c_void, + lp_proc_name: *const core::ffi::c_char, + ) -> *mut core::ffi::c_void; + fn GetModuleFileNameW( + h_module: *mut core::ffi::c_void, + lp_filename: *mut u16, + n_size: u32, + ) -> u32; + fn RtlGetCurrentPeb() -> *const ProcessEnvironmentBlock; + } + + macro_rules! print_diff_fields { + ($prefix:literal, $synthetic:expr, $host:expr, [$($field:ident),+ $(,)?]) => { + $( + print_diff_field!($prefix, $synthetic, $host, $field); + )+ + }; + } + + macro_rules! print_diff_field { + ($prefix:literal, $synthetic:expr, $host:expr, csd_version) => { + print_unicode_string_diff( + concat!($prefix, ".", stringify!(csd_version)), + ($synthetic).csd_version, + ($host).csd_version, + ); + }; + + ($prefix:literal, $synthetic:expr, $host:expr, static_unicode_string) => { + print_unicode_string_diff( + concat!($prefix, ".", stringify!(static_unicode_string)), + ($synthetic).static_unicode_string, + ($host).static_unicode_string, + ); + }; + ($prefix:literal, $synthetic:expr, $host:expr, $field:ident) => { + print_field_diff( + concat!($prefix, ".", stringify!($field)), + ($synthetic).$field, + ($host).$field, + ); + }; + } + + fn table_offset(base: u32, index: u32, entry_size: usize) -> Option { + (base as usize).checked_add((index as usize).checked_mul(entry_size)?) + } + + fn read_utf16_string(bytes: &[u8], offset: u32, len: u32) -> Option { + let offset = offset as usize; + let len = len as usize; + let end = offset.checked_add(len)?; + let bytes = bytes.get(offset..end)?; + let mut chunks = bytes.chunks_exact(size_of::()); + if !chunks.remainder().is_empty() { + return None; + } + let units = chunks + .by_ref() + .map(|chunk| u16::from_le_bytes(chunk.try_into().expect("u16 byte chunk"))) + .collect::>(); + Some(String::from_utf16_lossy(&units)) + } + + fn parse_api_set_value_entry(bytes: &[u8], offset: usize) -> Option { + Some( + ApiSetValueEntry::read_from_prefix(bytes.get(offset..)?) + .ok()? + .0, + ) + } + + fn api_set_value_entry_value(entry: ApiSetValueEntry, bytes: &[u8]) -> Option { + read_utf16_string(bytes, entry.value_offset, entry.value_length) + } + + fn api_set_value_entry_name(entry: ApiSetValueEntry, bytes: &[u8]) -> Option { + read_utf16_string(bytes, entry.name_offset, entry.name_length) + } + + fn parse_api_set_hash_entry(bytes: &[u8], offset: usize) -> Option { + Some( + ApiSetHashEntry::read_from_prefix(bytes.get(offset..)?) + .ok()? + .0, + ) + } + + fn parse_api_set_namespace_entry(bytes: &[u8], offset: usize) -> Option { + Some( + ApiSetNamespaceEntry::read_from_prefix(bytes.get(offset..)?) + .ok()? + .0, + ) + } + + fn api_set_namespace_entry_name(entry: ApiSetNamespaceEntry, bytes: &[u8]) -> Option { + read_utf16_string(bytes, entry.name_offset, entry.name_length) + } + + fn api_set_namespace_entry_value( + entry: ApiSetNamespaceEntry, + bytes: &[u8], + index: u32, + ) -> Option { + if index >= entry.value_count { + return None; + } + parse_api_set_value_entry( + bytes, + table_offset(entry.value_offset, index, size_of::())?, + ) + } + + fn parse_api_set_namespace(bytes: &[u8]) -> Option { + let namespace = ApiSetNamespace::read_from_prefix(bytes).ok()?.0; + let size = namespace.size as usize; + if size != bytes.len() + || !(size_of::()..=MAX_API_SET_NAMESPACE_SIZE).contains(&size) + { + return None; + } + if table_offset( + namespace.entry_offset, + namespace.count, + size_of::(), + )? > size + { + return None; + } + if table_offset( + namespace.hash_offset, + namespace.count, + size_of::(), + )? > size + { + return None; + } + Some(namespace) + } + + fn api_set_namespace_entry( + namespace: ApiSetNamespace, + bytes: &[u8], + index: u32, + ) -> Option { + if index >= namespace.count { + return None; + } + parse_api_set_namespace_entry( + bytes, + table_offset( + namespace.entry_offset, + index, + size_of::(), + )?, + ) + } + + fn api_set_namespace_hash_entry( + namespace: ApiSetNamespace, + bytes: &[u8], + index: u32, + ) -> Option { + if index >= namespace.count { + return None; + } + parse_api_set_hash_entry( + bytes, + table_offset(namespace.hash_offset, index, size_of::())?, + ) + } + + fn host_api_set_namespace_bytes() -> Vec { + let peb = unsafe { + // SAFETY: `RtlGetCurrentPeb` returns the current process PEB pointer on Windows. + RtlGetCurrentPeb().as_ref() + } + .expect("host PEB"); + let namespace_ptr = peb.api_set_map as *const ApiSetNamespace; + let namespace = unsafe { + // SAFETY: `ApiSetMap` points at the host process API_SET_NAMESPACE while the + // process is alive; we read only the fixed header first to learn its size. + namespace_ptr.as_ref() + } + .expect("host API_SET_NAMESPACE header"); + let size = namespace.size as usize; + assert!( + (size_of::()..=MAX_API_SET_NAMESPACE_SIZE).contains(&size), + "host API_SET_NAMESPACE has unexpected size {size:#x}" + ); + let bytes = unsafe { + // SAFETY: The size was read from the validated namespace header above, and the + // host API-set namespace is immutable process-wide data owned by ntdll. + core::slice::from_raw_parts(peb.api_set_map as *const u8, size) + }; + bytes.to_vec() + } + + fn api_set_default_value(bytes: &[u8], contract: &str) -> Option { + let namespace = parse_api_set_namespace(bytes)?; + for index in 0..namespace.count { + let entry = api_set_namespace_entry(namespace, bytes, index)?; + if api_set_namespace_entry_name(entry, bytes)?.eq_ignore_ascii_case(contract) { + let value = api_set_namespace_entry_value(entry, bytes, 0)?; + return api_set_value_entry_value(value, bytes); + } + } + None + } + + fn assert_api_set_hash_table(bytes: &[u8]) { + let namespace = parse_api_set_namespace(bytes).expect("valid API_SET_NAMESPACE"); + let mut previous = None; + for hash_index in 0..namespace.count { + let hash_entry = + api_set_namespace_hash_entry(namespace, bytes, hash_index).expect("hash entry"); + let entry = api_set_namespace_entry(namespace, bytes, hash_entry.index) + .expect("hash entry target"); + let name = api_set_namespace_entry_name(entry, bytes).expect("hash entry target name"); + let expected_hash = api_set_hash_with_hashed_length(&name, entry.hashed_length) + .expect("valid API-set hashed length"); + assert_eq!(hash_entry.hash, expected_hash, "hash for {name}"); + if let Some((previous_hash, previous_index)) = previous { + assert!( + (previous_hash, previous_index) <= (hash_entry.hash, hash_entry.index), + "API-set hash table is not sorted" + ); + } + previous = Some((hash_entry.hash, hash_entry.index)); + } + } + + fn api_set_hash_with_hashed_length(name: &str, hashed_length: u32) -> Option { + let code_unit_bytes = u32::try_from(size_of::()).ok()?; + if !name.is_ascii() || !hashed_length.is_multiple_of(code_unit_bytes) { + return None; + } + let hashed_units = usize::try_from(hashed_length / code_unit_bytes).ok()?; + let prefix = name.get(..hashed_units)?; + Some(api_set_hash_prefix(prefix)) + } + + fn dump_api_set_entries(bytes: &[u8], namespace: ApiSetNamespace) { + std::println!("entries:"); + for index in 0..namespace.count { + let entry = api_set_namespace_entry(namespace, bytes, index).expect("namespace entry"); + std::println!( + " {index:04} name={} flags={:#x} hashed_len={} values={}", + api_set_namespace_entry_name(entry, bytes) + .unwrap_or_else(|| String::from("")), + entry.flags, + entry.hashed_length, + entry.value_count + ); + for value_index in 0..entry.value_count { + let value = api_set_namespace_entry_value(entry, bytes, value_index) + .expect("namespace value"); + let name = api_set_value_entry_name(value, bytes).unwrap_or_default(); + let value_name = api_set_value_entry_value(value, bytes) + .unwrap_or_else(|| String::from("")); + std::println!( + " [{value_index}] name={} value={} flags={:#x}", + if name.is_empty() { "" } else { &name }, + value_name, + value.flags + ); + } + } + } + + fn dump_api_set_hash_entries(bytes: &[u8], namespace: ApiSetNamespace) { + std::println!("hash entries:"); + for index in 0..namespace.count { + let entry = api_set_namespace_hash_entry(namespace, bytes, index).expect("hash entry"); + std::println!( + " {index:04} hash={:#010x} index={}", + entry.hash, + entry.index + ); + } + } + + fn dump_api_set_namespace(api_set_map: ApiSetNamespace, bytes: &[u8], label: &str) { + std::println!("{label}"); + std::println!("API_SET_NAMESPACE len={:#x}", bytes.len()); + std::println!(" version: {:#010x}", api_set_map.version); + std::println!(" size: {:#010x}", api_set_map.size); + std::println!(" flags: {:#010x}", api_set_map.flags); + std::println!(" count: {:#010x}", api_set_map.count); + std::println!("entry_offset: {:#010x}", api_set_map.entry_offset); + std::println!(" hash_offset: {:#010x}", api_set_map.hash_offset); + std::println!(" hash_factor: {:#010x}", api_set_map.hash_factor); + std::println!(); + dump_api_set_entries(bytes, api_set_map); + std::println!(); + dump_api_set_hash_entries(bytes, api_set_map); + std::println!(); + } + + #[test] + fn dump_host_api_set_namespace() { + let host_bytes = host_api_set_namespace_bytes(); + let host = parse_api_set_namespace(&host_bytes).expect("valid host API_SET_NAMESPACE"); + dump_api_set_namespace(host, &host_bytes, "host"); + } + + #[test] + fn api_set_namespace_matches_host_invariants() { + let host_bytes = host_api_set_namespace_bytes(); + let host = parse_api_set_namespace(&host_bytes).expect("valid host API_SET_NAMESPACE"); + let synthetic_bytes = + build_api_set_namespace(API_SET_MAPPINGS).expect("LiteBox API_SET_NAMESPACE builds"); + let synthetic = + parse_api_set_namespace(&synthetic_bytes).expect("valid synthetic API_SET_NAMESPACE"); + + assert_eq!(synthetic.version, host.version); + assert_eq!(synthetic.hash_factor, host.hash_factor); + assert_eq!(synthetic.flags, 0); + assert_api_set_hash_table(&host_bytes); + assert_api_set_hash_table(&synthetic_bytes); + + let mut host_checked = 0; + let mut host_mismatches = Vec::new(); + for &(contract, expected_host) in API_SET_MAPPINGS { + let synthetic_host = api_set_default_value(&synthetic_bytes, contract); + assert_eq!( + synthetic_host.as_deref(), + Some(expected_host), + "synthetic mapping for {contract}" + ); + if let Some(host_value) = api_set_default_value(&host_bytes, contract) { + if !host_value.eq_ignore_ascii_case(expected_host) { + host_mismatches.push(std::format!( + "{contract}: expected {expected_host}, got {host_value}" + )); + } + host_checked += 1; + } + } + assert!( + host_mismatches.is_empty(), + "host API-set mapping mismatches:\n{}", + host_mismatches.join("\n") + ); + assert!( + host_checked >= 3, + "expected at least three synthetic API-set contracts on the host, found {host_checked}" + ); + + for (contract, expected_host) in [ + ("api-ms-win-core-rtlsupport-l1-1-0", "ntdll.dll"), + ("api-ms-win-core-file-l1-2-3", "kernelbase.dll"), + ("api-ms-win-eventing-consumer-l1-1-0", "sechost.dll"), + ] { + assert_eq!( + api_set_default_value(&synthetic_bytes, contract).as_deref(), + Some(expected_host), + "synthetic mapping for {contract}" + ); + } + } + + #[test] + fn prints_created_teb_host_diff() { + let synthetic = created_process_environment_snapshot(); + let current_teb = host_teb_snapshot(); + let teb_self = host_teb_address(); + let peb_address = host_peb_address(); + + assert_eq!(synthetic.teb.nt_tib.self_pointer, synthetic.environment.teb); + assert_eq!(current_teb.nt_tib.self_pointer, teb_self); + assert_eq!( + synthetic.teb.process_environment_block, + synthetic.environment.peb + ); + assert_eq!(current_teb.process_environment_block, peb_address); + assert_eq!(current_teb.client_id, host_client_id()); + + print_diff_header("synthetic TEB vs host TEB"); + print_diff_fields!( + "TEB.NtTib", + synthetic.teb.nt_tib, + current_teb.nt_tib, + [ + exception_list, + stack_base, + stack_limit, + sub_system_tib, + fiber_data_or_version, + arbitrary_user_pointer, + self_pointer, + ] + ); + print_diff_fields!( + "TEB", + synthetic.teb, + current_teb, + [ + environment_pointer, + client_id, + active_rpc_handle, + thread_local_storage_pointer, + process_environment_block, + last_error_value, + count_of_owned_critical_sections, + csr_client_thread, + win_32_thread_info, + user_32_reserved, + user_reserved, + padding_user_reserved, + wow_32_reserved, + current_locale, + fp_software_status_register, + reserved_for_debugger_instrumentation, + system_reserved_1, + heap_fls_data, + rng_state, + placeholder_compatibility_mode, + placeholder_hydration_always_explicit, + placeholder_reserved, + proxied_process_id, + activation_stack, + working_on_behalf_ticket, + exception_code, + padding_0, + activation_context_stack_pointer, + instrumentation_callback_sp, + instrumentation_callback_previous_pc, + instrumentation_callback_previous_sp, + tx_fs_context, + instrumentation_callback_disabled, + unaligned_load_store_exceptions, + padding_1, + gdi_teb_batch, + real_client_id, + gdi_cached_process_handle, + gdi_client_pid, + gdi_client_tid, + gdi_thread_local_info, + win_32_client_info, + gl_dispatch_table, + gl_reserved_1, + gl_reserved_2, + gl_section_info, + gl_section, + gl_table, + gl_current_rc, + gl_context, + last_status_value, + padding_2, + static_unicode_string, + static_unicode_buffer, + padding_3, + deallocation_stack, + tls_slots, + tls_links, + vdm, + reserved_for_nt_rpc, + dbg_ss_reserved, + hard_error_mode, + padding_4, + instrumentation, + activity_id, + sub_process_tag, + perflib_data, + etw_trace_data, + win_sock_data, + gdi_batch_count, + ideal_processor_value, + guaranteed_stack_bytes, + padding_5, + reserved_for_perf, + reserved_for_ole, + waiting_on_loader_lock, + padding_6, + saved_priority_state, + reserved_for_code_coverage, + thread_pool_data, + tls_expansion_slots, + chpe_v_2_cpu_area_info, + unused, + mui_generation, + is_impersonating, + nls_cache, + p_shim_data, + heap_data, + padding_7, + current_transaction_handle, + active_frame, + fls_data, + preferred_languages, + user_pref_languages, + merged_pref_languages, + mui_impersonation, + cross_teb_flags, + same_teb_flags, + txn_scope_enter_callback, + txn_scope_exit_callback, + txn_scope_context, + lock_count, + wow_teb_offset, + resource_ret_value, + reserved_for_wdf, + reserved_for_crt, + effective_container_id, + last_sleep_counter, + spin_call_count, + padding_8, + extended_feature_disable_mask, + scheduler_shared_data_slot, + heap_walk_context, + primary_group_affinity, + rcu, + ] + ); + } + + #[test] + fn prints_created_peb_host_diff() { + let created = created_process_environment_snapshot(); + let host_peb = host_peb_snapshot(); + + assert_eq!(created.peb.image_base_address, created.image_base_address); + assert_ne!(host_peb.image_base_address, 0); + + print_diff_header("synthetic PEB vs host PEB"); + print_diff_fields!( + "PEB", + created.peb, + host_peb, + [ + inherited_address_space, + read_image_file_exec_options, + being_debugged, + ] + ); + print_peb_bit_field_diff( + "PEB.bit_field", + crate::nt_types::PebBitField::from_bits_retain(created.peb.bit_field), + crate::nt_types::PebBitField::from_bits_retain(host_peb.bit_field), + ); + print_diff_fields!( + "PEB", + created.peb, + host_peb, + [ + padding_0, + mutant, + image_base_address, + ldr, + process_parameters, + sub_system_data, + process_heap, + fast_peb_lock, + atl_thunk_s_list_ptr, + ifeo_key, + cross_process_flags, + padding_1, + kernel_callback_table, + system_reserved, + atl_thunk_s_list_ptr_32, + api_set_map, + tls_expansion_counter, + padding_2, + tls_bitmap, + tls_bitmap_bits, + read_only_shared_memory_base, + shared_data, + read_only_static_server_data, + ansi_code_page_data, + oem_code_page_data, + unicode_case_table_data, + number_of_processors, + nt_global_flag, + critical_section_timeout, + heap_segment_reserve, + heap_segment_commit, + heap_de_commit_total_free_threshold, + heap_de_commit_free_block_threshold, + number_of_heaps, + maximum_number_of_heaps, + process_heaps, + gdi_shared_handle_table, + process_starter_helper, + gdi_dc_attribute_list, + padding_3, + loader_lock, + os_major_version, + os_minor_version, + os_build_number, + os_csd_version, + os_platform_id, + image_subsystem, + image_subsystem_major_version, + image_subsystem_minor_version, + padding_4, + active_process_affinity_mask, + gdi_handle_buffer, + post_process_init_routine, + tls_expansion_bitmap, + tls_expansion_bitmap_bits, + session_id, + padding_5, + app_compat_flags, + app_compat_flags_user, + p_shim_data, + app_compat_info, + csd_version, + activation_context_data, + process_assembly_storage_map, + system_default_activation_context_data, + system_assembly_storage_map, + minimum_stack_commit, + spare_pointers, + patch_loader_data, + chpe_v2_process_info, + app_model_feature_state, + spare_ulongs, + active_code_page, + oem_code_page, + use_case_mapping, + unused_nls_field, + padding_6a, + wer_registration_data, + wer_ship_assert_ptr, + ec_code_bit_map, + p_image_header_hash, + tracing_flags, + padding_6, + csr_server_read_only_shared_memory_base, + tpp_workerp_list_lock, + tpp_workerp_list, + wait_on_address_hash_table, + telemetry_coverage_header, + cloud_file_flags, + cloud_file_diag_flags, + placeholder_compatibility_mode, + placeholder_compatibility_mode_reserved, + leap_second_data, + leap_second_flags, + nt_global_flag_2, + extended_feature_disable_mask, + ] + ); + } + + #[test] + fn process_parameters_include_argv_and_environment() { + let created = created_process_environment_snapshot(); + + // `RTL_USER_PROCESS_PARAMETERS.CommandLine` stores the original command line for + // `CommandLineToArgvW`, and the environment is a sorted UTF-16 `name=value\0...\0\0` + // block as documented for `GetEnvironmentStringsW`. + assert_eq!( + decode_guest_unicode_string(created.process_parameters.command_line), + "test.exe \"arg with space\" \"quote\\\"arg\"" + ); + assert_ne!(created.process_parameters.environment, 0); + assert_eq!( + read_guest_utf16_units( + created.process_parameters.environment, + usize::try_from(created.process_parameters.environment_size) + .expect("environment size fits usize") + / size_of::(), + ), + utf16_environment_units(&["a=one", "B=two", "c=three"]) + ); + } + + #[test] + fn ntdll_exports_finds_ki_user_inverted_function_table() { + let ntdll = ntdll_module_base(); + let loaded_ntdll = loaded_module_image(ntdll); + + let exports = ntdll_exports::(&loaded_ntdll) + .expect("failed to parse ntdll exports"); + let expected_table = own_inverted_function_table() as usize; + + assert_eq!( + exports.ki_user_inverted_function_table, expected_table, + "ntdll export lookup returned the wrong KiUserInvertedFunctionTable address" + ); + } + + #[test] + fn dumps_own_inverted_function_table() { + assert_eq!( + core::mem::size_of::(), + 16 + ); + assert_eq!(core::mem::size_of::(), 24); + + let table = own_inverted_function_table(); + // SAFETY: `own_inverted_function_table` resolves a live data export from the + // current process's already-loaded ntdll. The header is copied immediately. + let header = unsafe { read_table_value::(table) }; + + std::println!( + "ntdll!KiUserInvertedFunctionTable @ {:#x}: current_size={} maximum_size={} epoch={} overflow={}", + table as usize, + header.current_size, + header.maximum_size, + header.epoch, + header.overflow + ); + + assert!(header.maximum_size > 0); + assert!(header.maximum_size <= MAXIMUM_INVERTED_FUNCTION_TABLE_SIZE); + assert!(header.current_size <= header.maximum_size); + assert!(header.current_size > 0); + + let entries = read_inverted_function_table_entries(table, header.current_size); + for (index, entry) in entries.iter().enumerate() { + let binary_name = module_name_from_base(entry.image_base); + std::println!( + " [{index}] binary=\"{}\" exception_directory={:#x} image_base={:#x} image_size={:#x} size_of_table={:#x}", + binary_name, + entry.exception_directory_address, + entry.image_base, + entry.image_size, + entry.size_of_table + ); + } + + assert_table_contains_entry( + &entries, + "ntdll.dll", + module_inverted_function_table_entry(ntdll_module_base()), + ); + assert_table_contains_entry( + &entries, + "the test executable", + module_inverted_function_table_entry(application_module_base()), + ); + } + + struct CreatedProcessEnvironmentSnapshot { + environment: WindowsProcessEnvironment, + peb: ProcessEnvironmentBlock, + teb: ThreadEnvironmentBlock, + process_parameters: RtlUserProcessParameters, + image_base_address: usize, + } + + fn created_process_environment_snapshot() -> CreatedProcessEnvironmentSnapshot { + let platform = crate::tests::test_platform(); + let litebox = litebox::LiteBox::new(platform); + let page_manager = crate::WindowsPageManager::::new(&litebox); + let fs = Arc::new(litebox::fs::in_mem::FileSystem::new(&litebox)); + let loader = PeLoader::new(platform, fs, &page_manager); + let image = loaded_module_image(application_module_base()); + + let image_base_address = image.mapping.base_addr; + let argv = [ + CString::new("test.exe").expect("valid argv[0]"), + CString::new("arg with space").expect("valid argv[1]"), + CString::new("quote\"arg").expect("valid argv[2]"), + ]; + let envp = [ + CString::new("c=three").expect("valid envp[0]"), + CString::new("B=two").expect("valid envp[1]"), + CString::new("a=one").expect("valid envp[2]"), + ]; + let environment = loader + .create_process_environment(ProcessEnvironmentInput { + image: &image.parsed, + image_base_address, + image_path: "test.exe", + argv: &argv, + envp: &envp, + stack_base: TEST_STACK_BASE, + stack_allocation_top: TEST_STACK_TOP, + }) + .expect("failed to create synthetic Windows process environment"); + let peb = read_guest_value::(environment.peb); + + CreatedProcessEnvironmentSnapshot { + process_parameters: read_guest_value(peb.process_parameters), + peb, + teb: read_guest_value(environment.teb), + environment, + image_base_address, + } + } + + fn read_guest_utf16_units(address: usize, units: usize) -> Vec { + let ptr = + ::RawConstPointer::::from_usize( + address, + ); + ptr.to_owned_slice(units) + .expect("guest UTF-16 block is readable") + .to_vec() + } + + fn utf16_environment_units(vars: &[&str]) -> Vec { + let mut units = Vec::new(); + for var in vars { + units.extend(var.encode_utf16()); + units.push(0); + } + units.push(0); + units + } + + fn print_field_diff(field: &str, synthetic: T, host: T) + where + T: core::fmt::Debug + IntoBytes + zerocopy::Immutable, + { + let status = if synthetic.as_bytes() == host.as_bytes() { + "✓" + } else { + "X" + }; + let synthetic = format_field_value(&synthetic); + let host = format_field_value(&host); + std::println!("{status:<2} {field:<48} {synthetic} | {host}"); + } + + fn print_peb_bit_field_diff( + field: &str, + synthetic: crate::nt_types::PebBitField, + host: crate::nt_types::PebBitField, + ) { + let status = if synthetic.bits() == host.bits() { + "✓" + } else { + "X" + }; + let synthetic = format_field_value(&synthetic); + let host = format_field_value(&host); + std::println!("{status:<2} {field:<48} {synthetic} | {host}"); + } + + fn print_unicode_string_diff(field: &str, synthetic: UnicodeString, host: UnicodeString) { + let synthetic = decode_guest_unicode_string(synthetic); + let host = decode_host_unicode_string(host); + let status = if synthetic == host { "✓" } else { "X" }; + let synthetic = format_field_value(&synthetic); + let host = format_field_value(&host); + std::println!("{status:<2} {field:<48} {synthetic} | {host}"); + } + + fn decode_guest_unicode_string(value: UnicodeString) -> String { + let Some(chars) = unicode_string_chars(value) else { + return std::format!("", value.length); + }; + if chars == 0 { + return String::new(); + } + if value.buffer == 0 { + return String::from(""); + } + + let ptr = + ::RawConstPointer::::from_usize( + value.buffer, + ); + let Some(units) = ptr.to_owned_slice(chars) else { + return String::from(""); + }; + String::from_utf16_lossy(&units) + } + + fn decode_host_unicode_string(value: UnicodeString) -> String { + let Some(chars) = unicode_string_chars(value) else { + return std::format!("", value.length); + }; + if chars == 0 { + return String::new(); + } + if value.buffer == 0 { + return String::from(""); + } + + // SAFETY: Host PEB/TEB snapshots contain pointers owned by the current + // process; the `UNICODE_STRING.Length` field bounds the UTF-16 slice. + let units = unsafe { core::slice::from_raw_parts(value.buffer as *const u16, chars) }; + String::from_utf16_lossy(units) + } + + fn unicode_string_chars(value: UnicodeString) -> Option { + if value.length.is_multiple_of(2) { + Some(usize::from(value.length / 2)) + } else { + None + } + } + + fn format_field_value(value: &T) -> String { + const MAX_VALUE_LEN: usize = 96; + + let mut value = std::format!("{value:x?}"); + if value.len() <= MAX_VALUE_LEN { + return value; + } + + let mut end = MAX_VALUE_LEN; + while !value.is_char_boundary(end) { + end -= 1; + } + value.truncate(end); + value.push_str("..."); + value + } + + fn print_diff_header(title: &str) { + std::println!("{title}"); + std::println!(" {:<48} synthetic | host", "field"); + std::println!(" {:<48} ----------------", "-----"); + } + + fn read_guest_value(address: usize) -> T + where + T: Copy + zerocopy::FromBytes, + { + let ptr = + ::RawConstPointer::::from_usize( + address, + ); + ptr.read_at_offset(0) + .expect("failed to read synthetic guest process environment value") + } + + fn host_teb_snapshot() -> ThreadEnvironmentBlock { + // SAFETY: `host_teb_address` returns the current thread's live host TEB pointer. + unsafe { read_host_value(host_teb_address() as *const ThreadEnvironmentBlock) } + } + + fn host_peb_snapshot() -> ProcessEnvironmentBlock { + // SAFETY: `host_peb_address` returns the current process's live host PEB pointer. + unsafe { read_host_value(host_peb_address() as *const ProcessEnvironmentBlock) } + } + + fn host_teb_address() -> usize { + let teb: usize; + // SAFETY: On x86_64 Windows, GS:[0x30] is the current thread's TEB pointer. + unsafe { + core::arch::asm!( + "mov {}, gs:[0x30]", + out(reg) teb, + options(nostack, preserves_flags, readonly), + ); + } + teb + } + + fn host_peb_address() -> usize { + let peb: usize; + // SAFETY: On x86_64 Windows, GS:[0x60] is the current process's PEB pointer. + unsafe { + core::arch::asm!( + "mov {}, gs:[0x60]", + out(reg) peb, + options(nostack, preserves_flags, readonly), + ); + } + peb + } + + fn host_client_id() -> ClientId { + // SAFETY: These kernel32 calls take no pointers and return IDs for the current process/thread. + let unique_process = unsafe { GetCurrentProcessId() }; + // SAFETY: These kernel32 calls take no pointers and return IDs for the current process/thread. + let unique_thread = unsafe { GetCurrentThreadId() }; + ClientId { + unique_process: usize::try_from(unique_process).unwrap(), + unique_thread: usize::try_from(unique_thread).unwrap(), + } + } + + unsafe fn read_host_value(address: *const T) -> T { + // SAFETY: The caller guarantees `address` points into live host loader/PEB/TEB state. + unsafe { core::ptr::read_volatile(address) } + } + + fn own_inverted_function_table() -> *const u8 { + let ntdll = ntdll_module_base(); + + // SAFETY: The module handle was returned by `GetModuleHandleW`, and the + // symbol name is a valid NUL-terminated C string literal. + let table = unsafe { GetProcAddress(ntdll, c"KiUserInvertedFunctionTable".as_ptr()) }; + assert!( + !table.is_null(), + "ntdll.dll does not export KiUserInvertedFunctionTable" + ); + + table.cast::() + } + + fn ntdll_module_base() -> *mut core::ffi::c_void { + module_base(Some("ntdll.dll")) + } + + fn application_module_base() -> *mut core::ffi::c_void { + module_base(None) + } + + fn module_base(name: Option<&str>) -> *mut core::ffi::c_void { + let module_name: Option> = name.map(|name| { + let mut name: Vec = name.encode_utf16().collect(); + name.push(0); + name + }); + let module_name_ptr = module_name.as_ref().map_or(core::ptr::null(), Vec::as_ptr); + // SAFETY: The string is NUL-terminated and points to a process-owned buffer + // that remains alive for the duration of the call. A null pointer asks for + // the current process's executable module. + let module = unsafe { GetModuleHandleW(module_name_ptr) }; + assert!(!module.is_null(), "module is not loaded in this process"); + + module + } + + fn module_inverted_function_table_entry( + module: *mut core::ffi::c_void, + ) -> KiUserInvertedFunctionTableEntry { + loaded_module_image(module) + .inverted_function_table_entry() + .expect("failed to build inverted function table entry") + .expect("loaded PE image has no exception directory") + } + + fn loaded_module_image(module: *mut core::ffi::c_void) -> LoadedImage { + let base_addr = module as usize; + let mut module_memory = ModuleMemory { + base: base_addr as *const u8, + }; + let parsed = PeParsedFile::parse(&mut module_memory) + .expect("failed to parse loaded PE image from memory"); + LoadedImage { + mapping: MappingInfo { + base_addr, + image_size: parsed.image_size(), + mapping_size: parsed.image_size(), + entry_point: base_addr + .checked_add(parsed.entry_point_rva()) + .expect("module entry point address fits usize"), + }, + pages: RangeMap::new(), + parsed, + } + } + + fn module_name_from_base(image_base: usize) -> String { + let module = image_base as *mut core::ffi::c_void; + let mut buffer = vec![0u16; 260]; + loop { + // SAFETY: `module` is the image base reported by ntdll's table, which is + // also the HMODULE for the loaded image. `buffer` is valid for `len` UTF-16 + // code units and remains alive for the duration of the call. + let len = unsafe { + GetModuleFileNameW( + module, + buffer.as_mut_ptr(), + u32::try_from(buffer.len()).unwrap(), + ) + } as usize; + if len == 0 { + return String::from(""); + } + if len < buffer.len() { + return String::from_utf16_lossy(&buffer[..len]); + } + buffer.resize(buffer.len() * 2, 0); + } + } + + fn read_inverted_function_table_entries( + table: *const u8, + current_size: u32, + ) -> Vec { + let entries = table.wrapping_add(core::mem::size_of::()); + let entry_size = core::mem::size_of::(); + (0..current_size as usize) + .map(|index| { + let entry_address = entries.wrapping_add(index * entry_size); + // SAFETY: The header just read from ntdll says `current_size` entries are + // initialized immediately after the header in this same exported table. + unsafe { read_table_value::(entry_address) } + }) + .collect() + } + + fn assert_table_contains_entry( + entries: &[KiUserInvertedFunctionTableEntry], + name: &str, + expected: KiUserInvertedFunctionTableEntry, + ) { + let actual = entries + .iter() + .find(|entry| entry.image_base == expected.image_base) + .unwrap_or_else(|| { + panic!("{name} was not present in the host inverted function table") + }); + + assert_eq!( + actual.exception_directory_address, + expected.exception_directory_address + ); + assert_eq!(actual.image_size, expected.image_size); + assert_eq!(actual.size_of_table, expected.size_of_table); + } + + unsafe fn read_table_value(address: *const u8) -> T { + // SAFETY: The caller guarantees that `address` points to at least + // `size_of::()` readable bytes. + let bytes = unsafe { core::slice::from_raw_parts(address, core::mem::size_of::()) }; + T::read_from_bytes(bytes).expect("failed to read table value") + } + + struct ModuleMemory { + base: *const u8, + } + + impl ReadAt for ModuleMemory { + type Error = core::convert::Infallible; + + fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), Self::Error> { + let offset: usize = offset.try_into().unwrap(); + // SAFETY: The test only constructs `ModuleMemory` from live module image + // bases returned by `GetModuleHandleW`. `PeParsedFile::parse` reads PE + // headers and section headers, which remain mapped in loaded images. + unsafe { + core::ptr::copy_nonoverlapping(self.base.add(offset), buf.as_mut_ptr(), buf.len()); + } + Ok(()) + } + + fn size(&mut self) -> Result { + Ok(u64::MAX) + } + } +} diff --git a/litebox_shim_windows/src/nt_types.rs b/litebox_shim_windows/src/nt_types.rs new file mode 100644 index 0000000000..1a0211eb83 --- /dev/null +++ b/litebox_shim_windows/src/nt_types.rs @@ -0,0 +1,901 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::string::String; +use core::mem::offset_of; +use litebox::platform::{RawConstPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes, KnownLayout}; + +use crate::{ConstPtr, syscalls::Handle}; + +bitflags::bitflags! { + /// Flags carried in `CONTEXT.ContextFlags`, selecting which register groups + /// a `CONTEXT` structure describes. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct ContextFlags: u32 { + const CONTROL = 0x0010_0001; + const INTEGER = 0x0010_0002; + const FLOATING_POINT = 0x0010_0008; + const DEBUG_REGISTERS = 0x0010_0010; + const XSTATE = 0x0010_0040; + + const _ = !0; + } +} + +const INITIAL_CONTEXT_MXCSR: u32 = 0x1f80; +const USER_MODE_CODE_SELECTOR: u16 = 0x33; +const USER_MODE_STACK_SELECTOR: u16 = 0x2b; +const INITIAL_CONTEXT_EFLAGS: u32 = 0x200; + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct Luid { + pub(crate) low_part: u32, + pub(crate) high_part: i32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct X64Context { + pub p1_home: u64, + pub p2_home: u64, + pub p3_home: u64, + pub p4_home: u64, + pub p5_home: u64, + pub p6_home: u64, + pub context_flags: u32, + pub mx_csr: u32, + pub seg_cs: u16, + pub seg_ds: u16, + pub seg_es: u16, + pub seg_fs: u16, + pub seg_gs: u16, + pub seg_ss: u16, + pub e_flags: u32, + pub dr0: u64, + pub dr1: u64, + pub dr2: u64, + pub dr3: u64, + pub dr6: u64, + pub dr7: u64, + pub rax: u64, + pub rcx: u64, + pub rdx: u64, + pub rbx: u64, + pub rsp: u64, + pub rbp: u64, + pub rsi: u64, + pub rdi: u64, + pub r8: u64, + pub r9: u64, + pub r10: u64, + pub r11: u64, + pub r12: u64, + pub r13: u64, + pub r14: u64, + pub r15: u64, + pub rip: u64, + pub extended_state: [u8; 0x3d0], +} + +impl Default for X64Context { + fn default() -> Self { + Self { + p1_home: 0, + p2_home: 0, + p3_home: 0, + p4_home: 0, + p5_home: 0, + p6_home: 0, + context_flags: 0, + mx_csr: 0, + seg_cs: 0, + seg_ds: 0, + seg_es: 0, + seg_fs: 0, + seg_gs: 0, + seg_ss: 0, + e_flags: 0, + dr0: 0, + dr1: 0, + dr2: 0, + dr3: 0, + dr6: 0, + dr7: 0, + rax: 0, + rcx: 0, + rdx: 0, + rbx: 0, + rsp: 0, + rbp: 0, + rsi: 0, + rdi: 0, + r8: 0, + r9: 0, + r10: 0, + r11: 0, + r12: 0, + r13: 0, + r14: 0, + r15: 0, + rip: 0, + extended_state: [0; 0x3d0], + } + } +} + +impl X64Context { + pub(crate) fn initial_thread_context( + thread_entry_point: usize, + application_entry_point: usize, + stack_top: usize, + peb: usize, + ) -> X64Context { + X64Context { + context_flags: ContextFlags::CONTROL + .union(ContextFlags::INTEGER) + .union(ContextFlags::FLOATING_POINT) + .union(ContextFlags::DEBUG_REGISTERS) + .bits(), + mx_csr: INITIAL_CONTEXT_MXCSR, + seg_cs: USER_MODE_CODE_SELECTOR, + seg_ss: USER_MODE_STACK_SELECTOR, + e_flags: INITIAL_CONTEXT_EFLAGS, + rcx: application_entry_point as u64, + rdx: peb as u64, + rsp: stack_top as u64, + rip: thread_entry_point as u64, + ..X64Context::default() + } + } +} + +bitflags::bitflags! { + /// Common Windows object-manager `ACCESS_MASK` rights shared by NT object types. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct AccessMask: u32 { + const DELETE = 0x0001_0000; + const READ_CONTROL = 0x0002_0000; + const WRITE_DAC = 0x0004_0000; + const WRITE_OWNER = 0x0008_0000; + const SYNCHRONIZE = 0x0010_0000; + const MAXIMUM_ALLOWED = 0x0200_0000; + const STANDARD_RIGHTS_READ = Self::READ_CONTROL.bits(); + const STANDARD_RIGHTS_WRITE = Self::READ_CONTROL.bits(); + const STANDARD_RIGHTS_EXECUTE = Self::READ_CONTROL.bits(); + const STANDARD_RIGHTS_ALL = Self::DELETE.bits() + | Self::READ_CONTROL.bits() + | Self::WRITE_DAC.bits() + | Self::WRITE_OWNER.bits() + | Self::SYNCHRONIZE.bits(); + + const GENERIC_ALL = 0x1000_0000; + const GENERIC_EXECUTE = 0x2000_0000; + const GENERIC_WRITE = 0x4000_0000; + const GENERIC_READ = 0x8000_0000; + + const _ = !0; + } +} + +bitflags::bitflags! { + /// Flags carried in `OBJECT_ATTRIBUTES.Attributes`. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct ObjectAttributesFlags: u32 { + const CASE_INSENSITIVE = 0x0000_0040; + const OPENIF = 0x0000_0080; + const OPENLINK = 0x0000_0100; + + const _ = !0; + } +} + +impl AccessMask { + pub(crate) fn expand_generic_access( + desired_access: u32, + generic_read: u32, + generic_write: u32, + generic_execute: u32, + generic_all: u32, + ) -> u32 { + let mut access = desired_access; + if desired_access & Self::GENERIC_READ.bits() != 0 { + access |= generic_read; + } + if desired_access & Self::GENERIC_WRITE.bits() != 0 { + access |= generic_write; + } + if desired_access & Self::GENERIC_EXECUTE.bits() != 0 { + access |= generic_execute; + } + if desired_access & Self::GENERIC_ALL.bits() != 0 { + access |= generic_all; + } + access + & !(Self::GENERIC_READ.bits() + | Self::GENERIC_WRITE.bits() + | Self::GENERIC_EXECUTE.bits() + | Self::GENERIC_ALL.bits()) + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable)] +pub(crate) struct ObjectAttributes { + pub(crate) length: u32, + pub(crate) root_directory: Handle, + pub(crate) object_name: usize, + pub(crate) attributes: u32, + pub(crate) security_descriptor: usize, + pub(crate) security_quality_of_service: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Default, FromBytes, IntoBytes, Immutable)] +pub(crate) struct IoStatusBlock { + pub(crate) status: i32, + pub(crate) padding_0: [u8; 4], + pub(crate) information: usize, +} + +const _: () = assert!(offset_of!(IoStatusBlock, information) == 0x8); + +impl IoStatusBlock { + pub(crate) const fn new(status: NtStatus, information: usize) -> Self { + Self { + status: status.as_raw(), + padding_0: [0; 4], + information, + } + } +} + +pub(crate) fn read_object_attributes( + object_attributes: ConstPtr, +) -> Result { + let Some(object_attributes) = object_attributes.read_at_offset(0) else { + return Err(NtStatus::ACCESS_VIOLATION); + }; + if object_attributes.length as usize != size_of::() { + return Err(NtStatus::INVALID_PARAMETER); + } + Ok(object_attributes) +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub(crate) struct UnicodeString { + pub(crate) length: u16, + pub(crate) maximum_length: u16, + pub(crate) padding_0: [u8; 4], + pub(crate) buffer: usize, +} + +impl UnicodeString { + pub(crate) fn read_string(self) -> Result { + if !self.length.is_multiple_of(2) { + return Err(NtStatus::INVALID_PARAMETER); + } + if self.maximum_length < self.length { + return Err(NtStatus::INVALID_PARAMETER); + } + if self.length == 0 { + return Ok(String::new()); + } + if self.buffer == 0 { + return Err(NtStatus::ACCESS_VIOLATION); + } + + let chars = usize::from(self.length / 2); + let buffer = + ::RawConstPointer::::from_usize( + self.buffer, + ); + let Some(units) = buffer.to_owned_slice(chars) else { + return Err(NtStatus::ACCESS_VIOLATION); + }; + Ok(String::from_utf16_lossy(&units)) + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub(crate) struct AhcServiceLookupCdb { + pub(crate) name: UnicodeString, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub(crate) struct AhcServiceData { + // TODO(ahc-service-data): model the full Win11 AHC_SERVICE_DATA sub-structs + // once their live boundaries are probed; phnt's ntmisc.h layout diverges + // from the observed guest layout before the verified fields below. + pub(crate) reserved_0: [u8; 0xf8], + pub(crate) lookup_cdb: AhcServiceLookupCdb, + pub(crate) reserved_1: [u8; 0x68], + pub(crate) driver_status: i32, + pub(crate) reserved_2: [u8; 4], + pub(crate) params_out: usize, + pub(crate) params_out_size: u32, + pub(crate) reserved_3: [u8; 4], +} + +bitflags::bitflags! { + /// Packed process flags stored in `PEB.BitField`. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub struct PebBitField: u8 { + const IMAGE_USES_LARGE_PAGES = 1 << 0; + const IS_PROTECTED_PROCESS = 1 << 1; + const IS_IMAGE_DYNAMICALLY_RELOCATED = 1 << 2; + const SKIP_PATCHING_USER32_FORWARDERS = 1 << 3; + const IS_PACKAGED_PROCESS = 1 << 4; + const IS_APP_CONTAINER = 1 << 5; + const IS_PROTECTED_PROCESS_LIGHT = 1 << 6; + const IS_LONG_PATH_AWARE_PROCESS = 1 << 7; + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct ProcessEnvironmentBlock { + pub inherited_address_space: u8, + pub read_image_file_exec_options: u8, + pub being_debugged: u8, + /// [`PebBitField`] + pub bit_field: u8, + pub padding_0: [u8; 4], + pub mutant: usize, + pub image_base_address: usize, + pub ldr: usize, + /// Pointer to [`RtlUserProcessParameters`]. + pub process_parameters: usize, + pub sub_system_data: usize, + pub process_heap: usize, + pub fast_peb_lock: usize, + pub atl_thunk_s_list_ptr: usize, + pub ifeo_key: usize, + pub cross_process_flags: u32, + pub padding_1: [u8; 4], + pub kernel_callback_table: usize, + pub system_reserved: u32, + pub atl_thunk_s_list_ptr_32: u32, + pub api_set_map: usize, + pub tls_expansion_counter: u32, + pub padding_2: [u8; 4], + pub tls_bitmap: usize, + pub tls_bitmap_bits: [u32; 2], + pub read_only_shared_memory_base: usize, + pub shared_data: usize, + pub read_only_static_server_data: usize, + pub ansi_code_page_data: usize, + pub oem_code_page_data: usize, + pub unicode_case_table_data: usize, + pub number_of_processors: u32, + pub nt_global_flag: u32, + pub critical_section_timeout: i64, + pub heap_segment_reserve: u64, + pub heap_segment_commit: u64, + pub heap_de_commit_total_free_threshold: u64, + pub heap_de_commit_free_block_threshold: u64, + pub number_of_heaps: u32, + pub maximum_number_of_heaps: u32, + pub process_heaps: usize, + pub gdi_shared_handle_table: usize, + pub process_starter_helper: usize, + pub gdi_dc_attribute_list: u32, + pub padding_3: [u8; 4], + pub loader_lock: usize, + pub os_major_version: u32, + pub os_minor_version: u32, + pub os_build_number: u16, + pub os_csd_version: u16, + pub os_platform_id: u32, + pub image_subsystem: u32, + pub image_subsystem_major_version: u32, + pub image_subsystem_minor_version: u32, + pub padding_4: [u8; 4], + pub active_process_affinity_mask: u64, + pub gdi_handle_buffer: [u32; 60], + pub post_process_init_routine: usize, + pub tls_expansion_bitmap: usize, + pub tls_expansion_bitmap_bits: [u32; 32], + pub session_id: u32, + pub padding_5: [u8; 4], + pub app_compat_flags: u64, + pub app_compat_flags_user: u64, + pub p_shim_data: usize, + pub app_compat_info: usize, + pub csd_version: UnicodeString, + pub activation_context_data: usize, + pub process_assembly_storage_map: usize, + pub system_default_activation_context_data: usize, + pub system_assembly_storage_map: usize, + pub minimum_stack_commit: u64, + pub spare_pointers: [usize; 2], + pub patch_loader_data: usize, + pub chpe_v2_process_info: usize, + pub app_model_feature_state: u32, + pub spare_ulongs: [u32; 2], + pub active_code_page: u16, + pub oem_code_page: u16, + pub use_case_mapping: u16, + pub unused_nls_field: u16, + pub padding_6a: [u8; 4], + pub wer_registration_data: usize, + pub wer_ship_assert_ptr: usize, + pub ec_code_bit_map: usize, + pub p_image_header_hash: usize, + pub tracing_flags: u32, + pub padding_6: [u8; 4], + pub csr_server_read_only_shared_memory_base: u64, + pub tpp_workerp_list_lock: u64, + pub tpp_workerp_list: ListEntry, + pub wait_on_address_hash_table: [usize; 128], + pub telemetry_coverage_header: usize, + pub cloud_file_flags: u32, + pub cloud_file_diag_flags: u32, + pub placeholder_compatibility_mode: i8, + pub placeholder_compatibility_mode_reserved: [i8; 7], + pub leap_second_data: usize, + pub leap_second_flags: u32, + pub nt_global_flag_2: u32, + pub extended_feature_disable_mask: u64, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct NtTib { + pub exception_list: usize, + pub stack_base: usize, + pub stack_limit: usize, + pub sub_system_tib: usize, + pub fiber_data_or_version: usize, + pub arbitrary_user_pointer: usize, + pub self_pointer: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct ActivationContextStack { + _reserved: [u8; 0x28], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct GdiTebBatch { + _reserved: [u8; 0x4e8], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, FromBytes, IntoBytes, Immutable)] +pub struct ClientId { + pub unique_process: usize, + pub unique_thread: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct ListEntry { + pub flink: usize, + pub blink: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct Guid { + pub data: [u8; 16], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct GroupAffinity { + pub mask: usize, + pub group: u16, + pub reserved: [u16; 3], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct ThreadEnvironmentBlock { + pub nt_tib: NtTib, + pub environment_pointer: usize, + pub client_id: ClientId, + pub active_rpc_handle: usize, + pub thread_local_storage_pointer: usize, + /// Pointer to [`ProcessEnvironmentBlock`]. + pub process_environment_block: usize, + pub last_error_value: u32, + pub count_of_owned_critical_sections: u32, + pub csr_client_thread: usize, + pub win_32_thread_info: usize, + pub user_32_reserved: [u32; 26], + pub user_reserved: [u32; 5], + pub padding_user_reserved: [u8; 4], + pub wow_32_reserved: usize, + pub current_locale: u32, + pub fp_software_status_register: u32, + pub reserved_for_debugger_instrumentation: [usize; 16], + pub system_reserved_1: [usize; 25], + pub heap_fls_data: usize, + pub rng_state: [u64; 4], + pub placeholder_compatibility_mode: i8, + pub placeholder_hydration_always_explicit: u8, + pub placeholder_reserved: [i8; 10], + pub proxied_process_id: u32, + pub activation_stack: ActivationContextStack, + pub working_on_behalf_ticket: [u8; 8], + pub exception_code: i32, + pub padding_0: [u8; 4], + pub activation_context_stack_pointer: usize, + pub instrumentation_callback_sp: u64, + pub instrumentation_callback_previous_pc: u64, + pub instrumentation_callback_previous_sp: u64, + pub tx_fs_context: u32, + pub instrumentation_callback_disabled: u8, + pub unaligned_load_store_exceptions: u8, + pub padding_1: [u8; 2], + pub gdi_teb_batch: GdiTebBatch, + pub real_client_id: ClientId, + pub gdi_cached_process_handle: usize, + pub gdi_client_pid: u32, + pub gdi_client_tid: u32, + pub gdi_thread_local_info: usize, + pub win_32_client_info: [u64; 62], + pub gl_dispatch_table: [usize; 233], + pub gl_reserved_1: [u64; 29], + pub gl_reserved_2: usize, + pub gl_section_info: usize, + pub gl_section: usize, + pub gl_table: usize, + pub gl_current_rc: usize, + pub gl_context: usize, + pub last_status_value: u32, + pub padding_2: [u8; 4], + pub static_unicode_string: UnicodeString, + pub static_unicode_buffer: [u16; 261], + pub padding_3: [u8; 6], + pub deallocation_stack: usize, + pub tls_slots: [usize; 64], + pub tls_links: ListEntry, + pub vdm: usize, + pub reserved_for_nt_rpc: usize, + pub dbg_ss_reserved: [usize; 2], + pub hard_error_mode: u32, + pub padding_4: [u8; 4], + pub instrumentation: [usize; 11], + pub activity_id: Guid, + pub sub_process_tag: usize, + pub perflib_data: usize, + pub etw_trace_data: usize, + pub win_sock_data: usize, + pub gdi_batch_count: u32, + pub ideal_processor_value: u32, + pub guaranteed_stack_bytes: u32, + pub padding_5: [u8; 4], + pub reserved_for_perf: usize, + pub reserved_for_ole: usize, + pub waiting_on_loader_lock: u32, + pub padding_6: [u8; 4], + pub saved_priority_state: usize, + pub reserved_for_code_coverage: u64, + pub thread_pool_data: usize, + pub tls_expansion_slots: usize, + pub chpe_v_2_cpu_area_info: usize, + pub unused: usize, + pub mui_generation: u32, + pub is_impersonating: u32, + pub nls_cache: usize, + pub p_shim_data: usize, + pub heap_data: u32, + pub padding_7: [u8; 4], + pub current_transaction_handle: usize, + pub active_frame: usize, + pub fls_data: usize, + pub preferred_languages: usize, + pub user_pref_languages: usize, + pub merged_pref_languages: usize, + pub mui_impersonation: u32, + pub cross_teb_flags: u16, + pub same_teb_flags: u16, + pub txn_scope_enter_callback: usize, + pub txn_scope_exit_callback: usize, + pub txn_scope_context: usize, + pub lock_count: u32, + pub wow_teb_offset: i32, + pub resource_ret_value: usize, + pub reserved_for_wdf: usize, + pub reserved_for_crt: u64, + pub effective_container_id: Guid, + pub last_sleep_counter: u64, + pub spin_call_count: u32, + pub padding_8: [u8; 4], + pub extended_feature_disable_mask: u64, + pub scheduler_shared_data_slot: usize, + pub heap_walk_context: usize, + pub primary_group_affinity: GroupAffinity, + pub rcu: [u32; 2], +} + +bitflags::bitflags! { + /// Flags stored in `RTL_USER_PROCESS_PARAMETERS.Flags`. + #[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] + pub struct RtlUserProcFlags: u32 { + /// Pointers in the process-parameter block are absolute addresses. + const NORMALIZED = 0x0000_0001; + const PROFILE_USER = 0x0000_0002; + const PROFILE_KERNEL = 0x0000_0004; + const PROFILE_SERVER = 0x0000_0008; + const UNKNOWN = 0x0000_0010; + /// Reserve low address space at process creation. + const RESERVE_1MB = 0x0000_0020; + /// Reserve low address space at process creation. + const RESERVE_16MB = 0x0000_0040; + const CASE_SENSITIVE = 0x0000_0080; + const DISABLE_HEAP_DECOMMIT = 0x0000_0100; + const PROCESS_OR_1 = 0x0000_0200; + const PROCESS_OR_2 = 0x0000_0400; + const DLL_REDIRECTION_LOCAL = 0x0000_1000; + /// An application manifest was detected during process creation. + const APP_MANIFEST_PRESENT = 0x0000_2000; + /// The corresponding Image File Execution Options key was missing at process creation. + const IMAGE_KEY_MISSING = 0x0000_4000; + /// System-global IFEO development override support is enabled. + const DEV_OVERRIDE_ENABLED = 0x0000_8000; + const OPTIN_PROCESS = 0x0002_0000; + const SESSION_OWNER = 0x0004_0000; + const HANDLE_USER_CALLBACK_EXCEPTIONS = 0x0008_0000; + const PROTECTED_PROCESS = 0x0040_0000; + const NO_IMAGE_EXPANSION_MITIGATION = 0x0200_0000; + const APPX_LOADER_ALTERNATE_FORWARDER = 0x0400_0000; + const APPX_GLOBAL_OVERRIDE = 0x0800_0000; + /// Allow the loader to use OneCore API-set forwarders when resolving imports. + const ONECORE_FORWARDERS_ENABLED = 0x2000_0000; + /// Opt back in to the normal `ExitProcess` path that detaches DLLs on exit. + const EXIT_PROCESS_NORMAL = 0x4000_0000; + const SECURE_PROCESS = 0x8000_0000; + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct CurDir { + pub dos_path: UnicodeString, + pub handle: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct RtlDriveLetterCurdir { + /// Per-drive current-directory flags. + pub flags: u16, + /// Length of the drive current-directory entry. + pub length: u16, + /// Timestamp associated with this drive current-directory entry. + pub time_stamp: u32, + /// DOS path for this drive's current directory. + pub dos_path: UnicodeString, +} + +/// Memory layout of this struct: +/// +/// ```text +/// +-------------------------------+ +/// | RTL_USER_PROCESS_PARAMETERS | +/// | fixed-size struct | +/// +-------------------------------+ +/// | CurrentDirectory.DosPath | +/// | (string buffer) | +/// +-------------------------------+ +/// | DllPath | +/// +-------------------------------+ +/// | ImagePathName | +/// +-------------------------------+ +/// | CommandLine | +/// +-------------------------------+ +/// | WindowTitle | +/// +-------------------------------+ +/// | DesktopInfo | +/// +-------------------------------+ +/// | ShellInfo | +/// +-------------------------------+ +/// | RuntimeData | +/// +-------------------------------+ +/// | RedirectionDllName | +/// +-------------------------------+ +/// ``` +/// +/// See for details on the fields of this struct. +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +pub struct RtlUserProcessParameters { + /// Total allocated size of this process-parameter buffer, in bytes. + pub maximum_length: u32, + /// Size of the process-parameter block, including any inline variable-length strings. + pub length: u32, + /// Process-parameter flags (see [`RtlUserProcFlags`]). + pub flags: u32, + /// Debug flags associated with these process parameters. + pub debug_flags: u32, + /// Console session handle, inherited or derived from process creation options. + pub console_handle: usize, + /// Console behavior flags, such as ignoring Ctrl+C requests. + pub console_flags: u32, + /// Reserved alignment padding. + pub padding_0: [u8; 4], + /// Standard input handle from `STARTUPINFO.hStdInput`. + pub standard_input: usize, + /// Standard output handle from `STARTUPINFO.hStdOutput`. + pub standard_output: usize, + /// Standard error handle from `STARTUPINFO.hStdError`. + pub standard_error: usize, + /// Current directory path and handle. + pub current_directory: CurDir, + /// Semicolon-separated DOS-style DLL search paths. + pub dll_path: UnicodeString, + /// Full DOS-style path to the executable image. + pub image_path_name: UnicodeString, + /// Command line string passed to the process. + pub command_line: UnicodeString, + /// Pointer to the separately allocated environment block. + pub environment: usize, + /// Initial window X position when `window_flags` requests a position. + pub starting_x: u32, + /// Initial window Y position when `window_flags` requests a position. + pub starting_y: u32, + /// Initial window width when `window_flags` requests a size. + pub count_x: u32, + /// Initial window height when `window_flags` requests a size. + pub count_y: u32, + /// Initial console screen-buffer width in character cells. + pub count_chars_x: u32, + /// Initial console screen-buffer height in character cells. + pub count_chars_y: u32, + /// Initial console text/background color attributes. + pub fill_attribute: u32, + /// `STARTUPINFO` flags describing which startup fields are valid. + pub window_flags: u32, + /// `ShowWindow` value used when `window_flags` includes `STARTF_USESHOWWINDOW`. + pub show_window_flags: u32, + /// Reserved alignment padding. + pub padding_1: [u8; 4], + /// Console window title, shortcut path, or AppUserModelID depending on `window_flags`. + pub window_title: UnicodeString, + /// Window station and desktop name, such as `WinSta0\Default`. + pub desktop_info: UnicodeString, + /// Startup shell data corresponding to `STARTUPINFO.lpReserved`. + pub shell_info: UnicodeString, + /// Runtime data corresponding to `STARTUPINFO.lpReserved2` and `cbReserved2`. + pub runtime_data: UnicodeString, + /// Per-drive current-directory entries for the 32 DOS drive letters. + pub current_directories: [RtlDriveLetterCurdir; 32], + /// Allocated size of the environment block, in bytes. + pub environment_size: u64, + /// Environment version incremented when environment strings change. + pub environment_version: u64, + /// Package dependency metadata pointer. + pub package_dependency_data: usize, + /// Console process group identifier used to scope control-signal delivery. + pub process_group_id: u32, + /// Requested worker-thread count for parallel DLL loading. + pub loader_threads: u32, + /// DLL path used for packaged-app import redirection. + pub redirection_dll_name: UnicodeString, + /// Heap partition name. + pub heap_partition_name: UnicodeString, + /// Pointer to default thread-pool CPU-set masks. + pub default_threadpool_cpu_set_masks: usize, + /// Number of default thread-pool CPU-set masks. + pub default_threadpool_cpu_set_mask_count: u32, + /// Maximum default thread-pool thread count. + pub default_threadpool_thread_maximum: u32, + /// Heap memory type mask. + pub heap_memory_type_mask: u32, + /// Reserved tail padding. + pub padding_2: [u8; 4], +} + +const _: [(); 0x1878] = [(); core::mem::size_of::()]; +const _: [(); 0x7d0] = [(); core::mem::size_of::()]; +const _: [(); 0x4d0] = [(); core::mem::size_of::()]; +const _: [(); 0x448] = [(); core::mem::size_of::()]; + +#[repr(C)] +#[derive(Clone, Copy, FromBytes, Immutable, IntoBytes, KnownLayout)] +pub struct KSystemTime { + pub low_part: u32, + pub high_1_time: i32, + pub high_2_time: i32, +} + +#[cfg(not(target_os = "windows"))] +const WINDOWS_KUSER_SHARED_DATA_XSTATE_CONFIGURATION_SIZE: usize = 0x348; + +/// Layout from Wine `include/ddk/wdm.h` and ReactOS `sdk/include/wine/ddk/wdm.h`. +#[cfg(not(target_os = "windows"))] +#[repr(C)] +#[derive(Clone, Copy, FromBytes, Immutable, IntoBytes, KnownLayout)] +pub struct KUserSharedData { + tick_count_low_deprecated: u32, + tick_count_multiplier: u32, + interrupt_time: KSystemTime, + system_time: KSystemTime, + time_zone_bias: KSystemTime, + image_number_low: u16, + image_number_high: u16, + pub nt_system_root: [u16; 260], + max_stack_trace_depth: u32, + crypto_exponent: u32, + time_zone_id: u32, + large_page_minimum: u32, + ait_sampling_value: u32, + app_compat_flag: u32, + rng_seed_version: u64, + global_validation_run_level: u32, + time_zone_bias_stamp: u32, + pub nt_build_number: u32, + pub nt_product_type: u32, + pub product_type_is_valid: u8, + reserved_0: u8, + native_processor_architecture: u16, + pub nt_major_version: u32, + pub nt_minor_version: u32, + processor_features: [u8; 64], + reserved_1: u32, + reserved_3: u32, + time_slip: u32, + alternative_architecture: u32, + boot_id: u32, + system_expiration_date: i64, + suite_mask: u32, + kd_debugger_enabled: u8, + nx_support_policy: u8, + cycles_per_yield: u16, + active_console_id: u32, + dismount_count: u32, + com_plus_package: u32, + last_system_rit_event_tick_count: u32, + number_of_physical_pages: u32, + safe_boot_mode: u8, + virtualization_flags: u8, + padding_2ee: [u8; 2], + shared_data_flags: u32, + data_flags_pad: [u32; 1], + test_ret_instruction: u64, + qpc_frequency: i64, + system_call: u32, + user_cet_available_environments: u32, + system_call_pad: [u64; 2], + tick_count: [u8; 0x10], + cookie: u32, + cookie_pad: [u32; 1], + console_session_foreground_process_id: i64, + time_update_lock: u64, + baseline_system_time_qpc: u64, + baseline_interrupt_time_qpc: u64, + qpc_system_time_increment: u64, + qpc_interrupt_time_increment: u64, + qpc_system_time_increment_shift: u8, + qpc_interrupt_time_increment_shift: u8, + unparked_processor_count: u16, + enclave_feature_mask: [u32; 4], + telemetry_coverage_round: u32, + user_mode_global_logger: [u16; 16], + image_file_execution_options: u32, + lang_generation_count: u32, + active_processor_affinity: u32, + padding_3ac: u32, + interrupt_time_bias: u64, + qpc_bias: u64, + active_processor_count: u32, + active_group_count: u8, + padding_3c5: u8, + qpc_data: u16, + time_zone_bias_effective_start: i64, + time_zone_bias_effective_end: i64, + x_state: [u8; WINDOWS_KUSER_SHARED_DATA_XSTATE_CONFIGURATION_SIZE], + feature_configuration_change_stamp: KSystemTime, + spare: u32, + user_pointer_auth_mask: u64, +} diff --git a/litebox_shim_windows/src/syscalls/apphelp.rs b/litebox_shim_windows/src/syscalls/apphelp.rs new file mode 100644 index 0000000000..6711325113 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/apphelp.rs @@ -0,0 +1,121 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use core::mem::offset_of; +use int_enum::IntEnum; + +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::AhcServiceData; +use crate::{MutPtr, ShimPlatform}; + +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +#[repr(u32)] +pub enum AhcServiceClass { + Lookup = 0, + Remove = 1, + Update = 2, + Clear = 3, + SnapStatistics = 4, + SnapCache = 5, + LookupCdb = 6, + RefreshCdb = 7, + MapQuirks = 8, + HwIdQuery = 9, + InitProcessData = 10, + LookupAndWriteToProcess = 11, +} + +fn handle_lookup_cdb( + service_data: Option>, +) -> NtStatus { + let Some(data_ptr) = service_data else { + return NtStatus::INVALID_PARAMETER; + }; + + let Some(service_data) = data_ptr.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + + if service_data.params_out == 0 || service_data.params_out_size != size_of::().trunc() { + return NtStatus::INVALID_PARAMETER; + } + + match service_data.lookup_cdb.name.read_string::() { + Ok(name) => { + litebox_util_log::debug!( + lookup_cdb_name:% = name, + params_out:% = format_args!("{:#x}", service_data.params_out), + params_out_size = service_data.params_out_size; + "Decoded NtApphelpCacheControl LookupCdb service data" + ); + } + Err(status) => { + litebox_util_log::warn!( + status:? = status, + params_out:% = format_args!("{:#x}", service_data.params_out), + params_out_size = service_data.params_out_size; + "Failed to decode NtApphelpCacheControl LookupCdb name" + ); + } + } + + // TODO: zero seems to indicate no matches. + let params_out = MutPtr::::from_usize(service_data.params_out); + if params_out.write_at_offset(0, 0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + + if crate::write_field_at_offset::( + data_ptr.as_usize(), + offset_of!(AhcServiceData, driver_status), + NtStatus::SUCCESS.as_raw(), + ) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS +} + +pub(crate) fn sys_nt_apphelp_cache_control( + service_class: u32, + service_data: Option>, +) -> NtStatus { + let Ok(service_class) = AhcServiceClass::try_from(service_class) else { + litebox_util_log::debug!( + service_class, + service_data:% = format_args!("{:#x}", service_data.map_or(0, |ptr| ptr.as_usize())); + "Rejected NtApphelpCacheControl service class" + ); + return NtStatus::INVALID_PARAMETER; + }; + + let status = match service_class { + AhcServiceClass::LookupCdb => handle_lookup_cdb::(service_data), + AhcServiceClass::Lookup | AhcServiceClass::LookupAndWriteToProcess => { + NtStatus::NOT_SUPPORTED + } + AhcServiceClass::Remove + | AhcServiceClass::Update + | AhcServiceClass::Clear + | AhcServiceClass::SnapStatistics + | AhcServiceClass::SnapCache + | AhcServiceClass::RefreshCdb + | AhcServiceClass::MapQuirks + | AhcServiceClass::HwIdQuery + | AhcServiceClass::InitProcessData => NtStatus::NOT_SUPPORTED, + }; + + litebox_util_log::debug!( + service_class:? = service_class, + service_data:% = format_args!("{:#x}", service_data.map_or(0, |ptr| ptr.as_usize())), + status:? = status; + "Handled NtApphelpCacheControl with empty apphelp cache" + ); + + status +} diff --git a/litebox_shim_windows/src/syscalls/condrv.rs b/litebox_shim_windows/src/syscalls/condrv.rs new file mode 100644 index 0000000000..956a1accd1 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/condrv.rs @@ -0,0 +1,428 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows console driver support. + +use alloc::sync::Arc; +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::IoStatusBlock; +use crate::{ConstPtr, MutPtr}; + +const FILE_DEVICE_CONSOLE: u32 = 0x50; +const CD_SERVER_EA_NAME: &[u8] = b"server"; + +#[repr(u8)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +pub(crate) enum CondrvObject { + Input = 0, + Output = 1, + CurrentInput = 2, + CurrentOutput = 3, + ScreenBuffer = 4, + Server = 5, + Reference = 6, + Connect = 7, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +pub(crate) enum CondrvStreamDirection { + Input, + Output, +} + +pub(crate) struct CondrvStreamObject { + id: u64, +} + +struct CondrvConsoleState { + next_object_id: u64, + bound_input: Arc, + active_output: Arc, +} + +pub(crate) struct CondrvConsole { + state: litebox::sync::Mutex, +} + +impl CondrvStreamObject { + pub(crate) fn id(&self) -> u64 { + self.id + } +} + +impl CondrvConsole { + pub(crate) fn new() -> Self { + let bound_input = Arc::new(CondrvStreamObject { id: 1 }); + let active_output = Arc::new(CondrvStreamObject { id: 2 }); + Self { + state: litebox::sync::Mutex::new(CondrvConsoleState { + next_object_id: 3, + bound_input, + active_output, + }), + } + } + + pub(crate) fn open_stream( + &self, + endpoint: CondrvObject, + ) -> Result, NtStatus> { + let mut state = self.state.lock(); + match endpoint { + CondrvObject::CurrentInput => Ok(Arc::clone(&state.bound_input)), + // TODO(condrv-activate-buffer): update this pointer when LiteBox implements and + // host-validates the ConDrv activate-buffer IOCTL. + CondrvObject::CurrentOutput => Ok(Arc::clone(&state.active_output)), + CondrvObject::Input | CondrvObject::Output | CondrvObject::ScreenBuffer => { + state.allocate_object() + } + CondrvObject::Server | CondrvObject::Reference | CondrvObject::Connect => { + Err(NtStatus::OBJECT_TYPE_MISMATCH) + } + } + } +} + +impl CondrvConsoleState { + fn allocate_object(&mut self) -> Result, NtStatus> { + let id = self.next_object_id; + self.next_object_id = id.checked_add(1).ok_or(NtStatus::QUOTA_EXCEEDED)?; + Ok(Arc::new(CondrvStreamObject { id })) + } +} + +impl CondrvObject { + pub(crate) fn from_device_name(name: &str) -> Result { + match Self::from_component(name) { + Some( + object @ (Self::Input + | Self::Output + | Self::CurrentInput + | Self::CurrentOutput + | Self::ScreenBuffer + | Self::Server), + ) => Ok(object), + Some(Self::Reference) => Err(NtStatus::INVALID_HANDLE), + Some(Self::Connect) => Err(NtStatus::OBJECT_TYPE_MISMATCH), + None => Err(NtStatus::OBJECT_NAME_NOT_FOUND), + } + } + + fn from_component(name: &str) -> Option { + if name.eq_ignore_ascii_case("Input") { + Some(Self::Input) + } else if name.eq_ignore_ascii_case("Output") { + Some(Self::Output) + } else if name.eq_ignore_ascii_case("CurrentIn") { + Some(Self::CurrentInput) + } else if name.eq_ignore_ascii_case("CurrentOut") { + Some(Self::CurrentOutput) + } else if name.eq_ignore_ascii_case("ScreenBuffer") { + Some(Self::ScreenBuffer) + } else if name.eq_ignore_ascii_case("Server") { + Some(Self::Server) + } else if name.eq_ignore_ascii_case("Reference") { + Some(Self::Reference) + } else if name.eq_ignore_ascii_case("Connect") { + Some(Self::Connect) + } else { + None + } + } + + pub(crate) fn relative_child(self, name: &str) -> Result { + let name = name.strip_prefix('\\').ok_or(NtStatus::NOT_FOUND)?; + let child = Self::from_component(name).ok_or(NtStatus::NOT_FOUND)?; + + match child { + Self::Server => Ok(child), + Self::Reference => match self { + Self::Server | Self::Connect => Ok(child), + _ => Err(NtStatus::OBJECT_TYPE_MISMATCH), + }, + Self::Connect => { + if self == Self::Reference { + Ok(child) + } else { + Err(NtStatus::INVALID_HANDLE) + } + } + Self::Input + | Self::Output + | Self::CurrentInput + | Self::CurrentOutput + | Self::ScreenBuffer => { + if self == Self::Server { + Err(NtStatus::INVALID_DEVICE_STATE) + } else { + Ok(child) + } + } + } + } + + pub(crate) fn handle_path(self) -> &'static str { + match self { + Self::Input | Self::CurrentInput => "/dev/stdin", + Self::Output | Self::CurrentOutput | Self::ScreenBuffer => "/dev/stdout", + Self::Server => r"\Device\ConDrv\Server", + Self::Reference => r"\Device\ConDrv\Reference", + Self::Connect => r"\Device\ConDrv\Connect", + } + } + + pub(crate) fn stream_direction(self) -> Option { + match self { + Self::Input | Self::CurrentInput => Some(CondrvStreamDirection::Input), + Self::Output | Self::CurrentOutput | Self::ScreenBuffer => { + Some(CondrvStreamDirection::Output) + } + Self::Server | Self::Reference | Self::Connect => None, + } + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum IoControlMethod { + Buffered = 0, + InDirect = 1, + OutDirect = 2, + Neither = 3, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct IoControlAccess: u32 { + const ANY = 0; + const READ = 1; + const WRITE = 2; + const _ = !0; + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum ConsoleIoControlFunction { + LaunchServer = 13, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct FileFullEaInformation { + next_entry_offset: u32, + flags: u8, + ea_name_length: u8, + ea_value_length: u16, +} + +#[cfg(test)] +pub(crate) fn ea_buffer(name: &[u8], value_length: usize) -> alloc::vec::Vec { + let header = FileFullEaInformation { + next_entry_offset: 0, + flags: 0, + ea_name_length: u8::try_from(name.len()).unwrap(), + ea_value_length: u16::try_from(value_length).unwrap(), + }; + let mut buffer = alloc::vec::Vec::new(); + buffer.extend_from_slice(header.as_bytes()); + buffer.extend_from_slice(name); + buffer.push(0); + buffer.resize(buffer.len() + value_length, 0); + buffer +} + +pub(crate) fn validate_connect_server_ea( + ea_buffer: Option>, + ea_length: u32, +) -> Result<(), NtStatus> { + let Some(ea_buffer) = ea_buffer else { + return Err(NtStatus::EAS_NOT_SUPPORTED); + }; + let ea_length = ea_length as usize; + let Some(entry) = ConstPtr::::from_usize(ea_buffer.as_usize()) + .read_at_offset(0) + else { + return Err(NtStatus::ACCESS_VIOLATION); + }; + + let name_offset = size_of::(); + let name_length = entry.ea_name_length as usize; + let value_length = entry.ea_value_length as usize; + let value_offset = name_offset + .checked_add(name_length) + .and_then(|offset| offset.checked_add(1)) + .ok_or(NtStatus::EAS_NOT_SUPPORTED)?; + let entry_length = value_offset + .checked_add(value_length) + .ok_or(NtStatus::EAS_NOT_SUPPORTED)?; + if entry_length > ea_length { + return Err(NtStatus::EAS_NOT_SUPPORTED); + } + + let Some(name_address) = ea_buffer.as_usize().checked_add(name_offset) else { + return Err(NtStatus::EAS_NOT_SUPPORTED); + }; + let Some(name_with_nul) = + ConstPtr::::from_usize(name_address).to_owned_slice(name_length + 1) + else { + return Err(NtStatus::ACCESS_VIOLATION); + }; + let Some((&0, name)) = name_with_nul.split_last() else { + return Err(NtStatus::EAS_NOT_SUPPORTED); + }; + if !name.eq_ignore_ascii_case(CD_SERVER_EA_NAME) { + return Err(NtStatus::EAS_NOT_SUPPORTED); + } + + let Some(value_address) = ea_buffer.as_usize().checked_add(value_offset) else { + return Err(NtStatus::EAS_NOT_SUPPORTED); + }; + // Windows rejects a structurally valid but zeroed server handshake with + // STATUS_PIPE_DISCONNECTED. + // TODO(condrv-handshake): fully decode the undocumented server payload; the current subset + // only pins the native all-zero rejection and validates its readable extent. + let Some(value) = + ConstPtr::::from_usize(value_address).to_owned_slice(value_length) + else { + return Err(NtStatus::ACCESS_VIOLATION); + }; + if value.iter().all(|byte| *byte == 0) { + return Err(NtStatus::PIPE_DISCONNECTED); + } + + Ok(()) +} + +pub(crate) fn handle_ioctl( + condrv_object: CondrvObject, + io_status_block: MutPtr, + io_control_code: u32, + input_buffer: Option>, + input_buffer_length: u32, + output_buffer: Option>, + output_buffer_length: u32, +) -> NtStatus { + let device_type = io_control_code >> 16; + let access = IoControlAccess::from_bits_retain((io_control_code >> 14) & 0x3); + let function = (io_control_code >> 2) & 0xfff; + let method = IoControlMethod::try_from(io_control_code & 0x3); + + if device_type != FILE_DEVICE_CONSOLE || method != Ok(IoControlMethod::Neither) { + litebox_util_log::debug!( + condrv_object:? = condrv_object, + io_control_code:% = format_args!("{io_control_code:#x}"); + "Unsupported ConDrv IOCTL shape" + ); + return complete_ioctl::(io_status_block, NtStatus::NOT_SUPPORTED, 0); + } + + let Ok(function) = ConsoleIoControlFunction::try_from(function) else { + litebox_util_log::debug!( + condrv_object:? = condrv_object, + io_control_code:% = format_args!("{io_control_code:#x}"); + "Unsupported ConDrv IOCTL function" + ); + return complete_ioctl::(io_status_block, NtStatus::NOT_SUPPORTED, 0); + }; + + match (condrv_object, function) { + (CondrvObject::Server, ConsoleIoControlFunction::LaunchServer) + if access.is_empty() + && input_buffer.is_some() + && input_buffer_length != 0 + && output_buffer.is_none() + && output_buffer_length == 0 => + { + if input_buffer + .and_then(|input_buffer| input_buffer.read_at_offset(0)) + .is_none() + { + return complete_ioctl::(io_status_block, NtStatus::ACCESS_VIOLATION, 0); + } + complete_ioctl::(io_status_block, NtStatus::SUCCESS, 0) + } + _ => { + litebox_util_log::debug!( + condrv_object:? = condrv_object, + function:? = function, + io_control_code:% = format_args!("{io_control_code:#x}"); + "Unsupported ConDrv IOCTL for object" + ); + complete_ioctl::(io_status_block, NtStatus::NOT_SUPPORTED, 0) + } + } +} + +pub(crate) fn complete_ioctl( + io_status_block: MutPtr, + status: NtStatus, + information: usize, +) -> NtStatus { + if io_status_block + .write_at_offset(0, IoStatusBlock::new(status, information)) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + status +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn relative_children_match_host_parse_contexts() { + use CondrvObject::{ + Connect, CurrentInput, CurrentOutput, Input, Output, Reference, ScreenBuffer, Server, + }; + + for (parent, name, expected) in [ + (Server, r"\Server", Ok(Server)), + (Server, r"\Reference", Ok(Reference)), + (Server, r"\Connect", Err(NtStatus::INVALID_HANDLE)), + (Server, r"\Input", Err(NtStatus::INVALID_DEVICE_STATE)), + (Server, r"\Output", Err(NtStatus::INVALID_DEVICE_STATE)), + (Reference, r"\Server", Ok(Server)), + (Reference, r"\Connect", Ok(Connect)), + (Reference, r"\Input", Ok(Input)), + (Reference, r"\Output", Ok(Output)), + ( + Reference, + r"\Reference", + Err(NtStatus::OBJECT_TYPE_MISMATCH), + ), + (Input, r"\Server", Ok(Server)), + (Input, r"\Input", Ok(Input)), + (Input, r"\Output", Ok(Output)), + (Input, r"\Reference", Err(NtStatus::OBJECT_TYPE_MISMATCH)), + (Input, r"\Connect", Err(NtStatus::INVALID_HANDLE)), + (Output, r"\Server", Ok(Server)), + (Output, r"\Input", Ok(Input)), + (Output, r"\Output", Ok(Output)), + (Output, r"\Reference", Err(NtStatus::OBJECT_TYPE_MISMATCH)), + (Output, r"\Connect", Err(NtStatus::INVALID_HANDLE)), + (Connect, r"\Input", Ok(Input)), + (Connect, r"\Output", Ok(Output)), + (Connect, r"\CurrentIn", Ok(CurrentInput)), + (Connect, r"\CurrentOut", Ok(CurrentOutput)), + (Connect, r"\ScreenBuffer", Ok(ScreenBuffer)), + (Connect, r"\Server", Ok(Server)), + (Connect, r"\Reference", Ok(Reference)), + (Connect, r"\Connect", Err(NtStatus::INVALID_HANDLE)), + (Connect, r"\Bogus", Err(NtStatus::NOT_FOUND)), + ] { + assert_eq!(parent.relative_child(name), expected, "{parent:?} + {name}"); + } + + assert_eq!(Server.relative_child("Reference"), Err(NtStatus::NOT_FOUND)); + assert_eq!(Server.relative_child(r"\Missing"), Err(NtStatus::NOT_FOUND)); + } +} diff --git a/litebox_shim_windows/src/syscalls/event.rs b/litebox_shim_windows/src/syscalls/event.rs new file mode 100644 index 0000000000..4696108b3c --- /dev/null +++ b/litebox_shim_windows/src/syscalls/event.rs @@ -0,0 +1,1172 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT event object syscalls. + +use alloc::sync::{Arc, Weak}; +use core::marker::PhantomData; +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::event::{Events, IOPollable, observer::Observer, polling::Pollee}; +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox::sync::Mutex; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::{ + AccessMask, ObjectAttributes, ObjectAttributesFlags, UnicodeString, read_object_attributes, +}; +use crate::syscalls::Handle; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +pub(crate) enum EventType { + Notification = 0, + Synchronization = 1, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct EventBasicInformation { + event_type: u32, + event_state: i32, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum EventInformationClass { + Basic = 0, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct EventAccess: u32 { + const QUERY_STATE = 0x0001; + const MODIFY_STATE = 0x0002; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY_STATE.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() | Self::MODIFY_STATE.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() | AccessMask::SYNCHRONIZE.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::QUERY_STATE.bits() + | Self::MODIFY_STATE.bits(); + + const _ = !0; + } +} + +impl EventAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +pub(crate) struct EventSubsystem(PhantomData); + +impl FdEnabledSubsystem for EventSubsystem { + type Entry = EventHandleObject; +} + +impl FdEnabledSubsystemEntry for EventHandleObject {} + +impl crate::WindowsHandleSubsystem for EventSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + EventAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct EventHandleObject { + event: Arc>, +} + +pub(crate) struct EventObject { + event_type: EventType, + signaled: Mutex, + pollee: Pollee, +} + +impl EventObject { + fn new(event_type: EventType, initial_state: bool) -> Self { + Self { + event_type, + signaled: Mutex::new(initial_state), + pollee: Pollee::new(), + } + } + + fn set(&self) -> i32 { + let previous = self.replace_state(true); + if previous == 0 { + self.pollee.notify_observers(Events::IN); + } + previous + } + + fn set_boost_priority(&self) -> Result { + if self.event_type != EventType::Synchronization { + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + } + Ok(self.set()) + } + + fn reset(&self) -> i32 { + self.replace_state(false) + } + + fn clear(&self) -> i32 { + self.replace_state(false) + } + + fn pulse(&self) -> i32 { + let previous = self.replace_state(true); + self.pollee.notify_observers(Events::IN); + self.replace_state(false); + previous + } + + fn query(&self) -> EventBasicInformation { + EventBasicInformation { + event_type: self.event_type as u32, + event_state: i32::from(*self.signaled.lock()), + } + } + + pub(crate) fn is_signaled(&self) -> bool { + *self.signaled.lock() + } + + fn replace_state(&self, next: bool) -> i32 { + let mut signaled = self.signaled.lock(); + let previous = i32::from(*signaled); + *signaled = next; + previous + } +} + +impl EventHandleObject { + pub(crate) fn is_signaled(&self) -> bool { + self.event.is_signaled() + } +} + +impl IOPollable for EventObject { + fn register_observer(&self, observer: Weak>, mask: Events) { + self.pollee.register_observer(observer, mask); + } + + fn check_io_events(&self) -> Events { + if *self.signaled.lock() { + Events::IN + } else { + Events::empty() + } + } +} + +struct EventName { + original_path: alloc::string::String, +} + +const EVENT_BASIC_INFORMATION_SIZE_U32: u32 = 8; +const _: () = + assert!(size_of::() == EVENT_BASIC_INFORMATION_SIZE_U32 as usize); + +fn read_event_name( + object_name: usize, + object_attributes: &ObjectAttributes, +) -> Result, NtStatus> { + if object_name == 0 { + if !object_attributes.root_directory.is_null() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + return Ok(None); + } + if !object_attributes.root_directory.is_null() { + return Err(NtStatus::OBJECT_PATH_NOT_FOUND); + } + + let unicode_string = ConstPtr::::from_usize(object_name) + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + if unicode_string.length == 0 || !unicode_string.length.is_multiple_of(2) { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + if unicode_string.buffer == 0 { + return Err(NtStatus::ACCESS_VIOLATION); + } + let original_path = unicode_string.read_string::()?; + if original_path.is_empty() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(Some(EventName { original_path })) +} + +fn read_event_object_attributes( + object_attributes: Option>, + require_name: bool, +) -> Result<(Option, Option), NtStatus> { + let Some(object_attributes_ptr) = object_attributes else { + if require_name { + return Err(NtStatus::INVALID_PARAMETER); + } + return Ok((None, None)); + }; + let object_attributes = read_object_attributes::(object_attributes_ptr)?; + if ObjectAttributesFlags::from_bits_retain(object_attributes.attributes) + .contains(ObjectAttributesFlags::OPENLINK) + { + return Err(NtStatus::INVALID_PARAMETER); + } + let event_name = + read_event_name::(object_attributes.object_name, &object_attributes)?; + if require_name && event_name.is_none() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok((Some(object_attributes), event_name)) +} + +impl Task { + fn insert_event_handle( + &self, + event: Arc>, + granted_access: EventAccess, + ) -> Result { + self.insert_typed_handle::>( + EventHandleObject { event }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_event_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_event(event: EventHandleObject) { + drop(event); + } + + pub(crate) fn sys_nt_create_event( + &self, + event_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + event_type: u32, + initial_state: u8, + ) -> NtStatus { + let Ok(event_type) = EventType::try_from(event_type) else { + return NtStatus::INVALID_PARAMETER; + }; + if let Err(status) = probe_guest_output_preserving_value::(event_handle) { + return status; + } + + let (object_attributes, event_name) = + match read_event_object_attributes::(object_attributes, false) { + Ok(value) => value, + Err(status) => return status, + }; + let granted_access = EventAccess::from_desired_access(desired_access); + + if let Some(event_name) = event_name { + let event = Arc::new(EventObject::new(event_type, initial_state != 0)); + return self.process.object_manager.create_event( + &event_name.original_path, + &event, + |event| { + let Some(object_attributes) = object_attributes else { + return NtStatus::INVALID_PARAMETER; + }; + if !ObjectAttributesFlags::from_bits_retain(object_attributes.attributes) + .contains(ObjectAttributesFlags::OPENIF) + { + return NtStatus::OBJECT_NAME_COLLISION; + } + let Ok(handle) = self.insert_event_handle(event, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if event_handle.write_at_offset(0, handle).is_none() { + self.close_event_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::OBJECT_NAME_EXISTS + }, + || { + let Ok(handle) = self.insert_event_handle(event.clone(), granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if event_handle.write_at_offset(0, handle).is_none() { + self.close_event_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + }, + ); + } + + let event = Arc::new(EventObject::new(event_type, initial_state != 0)); + let Ok(handle) = self.insert_event_handle(event, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if event_handle.write_at_offset(0, handle).is_none() { + self.close_event_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_open_event( + &self, + event_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(event_handle) { + return status; + } + let event_name = match read_event_object_attributes::(object_attributes, true) { + Ok((_, Some(event_name))) => event_name, + Ok((_, None)) => return NtStatus::OBJECT_NAME_INVALID, + Err(status) => return status, + }; + let event = match self + .process + .object_manager + .resolve_event(&event_name.original_path) + { + Ok(event) => event, + Err(status) => return status, + }; + + let Ok(handle) = + self.insert_event_handle(event, EventAccess::from_desired_access(desired_access)) + else { + return NtStatus::QUOTA_EXCEEDED; + }; + if event_handle.write_at_offset(0, handle).is_none() { + self.close_event_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_set_event( + &self, + event_handle: Handle, + previous_state: Option>, + ) -> NtStatus { + if let Some(previous_state) = previous_state + && let Err(status) = probe_guest_output_preserving_value::(previous_state) + { + return status; + } + + match self.modify_event(event_handle, previous_state, |event| Ok(event.set())) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn set_event(&self, event_handle: Handle) -> NtStatus { + match self.modify_event(event_handle, None, |event| Ok(event.set())) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn clear_event(&self, event_handle: Handle) -> Result<(), NtStatus> { + self.modify_event(event_handle, None, |event| Ok(event.clear())) + } + + pub(crate) fn check_event_modify_access(&self, event_handle: Handle) -> Result<(), NtStatus> { + self.require_handle_access::>( + event_handle, + EventAccess::MODIFY_STATE.bits(), + ) + } + + pub(crate) fn sys_nt_reset_event( + &self, + event_handle: Handle, + previous_state: Option>, + ) -> NtStatus { + if let Some(previous_state) = previous_state + && let Err(status) = probe_guest_output_preserving_value::(previous_state) + { + return status; + } + + match self.modify_event(event_handle, previous_state, |event| Ok(event.reset())) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn sys_nt_clear_event(&self, event_handle: Handle) -> NtStatus { + match self.modify_event(event_handle, None, |event| Ok(event.clear())) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn sys_nt_pulse_event( + &self, + event_handle: Handle, + previous_state: Option>, + ) -> NtStatus { + if let Some(previous_state) = previous_state + && let Err(status) = probe_guest_output_preserving_value::(previous_state) + { + return status; + } + + match self.modify_event(event_handle, previous_state, |event| Ok(event.pulse())) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn sys_nt_set_event_boost_priority(&self, event_handle: Handle) -> NtStatus { + match self.modify_event(event_handle, None, EventObject::set_boost_priority) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + pub(crate) fn sys_nt_query_event( + &self, + event_handle: Handle, + event_information_class: u32, + event_information: MutPtr, + event_information_length: u32, + return_length: Option>, + ) -> NtStatus { + let Ok(EventInformationClass::Basic) = + EventInformationClass::try_from(event_information_class) + else { + return NtStatus::INVALID_INFO_CLASS; + }; + if event_information_length as usize != size_of::() { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if let Err(status) = probe_guest_output_preserving_value::(event_information) { + return status; + } + if let Some(return_length) = return_length + && let Err(status) = probe_guest_output_preserving_value::(return_length) + { + return status; + } + + let entry = match self.typed_handle_entry_with_access::>( + event_handle, + EventAccess::QUERY_STATE.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + let info = entry.with_entry(|entry| entry.event.query()); + if event_information.write_at_offset(0, info).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = return_length + && return_length + .write_at_offset(0, EVENT_BASIC_INFORMATION_SIZE_U32) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + fn modify_event( + &self, + event_handle: Handle, + previous_state: Option>, + op: impl FnOnce(&EventObject) -> Result, + ) -> Result<(), NtStatus> { + let entry = self.typed_handle_entry_with_access::>( + event_handle, + EventAccess::MODIFY_STATE.bits(), + )?; + let previous = entry.with_entry(|entry| op(&entry.event))?; + if let Some(previous_state) = previous_state + && previous_state.write_at_offset(0, previous).is_none() + { + return Err(NtStatus::ACCESS_VIOLATION); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use core::mem::size_of; + + use litebox::utils::TruncateExt as _; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::{ObjectAttributes, ObjectAttributesFlags}; + use crate::tests::{ + const_ptr, mut_ptr, object_attributes, test_task, unicode_string, utf16_units, + }; + + const EVENT_QUERY_STATE: u32 = 0x0001; + const EVENT_MODIFY_STATE: u32 = 0x0002; + const EVENT_ALL_ACCESS: u32 = 0x001f_0003; + + fn event_basic_information_size() -> u32 { + size_of::().trunc() + } + + fn object_attributes_size() -> u32 { + size_of::().trunc() + } + + #[test] + fn create_rejects_invalid_event_type() { + let task = test_task(); + let mut handle = Handle::from_raw(usize::MAX); + + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut handle), EVENT_ALL_ACCESS, None, 2, 0), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + } + + #[test] + fn set_reset_clear_pulse_return_previous_state() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut handle), + EVENT_ALL_ACCESS, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + + let mut previous = -1; + assert_eq!( + task.sys_nt_set_event(handle, Some(mut_ptr(&mut previous))), + NtStatus::SUCCESS + ); + assert_eq!(previous, 0); + assert_eq!( + task.sys_nt_set_event(handle, Some(mut_ptr(&mut previous))), + NtStatus::SUCCESS + ); + assert_eq!(previous, 1); + assert_eq!( + task.sys_nt_reset_event(handle, Some(mut_ptr(&mut previous))), + NtStatus::SUCCESS + ); + assert_eq!(previous, 1); + assert_eq!( + task.sys_nt_reset_event(handle, Some(mut_ptr(&mut previous))), + NtStatus::SUCCESS + ); + assert_eq!(previous, 0); + + assert_eq!(task.sys_nt_set_event(handle, None), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_clear_event(handle), NtStatus::SUCCESS); + assert_eq!( + task.sys_nt_pulse_event(handle, Some(mut_ptr(&mut previous))), + NtStatus::SUCCESS + ); + assert_eq!(previous, 0); + } + + #[test] + fn query_event_reports_type_state_and_return_length() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut handle), + EVENT_ALL_ACCESS, + None, + EventType::Synchronization as u32, + 1, + ), + NtStatus::SUCCESS + ); + + let mut info = EventBasicInformation { + event_type: 99, + event_state: -1, + }; + let mut return_length = 0; + assert_eq!( + task.sys_nt_query_event( + handle, + EventInformationClass::Basic as u32, + mut_ptr(&mut info), + event_basic_information_size(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + assert_eq!( + info, + EventBasicInformation { + event_type: EventType::Synchronization as u32, + event_state: 1, + } + ); + assert_eq!(return_length, event_basic_information_size()); + } + + #[test] + fn query_validates_class_and_exact_length() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut handle), + EVENT_ALL_ACCESS, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + let mut info = EventBasicInformation { + event_type: 0, + event_state: 0, + }; + + assert_eq!( + task.sys_nt_query_event( + handle, + 1, + mut_ptr(&mut info), + event_basic_information_size(), + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + assert_eq!( + task.sys_nt_query_event( + handle, + EventInformationClass::Basic as u32, + mut_ptr(&mut info), + event_basic_information_size() - 1, + None, + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + } + + #[test] + fn handle_access_is_enforced() { + let task = test_task(); + let mut query_only = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut query_only), + EVENT_QUERY_STATE, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + let mut modify_only = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut modify_only), + EVENT_MODIFY_STATE, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + + assert_eq!( + task.sys_nt_set_event(query_only, None), + NtStatus::ACCESS_DENIED + ); + + let mut info = EventBasicInformation { + event_type: 0, + event_state: 0, + }; + assert_eq!( + task.sys_nt_query_event( + modify_only, + EventInformationClass::Basic as u32, + mut_ptr(&mut info), + event_basic_information_size(), + None, + ), + NtStatus::ACCESS_DENIED + ); + } + + #[test] + fn named_event_open_shares_state() { + let task = test_task(); + let name_units = utf16_units("\\BaseNamedObjects\\LiteBoxEvent"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + + let mut created = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut created), + EVENT_ALL_ACCESS, + Some(const_ptr(&attrs)), + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_open_event( + mut_ptr(&mut opened), + EVENT_ALL_ACCESS, + Some(const_ptr(&attrs)) + ), + NtStatus::SUCCESS + ); + assert_ne!(created, opened); + + assert_eq!(task.sys_nt_set_event(created, None), NtStatus::SUCCESS); + let mut info = EventBasicInformation { + event_type: 0, + event_state: 0, + }; + assert_eq!( + task.sys_nt_query_event( + opened, + EventInformationClass::Basic as u32, + mut_ptr(&mut info), + event_basic_information_size(), + None, + ), + NtStatus::SUCCESS + ); + assert_eq!(info.event_state, 1); + } + + #[test] + fn open_event_requires_existing_name() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_event(mut_ptr(&mut handle), EVENT_ALL_ACCESS, None), + NtStatus::INVALID_PARAMETER + ); + + let unnamed_attrs = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + assert_eq!( + task.sys_nt_open_event( + mut_ptr(&mut handle), + EVENT_ALL_ACCESS, + Some(const_ptr(&unnamed_attrs)), + ), + NtStatus::OBJECT_NAME_INVALID + ); + } + + #[test] + fn create_openif_existing_named_event_returns_name_exists() { + let task = test_task(); + let name_units = utf16_units("\\BaseNamedObjects\\LiteBoxOpenIf"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let openif_attrs = ObjectAttributes { + attributes: (ObjectAttributesFlags::CASE_INSENSITIVE | ObjectAttributesFlags::OPENIF) + .bits(), + ..attrs + }; + + let mut first = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut first), + EVENT_ALL_ACCESS, + Some(const_ptr(&attrs)), + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + let mut collision = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut collision), + EVENT_ALL_ACCESS, + Some(const_ptr(&attrs)), + EventType::Notification as u32, + 0, + ), + NtStatus::OBJECT_NAME_COLLISION + ); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut collision), + EVENT_MODIFY_STATE, + Some(const_ptr(&openif_attrs)), + EventType::Notification as u32, + 0, + ), + NtStatus::OBJECT_NAME_EXISTS + ); + assert_eq!(task.sys_nt_set_event(collision, None), NtStatus::SUCCESS); + } + + #[test] + fn close_invalidates_event_handle() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut handle), + EVENT_ALL_ACCESS, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert_eq!( + task.sys_nt_set_event(handle, None), + NtStatus::INVALID_HANDLE + ); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + mod host_fidelity { + use core::ffi::c_void; + + use super::*; + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtCreateEvent( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + event_type: u32, + initial_state: u8, + ) -> i32; + fn NtOpenEvent( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + ) -> i32; + fn NtSetEvent(handle: *mut c_void, previous_state: *mut i32) -> i32; + fn NtResetEvent(handle: *mut c_void, previous_state: *mut i32) -> i32; + fn NtClearEvent(handle: *mut c_void) -> i32; + fn NtPulseEvent(handle: *mut c_void, previous_state: *mut i32) -> i32; + fn NtSetEventBoostPriority(handle: *mut c_void) -> i32; + fn NtQueryEvent( + handle: *mut c_void, + event_information_class: u32, + event_information: *mut EventBasicInformation, + event_information_length: u32, + return_length: *mut u32, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + fn assert_status_eq(shim: NtStatus, host: i32) { + assert_eq!(shim.as_raw(), host); + } + + fn close_host_handle(handle: *mut c_void) { + if !handle.is_null() { + // SAFETY: The handle was returned by a successful host ntdll call in this test. + let status = unsafe { NtClose(handle) }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + } + } + + fn host_query_event(handle: *mut c_void) -> (i32, EventBasicInformation, u32) { + let mut info = EventBasicInformation { + event_type: 0, + event_state: 0, + }; + let mut return_length = 0; + // SAFETY: `handle` is a live host event handle and the output pointers reference + // stack locals that are valid for the duration of the call. + let status = unsafe { + NtQueryEvent( + handle, + EventInformationClass::Basic as u32, + &raw mut info, + event_basic_information_size(), + &raw mut return_length, + ) + }; + (status, info, return_length) + } + + fn shim_query_event( + task: &Task, + handle: Handle, + ) -> (NtStatus, EventBasicInformation, u32) { + let mut info = EventBasicInformation { + event_type: 0, + event_state: 0, + }; + let mut return_length = 0; + let status = task.sys_nt_query_event( + handle, + EventInformationClass::Basic as u32, + mut_ptr(&mut info), + event_basic_information_size(), + Some(mut_ptr(&mut return_length)), + ); + (status, info, return_length) + } + + fn assert_queries_match( + task: &Task, + host_handle: *mut c_void, + shim_handle: Handle, + ) { + let (host_status, host_info, host_length) = host_query_event(host_handle); + let (shim_status, shim_info, shim_length) = shim_query_event(task, shim_handle); + assert_status_eq(shim_status, host_status); + assert_eq!(shim_info, host_info); + assert_eq!(shim_length, host_length); + } + + #[test] + fn create_query_reset_matches_host_outputs() { + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer references a live stack local, null object attributes are + // accepted by NtCreateEvent, and the handle is closed before the test returns. + let host_create_status = unsafe { + NtCreateEvent( + &raw mut host_handle, + EVENT_ALL_ACCESS, + core::ptr::null(), + EventType::Notification as u32, + 1, + ) + }; + assert_eq!(host_create_status, NtStatus::SUCCESS.as_raw()); + let (host_query_status, host_info, host_length) = host_query_event(host_handle); + assert_eq!(host_query_status, NtStatus::SUCCESS.as_raw()); + let mut host_previous = 0; + // SAFETY: `host_handle` is a live event handle and `host_previous` is a valid output. + let host_reset_status = unsafe { NtResetEvent(host_handle, &raw mut host_previous) }; + + let task = test_task(); + let mut shim_handle = Handle::default(); + let shim_create_status = task.sys_nt_create_event( + mut_ptr(&mut shim_handle), + EVENT_ALL_ACCESS, + None, + EventType::Notification as u32, + 1, + ); + assert_status_eq(shim_create_status, host_create_status); + let (shim_query_status, shim_info, shim_length) = shim_query_event(&task, shim_handle); + assert_status_eq(shim_query_status, host_query_status); + let mut shim_previous = 0; + let shim_reset_status = + task.sys_nt_reset_event(shim_handle, Some(mut_ptr(&mut shim_previous))); + + assert_status_eq(shim_reset_status, host_reset_status); + assert_eq!(shim_info, host_info); + assert_eq!(shim_length, host_length); + assert_eq!(shim_previous, host_previous); + + close_host_handle(host_handle); + } + + #[test] + fn set_clear_pulse_and_boost_match_host_state() { + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer references a live stack local, null object attributes are + // accepted by NtCreateEvent, and the handle is closed before the test returns. + let status = unsafe { + NtCreateEvent( + &raw mut host_handle, + EVENT_ALL_ACCESS, + core::ptr::null(), + EventType::Notification as u32, + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + + let task = test_task(); + let mut shim_handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut shim_handle), + EVENT_ALL_ACCESS, + None, + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + + let mut host_previous = -1; + let mut shim_previous = -1; + // SAFETY: `host_handle` is a live event handle and `host_previous` is a valid output. + let host_status = unsafe { NtSetEvent(host_handle, &raw mut host_previous) }; + let shim_status = task.sys_nt_set_event(shim_handle, Some(mut_ptr(&mut shim_previous))); + assert_status_eq(shim_status, host_status); + assert_eq!(shim_previous, host_previous); + assert_queries_match(&task, host_handle, shim_handle); + + // SAFETY: `host_handle` is a live event handle. + let host_status = unsafe { NtClearEvent(host_handle) }; + let shim_status = task.sys_nt_clear_event(shim_handle); + assert_status_eq(shim_status, host_status); + assert_queries_match(&task, host_handle, shim_handle); + + host_previous = -1; + shim_previous = -1; + // SAFETY: `host_handle` is a live event handle and `host_previous` is a valid output. + let host_status = unsafe { NtPulseEvent(host_handle, &raw mut host_previous) }; + let shim_status = + task.sys_nt_pulse_event(shim_handle, Some(mut_ptr(&mut shim_previous))); + assert_status_eq(shim_status, host_status); + assert_eq!(shim_previous, host_previous); + assert_queries_match(&task, host_handle, shim_handle); + + // SAFETY: `host_handle` is a live event handle. + let host_status = unsafe { NtSetEventBoostPriority(host_handle) }; + let shim_status = task.sys_nt_set_event_boost_priority(shim_handle); + assert_status_eq(shim_status, host_status); + assert_queries_match(&task, host_handle, shim_handle); + + close_host_handle(host_handle); + } + + #[test] + fn boost_priority_sets_synchronization_event() { + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer references a live stack local, null object attributes are + // accepted by NtCreateEvent, and the handle is closed before the test returns. + let status = unsafe { + NtCreateEvent( + &raw mut host_handle, + EVENT_ALL_ACCESS, + core::ptr::null(), + EventType::Synchronization as u32, + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + + let task = test_task(); + let mut shim_handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut shim_handle), + EVENT_ALL_ACCESS, + None, + EventType::Synchronization as u32, + 0, + ), + NtStatus::SUCCESS + ); + + // SAFETY: `host_handle` is a live synchronization event handle. + let host_status = unsafe { NtSetEventBoostPriority(host_handle) }; + let shim_status = task.sys_nt_set_event_boost_priority(shim_handle); + assert_status_eq(shim_status, host_status); + assert_queries_match(&task, host_handle, shim_handle); + + close_host_handle(host_handle); + } + + #[test] + fn named_open_matches_host_state_sharing() { + let unique = 0u8; + let name_units = utf16_units(&alloc::format!( + r"\BaseNamedObjects\LiteBoxEventFidelity{:p}", + &unique, + )); + let name = unicode_string(&name_units); + let attributes = + object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + + let mut host_created = core::ptr::null_mut(); + let mut host_opened = core::ptr::null_mut(); + // SAFETY: Pointers reference live stack locals and ObjectAttributes points to a live + // UnicodeString naming a BaseNamedObjects event for the duration of the calls. + let host_create_status = unsafe { + NtCreateEvent( + &raw mut host_created, + EVENT_ALL_ACCESS, + &raw const attributes, + EventType::Notification as u32, + 0, + ) + }; + assert_eq!(host_create_status, NtStatus::SUCCESS.as_raw()); + // SAFETY: Same live ObjectAttributes as above, and output pointer is valid. + let host_open_status = unsafe { + NtOpenEvent( + &raw mut host_opened, + EVENT_MODIFY_STATE, + &raw const attributes, + ) + }; + assert_eq!(host_open_status, NtStatus::SUCCESS.as_raw()); + + let task = test_task(); + let mut shim_created = Handle::default(); + let shim_create_status = task.sys_nt_create_event( + mut_ptr(&mut shim_created), + EVENT_ALL_ACCESS, + Some(const_ptr(&attributes)), + EventType::Notification as u32, + 0, + ); + assert_status_eq(shim_create_status, host_create_status); + let mut shim_opened = Handle::default(); + let shim_open_status = task.sys_nt_open_event( + mut_ptr(&mut shim_opened), + EVENT_MODIFY_STATE, + Some(const_ptr(&attributes)), + ); + assert_status_eq(shim_open_status, host_open_status); + + // SAFETY: `host_opened` is a live event handle opened with EVENT_MODIFY_STATE. + let host_set_status = unsafe { NtSetEvent(host_opened, core::ptr::null_mut()) }; + let shim_set_status = task.sys_nt_set_event(shim_opened, None); + assert_status_eq(shim_set_status, host_set_status); + assert_queries_match(&task, host_created, shim_created); + + close_host_handle(host_opened); + close_host_handle(host_created); + } + } +} diff --git a/litebox_shim_windows/src/syscalls/file.rs b/litebox_shim_windows/src/syscalls/file.rs new file mode 100644 index 0000000000..bdbd5b036a --- /dev/null +++ b/litebox_shim_windows/src/syscalls/file.rs @@ -0,0 +1,3044 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::string::String; +use alloc::sync::Arc; +use core::marker::PhantomData; +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry, TypedFd}; +use litebox::fs::errors::{FileStatusError, MkdirError, OpenError, PathError, WriteError}; +use litebox::fs::{FileType, Mode, OFlags, SeekWhence}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::{ + AccessMask, IoStatusBlock, ObjectAttributes, UnicodeString, read_object_attributes, +}; +use crate::syscalls::Handle; +use crate::syscalls::condrv::{self, CondrvObject, CondrvStreamDirection, CondrvStreamObject}; +use crate::syscalls::file_path::{FilePathResolver, FilePathRoot, FileTarget}; +use crate::{ + ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value, raw_handle_entry, +}; + +const FILE_ATTRIBUTE_READONLY: u32 = 0x0000_0001; + +const FILE_SHARE_READ: u32 = 0x0000_0001; +const FILE_SHARE_WRITE: u32 = 0x0000_0002; +const FILE_SHARE_DELETE: u32 = 0x0000_0004; + +/// Append at the current end of file +const FILE_WRITE_TO_END_OF_FILE: i64 = -1; + +/// Use the file object's current position +const FILE_USE_FILE_POINTER_POSITION: i64 = -2; + +// These names and values are Windows ABI constants from WDK headers; Wine's +// regular file/directory branch and ReactOS' filesystem device query path use +// the same FILE_DEVICE_* and FILE_DEVICE_IS_MOUNTED vocabulary. +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum FileDeviceType { + Disk = 0x0000_0007, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct FileDeviceCharacteristics: u32 { + const IS_MOUNTED = 0x0000_0020; + const _ = !0; + } +} + +#[repr(usize)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum FileCreateInformation { + /// An existing file was deleted and a new file was created in its place. + Superseded = 0, + /// An existing file was opened. + Opened = 1, + Created = 2, + /// An existing file was overwritten. + Overwritten = 3, + Exists = 4, + DoesNotExist = 5, +} + +// TODO: NtSetVolumeInformationFile and sibling query classes +// (FileFsVolumeInformation=1, FileFsSizeInformation=3, FileFsAttributeInformation=5) +// are deferred until a guest exercises them; each needs host-grounded volume +// metadata LiteBox does not model yet. Add the variant and match arm when that boundary lands. +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum FsInformationClass { + FileFsDeviceInformation = 4, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +struct FileFsDeviceInformation { + device_type: u32, + characteristics: u32, +} + +const _: () = assert!(size_of::() == 8); + +pub(crate) struct FileObjectSubsystem(PhantomData); + +impl FdEnabledSubsystem for FileObjectSubsystem { + type Entry = FileObject; +} + +impl FdEnabledSubsystemEntry for FileObject {} + +impl crate::WindowsHandleSubsystem for FileObjectSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + FileAccess::from_desired_access(desired_access).bits() + } + + fn resolve_duplicate_access(entry: &Self::Entry, desired_access: u32) -> Result { + let maximum_allowed = desired_access & AccessMask::MAXIMUM_ALLOWED.bits() != 0; + let explicit_access = + FileAccess::from_desired_access(desired_access & !AccessMask::MAXIMUM_ALLOWED.bits()); + if !entry.create_time_access.contains(explicit_access) { + return Err(NtStatus::ACCESS_DENIED); + } + Ok(if maximum_allowed { + // TODO(dacl-access-check): Replace this original-open ceiling with a token and + // security-descriptor access check when the shim models DACLs. + entry.create_time_access.bits() + } else { + explicit_access.bits() + }) + } +} + +pub(crate) struct FileObject { + path: String, + backing: FileObjectBacking, + create_time_access: FileAccess, + share_access: FileShareAccess, + create_options: FileCreateOptions, +} + +enum FileObjectBacking { + Filesystem { + fd: TypedFd, + is_directory: bool, + }, + CondrvStream { + object: CondrvObject, + stream_object: Arc, + fd: TypedFd, + }, + CondrvControl(CondrvObject), +} + +#[derive(Clone, Copy)] +enum FileSharingIdentity<'a> { + Path(&'a str), + // TODO(condrv-share-access): native CONIN$/CONOUT$ permits multiple share-access-zero opens + // of the same bound object; determine which ConDrv opens ignore sharing before enforcing it. + CondrvObject(u64), +} + +impl FileSharingIdentity<'_> { + fn matches(self, file: &FileObject) -> bool { + match self { + Self::Path(path) => file.condrv_stream_object_id().is_none() && file.path == path, + Self::CondrvObject(object_id) => file.condrv_stream_object_id() == Some(object_id), + } + } +} + +impl FileObject { + fn condrv_object(&self) -> Option { + match self.backing { + FileObjectBacking::CondrvStream { object, .. } + | FileObjectBacking::CondrvControl(object) => Some(object), + FileObjectBacking::Filesystem { .. } => None, + } + } + + fn condrv_stream_object_id(&self) -> Option { + match &self.backing { + FileObjectBacking::CondrvStream { stream_object, .. } => Some(stream_object.id()), + FileObjectBacking::Filesystem { .. } | FileObjectBacking::CondrvControl(_) => None, + } + } + + fn is_directory(&self) -> bool { + matches!( + self.backing, + FileObjectBacking::Filesystem { + is_directory: true, + .. + } + ) + } +} + +bitflags::bitflags! { + /// File object `ACCESS_MASK` rights accepted by `NtOpenFile`/`NtCreateFile`. + /// + /// Generic-right mappings and create/open disposition behavior follow + /// Microsoft Learn's `NtCreateFile` documentation. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct FileAccess: u32 { + const READ_DATA = 0x0001; + const LIST_DIRECTORY = Self::READ_DATA.bits(); + const WRITE_DATA = 0x0002; + const ADD_FILE = Self::WRITE_DATA.bits(); + const APPEND_DATA = 0x0004; + const ADD_SUBDIRECTORY = Self::APPEND_DATA.bits(); + const READ_EA = 0x0008; + const WRITE_EA = 0x0010; + const EXECUTE = 0x0020; + const TRAVERSE = Self::EXECUTE.bits(); + const DELETE_CHILD = 0x0040; + const READ_ATTRIBUTES = 0x0080; + const WRITE_ATTRIBUTES = 0x0100; + const DELETE = AccessMask::DELETE.bits(); + const SYNCHRONIZE = AccessMask::SYNCHRONIZE.bits(); + + const GENERIC_READ_EXPANSION = AccessMask::STANDARD_RIGHTS_READ.bits() + | Self::READ_DATA.bits() + | Self::READ_ATTRIBUTES.bits() + | Self::READ_EA.bits() + | Self::SYNCHRONIZE.bits(); + const GENERIC_WRITE_EXPANSION = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | Self::WRITE_DATA.bits() + | Self::WRITE_ATTRIBUTES.bits() + | Self::WRITE_EA.bits() + | Self::APPEND_DATA.bits() + | Self::SYNCHRONIZE.bits(); + const GENERIC_EXECUTE_EXPANSION = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() + | Self::EXECUTE.bits() + | Self::READ_ATTRIBUTES.bits() + | Self::SYNCHRONIZE.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::READ_DATA.bits() + | Self::WRITE_DATA.bits() + | Self::APPEND_DATA.bits() + | Self::READ_EA.bits() + | Self::WRITE_EA.bits() + | Self::EXECUTE.bits() + | Self::DELETE_CHILD.bits() + | Self::READ_ATTRIBUTES.bits() + | Self::WRITE_ATTRIBUTES.bits(); + + const FS_READ_ACCESS = Self::READ_DATA.bits() + | Self::READ_EA.bits() + | Self::READ_ATTRIBUTES.bits() + | Self::EXECUTE.bits(); + const FS_WRITE_ACCESS = Self::WRITE_DATA.bits() + | Self::APPEND_DATA.bits() + | Self::WRITE_EA.bits() + | Self::WRITE_ATTRIBUTES.bits() + | Self::DELETE.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits(); + + const SHARE_READ_ACCESS = Self::READ_DATA.bits() + | Self::READ_EA.bits() + | Self::READ_ATTRIBUTES.bits() + | Self::EXECUTE.bits(); + const SHARE_WRITE_ACCESS = Self::WRITE_DATA.bits() + | Self::APPEND_DATA.bits() + | Self::WRITE_EA.bits() + | Self::WRITE_ATTRIBUTES.bits(); + const SHARE_DELETE_ACCESS = Self::DELETE.bits() + | Self::DELETE_CHILD.bits(); + + const _ = !0; + } +} + +impl FileAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::GENERIC_READ_EXPANSION.bits(), + Self::GENERIC_WRITE_EXPANSION.bits(), + Self::GENERIC_EXECUTE_EXPANSION.bits(), + Self::ALL_ACCESS.bits(), + )) + } + + fn open_flags( + self, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, + ) -> OFlags { + let wants_read = self.intersects(Self::FS_READ_ACCESS); + let wants_write = self.intersects(Self::FS_WRITE_ACCESS) + || matches!( + create_disposition, + CreateDisposition::Supersede + | CreateDisposition::Overwrite + | CreateDisposition::OverwriteIf + ); + + let mut flags = match (wants_read, wants_write) { + (true, true) => OFlags::RDWR, + (false, true) => OFlags::WRONLY, + _ => OFlags::RDONLY, + }; + + match create_disposition { + CreateDisposition::Overwrite => { + flags.insert(OFlags::TRUNC); + } + CreateDisposition::Supersede | CreateDisposition::OverwriteIf => { + flags.insert(OFlags::CREAT | OFlags::TRUNC); + } + CreateDisposition::Create => flags.insert(OFlags::CREAT | OFlags::EXCL), + CreateDisposition::OpenIf => flags.insert(OFlags::CREAT), + CreateDisposition::Open => {} + } + + if create_options.contains(FileCreateOptions::DIRECTORY_FILE) { + flags.insert(OFlags::DIRECTORY); + } + if create_options.contains(FileCreateOptions::NON_DIRECTORY_FILE) { + flags.insert(OFlags::NOFOLLOW); + } + + flags + } + + fn conflicts_with_share(self, share_access: FileShareAccess) -> bool { + self.intersects(Self::SHARE_READ_ACCESS) && !share_access.contains(FileShareAccess::READ) + || self.intersects(Self::SHARE_WRITE_ACCESS) + && !share_access.contains(FileShareAccess::WRITE) + || self.intersects(Self::SHARE_DELETE_ACCESS) + && !share_access.contains(FileShareAccess::DELETE) + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct FileShareAccess: u32 { + const READ = FILE_SHARE_READ; + const WRITE = FILE_SHARE_WRITE; + const DELETE = FILE_SHARE_DELETE; + const _ = !0; + } +} + +impl FileShareAccess { + const VALID_BITS: u32 = FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE; + + fn from_share_access(share_access: u32) -> Result { + if share_access & !Self::VALID_BITS != 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + Ok(Self::from_bits_retain(share_access)) + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct FileCreateOptions: u32 { + const DIRECTORY_FILE = 0x0000_0001; + const WRITE_THROUGH = 0x0000_0002; + const SEQUENTIAL_ONLY = 0x0000_0004; + const NO_INTERMEDIATE_BUFFERING = 0x0000_0008; + const SYNCHRONOUS_IO_ALERT = 0x0000_0010; + const SYNCHRONOUS_IO_NONALERT = 0x0000_0020; + const NON_DIRECTORY_FILE = 0x0000_0040; + const CREATE_TREE_CONNECTION = 0x0000_0080; + const COMPLETE_IF_OPLOCKED = 0x0000_0100; + const NO_EA_KNOWLEDGE = 0x0000_0200; + const OPEN_REMOTE_INSTANCE = 0x0000_0400; + const RANDOM_ACCESS = 0x0000_0800; + const DELETE_ON_CLOSE = 0x0000_1000; + const OPEN_BY_FILE_ID = 0x0000_2000; + const OPEN_FOR_BACKUP_INTENT = 0x0000_4000; + const NO_COMPRESSION = 0x0000_8000; + const OPEN_REQUIRING_OPLOCK = 0x0001_0000; + const DISALLOW_EXCLUSIVE = 0x0002_0000; + const SESSION_AWARE = 0x0004_0000; + const RESERVE_OPFILTER = 0x0010_0000; + const OPEN_REPARSE_POINT = 0x0020_0000; + const OPEN_NO_RECALL = 0x0040_0000; + const OPEN_FOR_FREE_SPACE_QUERY = 0x0080_0000; + const CONTAINS_EXTENDED_CREATE_INFORMATION = 0x1000_0000; + + const _ = !0; + } +} + +impl FileCreateOptions { + const SYNCHRONOUS_IO: Self = Self::SYNCHRONOUS_IO_ALERT.union(Self::SYNCHRONOUS_IO_NONALERT); + + const DIRECTORY_COMPATIBLE: Self = Self::DIRECTORY_FILE + .union(Self::SYNCHRONOUS_IO_ALERT) + .union(Self::SYNCHRONOUS_IO_NONALERT) + .union(Self::WRITE_THROUGH) + .union(Self::COMPLETE_IF_OPLOCKED) + .union(Self::OPEN_FOR_BACKUP_INTENT) + .union(Self::DELETE_ON_CLOSE) + .union(Self::OPEN_BY_FILE_ID) + .union(Self::NO_COMPRESSION) + .union(Self::OPEN_REPARSE_POINT) + .union(Self::OPEN_FOR_FREE_SPACE_QUERY); +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum CreateDisposition { + Supersede = 0, + Create = 1, + Open = 2, + OpenIf = 3, + Overwrite = 4, + OverwriteIf = 5, +} + +impl CreateDisposition { + fn success_information(self, existed_before_open: bool) -> FileCreateInformation { + match (self, existed_before_open) { + (Self::Supersede, true) => FileCreateInformation::Superseded, + (Self::Supersede | Self::Create | Self::OpenIf | Self::OverwriteIf, false) => { + FileCreateInformation::Created + } + (Self::Overwrite | Self::OverwriteIf, true) => FileCreateInformation::Overwritten, + _ => FileCreateInformation::Opened, + } + } +} + +impl Task { + fn file_entry( + &self, + handle: Handle, + ) -> Result>, NtStatus> { + raw_handle_entry::>( + &self.global.litebox, + &self.process.handles, + handle, + ) + .ok_or(NtStatus::INVALID_HANDLE) + } + + fn insert_file_handle(&self, file: FileObject) -> Result { + let granted_access = file.create_time_access.bits(); + self.insert_typed_handle::>(file, granted_access, |file| { + self.close_file(file); + }) + } + + pub(crate) fn close_file_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, |file| self.close_file(file)); + } + + pub(crate) fn close_file(&self, file: FileObject) { + match file.backing { + FileObjectBacking::Filesystem { fd, is_directory } => { + let _ = self.fs.close(&fd); + if file + .create_options + .contains(FileCreateOptions::DELETE_ON_CLOSE) + { + if is_directory { + let _ = self.fs.rmdir(&file.path); + } else { + let _ = self.fs.unlink(&file.path); + } + } + } + FileObjectBacking::CondrvStream { fd, .. } => { + let _ = self.fs.close(&fd); + } + FileObjectBacking::CondrvControl(_) => {} + } + } + + pub(crate) fn sys_nt_open_file( + &self, + file_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + io_status_block: MutPtr, + share_access: u32, + open_options: u32, + ) -> NtStatus { + let Some(object_attributes) = object_attributes else { + return NtStatus::INVALID_PARAMETER; + }; + let object_attributes = match read_object_attributes::(object_attributes) { + Ok(object_attributes) => object_attributes, + Err(status) => return status, + }; + if let Err(status) = probe_file_outputs::(file_handle, io_status_block) { + return status; + } + let result = self.do_nt_create_file( + desired_access, + object_attributes, + io_status_block, + FILE_ATTRIBUTE_READONLY, + share_access, + CreateDisposition::Open, + open_options, + None, + 0, + ); + write_file_result::(file_handle, io_status_block, result, |handle| { + self.close_file_handle(handle); + }) + } + + #[expect( + clippy::too_many_arguments, + reason = "NtCreateFile has eleven ABI parameters; keeping the syscall handler aligned with that shape avoids argument reshuffling bugs" + )] + pub(crate) fn sys_nt_create_file( + &self, + file_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + io_status_block: MutPtr, + _allocation_size: Option>, + file_attributes: u32, + share_access: u32, + create_disposition: u32, + create_options: u32, + ea_buffer: Option>, + ea_length: u32, + ) -> NtStatus { + let Some(object_attributes) = object_attributes else { + return NtStatus::INVALID_PARAMETER; + }; + let object_attributes = match read_object_attributes::(object_attributes) { + Ok(object_attributes) => object_attributes, + Err(status) => return status, + }; + let Ok(create_disposition) = CreateDisposition::try_from(create_disposition) else { + return NtStatus::INVALID_PARAMETER; + }; + if let Err(status) = probe_file_outputs::(file_handle, io_status_block) { + return status; + } + let result = self.do_nt_create_file( + desired_access, + object_attributes, + io_status_block, + file_attributes, + share_access, + create_disposition, + create_options, + ea_buffer, + ea_length, + ); + write_file_result::(file_handle, io_status_block, result, |handle| { + self.close_file_handle(handle); + }) + } + + #[expect( + clippy::too_many_arguments, + reason = "NtWriteFile has nine ABI parameters; keeping them explicit preserves syscall ordering" + )] + pub(crate) fn sys_nt_write_file( + &self, + file_handle: Handle, + event: Handle, + apc_routine: Option>, + apc_context: Option>, + io_status_block: MutPtr, + buffer: ConstPtr, + length: u32, + byte_offset: Option>, + key: Option>, + ) -> NtStatus { + if probe_guest_output_preserving_value::(io_status_block).is_err() + { + return NtStatus::ACCESS_VIOLATION; + } + let Some(buffer) = buffer.to_owned_slice(length as usize) else { + return NtStatus::ACCESS_VIOLATION; + }; + if !event.is_null() + && let Err(status) = self.check_event_modify_access(event) + { + return status; + } + let offset = match byte_offset { + Some(byte_offset) => match byte_offset.read_at_offset(0) { + Some(FILE_USE_FILE_POINTER_POSITION) => None, + Some(FILE_WRITE_TO_END_OF_FILE) => { + let file = match self.file_entry(file_handle) { + Ok(file) => file, + Err(status) => return status, + }; + match file.with_entry(|file| self.fs.file_status(&file.path)) { + Ok(status) => Some(status.size), + Err(error) => return map_file_status_error(error), + } + } + Some(offset) if offset >= 0 => match usize::try_from(offset) { + Ok(offset) => Some(offset), + Err(_) => return NtStatus::INVALID_PARAMETER, + }, + Some(_) => return NtStatus::INVALID_PARAMETER, + None => return NtStatus::ACCESS_VIOLATION, + }, + None => None, + }; + if let Some(key) = key { + let Some(key) = key.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + litebox_util_log::debug!( + file_handle = file_handle.as_raw(), + key = key; + "Ignoring NtWriteFile byte-range lock key; byte-range locking is not supported yet" + ); + } + + let file = match self.file_entry(file_handle) { + Ok(file) => file, + Err(status) => return status, + }; + if !event.is_null() + && let Err(status) = self.clear_event(event) + { + return status; + } + if apc_routine.is_some() || apc_context.is_some() { + litebox_util_log::debug!( + file_handle = file_handle.as_raw(); + "Ignoring NtWriteFile APC completion arguments for synchronous completion" + ); + } + let result = file.with_entry(|file| match &file.backing { + FileObjectBacking::Filesystem { fd, is_directory } => { + if *is_directory { + return Err(WriteError::NotAFile); + } + let written = self.fs.write(fd, &buffer, offset)?; + // A positional write leaves the backing file offset untouched, but NT advances a + // synchronous file object's position past the end of every write, including + // explicit-offset and append writes. Asynchronous handles keep their position. + if let Some(offset) = offset + && file + .create_options + .intersects(FileCreateOptions::SYNCHRONOUS_IO) + { + let _ = self.fs.seek( + fd, + (offset + written).cast_signed(), + SeekWhence::RelativeToBeginning, + ); + } + Ok(written) + } + FileObjectBacking::CondrvStream { fd, .. } => self.fs.write(fd, &buffer, offset), + FileObjectBacking::CondrvControl(_) => Err(WriteError::NotAFile), + }); + let (status, information) = match result { + Ok(written) => (NtStatus::SUCCESS, written), + Err(WriteError::ClosedFd) => (NtStatus::INVALID_HANDLE, 0), + Err(WriteError::NotForWriting) => (NtStatus::ACCESS_DENIED, 0), + Err(WriteError::NotAFile) => (NtStatus::INVALID_DEVICE_REQUEST, 0), + Err(_) => (NtStatus::UNSUCCESSFUL, 0), + }; + if io_status_block + .write_at_offset(0, IoStatusBlock::new(status, information)) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if !event.is_null() { + let event_status = self.set_event(event); + if event_status != NtStatus::SUCCESS { + return event_status; + } + } + status + } + + pub(crate) fn sys_nt_query_volume_information_file( + &self, + file_handle: Handle, + io_status_block: MutPtr, + fs_information: MutPtr, + length: u32, + fs_information_class: u32, + ) -> NtStatus { + let Ok(fs_information_class) = FsInformationClass::try_from(fs_information_class) else { + litebox_util_log::debug!( + fs_information_class = fs_information_class; + "Unsupported NtQueryVolumeInformationFile class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match fs_information_class { + FsInformationClass::FileFsDeviceInformation => self.write_file_fs_device_information( + file_handle, + io_status_block, + fs_information, + length, + ), + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + file_handle = file_handle.as_raw(), + length = length, + fs_information_class:? = fs_information_class; + "Handled NtQueryVolumeInformationFile syscall" + ); + } + + status + } + + #[expect( + clippy::too_many_arguments, + reason = "NtDeviceIoControlFile has ten ABI parameters; keeping them explicit preserves syscall ordering" + )] + pub(crate) fn sys_nt_device_io_control_file( + &self, + file_handle: Handle, + event: Handle, + apc_routine: Option>, + apc_context: Option>, + io_status_block: MutPtr, + io_control_code: u32, + input_buffer: Option>, + input_buffer_length: u32, + output_buffer: Option>, + output_buffer_length: u32, + ) -> NtStatus { + if let Err(status) = + probe_guest_output_preserving_value::(io_status_block) + { + return status; + } + if !event.is_null() + && let Err(status) = self.check_event_modify_access(event) + { + return status; + } + + let condrv_object = match self.file_entry(file_handle) { + Ok(entry) => entry.with_entry(FileObject::condrv_object), + Err(status) => return status, + }; + if !event.is_null() + && let Err(status) = self.clear_event(event) + { + return status; + } + let Some(condrv_object) = condrv_object else { + litebox_util_log::debug!( + file_handle = file_handle.as_raw(), + io_control_code:% = format_args!("{io_control_code:#x}"); + "Unsupported NtDeviceIoControlFile for non-ConDrv file handle" + ); + return NtStatus::INVALID_DEVICE_REQUEST; + }; + if apc_routine.is_some() || apc_context.is_some() { + litebox_util_log::debug!( + file_handle = file_handle.as_raw(), + apc_context = apc_context.map_or(0, |context| context.as_usize()); + "Ignoring NtDeviceIoControlFile APC completion arguments for synchronous completion" + ); + } + let status = condrv::handle_ioctl::( + condrv_object, + io_status_block, + io_control_code, + input_buffer, + input_buffer_length, + output_buffer, + output_buffer_length, + ); + if !event.is_null() { + let event_status = self.set_event(event); + if event_status != NtStatus::SUCCESS { + return event_status; + } + } + status + } + + fn write_file_fs_device_information( + &self, + file_handle: Handle, + io_status_block: MutPtr, + fs_information: MutPtr, + length: u32, + ) -> NtStatus { + if length < u32::try_from(size_of::()).unwrap() { + return NtStatus::INFO_LENGTH_MISMATCH; + } + + let fs_information = + MutPtr::::from_usize(fs_information.as_usize()); + if probe_guest_output_preserving_value::(io_status_block).is_err() + || probe_guest_output_preserving_value::( + fs_information, + ) + .is_err() + { + return NtStatus::ACCESS_VIOLATION; + } + + let Ok(_file) = self.file_entry(file_handle) else { + return NtStatus::INVALID_HANDLE; + }; + + let info = FileFsDeviceInformation { + device_type: FileDeviceType::Disk as u32, + characteristics: FileDeviceCharacteristics::IS_MOUNTED.bits(), + }; + if fs_information.write_at_offset(0, info).is_none() + || io_status_block + .write_at_offset( + 0, + IoStatusBlock::new(NtStatus::SUCCESS, size_of::()), + ) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + // Microsoft Learn documents `NtCreateFile` as the common create/open primitive, + // with `NtOpenFile` being its open-existing subset. + #[expect( + clippy::too_many_arguments, + reason = "This helper carries the parsed NtCreateFile ABI fields through one shared NtOpenFile/NtCreateFile path" + )] + fn do_nt_create_file( + &self, + desired_access: u32, + object_attributes: ObjectAttributes, + io_status_block: MutPtr, + file_attributes: u32, + share_access: u32, + create_disposition: CreateDisposition, + create_options: u32, + ea_buffer: Option>, + ea_length: u32, + ) -> Result<(Handle, FileCreateInformation), NtStatus> { + if io_status_block.as_usize() == 0 { + return Err(NtStatus::ACCESS_VIOLATION); + } + if object_attributes.object_name == 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + let desired_access = FileAccess::from_desired_access(desired_access); + let create_options = FileCreateOptions::from_bits_retain(create_options); + validate_create_options(desired_access, create_disposition, create_options)?; + + let share_access = FileShareAccess::from_share_access(share_access)?; + let (file, information) = match self.object_attributes_to_file_target(object_attributes)? { + FileTarget::Filesystem(path) => { + if ea_buffer.is_some() || ea_length != 0 { + return Err(NtStatus::EAS_NOT_SUPPORTED); + } + self.open_filesystem_target( + path, + desired_access, + share_access, + create_disposition, + create_options, + file_attributes, + ) + } + FileTarget::Condrv(object) => self.open_condrv_target( + object, + desired_access, + share_access, + create_disposition, + create_options, + ea_buffer, + ea_length, + ), + }?; + let handle = self.insert_file_handle(file)?; + Ok((handle, information)) + } + + fn open_filesystem_target( + &self, + path: String, + desired_access: FileAccess, + share_access: FileShareAccess, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, + file_attributes: u32, + ) -> Result<(FileObject, FileCreateInformation), NtStatus> { + self.check_file_sharing( + FileSharingIdentity::Path(&path), + desired_access, + share_access, + )?; + if create_options.contains(FileCreateOptions::DIRECTORY_FILE) { + return self.open_or_create_directory( + &path, + desired_access, + share_access, + create_disposition, + create_options, + file_attributes, + ); + } + + let (fd, is_directory, information) = self.open_backing_fd( + &path, + desired_access, + create_disposition, + create_options, + create_mode(file_attributes), + )?; + Ok(( + FileObject { + path, + backing: FileObjectBacking::Filesystem { fd, is_directory }, + create_time_access: desired_access, + share_access, + create_options, + }, + information, + )) + } + + #[expect( + clippy::too_many_arguments, + reason = "ConDrv creation validates the parsed NtCreateFile fields at the device boundary" + )] + fn open_condrv_target( + &self, + object: CondrvObject, + desired_access: FileAccess, + share_access: FileShareAccess, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, + ea_buffer: Option>, + ea_length: u32, + ) -> Result<(FileObject, FileCreateInformation), NtStatus> { + if object == CondrvObject::Connect { + condrv::validate_connect_server_ea::(ea_buffer, ea_length)?; + } else if ea_buffer.is_some() || ea_length != 0 { + return Err(NtStatus::EAS_NOT_SUPPORTED); + } + if create_options.contains(FileCreateOptions::DIRECTORY_FILE) { + return Err(NtStatus::NOT_A_DIRECTORY); + } + + let path = String::from(object.handle_path()); + let (backing, information) = if let Some(direction) = object.stream_direction() { + let stream_object = self.process.condrv_console.open_stream(object)?; + self.check_file_sharing( + FileSharingIdentity::CondrvObject(stream_object.id()), + desired_access, + share_access, + )?; + let backing_access = match direction { + CondrvStreamDirection::Input => FileAccess::READ_DATA, + CondrvStreamDirection::Output => FileAccess::WRITE_DATA, + }; + let (fd, _, information) = self.open_backing_fd( + &path, + backing_access, + create_disposition, + create_options, + Mode::empty(), + )?; + ( + FileObjectBacking::CondrvStream { + object, + stream_object, + fd, + }, + information, + ) + } else { + self.check_file_sharing( + FileSharingIdentity::Path(&path), + desired_access, + share_access, + )?; + ( + FileObjectBacking::CondrvControl(object), + FileCreateInformation::Opened, + ) + }; + Ok(( + FileObject { + path, + backing, + create_time_access: desired_access, + share_access, + create_options, + }, + information, + )) + } + + fn open_backing_fd( + &self, + path: &str, + desired_access: FileAccess, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, + mode: Mode, + ) -> Result<(TypedFd, bool, FileCreateInformation), NtStatus> { + let existed_before_open = self.fs.file_status(path).is_ok(); + if create_disposition == CreateDisposition::Supersede + && existed_before_open + && !desired_access.contains(FileAccess::DELETE) + { + return Err(NtStatus::ACCESS_DENIED); + } + let flags = desired_access.open_flags(create_disposition, create_options); + let fd = self + .fs + .open(path, flags, mode) + .map_err(|error| map_open_error(error, create_disposition))?; + let file_status = match self.fs.fd_file_status(&fd) { + Ok(file_status) => file_status, + Err(error) => { + let _ = self.fs.close(&fd); + return Err(map_file_status_error(error)); + } + }; + if create_options.contains(FileCreateOptions::NON_DIRECTORY_FILE) + && file_status.file_type == FileType::Directory + { + let _ = self.fs.close(&fd); + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + } + let information = create_disposition.success_information(existed_before_open); + Ok(( + fd, + file_status.file_type == FileType::Directory, + information, + )) + } + + fn open_or_create_directory( + &self, + path: &str, + desired_access: FileAccess, + share_access: FileShareAccess, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, + file_attributes: u32, + ) -> Result<(FileObject, FileCreateInformation), NtStatus> { + if matches!( + create_disposition, + CreateDisposition::Supersede + | CreateDisposition::Overwrite + | CreateDisposition::OverwriteIf + ) { + return Err(NtStatus::INVALID_PARAMETER); + } + + let existed_before_open = match self.fs.file_status(path) { + Ok(status) => { + if status.file_type != FileType::Directory { + return Err(NtStatus::NOT_A_DIRECTORY); + } + true + } + Err(_) + if matches!( + create_disposition, + CreateDisposition::Create | CreateDisposition::OpenIf + ) => + { + self.fs + .mkdir(path, create_directory_mode(file_attributes)) + .map_err(map_mkdir_error)?; + false + } + Err(error) => return Err(map_file_status_error(error)), + }; + + let open_disposition = if existed_before_open { + create_disposition + } else { + CreateDisposition::Open + }; + let flags = desired_access.open_flags(open_disposition, create_options); + let fd = self + .fs + .open(path, flags, Mode::empty()) + .map_err(|error| map_open_error(error, create_disposition))?; + let information = create_disposition.success_information(existed_before_open); + Ok(( + FileObject { + path: String::from(path), + backing: FileObjectBacking::Filesystem { + fd, + is_directory: true, + }, + create_time_access: desired_access, + share_access, + create_options, + }, + information, + )) + } + + fn object_attributes_to_file_target( + &self, + object_attributes: ObjectAttributes, + ) -> Result { + let object_name_ptr = + ConstPtr::::from_usize(object_attributes.object_name); + let object_name = object_name_ptr + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + let object_name = object_name.read_string::()?; + let resolver = FilePathResolver::new(&self.process.object_manager); + if object_attributes.root_directory.is_null() { + return resolver.resolve(FilePathRoot::Namespace, &object_name); + } + + let root_file = self.file_entry(object_attributes.root_directory)?; + root_file.with_entry(|root_file| { + if let Some(parent) = root_file.condrv_object() { + return resolver.resolve(FilePathRoot::Condrv(parent), &object_name); + } + resolver.resolve( + FilePathRoot::Filesystem { + path: &root_file.path, + is_directory: root_file.is_directory(), + }, + &object_name, + ) + }) + } + + fn check_file_sharing( + &self, + identity: FileSharingIdentity<'_>, + desired_access: FileAccess, + share_access: FileShareAccess, + ) -> Result<(), NtStatus> { + let raw_handles: alloc::vec::Vec = + self.process.handles.read().iter_alive().collect(); + for raw_handle in raw_handles { + let Some(handle) = Handle::from_raw_fd(raw_handle) else { + continue; + }; + let Some(entry) = raw_handle_entry::>( + &self.global.litebox, + &self.process.handles, + handle, + ) else { + continue; + }; + let conflicts = entry.with_entry(|file| { + identity.matches(file) + && (desired_access.conflicts_with_share(file.share_access) + || file.create_time_access.conflicts_with_share(share_access)) + }); + if conflicts { + return Err(NtStatus::SHARING_VIOLATION); + } + } + Ok(()) + } +} + +fn probe_file_outputs( + file_handle: MutPtr, + io_status_block: MutPtr, +) -> Result<(), NtStatus> { + probe_guest_output_preserving_value::(file_handle)?; + probe_guest_output_preserving_value::(io_status_block) +} + +fn write_file_result( + file_handle: MutPtr, + io_status_block: MutPtr, + result: Result<(Handle, FileCreateInformation), NtStatus>, + cleanup_handle: impl FnOnce(Handle), +) -> NtStatus { + match result { + Ok((handle, information)) => { + if write_file_success::(file_handle, io_status_block, handle, information) + .is_none() + { + cleanup_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + Err(status) => { + let _ = io_status_block + .write_at_offset(0, IoStatusBlock::new(status, failure_information(status))); + status + } + } +} + +fn write_file_success( + file_handle: MutPtr, + io_status_block: MutPtr, + handle: Handle, + information: FileCreateInformation, +) -> Option<()> { + file_handle.write_at_offset(0, Handle::default())?; + io_status_block + .write_at_offset(0, IoStatusBlock::new(NtStatus::SUCCESS, information.into()))?; + file_handle.write_at_offset(0, handle) +} + +fn failure_information(status: NtStatus) -> usize { + match status { + NtStatus::OBJECT_NAME_COLLISION => FileCreateInformation::Exists.into(), + NtStatus::OBJECT_NAME_NOT_FOUND | NtStatus::OBJECT_PATH_NOT_FOUND => { + FileCreateInformation::DoesNotExist.into() + } + _ => 0, + } +} + +fn validate_create_options( + desired_access: FileAccess, + create_disposition: CreateDisposition, + create_options: FileCreateOptions, +) -> Result<(), NtStatus> { + if create_options.contains(FileCreateOptions::DIRECTORY_FILE) + && matches!( + create_disposition, + CreateDisposition::Supersede + | CreateDisposition::Overwrite + | CreateDisposition::OverwriteIf + ) + { + return Err(NtStatus::INVALID_PARAMETER); + } + + if create_options.contains(FileCreateOptions::DIRECTORY_FILE) + && !create_options + .difference(FileCreateOptions::DIRECTORY_COMPATIBLE) + .is_empty() + { + return Err(NtStatus::INVALID_PARAMETER); + } + + if create_options.contains(FileCreateOptions::SYNCHRONOUS_IO) { + return Err(NtStatus::INVALID_PARAMETER); + } + + if create_options.intersects(FileCreateOptions::SYNCHRONOUS_IO) + && !desired_access.contains(FileAccess::SYNCHRONIZE) + { + return Err(NtStatus::INVALID_PARAMETER); + } + + if create_options.contains(FileCreateOptions::NO_INTERMEDIATE_BUFFERING) + && desired_access.contains(FileAccess::APPEND_DATA) + { + return Err(NtStatus::INVALID_PARAMETER); + } + + if create_options.contains(FileCreateOptions::DELETE_ON_CLOSE) + && !desired_access.contains(FileAccess::DELETE) + { + return Err(NtStatus::INVALID_PARAMETER); + } + + Ok(()) +} + +fn create_mode(file_attributes: u32) -> Mode { + if file_attributes & FILE_ATTRIBUTE_READONLY == 0 { + Mode::RUSR | Mode::WUSR + } else { + Mode::RUSR + } +} + +fn create_directory_mode(file_attributes: u32) -> Mode { + create_mode(file_attributes) | Mode::XUSR +} + +fn map_open_error(error: OpenError, create_disposition: CreateDisposition) -> NtStatus { + match error { + OpenError::PathError(error) => match error { + PathError::NoSuchFileOrDirectory => match create_disposition { + CreateDisposition::Create + | CreateDisposition::OpenIf + | CreateDisposition::OverwriteIf + | CreateDisposition::Supersede => NtStatus::OBJECT_PATH_NOT_FOUND, + CreateDisposition::Open | CreateDisposition::Overwrite => { + NtStatus::OBJECT_NAME_NOT_FOUND + } + }, + PathError::MissingComponent => NtStatus::OBJECT_PATH_NOT_FOUND, + PathError::ComponentNotADirectory => NtStatus::NOT_A_DIRECTORY, + PathError::InvalidPathname => NtStatus::INVALID_PARAMETER, + PathError::NoSearchPerms { .. } => NtStatus::UNSUCCESSFUL, + }, + OpenError::AccessNotAllowed | OpenError::NoWritePerms | OpenError::ReadOnlyFileSystem => { + NtStatus::ACCESS_DENIED + } + OpenError::AlreadyExists => NtStatus::OBJECT_NAME_COLLISION, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_file_status_error(error: FileStatusError) -> NtStatus { + match error { + FileStatusError::PathError(PathError::NoSuchFileOrDirectory) => { + NtStatus::OBJECT_NAME_NOT_FOUND + } + FileStatusError::PathError(PathError::MissingComponent) => NtStatus::OBJECT_PATH_NOT_FOUND, + FileStatusError::PathError(PathError::ComponentNotADirectory) => NtStatus::NOT_A_DIRECTORY, + FileStatusError::PathError(PathError::InvalidPathname) => NtStatus::INVALID_PARAMETER, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_mkdir_error(error: MkdirError) -> NtStatus { + match error { + MkdirError::AlreadyExists => NtStatus::OBJECT_NAME_COLLISION, + MkdirError::PathError(PathError::NoSuchFileOrDirectory | PathError::MissingComponent) => { + NtStatus::OBJECT_PATH_NOT_FOUND + } + MkdirError::PathError(PathError::ComponentNotADirectory) => NtStatus::NOT_A_DIRECTORY, + MkdirError::PathError(PathError::InvalidPathname) => NtStatus::INVALID_PARAMETER, + MkdirError::NoWritePerms | MkdirError::ReadOnlyFileSystem => NtStatus::ACCESS_DENIED, + _ => NtStatus::UNSUCCESSFUL, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{ + TestFS, TestPlatform, const_ptr, mut_byte_ptr, mut_ptr, null_mut_ptr, object_attributes, + unicode_string, utf16_units as utf16, + }; + use litebox::fs::FileSystem as _; + + extern crate std; + + const FILE_GENERIC_READ: u32 = AccessMask::STANDARD_RIGHTS_READ.bits() + | FileAccess::READ_DATA.bits() + | FileAccess::READ_ATTRIBUTES.bits() + | FileAccess::READ_EA.bits() + | AccessMask::SYNCHRONIZE.bits(); + const FILE_GENERIC_WRITE: u32 = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | FileAccess::WRITE_DATA.bits() + | FileAccess::WRITE_ATTRIBUTES.bits() + | FileAccess::WRITE_EA.bits() + | FileAccess::APPEND_DATA.bits() + | AccessMask::SYNCHRONIZE.bits(); + const FILE_SUPERSEDE: u32 = 0; + const FILE_OPEN: u32 = 2; + const FILE_CREATE: u32 = 1; + const FILE_OVERWRITE: u32 = 4; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + fn open_object_attributes( + path: &str, + ) -> ( + std::vec::Vec, + std::boxed::Box, + ObjectAttributes, + ) { + let path = utf16(path); + let name = std::boxed::Box::new(unicode_string(&path)); + let attributes = object_attributes(&name, 0); + (path, name, attributes) + } + + fn create_existing_file(task: &Task, path: &str, data: &[u8]) { + let fd = task + .fs + .open(path, OFlags::CREAT | OFlags::RDWR, Mode::RUSR | Mode::WUSR) + .unwrap(); + assert_eq!(task.fs.write(&fd, data, Some(0)).unwrap(), data.len()); + task.fs.close(&fd).unwrap(); + } + + fn create_file( + task: &Task, + path: &str, + desired_access: u32, + create_disposition: u32, + ) -> (NtStatus, Handle, IoStatusBlock) { + let (_path, _name, attributes) = open_object_attributes(path); + let mut handle = Handle::default(); + let mut io_status = IoStatusBlock::default(); + let status = task.sys_nt_create_file( + mut_ptr(&mut handle), + desired_access, + Some(const_ptr(&attributes)), + mut_ptr(&mut io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + create_disposition, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ); + (status, handle, io_status) + } + + fn open_fs_root(task: &Task) -> Handle { + let (_path, _name, attributes) = open_object_attributes("/"); + let mut handle = Handle::default(); + let mut io_status = IoStatusBlock::default(); + assert_eq!( + task.sys_nt_open_file( + mut_ptr(&mut handle), + FILE_GENERIC_READ, + Some(const_ptr(&attributes)), + mut_ptr(&mut io_status), + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + (FileCreateOptions::DIRECTORY_FILE | FileCreateOptions::SYNCHRONOUS_IO_NONALERT) + .bits(), + ), + NtStatus::SUCCESS + ); + handle + } + + fn open_condrv_server(task: &Task) -> Handle { + let (_server_path, _server_name, server_attributes) = + open_object_attributes(r"\Device\ConDrv\Server"); + let mut io_status = IoStatusBlock::default(); + task.do_nt_create_file( + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + server_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0 + } + + fn open_condrv_reference(task: &Task, server_handle: Handle) -> Handle { + let (_reference_path, _reference_name, mut reference_attributes) = + open_object_attributes(r"\Reference"); + reference_attributes.root_directory = server_handle; + let mut io_status = IoStatusBlock::default(); + task.do_nt_create_file( + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + reference_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0 + } + + fn open_condrv_child( + task: &Task, + root: Handle, + name: &str, + desired_access: u32, + ) -> (NtStatus, Handle) { + let (_path, _name, mut attributes) = open_object_attributes(name); + attributes.root_directory = root; + let mut handle = Handle::default(); + let mut io_status = IoStatusBlock::default(); + let status = task.sys_nt_create_file( + mut_ptr(&mut handle), + desired_access, + Some(const_ptr(&attributes)), + mut_ptr(&mut io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_OPEN, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ); + (status, handle) + } + + #[test] + fn nt_duplicate_object_rejects_file_access_escalation() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/duplicate-read-only.txt", b"data"); + let (status, source, _) = create_file( + &task, + "/tmp/duplicate-read-only.txt", + FILE_GENERIC_READ, + FILE_OPEN, + ); + assert_eq!(status, NtStatus::SUCCESS); + + let mut write_duplicate = Handle::default(); + assert_eq!( + task.sys_nt_duplicate_object( + crate::syscalls::ProcessHandle::CURRENT, + source, + crate::syscalls::ProcessHandle::CURRENT, + Some(mut_ptr(&mut write_duplicate)), + FileAccess::WRITE_DATA.bits(), + 0, + 0, + ), + NtStatus::ACCESS_DENIED + ); + assert!(write_duplicate.is_null()); + + let mut maximum_duplicate = Handle::default(); + assert_eq!( + task.sys_nt_duplicate_object( + crate::syscalls::ProcessHandle::CURRENT, + source, + crate::syscalls::ProcessHandle::CURRENT, + Some(mut_ptr(&mut maximum_duplicate)), + AccessMask::MAXIMUM_ALLOWED.bits(), + 0, + 0, + ), + NtStatus::SUCCESS + ); + assert_eq!( + task.typed_handle::>(maximum_duplicate) + .and_then(|typed| { + task.typed_handle_metadata(&typed) + .map(|metadata| metadata.granted_access) + }), + Ok(FileAccess::from_desired_access(FILE_GENERIC_READ).bits()) + ); + assert_eq!(task.sys_nt_close(source), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(maximum_duplicate), NtStatus::SUCCESS); + } + + #[test] + fn nt_create_file_follows_condrv_connection_through_standard_streams() { + let task = crate::tests::test_task(); + let server_handle = open_condrv_server(&task); + let reference_handle = open_condrv_reference(&task, server_handle); + let (_connect_path, _connect_name, mut connect_attributes) = + open_object_attributes(r"\Connect"); + connect_attributes.root_directory = reference_handle; + let ea = condrv::ea_buffer(b"server", 1340); + let mut connect_handle = Handle::default(); + let mut io_status = IoStatusBlock::default(); + + assert_eq!( + task.file_entry(server_handle) + .unwrap() + .with_entry(FileObject::condrv_object), + Some(CondrvObject::Server) + ); + assert_eq!( + task.file_entry(reference_handle) + .unwrap() + .with_entry(FileObject::condrv_object), + Some(CondrvObject::Reference) + ); + + assert_eq!( + task.sys_nt_create_file( + mut_ptr(&mut connect_handle), + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + Some(const_ptr(&connect_attributes)), + mut_ptr(&mut io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_OPEN, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ), + NtStatus::EAS_NOT_SUPPORTED + ); + assert!(connect_handle.is_null()); + + assert_eq!( + task.sys_nt_create_file( + mut_ptr(&mut connect_handle), + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + Some(const_ptr(&connect_attributes)), + mut_ptr(&mut io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_OPEN, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + Some(const_ptr(&ea[0])), + u32::try_from(ea.len()).unwrap(), + ), + NtStatus::PIPE_DISCONNECTED + ); + assert!(connect_handle.is_null()); + + assert_eq!(task.sys_nt_close(server_handle), NtStatus::SUCCESS); + let mut ea = ea; + *ea.last_mut().unwrap() = 1; + assert_eq!( + task.sys_nt_create_file( + mut_ptr(&mut connect_handle), + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + Some(const_ptr(&connect_attributes)), + mut_ptr(&mut io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_OPEN, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + Some(const_ptr(&ea[0])), + u32::try_from(ea.len()).unwrap(), + ), + NtStatus::SUCCESS + ); + assert_eq!( + task.file_entry(connect_handle) + .unwrap() + .with_entry(FileObject::condrv_object), + Some(CondrvObject::Connect) + ); + + assert_eq!(connect_handle, server_handle); + let (input_status, input_handle) = + open_condrv_child(&task, connect_handle, r"\Input", FILE_GENERIC_READ); + let (output_status, output_handle) = + open_condrv_child(&task, connect_handle, r"\Output", FILE_GENERIC_WRITE); + let (current_input_status, current_input_handle) = + open_condrv_child(&task, connect_handle, r"\CurrentIn", FILE_GENERIC_READ); + let (current_output_status, current_output_handle) = + open_condrv_child(&task, connect_handle, r"\CurrentOut", FILE_GENERIC_WRITE); + let (screen_buffer_status, screen_buffer_handle) = + open_condrv_child(&task, connect_handle, r"\ScreenBuffer", FILE_GENERIC_WRITE); + assert_eq!(input_status, NtStatus::SUCCESS); + assert_eq!(output_status, NtStatus::SUCCESS); + assert_eq!(current_input_status, NtStatus::SUCCESS); + assert_eq!(current_output_status, NtStatus::SUCCESS); + assert_eq!(screen_buffer_status, NtStatus::SUCCESS); + let stream_identity = |handle| { + task.file_entry(handle).unwrap().with_entry(|file| { + ( + file.condrv_object().unwrap(), + file.condrv_stream_object_id().unwrap(), + ) + }) + }; + let input_identity = stream_identity(input_handle); + let output_identity = stream_identity(output_handle); + let current_input_identity = stream_identity(current_input_handle); + let current_output_identity = stream_identity(current_output_handle); + let screen_buffer_identity = stream_identity(screen_buffer_handle); + assert_eq!(current_input_identity.0, CondrvObject::CurrentInput); + assert_eq!(current_output_identity.0, CondrvObject::CurrentOutput); + assert_eq!(screen_buffer_identity.0, CondrvObject::ScreenBuffer); + assert_ne!(input_identity.1, current_input_identity.1); + assert_ne!(output_identity.1, current_output_identity.1); + assert_ne!(output_identity.1, screen_buffer_identity.1); + assert_ne!(current_output_identity.1, screen_buffer_identity.1); + for handle in [output_handle, current_output_handle, screen_buffer_handle] { + assert_eq!( + task.file_entry(handle) + .unwrap() + .with_entry(|file| file.path.clone()), + "/dev/stdout" + ); + } + + for (path, desired_access, expected_object, expected_bound_id) in [ + ( + r"\Device\ConDrv\CurrentIn", + FILE_GENERIC_READ, + CondrvObject::CurrentInput, + Some(current_input_identity.1), + ), + ( + r"\Device\ConDrv\CurrentOut", + FILE_GENERIC_WRITE, + CondrvObject::CurrentOutput, + Some(current_output_identity.1), + ), + ( + r"\Device\ConDrv\ScreenBuffer", + FILE_GENERIC_WRITE, + CondrvObject::ScreenBuffer, + None, + ), + ] { + let (status, handle, _) = create_file(&task, path, desired_access, FILE_OPEN); + assert_eq!(status, NtStatus::SUCCESS, "{path}"); + let identity = stream_identity(handle); + assert_eq!(identity.0, expected_object, "{path}"); + if let Some(expected_bound_id) = expected_bound_id { + assert_eq!(identity.1, expected_bound_id, "{path}"); + } else { + assert_ne!(identity.1, screen_buffer_identity.1, "{path}"); + } + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + } + + assert_eq!(task.sys_nt_close(current_output_handle), NtStatus::SUCCESS); + let mut exclusive_handles = alloc::vec::Vec::new(); + for path in [ + r"\Device\ConDrv\Output", + r"\Device\ConDrv\CurrentOut", + r"\Device\ConDrv\ScreenBuffer", + r"\Device\ConDrv\ScreenBuffer", + ] { + let (_path, _name, attributes) = open_object_attributes(path); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_file( + mut_ptr(&mut handle), + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + Some(const_ptr(&attributes)), + mut_ptr(&mut io_status), + None, + 0, + 0, + FILE_OPEN, + FileCreateOptions::NON_DIRECTORY_FILE.bits(), + None, + 0, + ), + NtStatus::SUCCESS, + "{path}" + ); + exclusive_handles.push(handle); + } + let exclusive_identities: alloc::vec::Vec<_> = exclusive_handles + .iter() + .map(|handle| stream_identity(*handle)) + .collect(); + assert_eq!(exclusive_identities[0].0, CondrvObject::Output); + assert_eq!(exclusive_identities[1].0, CondrvObject::CurrentOutput); + assert_eq!( + exclusive_identities[1].1, current_output_identity.1, + "CurrentOut must reference the active output object" + ); + assert_ne!(exclusive_identities[0].1, exclusive_identities[1].1); + assert_ne!(exclusive_identities[2].1, exclusive_identities[3].1); + let closed_screen_buffer_id = exclusive_identities[2].1; + for handle in exclusive_handles { + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + } + let (status, reopened_screen_buffer, _) = create_file( + &task, + r"\Device\ConDrv\ScreenBuffer", + FILE_GENERIC_WRITE, + FILE_OPEN, + ); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!( + stream_identity(reopened_screen_buffer).1, + closed_screen_buffer_id + ); + assert_eq!(task.sys_nt_close(reopened_screen_buffer), NtStatus::SUCCESS); + + assert_eq!(task.sys_nt_close(screen_buffer_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(current_input_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(output_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(input_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(connect_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(reference_handle), NtStatus::SUCCESS); + } + + #[test] + fn nt_query_volume_information_file_returns_fs_device_information() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let handle = open_fs_root(&task); + let mut io_status = IoStatusBlock::default(); + let mut output = FileFsDeviceInformation { + device_type: 0, + characteristics: 0, + }; + + assert_eq!( + task.sys_nt_query_volume_information_file( + handle, + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::SUCCESS + ); + assert_eq!( + FileDeviceType::try_from(output.device_type), + Ok(FileDeviceType::Disk), + "Wine's regular file/directory branch reports FILE_DEVICE_DISK" + ); + assert_eq!( + FileDeviceCharacteristics::from_bits_retain(output.characteristics), + FileDeviceCharacteristics::IS_MOUNTED, + "Wine's regular file/directory branch reports FILE_DEVICE_IS_MOUNTED" + ); + assert_eq!((output.device_type, output.characteristics), (0x7, 0x20)); + assert_eq!(io_status.status, NtStatus::SUCCESS.as_raw()); + assert_eq!(io_status.information, size_of::()); + }); + } + + #[test] + fn nt_query_volume_information_file_leaves_iosb_untouched_on_failures() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let handle = open_fs_root(&task); + let sentinel = IoStatusBlock::new(NtStatus::from_raw(0x1111_1111), 0x2222_2222); + let mut io_status = sentinel; + let mut output = FileFsDeviceInformation { + device_type: 0xcccc_cccc, + characteristics: 0xcccc_cccc, + }; + + assert_eq!( + task.sys_nt_query_volume_information_file( + handle, + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap() - 1, + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(io_status.status, sentinel.status); + assert_eq!(io_status.information, sentinel.information); + assert_eq!( + (output.device_type, output.characteristics), + (0xcccc_cccc, 0xcccc_cccc) + ); + + assert_eq!( + task.sys_nt_query_volume_information_file( + handle, + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + 0xffff, + ), + NtStatus::INVALID_INFO_CLASS + ); + assert_eq!(io_status.status, sentinel.status); + assert_eq!(io_status.information, sentinel.information); + + assert_eq!( + task.sys_nt_query_volume_information_file( + Handle::from_raw(0x1234), + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::INVALID_HANDLE + ); + assert_eq!(io_status.status, sentinel.status); + assert_eq!(io_status.information, sentinel.information); + + assert_eq!( + task.sys_nt_query_volume_information_file( + Handle::from_raw(0x1234), + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap() - 1, + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(io_status.status, sentinel.status); + assert_eq!(io_status.information, sentinel.information); + + assert_eq!( + task.sys_nt_query_volume_information_file( + Handle::from_raw(0x1234), + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + 0xffff, + ), + NtStatus::INVALID_INFO_CLASS + ); + assert_eq!(io_status.status, sentinel.status); + assert_eq!(io_status.information, sentinel.information); + + assert_eq!( + task.sys_nt_query_volume_information_file( + handle, + null_mut_ptr::(), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[test] + fn nt_open_file_opens_existing_absolute_and_relative_files() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/dir-file-root.txt", b"root"); + task.fs + .mkdir("/tmp/dir", Mode::RUSR | Mode::WUSR | Mode::XUSR) + .unwrap(); + create_existing_file(&task, "/tmp/dir/child.txt", b"child"); + + let (_path, _name, attributes) = + open_object_attributes(r"\Device\HarddiskVolume1\tmp\dir-file-root.txt"); + let mut handle = Handle::default(); + let mut io_status = IoStatusBlock::default(); + assert_eq!( + task.sys_nt_open_file( + mut_ptr(&mut handle), + FILE_GENERIC_READ, + Some(const_ptr(&attributes)), + mut_ptr(&mut io_status), + FILE_SHARE_READ, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + ), + NtStatus::SUCCESS + ); + assert_ne!(handle, Handle::default()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Opened) + ); + + let (_path, _name, directory_attributes) = + open_object_attributes(r"\Device\HarddiskVolume1\tmp\dir"); + let directory_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + directory_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + (FileCreateOptions::DIRECTORY_FILE | FileCreateOptions::SYNCHRONOUS_IO_NONALERT) + .bits(), + None, + 0, + ) + .unwrap() + .0; + let (_path, _child_name, mut child_attributes) = open_object_attributes("child.txt"); + child_attributes.root_directory = directory_handle; + let (child_handle, information) = task + .do_nt_create_file( + FILE_GENERIC_READ, + child_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap(); + assert_ne!(child_handle, Handle::default()); + assert_eq!(information, FileCreateInformation::Opened); + } + + #[test] + fn nt_create_file_reports_disposition_information() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/existing.txt", b"old"); + + let (status, handle, io_status) = + create_file(&task, "/tmp/existing.txt", FILE_GENERIC_READ, FILE_OPEN); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!(handle, Handle::default()); + assert_eq!(io_status.status, NtStatus::SUCCESS.as_raw()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Opened) + ); + + let (status, handle, io_status) = create_file( + &task, + "/tmp/created.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + FILE_CREATE, + ); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!(handle, Handle::default()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Created) + ); + + let (status, handle, io_status) = create_file( + &task, + "/tmp/supersede-created.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + FILE_SUPERSEDE, + ); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!(handle, Handle::default()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Created) + ); + + let (status, _handle, _io_status) = create_file( + &task, + "/tmp/existing.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + FILE_SUPERSEDE, + ); + assert_eq!(status, NtStatus::ACCESS_DENIED); + + let (status, handle, io_status) = create_file( + &task, + "/tmp/existing.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE | AccessMask::DELETE.bits(), + FILE_SUPERSEDE, + ); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!(handle, Handle::default()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Superseded) + ); + + let (status, handle, io_status) = create_file( + &task, + "/tmp/created.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + FILE_OVERWRITE, + ); + assert_eq!(status, NtStatus::SUCCESS); + assert_ne!(handle, Handle::default()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Overwritten) + ); + } + + #[test] + fn nt_create_file_reports_missing_and_collision_information() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/existing-collision.txt", b"old"); + + let (status, _handle, io_status) = + create_file(&task, "/tmp/missing.txt", FILE_GENERIC_READ, FILE_OPEN); + assert_eq!(status, NtStatus::OBJECT_NAME_NOT_FOUND); + assert_eq!(io_status.status, NtStatus::OBJECT_NAME_NOT_FOUND.as_raw()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::DoesNotExist) + ); + + let (status, _handle, io_status) = create_file( + &task, + "/tmp/existing-collision.txt", + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + FILE_CREATE, + ); + assert_eq!(status, NtStatus::OBJECT_NAME_COLLISION); + assert_eq!(io_status.status, NtStatus::OBJECT_NAME_COLLISION.as_raw()); + assert_eq!( + io_status.information, + usize::from(FileCreateInformation::Exists) + ); + } + + #[test] + fn nt_create_file_rejects_invalid_share_access() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/invalid-share.txt", b"old"); + let (_path, _name, attributes) = open_object_attributes("/tmp/invalid-share.txt"); + let mut io_status = IoStatusBlock::default(); + + assert_eq!( + task.do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + 0x8, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap_err(), + NtStatus::INVALID_PARAMETER + ); + } + + #[test] + fn nt_create_file_directory_handles_can_root_relative_opens() { + let task = crate::tests::test_task(); + let (_path, _name, attributes) = open_object_attributes("/tmp/created-dir"); + let mut io_status = IoStatusBlock::default(); + let directory_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Create, + (FileCreateOptions::DIRECTORY_FILE | FileCreateOptions::SYNCHRONOUS_IO_NONALERT) + .bits(), + None, + 0, + ) + .unwrap() + .0; + let (_path, _name, mut child_attributes) = open_object_attributes("child.txt"); + child_attributes.root_directory = directory_handle; + let child_handle = task + .do_nt_create_file( + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + child_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Create, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + assert_ne!(child_handle, Handle::default()); + } + + #[test] + fn nt_create_file_actual_directory_handles_can_root_relative_opens() { + let task = crate::tests::test_task(); + task.fs + .mkdir("/tmp/implicit-dir", Mode::RUSR | Mode::WUSR | Mode::XUSR) + .unwrap(); + create_existing_file(&task, "/tmp/implicit-dir/child.txt", b"child"); + let (_path, _name, attributes) = open_object_attributes("/tmp/implicit-dir"); + let mut io_status = IoStatusBlock::default(); + let directory_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + let (_path, _name, mut child_attributes) = open_object_attributes("child.txt"); + child_attributes.root_directory = directory_handle; + + let (child_handle, information) = task + .do_nt_create_file( + FILE_GENERIC_READ, + child_attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap(); + assert_ne!(child_handle, Handle::default()); + assert_eq!(information, FileCreateInformation::Opened); + } + + #[test] + fn nt_create_file_validates_create_options() { + let generic_read = FileAccess::from_desired_access(FILE_GENERIC_READ); + let synchronize = FileAccess::SYNCHRONIZE; + + assert_eq!( + validate_create_options( + generic_read, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + FileAccess::READ_DATA, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + FileAccess::APPEND_DATA | synchronize, + CreateDisposition::Open, + FileCreateOptions::NO_INTERMEDIATE_BUFFERING, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + generic_read, + CreateDisposition::Open, + FileCreateOptions::DELETE_ON_CLOSE, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + generic_read, + CreateDisposition::Overwrite, + FileCreateOptions::DIRECTORY_FILE, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + generic_read, + CreateDisposition::Open, + FileCreateOptions::DIRECTORY_FILE | FileCreateOptions::SEQUENTIAL_ONLY, + ), + Err(NtStatus::INVALID_PARAMETER) + ); + assert_eq!( + validate_create_options( + generic_read, + CreateDisposition::Open, + FileCreateOptions::DIRECTORY_FILE + | FileCreateOptions::WRITE_THROUGH + | FileCreateOptions::SYNCHRONOUS_IO_NONALERT, + ), + Ok(()) + ); + assert_eq!( + validate_create_options( + generic_read | FileAccess::DELETE, + CreateDisposition::Open, + FileCreateOptions::DIRECTORY_FILE + | FileCreateOptions::DELETE_ON_CLOSE + | FileCreateOptions::COMPLETE_IF_OPLOCKED + | FileCreateOptions::OPEN_REPARSE_POINT + | FileCreateOptions::OPEN_FOR_FREE_SPACE_QUERY + | FileCreateOptions::NO_COMPRESSION + | FileCreateOptions::SYNCHRONOUS_IO_NONALERT, + ), + Ok(()) + ); + assert!( + generic_read + .open_flags( + CreateDisposition::Open, + FileCreateOptions::NON_DIRECTORY_FILE + ) + .contains(OFlags::NOFOLLOW) + ); + } + + #[test] + fn nt_create_file_enforces_share_access() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/shared.txt", b"old"); + let (_path, _name, attributes) = open_object_attributes("/tmp/shared.txt"); + let mut io_status = IoStatusBlock::default(); + let first_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + 0, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + assert_ne!(first_handle, Handle::default()); + + let (_path, _name, attributes) = open_object_attributes("/tmp/shared.txt"); + assert_eq!( + task.do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap_err(), + NtStatus::SHARING_VIOLATION + ); + } + + #[test] + fn nt_close_releases_file_handle_and_share_lock() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/close-shared.txt", b"old"); + let (_path, _name, attributes) = open_object_attributes("/tmp/close-shared.txt"); + let mut io_status = IoStatusBlock::default(); + let first_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + 0, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + + let (_path, _name, attributes) = open_object_attributes("/tmp/close-shared.txt"); + assert_eq!( + task.do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap_err(), + NtStatus::SHARING_VIOLATION + ); + + assert_eq!(task.sys_nt_close(first_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(first_handle), NtStatus::INVALID_HANDLE); + + let (_path, _name, attributes) = open_object_attributes("/tmp/close-shared.txt"); + let second_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + assert_eq!(task.sys_nt_close(second_handle), NtStatus::SUCCESS); + } + + #[test] + fn nt_close_deletes_delete_on_close_file() { + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/delete-on-close.txt", b"old"); + let (_path, _name, attributes) = open_object_attributes("/tmp/delete-on-close.txt"); + let mut io_status = IoStatusBlock::default(); + let handle = task + .do_nt_create_file( + FILE_GENERIC_READ | AccessMask::DELETE.bits(), + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Open, + (FileCreateOptions::SYNCHRONOUS_IO_NONALERT | FileCreateOptions::DELETE_ON_CLOSE) + .bits(), + None, + 0, + ) + .unwrap() + .0; + + assert!(task.fs.file_status("/tmp/delete-on-close.txt").is_ok()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert!(matches!( + task.fs.file_status("/tmp/delete-on-close.txt"), + Err(FileStatusError::PathError(PathError::NoSuchFileOrDirectory)) + )); + } + + #[test] + fn nt_close_deletes_delete_on_close_directory() { + let task = crate::tests::test_task(); + let (_path, _name, attributes) = open_object_attributes("/tmp/delete-on-close-dir"); + let mut io_status = IoStatusBlock::default(); + let handle = task + .do_nt_create_file( + FILE_GENERIC_READ | AccessMask::DELETE.bits(), + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Create, + (FileCreateOptions::DIRECTORY_FILE + | FileCreateOptions::SYNCHRONOUS_IO_NONALERT + | FileCreateOptions::DELETE_ON_CLOSE) + .bits(), + None, + 0, + ) + .unwrap() + .0; + + assert!(task.fs.file_status("/tmp/delete-on-close-dir").is_ok()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert!(matches!( + task.fs.file_status("/tmp/delete-on-close-dir"), + Err(FileStatusError::PathError(PathError::NoSuchFileOrDirectory)) + )); + } + + #[test] + fn write_file_result_clears_handle_output_when_iosb_write_fails() { + let task = crate::tests::test_task(); + let (_path, _name, attributes) = open_object_attributes("/tmp/iosb-fault.txt"); + let mut io_status = IoStatusBlock::default(); + let created_handle = task + .do_nt_create_file( + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Create, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + let mut handle_output = created_handle; + + let status = run_with_test_platform_pointers(|| { + write_file_result::( + mut_ptr(&mut handle_output), + null_mut_ptr::(), + Ok((created_handle, FileCreateInformation::Created)), + |handle| task.close_file_handle(handle), + ) + }); + + assert_eq!(status, NtStatus::ACCESS_VIOLATION); + assert_eq!(handle_output, Handle::default()); + assert_eq!(task.sys_nt_close(created_handle), NtStatus::INVALID_HANDLE); + let (_path, _name, attributes) = open_object_attributes("/tmp/iosb-fault.txt"); + let reopened_handle = task + .do_nt_create_file( + FILE_GENERIC_READ, + attributes, + mut_ptr(&mut io_status), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + CreateDisposition::Open, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ) + .unwrap() + .0; + assert_eq!(task.sys_nt_close(reopened_handle), NtStatus::SUCCESS); + } + + #[test] + fn probe_file_outputs_preserves_handle_output_when_iosb_probe_fails() { + let original_handle = Handle::from_raw_fd(0).unwrap(); + let mut handle = original_handle; + + let status = run_with_test_platform_pointers(|| { + probe_file_outputs::(mut_ptr(&mut handle), null_mut_ptr()) + }); + + assert_eq!(status, Err(NtStatus::ACCESS_VIOLATION)); + assert_eq!(handle, original_handle); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + mod host_fidelity { + use super::*; + use crate::nt_types::{ProcessEnvironmentBlock, RtlUserProcessParameters}; + use core::ffi::c_void; + + #[link(name = "ntdll")] + unsafe extern "system" { + fn RtlGetCurrentPeb() -> *const ProcessEnvironmentBlock; + fn NtCreateFile( + FileHandle: *mut *mut c_void, + DesiredAccess: u32, + ObjectAttributes: *const ObjectAttributes, + IoStatusBlock: *mut IoStatusBlock, + AllocationSize: *const i64, + FileAttributes: u32, + ShareAccess: u32, + CreateDisposition: u32, + CreateOptions: u32, + EaBuffer: *const c_void, + EaLength: u32, + ) -> i32; + fn NtOpenFile( + FileHandle: *mut *mut c_void, + DesiredAccess: u32, + ObjectAttributes: *const ObjectAttributes, + IoStatusBlock: *mut IoStatusBlock, + ShareAccess: u32, + OpenOptions: u32, + ) -> i32; + fn NtQueryVolumeInformationFile( + FileHandle: *mut c_void, + IoStatusBlock: *mut IoStatusBlock, + FsInformation: *mut c_void, + Length: u32, + FsInformationClass: u32, + ) -> i32; + fn NtDuplicateObject( + SourceProcessHandle: *mut c_void, + SourceHandle: *mut c_void, + TargetProcessHandle: *mut c_void, + TargetHandle: *mut c_void, + DesiredAccess: u32, + HandleAttributes: u32, + Options: u32, + ) -> i32; + fn NtClose(Handle: *mut c_void) -> i32; + } + + #[link(name = "kernel32")] + unsafe extern "system" { + fn AllocConsole() -> i32; + fn GetLastError() -> u32; + } + + fn host_nt_path(path: &std::path::Path) -> std::string::String { + std::format!(r"\??\{}", path.display()) + } + + fn test_tmp_dir(name: &str) -> std::path::PathBuf { + std::env::var_os("CARGO_TARGET_TMPDIR") + .map_or_else(std::env::temp_dir, std::path::PathBuf::from) + .join(name) + } + + fn host_object_attributes(name: &UnicodeString) -> ObjectAttributes { + object_attributes(name, 0) + } + + fn close_host_handle(handle: *mut c_void) { + if !handle.is_null() { + // SAFETY: The handle was returned by `NtCreateFile`/`NtOpenFile` in this test. + let status = unsafe { NtClose(handle) }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + } + } + + fn host_status(status: i32) -> NtStatus { + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + fn host_create_file(root: Handle, name: &str) -> (NtStatus, *mut c_void) { + let path = utf16(name); + let name = unicode_string(&path); + let mut attributes = host_object_attributes(&name); + attributes.root_directory = root; + let mut handle = core::ptr::null_mut(); + let mut io_status = IoStatusBlock::default(); + // SAFETY: All pointers reference live local typed values for the call. + let status = unsafe { + NtCreateFile( + &raw mut handle, + FILE_GENERIC_READ | FILE_GENERIC_WRITE, + &raw const attributes, + &raw mut io_status, + core::ptr::null(), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_OPEN, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + core::ptr::null(), + 0, + ) + }; + (host_status(status), handle) + } + + #[test] + fn connected_console_child_matrix_matches_host() { + // SAFETY: RtlGetCurrentPeb returns the live typed PEB for this process. + let mut peb = unsafe { &*RtlGetCurrentPeb() }; + // SAFETY: The current process owns a live RTL_USER_PROCESS_PARAMETERS block. + let mut process_parameters = + unsafe { &*(peb.process_parameters as *const RtlUserProcessParameters) }; + let console_handle = process_parameters.console_handle; + // ReactOS and Wine model detached/new/no-window console states as null or + // the reserved pseudo-handles -1 through -4. + if console_handle == 0 || console_handle >= usize::MAX - 3 { + // SAFETY: The test process has no connected console, so AllocConsole may attach one. + let allocated = unsafe { AllocConsole() }; + assert_ne!( + allocated, + 0, + "AllocConsole failed with Win32 error {}", + // SAFETY: GetLastError has no preconditions. + unsafe { GetLastError() } + ); + // AllocConsole updates the live process parameters. + // SAFETY: RtlGetCurrentPeb returns the live typed PEB for this process. + peb = unsafe { &*RtlGetCurrentPeb() }; + // SAFETY: The PEB owns a live RTL_USER_PROCESS_PARAMETERS block. + process_parameters = + unsafe { &*(peb.process_parameters as *const RtlUserProcessParameters) }; + } + assert_ne!( + process_parameters.console_handle, 0, + "console handle remained null after ensuring a console" + ); + let console_handle = Handle::from_raw(process_parameters.console_handle); + + let success = [NtStatus::SUCCESS]; + let screen_buffer = [NtStatus::SUCCESS, NtStatus::INVALID_PARAMETER]; + let invalid_handle = [NtStatus::INVALID_HANDLE]; + let not_found = [NtStatus::NOT_FOUND]; + for (name, expected) in [ + (r"\Input", success.as_slice()), + (r"\Output", success.as_slice()), + (r"\CurrentIn", success.as_slice()), + (r"\CurrentOut", success.as_slice()), + // Headless and pseudoconsole hosts may not support creating a bound legacy + // screen buffer even though their connected root supports CurrentOut. + (r"\ScreenBuffer", screen_buffer.as_slice()), + (r"\Server", success.as_slice()), + (r"\Reference", success.as_slice()), + (r"\Connect", invalid_handle.as_slice()), + (r"\Bogus", not_found.as_slice()), + ] { + let (status, handle) = host_create_file(console_handle, name); + assert!( + expected.contains(&status), + "{name:?} under console handle {:#x}: expected one of {expected:?}, got {status:?}", + console_handle.as_raw(), + ); + if status == NtStatus::SUCCESS { + assert!(!handle.is_null()); + close_host_handle(handle); + } else { + assert!(handle.is_null()); + } + } + } + + #[test] + fn nt_query_volume_information_file_device_information_matches_host_statuses() { + let test_dir = test_tmp_dir( + "nt_query_volume_information_file_device_information_matches_host_statuses", + ); + let _ = std::fs::remove_dir_all(&test_dir); + std::fs::create_dir_all(&test_dir).unwrap(); + let host_file = test_dir.join("existing.txt"); + std::fs::write(&host_file, b"host").unwrap(); + + let host_name_units = utf16(&host_nt_path(&host_file)); + let host_name = unicode_string(&host_name_units); + let host_attributes = host_object_attributes(&host_name); + let mut host_handle = core::ptr::null_mut(); + let mut host_io_status = IoStatusBlock::default(); + // SAFETY: All pointers reference live test locals, and ObjectName is an NT path + // to the temporary file created above. + let host_open = unsafe { + NtOpenFile( + &raw mut host_handle, + FILE_GENERIC_READ, + &raw const host_attributes, + &raw mut host_io_status, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + 0, + ) + }; + assert_eq!(host_open, NtStatus::SUCCESS.as_raw()); + + let mut host_output = FileFsDeviceInformation { + device_type: 0, + characteristics: 0, + }; + let mut host_query_iosb = IoStatusBlock::default(); + // SAFETY: The handle was opened above and output pointers reference live locals. + let host_query = unsafe { + NtQueryVolumeInformationFile( + host_handle, + &raw mut host_query_iosb, + (&raw mut host_output).cast::(), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ) + }; + close_host_handle(host_handle); + + assert_eq!(host_status(host_query), NtStatus::SUCCESS); + assert_eq!(host_query_iosb.status, NtStatus::SUCCESS.as_raw()); + assert_eq!( + host_query_iosb.information, + size_of::() + ); + assert_eq!( + FileDeviceType::try_from(host_output.device_type), + Ok(FileDeviceType::Disk) + ); + assert!( + FileDeviceCharacteristics::from_bits_retain(host_output.characteristics) + .contains(FileDeviceCharacteristics::IS_MOUNTED) + ); + + let task = crate::tests::test_task(); + let handle = open_fs_root(&task); + let mut output = FileFsDeviceInformation { + device_type: 0, + characteristics: 0, + }; + let mut io_status = IoStatusBlock::default(); + assert_eq!( + task.sys_nt_query_volume_information_file( + handle, + mut_ptr(&mut io_status), + mut_byte_ptr(&mut output), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ), + host_status(host_query) + ); + assert_eq!(io_status.status, host_query_iosb.status); + assert_eq!(io_status.information, host_query_iosb.information); + assert_eq!( + FileDeviceType::try_from(output.device_type), + Ok(FileDeviceType::Disk) + ); + assert_eq!( + FileDeviceCharacteristics::from_bits_retain(output.characteristics), + FileDeviceCharacteristics::IS_MOUNTED + ); + assert_eq!((output.device_type, output.characteristics), (0x7, 0x20)); + + let sentinel = IoStatusBlock::new(NtStatus::from_raw(0x1111_1111), 0x2222_2222); + for (length, class, expected) in [ + ( + u32::try_from(size_of::()).unwrap() - 1, + FsInformationClass::FileFsDeviceInformation as u32, + NtStatus::INFO_LENGTH_MISMATCH, + ), + ( + u32::try_from(size_of::()).unwrap(), + 0xffff, + NtStatus::INVALID_INFO_CLASS, + ), + ] { + let mut host_iosb = sentinel; + let mut host_output = FileFsDeviceInformation { + device_type: 0xcccc_cccc, + characteristics: 0xcccc_cccc, + }; + // SAFETY: `host_handle` is intentionally invalid only in the separate bad-handle + // case below; here all pointers reference live locals. + let host = unsafe { + NtQueryVolumeInformationFile( + core::ptr::null_mut(), + &raw mut host_iosb, + (&raw mut host_output).cast::(), + length, + class, + ) + }; + let mut shim_iosb = sentinel; + let mut shim_output = host_output; + let shim = task.sys_nt_query_volume_information_file( + handle, + mut_ptr(&mut shim_iosb), + mut_byte_ptr(&mut shim_output), + length, + class, + ); + + assert_eq!(shim, expected); + assert_eq!(shim, host_status(host)); + assert_eq!(shim_iosb.status, host_iosb.status); + assert_eq!(shim_iosb.information, host_iosb.information); + } + + let mut shim_iosb = sentinel; + let mut shim_output = FileFsDeviceInformation { + device_type: 0xcccc_cccc, + characteristics: 0xcccc_cccc, + }; + assert_eq!( + task.sys_nt_query_volume_information_file( + Handle::from_raw(0x1234), + mut_ptr(&mut shim_iosb), + mut_byte_ptr(&mut shim_output), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ), + NtStatus::INVALID_HANDLE + ); + assert_eq!(shim_iosb.status, sentinel.status); + assert_eq!(shim_iosb.information, sentinel.information); + + let mut host_iosb = sentinel; + let mut host_output = FileFsDeviceInformation { + device_type: 0xcccc_cccc, + characteristics: 0xcccc_cccc, + }; + // SAFETY: The bad handle is deliberately invalid to observe NTSTATUS; the output + // pointers reference live locals and are not retained. + let host_bad_handle = unsafe { + NtQueryVolumeInformationFile( + 0x1234usize as *mut c_void, + &raw mut host_iosb, + (&raw mut host_output).cast::(), + u32::try_from(size_of::()).unwrap(), + FsInformationClass::FileFsDeviceInformation as u32, + ) + }; + assert_eq!(host_status(host_bad_handle), NtStatus::INVALID_HANDLE); + assert_eq!(host_iosb.status, sentinel.status); + assert_eq!(host_iosb.information, sentinel.information); + + for (length, class, expected) in [ + ( + u32::try_from(size_of::()).unwrap() - 1, + FsInformationClass::FileFsDeviceInformation as u32, + NtStatus::INFO_LENGTH_MISMATCH, + ), + ( + u32::try_from(size_of::()).unwrap(), + 0xffff, + NtStatus::INVALID_INFO_CLASS, + ), + ] { + let mut host_iosb = sentinel; + let mut host_output = FileFsDeviceInformation { + device_type: 0xcccc_cccc, + characteristics: 0xcccc_cccc, + }; + // SAFETY: The bad handle is deliberately invalid to observe validation + // precedence; output pointers reference live locals and are not retained. + let host = unsafe { + NtQueryVolumeInformationFile( + 0x1234usize as *mut c_void, + &raw mut host_iosb, + (&raw mut host_output).cast::(), + length, + class, + ) + }; + let mut shim_iosb = sentinel; + let mut shim_output = host_output; + let shim = task.sys_nt_query_volume_information_file( + Handle::from_raw(0x1234), + mut_ptr(&mut shim_iosb), + mut_byte_ptr(&mut shim_output), + length, + class, + ); + + assert_eq!(shim, expected); + assert_eq!(shim, host_status(host)); + assert_eq!(shim_iosb.status, host_iosb.status); + assert_eq!(shim_iosb.information, host_iosb.information); + } + } + + #[test] + fn nt_open_file_existing_file_matches_host_status_and_information() { + let test_dir = + test_tmp_dir("nt_open_file_existing_file_matches_host_status_and_information"); + let _ = std::fs::remove_dir_all(&test_dir); + std::fs::create_dir_all(&test_dir).unwrap(); + let host_file = test_dir.join("existing.txt"); + std::fs::write(&host_file, b"host").unwrap(); + + let host_name_units = utf16(&host_nt_path(&host_file)); + let host_name = unicode_string(&host_name_units); + let host_attributes = host_object_attributes(&host_name); + let mut host_handle = core::ptr::null_mut(); + let mut host_io_status = IoStatusBlock::default(); + // SAFETY: All pointers reference live test locals, and ObjectName is an NT path + // to the temporary file created above. + let host_status = unsafe { + NtOpenFile( + &raw mut host_handle, + FILE_GENERIC_READ, + &raw const host_attributes, + &raw mut host_io_status, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + 0, + ) + }; + close_host_handle(host_handle); + + let task = crate::tests::test_task(); + create_existing_file(&task, "/tmp/existing.txt", b"litebox"); + let (_path, _name, attributes) = open_object_attributes("/tmp/existing.txt"); + let mut litebox_handle = Handle::default(); + let mut litebox_io_status = IoStatusBlock::default(); + let litebox_status = task.sys_nt_open_file( + mut_ptr(&mut litebox_handle), + FILE_GENERIC_READ, + Some(const_ptr(&attributes)), + mut_ptr(&mut litebox_io_status), + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + ); + + assert_eq!(host_status, litebox_status.as_raw()); + assert_eq!(host_io_status.status, litebox_io_status.status); + assert_eq!(host_io_status.information, litebox_io_status.information); + } + + #[test] + fn nt_duplicate_object_file_access_matrix_matches_host() { + let test_dir = test_tmp_dir("nt_duplicate_object_file_access_matrix_matches_host"); + let _ = std::fs::remove_dir_all(&test_dir); + std::fs::create_dir_all(&test_dir).unwrap(); + let host_file = test_dir.join("read-only-source.txt"); + std::fs::write(&host_file, b"host").unwrap(); + + let host_name_units = utf16(&host_nt_path(&host_file)); + let host_name = unicode_string(&host_name_units); + let host_attributes = host_object_attributes(&host_name); + let mut host_source = core::ptr::null_mut(); + let mut host_io_status = IoStatusBlock::default(); + // SAFETY: All pointers reference live locals and ObjectName names the test file. + assert_eq!( + unsafe { + NtOpenFile( + &raw mut host_source, + FILE_GENERIC_READ, + &raw const host_attributes, + &raw mut host_io_status, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut host_write_duplicate: *mut c_void = core::ptr::null_mut(); + // SAFETY: The process pseudo-handles and source handle are valid; output is a local. + assert_eq!( + unsafe { + NtDuplicateObject( + usize::MAX as *mut c_void, + host_source, + usize::MAX as *mut c_void, + (&raw mut host_write_duplicate).cast(), + FileAccess::WRITE_DATA.bits(), + 0, + 0, + ) + }, + NtStatus::ACCESS_DENIED.as_raw() + ); + assert!(host_write_duplicate.is_null()); + let mut host_maximum_duplicate: *mut c_void = core::ptr::null_mut(); + // SAFETY: The process pseudo-handles and source handle are valid; output is a local. + assert_eq!( + unsafe { + NtDuplicateObject( + usize::MAX as *mut c_void, + host_source, + usize::MAX as *mut c_void, + (&raw mut host_maximum_duplicate).cast(), + AccessMask::MAXIMUM_ALLOWED.bits(), + 0, + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + close_host_handle(host_source); + close_host_handle(host_maximum_duplicate); + } + + #[test] + fn nt_create_file_supersede_missing_matches_host_status_and_information() { + let test_dir = test_tmp_dir( + "nt_create_file_supersede_missing_matches_host_status_and_information", + ); + let _ = std::fs::remove_dir_all(&test_dir); + std::fs::create_dir_all(&test_dir).unwrap(); + let host_file = test_dir.join("created.txt"); + + let host_name_units = utf16(&host_nt_path(&host_file)); + let host_name = unicode_string(&host_name_units); + let host_attributes = host_object_attributes(&host_name); + let mut host_handle = core::ptr::null_mut(); + let mut host_io_status = IoStatusBlock::default(); + // SAFETY: All pointers reference live test locals, the optional pointer + // arguments are null, and ObjectName points to a path in the test directory. + let host_status = unsafe { + NtCreateFile( + &raw mut host_handle, + FILE_GENERIC_READ | FILE_GENERIC_WRITE | AccessMask::DELETE.bits(), + &raw const host_attributes, + &raw mut host_io_status, + core::ptr::null(), + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_SUPERSEDE, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + core::ptr::null(), + 0, + ) + }; + close_host_handle(host_handle); + + let task = crate::tests::test_task(); + let (_path, _name, attributes) = open_object_attributes("/tmp/supersede-created.txt"); + let mut litebox_handle = Handle::default(); + let mut litebox_io_status = IoStatusBlock::default(); + let litebox_status = task.sys_nt_create_file( + mut_ptr(&mut litebox_handle), + FILE_GENERIC_READ | FILE_GENERIC_WRITE | AccessMask::DELETE.bits(), + Some(const_ptr(&attributes)), + mut_ptr(&mut litebox_io_status), + None, + 0, + FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, + FILE_SUPERSEDE, + FileCreateOptions::SYNCHRONOUS_IO_NONALERT.bits(), + None, + 0, + ); + + assert_eq!(host_status, litebox_status.as_raw()); + assert_eq!(host_io_status.status, litebox_io_status.status); + assert_eq!(host_io_status.information, litebox_io_status.information); + } + } +} diff --git a/litebox_shim_windows/src/syscalls/file_path.rs b/litebox_shim_windows/src/syscalls/file_path.rs new file mode 100644 index 0000000000..5ac9e7a39c --- /dev/null +++ b/litebox_shim_windows/src/syscalls/file_path.rs @@ -0,0 +1,254 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::string::String; + +use litebox_common_windows::nt_status::NtStatus; + +use crate::syscalls::condrv::CondrvObject; +use crate::syscalls::object_manager::{FileDeviceObject, ObjectManager}; + +#[derive(Debug, Eq, PartialEq)] +pub(crate) enum FileTarget { + Filesystem(String), + Condrv(CondrvObject), +} + +pub(crate) enum FilePathRoot<'a> { + Namespace, + Filesystem { path: &'a str, is_directory: bool }, + Condrv(CondrvObject), +} + +pub(crate) struct FilePathResolver<'a, Platform: crate::ShimPlatform> { + object_manager: &'a ObjectManager, +} + +impl<'a, Platform: crate::ShimPlatform> FilePathResolver<'a, Platform> { + pub(crate) fn new(object_manager: &'a ObjectManager) -> Self { + Self { object_manager } + } + + pub(crate) fn resolve( + &self, + root: FilePathRoot<'_>, + name: &str, + ) -> Result { + match root { + FilePathRoot::Condrv(parent) => parent.relative_child(name).map(FileTarget::Condrv), + FilePathRoot::Namespace => self.resolve_absolute(name), + FilePathRoot::Filesystem { .. } if is_absolute_windows_path(name) => { + self.resolve_absolute(name) + } + FilePathRoot::Filesystem { + is_directory: false, + .. + } => Err(NtStatus::NOT_A_DIRECTORY), + FilePathRoot::Filesystem { path, .. } => { + join_absolute_components(path, name).map(FileTarget::Filesystem) + } + } + } + + fn resolve_absolute(&self, name: &str) -> Result { + if name.starts_with('/') { + return join_absolute_components("/", name).map(FileTarget::Filesystem); + } + if !is_absolute_windows_path(name) { + return Err(NtStatus::OBJECT_PATH_SYNTAX_BAD); + } + + let object_path = absolute_windows_file_name_to_object_path(name); + let (device, remaining) = self.object_manager.resolve_file_device(&object_path)?; + file_device_path_to_file_target(device, &remaining) + } +} + +fn absolute_windows_file_name_to_object_path(name: &str) -> String { + if let Some(rest) = strip_case_insensitive_prefix(name, "\\\\?\\") { + return alloc::format!(r"\??\{}", normalize_file_name_separators(rest)); + } + if name.starts_with('\\') { + return normalize_file_name_separators(name); + } + alloc::format!(r"\??\{}", normalize_file_name_separators(name)) +} + +fn normalize_file_name_separators(name: &str) -> String { + name.replace('/', "\\") +} + +fn file_device_path_to_file_target( + device: FileDeviceObject, + remaining: &str, +) -> Result { + match device { + FileDeviceObject::Filesystem { root_path } => { + join_absolute_components(&root_path, remaining).map(FileTarget::Filesystem) + } + FileDeviceObject::ConsoleDriver => { + CondrvObject::from_device_name(remaining).map(FileTarget::Condrv) + } + } +} + +fn join_absolute_components(root_path: &str, components: &str) -> Result { + let mut path = String::from(root_path.trim_end_matches('/')); + if path.is_empty() { + path.push('/'); + } + for component in components.split(['\\', '/']) { + if component.is_empty() || component == "." { + continue; + } + if component == ".." { + return Err(NtStatus::INVALID_PARAMETER); + } + if !path.ends_with('/') { + path.push('/'); + } + path.push_str(component); + } + Ok(path) +} + +fn strip_case_insensitive_prefix<'a>(value: &'a str, prefix: &str) -> Option<&'a str> { + value + .get(..prefix.len()) + .is_some_and(|head| head.eq_ignore_ascii_case(prefix)) + .then(|| &value[prefix.len()..]) +} + +fn is_absolute_windows_path(name: &str) -> bool { + name.starts_with(['\\', '/']) || is_absolute_windows_drive_path(name) +} + +fn is_absolute_windows_drive_path(name: &str) -> bool { + name.as_bytes() + .get(1..3) + .is_some_and(|bytes| bytes[0] == b':' && matches!(bytes[1], b'\\' | b'/')) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn namespace_paths_resolve_through_the_object_manager() { + let task = crate::tests::test_task(); + let resolver = FilePathResolver::new(&task.process.object_manager); + let resolve = |name| resolver.resolve(FilePathRoot::Namespace, name); + let resolve_path = |name| { + resolve(name).map(|target| match target { + FileTarget::Filesystem(path) => path, + FileTarget::Condrv(object) => String::from(object.handle_path()), + }) + }; + + assert_eq!( + resolve_path(r"\??\C:\Windows\System32\ntdll.dll").unwrap(), + "/Windows/System32/ntdll.dll" + ); + assert_eq!( + resolve_path(r"\??\c:\windows\system32\KERNEL32.DLL").unwrap(), + "/windows/system32/KERNEL32.DLL" + ); + assert_eq!( + resolve_path(r"\Device\HarddiskVolume1\Windows\System32\c_1252.NLS").unwrap(), + "/Windows/System32/c_1252.NLS" + ); + assert_eq!( + resolve_path(r"\SystemRoot\System32\kernel32.dll").unwrap(), + "/Windows/System32/kernel32.dll" + ); + assert_eq!( + resolve_path(r"\Device\ConDrv\Output").unwrap(), + "/dev/stdout" + ); + assert_eq!( + resolve_path(r"\Device\ConDrv\Reference"), + Err(NtStatus::INVALID_HANDLE) + ); + assert_eq!( + resolve_path(r"\Device\ConDrv\Connect"), + Err(NtStatus::OBJECT_TYPE_MISMATCH) + ); + assert_eq!( + resolve(r"\Missing\file.txt"), + Err(NtStatus::OBJECT_PATH_NOT_FOUND) + ); + assert_eq!( + resolve("/tmp/compatibility-path.txt"), + Ok(FileTarget::Filesystem(String::from( + "/tmp/compatibility-path.txt" + ))) + ); + assert_eq!( + resolve("/SystemRoot/not-an-object-path"), + Ok(FileTarget::Filesystem(String::from( + "/SystemRoot/not-an-object-path" + ))) + ); + assert_eq!( + resolve("relative.txt"), + Err(NtStatus::OBJECT_PATH_SYNTAX_BAD) + ); + } + + #[test] + fn root_kind_controls_relative_path_resolution() { + let task = crate::tests::test_task(); + let resolver = FilePathResolver::new(&task.process.object_manager); + + assert_eq!( + resolver.resolve( + FilePathRoot::Filesystem { + path: "/tmp/root", + is_directory: true, + }, + r"child\file.txt", + ), + Ok(FileTarget::Filesystem(String::from( + "/tmp/root/child/file.txt" + ))) + ); + assert_eq!( + resolver.resolve( + FilePathRoot::Filesystem { + path: "/tmp/root", + is_directory: true, + }, + r"C:\Windows\System32\ntdll.dll", + ), + Ok(FileTarget::Filesystem(String::from( + "/Windows/System32/ntdll.dll" + ))) + ); + assert_eq!( + resolver.resolve( + FilePathRoot::Filesystem { + path: "/tmp/root", + is_directory: true, + }, + r"MixedCase\File.TXT", + ), + Ok(FileTarget::Filesystem(String::from( + "/tmp/root/MixedCase/File.TXT" + ))) + ); + assert_eq!( + resolver.resolve( + FilePathRoot::Filesystem { + path: "/tmp/root.txt", + is_directory: false, + }, + "child.txt", + ), + Err(NtStatus::NOT_A_DIRECTORY) + ); + assert_eq!( + resolver.resolve(FilePathRoot::Condrv(CondrvObject::Reference), r"\Connect"), + Ok(FileTarget::Condrv(CondrvObject::Connect)) + ); + } +} diff --git a/litebox_shim_windows/src/syscalls/iocp.rs b/litebox_shim_windows/src/syscalls/iocp.rs new file mode 100644 index 0000000000..c4b8a008af --- /dev/null +++ b/litebox_shim_windows/src/syscalls/iocp.rs @@ -0,0 +1,318 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT I/O completion port syscalls. + +use alloc::sync::Arc; +use core::marker::PhantomData; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::{AccessMask, ObjectAttributes, read_object_attributes}; +use crate::syscalls::Handle; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct IoCompletionAccess: u32 { + const QUERY_STATE = 0x0001; + const MODIFY_STATE = 0x0002; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY_STATE.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() | Self::MODIFY_STATE.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() | AccessMask::SYNCHRONIZE.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::QUERY_STATE.bits() + | Self::MODIFY_STATE.bits(); + + const _ = !0; + } +} + +impl IoCompletionAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +pub(crate) struct IoCompletionSubsystem(PhantomData); + +impl FdEnabledSubsystem for IoCompletionSubsystem { + type Entry = IoCompletionHandleObject; +} + +impl FdEnabledSubsystemEntry for IoCompletionHandleObject {} + +impl crate::WindowsHandleSubsystem + for IoCompletionSubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + IoCompletionAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct IoCompletionHandleObject { + port: Arc>, +} + +pub(crate) struct IoCompletionObject { + _number_of_concurrent_threads: u32, + _not_send_without_platform: PhantomData, +} + +impl IoCompletionObject { + fn new(number_of_concurrent_threads: u32) -> Self { + Self { + _number_of_concurrent_threads: number_of_concurrent_threads, + _not_send_without_platform: PhantomData, + } + } +} + +impl IoCompletionHandleObject { + pub(crate) fn port(&self) -> Arc> { + self.port.clone() + } +} + +fn validate_io_completion_object_attributes( + object_attributes: Option>, +) -> Result<(), NtStatus> { + let Some(object_attributes) = object_attributes else { + return Ok(()); + }; + let object_attributes = read_object_attributes::(object_attributes)?; + if object_attributes.object_name == 0 && !object_attributes.root_directory.is_null() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(()) +} + +impl Task { + fn insert_io_completion_handle( + &self, + port: Arc>, + granted_access: IoCompletionAccess, + ) -> Result { + self.insert_typed_handle::>( + IoCompletionHandleObject { port }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_io_completion_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_io_completion(io_completion: IoCompletionHandleObject) { + drop(io_completion); + } + + pub(crate) fn sys_nt_create_io_completion( + &self, + io_completion_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + number_of_concurrent_threads: u32, + ) -> NtStatus { + if let Err(status) = + probe_guest_output_preserving_value::(io_completion_handle) + { + return status; + } + if let Err(status) = validate_io_completion_object_attributes::(object_attributes) + { + return status; + } + + // TODO: model the IOCP packet queue, concurrency accounting, named-object lookup, + // and file-handle association once completion posting/removal and file completion + // context syscalls are implemented. + let port = Arc::new(IoCompletionObject::new(number_of_concurrent_threads)); + let granted_access = IoCompletionAccess::from_desired_access(desired_access); + let Ok(handle) = self.insert_io_completion_handle(port, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if io_completion_handle.write_at_offset(0, handle).is_none() { + self.close_io_completion_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } +} + +#[cfg(test)] +mod tests { + use core::mem::size_of; + + use litebox::utils::TruncateExt as _; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::ObjectAttributes; + use crate::tests::{const_ptr, mut_ptr, test_task}; + + const IO_COMPLETION_ALL_ACCESS: u32 = 0x001f_0003; + + fn object_attributes_size() -> u32 { + size_of::().trunc() + } + + #[test] + fn create_validates_object_attributes_without_clobbering_output() { + let task = test_task(); + let mut handle = Handle::from_raw(usize::MAX); + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut handle), + IO_COMPLETION_ALL_ACCESS, + Some(const_ptr(&bad_length)), + 0, + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + + let root_without_name = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::from_raw(4), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut handle), + IO_COMPLETION_ALL_ACCESS, + Some(const_ptr(&root_without_name)), + 0, + ), + NtStatus::OBJECT_NAME_INVALID + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_create_io_completion_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateIoCompletion( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + number_of_concurrent_threads: u32, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, object attributes are null as accepted by native + // NtCreateIoCompletion, and the returned host handle is closed before leaving the test. + let host_success = unsafe { + let status = NtCreateIoCompletion( + &raw mut host_handle, + IO_COMPLETION_ALL_ACCESS, + core::ptr::null(), + 0, + ); + if status == NtStatus::SUCCESS.as_raw() && !host_handle.is_null() { + assert_eq!(NtClose(host_handle), NtStatus::SUCCESS.as_raw()); + } + status + }; + + let task = test_task(); + let mut shim_handle = Handle::default(); + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut shim_handle), + IO_COMPLETION_ALL_ACCESS, + None, + 0, + ) + .as_raw(), + host_success + ); + assert!(!shim_handle.is_null()); + + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + // SAFETY: The host output and attributes pointers are valid locals; the bad length is the + // parameter being tested. + let host_bad_length = unsafe { + NtCreateIoCompletion( + &raw mut host_handle, + IO_COMPLETION_ALL_ACCESS, + &raw const bad_length, + 0, + ) + }; + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut shim_handle), + IO_COMPLETION_ALL_ACCESS, + Some(const_ptr(&bad_length)), + 0, + ) + .as_raw(), + host_bad_length + ); + + let root_without_name = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::from_raw(4), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + // SAFETY: The host output and attributes pointers are valid locals; root without an object + // name is the probed native behavior. + let host_root_without_name = unsafe { + NtCreateIoCompletion( + &raw mut host_handle, + IO_COMPLETION_ALL_ACCESS, + &raw const root_without_name, + 0, + ) + }; + let mut shim_root_without_name_handle = Handle::from_raw(usize::MAX); + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut shim_root_without_name_handle), + IO_COMPLETION_ALL_ACCESS, + Some(const_ptr(&root_without_name)), + 0, + ) + .as_raw(), + host_root_without_name + ); + } +} diff --git a/litebox_shim_windows/src/syscalls/lpc.rs b/litebox_shim_windows/src/syscalls/lpc.rs new file mode 100644 index 0000000000..e9e11c94f3 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/lpc.rs @@ -0,0 +1,438 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::string::String; +use core::marker::PhantomData; +use core::mem::size_of; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use super::Handle; +use crate::nt_types::{ProcessEnvironmentBlock, ThreadEnvironmentBlock, UnicodeString}; +use crate::{ConstPtr, MutPtr, ShimFS, ShimPlatform, Task, probe_guest_output_preserving_value}; + +const CSR_MAX_MESSAGE_LENGTH: u32 = 0x148; +const CSR_SERVER_PROCESS_ID: usize = 1; +// TODO(csr-server-dll-names): report names once the CSR connect contract models them. +const CSR_NUMBER_OF_SERVER_DLL_NAMES: u32 = 0; + +pub(crate) struct LpcPortSubsystem(PhantomData); + +impl FdEnabledSubsystem for LpcPortSubsystem { + type Entry = LpcPortHandleObject; +} + +impl FdEnabledSubsystemEntry for LpcPortHandleObject {} + +impl crate::WindowsHandleSubsystem for LpcPortSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + desired_access + } + + fn resolve_duplicate_access( + _entry: &Self::Entry, + desired_access: u32, + ) -> Result { + Ok(desired_access & !crate::nt_types::AccessMask::MAXIMUM_ALLOWED.bits()) + } +} + +pub(crate) struct LpcPortHandleObject { + _port_name: String, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +pub(crate) struct SecurityQualityOfService { + length: u32, + impersonation_level: u32, + context_tracking_mode: u8, + effective_only: u8, + padding: [u8; 2], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +pub(crate) struct PortView { + length: u32, + padding: u32, + section_handle: Handle, + section_offset: u64, + view_size: usize, + view_base: usize, + view_remote_base: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +pub(crate) struct RemotePortView { + length: u32, + padding: u32, + view_size: usize, + view_base: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct CsrApiConnectInfo { + shared_section_base: usize, + shared_static_server_data: usize, + shared_section_heap: usize, + debug_flags: u32, + size_of_peb_data: u32, + size_of_teb_data: u32, + number_of_server_dll_names: u32, + server_process_id: usize, +} + +pub(crate) struct ConnectPortParameters { + pub(crate) port_handle: MutPtr, + pub(crate) port_name: ConstPtr, + pub(crate) security_qos: ConstPtr, + pub(crate) client_view: Option>, + pub(crate) server_view: Option>, + pub(crate) max_message_length: Option>, + pub(crate) connection_information: Option>, + pub(crate) connection_information_length: Option>, +} + +impl Task { + pub(crate) fn sys_nt_connect_port(&self, params: ConnectPortParameters) -> NtStatus { + if params.security_qos.read_at_offset(0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + let port_name = match params + .port_name + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION) + .and_then(UnicodeString::read_string::) + { + Ok(name) => name, + Err(status) => return status, + }; + if let Err(status) = self.process.object_manager.resolve_port(&port_name) { + return status; + } + + let Some(client_view) = params.client_view else { + return NtStatus::INVALID_PARAMETER; + }; + let Some(connection_information) = params.connection_information else { + return NtStatus::INVALID_PARAMETER; + }; + let Some(connection_information_length) = params.connection_information_length else { + return NtStatus::INVALID_PARAMETER; + }; + + let Some(client_view_value) = client_view.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if client_view_value.length as usize != size_of::() + || client_view_value.section_offset != 0 + || client_view_value.view_size == 0 + { + return NtStatus::INVALID_PARAMETER; + } + let server_view_value = match params.server_view { + Some(server_view) => match server_view.read_at_offset(0) { + Some(view) if view.length as usize == size_of::() => Some(view), + Some(_) => return NtStatus::INVALID_PARAMETER, + None => return NtStatus::ACCESS_VIOLATION, + }, + None => None, + }; + + let connection_info_len = match connection_information_length.read_at_offset(0) { + Some(length) => length as usize, + None => return NtStatus::ACCESS_VIOLATION, + }; + if connection_info_len != size_of::() { + return NtStatus::INFO_LENGTH_MISMATCH; + } + + if let Err(status) = probe_lpc_outputs::( + params.port_handle, + client_view, + params.server_view, + params.max_message_length, + connection_information, + connection_information_length, + connection_info_len, + ) { + return status; + } + + let Some(connect_info) = self.csr_api_connect_info() else { + return NtStatus::ACCESS_VIOLATION; + }; + let mapped_view = match self.map_client_port_section( + client_view_value.section_handle, + client_view_value.view_size, + ) { + Ok(mapped_view) => mapped_view, + Err(status) => return status, + }; + let port = LpcPortHandleObject { + _port_name: port_name.clone(), + }; + let handle = match self.insert_typed_handle::>(port, 0, drop) { + Ok(handle) => handle, + Err(status) => { + self.rollback_pagefile_section_view(mapped_view.base); + return status; + } + }; + + let mut written_client_view = client_view_value; + written_client_view.view_size = mapped_view.view_size; + written_client_view.view_base = mapped_view.base; + written_client_view.view_remote_base = mapped_view.base; + + let write_failed = client_view + .write_at_offset(0, written_client_view) + .is_none() + || params + .max_message_length + .is_some_and(|ptr| ptr.write_at_offset(0, CSR_MAX_MESSAGE_LENGTH).is_none()) + || connection_information + .write_slice_at_offset(0, connect_info.as_bytes()) + .is_none() + || connection_information_length + .write_at_offset(0, size_of::().trunc()) + .is_none() + || params.port_handle.write_at_offset(0, handle).is_none(); + if write_failed { + self.close_lpc_port_handle(handle); + self.rollback_pagefile_section_view(mapped_view.base); + return NtStatus::ACCESS_VIOLATION; + } + + if let (Some(server_view), Some(mut server_view_value)) = + (params.server_view, server_view_value) + { + server_view_value.view_size = mapped_view.mapped_size; + server_view_value.view_base = mapped_view.base; + if server_view.write_at_offset(0, server_view_value).is_none() { + self.close_lpc_port_handle(handle); + self.rollback_pagefile_section_view(mapped_view.base); + return NtStatus::ACCESS_VIOLATION; + } + } + + litebox_util_log::debug!( + port_name:% = port_name, + handle:% = format_args!("{:#x}", handle.as_raw()), + client_view_base:% = format_args!("{:#x}", mapped_view.base), + client_view_size = mapped_view.view_size; + "Handled NtConnectPort for CSR API port" + ); + NtStatus::SUCCESS + } + + pub(crate) fn close_lpc_port_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_lpc_port(port: LpcPortHandleObject) { + drop(port); + } + + fn csr_api_connect_info(&self) -> Option { + let read_only_shared_memory_base = crate::read_field_at_offset::( + self.process.peb_address, + core::mem::offset_of!(ProcessEnvironmentBlock, read_only_shared_memory_base), + )?; + let read_only_static_server_data = crate::read_field_at_offset::( + self.process.peb_address, + core::mem::offset_of!(ProcessEnvironmentBlock, read_only_static_server_data), + )?; + Some(CsrApiConnectInfo { + shared_section_base: read_only_shared_memory_base, + shared_static_server_data: read_only_static_server_data, + shared_section_heap: read_only_shared_memory_base, + debug_flags: 0, + size_of_peb_data: size_of::().trunc(), + size_of_teb_data: size_of::().trunc(), + number_of_server_dll_names: CSR_NUMBER_OF_SERVER_DLL_NAMES, + server_process_id: CSR_SERVER_PROCESS_ID, + }) + } +} + +fn probe_lpc_outputs( + port_handle: MutPtr, + client_view: MutPtr, + server_view: Option>, + max_message_length: Option>, + connection_information: MutPtr, + connection_information_length: MutPtr, + connection_information_len: usize, +) -> Result<(), NtStatus> { + probe_guest_output_preserving_value::(port_handle)?; + probe_guest_output_preserving_value::(client_view)?; + if let Some(server_view) = server_view { + probe_guest_output_preserving_value::(server_view)?; + } + if let Some(max_message_length) = max_message_length { + probe_guest_output_preserving_value::(max_message_length)?; + } + probe_guest_byte_buffer_preserving::( + connection_information, + connection_information_len, + size_of::(), + )?; + probe_guest_output_preserving_value::(connection_information_length) +} + +fn probe_guest_byte_buffer_preserving( + ptr: MutPtr, + len: usize, + max_len: usize, +) -> Result<(), NtStatus> { + if len > max_len { + return Err(NtStatus::INFO_LENGTH_MISMATCH); + } + let bytes = ptr.to_owned_slice(len).ok_or(NtStatus::ACCESS_VIOLATION)?; + ptr.write_slice_at_offset(0, bytes.as_ref()) + .ok_or(NtStatus::ACCESS_VIOLATION) +} + +#[cfg(test)] +mod tests { + use zerocopy::FromZeros as _; + + use super::*; + use crate::syscalls::mm::PageProtection; + use crate::syscalls::object_manager::WINDOWS_API_PORT; + use crate::tests::{ + TestFS, TestPlatform, const_ptr, mut_byte_ptr, mut_ptr, test_task, unicode_string, + utf16_units, + }; + + const SECTION_MAP_WRITE: u32 = 0x0002; + const SECTION_MAP_READ: u32 = 0x0004; + const SEC_COMMIT: u32 = 0x0800_0000; + + fn task_with_peb(peb: &mut ProcessEnvironmentBlock) -> Task { + let mut task = test_task(); + alloc::sync::Arc::get_mut(&mut task.process) + .expect("test task has a unique process reference") + .peb_address = core::ptr::from_mut(peb) as usize; + task + } + + fn security_qos() -> SecurityQualityOfService { + SecurityQualityOfService { + length: size_of::().trunc(), + impersonation_level: 2, + context_tracking_mode: 0, + effective_only: 1, + padding: [0; 2], + } + } + + fn api_port_name(value: &str) -> (alloc::vec::Vec, UnicodeString) { + let units = utf16_units(value); + let unicode = unicode_string(&units); + (units, unicode) + } + + fn empty_connect_info() -> CsrApiConnectInfo { + CsrApiConnectInfo { + shared_section_base: 0, + shared_static_server_data: 0, + shared_section_heap: 0, + debug_flags: 0, + size_of_peb_data: 0, + size_of_teb_data: 0, + number_of_server_dll_names: 0, + server_process_id: 0, + } + } + + fn create_client_section(task: &Task, access: u32) -> Handle { + let mut handle = Handle::default(); + let size = i64::try_from(crate::PAGE_SIZE).expect("test section size fits in i64"); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut handle), + access, + None, + Some(const_ptr(&size)), + PageProtection::PAGE_READWRITE.bits(), + SEC_COMMIT, + Handle::default(), + ), + NtStatus::SUCCESS + ); + handle + } + + #[test] + fn nt_connect_port_fills_csr_info_and_maps_client_section() { + let mut peb = ProcessEnvironmentBlock::new_zeroed(); + peb.read_only_shared_memory_base = 0x7000_0000; + peb.read_only_static_server_data = 0x7000_1000; + peb.csr_server_read_only_shared_memory_base = 0x7100_0000; + let task = task_with_peb(&mut peb); + let (_name_units, name) = api_port_name(WINDOWS_API_PORT); + let qos = security_qos(); + let section_handle = create_client_section(&task, SECTION_MAP_READ | SECTION_MAP_WRITE); + let mut handle = Handle::default(); + let mut client_view = PortView { + length: size_of::().trunc(), + padding: 0, + section_handle, + section_offset: 0, + view_size: crate::PAGE_SIZE, + view_base: 0, + view_remote_base: 0, + }; + let mut max_message_length = 0u32; + let mut connection_info = empty_connect_info(); + let mut connection_info_len = size_of::().trunc(); + + assert_eq!( + task.sys_nt_connect_port(ConnectPortParameters { + port_handle: mut_ptr(&mut handle), + port_name: const_ptr(&name), + security_qos: const_ptr(&qos), + client_view: Some(mut_ptr(&mut client_view)), + server_view: None, + max_message_length: Some(mut_ptr(&mut max_message_length)), + connection_information: Some(mut_byte_ptr(&mut connection_info)), + connection_information_length: Some(mut_ptr(&mut connection_info_len)), + }), + NtStatus::SUCCESS + ); + + assert!(!handle.is_null()); + assert_ne!(client_view.view_base, 0); + assert_eq!(client_view.view_remote_base, client_view.view_base); + assert_eq!(max_message_length, CSR_MAX_MESSAGE_LENGTH); + assert_eq!( + connection_info.shared_section_base, + peb.read_only_shared_memory_base + ); + assert_eq!( + connection_info.shared_static_server_data, + peb.read_only_static_server_data + ); + assert_ne!( + u64::try_from(connection_info.shared_section_base).unwrap(), + peb.csr_server_read_only_shared_memory_base + ); + assert_eq!( + connection_info.size_of_peb_data, + size_of::().trunc() + ); + assert_eq!( + connection_info.size_of_teb_data, + size_of::().trunc() + ); + } +} diff --git a/litebox_shim_windows/src/syscalls/mm.rs b/litebox_shim_windows/src/syscalls/mm.rs new file mode 100644 index 0000000000..919075073e --- /dev/null +++ b/litebox_shim_windows/src/syscalls/mm.rs @@ -0,0 +1,2889 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::mm::linux::{CreatePagesFlags, MappingError, NonZeroAddress, NonZeroPageSize}; +use litebox::platform::page_mgmt::{AllocationError, MemoryRegionPermissions}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox_common_windows::nt_status::NtStatus; +use rangemap::RangeMap; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::syscalls::ProcessHandle; +use crate::{ + ConstPtr, MutPtr, PAGE_SIZE, ShimFS, ShimPlatform, Task, WindowsPageManager, + WindowsVirtualAllocation, WindowsVirtualAllocations, +}; + +pub(super) const ALLOCATION_GRANULARITY: usize = 0x1_0000; +const ALLOCATION_SEARCH_ATTEMPTS: usize = 8; +const MEMORY_WORKING_SET_LIST_MIN_SIZE: usize = 16; +const MEM_EXTENDED_PARAMETER_TYPE_MASK: u64 = 0xff; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct PageProtection: u32 { + const PAGE_NOACCESS = 0x01; + const PAGE_READONLY = 0x02; + const PAGE_READWRITE = 0x04; + const PAGE_WRITECOPY = 0x08; + const PAGE_EXECUTE = 0x10; + const PAGE_EXECUTE_READ = 0x20; + const PAGE_EXECUTE_READWRITE = 0x40; + const PAGE_EXECUTE_WRITECOPY = 0x80; + const PAGE_GUARD = 0x100; + const PAGE_NOCACHE = 0x200; + const PAGE_WRITECOMBINE = 0x400; + } +} + +impl PageProtection { + pub(super) const BASE_MASK: u32 = 0xff; + + fn base(self) -> u32 { + self.bits() & Self::BASE_MASK + } + + fn has_valid_modifier_combination(self) -> bool { + let noaccess = self.base() == Self::PAGE_NOACCESS.bits(); + let guard = self.contains(Self::PAGE_GUARD); + let nocache = self.contains(Self::PAGE_NOCACHE); + let writecombine = self.contains(Self::PAGE_WRITECOMBINE); + + !(noaccess && (guard || nocache || writecombine) + || guard && (nocache || writecombine) + || nocache && writecombine) + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct AllocationType: u32 { + const MEM_COMMIT = 0x1000; + const MEM_RESERVE = 0x2000; + const MEM_RESET = 0x80000; + const MEM_TOP_DOWN = 0x100000; + const MEM_WRITE_WATCH = 0x200000; + const MEM_PHYSICAL = 0x400000; + const MEM_RESET_UNDO = 0x1000000; + const MEM_LARGE_PAGES = 0x20000000; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct FreeType: u32 { + const MEM_COALESCE_PLACEHOLDERS = 0x1; + const MEM_PRESERVE_PLACEHOLDER = 0x2; + const MEM_DECOMMIT = 0x4000; + const MEM_RELEASE = 0x8000; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct MemoryState: u32 { + const MEM_COMMIT = 0x1000; + const MEM_RESERVE = 0x2000; + const MEM_FREE = 0x10000; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct MemoryType: u32 { + const MEM_PRIVATE = 0x20000; + const MEM_MAPPED = 0x40000; + const MEM_IMAGE = 0x1000000; + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum MemoryInformationClass { + Basic = 0, + WorkingSetList = 4, + Image = 6, + ImageExtension = 14, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Default, FromBytes, Immutable, IntoBytes)] +struct MemoryImageInformation { + image_base: usize, + size_of_image: usize, + image_flags: u32, + _padding: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, FromBytes, Immutable, IntoBytes)] +struct MemoryImageExtensionInformation { + extension_type: u32, + flags: u32, + extension_image_base_rva: usize, + extension_size: usize, +} + +#[repr(u64)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum MemoryExtendedParameterType { + AddressRequirements = 1, + NumaNode = 2, + AttributeFlags = 5, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Default, FromBytes, Immutable, IntoBytes)] +struct MemoryBasicInformation { + base_address: usize, + allocation_base: usize, + allocation_protect: u32, + partition_id: u16, + _padding0: u16, + region_size: usize, + state: u32, + protect: u32, + type_: u32, + _padding1: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +pub(crate) struct MemoryExtendedParameter { + type_: u64, + value: usize, +} + +pub(crate) struct MemoryExtendedParameters { + pub(crate) parameters: Option>, + pub(crate) count: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct MemoryAddressRequirements { + lowest_starting_address: usize, + highest_ending_address: usize, + alignment: usize, +} + +fn validate_memory_extended_parameters( + extended_parameters: MemoryExtendedParameters, +) -> Result<(), NtStatus> { + if extended_parameters.count == 0 { + return Ok(()); + } + + let Some(parameters) = extended_parameters.parameters else { + return Err(NtStatus::INVALID_PARAMETER); + }; + + let mut present = 0u32; + for index in 0..extended_parameters.count { + let parameter = parameters + .read_at_offset(index.try_into().map_err(|_| NtStatus::INVALID_PARAMETER)?) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + validate_memory_extended_parameter::(parameter, &mut present)?; + } + + Ok(()) +} + +fn validate_memory_extended_parameter( + parameter: MemoryExtendedParameter, + present: &mut u32, +) -> Result<(), NtStatus> { + if parameter.type_ & !MEM_EXTENDED_PARAMETER_TYPE_MASK != 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + + let parameter_type_raw = parameter.type_ & MEM_EXTENDED_PARAMETER_TYPE_MASK; + let parameter_type = MemoryExtendedParameterType::try_from(parameter_type_raw) + .map_err(|_| NtStatus::INVALID_PARAMETER)?; + let parameter_bit = u32::try_from(parameter_type_raw) + .ok() + .and_then(|parameter_type| 1u32.checked_shl(parameter_type)) + .ok_or(NtStatus::INVALID_PARAMETER)?; + if *present & parameter_bit != 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + *present |= parameter_bit; + + match parameter_type { + MemoryExtendedParameterType::AddressRequirements => { + let address_requirements = + ConstPtr::::from_usize(parameter.value) + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + if address_requirements.lowest_starting_address != 0 + || address_requirements.highest_ending_address != 0 + || !matches!(address_requirements.alignment, 0 | ALLOCATION_GRANULARITY) + { + return Err(NtStatus::INVALID_PARAMETER); + } + Ok(()) + } + MemoryExtendedParameterType::NumaNode => Ok(()), + MemoryExtendedParameterType::AttributeFlags => { + if parameter.value == 0 { + Ok(()) + } else { + Err(NtStatus::INVALID_PARAMETER) + } + } + } +} + +impl Task { + pub(crate) fn sys_nt_allocate_virtual_memory_ex( + &self, + process_handle: ProcessHandle, + base_address: MutPtr, + region_size: MutPtr, + allocation_type: u32, + protect: u32, + extended_parameters: MemoryExtendedParameters, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + if let Err(status) = validate_memory_extended_parameters::(extended_parameters) { + return status; + } + + // TODO: Apply supported extended parameters (especially MEM_ADDRESS_REQUIREMENTS) to the + // allocation search once PageManager can honor caller-specified placement constraints. + self.sys_nt_allocate_virtual_memory( + process_handle, + base_address, + 0, + region_size, + allocation_type, + protect, + ) + } + + pub(crate) fn sys_nt_allocate_virtual_memory( + &self, + process_handle: ProcessHandle, + base_address: MutPtr, + zero_bits: usize, + region_size: MutPtr, + allocation_type: u32, + protect: u32, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + let Some(base) = base_address.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(size) = region_size.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if base_address.write_at_offset(0, base).is_none() + || region_size.write_at_offset(0, size).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + let Some(allocation_type) = AllocationType::from_bits(allocation_type) else { + return NtStatus::INVALID_PARAMETER; + }; + let supported_allocation_types = AllocationType::MEM_COMMIT + | AllocationType::MEM_RESERVE + | AllocationType::MEM_RESET + | AllocationType::MEM_TOP_DOWN; + if size == 0 + || !supported_allocation_types.contains(allocation_type) + || (zero_bits > 21 && zero_bits < 32) + || (zero_bits != 0 && base != 0) + { + return NtStatus::INVALID_PARAMETER; + } + if allocation_type.contains(AllocationType::MEM_RESET) { + if allocation_type != AllocationType::MEM_RESET { + return NtStatus::INVALID_PARAMETER; + } + return self.reset_virtual_memory(base, size, protect, base_address, region_size); + } + if !allocation_type.intersects(AllocationType::MEM_COMMIT | AllocationType::MEM_RESERVE) { + return NtStatus::INVALID_PARAMETER; + } + + let new_allocation = base == 0 || allocation_type.contains(AllocationType::MEM_RESERVE); + let Some((aligned_base, aligned_len)) = (if new_allocation { + reserve_allocation_region(base, size) + } else { + page_aligned_region(base, size) + }) else { + return NtStatus::INVALID_PARAMETER; + }; + let Some((protect, permissions)) = parse_page_protection(protect) else { + return NtStatus::INVALID_PAGE_PROTECTION; + }; + + if !new_allocation { + return self.commit_existing_virtual_memory( + aligned_base, + aligned_len, + protect, + permissions, + base_address, + region_size, + ); + } + + let Some(length) = NonZeroPageSize::new(aligned_len) else { + return NtStatus::INVALID_PARAMETER; + }; + let initial_permissions = if allocation_type.contains(AllocationType::MEM_COMMIT) { + permissions + } else { + MemoryRegionPermissions::empty() + }; + let top_down = allocation_type.contains(AllocationType::MEM_TOP_DOWN); + let allocation = if base == 0 { + create_allocation_granularity_aligned_pages::( + &self.global.page_manager, + length, + initial_permissions, + zero_bits, + top_down, + ) + } else { + create_pages::( + &self.global.page_manager, + NonZeroAddress::new(aligned_base), + length, + CreatePagesFlags::FIXED_ADDR | CreatePagesFlags::NOREPLACE, + initial_permissions, + |_| Ok(0), + ) + .map_err(mapping_error_to_nt_status) + }; + let ptr = match allocation { + Ok(ptr) => ptr, + Err(status) => return status, + }; + + if base_address.write_at_offset(0, ptr.as_usize()).is_none() + || region_size.write_at_offset(0, aligned_len).is_none() + { + let ptr = MutPtr::::from_usize(ptr.as_usize()); + // SAFETY: The mapping was just created by this syscall and has not been published in + // the allocation table. Removing it rolls back failed output writeback. + let _ = unsafe { self.global.page_manager.remove_pages(ptr, aligned_len) }; + return NtStatus::ACCESS_VIOLATION; + } + self.process.virtual_allocations.write().insert( + ptr.as_usize(), + WindowsVirtualAllocation { + base: ptr.as_usize(), + size: aligned_len, + allocation_protect: protect, + type_: MemoryType::MEM_PRIVATE, + pages: if allocation_type.contains(AllocationType::MEM_COMMIT) { + committed_pages(ptr.as_usize(), aligned_len, protect) + } else { + RangeMap::new() + }, + }, + ); + + litebox_util_log::debug!( + base:% = format_args!("{:#x}", base), + aligned_base:% = format_args!("{:#x}", ptr.as_usize()), + aligned_len, + allocation_type:% = format_args!("{:#x}", allocation_type.bits()), + protect:% = format_args!("{:#x}", protect.bits()); + "Handled NtAllocateVirtualMemory syscall" + ); + NtStatus::SUCCESS + } + + fn reset_virtual_memory( + &self, + base: usize, + size: usize, + protect: u32, + base_address: MutPtr, + region_size: MutPtr, + ) -> NtStatus { + if parse_page_protection(protect).is_none() { + return NtStatus::INVALID_PAGE_PROTECTION; + } + let Some((aligned_base, aligned_len)) = page_aligned_region(base, size) else { + return NtStatus::INVALID_PARAMETER; + }; + let Some(allocation) = + find_virtual_allocation(&self.process.virtual_allocations, aligned_base, aligned_len) + else { + return NtStatus::INVALID_PARAMETER; + }; + if allocation.type_ != MemoryType::MEM_PRIVATE { + return NtStatus::INVALID_PARAMETER; + } + if !matches!( + scan_allocation_pages(&allocation, aligned_base, aligned_len), + Some(PageRangeScan::FullyCommitted(_)) + ) { + return NtStatus::CONFLICTING_ADDRESSES; + } + + if base_address.write_at_offset(0, aligned_base).is_none() + || region_size.write_at_offset(0, aligned_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + fn commit_existing_virtual_memory( + &self, + aligned_base: usize, + aligned_len: usize, + protect: PageProtection, + permissions: MemoryRegionPermissions, + base_address: MutPtr, + region_size: MutPtr, + ) -> NtStatus { + if find_private_virtual_allocation( + &self.process.virtual_allocations, + aligned_base, + aligned_len, + ) + .is_none() + { + return NtStatus::INVALID_PARAMETER; + } + if update_permissions( + &self.global.page_manager, + aligned_base, + aligned_len, + permissions, + ) + .is_err() + { + return NtStatus::INVALID_PARAMETER; + } + if base_address.write_at_offset(0, aligned_base).is_none() + || region_size.write_at_offset(0, aligned_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + set_committed_pages_protect( + &self.process.virtual_allocations, + aligned_base, + aligned_len, + protect, + ); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_free_virtual_memory( + &self, + process_handle: ProcessHandle, + base_address: MutPtr, + region_size: MutPtr, + free_type: u32, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + let Some(base) = base_address.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(size) = region_size.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if base_address.write_at_offset(0, base).is_none() + || region_size.write_at_offset(0, size).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + let Some(free_type) = FreeType::from_bits(free_type) else { + return NtStatus::INVALID_PARAMETER; + }; + if base == 0 || !matches!(free_type, FreeType::MEM_DECOMMIT | FreeType::MEM_RELEASE) { + return NtStatus::INVALID_PARAMETER; + } + + let Some((aligned_base, aligned_len)) = + free_region(&self.process.virtual_allocations, base, size, free_type) + else { + return NtStatus::INVALID_PARAMETER; + }; + let ptr = MutPtr::::from_usize(aligned_base); + if free_type == FreeType::MEM_DECOMMIT { + // SAFETY: The range is page-aligned and belongs to a private allocation tracked for + // this process. Decommit discards page contents while leaving the address range + // reserved for later recommit. + if unsafe { self.global.page_manager.reset_pages(ptr, aligned_len, true) }.is_err() { + return NtStatus::UNABLE_TO_FREE_VM; + } + if update_permissions( + &self.global.page_manager, + aligned_base, + aligned_len, + MemoryRegionPermissions::empty(), + ) + .is_err() + { + return NtStatus::UNABLE_TO_FREE_VM; + } + mark_pages_decommitted(&self.process.virtual_allocations, aligned_base, aligned_len); + } else { + // SAFETY: The range is page-aligned and belongs to an allocation tracked for this + // process. The guest requested release, so the pages must not be used after success. + if unsafe { self.global.page_manager.remove_pages(ptr, aligned_len) }.is_err() { + return NtStatus::UNABLE_TO_FREE_VM; + } + self.process + .virtual_allocations + .write() + .remove(&aligned_base); + } + if base_address.write_at_offset(0, aligned_base).is_none() + || region_size.write_at_offset(0, aligned_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_protect_virtual_memory( + &self, + process_handle: ProcessHandle, + base_address: MutPtr, + region_size: MutPtr, + new_protect: u32, + old_protect: MutPtr, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + let Some(base) = base_address.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(size) = region_size.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(old_protect_probe) = old_protect.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if base_address.write_at_offset(0, base).is_none() + || region_size.write_at_offset(0, size).is_none() + || old_protect.write_at_offset(0, old_protect_probe).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if base == 0 || size == 0 { + return NtStatus::INVALID_PARAMETER; + } + let Some((aligned_base, aligned_len)) = page_aligned_region(base, size) else { + return NtStatus::INVALID_PARAMETER; + }; + let Some((new_protect, new_permissions)) = parse_page_protection(new_protect) else { + return NtStatus::INVALID_PAGE_PROTECTION; + }; + let old_protect_value = match scan_protect_range( + &self.process.virtual_allocations, + aligned_base, + aligned_len, + ) { + Some(PageRangeScan::FullyCommitted(first_protect)) => first_protect, + Some(PageRangeScan::ContainsUncommitted) => { + if old_protect + .write_at_offset(0, PageProtection::PAGE_NOACCESS.bits()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + return NtStatus::NOT_COMMITTED; + } + None => return NtStatus::NOT_COMMITTED, + }; + + if update_permissions( + &self.global.page_manager, + aligned_base, + aligned_len, + new_permissions, + ) + .is_err() + { + return NtStatus::ACCESS_VIOLATION; + } + set_committed_pages_protect( + &self.process.virtual_allocations, + aligned_base, + aligned_len, + new_protect, + ); + + // ReactOS NtProtectVirtualMemory writes OldProtection, BaseAddress, then RegionSize after + // MiProtectVirtualMemory succeeds; failed writeback does not roll back the protection. + if old_protect + .write_at_offset(0, old_protect_value.bits()) + .is_none() + || base_address.write_at_offset(0, aligned_base).is_none() + || region_size.write_at_offset(0, aligned_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + litebox_util_log::debug!( + process_handle:? = process_handle, + base:% = format_args!("{:#x}", base), + size = size, + aligned_base:% = format_args!("{:#x}", aligned_base), + aligned_len = aligned_len, + new_protect:% = format_args!("{:#x}", new_protect), + old_protect:% = format_args!("{:#x}", old_protect_value); + "Handled NtProtectVirtualMemory syscall" + ); + + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_virtual_memory( + &self, + process_handle: ProcessHandle, + base_address: usize, + memory_information_class: u32, + memory_information: MutPtr, + memory_information_length: usize, + return_length: Option>, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + let Ok(memory_information_class) = + MemoryInformationClass::try_from(memory_information_class) + else { + return NtStatus::INVALID_INFO_CLASS; + }; + + match memory_information_class { + MemoryInformationClass::Basic => self.write_memory_basic_information( + base_address, + memory_information, + memory_information_length, + return_length, + ), + MemoryInformationClass::WorkingSetList => Self::write_memory_working_set_list( + memory_information, + memory_information_length, + return_length, + ), + MemoryInformationClass::Image => self.write_memory_image_information( + process_handle, + base_address, + memory_information, + memory_information_length, + return_length, + ), + MemoryInformationClass::ImageExtension => self + .write_memory_image_extension_information( + process_handle, + base_address, + memory_information, + memory_information_length, + return_length, + ), + } + } + + fn write_memory_basic_information( + &self, + base_address: usize, + memory_information: MutPtr, + memory_information_length: usize, + return_length: Option>, + ) -> NtStatus { + if let Err(status) = check_and_write_length::( + return_length, + memory_information_length, + size_of::(), + ) { + return status; + } + + let Some(info) = query_memory_basic_information::( + &self.global.page_manager, + &self.process.virtual_allocations, + base_address, + ) else { + return NtStatus::INVALID_PARAMETER; + }; + let output = + MutPtr::::from_usize(memory_information.as_usize()); + if output.write_at_offset(0, info).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + fn write_memory_image_information( + &self, + process_handle: ProcessHandle, + base_address: usize, + memory_information: MutPtr, + memory_information_length: usize, + return_length: Option>, + ) -> NtStatus { + if let Err(status) = check_and_write_length::( + return_length, + memory_information_length, + size_of::(), + ) { + return status; + } + + let Some(allocation) = + find_image_allocation_containing(&self.process.virtual_allocations, base_address) + else { + return NtStatus::INVALID_PARAMETER; + }; + + let info = MemoryImageInformation { + image_base: allocation.base, + size_of_image: allocation.size, + image_flags: 0, + _padding: 0, + }; + let output = + MutPtr::::from_usize(memory_information.as_usize()); + if output.write_at_offset(0, info).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + + litebox_util_log::debug!( + process_handle:? = process_handle, + base:% = format_args!("{base_address:#x}"), + image_base:% = format_args!("{:#x}", allocation.base), + image_size = allocation.size; + "Handled NtQueryVirtualMemory MemoryImageInformation syscall" + ); + + NtStatus::SUCCESS + } + + fn write_memory_image_extension_information( + &self, + process_handle: ProcessHandle, + base_address: usize, + memory_information: MutPtr, + memory_information_length: usize, + return_length: Option>, + ) -> NtStatus { + if let Err(status) = check_and_write_length::( + return_length, + memory_information_length, + size_of::(), + ) { + return status; + } + + let Some(allocation) = + find_image_allocation_containing(&self.process.virtual_allocations, base_address) + else { + return NtStatus::INVALID_PARAMETER; + }; + + let image_extension_information = + MutPtr::::from_usize( + memory_information.as_usize(), + ); + let Some(request) = image_extension_information.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + // TODO: The buffer is an input request before it becomes output; only the default request for + // absent image extension information is supported for now. + if request != MemoryImageExtensionInformation::default() { + return NtStatus::INVALID_PARAMETER; + } + + // TODO: Report real image extension metadata when PE image extension data is modeled. + if image_extension_information + .write_at_offset(0, MemoryImageExtensionInformation::default()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + litebox_util_log::debug!( + process_handle:? = process_handle, + base:% = format_args!("{base_address:#x}"), + image_base:% = format_args!("{:#x}", allocation.base), + image_size = allocation.size; + "Handled NtQueryVirtualMemory MemoryImageExtensionInformation syscall" + ); + + NtStatus::SUCCESS + } + + fn write_memory_working_set_list( + memory_information: MutPtr, + memory_information_length: usize, + return_length: Option>, + ) -> NtStatus { + if memory_information_length < MEMORY_WORKING_SET_LIST_MIN_SIZE { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if let Some(return_length) = return_length + && return_length + .write_at_offset(0, memory_information_length) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + // TODO: Model working set residency and report real entries instead of an empty list. + for offset in 0..memory_information_length { + let Ok(offset) = isize::try_from(offset) else { + return NtStatus::INVALID_PARAMETER; + }; + if memory_information.write_at_offset(offset, 0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + } + + litebox_util_log::debug!( + memory_information_length; + "Handled NtQueryVirtualMemory MemoryWorkingSetList syscall" + ); + + NtStatus::SUCCESS + } +} + +fn check_and_write_length( + return_length: Option>, + memory_information_length: usize, + required_len: usize, +) -> Result<(), NtStatus> { + if let Some(return_length) = return_length + && return_length.write_at_offset(0, required_len).is_none() + { + return Err(NtStatus::ACCESS_VIOLATION); + } + if memory_information_length < required_len { + return Err(NtStatus::INFO_LENGTH_MISMATCH); + } + Ok(()) +} + +fn page_aligned_region(base: usize, size: usize) -> Option<(usize, usize)> { + let aligned_base = base & !(PAGE_SIZE - 1); + let end = base.checked_add(size)?; + let aligned_end = end.checked_add(PAGE_SIZE - 1)? & !(PAGE_SIZE - 1); + let aligned_len = aligned_end.checked_sub(aligned_base)?; + if aligned_base == 0 || aligned_len == 0 { + return None; + } + Some((aligned_base, aligned_len)) +} + +fn reserve_allocation_region(base: usize, size: usize) -> Option<(usize, usize)> { + let aligned_base = if base == 0 { + 0 + } else { + base & !(ALLOCATION_GRANULARITY - 1) + }; + if base != 0 && aligned_base == 0 { + return None; + } + let end = base.checked_add(size)?; + let aligned_end = end.checked_add(PAGE_SIZE - 1)? & !(PAGE_SIZE - 1); + let aligned_len = aligned_end.checked_sub(aligned_base)?; + if aligned_len == 0 { + return None; + } + Some((aligned_base, aligned_len)) +} + +fn free_region( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, + free_type: FreeType, +) -> Option<(usize, usize)> { + if size == 0 { + let allocation = virtual_allocations + .read() + .get(&base) + .filter(|allocation| allocation.type_ == MemoryType::MEM_PRIVATE) + .cloned()?; + return Some((allocation.base, allocation.size)); + } + + if free_type == FreeType::MEM_RELEASE { + return None; + } + + let (aligned_base, aligned_len) = page_aligned_region(base, size)?; + find_private_virtual_allocation(virtual_allocations, aligned_base, aligned_len)?; + Some((aligned_base, aligned_len)) +} + +fn committed_pages( + base: usize, + size: usize, + protect: PageProtection, +) -> RangeMap { + let mut pages = RangeMap::new(); + let Some(end) = base.checked_add(size) else { + return pages; + }; + pages.insert(base..end, protect); + pages +} + +fn find_virtual_allocation( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, +) -> Option { + let end = base.checked_add(size)?; + virtual_allocations + .read() + .range(..=base) + .next_back() + .map(|(_, allocation)| allocation.clone()) + .filter(|allocation| { + allocation + .base + .checked_add(allocation.size) + .is_some_and(|allocation_end| end <= allocation_end) + }) +} + +fn find_private_virtual_allocation( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, +) -> Option { + find_virtual_allocation(virtual_allocations, base, size) + .filter(|allocation| allocation.type_ == MemoryType::MEM_PRIVATE) +} + +fn find_image_allocation_containing( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, +) -> Option { + find_virtual_allocation(virtual_allocations, base, 1) + .filter(|allocation| allocation.type_ == MemoryType::MEM_IMAGE) +} + +fn find_virtual_allocation_containing( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, +) -> Option { + find_virtual_allocation(virtual_allocations, base, 1) +} + +enum PageRangeScan { + FullyCommitted(PageProtection), + ContainsUncommitted, +} + +fn scan_allocation_pages( + allocation: &WindowsVirtualAllocation, + base: usize, + size: usize, +) -> Option { + let end = base.checked_add(size)?; + let allocation_end = allocation.base.checked_add(allocation.size)?; + let scan_end = end.min(allocation_end); + let mut first_protect = None; + let mut cursor = base; + for (range, protect) in allocation.pages.overlapping(base..scan_end) { + let range_start = range.start.max(base); + if cursor < range_start { + return Some(PageRangeScan::ContainsUncommitted); + } + first_protect.get_or_insert(*protect); + cursor = cursor.max(range.end.min(scan_end)); + if cursor == end { + break; + } + } + + if cursor == end { + Some(PageRangeScan::FullyCommitted(first_protect?)) + } else { + Some(PageRangeScan::ContainsUncommitted) + } +} + +fn scan_protect_range( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, +) -> Option { + let allocation = find_virtual_allocation_containing(virtual_allocations, base)?; + scan_allocation_pages(&allocation, base, size) +} + +fn set_committed_pages_protect( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, + protect: PageProtection, +) { + let Some(end) = base.checked_add(size) else { + return; + }; + let mut allocations = virtual_allocations.write(); + let Some((_, allocation)) = allocations.range_mut(..=base).next_back() else { + return; + }; + allocation.pages.insert(base..end, protect); +} + +fn mark_pages_decommitted( + virtual_allocations: &WindowsVirtualAllocations, + base: usize, + size: usize, +) { + let Some(end) = base.checked_add(size) else { + return; + }; + let mut allocations = virtual_allocations.write(); + let Some((_, allocation)) = allocations.range_mut(..=base).next_back() else { + return; + }; + allocation.pages.remove(base..end); +} + +pub(super) fn parse_page_protection( + protect: u32, +) -> Option<(PageProtection, MemoryRegionPermissions)> { + let protect = PageProtection::from_bits(protect)?; + let permissions = page_protect_to_permissions(protect)?; + Some((protect, permissions)) +} + +fn page_protect_to_permissions(protect: PageProtection) -> Option { + if !protect.has_valid_modifier_combination() { + return None; + } + + match protect.base() { + value if value == PageProtection::PAGE_NOACCESS.bits() => { + Some(MemoryRegionPermissions::empty()) + } + value if value == PageProtection::PAGE_READONLY.bits() => { + Some(MemoryRegionPermissions::READ) + } + value + if value == PageProtection::PAGE_READWRITE.bits() + || value == PageProtection::PAGE_WRITECOPY.bits() => + { + Some(MemoryRegionPermissions::READ | MemoryRegionPermissions::WRITE) + } + value + if value == PageProtection::PAGE_EXECUTE.bits() + || value == PageProtection::PAGE_EXECUTE_READ.bits() => + { + Some(MemoryRegionPermissions::READ | MemoryRegionPermissions::EXEC) + } + value + if value == PageProtection::PAGE_EXECUTE_READWRITE.bits() + || value == PageProtection::PAGE_EXECUTE_WRITECOPY.bits() => + { + Some( + MemoryRegionPermissions::READ + | MemoryRegionPermissions::WRITE + | MemoryRegionPermissions::EXEC, + ) + } + _ => None, + } +} + +fn permissions_to_page_protect(permissions: MemoryRegionPermissions) -> PageProtection { + match ( + permissions.contains(MemoryRegionPermissions::READ), + permissions.contains(MemoryRegionPermissions::WRITE), + permissions.contains(MemoryRegionPermissions::EXEC), + ) { + (false, false, false) => PageProtection::PAGE_NOACCESS, + (true, false, false) => PageProtection::PAGE_READONLY, + (_, true, false) => PageProtection::PAGE_READWRITE, + (false, false, true) => PageProtection::PAGE_EXECUTE, + (true, false, true) => PageProtection::PAGE_EXECUTE_READ, + (_, true, true) => PageProtection::PAGE_EXECUTE_READWRITE, + } +} + +pub(super) fn create_pages( + page_manager: &WindowsPageManager, + suggested_address: Option>, + length: NonZeroPageSize, + flags: CreatePagesFlags, + permissions: MemoryRegionPermissions, + op: impl FnOnce(MutPtr) -> Result, +) -> Result, MappingError> { + // SAFETY: This creates guest mappings through the LiteBox page manager. The caller controls + // fixed-address behavior, and `op` only initializes the new mapping before it is exposed. + unsafe { + match permissions { + permissions if permissions.is_empty() => { + page_manager.create_inaccessible_pages(suggested_address, length, flags, op) + } + MemoryRegionPermissions::READ => { + page_manager.create_readable_pages(suggested_address, length, flags, op) + } + permissions + if permissions + == MemoryRegionPermissions::READ | MemoryRegionPermissions::WRITE => + { + page_manager.create_writable_pages(suggested_address, length, flags, op) + } + permissions + if permissions == MemoryRegionPermissions::READ | MemoryRegionPermissions::EXEC => + { + page_manager.create_executable_pages(suggested_address, length, flags, op) + } + permissions + if permissions + == MemoryRegionPermissions::READ + | MemoryRegionPermissions::WRITE + | MemoryRegionPermissions::EXEC => + { + let ptr = + page_manager.create_writable_pages(suggested_address, length, flags, op)?; + page_manager + .make_pages_rwx(ptr, length.as_usize()) + .map_err(|_| MappingError::OutOfMemory)?; + Ok(ptr) + } + _ => unreachable!("Windows page protection parser produced unsupported permissions"), + } + } +} + +enum HoleSearchResult { + Allocated(MutPtr), + RetryWithFreshMappings, + Exhausted, +} + +fn create_aligned_pages_in_hole( + page_manager: &WindowsPageManager, + hole_start: usize, + hole_end: usize, + length: NonZeroPageSize, + permissions: MemoryRegionPermissions, + top_down: bool, +) -> Result, MappingError> { + let Some(mut candidate) = + allocation_granularity_aligned_candidate(hole_start, hole_end, length.as_usize(), top_down) + else { + return Ok(HoleSearchResult::Exhausted); + }; + + loop { + match create_pages( + page_manager, + NonZeroAddress::new(candidate), + length, + CreatePagesFlags::FIXED_ADDR | CreatePagesFlags::NOREPLACE, + permissions, + |_| Ok(0), + ) { + Ok(ptr) => return Ok(HoleSearchResult::Allocated(ptr)), + Err(MappingError::MapError(AllocationError::AddressInUse)) => { + return Ok(HoleSearchResult::RetryWithFreshMappings); + } + Err(MappingError::MapError(AllocationError::AddressInUseByPlatform)) => {} + Err(error) => return Err(error), + } + + let Some(next_candidate) = next_allocation_granularity_candidate( + candidate, + length.as_usize(), + hole_start, + hole_end, + top_down, + ) else { + return Ok(HoleSearchResult::Exhausted); + }; + candidate = next_candidate; + } +} + +fn next_allocation_granularity_candidate( + candidate: usize, + length: usize, + hole_start: usize, + hole_end: usize, + top_down: bool, +) -> Option { + if top_down { + candidate + .checked_sub(ALLOCATION_GRANULARITY) + .filter(|next| *next >= hole_start) + } else { + let next_candidate = candidate.checked_add(ALLOCATION_GRANULARITY)?; + let next_end = next_candidate.checked_add(length)?; + (next_end <= hole_end).then_some(next_candidate) + } +} + +fn zero_bits_address_limit(zero_bits: usize) -> Option { + if zero_bits > 32 { + // NtAllocateVirtualMemory treats ZeroBits as a bitmask when > 32. + zero_bits.checked_add(1) + } else if zero_bits < usize::BITS as usize { + let shift = (usize::BITS as usize - zero_bits).try_into().ok()?; + 1usize.checked_shl(shift) + } else { + None + } +} + +fn mapping_error_to_nt_status(error: MappingError) -> NtStatus { + match error { + MappingError::UnAligned + | MappingError::BadFD(_) + | MappingError::NotAFile + | MappingError::NotForReading + | MappingError::MapError( + AllocationError::Unaligned + | AllocationError::BelowMinAddress + | AllocationError::AboveMaxAddress, + ) => NtStatus::INVALID_PARAMETER, + MappingError::MapError( + AllocationError::AddressInUse + | AllocationError::AddressInUseByPlatform + | AllocationError::AddressPartiallyInUse, + ) => NtStatus::CONFLICTING_ADDRESSES, + MappingError::OutOfMemory | MappingError::MapError(AllocationError::OutOfMemory) | _ => { + NtStatus::NO_MEMORY + } + } +} + +fn create_allocation_granularity_aligned_pages( + page_manager: &WindowsPageManager, + length: NonZeroPageSize, + permissions: MemoryRegionPermissions, + zero_bits: usize, + top_down: bool, +) -> Result, NtStatus> { + let mut max_start = Platform::TASK_ADDR_MAX + .checked_sub(length.as_usize()) + .ok_or(NtStatus::NO_MEMORY)?; + if let Some(limit) = zero_bits_address_limit(zero_bits) { + max_start = max_start.min( + limit + .checked_sub(length.as_usize()) + .ok_or(NtStatus::NO_MEMORY)?, + ); + } + let min_start = Platform::TASK_ADDR_MIN.next_multiple_of(ALLOCATION_GRANULARITY); + let search_end = max_start + .checked_add(length.as_usize()) + .ok_or(NtStatus::NO_MEMORY)?; + + // TODO: consider adding support for different allocation strategies and granularity to page manager + 'search: for _ in 0..ALLOCATION_SEARCH_ATTEMPTS { + let mut mappings = page_manager.mappings(); + mappings.sort_by_key(|(range, _)| range.start); + + if top_down { + let mut hole_end = search_end; + for (range, _) in mappings.iter().rev() { + if range.end <= min_start { + break; + } + if range.start >= search_end { + continue; + } + if range.end < hole_end { + match create_aligned_pages_in_hole( + page_manager, + range.end.max(min_start), + hole_end, + length, + permissions, + true, + ) + .map_err(mapping_error_to_nt_status)? + { + HoleSearchResult::Allocated(ptr) => return Ok(ptr), + HoleSearchResult::RetryWithFreshMappings => continue 'search, + HoleSearchResult::Exhausted => {} + } + } + if range.start < hole_end { + hole_end = range.start; + } + if hole_end <= min_start { + break; + } + } + + match create_aligned_pages_in_hole( + page_manager, + min_start, + hole_end, + length, + permissions, + true, + ) + .map_err(mapping_error_to_nt_status)? + { + HoleSearchResult::Allocated(ptr) => return Ok(ptr), + HoleSearchResult::RetryWithFreshMappings => continue 'search, + HoleSearchResult::Exhausted => {} + } + } else { + let mut hole_start = min_start; + for (range, _) in &mappings { + if range.start >= search_end { + break; + } + if range.end <= hole_start { + continue; + } + if range.start > hole_start { + match create_aligned_pages_in_hole( + page_manager, + hole_start, + range.start.min(search_end), + length, + permissions, + false, + ) + .map_err(mapping_error_to_nt_status)? + { + HoleSearchResult::Allocated(ptr) => return Ok(ptr), + HoleSearchResult::RetryWithFreshMappings => continue 'search, + HoleSearchResult::Exhausted => {} + } + } + if range.end > hole_start { + hole_start = range.end; + } + if hole_start >= search_end { + break; + } + } + + match create_aligned_pages_in_hole( + page_manager, + hole_start, + search_end, + length, + permissions, + false, + ) + .map_err(mapping_error_to_nt_status)? + { + HoleSearchResult::Allocated(ptr) => return Ok(ptr), + HoleSearchResult::RetryWithFreshMappings => continue 'search, + HoleSearchResult::Exhausted => {} + } + } + + return Err(NtStatus::NO_MEMORY); + } + + Err(NtStatus::NO_MEMORY) +} + +fn allocation_granularity_aligned_candidate( + hole_start: usize, + hole_end: usize, + length: usize, + top_down: bool, +) -> Option { + if top_down { + let max_candidate = hole_end.checked_sub(length)? & !(ALLOCATION_GRANULARITY - 1); + (max_candidate >= hole_start).then_some(max_candidate) + } else { + let min_candidate = hole_start.next_multiple_of(ALLOCATION_GRANULARITY); + min_candidate + .checked_add(length) + .is_some_and(|end| end <= hole_end) + .then_some(min_candidate) + } +} + +fn update_permissions( + page_manager: &WindowsPageManager, + aligned_base: usize, + aligned_len: usize, + permissions: MemoryRegionPermissions, +) -> Result<(), ()> { + let ptr = MutPtr::::from_usize(aligned_base); + // SAFETY: This applies the guest's explicit VM protection/free request to a page-aligned range + // tracked by the LiteBox page manager. The page manager serializes the VMA update. + let result = unsafe { + match permissions { + permissions if permissions.is_empty() => { + page_manager.make_pages_inaccessible(ptr, aligned_len) + } + MemoryRegionPermissions::READ => page_manager.make_pages_readable(ptr, aligned_len), + permissions + if permissions + == MemoryRegionPermissions::READ | MemoryRegionPermissions::WRITE => + { + page_manager.make_pages_writable(ptr, aligned_len) + } + permissions + if permissions == MemoryRegionPermissions::READ | MemoryRegionPermissions::EXEC => + { + page_manager.make_pages_executable(ptr, aligned_len) + } + permissions + if permissions + == MemoryRegionPermissions::READ + | MemoryRegionPermissions::WRITE + | MemoryRegionPermissions::EXEC => + { + page_manager.make_pages_rwx(ptr, aligned_len) + } + _ => return Err(()), + } + }; + + result.map_err(|_| ()) +} + +fn query_memory_basic_information( + page_manager: &WindowsPageManager, + virtual_allocations: &WindowsVirtualAllocations, + base_address: usize, +) -> Option { + let query_base = base_address & !(PAGE_SIZE - 1); + if query_base >= Platform::TASK_ADDR_MAX { + return None; + } + + let mut mappings = page_manager.mappings(); + mappings.sort_by_key(|(range, _)| range.start); + if let Some(allocation) = find_virtual_allocation_containing(virtual_allocations, query_base) { + return query_allocation_basic_information(allocation, query_base); + } + + if let Some((range, flags)) = mappings + .iter() + .find(|(range, _)| range.contains(&base_address)) + { + let protect = permissions_to_page_protect(MemoryRegionPermissions::from(*flags)); + return Some(MemoryBasicInformation { + base_address: range.start, + allocation_base: range.start, + allocation_protect: protect.bits(), + partition_id: 0, + _padding0: 0, + region_size: range.end - range.start, + state: MemoryState::MEM_COMMIT.bits(), + protect: protect.bits(), + type_: MemoryType::MEM_PRIVATE.bits(), + _padding1: 0, + }); + } + + let next_mapping_start = mappings + .iter() + .find(|(range, _)| range.start > query_base) + .map_or(Platform::TASK_ADDR_MAX, |(range, _)| range.start); + + Some(MemoryBasicInformation { + base_address: query_base, + allocation_base: 0, + allocation_protect: 0, + partition_id: 0, + _padding0: 0, + region_size: next_mapping_start.saturating_sub(query_base), + state: MemoryState::MEM_FREE.bits(), + protect: 0, + type_: 0, + _padding1: 0, + }) +} + +fn query_allocation_basic_information( + allocation: WindowsVirtualAllocation, + query_base: usize, +) -> Option { + let allocation_end = allocation.base.checked_add(allocation.size)?; + let (state, protect) = private_page_state_and_protect(&allocation, query_base); + let mut region_end = query_base.checked_add(PAGE_SIZE)?; + while region_end < allocation_end { + if private_page_state_and_protect(&allocation, region_end) != (state, protect) { + break; + } + region_end = region_end.checked_add(PAGE_SIZE)?; + } + + Some(MemoryBasicInformation { + base_address: query_base, + allocation_base: allocation.base, + allocation_protect: allocation.allocation_protect.bits(), + partition_id: 0, + _padding0: 0, + region_size: region_end - query_base, + state, + protect, + type_: allocation.type_.bits(), + _padding1: 0, + }) +} + +fn private_page_state_and_protect( + allocation: &WindowsVirtualAllocation, + page: usize, +) -> (u32, u32) { + allocation + .pages + .get(&page) + .map_or((MemoryState::MEM_RESERVE.bits(), 0), |protect| { + (MemoryState::MEM_COMMIT.bits(), protect.bits()) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{mut_byte_ptr, mut_ptr}; + use litebox::platform::ThreadProvider; + + extern crate std; + + type TestPlatform = crate::tests::TestPlatform; + type TestTask = Task; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + fn allocate_committed_rw(task: &TestTask, size: usize) -> (usize, usize) { + let mut base = 0usize; + let mut region_size = size; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + (AllocationType::MEM_RESERVE | AllocationType::MEM_COMMIT).bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + (base, region_size) + } + + fn release_allocation(task: &TestTask, base: usize) { + let mut release_base = base; + let mut release_size = 0usize; + assert_eq!( + task.sys_nt_free_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut release_base), + mut_ptr(&mut release_size), + FreeType::MEM_RELEASE.bits(), + ), + NtStatus::SUCCESS + ); + } + + fn query_basic_information(task: &TestTask, base: usize) -> MemoryBasicInformation { + let mut info = MemoryBasicInformation::default(); + let mut return_length = 0usize; + assert_eq!( + task.sys_nt_query_virtual_memory( + ProcessHandle::CURRENT, + base, + MemoryInformationClass::Basic as u32, + mut_byte_ptr(&mut info), + size_of::(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + assert_eq!(return_length, size_of::()); + info + } + + #[test] + fn allocate_virtual_memory_commit_only_null_base_creates_committed_region() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + assert_ne!(base, 0); + assert_eq!(region_size, PAGE_SIZE); + + let info = query_basic_information(&task, base); + assert_eq!(info.state, MemoryState::MEM_COMMIT.bits()); + assert_eq!(info.protect, PageProtection::PAGE_READWRITE.bits()); + + release_allocation(&task, base); + }); + } + + #[test] + fn allocate_virtual_memory_fixed_reserve_rounds_base_to_allocation_granularity() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let requested_base = ALLOCATION_GRANULARITY * 8 + PAGE_SIZE + 123; + let expected_base = requested_base & !(ALLOCATION_GRANULARITY - 1); + let mut base = requested_base; + let mut region_size = 1usize; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + assert_eq!(base, expected_base); + assert_eq!(region_size, PAGE_SIZE * 2); + + release_allocation(&task, base); + }); + } + + #[test] + fn allocate_virtual_memory_zero_bits_is_allowed_for_null_base() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 1, + mut_ptr(&mut region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + assert_ne!(base, 0); + assert_eq!(region_size, PAGE_SIZE); + + release_allocation(&task, base); + }); + } + + #[test] + fn allocate_virtual_memory_fixed_collision_returns_conflicting_addresses() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + + let mut fixed_base = base; + let mut fixed_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut fixed_base), + 0, + mut_ptr(&mut fixed_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::CONFLICTING_ADDRESSES + ); + + release_allocation(&task, base); + }); + } + + #[test] + fn allocate_virtual_memory_mem_top_down_prefers_higher_addresses() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + + let mut bottom_base = 0usize; + let mut bottom_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut bottom_base), + 0, + mut_ptr(&mut bottom_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + + let mut top_base = 0usize; + let mut top_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut top_base), + 0, + mut_ptr(&mut top_size), + (AllocationType::MEM_RESERVE | AllocationType::MEM_TOP_DOWN).bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + + assert!(top_base > bottom_base); + + release_allocation(&task, top_base); + release_allocation(&task, bottom_base); + }); + } + + #[test] + fn allocate_virtual_memory_mem_reset_preserves_committed_state_and_protection() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, _) = allocate_committed_rw(&task, PAGE_SIZE); + + let mut reset_base = base + 1; + let mut reset_size = 1usize; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut reset_base), + 0, + mut_ptr(&mut reset_size), + AllocationType::MEM_RESET.bits(), + PageProtection::PAGE_NOACCESS.bits(), + ), + NtStatus::SUCCESS + ); + assert_eq!(reset_base, base); + assert_eq!(reset_size, PAGE_SIZE); + + let info = query_basic_information(&task, base); + assert_eq!(info.state, MemoryState::MEM_COMMIT.bits()); + assert_eq!(info.protect, PageProtection::PAGE_READWRITE.bits()); + + release_allocation(&task, base); + }); + } + + #[test] + fn allocate_virtual_memory_mem_reset_rejects_combined_flags() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, _) = allocate_committed_rw(&task, PAGE_SIZE); + + let mut reset_base = base; + let mut reset_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut reset_base), + 0, + mut_ptr(&mut reset_size), + (AllocationType::MEM_RESET | AllocationType::MEM_COMMIT).bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::INVALID_PARAMETER + ); + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_rounds_outputs_and_reports_old_protection() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, allocation_size) = allocate_committed_rw(&task, PAGE_SIZE * 2 - 1); + assert_eq!(base % ALLOCATION_GRANULARITY, 0); + assert_eq!(allocation_size, PAGE_SIZE * 2); + + let mut protect_base = base + 1; + let mut protect_size = 1usize; + let mut old_protect = 0u32; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::SUCCESS + ); + assert_eq!(protect_base, base); + assert_eq!(protect_size, PAGE_SIZE); + assert_eq!(old_protect, PageProtection::PAGE_READWRITE.bits()); + + let info = query_basic_information(&task, base); + assert_eq!(info.base_address, base); + assert_eq!(info.allocation_base, base); + assert_eq!( + info.allocation_protect, + PageProtection::PAGE_READWRITE.bits() + ); + assert_eq!(info.region_size, PAGE_SIZE); + assert_eq!(info.state, MemoryState::MEM_COMMIT.bits()); + assert_eq!(info.protect, PageProtection::PAGE_READONLY.bits()); + assert_eq!(info.type_, MemoryType::MEM_PRIVATE.bits()); + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_allows_committed_page_noaccess() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + (AllocationType::MEM_RESERVE | AllocationType::MEM_COMMIT).bits(), + PageProtection::PAGE_NOACCESS.bits(), + ), + NtStatus::SUCCESS + ); + + let info = query_basic_information(&task, base); + assert_eq!(info.state, MemoryState::MEM_COMMIT.bits()); + assert_eq!(info.protect, PageProtection::PAGE_NOACCESS.bits()); + + let mut protect_base = base; + let mut protect_size = PAGE_SIZE; + let mut old_protect = 0u32; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::SUCCESS + ); + assert_eq!(old_protect, PageProtection::PAGE_NOACCESS.bits()); + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_rejects_reserved_pages() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + + let mut protect_base = base; + let mut protect_size = PAGE_SIZE; + let mut old_protect = u32::MAX; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::NOT_COMMITTED + ); + assert_eq!(old_protect, PageProtection::PAGE_NOACCESS.bits()); + + for invalid_protect in [ + PageProtection::PAGE_NOACCESS | PageProtection::PAGE_GUARD, + PageProtection::PAGE_NOACCESS | PageProtection::PAGE_NOCACHE, + PageProtection::PAGE_NOACCESS | PageProtection::PAGE_WRITECOMBINE, + PageProtection::PAGE_READWRITE + | PageProtection::PAGE_NOCACHE + | PageProtection::PAGE_WRITECOMBINE, + PageProtection::PAGE_READWRITE + | PageProtection::PAGE_GUARD + | PageProtection::PAGE_NOCACHE, + ] { + let mut protect_base = base; + let mut protect_size = PAGE_SIZE; + let mut old_protect = u32::MAX; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + invalid_protect.bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::INVALID_PAGE_PROTECTION + ); + assert_eq!(old_protect, u32::MAX); + } + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_rejects_partly_reserved_range() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE * 2; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + let mut commit_base = base; + let mut commit_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut commit_base), + 0, + mut_ptr(&mut commit_size), + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + + let mut protect_base = base; + let mut protect_size = PAGE_SIZE * 2; + let mut old_protect = u32::MAX; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::NOT_COMMITTED + ); + assert_eq!(old_protect, PageProtection::PAGE_NOACCESS.bits()); + assert_eq!( + query_basic_information(&task, base).protect, + PageProtection::PAGE_READWRITE.bits() + ); + + release_allocation(&task, base); + }); + } + + #[test] + fn free_virtual_memory_decommit_zero_size_at_allocation_base_decommits_whole_region() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, allocation_size) = allocate_committed_rw(&task, PAGE_SIZE * 2); + + let mut decommit_base = base; + let mut decommit_size = 0usize; + assert_eq!( + task.sys_nt_free_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut decommit_base), + mut_ptr(&mut decommit_size), + FreeType::MEM_DECOMMIT.bits(), + ), + NtStatus::SUCCESS + ); + assert_eq!(decommit_base, base); + assert_eq!(decommit_size, allocation_size); + + let info = query_basic_information(&task, base); + assert_eq!(info.state, MemoryState::MEM_RESERVE.bits()); + assert_eq!(info.protect, 0); + assert_eq!(info.region_size, allocation_size); + + release_allocation(&task, base); + }); + } + + #[test] + fn free_virtual_memory_decommit_discards_page_contents() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, _) = allocate_committed_rw(&task, PAGE_SIZE); + let ptr = MutPtr::::from_usize(base); + assert_eq!(ptr.write_at_offset(0, 0xa5), Some(())); + + let mut decommit_base = base; + let mut decommit_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_free_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut decommit_base), + mut_ptr(&mut decommit_size), + FreeType::MEM_DECOMMIT.bits(), + ), + NtStatus::SUCCESS + ); + + let mut commit_base = base; + let mut commit_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut commit_base), + 0, + mut_ptr(&mut commit_size), + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + NtStatus::SUCCESS + ); + assert_eq!(ptr.read_at_offset(0), Some(0)); + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_preserves_page_modifier_bits() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut base = 0usize; + let mut region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut base), + 0, + mut_ptr(&mut region_size), + (AllocationType::MEM_RESERVE | AllocationType::MEM_COMMIT).bits(), + (PageProtection::PAGE_READWRITE | PageProtection::PAGE_NOCACHE).bits(), + ), + NtStatus::SUCCESS + ); + assert_eq!( + query_basic_information(&task, base).protect, + (PageProtection::PAGE_READWRITE | PageProtection::PAGE_NOCACHE).bits() + ); + + let mut protect_base = base; + let mut protect_size = PAGE_SIZE; + let mut old_protect = 0u32; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + (PageProtection::PAGE_READONLY | PageProtection::PAGE_WRITECOMBINE).bits(), + mut_ptr(&mut old_protect), + ), + NtStatus::SUCCESS + ); + assert_eq!( + old_protect, + (PageProtection::PAGE_READWRITE | PageProtection::PAGE_NOCACHE).bits() + ); + assert_eq!( + query_basic_information(&task, base).protect, + (PageProtection::PAGE_READONLY | PageProtection::PAGE_WRITECOMBINE).bits() + ); + + release_allocation(&task, base); + }); + } + + #[test] + fn protect_virtual_memory_rejects_invalid_page_protection() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let (base, _) = allocate_committed_rw(&task, PAGE_SIZE); + + let mut protect_base = base; + let mut protect_size = PAGE_SIZE; + let mut old_protect = u32::MAX; + assert_eq!( + task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut protect_base), + mut_ptr(&mut protect_size), + 0, + mut_ptr(&mut old_protect), + ), + NtStatus::INVALID_PAGE_PROTECTION + ); + assert_eq!(old_protect, u32::MAX); + + release_allocation(&task, base); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + mod host_fidelity { + use super::*; + use core::ffi::c_void; + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtAllocateVirtualMemory( + process_handle: *mut c_void, + base_address: *mut *mut c_void, + zero_bits: usize, + region_size: *mut usize, + allocation_type: u32, + protect: u32, + ) -> i32; + + fn NtProtectVirtualMemory( + process_handle: *mut c_void, + base_address: *mut *mut c_void, + region_size: *mut usize, + new_protect: u32, + old_protect: *mut u32, + ) -> i32; + + fn NtQueryVirtualMemory( + process_handle: *mut c_void, + base_address: *const c_void, + memory_information_class: u32, + memory_information: *mut c_void, + memory_information_length: usize, + return_length: *mut usize, + ) -> i32; + + fn NtFreeVirtualMemory( + process_handle: *mut c_void, + base_address: *mut *mut c_void, + region_size: *mut usize, + free_type: u32, + ) -> i32; + } + + fn current_process() -> *mut c_void { + usize::MAX as *mut c_void + } + + fn host_status(status: i32) -> NtStatus { + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + #[test] + fn allocate_query_free_outputs_match_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE; + // SAFETY: The output pointers are valid locals and the current-process pseudo + // handle targets this process. The allocation is released before return. + let host_allocate_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_NOACCESS.bits(), + )) + }; + assert_eq!(host_allocate_status, NtStatus::SUCCESS); + + let mut host_info = MemoryBasicInformation::default(); + let mut host_return_length = 0usize; + // SAFETY: The host allocation is live, and the output buffer and return length are + // valid locals that ntdll writes synchronously. + let host_query_status = unsafe { + host_status(NtQueryVirtualMemory( + current_process(), + host_base, + MemoryInformationClass::Basic as u32, + (&raw mut host_info).cast(), + size_of::(), + &raw mut host_return_length, + )) + }; + assert_eq!(host_query_status, NtStatus::SUCCESS); + + let task = crate::tests::test_task(); + let mut guest_base = 0usize; + let mut guest_region_size = PAGE_SIZE; + let guest_allocate_status = task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_base), + 0, + mut_ptr(&mut guest_region_size), + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_NOACCESS.bits(), + ); + assert_eq!(guest_allocate_status, host_allocate_status); + assert_eq!(guest_region_size, host_region_size); + + let guest_info = query_basic_information(&task, guest_base); + assert_eq!(guest_info.base_address, guest_base); + assert_eq!(host_info.base_address, host_base as usize); + assert_eq!(guest_info.allocation_base, guest_base); + assert_eq!(host_info.allocation_base, host_base as usize); + assert_eq!(guest_info.allocation_protect, host_info.allocation_protect); + assert_eq!(guest_info.region_size, host_info.region_size); + assert_eq!(guest_info.state, host_info.state); + assert_eq!(guest_info.protect, host_info.protect); + assert_eq!(guest_info.type_, host_info.type_); + + let mut guest_release_base = guest_base; + let mut guest_release_size = 0usize; + let guest_free_status = task.sys_nt_free_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_release_base), + mut_ptr(&mut guest_release_size), + FreeType::MEM_RELEASE.bits(), + ); + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(guest_free_status, host_free_status); + assert_eq!(guest_release_size, host_release_size); + }); + } + + #[test] + fn reserve_alignment_outputs_match_host_ntdll() { + run_with_test_platform_pointers(|| { + // The probe-free-reuse dance below races the process's own allocator: anything + // (heap growth, loader activity, another test's allocations) can grab the + // just-released range between the free and the fixed-address reuse, turning the + // reuse into STATUS_CONFLICTING_ADDRESSES -- observed repeatedly on CI runners. + // A conflict means the probe address went stale, not that alignment fidelity + // broke, so re-probe at a fresh address, bounded. + // + // Pre-warm the guest allocation path once before any probe: its first call + // grows the shim's own heap and tracking structures, and on some runners that + // growth deterministically landed exactly in the just-freed probe range, + // exhausting every retry. After a throwaway round trip, the machinery is + // allocated and the probe-to-attempt window contains no shim-side heap growth. + { + let task = crate::tests::test_task(); + let mut warm_base = 0usize; + let mut warm_size = 1usize; + let warm_status = task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut warm_base), + 0, + mut_ptr(&mut warm_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ); + assert_eq!(warm_status, NtStatus::SUCCESS); + release_allocation(&task, warm_base); + } + let mut attempts_left = 32; + loop { + attempts_left -= 1; + + let mut probe_base = core::ptr::null_mut::(); + let mut probe_region_size = ALLOCATION_GRANULARITY * 2; + // SAFETY: The output pointers are valid locals and the current-process + // pseudo handle targets this process. The allocation is released before + // reuse below. + let probe_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut probe_base, + 0, + &raw mut probe_region_size, + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(probe_status, NtStatus::SUCCESS); + + let mut probe_release_base = probe_base; + let mut probe_release_size = 0usize; + // SAFETY: Releases the host allocation created above so the fixed-address + // probe can reuse the same address range. + let probe_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut probe_release_base, + &raw mut probe_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(probe_free_status, NtStatus::SUCCESS); + + let requested_base = probe_base.wrapping_byte_add(PAGE_SIZE + 123); + let mut host_base = requested_base; + let mut host_region_size = 1usize; + // SAFETY: The fixed address range was just released and the output pointers + // are valid locals. The allocation is released before the guest probe runs. + let host_allocate_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + if host_allocate_status != NtStatus::SUCCESS && attempts_left > 0 { + continue; + } + assert_eq!(host_allocate_status, NtStatus::SUCCESS); + + let mut host_release_base = host_base; + let mut host_release_size = 0usize; + // SAFETY: Releases the fixed host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_release_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + + let task = crate::tests::test_task(); + let mut guest_base = requested_base as usize; + let mut guest_region_size = 1usize; + let guest_allocate_status = task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_base), + 0, + mut_ptr(&mut guest_region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ); + if guest_allocate_status != host_allocate_status && attempts_left > 0 { + continue; + } + assert_eq!(guest_allocate_status, host_allocate_status); + assert_eq!(guest_base, host_base as usize); + assert_eq!(guest_region_size, host_region_size); + + release_allocation(&task, guest_base); + break; + } + }); + } + + #[test] + fn allocate_virtual_memory_zero_bits_bitmask_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + // When ZeroBits > 32, Windows treats it as the maximum virtual address for the + // allocation (exclusive upper bound = zero_bits + 1). A value of 0x7FFF_FFFF + // restricts the allocation to below 2 GiB. + let zero_bits_max_addr: usize = 0x7FFF_FFFF; + let limit: usize = zero_bits_max_addr + 1; + + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE; + // SAFETY: Output pointers are valid locals and the current-process pseudo handle + // targets this process. The allocation is released before return. + let host_allocate_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + zero_bits_max_addr, + &raw mut host_region_size, + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_allocate_status, NtStatus::SUCCESS); + assert!( + host_base as usize + host_region_size <= limit, + "host allocation exceeds ZeroBits max address" + ); + + let task = crate::tests::test_task(); + let mut guest_base = 0usize; + let mut guest_region_size = PAGE_SIZE; + let guest_allocate_status = task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_base), + zero_bits_max_addr, + mut_ptr(&mut guest_region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ); + assert_eq!(guest_allocate_status, host_allocate_status); + assert!( + guest_base + guest_region_size <= limit, + "guest allocation exceeds ZeroBits max address" + ); + + release_allocation(&task, guest_base); + + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + }); + } + + #[test] + fn mem_reset_reserved_pages_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE; + // SAFETY: The output pointers are valid locals and the current-process pseudo + // handle targets this process. The allocation is released before return. + let host_reserve_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_reserve_status, NtStatus::SUCCESS); + + let mut host_reset_base = host_base.wrapping_byte_add(1); + let mut host_reset_size = 1usize; + // SAFETY: The host allocation is reserved but uncommitted; the output pointers are + // valid locals and ntdll does not retain them. + let host_reset_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_reset_base, + 0, + &raw mut host_reset_size, + AllocationType::MEM_RESET.bits(), + PageProtection::PAGE_NOACCESS.bits(), + )) + }; + assert_eq!(host_reset_status, NtStatus::CONFLICTING_ADDRESSES); + + let task = crate::tests::test_task(); + let mut guest_base = 0usize; + let mut guest_region_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_base), + 0, + mut_ptr(&mut guest_region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + host_reserve_status + ); + + let mut guest_reset_base = guest_base + 1; + let mut guest_reset_size = 1usize; + let guest_reset_status = task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_reset_base), + 0, + mut_ptr(&mut guest_reset_size), + AllocationType::MEM_RESET.bits(), + PageProtection::PAGE_NOACCESS.bits(), + ); + assert_eq!(guest_reset_status, host_reset_status); + assert_eq!( + guest_reset_base - guest_base, + host_reset_base as usize - host_base as usize + ); + assert_eq!(guest_reset_size, host_reset_size); + + release_allocation(&task, guest_base); + + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + }); + } + + #[test] + fn protect_virtual_memory_outputs_match_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE * 2 - 1; + // SAFETY: The output pointers are valid local variables and the pseudo process + // handle targets the current process. The allocation is released before return. + let host_allocate_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + (AllocationType::MEM_RESERVE | AllocationType::MEM_COMMIT).bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_allocate_status, NtStatus::SUCCESS); + + let mut host_protect_base = host_base.wrapping_byte_add(1); + let mut host_protect_size = 1usize; + let mut host_old_protect = 0u32; + // SAFETY: The host allocation above covers the requested byte; all output pointers + // are valid locals and ntdll does not retain them. + let host_protect_status = unsafe { + host_status(NtProtectVirtualMemory( + current_process(), + &raw mut host_protect_base, + &raw mut host_protect_size, + PageProtection::PAGE_READONLY.bits(), + &raw mut host_old_protect, + )) + }; + + let task = crate::tests::test_task(); + let (guest_base, _) = allocate_committed_rw(&task, PAGE_SIZE * 2 - 1); + let mut guest_protect_base = guest_base + 1; + let mut guest_protect_size = 1usize; + let mut guest_old_protect = 0u32; + let guest_protect_status = task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_protect_base), + mut_ptr(&mut guest_protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut guest_old_protect), + ); + + assert_eq!(guest_protect_status, host_protect_status); + assert_eq!(guest_old_protect, host_old_protect); + assert_eq!(guest_protect_base, guest_base); + assert_eq!(guest_protect_size, host_protect_size); + + release_allocation(&task, guest_base); + + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + }); + } + + #[test] + fn protect_virtual_memory_mixed_committed_protections_match_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE * 2; + // SAFETY: The output pointers are valid local variables and the pseudo process + // handle targets the current process. The allocation is released before return. + let host_allocate_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + (AllocationType::MEM_RESERVE | AllocationType::MEM_COMMIT).bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_allocate_status, NtStatus::SUCCESS); + + let mut host_second_page_base = host_base.wrapping_byte_add(PAGE_SIZE); + let mut host_second_page_size = PAGE_SIZE; + let mut host_second_old_protect = 0u32; + // SAFETY: The host allocation above covers the requested second page; output + // pointers are valid locals and ntdll does not retain them. + let host_second_protect_status = unsafe { + host_status(NtProtectVirtualMemory( + current_process(), + &raw mut host_second_page_base, + &raw mut host_second_page_size, + PageProtection::PAGE_READONLY.bits(), + &raw mut host_second_old_protect, + )) + }; + assert_eq!(host_second_protect_status, NtStatus::SUCCESS); + + let task = crate::tests::test_task(); + let (guest_base, _) = allocate_committed_rw(&task, PAGE_SIZE * 2); + let mut guest_second_page_base = guest_base + PAGE_SIZE; + let mut guest_second_page_size = PAGE_SIZE; + let mut guest_second_old_protect = 0u32; + let guest_second_protect_status = task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_second_page_base), + mut_ptr(&mut guest_second_page_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut guest_second_old_protect), + ); + assert_eq!(guest_second_protect_status, host_second_protect_status); + assert_eq!(guest_second_old_protect, host_second_old_protect); + assert_eq!( + guest_second_page_base - guest_base, + host_second_page_base as usize - host_base as usize + ); + assert_eq!(guest_second_page_size, host_second_page_size); + + let mut host_mixed_base = host_base; + let mut host_mixed_size = PAGE_SIZE * 2; + let mut host_mixed_old_protect = 0u32; + // SAFETY: The host range is fully committed with mixed protections; output + // pointers are valid locals and ntdll does not retain them. + let host_mixed_protect_status = unsafe { + host_status(NtProtectVirtualMemory( + current_process(), + &raw mut host_mixed_base, + &raw mut host_mixed_size, + PageProtection::PAGE_EXECUTE_READ.bits(), + &raw mut host_mixed_old_protect, + )) + }; + + let mut guest_mixed_base = guest_base; + let mut guest_mixed_size = PAGE_SIZE * 2; + let mut guest_mixed_old_protect = 0u32; + let guest_mixed_protect_status = task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_mixed_base), + mut_ptr(&mut guest_mixed_size), + PageProtection::PAGE_EXECUTE_READ.bits(), + mut_ptr(&mut guest_mixed_old_protect), + ); + assert_eq!(guest_mixed_protect_status, host_mixed_protect_status); + assert_eq!(guest_mixed_old_protect, host_mixed_old_protect); + assert_eq!(guest_mixed_base, guest_base); + assert_eq!(host_mixed_base, host_base); + assert_eq!(guest_mixed_size, host_mixed_size); + + release_allocation(&task, guest_base); + + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + }); + } + + #[test] + fn protect_virtual_memory_uncommitted_ranges_match_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_base = core::ptr::null_mut::(); + let mut host_region_size = PAGE_SIZE * 2; + // SAFETY: The output pointers are valid locals and the current-process pseudo + // handle targets this process. The allocation is released before return. + let host_reserve_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_base, + 0, + &raw mut host_region_size, + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_reserve_status, NtStatus::SUCCESS); + + let task = crate::tests::test_task(); + let mut guest_base = 0usize; + let mut guest_region_size = PAGE_SIZE * 2; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_base), + 0, + mut_ptr(&mut guest_region_size), + AllocationType::MEM_RESERVE.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + host_reserve_status + ); + + let mut host_protect_base = host_base; + let mut host_protect_size = PAGE_SIZE; + let mut host_old_protect = u32::MAX; + // SAFETY: The host range is reserved but uncommitted; output pointers are valid + // locals and ntdll does not retain them. + let host_reserved_protect_status = unsafe { + host_status(NtProtectVirtualMemory( + current_process(), + &raw mut host_protect_base, + &raw mut host_protect_size, + PageProtection::PAGE_READONLY.bits(), + &raw mut host_old_protect, + )) + }; + + let mut guest_protect_base = guest_base; + let mut guest_protect_size = PAGE_SIZE; + let mut guest_old_protect = u32::MAX; + let guest_reserved_protect_status = task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_protect_base), + mut_ptr(&mut guest_protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut guest_old_protect), + ); + assert_eq!(guest_reserved_protect_status, host_reserved_protect_status); + assert_eq!( + guest_protect_base - guest_base, + host_protect_base as usize - host_base as usize + ); + assert_eq!(guest_protect_size, host_protect_size); + assert_eq!(guest_old_protect, host_old_protect); + + let mut host_commit_base = host_base; + let mut host_commit_size = PAGE_SIZE; + // SAFETY: Commits the first page inside the live host reservation; output pointers + // are valid locals and the reservation is released before return. + let host_commit_status = unsafe { + host_status(NtAllocateVirtualMemory( + current_process(), + &raw mut host_commit_base, + 0, + &raw mut host_commit_size, + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_READWRITE.bits(), + )) + }; + assert_eq!(host_commit_status, NtStatus::SUCCESS); + + let mut guest_commit_base = guest_base; + let mut guest_commit_size = PAGE_SIZE; + assert_eq!( + task.sys_nt_allocate_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_commit_base), + 0, + mut_ptr(&mut guest_commit_size), + AllocationType::MEM_COMMIT.bits(), + PageProtection::PAGE_READWRITE.bits(), + ), + host_commit_status + ); + + let mut host_mixed_protect_base = host_base; + let mut host_mixed_protect_size = PAGE_SIZE * 2; + let mut host_mixed_old_protect = u32::MAX; + // SAFETY: The host range spans one committed page and one reserved page; output + // pointers are valid locals and ntdll does not retain them. + let host_mixed_protect_status = unsafe { + host_status(NtProtectVirtualMemory( + current_process(), + &raw mut host_mixed_protect_base, + &raw mut host_mixed_protect_size, + PageProtection::PAGE_READONLY.bits(), + &raw mut host_mixed_old_protect, + )) + }; + + let mut guest_mixed_protect_base = guest_base; + let mut guest_mixed_protect_size = PAGE_SIZE * 2; + let mut guest_mixed_old_protect = u32::MAX; + let guest_mixed_protect_status = task.sys_nt_protect_virtual_memory( + ProcessHandle::CURRENT, + mut_ptr(&mut guest_mixed_protect_base), + mut_ptr(&mut guest_mixed_protect_size), + PageProtection::PAGE_READONLY.bits(), + mut_ptr(&mut guest_mixed_old_protect), + ); + assert_eq!(guest_mixed_protect_status, host_mixed_protect_status); + assert_eq!( + guest_mixed_protect_base - guest_base, + host_mixed_protect_base as usize - host_base as usize + ); + assert_eq!(guest_mixed_protect_size, host_mixed_protect_size); + assert_eq!(guest_mixed_old_protect, host_mixed_old_protect); + + release_allocation(&task, guest_base); + + let mut host_release_size = 0usize; + // SAFETY: Releases the host allocation created by this test. + let host_free_status = unsafe { + host_status(NtFreeVirtualMemory( + current_process(), + &raw mut host_base, + &raw mut host_release_size, + FreeType::MEM_RELEASE.bits(), + )) + }; + assert_eq!(host_free_status, NtStatus::SUCCESS); + }); + } + } +} diff --git a/litebox_shim_windows/src/syscalls/mod.rs b/litebox_shim_windows/src/syscalls/mod.rs new file mode 100644 index 0000000000..a78a929fcd --- /dev/null +++ b/litebox_shim_windows/src/syscalls/mod.rs @@ -0,0 +1,1312 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +pub(crate) mod apphelp; +pub(crate) mod condrv; +pub(crate) mod event; +pub(crate) mod file; +pub(crate) mod file_path; +pub(crate) mod iocp; +pub(crate) mod lpc; +pub(crate) mod mm; +pub(crate) mod nls; +pub(crate) mod object_manager; +pub(crate) mod process; +pub(crate) mod registry; +pub(crate) mod section; +pub(crate) mod symlink; +pub(crate) mod sysinfo; +pub(crate) mod thread; +pub(crate) mod timer; +pub(crate) mod token; +pub(crate) mod wait_completion_packet; +pub(crate) mod wnf; +pub(crate) mod worker_factory; + +use litebox::platform::{RawConstPointer as _, RawPointerProvider}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::NtSysno; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes, KnownLayout}; + +use crate::nt_types; + +const FIRST_STACK_ARGUMENT_OFFSET: usize = 0x28; +const HANDLE_SHIFT: u32 = 2; +const HANDLE_TAG_MASK: usize = (1usize << HANDLE_SHIFT) - 1; + +#[repr(transparent)] +#[derive( + Clone, Copy, Debug, Default, Eq, PartialEq, FromBytes, IntoBytes, Immutable, KnownLayout, +)] +pub(crate) struct Handle(usize); + +impl Handle { + #[must_use] + pub(crate) const fn from_raw(raw: usize) -> Self { + Self(raw) + } + + #[must_use] + pub(crate) fn from_raw_fd(raw_fd: usize) -> Option { + raw_fd + .checked_add(1)? + .checked_mul(1usize << HANDLE_SHIFT) + .map(Self) + } + + #[must_use] + pub(crate) fn raw_fd(self) -> Option { + if self.0 & HANDLE_TAG_MASK != 0 { + return None; + } + (self.0 >> HANDLE_SHIFT).checked_sub(1) + } + + #[must_use] + pub(crate) const fn as_raw(self) -> usize { + self.0 + } + + #[must_use] + pub(crate) const fn is_null(self) -> bool { + self.as_raw() == 0 + } +} + +#[repr(transparent)] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub(crate) struct ProcessHandle(Handle); + +impl ProcessHandle { + pub(crate) const CURRENT: Self = Self::from_raw(usize::MAX); + + #[must_use] + pub(crate) const fn from_raw(raw: usize) -> Self { + Self(Handle::from_raw(raw)) + } + + #[must_use] + pub(crate) const fn is_null(self) -> bool { + self.0.is_null() + } + + #[must_use] + pub(crate) fn is_current(self) -> bool { + self == Self::CURRENT + } + + #[must_use] + pub(crate) const fn as_handle(self) -> Handle { + self.0 + } +} + +#[repr(transparent)] +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)] +pub(crate) struct ThreadHandle(Handle); + +impl ThreadHandle { + pub(crate) const CURRENT: Self = Self::from_raw(usize::MAX - 1); + + #[must_use] + pub(crate) const fn from_raw(raw: usize) -> Self { + Self(Handle::from_raw(raw)) + } + + #[must_use] + pub(crate) fn is_current(self) -> bool { + self == Self::CURRENT + } +} + +#[allow(clippy::enum_variant_names)] +#[derive(Debug)] +pub(crate) enum SyscallRequest { + NtClose { + handle: Handle, + }, + NtDuplicateObject { + source_process_handle: ProcessHandle, + source_handle: Handle, + target_process_handle: ProcessHandle, + target_handle: Option>, + desired_access: u32, + handle_attributes: u32, + options: u32, + }, + NtCreateEvent { + event_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + event_type: u32, + initial_state: u8, + }, + NtCreateDirectoryObject { + directory_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtCreateDirectoryObjectEx { + directory_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + shadow_directory_handle: Handle, + flags: u32, + }, + NtOpenDirectoryObject { + directory_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtOpenSection { + section_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtQueryDirectoryObject { + directory_handle: Handle, + buffer: Platform::RawMutPointer, + buffer_length: u32, + return_single_entry: u8, + restart_scan: u8, + context: Platform::RawMutPointer, + return_length: Option>, + }, + NtCreateSymbolicLinkObject { + link_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + link_target: Platform::RawConstPointer, + }, + NtOpenSymbolicLinkObject { + link_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtQuerySymbolicLinkObject { + link_handle: Handle, + link_target: Platform::RawMutPointer, + returned_length: Option>, + }, + NtCreateIoCompletion { + io_completion_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + number_of_concurrent_threads: u32, + }, + NtConnectPort { + port_handle: Platform::RawMutPointer, + port_name: Platform::RawConstPointer, + security_qos: Platform::RawConstPointer, + client_view: Option>, + server_view: Option>, + max_message_length: Option>, + connection_information: Option>, + connection_information_length: Option>, + }, + /// `NtSecureConnectPort` carries SID and server-view semantics that are + /// deliberately outside the current CSR `NtConnectPort` subset. + NtSecureConnectPort, + NtCreateSection { + section_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + maximum_size: Option>, + section_page_protection: u32, + allocation_attributes: u32, + file_handle: Handle, + }, + NtCreateSectionEx { + section_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + maximum_size: Option>, + section_page_protection: u32, + allocation_attributes: u32, + file_handle: Handle, + extended_parameters: Option>, + extended_parameter_count: u32, + }, + NtCreateWaitCompletionPacket { + wait_completion_packet_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtAssociateWaitCompletionPacket { + wait_completion_packet_handle: Handle, + io_completion_handle: Handle, + target_object_handle: Handle, + key_context: usize, + apc_context: usize, + io_status: i32, + io_status_information: usize, + already_signaled: Option>, + }, + NtCancelWaitCompletionPacket { + wait_completion_packet_handle: Handle, + remove_signaled_packet: u8, + }, + NtCreateWorkerFactory { + worker_factory_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + completion_port_handle: Handle, + worker_process_handle: ProcessHandle, + start_routine: usize, + start_parameter: usize, + max_thread_count: u32, + stack_reserve: usize, + stack_commit: usize, + }, + NtSetInformationWorkerFactory { + worker_factory_handle: Handle, + worker_factory_information_class: u32, + worker_factory_information: Platform::RawConstPointer, + worker_factory_information_length: u32, + }, + NtShutdownWorkerFactory { + worker_factory_handle: Handle, + pending_worker_count: Platform::RawMutPointer, + }, + NtCreateTimer2 { + timer_handle: Platform::RawMutPointer, + timer_id: Option>, + object_attributes: Option>, + attributes: u32, + desired_access: u32, + }, + NtSetTimer2 { + timer_handle: Handle, + due_time: Option>, + period: Option>, + parameters: Option>, + }, + NtOpenEvent { + event_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtSetEvent { + event_handle: Handle, + previous_state: Option>, + }, + NtResetEvent { + event_handle: Handle, + previous_state: Option>, + }, + NtClearEvent { + event_handle: Handle, + }, + NtPulseEvent { + event_handle: Handle, + previous_state: Option>, + }, + NtQueryEvent { + event_handle: Handle, + event_information_class: u32, + event_information: Platform::RawMutPointer, + event_information_length: u32, + return_length: Option>, + }, + NtSetEventBoostPriority { + event_handle: Handle, + }, + NtOpenFile { + file_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + io_status_block: Platform::RawMutPointer, + share_access: u32, + open_options: u32, + }, + NtCreateFile { + file_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + io_status_block: Platform::RawMutPointer, + allocation_size: Option>, + file_attributes: u32, + share_access: u32, + create_disposition: u32, + create_options: u32, + ea_buffer: Option>, + ea_length: u32, + }, + NtWriteFile { + file_handle: Handle, + event: Handle, + apc_routine: Option>, + apc_context: Option>, + io_status_block: Platform::RawMutPointer, + buffer: Platform::RawConstPointer, + length: u32, + byte_offset: Option>, + key: Option>, + }, + NtQueryVolumeInformationFile { + file_handle: Handle, + io_status_block: Platform::RawMutPointer, + fs_information: Platform::RawMutPointer, + length: u32, + fs_information_class: u32, + }, + NtDeviceIoControlFile { + file_handle: Handle, + event: Handle, + apc_routine: Option>, + apc_context: Option>, + io_status_block: Platform::RawMutPointer, + io_control_code: u32, + input_buffer: Option>, + input_buffer_length: u32, + output_buffer: Option>, + output_buffer_length: u32, + }, + NtApphelpCacheControl { + service_class: u32, + service_data: Option>, + }, + NtOpenKey { + key_handle: Platform::RawMutPointer, + desired_access: u32, + object_attributes: Option>, + }, + NtQueryValueKey { + key_handle: Handle, + value_name: Platform::RawConstPointer, + key_value_information_class: u32, + key_value_information: Platform::RawMutPointer, + length: u32, + result_length: Platform::RawMutPointer, + }, + NtGetNlsSectionPtr { + section_type: u32, + section_data: u32, + context_data: usize, + section_pointer: Platform::RawMutPointer, + section_size: Option>, + }, + NtInitializeNlsFiles { + base_address: Platform::RawMutPointer, + default_locale_id: Platform::RawMutPointer, + default_casing_table_size: Platform::RawMutPointer, + }, + NtQueryDefaultLocale { + user_profile: u8, + default_locale_id: Platform::RawMutPointer, + }, + NtSetDefaultLocale { + user_profile: u8, + default_locale_id: u32, + }, + NtQueryDefaultUILanguage { + default_ui_language: Platform::RawMutPointer, + }, + NtSetDefaultUILanguage { + default_ui_language: u16, + }, + NtQueryInstallUILanguage { + install_ui_language: Platform::RawMutPointer, + }, + NtQueryPerformanceCounter { + performance_counter: Platform::RawMutPointer, + performance_frequency: Option>, + }, + NtQuerySystemInformation { + system_information_class: u32, + system_information: Platform::RawMutPointer, + system_information_length: u32, + return_length: Option>, + }, + NtQuerySystemInformationEx { + system_information_class: u32, + input_buffer: Option>, + input_buffer_length: u32, + system_information: Platform::RawMutPointer, + system_information_length: u32, + return_length: Option>, + }, + NtQueryWnfStateData { + state_name: Platform::RawConstPointer, + type_id: Option>, + explicit_scope: Option>, + change_stamp: Platform::RawMutPointer, + buffer: Platform::RawMutPointer, + buffer_size: Platform::RawMutPointer, + }, + NtCreateWnfStateName { + state_name: Platform::RawMutPointer, + name_lifetime: u32, + data_scope: u32, + persist_data: u8, + type_id: Option>, + maximum_state_size: u32, + security_descriptor: Platform::RawConstPointer, + }, + NtUpdateWnfStateData { + state_name: Platform::RawConstPointer, + buffer: Option>, + buffer_size: u32, + type_id: Option>, + explicit_scope: Option>, + matching_change_stamp: u32, + check_stamp: i32, + }, + NtDeleteWnfStateData { + state_name: Platform::RawConstPointer, + explicit_scope: Option>, + }, + NtDeleteWnfStateName { + state_name: Platform::RawConstPointer, + }, + NtQueryWnfStateNameInformation { + state_name: Platform::RawConstPointer, + name_information_class: u32, + explicit_scope: Option>, + buffer: Platform::RawMutPointer, + buffer_size: u32, + }, + NtQuerySection { + section_handle: Handle, + section_information_class: u32, + section_information: Platform::RawMutPointer, + section_information_length: usize, + return_length: Option>, + }, + NtQueryInformationProcess { + process_handle: ProcessHandle, + process_information_class: u32, + process_information: Platform::RawMutPointer, + process_information_length: u32, + return_length: Option>, + }, + NtSetInformationProcess { + process_handle: ProcessHandle, + process_information_class: u32, + process_information: Platform::RawMutPointer, + process_information_length: u32, + }, + NtSetInformationThread { + thread_handle: ThreadHandle, + thread_information_class: u32, + thread_information: Platform::RawConstPointer, + thread_information_length: u32, + }, + NtOpenThreadToken { + thread_handle: ThreadHandle, + desired_access: u32, + open_as_self: u32, + token_handle: Platform::RawMutPointer, + }, + NtOpenThreadTokenEx { + thread_handle: ThreadHandle, + desired_access: u32, + open_as_self: u32, + handle_attributes: u32, + token_handle: Platform::RawMutPointer, + }, + NtOpenProcessToken { + process_handle: ProcessHandle, + desired_access: u32, + token_handle: Platform::RawMutPointer, + }, + NtOpenProcessTokenEx { + process_handle: ProcessHandle, + desired_access: u32, + handle_attributes: u32, + token_handle: Platform::RawMutPointer, + }, + NtQueryInformationToken { + token_handle: Handle, + token_information_class: u32, + token_information: Platform::RawMutPointer, + token_information_length: u32, + return_length: Platform::RawMutPointer, + }, + NtQuerySecurityAttributesToken { + token_handle: Handle, + attributes: Platform::RawConstPointer, + number_of_attributes: u32, + buffer: Platform::RawMutPointer, + length: u32, + return_length: Platform::RawMutPointer, + }, + NtConvertBetweenAuxiliaryCounterAndPerformanceCounter { + flag: u32, + source: Platform::RawConstPointer, + destination: Platform::RawMutPointer, + conversion_error: Option>, + }, + NtAllocateVirtualMemory { + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + zero_bits: usize, + region_size: Platform::RawMutPointer, + allocation_type: u32, + protect: u32, + }, + NtAllocateVirtualMemoryEx { + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + region_size: Platform::RawMutPointer, + allocation_type: u32, + protect: u32, + extended_parameters: Option>, + extended_parameter_count: u32, + }, + NtFreeVirtualMemory { + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + region_size: Platform::RawMutPointer, + free_type: u32, + }, + NtProtectVirtualMemory { + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + region_size: Platform::RawMutPointer, + new_protect: u32, + old_protect: Platform::RawMutPointer, + }, + NtQueryVirtualMemory { + process_handle: ProcessHandle, + base_address: usize, + memory_information_class: u32, + memory_information: Platform::RawMutPointer, + memory_information_length: usize, + return_length: Option>, + }, + NtMapViewOfSection { + section_handle: Handle, + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + zero_bits: usize, + commit_size: usize, + section_offset: Option>, + view_size: Platform::RawMutPointer, + inherit_disposition: u32, + allocation_type: u32, + page_protection: u32, + }, + NtMapViewOfSectionEx { + section_handle: Handle, + process_handle: ProcessHandle, + base_address: Platform::RawMutPointer, + zero_bits: usize, + commit_size: usize, + section_offset: Option>, + view_size: Platform::RawMutPointer, + inherit_disposition: u32, + allocation_type: u32, + page_protection: u32, + extended_parameters: Option>, + extended_parameter_count: u32, + }, + NtUnmapViewOfSection { + process_handle: ProcessHandle, + base_address: usize, + }, + NtUnmapViewOfSectionEx { + process_handle: ProcessHandle, + base_address: usize, + flags: u32, + }, + /// Restores the selected portions of a thread context and resumes execution. + NtContinue { + context: Platform::RawConstPointer, + test_alert: bool, + }, + NtTerminateProcess { + process_handle: ProcessHandle, + exit_status: i32, + }, + NtTestAlert, + /// TODO: not supported yet + NtManageHotPatch, +} + +impl SyscallRequest { + pub(crate) fn try_from_raw(pt_regs: &litebox_common_linux::PtRegs) -> Option { + macro_rules! sys_req { + ($id:ident { $( $field:ident $(:$star:tt)? ),* $(,)? }) => { + sys_req!(@[$id] [ $( $field $(:$star)? ),* ] [ 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11 ] [ ]) + }; + (@[$id:ident] [ $f:ident $(,)? $($field:ident $(:$star:tt)?),* ] [ $n:literal $(,)? $($ns:literal),* ] [ $($tail:tt)* ]) => { + sys_req!(@[$id] [ $( $field $(:$star)? ),* ] [ $($ns),* ] [ $($tail)* $f: win_sys_req_arg::(pt_regs, $n)?, ]) + }; + (@[$id:ident] [ $f:ident : * $(,)? $($field:ident $(:$star:tt)?),* ] [ $n:literal $(,)? $($ns:literal),* ] [ $($tail:tt)* ]) => { + sys_req!(@[$id] [ $( $field $(:$star)? ),* ] [ $($ns),* ] [ $($tail)* $f: win_sys_req_ptr::(pt_regs, $n)?, ]) + }; + (@[$id:ident] [ $f:ident : { $expr:expr } $(,)? $($field:ident $(:$star:tt)?),* ] [ $n:literal $(,)? $($ns:literal),* ] [ $($tail:tt)* ]) => { + sys_req!(@[$id] [ $( $field $(:$star)? ),* ] [ $($ns),* ] [ $($tail)* $f: ($expr)(win_sys_req_arg::(pt_regs, $n)?), ]) + }; + (@[$id:ident] [ ] [ $($ns:literal),* ] [ $($tail:tt)* ]) => { + SyscallRequest::$id { $($tail)* } + }; + } + + match NtSysno::from_raw(pt_regs.orig_rax)? { + NtSysno::NtClose => Some(sys_req!(NtClose { + handle: { Handle::from_raw }, + })), + NtSysno::NtDuplicateObject => Some(sys_req!(NtDuplicateObject { + source_process_handle: { ProcessHandle::from_raw }, + source_handle: { Handle::from_raw }, + target_process_handle: { ProcessHandle::from_raw }, + target_handle:*, + desired_access, + handle_attributes, + options, + })), + NtSysno::NtCreateEvent => Some(sys_req!(NtCreateEvent { + event_handle:*, + desired_access, + object_attributes:*, + event_type, + initial_state, + })), + NtSysno::NtCreateDirectoryObject => Some(sys_req!(NtCreateDirectoryObject { + directory_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtCreateDirectoryObjectEx => Some(sys_req!(NtCreateDirectoryObjectEx { + directory_handle:*, + desired_access, + object_attributes:*, + shadow_directory_handle:{Handle::from_raw}, + flags, + })), + NtSysno::NtOpenDirectoryObject => Some(sys_req!(NtOpenDirectoryObject { + directory_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtOpenSection => Some(sys_req!(NtOpenSection { + section_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtQueryDirectoryObject => Some(sys_req!(NtQueryDirectoryObject { + directory_handle:{ Handle::from_raw }, + buffer:*, + buffer_length, + return_single_entry, + restart_scan, + context:*, + return_length:*, + })), + NtSysno::NtCreateSymbolicLinkObject => Some(sys_req!(NtCreateSymbolicLinkObject { + link_handle:*, + desired_access, + object_attributes:*, + link_target:*, + })), + NtSysno::NtOpenSymbolicLinkObject => Some(sys_req!(NtOpenSymbolicLinkObject { + link_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtQuerySymbolicLinkObject => Some(sys_req!(NtQuerySymbolicLinkObject { + link_handle:{Handle::from_raw}, + link_target:*, + returned_length:*, + })), + NtSysno::NtCreateIoCompletion => Some(sys_req!(NtCreateIoCompletion { + io_completion_handle:*, + desired_access, + object_attributes:*, + number_of_concurrent_threads, + })), + NtSysno::NtConnectPort => Some(sys_req!(NtConnectPort { + port_handle:*, + port_name:*, + security_qos:*, + client_view:*, + server_view:*, + max_message_length:*, + connection_information:*, + connection_information_length:*, + })), + NtSysno::NtSecureConnectPort => Some(SyscallRequest::NtSecureConnectPort), + NtSysno::NtCreateSection => Some(sys_req!(NtCreateSection { + section_handle:*, + desired_access, + object_attributes:*, + maximum_size:*, + section_page_protection, + allocation_attributes, + file_handle:{ Handle::from_raw }, + })), + NtSysno::NtCreateSectionEx => Some(sys_req!(NtCreateSectionEx { + section_handle:*, + desired_access, + object_attributes:*, + maximum_size:*, + section_page_protection, + allocation_attributes, + file_handle:{ Handle::from_raw }, + extended_parameters:*, + extended_parameter_count, + })), + NtSysno::NtCreateWaitCompletionPacket => Some(sys_req!( + NtCreateWaitCompletionPacket { + wait_completion_packet_handle:*, + desired_access, + object_attributes:*, + } + )), + NtSysno::NtAssociateWaitCompletionPacket => Some(sys_req!( + NtAssociateWaitCompletionPacket { + wait_completion_packet_handle:{Handle::from_raw}, + io_completion_handle:{Handle::from_raw}, + target_object_handle:{Handle::from_raw}, + key_context, + apc_context, + io_status, + io_status_information, + already_signaled:*, + } + )), + NtSysno::NtCancelWaitCompletionPacket => Some(sys_req!(NtCancelWaitCompletionPacket { + wait_completion_packet_handle: { Handle::from_raw }, + remove_signaled_packet, + })), + NtSysno::NtCreateWorkerFactory => Some(sys_req!(NtCreateWorkerFactory { + worker_factory_handle:*, + desired_access, + object_attributes:*, + completion_port_handle:{Handle::from_raw}, + worker_process_handle:{ProcessHandle::from_raw}, + start_routine, + start_parameter, + max_thread_count, + stack_reserve, + stack_commit, + })), + NtSysno::NtSetInformationWorkerFactory => Some(sys_req!( + NtSetInformationWorkerFactory { + worker_factory_handle:{Handle::from_raw}, + worker_factory_information_class, + worker_factory_information:*, + worker_factory_information_length, + } + )), + NtSysno::NtShutdownWorkerFactory => Some(sys_req!(NtShutdownWorkerFactory { + worker_factory_handle:{Handle::from_raw}, + pending_worker_count:*, + })), + NtSysno::NtCreateTimer2 => Some(sys_req!(NtCreateTimer2 { + timer_handle:*, + timer_id:*, + object_attributes:*, + attributes, + desired_access, + })), + NtSysno::NtSetTimer2 => Some(sys_req!(NtSetTimer2 { + timer_handle:{Handle::from_raw}, + due_time:*, + period:*, + parameters:*, + })), + NtSysno::NtOpenEvent => Some(sys_req!(NtOpenEvent { + event_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtSetEvent => Some(sys_req!(NtSetEvent { + event_handle:{Handle::from_raw}, + previous_state:*, + })), + NtSysno::NtResetEvent => Some(sys_req!(NtResetEvent { + event_handle:{Handle::from_raw}, + previous_state:*, + })), + NtSysno::NtClearEvent => Some(sys_req!(NtClearEvent { + event_handle: { Handle::from_raw }, + })), + NtSysno::NtPulseEvent => Some(sys_req!(NtPulseEvent { + event_handle:{Handle::from_raw}, + previous_state:*, + })), + NtSysno::NtQueryEvent => Some(sys_req!(NtQueryEvent { + event_handle:{Handle::from_raw}, + event_information_class, + event_information:*, + event_information_length, + return_length:*, + })), + NtSysno::NtSetEventBoostPriority => Some(sys_req!(NtSetEventBoostPriority { + event_handle: { Handle::from_raw }, + })), + NtSysno::NtOpenFile => Some(sys_req!(NtOpenFile { + file_handle:*, + desired_access, + object_attributes:*, + io_status_block:*, + share_access, + open_options, + })), + NtSysno::NtCreateFile => Some(sys_req!(NtCreateFile { + file_handle:*, + desired_access, + object_attributes:*, + io_status_block:*, + allocation_size:*, + file_attributes, + share_access, + create_disposition, + create_options, + ea_buffer:*, + ea_length, + })), + NtSysno::NtWriteFile => Some(sys_req!(NtWriteFile { + file_handle:{Handle::from_raw}, + event:{Handle::from_raw}, + apc_routine:*, + apc_context:*, + io_status_block:*, + buffer:*, + length, + byte_offset:*, + key:*, + })), + NtSysno::NtQueryVolumeInformationFile => Some(sys_req!(NtQueryVolumeInformationFile { + file_handle:{Handle::from_raw}, + io_status_block:*, + fs_information:*, + length, + fs_information_class, + })), + NtSysno::NtDeviceIoControlFile => Some(sys_req!(NtDeviceIoControlFile { + file_handle:{Handle::from_raw}, + event:{Handle::from_raw}, + apc_routine:*, + apc_context:*, + io_status_block:*, + io_control_code, + input_buffer:*, + input_buffer_length, + output_buffer:*, + output_buffer_length, + })), + NtSysno::NtApphelpCacheControl => Some(sys_req!(NtApphelpCacheControl { + service_class, + service_data:*, + })), + NtSysno::NtOpenKey => Some(sys_req!(NtOpenKey { + key_handle:*, + desired_access, + object_attributes:*, + })), + NtSysno::NtQueryValueKey => Some(sys_req!(NtQueryValueKey { + key_handle:{Handle::from_raw}, + value_name:*, + key_value_information_class, + key_value_information:*, + length, + result_length:*, + })), + NtSysno::NtGetNlsSectionPtr => Some(sys_req!(NtGetNlsSectionPtr { + section_type, + section_data, + context_data, + section_pointer:*, + section_size:*, + })), + NtSysno::NtInitializeNlsFiles => Some(sys_req!(NtInitializeNlsFiles { + base_address:*, + default_locale_id:*, + default_casing_table_size:*, + })), + NtSysno::NtQueryDefaultLocale => Some(sys_req!(NtQueryDefaultLocale { + user_profile, + default_locale_id:*, + })), + NtSysno::NtSetDefaultLocale => Some(sys_req!(NtSetDefaultLocale { + user_profile, + default_locale_id, + })), + NtSysno::NtQueryDefaultUILanguage => Some(sys_req!(NtQueryDefaultUILanguage { + default_ui_language:*, + })), + NtSysno::NtSetDefaultUILanguage => Some(sys_req!(NtSetDefaultUILanguage { + default_ui_language, + })), + NtSysno::NtQueryInstallUILanguage => Some(sys_req!(NtQueryInstallUILanguage { + install_ui_language:*, + })), + NtSysno::NtQueryPerformanceCounter => Some(sys_req!(NtQueryPerformanceCounter { + performance_counter:*, + performance_frequency:*, + })), + NtSysno::NtQuerySystemInformation => Some(sys_req!(NtQuerySystemInformation { + system_information_class, + system_information:*, + system_information_length, + return_length:*, + })), + NtSysno::NtQuerySystemInformationEx => Some(sys_req!(NtQuerySystemInformationEx { + system_information_class, + input_buffer:*, + input_buffer_length, + system_information:*, + system_information_length, + return_length:*, + })), + NtSysno::NtQueryWnfStateData => Some(sys_req!(NtQueryWnfStateData { + state_name:*, + type_id:*, + explicit_scope:*, + change_stamp:*, + buffer:*, + buffer_size:*, + })), + NtSysno::NtCreateWnfStateName => Some(sys_req!(NtCreateWnfStateName { + state_name:*, + name_lifetime, + data_scope, + persist_data, + type_id:*, + maximum_state_size, + security_descriptor:*, + })), + NtSysno::NtUpdateWnfStateData => Some(sys_req!(NtUpdateWnfStateData { + state_name:*, + buffer:*, + buffer_size, + type_id:*, + explicit_scope:*, + matching_change_stamp, + check_stamp, + })), + NtSysno::NtDeleteWnfStateData => Some(sys_req!(NtDeleteWnfStateData { + state_name:*, + explicit_scope:*, + })), + NtSysno::NtDeleteWnfStateName => Some(sys_req!(NtDeleteWnfStateName { + state_name:*, + })), + NtSysno::NtQueryWnfStateNameInformation => { + Some(sys_req!(NtQueryWnfStateNameInformation { + state_name:*, + name_information_class, + explicit_scope:*, + buffer:*, + buffer_size, + })) + } + NtSysno::NtQuerySection => Some(sys_req!(NtQuerySection { + section_handle: { Handle::from_raw }, + section_information_class, + section_information:*, + section_information_length, + return_length:*, + })), + NtSysno::NtQueryInformationProcess => Some(sys_req!(NtQueryInformationProcess { + process_handle: { ProcessHandle::from_raw }, + process_information_class, + process_information:*, + process_information_length, + return_length:*, + })), + NtSysno::NtSetInformationProcess => Some(sys_req!(NtSetInformationProcess { + process_handle: { ProcessHandle::from_raw }, + process_information_class, + process_information:*, + process_information_length, + })), + NtSysno::NtSetInformationThread => Some(sys_req!(NtSetInformationThread { + thread_handle: { ThreadHandle::from_raw }, + thread_information_class, + thread_information:*, + thread_information_length, + })), + NtSysno::NtOpenThreadToken => Some(sys_req!(NtOpenThreadToken { + thread_handle: { ThreadHandle::from_raw }, + desired_access, + open_as_self, + token_handle:*, + })), + NtSysno::NtOpenThreadTokenEx => Some(sys_req!(NtOpenThreadTokenEx { + thread_handle: { ThreadHandle::from_raw }, + desired_access, + open_as_self, + handle_attributes, + token_handle:*, + })), + NtSysno::NtOpenProcessToken => Some(sys_req!(NtOpenProcessToken { + process_handle: { ProcessHandle::from_raw }, + desired_access, + token_handle:*, + })), + NtSysno::NtOpenProcessTokenEx => Some(sys_req!(NtOpenProcessTokenEx { + process_handle: { ProcessHandle::from_raw }, + desired_access, + handle_attributes, + token_handle:*, + })), + NtSysno::NtQueryInformationToken => Some(sys_req!(NtQueryInformationToken { + token_handle: { Handle::from_raw }, + token_information_class, + token_information:*, + token_information_length, + return_length:*, + })), + NtSysno::NtQuerySecurityAttributesToken => { + Some(sys_req!(NtQuerySecurityAttributesToken { + token_handle: { Handle::from_raw }, + attributes:*, + number_of_attributes, + buffer:*, + length, + return_length:*, + })) + } + NtSysno::NtConvertBetweenAuxiliaryCounterAndPerformanceCounter => Some( + sys_req!(NtConvertBetweenAuxiliaryCounterAndPerformanceCounter { + flag, + source:*, + destination:*, + conversion_error:*, + }), + ), + NtSysno::NtAllocateVirtualMemory => Some(sys_req!(NtAllocateVirtualMemory { + process_handle: { ProcessHandle::from_raw }, + base_address:*, + zero_bits, + region_size:*, + allocation_type, + protect, + })), + NtSysno::NtAllocateVirtualMemoryEx => Some(sys_req!(NtAllocateVirtualMemoryEx { + process_handle: { ProcessHandle::from_raw }, + base_address:*, + region_size:*, + allocation_type, + protect, + extended_parameters:*, + extended_parameter_count, + })), + NtSysno::NtFreeVirtualMemory => Some(sys_req!(NtFreeVirtualMemory { + process_handle: { ProcessHandle::from_raw }, + base_address:*, + region_size:*, + free_type, + })), + NtSysno::NtProtectVirtualMemory => Some(sys_req!(NtProtectVirtualMemory { + process_handle: { ProcessHandle::from_raw }, + base_address:*, + region_size:*, + new_protect, + old_protect:*, + })), + NtSysno::NtQueryVirtualMemory => Some(sys_req!(NtQueryVirtualMemory { + process_handle: { ProcessHandle::from_raw }, + base_address, + memory_information_class, + memory_information:*, + memory_information_length, + return_length:*, + })), + NtSysno::NtMapViewOfSection => Some(sys_req!(NtMapViewOfSection { + section_handle: { Handle::from_raw }, + process_handle: { ProcessHandle::from_raw }, + base_address:*, + zero_bits, + commit_size, + section_offset:*, + view_size:*, + inherit_disposition, + allocation_type, + page_protection, + })), + NtSysno::NtMapViewOfSectionEx => Some(sys_req!(NtMapViewOfSectionEx { + section_handle: { Handle::from_raw }, + process_handle: { ProcessHandle::from_raw }, + base_address:*, + zero_bits, + commit_size, + section_offset:*, + view_size:*, + inherit_disposition, + allocation_type, + page_protection, + extended_parameters:*, + extended_parameter_count, + })), + NtSysno::NtUnmapViewOfSection => Some(sys_req!(NtUnmapViewOfSection { + process_handle: { ProcessHandle::from_raw }, + base_address, + })), + NtSysno::NtUnmapViewOfSectionEx => Some(sys_req!(NtUnmapViewOfSectionEx { + process_handle: { ProcessHandle::from_raw }, + base_address, + flags, + })), + NtSysno::NtContinue => Some(sys_req!(NtContinue { + context:*, + test_alert: { |value: u8| value != 0 }, + })), + NtSysno::NtTerminateProcess => Some(sys_req!(NtTerminateProcess { + process_handle: { ProcessHandle::from_raw }, + exit_status, + })), + NtSysno::NtTestAlert => Some(SyscallRequest::NtTestAlert), + NtSysno::NtManageHotPatch => Some(SyscallRequest::NtManageHotPatch), + _ => None, + } + } +} + +fn win_syscall_arg( + pt_regs: &litebox_common_linux::PtRegs, + idx: usize, +) -> Option { + match idx { + 0 => Some(pt_regs.r10), + 1 => Some(pt_regs.rdx), + 2 => Some(pt_regs.r8), + 3 => Some(pt_regs.r9), + idx => { + // The first stack argument sits after the return address and x64 shadow space. + let stack_offset = FIRST_STACK_ARGUMENT_OFFSET + .checked_add((idx - 4).checked_mul(size_of::())?)?; + let stack_address = pt_regs.rsp.checked_add(stack_offset)?; + let stack_arg = Platform::RawConstPointer::::from_usize(stack_address); + stack_arg.read_at_offset(0) + } + } +} + +fn win_sys_req_arg( + pt_regs: &litebox_common_linux::PtRegs, + idx: usize, +) -> Option { + Some(T::reinterpret_truncated_from_usize(win_syscall_arg::< + Platform, + >(pt_regs, idx)?)) +} + +fn win_sys_req_ptr< + Platform: RawPointerProvider, + T: zerocopy::FromBytes, + P: ReinterpretUsizeAsPtr, +>( + pt_regs: &litebox_common_linux::PtRegs, + idx: usize, +) -> Option

{ + Some(P::reinterpret_usize_as_ptr(win_syscall_arg::( + pt_regs, idx, + )?)) +} + +trait ReinterpretTruncatedFromUsize: Sized { + fn reinterpret_truncated_from_usize(value: usize) -> Self; +} + +impl ReinterpretTruncatedFromUsize for usize { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + value + } +} + +impl ReinterpretTruncatedFromUsize for u64 { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + value as u64 + } +} + +impl ReinterpretTruncatedFromUsize for isize { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + value.cast_signed() + } +} + +impl ReinterpretTruncatedFromUsize for NtStatus { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + Self::from_raw(value.trunc()) + } +} + +macro_rules! reinterpret_truncated_unsigned { + ($($ty:ty),* $(,)?) => { + $( + impl ReinterpretTruncatedFromUsize for $ty { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + value.trunc() + } + } + )* + }; +} + +macro_rules! reinterpret_truncated_signed { + ($($sty:ty),* $(,)?) => { + $( + impl ReinterpretTruncatedFromUsize for $sty { + fn reinterpret_truncated_from_usize(value: usize) -> Self { + value.cast_signed().trunc() + } + } + )* + }; +} + +reinterpret_truncated_unsigned!(u8, u16, u32); +reinterpret_truncated_signed!(i8, i16, i32); + +trait ReinterpretUsizeAsPtr: Sized { + fn reinterpret_usize_as_ptr(value: usize) -> Self; +} + +impl> + ReinterpretUsizeAsPtr> for P +{ + fn reinterpret_usize_as_ptr(value: usize) -> Self { + P::from_usize(value) + } +} + +impl> + ReinterpretUsizeAsPtr> for Option

+{ + fn reinterpret_usize_as_ptr(value: usize) -> Self { + if value == 0 { + None + } else { + Some(P::from_usize(value)) + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn handle_encodes_raw_fds_and_rejects_invalid_values() { + let first_handle = Handle::from_raw_fd(0).expect("raw fd 0 should encode"); + assert_eq!(first_handle, Handle::from_raw(1usize << HANDLE_SHIFT)); + assert_eq!(first_handle.raw_fd(), Some(0)); + + let max_raw_fd = (usize::MAX >> HANDLE_SHIFT) - 1; + for raw_fd in [1, 42, max_raw_fd] { + let handle = Handle::from_raw_fd(raw_fd).expect("raw fd should encode"); + assert_eq!(handle.raw_fd(), Some(raw_fd)); + } + + assert_eq!(Handle::from_raw(0).raw_fd(), None); + + for tag in 1..=HANDLE_TAG_MASK { + assert_eq!(Handle::from_raw(tag).raw_fd(), None); + assert_eq!( + Handle::from_raw((2usize << HANDLE_SHIFT) | tag).raw_fd(), + None + ); + } + + assert_eq!(Handle::from_raw_fd(usize::MAX >> HANDLE_SHIFT), None); + assert_eq!(Handle::from_raw_fd(usize::MAX), None); + } +} diff --git a/litebox_shim_windows/src/syscalls/nls.rs b/litebox_shim_windows/src/syscalls/nls.rs new file mode 100644 index 0000000000..d0935ee3e6 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/nls.rs @@ -0,0 +1,956 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::format; +use alloc::string::String; +use litebox::fd::TypedFd; +use litebox::fs::errors::{FileStatusError, OpenError, PathError, ReadError}; +use litebox::fs::{FileType, Mode, OFlags}; +use litebox::mm::linux::{CreatePagesFlags, MappingError, NonZeroPageSize}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::loader::PAGE_SIZE; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::ProcessEnvironmentBlock; +use crate::{MutPtr, ShimFS, ShimPlatform, Task, probe_guest_output_preserving_value, write_value}; + +pub(crate) const DEFAULT_LOCALE_ID: u32 = 0x0409; + +const ANSI_CODE_PAGE: u32 = 1252; +const OEM_CODE_PAGE: u32 = 437; +const UNICODE_CASE_TABLE: u32 = 10000; +const NLS_SECTION_LOCALE: u32 = 2; +const NLS_SECTION_SORTKEYS: u32 = 9; +const NLS_SECTION_CASEMAP: u32 = 10; +const NLS_SECTION_CODEPAGE: u32 = 11; +const NLS_SECTION_NORMALIZE: u32 = 12; + +struct NlsSectionRequest { + section_type: u32, + section_data: u32, + context_data: usize, + section_pointer: MutPtr, + section_size: Option>, +} + +impl Clone for NlsSectionRequest { + fn clone(&self) -> Self { + *self + } +} + +impl Copy for NlsSectionRequest {} + +#[derive(Clone, Copy)] +struct MappedNlsSection { + address: usize, + len: usize, +} + +struct NlsSectionFile { + fd: TypedFd, + len: usize, +} + +impl Task { + pub(crate) fn sys_nt_get_nls_section_ptr( + &self, + section_type: u32, + section_data: u32, + context_data: usize, + section_pointer: MutPtr, + section_size: Option>, + ) -> NtStatus { + let request = NlsSectionRequest { + section_type, + section_data, + context_data, + section_pointer, + section_size, + }; + + if as litebox::platform::RawConstPointer>::as_usize( + &request.section_pointer, + ) == 0 + { + return NtStatus::INVALID_PARAMETER; + } + + let cache_key = (request.section_type, request.section_data); + if let Some(mapped_section) = self.cached_nls_section(cache_key) { + return self.write_nls_section_result(request, mapped_section, true); + } + + let mapped_section = match self.map_nls_section_file(request) { + Ok(mapped_section) => mapped_section, + Err(status) => { + litebox_util_log::debug!( + section_type = request.section_type, + section_data = request.section_data, + status:? = status; + "NtGetNlsSectionPtr section is not available" + ); + return status; + } + }; + + let (mapped_section, cached) = self.publish_nls_section_mapping(cache_key, mapped_section); + let status = self.write_nls_section_result(request, mapped_section, cached); + if status != NtStatus::SUCCESS && !cached { + self.remove_owned_cached_nls_section(cache_key, mapped_section); + } + status + } + + pub(crate) fn sys_nt_initialize_nls_files( + &self, + base_address: MutPtr, + default_locale_id: MutPtr, + _default_casing_table_size: MutPtr, + ) -> NtStatus { + if base_address.as_usize() == 0 { + return NtStatus::ACCESS_VIOLATION; + } + + let request = NlsSectionRequest { + section_type: NLS_SECTION_LOCALE, + section_data: 0, + context_data: 0, + section_pointer: base_address, + section_size: None, + }; + let cache_key = (NLS_SECTION_LOCALE, 0); + let (mapped_section, cached) = + if let Some(mapped_section) = self.cached_nls_section(cache_key) { + (mapped_section, true) + } else { + let mapped_section = match self.map_nls_section_file(request) { + Ok(mapped_section) => mapped_section, + Err(status) => return status, + }; + self.publish_nls_section_mapping(cache_key, mapped_section) + }; + + let locale_id = self + .process + .system_lcid + .load(core::sync::atomic::Ordering::Relaxed); + if probe_guest_output_preserving_value::(base_address).is_err() + || probe_guest_output_preserving_value::(default_locale_id).is_err() + || default_locale_id.write_at_offset(0, locale_id).is_none() + || base_address + .write_at_offset(0, mapped_section.address) + .is_none() + { + if !cached { + self.remove_owned_cached_nls_section(cache_key, mapped_section); + } + return NtStatus::ACCESS_VIOLATION; + } + + litebox_util_log::debug!( + base:% = format_args!("{:#x}", mapped_section.address), + default_locale_id = locale_id; + "Handled NtInitializeNlsFiles syscall" + ); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_default_locale( + &self, + user_profile: u8, + default_locale_id: MutPtr, + ) -> NtStatus { + let locale_id = if user_profile == 0 { + self.process + .system_lcid + .load(core::sync::atomic::Ordering::Relaxed) + } else { + self.process + .user_lcid + .load(core::sync::atomic::Ordering::Relaxed) + }; + write_required_output::(default_locale_id, locale_id) + } + + pub(crate) fn sys_nt_set_default_locale( + &self, + user_profile: u8, + default_locale_id: u32, + ) -> NtStatus { + if user_profile == 0 { + self.process + .system_lcid + .store(default_locale_id, core::sync::atomic::Ordering::Relaxed); + } else { + self.process + .user_lcid + .store(default_locale_id, core::sync::atomic::Ordering::Relaxed); + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_default_ui_language( + &self, + default_ui_language: MutPtr, + ) -> NtStatus { + let lang_id = lang_id_from_locale_id( + self.process + .user_ui_language + .load(core::sync::atomic::Ordering::Relaxed), + ); + write_required_output::(default_ui_language, lang_id) + } + + pub(crate) fn sys_nt_set_default_ui_language(&self, default_ui_language: u16) -> NtStatus { + self.process.user_ui_language.store( + u32::from(default_ui_language), + core::sync::atomic::Ordering::Relaxed, + ); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_install_ui_language( + &self, + install_ui_language: MutPtr, + ) -> NtStatus { + let lang_id = lang_id_from_locale_id( + self.process + .system_lcid + .load(core::sync::atomic::Ordering::Relaxed), + ); + write_required_output::(install_ui_language, lang_id) + } + + fn map_nls_section_file( + &self, + request: NlsSectionRequest, + ) -> Result { + let section_file = self.open_nls_section_file(request)?; + let section_len = section_file.len; + let alloc_len = match nls_section_alloc_len(section_len) { + Ok(alloc_len) => alloc_len, + Err(status) => { + let _ = self.fs.close(§ion_file.fd); + return Err(status); + } + }; + let Some(page_len) = NonZeroPageSize::::new(alloc_len) else { + let _ = self.fs.close(§ion_file.fd); + return Err(NtStatus::INVALID_PARAMETER); + }; + + let mut copy_status = None; + // SAFETY: No fixed address is requested, so the page manager chooses an unused guest + // range. The callback only initializes the newly allocated pages before they are exposed. + let mapping = unsafe { + self.global.page_manager.create_readable_pages( + None, + page_len, + CreatePagesFlags::POPULATE_PAGES_IMMEDIATELY, + |ptr| match self.copy_nls_section_file(§ion_file.fd, section_len, ptr) { + Ok(copied) => Ok(copied), + Err(status) => { + copy_status = Some(status); + Err(MappingError::OutOfMemory) + } + }, + ) + }; + let _ = self.fs.close(§ion_file.fd); + let mapping = mapping.map_err(|_| copy_status.unwrap_or(NtStatus::NO_MEMORY))?; + Ok(MappedNlsSection { + address: mapping.as_usize(), + len: alloc_len, + }) + } + + fn open_nls_section_file( + &self, + request: NlsSectionRequest, + ) -> Result, NtStatus> { + let path = nls_section_file_path(request.section_type, request.section_data)?; + let fd = self + .fs + .open(path.as_str(), OFlags::RDONLY, Mode::empty()) + .map_err(map_nls_open_error)?; + + let status = match self.fs.fd_file_status(&fd) { + Ok(status) => status, + Err(error) => { + let _ = self.fs.close(&fd); + return Err(map_nls_file_status_error(error)); + } + }; + if status.file_type != FileType::RegularFile { + let _ = self.fs.close(&fd); + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + } + if status.size == 0 { + let _ = self.fs.close(&fd); + return Err(NtStatus::OBJECT_NAME_NOT_FOUND); + } + + Ok(NlsSectionFile { + fd, + len: status.size, + }) + } + + fn copy_nls_section_file( + &self, + fd: &TypedFd, + section_len: usize, + output: MutPtr, + ) -> Result { + let mut offset = 0; + while offset < section_len { + let mut chunk = [0; PAGE_SIZE]; + let remaining = section_len - offset; + let chunk_len = remaining.min(PAGE_SIZE); + let read = self + .fs + .read(fd, &mut chunk[..chunk_len], Some(offset)) + .map_err(map_nls_read_error)?; + if read == 0 { + return Err(NtStatus::END_OF_FILE); + } + let Ok(output_offset) = isize::try_from(offset) else { + return Err(NtStatus::INVALID_PARAMETER); + }; + if output + .write_slice_at_offset(output_offset, &chunk[..read]) + .is_none() + { + return Err(NtStatus::ACCESS_VIOLATION); + } + offset += read; + } + Ok(offset) + } + + fn write_nls_section_result( + &self, + request: NlsSectionRequest, + mapped_section: MappedNlsSection, + cached: bool, + ) -> NtStatus { + if probe_guest_output_preserving_value::(request.section_pointer).is_err() + { + return NtStatus::ACCESS_VIOLATION; + } + let Ok(len) = u32::try_from(mapped_section.len) else { + return NtStatus::SECTION_TOO_BIG; + }; + if let Some(section_size) = request.section_size + && section_size.write_at_offset(0, len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if request + .section_pointer + .write_at_offset(0, mapped_section.address) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + self.set_peb_nls_pointer(request.section_data, mapped_section.address); + + litebox_util_log::debug!( + section_type = request.section_type, + section_data = request.section_data, + context_data:% = format_args!("{:#x}", request.context_data), + mapped_address:% = format_args!("{:#x}", mapped_section.address), + section_len = mapped_section.len, + cached = cached; + "Handled NtGetNlsSectionPtr syscall" + ); + + NtStatus::SUCCESS + } + + fn cached_nls_section(&self, cache_key: (u32, u32)) -> Option { + self.process + .nls_section_mappings + .read() + .get(&cache_key) + .copied() + .map(|(address, len)| MappedNlsSection { address, len }) + } + + fn publish_nls_section_mapping( + &self, + cache_key: (u32, u32), + mapped_section: MappedNlsSection, + ) -> (MappedNlsSection, bool) { + let mut mappings = self.process.nls_section_mappings.write(); + if let Some((mapped_address, section_len)) = mappings.get(&cache_key).copied() { + drop(mappings); + self.unmap_owned_nls_section(mapped_section); + return ( + MappedNlsSection { + address: mapped_address, + len: section_len, + }, + true, + ); + } + + mappings.insert(cache_key, (mapped_section.address, mapped_section.len)); + (mapped_section, false) + } + + fn remove_owned_cached_nls_section( + &self, + cache_key: (u32, u32), + mapped_section: MappedNlsSection, + ) { + let mut mappings = self.process.nls_section_mappings.write(); + let remove_cached_mapping = + mappings.get(&cache_key).copied() == Some((mapped_section.address, mapped_section.len)); + if remove_cached_mapping { + mappings.remove(&cache_key); + } + drop(mappings); + + if remove_cached_mapping { + self.unmap_owned_nls_section(mapped_section); + } + } + + fn set_peb_nls_pointer(&self, section_data: u32, mapped_address: usize) { + if self.process.peb_address == 0 { + return; + } + + let Some(field_offset) = (match section_data { + ANSI_CODE_PAGE => Some(core::mem::offset_of!( + ProcessEnvironmentBlock, + ansi_code_page_data + )), + OEM_CODE_PAGE => Some(core::mem::offset_of!( + ProcessEnvironmentBlock, + oem_code_page_data + )), + UNICODE_CASE_TABLE => Some(core::mem::offset_of!( + ProcessEnvironmentBlock, + unicode_case_table_data + )), + _ => None, + }) else { + return; + }; + + let peb_field = + MutPtr::::from_usize(self.process.peb_address + field_offset); + let _ = peb_field.write_at_offset(0, mapped_address); + } + + fn unmap_owned_nls_section(&self, mapped_section: MappedNlsSection) { + // SAFETY: The mapping was created by this syscall path and has not been published on the + // failing path, so no guest execution can hold a valid reference to it yet. + let _ = unsafe { + self.global.page_manager.remove_pages( + MutPtr::::from_usize(mapped_section.address), + mapped_section.len, + ) + }; + } +} + +fn nls_section_file_path(section_type: u32, section_data: u32) -> Result { + match section_type { + NLS_SECTION_LOCALE if section_data == 0 => Ok(String::from("/Windows/System32/locale.nls")), + NLS_SECTION_SORTKEYS if section_data == 0 => Ok(String::from( + "/Windows/Globalization/Sorting/sortdefault.nls", + )), + NLS_SECTION_CASEMAP if section_data == 0 => { + Ok(String::from("/Windows/System32/l_intl.nls")) + } + NLS_SECTION_CASEMAP => Err(NtStatus::UNSUCCESSFUL), + NLS_SECTION_CODEPAGE => Ok(format!("/Windows/System32/c_{section_data:03}.nls")), + NLS_SECTION_NORMALIZE => normalize_nls_file_name(section_data) + .map(|name| format!("/Windows/System32/{name}.nls")) + .ok_or(NtStatus::OBJECT_NAME_NOT_FOUND), + _ => Err(NtStatus::INVALID_PARAMETER_1), + } +} + +fn nls_section_alloc_len(section_len: usize) -> Result { + let alloc_len = section_len + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(NtStatus::SECTION_TOO_BIG)?; + if u32::try_from(alloc_len).is_err() { + return Err(NtStatus::SECTION_TOO_BIG); + } + Ok(alloc_len) +} + +fn lang_id_from_locale_id(locale_id: u32) -> u16 { + u16::try_from(locale_id & u32::from(u16::MAX)).expect("masked locale id fits in a LANGID") +} + +fn normalize_nls_file_name(section_data: u32) -> Option<&'static str> { + match section_data { + 1 => Some("normnfc"), + 2 => Some("normnfd"), + 5 => Some("normnfkc"), + 6 => Some("normnfkd"), + 13 => Some("normidna"), + _ => None, + } +} + +fn write_required_output(output: MutPtr, value: T) -> NtStatus +where + Platform: RawPointerProvider, + T: zerocopy::FromBytes + zerocopy::IntoBytes, +{ + if write_value::(output.as_usize(), value).is_some() { + NtStatus::SUCCESS + } else { + NtStatus::ACCESS_VIOLATION + } +} + +fn map_nls_open_error(error: OpenError) -> NtStatus { + match error { + OpenError::PathError( + PathError::NoSuchFileOrDirectory + | PathError::MissingComponent + | PathError::ComponentNotADirectory, + ) => NtStatus::OBJECT_NAME_NOT_FOUND, + OpenError::PathError(PathError::NoSearchPerms { .. }) | OpenError::AccessNotAllowed => { + NtStatus::ACCESS_DENIED + } + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_nls_file_status_error(error: FileStatusError) -> NtStatus { + match error { + FileStatusError::PathError( + PathError::NoSuchFileOrDirectory + | PathError::MissingComponent + | PathError::ComponentNotADirectory, + ) => NtStatus::OBJECT_NAME_NOT_FOUND, + FileStatusError::PathError(PathError::NoSearchPerms { .. }) => NtStatus::ACCESS_DENIED, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_nls_read_error(error: ReadError) -> NtStatus { + match error { + ReadError::NotForReading => NtStatus::ACCESS_DENIED, + _ => NtStatus::UNSUCCESSFUL, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::mut_ptr; + use alloc::vec; + use litebox::platform::RawPointerProvider; + + extern crate std; + + type TestPlatform = crate::tests::TestPlatform; + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + unsafe extern "system" { + fn NtGetNlsSectionPtr( + section_type: u32, + section_data: u32, + context_data: *mut core::ffi::c_void, + section_pointer: *mut *const u8, + section_size: *mut u32, + ) -> i32; + + fn NtInitializeNlsFiles( + base_address: *mut *const u8, + default_locale_id: *mut u32, + default_casing_table_size: *mut i64, + ) -> i32; + + fn NtQueryDefaultLocale(user_profile: u8, default_locale_id: *mut u32) -> i32; + + fn NtQueryDefaultUILanguage(default_ui_language: *mut u16) -> i32; + + fn NtQueryInstallUILanguage(install_ui_language: *mut u16) -> i32; + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_system32_file_bytes(file_name: &str) -> std::vec::Vec { + std::fs::read( + std::path::PathBuf::from( + std::env::var_os("SystemRoot") + .unwrap_or_else(|| std::ffi::OsString::from(r"C:\Windows")), + ) + .join("System32") + .join(file_name), + ) + .unwrap() + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_status(status: i32) -> NtStatus { + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + #[test] + fn nt_get_nls_section_ptr_maps_file_backed_section() { + let section_bytes = vec![1, 2, 3, 4, 5]; + let task = crate::tests::test_task_with_nls_files(&[( + "/Windows/System32/c_1252.nls", + section_bytes.as_slice(), + )]); + let mut section_pointer = 0usize; + let mut section_size = 0u32; + + assert_eq!( + task.sys_nt_get_nls_section_ptr( + NLS_SECTION_CODEPAGE, + ANSI_CODE_PAGE, + 0, + mut_ptr(&mut section_pointer), + Some(mut_ptr(&mut section_size)), + ), + NtStatus::SUCCESS + ); + + assert_ne!(section_pointer, 0); + assert_eq!(section_size, u32::try_from(PAGE_SIZE).unwrap()); + let mapped = ::RawConstPointer::::from_usize( + section_pointer, + ); + assert_eq!( + mapped.to_owned_slice(section_bytes.len()).unwrap().as_ref(), + section_bytes.as_slice() + ); + + let mut second_section_pointer = 0usize; + assert_eq!( + task.sys_nt_get_nls_section_ptr( + NLS_SECTION_CODEPAGE, + ANSI_CODE_PAGE, + 0, + mut_ptr(&mut second_section_pointer), + None, + ), + NtStatus::SUCCESS + ); + assert_eq!(second_section_pointer, section_pointer); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_get_nls_section_ptr_matches_host_section_content() { + let host_file_bytes = host_system32_file_bytes("c_1252.nls"); + let task = crate::tests::test_task_with_nls_files(&[( + "/Windows/System32/c_1252.nls", + host_file_bytes.as_slice(), + )]); + + let mut host_section_pointer = core::ptr::null::(); + let mut host_section_size = 0u32; + // SAFETY: The pointers reference local output variables, and the section type/data pair is + // the same supported codepage section requested by normal Windows process startup. + let status = unsafe { + NtGetNlsSectionPtr( + NLS_SECTION_CODEPAGE, + ANSI_CODE_PAGE, + core::ptr::null_mut(), + core::ptr::addr_of_mut!(host_section_pointer), + core::ptr::addr_of_mut!(host_section_size), + ) + }; + assert_eq!(host_status(status), NtStatus::SUCCESS); + assert!(!host_section_pointer.is_null()); + + let mut section_pointer = 0usize; + let mut section_size = 0u32; + assert_eq!( + task.sys_nt_get_nls_section_ptr( + NLS_SECTION_CODEPAGE, + ANSI_CODE_PAGE, + 0, + mut_ptr(&mut section_pointer), + Some(mut_ptr(&mut section_size)), + ), + NtStatus::SUCCESS + ); + + let host_section_len = usize::try_from(host_section_size).unwrap(); + assert_eq!(section_size, host_section_size); + let mapped = ::RawConstPointer::::from_usize( + section_pointer, + ); + // SAFETY: A successful host NtGetNlsSectionPtr returned a non-null pointer and size for a + // process-lifetime read-only NLS mapping. + let host_section = + unsafe { core::slice::from_raw_parts(host_section_pointer, host_section_len) }; + assert_eq!( + mapped.to_owned_slice(host_section_len).unwrap().as_ref(), + host_section + ); + } + + #[test] + fn nt_get_nls_section_ptr_rejects_invalid_arguments() { + let bytes = [0xaa]; + let task = crate::tests::test_task_with_nls_files(&[( + "/Windows/System32/c_437.nls", + bytes.as_slice(), + )]); + let mut section_pointer = 0usize; + + assert_eq!( + task.sys_nt_get_nls_section_ptr( + NLS_SECTION_CODEPAGE, + OEM_CODE_PAGE, + 0, + MutPtr::::from_usize(0), + None, + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!( + task.sys_nt_get_nls_section_ptr( + NLS_SECTION_CODEPAGE, + ANSI_CODE_PAGE, + 0, + mut_ptr(&mut section_pointer), + None, + ), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!(section_pointer, 0); + } + + #[test] + fn nls_section_file_path_formats_codepage_names() { + assert_eq!( + nls_section_file_path(NLS_SECTION_CODEPAGE, 37).unwrap(), + "/Windows/System32/c_037.nls" + ); + assert_eq!( + nls_section_file_path(NLS_SECTION_CODEPAGE, ANSI_CODE_PAGE).unwrap(), + "/Windows/System32/c_1252.nls" + ); + } + + #[test] + fn nls_section_alloc_len_rejects_unrepresentable_sections() { + assert_eq!(nls_section_alloc_len(1).unwrap(), PAGE_SIZE); + assert_eq!( + nls_section_alloc_len(usize::MAX), + Err(NtStatus::SECTION_TOO_BIG) + ); + assert_eq!( + nls_section_alloc_len(usize::try_from(u32::MAX).unwrap()), + Err(NtStatus::SECTION_TOO_BIG) + ); + } + + #[test] + fn nt_initialize_nls_files_maps_locale_file() { + let locale_bytes = vec![0x44; PAGE_SIZE + 1]; + let task = crate::tests::test_task_with_nls_files(&[( + "/Windows/System32/locale.nls", + locale_bytes.as_slice(), + )]); + let mut base_address = 0usize; + let mut locale_id = 0u32; + let mut casing_table_size = 0x1234_5678i64; + + assert_eq!( + task.sys_nt_initialize_nls_files( + mut_ptr(&mut base_address), + mut_ptr(&mut locale_id), + mut_ptr(&mut casing_table_size), + ), + NtStatus::SUCCESS + ); + + assert_ne!(base_address, 0); + assert_eq!(locale_id, DEFAULT_LOCALE_ID); + assert_eq!(casing_table_size, 0x1234_5678); + let mapped = + ::RawConstPointer::::from_usize(base_address); + assert_eq!( + mapped.to_owned_slice(locale_bytes.len()).unwrap().as_ref(), + locale_bytes.as_slice() + ); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_initialize_nls_files_matches_host_outputs() { + let host_file_bytes = host_system32_file_bytes("locale.nls"); + let task = crate::tests::test_task_with_nls_files(&[( + "/Windows/System32/locale.nls", + host_file_bytes.as_slice(), + )]); + + let mut host_base_address = core::ptr::null::(); + let mut host_locale_id = 0x1234_5678u32; + let mut host_casing_table_size = 0x1234_5678i64; + // SAFETY: The pointers reference local output variables and mirror the normal process + // startup call shape; the returned mapping is process-lifetime read-only NLS data. + let status = unsafe { + NtInitializeNlsFiles( + core::ptr::addr_of_mut!(host_base_address), + core::ptr::addr_of_mut!(host_locale_id), + core::ptr::addr_of_mut!(host_casing_table_size), + ) + }; + assert_eq!(host_status(status), NtStatus::SUCCESS); + assert!(!host_base_address.is_null()); + + task.process + .system_lcid + .store(host_locale_id, core::sync::atomic::Ordering::Relaxed); + let mut base_address = 0usize; + let mut locale_id = 0x1234_5678u32; + let mut casing_table_size = 0x1234_5678i64; + assert_eq!( + task.sys_nt_initialize_nls_files( + mut_ptr(&mut base_address), + mut_ptr(&mut locale_id), + mut_ptr(&mut casing_table_size), + ), + NtStatus::SUCCESS + ); + + assert_eq!(locale_id, host_locale_id); + assert_eq!(casing_table_size, host_casing_table_size); + let mapped = + ::RawConstPointer::::from_usize(base_address); + // SAFETY: A successful host NtInitializeNlsFiles returned a non-null process-lifetime NLS + // mapping, and the fixture file length bounds the comparison. + let host_section = + unsafe { core::slice::from_raw_parts(host_base_address, host_file_bytes.len()) }; + assert_eq!( + mapped + .to_owned_slice(host_file_bytes.len()) + .unwrap() + .as_ref(), + host_section + ); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn locale_query_syscalls_match_host_outputs() { + let task = crate::tests::test_task(); + let mut host_system_locale = 0u32; + let mut host_user_locale = 0u32; + let mut host_user_ui_language = 0u16; + let mut host_install_ui_language = 0u16; + + // SAFETY: The pointers reference local output variables for read-only host locale queries. + unsafe { + assert_eq!( + host_status(NtQueryDefaultLocale( + 0, + core::ptr::addr_of_mut!(host_system_locale), + )), + NtStatus::SUCCESS + ); + assert_eq!( + host_status(NtQueryDefaultLocale( + 1, + core::ptr::addr_of_mut!(host_user_locale), + )), + NtStatus::SUCCESS + ); + assert_eq!( + host_status(NtQueryDefaultUILanguage(core::ptr::addr_of_mut!( + host_user_ui_language + ))), + NtStatus::SUCCESS + ); + assert_eq!( + host_status(NtQueryInstallUILanguage(core::ptr::addr_of_mut!( + host_install_ui_language + ))), + NtStatus::SUCCESS + ); + } + + task.process + .system_lcid + .store(host_system_locale, core::sync::atomic::Ordering::Relaxed); + task.process + .user_lcid + .store(host_user_locale, core::sync::atomic::Ordering::Relaxed); + task.process.user_ui_language.store( + u32::from(host_user_ui_language), + core::sync::atomic::Ordering::Relaxed, + ); + + let mut locale_id = 0u32; + let mut language = 0u16; + assert_eq!( + task.sys_nt_query_default_locale(0, mut_ptr(&mut locale_id)), + NtStatus::SUCCESS + ); + assert_eq!(locale_id, host_system_locale); + assert_eq!( + task.sys_nt_query_default_locale(1, mut_ptr(&mut locale_id)), + NtStatus::SUCCESS + ); + assert_eq!(locale_id, host_user_locale); + assert_eq!( + task.sys_nt_query_default_ui_language(mut_ptr(&mut language)), + NtStatus::SUCCESS + ); + assert_eq!(language, host_user_ui_language); + task.process.system_lcid.store( + u32::from(host_install_ui_language), + core::sync::atomic::Ordering::Relaxed, + ); + assert_eq!( + task.sys_nt_query_install_ui_language(mut_ptr(&mut language)), + NtStatus::SUCCESS + ); + assert_eq!(language, host_install_ui_language); + } + + #[test] + fn locale_syscalls_query_and_update_process_locale_state() { + let task = crate::tests::test_task(); + let mut locale_id = 0u32; + let mut language = 0u16; + + assert_eq!( + task.sys_nt_query_default_locale(0, mut_ptr(&mut locale_id)), + NtStatus::SUCCESS + ); + assert_eq!(locale_id, DEFAULT_LOCALE_ID); + + assert_eq!(task.sys_nt_set_default_locale(0, 0x0411), NtStatus::SUCCESS); + assert_eq!( + task.sys_nt_query_default_locale(0, mut_ptr(&mut locale_id)), + NtStatus::SUCCESS + ); + assert_eq!(locale_id, 0x0411); + + assert_eq!( + task.sys_nt_set_default_ui_language(0x040c), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_query_default_ui_language(mut_ptr(&mut language)), + NtStatus::SUCCESS + ); + assert_eq!(language, 0x040c); + + assert_eq!( + task.sys_nt_query_install_ui_language(mut_ptr(&mut language)), + NtStatus::SUCCESS + ); + assert_eq!(language, 0x0411); + } +} diff --git a/litebox_shim_windows/src/syscalls/object_manager.rs b/litebox_shim_windows/src/syscalls/object_manager.rs new file mode 100644 index 0000000000..15cb72985b --- /dev/null +++ b/litebox_shim_windows/src/syscalls/object_manager.rs @@ -0,0 +1,2428 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT object manager. + +use alloc::collections::BTreeMap; +use alloc::string::{String, ToString as _}; +use alloc::sync::{Arc, Weak}; +use alloc::vec::Vec; +use core::cmp::Ordering; +use core::hash::{Hash, Hasher}; +use core::marker::PhantomData; +use core::mem::size_of; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::{ + AccessMask, ObjectAttributes, ObjectAttributesFlags, UnicodeString, read_object_attributes, +}; +use crate::syscalls::Handle; +use crate::syscalls::event::EventObject; +use crate::syscalls::section::{ + SectionObject, WINDOWS_SESSION_SHARED_SECTION_OBJECT, WINDOWS_SHARED_SECTION_OBJECT, +}; +use crate::{ + ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_buffer, probe_guest_output_preserving_value, +}; + +const MAX_SYMLINK_REPARSE_DEPTH: usize = 64; +pub(crate) const WINDOWS_API_PORT: &str = r"\Windows\ApiPort"; +const STANDARD_RIGHTS_REQUIRED: u32 = AccessMask::DELETE.bits() + | AccessMask::READ_CONTROL.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits(); + +// Wine's server seeds these object-manager directories during init_directories/create_session; +// ReactOS initializes the same root-style namespace through ObpRootDirectoryObject. +const SEEDED_DIRECTORY_PATHS: &[&str] = &[ + r"\", + r"\??", + r"\BaseNamedObjects", + r"\Device", + r"\Driver", + r"\KnownDlls", + r"\KernelObjects", + r"\NLS", + r"\ObjectTypes", + r"\Sessions", + r"\Sessions\0", + r"\Sessions\0\BaseNamedObjects", + r"\Sessions\0\DosDevices", + r"\Sessions\0\Windows", + r"\Sessions\0\Windows\WindowStations", + r"\Sessions\BNOLINKS", + r"\Windows", +]; + +// Wine's wineboot and ReactOS SMSS create KnownDllPath so ntdll can open/query +// the DOS path prefix for known DLL lookups during loader initialization. +const SEEDED_SYMLINK_PATHS: &[(&str, &str)] = &[ + (r"\??\C:", r"\Device\HarddiskVolume1"), + (r"\SystemRoot", r"\Device\HarddiskVolume1\Windows"), + (r"\KnownDlls\KnownDllPath", r"C:\Windows\System32"), + // TODO(windows-sessions): resolve this through the current session id once + // the shim supports multiple Windows sessions. + ( + WINDOWS_SHARED_SECTION_OBJECT, + WINDOWS_SESSION_SHARED_SECTION_OBJECT, + ), +]; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct DirectoryAccess: u32 { + const QUERY = 0x0001; + const TRAVERSE = 0x0002; + const CREATE_OBJECT = 0x0004; + const CREATE_SUBDIRECTORY = 0x0008; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() + | Self::QUERY.bits() + | Self::TRAVERSE.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | Self::CREATE_OBJECT.bits() + | Self::CREATE_SUBDIRECTORY.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() + | Self::QUERY.bits() + | Self::TRAVERSE.bits(); + const ALL_ACCESS = STANDARD_RIGHTS_REQUIRED + | Self::QUERY.bits() + | Self::TRAVERSE.bits() + | Self::CREATE_OBJECT.bits() + | Self::CREATE_SUBDIRECTORY.bits(); + + const _ = !0; + } +} + +impl DirectoryAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +pub(crate) struct DirectoryObjectSubsystem(PhantomData); + +impl FdEnabledSubsystem for DirectoryObjectSubsystem { + type Entry = DirectoryHandleObject; +} + +impl FdEnabledSubsystemEntry for DirectoryHandleObject {} + +impl crate::WindowsHandleSubsystem + for DirectoryObjectSubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + DirectoryAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct DirectoryHandleObject { + directory: Arc>, +} + +pub(super) struct ObjectNode { + path: String, + name: String, + parent: Option>>, + body: litebox::sync::RwLock>, +} + +pub(crate) struct ObjectManager { + root: Arc>, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub(crate) enum FileDeviceObject { + Filesystem { root_path: String }, + ConsoleDriver, +} + +enum NamedObject { + Directory { + children: BTreeMap>>, + }, + Symlink { + target: String, + }, + Event { + event: Weak>, + }, + Section { + section: Weak>, + }, + FileDevice { + device: FileDeviceObject, + }, + Port, +} + +pub(super) enum ObjectLeafLookup { + Live(T), + Stale, + TypeMismatch, +} + +impl ObjectLeafLookup { + fn map(self, f: impl FnOnce(T) -> U) -> ObjectLeafLookup { + match self { + Self::Live(object) => ObjectLeafLookup::Live(f(object)), + Self::Stale => ObjectLeafLookup::Stale, + Self::TypeMismatch => ObjectLeafLookup::TypeMismatch, + } + } + + pub(super) fn into_result(self) -> Result { + match self { + Self::Live(object) => Ok(object), + Self::Stale => Err(NtStatus::OBJECT_NAME_NOT_FOUND), + Self::TypeMismatch => Err(NtStatus::OBJECT_TYPE_MISMATCH), + } + } +} + +impl ObjectLeafLookup> { + fn from_weak(object: &Weak) -> Self { + object.upgrade().map_or(Self::Stale, Self::Live) + } +} + +macro_rules! object_leaf_accessors { + ($($vis:vis $method:ident, $lookup:ty, $pattern:pat => $value:expr;)+) => { + $( + $vis fn $method(&self) -> $lookup { + match &*self.body.read() { + $pattern => $value, + _ => ObjectLeafLookup::TypeMismatch, + } + } + )+ + }; +} + +macro_rules! object_node_constructors { + ($($method:ident($($arg:ident: $arg_ty:ty),*) => $body:expr;)+) => { + $( + fn $method( + path: String, + parent: Option>>, + name: String, + $($arg: $arg_ty),* + ) -> Self { + Self::new(path, parent, name, $body) + } + )+ + }; +} + +#[derive(Clone, Debug)] +struct ObjectName(String); + +impl ObjectName { + // ReactOS and Wine keep the creator's object name but compare through a + // case-insensitive object-manager lookup key. + fn new(name: &str) -> Self { + Self(name.to_string()) + } +} + +impl PartialEq for ObjectName { + fn eq(&self, other: &Self) -> bool { + self.0.eq_ignore_ascii_case(&other.0) + } +} + +impl Eq for ObjectName {} + +impl PartialOrd for ObjectName { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} + +impl Ord for ObjectName { + fn cmp(&self, other: &Self) -> Ordering { + self.0 + .bytes() + .map(|byte| byte.to_ascii_lowercase()) + .cmp(other.0.bytes().map(|byte| byte.to_ascii_lowercase())) + } +} + +impl Hash for ObjectName { + fn hash(&self, state: &mut H) { + for byte in self.0.bytes() { + byte.to_ascii_lowercase().hash(state); + } + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, IntoBytes, Immutable)] +struct ObjectDirectoryInformation { + name: UnicodeString, + type_name: UnicodeString, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +struct DirectoryEntrySnapshot { + name: String, + type_name: &'static str, +} + +impl ObjectDirectoryInformation { + const fn new(name: UnicodeString, type_name: UnicodeString) -> Self { + Self { name, type_name } + } + + const fn zero() -> Self { + Self { + name: UnicodeString { + length: 0, + maximum_length: 0, + padding_0: [0; 4], + buffer: 0, + }, + type_name: UnicodeString { + length: 0, + maximum_length: 0, + padding_0: [0; 4], + buffer: 0, + }, + } + } +} + +pub(crate) struct DirectoryQueryParameters { + pub(crate) directory_handle: Handle, + pub(crate) buffer: MutPtr, + pub(crate) buffer_length: u32, + pub(crate) return_single_entry: u8, + pub(crate) restart_scan: u8, + pub(crate) context: MutPtr, + pub(crate) return_length: Option>, +} + +impl ObjectNode { + fn new( + path: String, + parent: Option>>, + name: String, + body: NamedObject, + ) -> Self { + Self { + path, + name, + parent, + body: litebox::sync::RwLock::::new(body), + } + } + + object_node_constructors! { + new_directory() => NamedObject::Directory { children: BTreeMap::new() }; + new_symlink(target: String) => NamedObject::Symlink { target }; + new_event(event: Weak>) => NamedObject::Event { event }; + new_section(section: Weak>) => NamedObject::Section { section }; + new_file_device(device: FileDeviceObject) => NamedObject::FileDevice { device }; + new_port() => NamedObject::Port; + } + + fn child(&self, name: &str) -> Option> { + let body = self.body.read(); + let NamedObject::Directory { children } = &*body else { + return None; + }; + children.get(&ObjectName::new(name)).cloned() + } + + fn children_snapshot(&self) -> Result, NtStatus> { + let body = self.body.read(); + let NamedObject::Directory { children } = &*body else { + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + }; + Ok(children + .values() + .filter_map(|child| { + child.type_name().map(|type_name| DirectoryEntrySnapshot { + name: child.name.clone(), + type_name, + }) + }) + .collect()) + } + + pub(super) fn is_directory(&self) -> bool { + matches!(&*self.body.read(), NamedObject::Directory { .. }) + } + + pub(super) fn is_symlink(&self) -> bool { + matches!(&*self.body.read(), NamedObject::Symlink { .. }) + } + + fn is_file_device(&self) -> bool { + matches!(&*self.body.read(), NamedObject::FileDevice { .. }) + } + + object_leaf_accessors! { + directory_object, ObjectLeafLookup<()>, NamedObject::Directory { .. } => ObjectLeafLookup::Live(()); + pub(super) symlink_target, ObjectLeafLookup, NamedObject::Symlink { target } => ObjectLeafLookup::Live(target.clone()); + event_object, ObjectLeafLookup>>, NamedObject::Event { event } => ObjectLeafLookup::from_weak(event); + section_object, ObjectLeafLookup>>, NamedObject::Section { section } => ObjectLeafLookup::from_weak(section); + file_device_object, ObjectLeafLookup, NamedObject::FileDevice { device } => ObjectLeafLookup::Live(device.clone()); + port_object, ObjectLeafLookup<()>, NamedObject::Port => ObjectLeafLookup::Live(()); + } + + fn type_name(&self) -> Option<&'static str> { + match &*self.body.read() { + NamedObject::Directory { .. } => Some("Directory"), + NamedObject::Symlink { .. } => Some("SymbolicLink"), + NamedObject::Event { event } => event.upgrade().map(|_| "Event"), + NamedObject::Section { section } => section.upgrade().map(|_| "Section"), + NamedObject::FileDevice { .. } => Some("Device"), + NamedObject::Port => Some("Port"), + } + } + + fn parent(&self) -> Option> { + self.parent.as_ref().and_then(Weak::upgrade) + } +} + +impl ObjectManager { + fn new() -> Self { + Self { + root: Arc::new(ObjectNode::new_directory( + r"\".to_string(), + None, + String::new(), + )), + } + } + + pub(super) fn parent_directory_exists(&self, path: &str) -> bool { + let path = trim_trailing_directory_path(path); + if path == r"\" { + return false; + } + let Some(index) = path.rfind('\\') else { + return false; + }; + let parent = if index == 0 { r"\" } else { &path[..index] }; + self.resolve_directory(parent).is_ok() + } + + fn create_directory( + &self, + path: &str, + on_exists: impl FnOnce(Arc>) -> NtStatus, + on_created: impl FnOnce(Arc>) -> NtStatus, + ) -> NtStatus { + self.create_child( + path, + |node| { + if node.is_directory() { + ObjectLeafLookup::Live(Arc::clone(node)) + } else { + ObjectLeafLookup::TypeMismatch + } + }, + ObjectNode::new_directory, + NtStatus::OBJECT_TYPE_MISMATCH, + on_exists, + on_created, + ) + } + + pub(super) fn create_symlink( + &self, + path: &str, + target: String, + on_exists: impl FnOnce(Arc>) -> NtStatus, + on_created: impl FnOnce(Arc>) -> NtStatus, + ) -> NtStatus { + self.create_child( + path, + |node| { + if node.is_symlink() { + ObjectLeafLookup::Live(Arc::clone(node)) + } else { + ObjectLeafLookup::TypeMismatch + } + }, + |path, parent, name| ObjectNode::new_symlink(path, parent, name, target), + NtStatus::OBJECT_TYPE_MISMATCH, + on_exists, + on_created, + ) + } + + pub(super) fn create_event( + &self, + path: &str, + event: &Arc>, + on_exists: impl FnOnce(Arc>) -> NtStatus, + on_created: impl FnOnce() -> NtStatus, + ) -> NtStatus { + let event = Arc::downgrade(event); + self.create_child( + path, + |node| node.event_object(), + |path, parent, name| ObjectNode::new_event(path, parent, name, event), + NtStatus::OBJECT_TYPE_MISMATCH, + on_exists, + |_| on_created(), + ) + } + + pub(crate) fn create_section( + &self, + path: &str, + section: &Arc>, + ) -> NtStatus { + let section = Arc::downgrade(section); + self.create_child( + path, + |node| node.section_object(), + |path, parent, name| ObjectNode::new_section(path, parent, name, section), + NtStatus::OBJECT_NAME_EXISTS, + |_| NtStatus::OBJECT_NAME_EXISTS, + |_| NtStatus::SUCCESS, + ) + } + + fn create_file_device(&self, path: &str, device: FileDeviceObject) -> NtStatus { + self.create_child( + path, + |node| node.file_device_object(), + |path, parent, name| ObjectNode::new_file_device(path, parent, name, device), + NtStatus::OBJECT_TYPE_MISMATCH, + |_| NtStatus::OBJECT_NAME_EXISTS, + |_| NtStatus::SUCCESS, + ) + } + + fn create_port(&self, path: &str) -> NtStatus { + self.create_child( + path, + |node| node.port_object(), + ObjectNode::new_port, + NtStatus::OBJECT_TYPE_MISMATCH, + |()| NtStatus::OBJECT_NAME_EXISTS, + |_| NtStatus::SUCCESS, + ) + } + + fn create_child( + &self, + path: &str, + existing_object: impl Fn(&Arc>) -> ObjectLeafLookup, + construct: impl FnOnce( + String, + Option>>, + String, + ) -> ObjectNode, + mismatch_status: NtStatus, + on_exists: impl FnOnce(T) -> NtStatus, + on_created: impl FnOnce(Arc>) -> NtStatus, + ) -> NtStatus { + let tail = match absolute_path_tail(path) { + Ok(tail) => tail, + Err(status) => return status, + }; + if tail.is_empty() { + return match existing_object(&self.root) { + ObjectLeafLookup::Live(object) => on_exists(object), + ObjectLeafLookup::Stale => NtStatus::OBJECT_NAME_NOT_FOUND, + ObjectLeafLookup::TypeMismatch => mismatch_status, + }; + } + + let (parent_tail, leaf_name) = match tail.rsplit_once('\\') { + Some((parent, leaf)) => (parent, leaf), + None => ("", tail), + }; + if leaf_name.is_empty() { + return NtStatus::OBJECT_NAME_INVALID; + } + + let parent = match self.resolve_tail(parent_tail, NtStatus::OBJECT_PATH_NOT_FOUND, true) { + Ok((parent, remaining)) if remaining.is_empty() => parent, + Ok((_, remaining)) => { + return unresolved_tail_status(&remaining, NtStatus::OBJECT_PATH_NOT_FOUND); + } + Err(status) => return status, + }; + let mut body = parent.body.write(); + let NamedObject::Directory { children } = &mut *body else { + return NtStatus::OBJECT_TYPE_MISMATCH; + }; + let leaf_key = ObjectName::new(leaf_name); + if let Some(existing) = children.get(&leaf_key).cloned() { + match existing_object(&existing) { + ObjectLeafLookup::Live(object) => return on_exists(object), + ObjectLeafLookup::Stale => { + children.remove(&leaf_key); + } + ObjectLeafLookup::TypeMismatch => return mismatch_status, + } + } + + let node = Arc::new(construct( + join_directory_path(&parent.path, leaf_name), + Some(Arc::downgrade(&parent)), + leaf_name.to_string(), + )); + debug_assert!(node.parent().is_some()); + let status = on_created(Arc::clone(&node)); + if status == NtStatus::SUCCESS { + children.insert(leaf_key, node); + } + status + } + + pub(super) fn resolve_directory( + &self, + path: &str, + ) -> Result>, NtStatus> { + self.resolve_object_leaf(path, true, |node| { + node.directory_object().map(|()| Arc::clone(node)) + }) + } + + pub(super) fn resolve_symlink( + &self, + path: &str, + follow_final_symlink: bool, + ) -> Result>, NtStatus> { + self.resolve_object_leaf(path, follow_final_symlink, |node| { + node.symlink_target().map(|_| Arc::clone(node)) + }) + } + + pub(super) fn resolve_event(&self, path: &str) -> Result>, NtStatus> { + self.resolve_object_leaf(path, false, |node| node.event_object()) + } + + pub(super) fn resolve_section( + &self, + path: &str, + ) -> Result>, NtStatus> { + self.resolve_object_leaf(path, true, |node| node.section_object()) + } + + pub(crate) fn resolve_file_device( + &self, + path: &str, + ) -> Result<(FileDeviceObject, String), NtStatus> { + let tail = absolute_path_tail(path)?; + let (node, remaining) = self.resolve_tail(tail, NtStatus::OBJECT_NAME_NOT_FOUND, true)?; + if node.is_file_device() { + return Ok((node.file_device_object().into_result()?, remaining)); + } + if remaining.is_empty() { + Err(NtStatus::OBJECT_TYPE_MISMATCH) + } else { + Err(unresolved_tail_status( + &remaining, + NtStatus::OBJECT_NAME_NOT_FOUND, + )) + } + } + + pub(crate) fn resolve_port(&self, path: &str) -> Result<(), NtStatus> { + self.resolve_object_leaf(path, false, |node| { + if node.path == path { + node.port_object() + } else { + ObjectLeafLookup::Stale + } + }) + } + + fn resolve_object_leaf( + &self, + path: &str, + follow_final_symlink: bool, + lookup: impl FnOnce(&Arc>) -> ObjectLeafLookup, + ) -> Result { + let tail = absolute_path_tail(path)?; + let (node, remaining) = + self.resolve_tail(tail, NtStatus::OBJECT_NAME_NOT_FOUND, follow_final_symlink)?; + if !remaining.is_empty() { + return Err(unresolved_tail_status( + &remaining, + NtStatus::OBJECT_NAME_NOT_FOUND, + )); + } + lookup(&node).into_result() + } + + fn seed_directory(&self, path: &str) { + let status = self.create_directory(path, |_| NtStatus::SUCCESS, |_| NtStatus::SUCCESS); + assert!( + status == NtStatus::SUCCESS, + "seeded NT object directory must have seeded ancestors: {status:?}" + ); + } + + fn seed_symlink(&self, path: &str, target: &str) { + let status = self.create_symlink( + path, + target.to_string(), + |_| NtStatus::SUCCESS, + |_| NtStatus::SUCCESS, + ); + assert!( + status == NtStatus::SUCCESS, + "seeded NT object symbolic link must have seeded ancestors: {status:?}" + ); + } + + fn seed_file_device(&self, path: &str, device: FileDeviceObject) { + let status = self.create_file_device(path, device); + assert!( + status == NtStatus::SUCCESS, + "seeded NT file device must have seeded ancestors: {status:?}" + ); + } + + fn seed_port(&self, path: &str) { + let status = self.create_port(path); + assert!( + status == NtStatus::SUCCESS, + "seeded NT port must have seeded ancestors: {status:?}" + ); + } + + fn resolve_tail( + &self, + tail: &str, + final_missing_status: NtStatus, + follow_final_symlink: bool, + ) -> Result<(Arc>, String), NtStatus> { + let mut tail = tail.to_string(); + for _ in 0..=MAX_SYMLINK_REPARSE_DEPTH { + let (node, remaining) = match self.resolve_tail_once(&tail) { + Ok(resolution) => resolution, + Err(NtStatus::OBJECT_NAME_NOT_FOUND) => return Err(final_missing_status), + Err(status) => return Err(status), + }; + if node.is_symlink() && (!remaining.is_empty() || follow_final_symlink) { + tail = reparse_tail(&node, &remaining)?; + continue; + } + return Ok((node, remaining)); + } + Err(NtStatus::NAME_TOO_LONG) + } + + fn resolve_tail_once( + &self, + tail: &str, + ) -> Result<(Arc>, String), NtStatus> { + if tail.is_empty() { + return Ok((Arc::clone(&self.root), String::new())); + } + + let mut current = Arc::clone(&self.root); + let mut components = tail.split('\\').peekable(); + while let Some(component) = components.next() { + if component.is_empty() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + let final_component = components.peek().is_none(); + let missing_status = if final_component { + NtStatus::OBJECT_NAME_NOT_FOUND + } else { + NtStatus::OBJECT_PATH_NOT_FOUND + }; + let child = current.child(component).ok_or(missing_status)?; + if !child.is_directory() { + return Ok((child, components.collect::>().join("\\"))); + } + current = child; + } + Ok((current, String::new())) + } +} + +fn reparse_tail( + node: &ObjectNode, + remaining: &str, +) -> Result { + // This is the lazy-resolution point paired with NtCreateSymbolicLinkObject + // storing the target without lookup. + let target = normalize_reparse_target(&node.symlink_target().into_result()?)?; + let target_tail = absolute_path_tail(&target)?; + if target_tail.is_empty() { + Ok(remaining.to_string()) + } else if remaining.is_empty() { + Ok(target_tail.to_string()) + } else { + Ok(alloc::format!("{target_tail}\\{remaining}")) + } +} + +fn unresolved_tail_status(remaining: &str, final_missing_status: NtStatus) -> NtStatus { + debug_assert!(!remaining.is_empty()); + if remaining.split('\\').any(str::is_empty) { + NtStatus::OBJECT_NAME_INVALID + } else if remaining.contains('\\') { + NtStatus::OBJECT_PATH_NOT_FOUND + } else { + final_missing_status + } +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub(super) struct DirectoryName { + pub(super) original_path: String, +} + +fn trim_trailing_directory_path(path: &str) -> &str { + if path == r"\" { + path + } else { + path.trim_end_matches('\\') + } +} + +fn normalize_reparse_target(path: &str) -> Result { + if !path.starts_with('\\') { + return Err(NtStatus::OBJECT_PATH_SYNTAX_BAD); + } + if path.len() > 1 && path[1..].contains(r"\\") { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(trim_trailing_directory_path(path).to_string()) +} + +fn absolute_path_tail(path: &str) -> Result<&str, NtStatus> { + let path = trim_trailing_directory_path(path); + if path == r"\" { + return Ok(""); + } + path.strip_prefix('\\') + .ok_or(NtStatus::OBJECT_PATH_SYNTAX_BAD) +} + +fn join_directory_path(root_path: &str, name: &str) -> String { + if root_path == r"\" { + alloc::format!(r"\{name}") + } else { + alloc::format!(r"{root_path}\{name}") + } +} + +fn read_directory_name_string( + object_name: usize, +) -> Result, NtStatus> { + debug_assert!(object_name != 0); + let unicode_string = ConstPtr::::from_usize(object_name) + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + if unicode_string.length == 0 { + return Ok(None); + } + if !unicode_string.length.is_multiple_of(2) { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + if unicode_string.buffer == 0 { + return Err(NtStatus::ACCESS_VIOLATION); + } + Ok(Some(unicode_string.read_string::()?)) +} + +fn utf16_byte_len(value: &str) -> Result { + let len = value + .encode_utf16() + .count() + .checked_mul(size_of::()) + .ok_or(NtStatus::NAME_TOO_LONG)?; + if len > u16::MAX as usize { + return Err(NtStatus::NAME_TOO_LONG); + } + Ok(len) +} + +fn directory_record_size(entry: &DirectoryEntrySnapshot) -> Result { + size_of::() + .checked_add(utf16_byte_len(&entry.name)?) + .and_then(|size| size.checked_add(size_of::())) + .and_then(|size| size.checked_add(utf16_byte_len(entry.type_name).ok()?)) + .and_then(|size| size.checked_add(size_of::())) + .ok_or(NtStatus::NAME_TOO_LONG) +} + +fn directory_query_required_size(entries: &[DirectoryEntrySnapshot]) -> Result { + entries + .iter() + .try_fold(size_of::(), |size, entry| { + size.checked_add(directory_record_size(entry)?) + .ok_or(NtStatus::NAME_TOO_LONG) + }) +} + +fn byte_offset(offset: usize) -> Result { + isize::try_from(offset).map_err(|_| NtStatus::BUFFER_TOO_SMALL) +} + +fn write_utf16_nul_terminated( + buffer: MutPtr, + offset: usize, + value: &str, +) -> Result<(), NtStatus> { + let mut bytes = Vec::new(); + for unit in value.encode_utf16() { + bytes.extend_from_slice(&unit.to_le_bytes()); + } + bytes.extend_from_slice(&0u16.to_le_bytes()); + buffer + .write_slice_at_offset(byte_offset(offset)?, &bytes) + .ok_or(NtStatus::ACCESS_VIOLATION) +} + +fn output_unicode_string( + buffer_base: usize, + offset: usize, + len: usize, +) -> Result { + let len = u16::try_from(len).map_err(|_| NtStatus::NAME_TOO_LONG)?; + let maximum_length = len + .checked_add(u16::try_from(size_of::()).expect("WCHAR size fits in USHORT")) + .ok_or(NtStatus::NAME_TOO_LONG)?; + Ok(UnicodeString { + length: len, + maximum_length, + padding_0: [0; 4], + buffer: buffer_base + .checked_add(offset) + .ok_or(NtStatus::NAME_TOO_LONG)?, + }) +} + +fn write_directory_records( + buffer: MutPtr, + buffer_base: usize, + entries: &[DirectoryEntrySnapshot], +) -> Result<(), NtStatus> { + let header_size = size_of::(); + let mut string_offset = entries + .len() + .checked_add(1) + .and_then(|records| records.checked_mul(header_size)) + .ok_or(NtStatus::NAME_TOO_LONG)?; + + for (index, entry) in entries.iter().enumerate() { + let name_len = utf16_byte_len(&entry.name)?; + let type_len = utf16_byte_len(entry.type_name)?; + let name_offset = string_offset; + let type_offset = name_offset + .checked_add(name_len) + .and_then(|offset| offset.checked_add(size_of::())) + .ok_or(NtStatus::NAME_TOO_LONG)?; + let record = ObjectDirectoryInformation::new( + output_unicode_string(buffer_base, name_offset, name_len)?, + output_unicode_string(buffer_base, type_offset, type_len)?, + ); + buffer + .write_slice_at_offset( + byte_offset( + index + .checked_mul(header_size) + .ok_or(NtStatus::NAME_TOO_LONG)?, + )?, + record.as_bytes(), + ) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + write_utf16_nul_terminated::(buffer, name_offset, &entry.name)?; + write_utf16_nul_terminated::(buffer, type_offset, entry.type_name)?; + string_offset = type_offset + .checked_add(type_len) + .and_then(|offset| offset.checked_add(size_of::())) + .ok_or(NtStatus::NAME_TOO_LONG)?; + } + + let terminator_offset = entries + .len() + .checked_mul(header_size) + .ok_or(NtStatus::NAME_TOO_LONG)?; + buffer + .write_slice_at_offset( + byte_offset(terminator_offset)?, + ObjectDirectoryInformation::zero().as_bytes(), + ) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + Ok(()) +} + +impl Task { + fn directory_entry( + &self, + handle: Handle, + ) -> Result>, NtStatus> + { + self.typed_handle_entry::>(handle) + } + + fn directory_object_for_name_resolution( + &self, + handle: Handle, + ) -> Result>, NtStatus> { + let entry = self.typed_handle_entry_with_access::>( + handle, + DirectoryAccess::TRAVERSE.bits(), + )?; + Ok(entry.with_entry(|entry| Arc::clone(&entry.directory))) + } + + pub(super) fn read_directory_object_attributes( + &self, + object_attributes: Option>, + require_name: bool, + ) -> Result<(Option, Option), NtStatus> { + let Some(object_attributes_ptr) = object_attributes else { + if require_name { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + return Ok((None, None)); + }; + let object_attributes = read_object_attributes::(object_attributes_ptr)?; + + if object_attributes.object_name == 0 { + if require_name { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + if !object_attributes.root_directory.is_null() { + // Wine and ReactOS match Windows: a NULL ObjectName plus RootDirectory + // is invalid, while a present zero-length UNICODE_STRING creates unnamed. + return Err(NtStatus::OBJECT_NAME_INVALID); + } + return Ok((Some(object_attributes), None)); + } + + let Some(raw_name) = read_directory_name_string::(object_attributes.object_name)? + else { + if require_name { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + return Ok((Some(object_attributes), None)); + }; + if raw_name.is_empty() { + if require_name { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + return Ok((Some(object_attributes), None)); + } + + let original_path = if object_attributes.root_directory.is_null() { + if !raw_name.starts_with('\\') { + return Err(NtStatus::OBJECT_PATH_SYNTAX_BAD); + } + raw_name + } else { + if raw_name.starts_with('\\') { + return Err(NtStatus::OBJECT_PATH_SYNTAX_BAD); + } + let root = + self.directory_object_for_name_resolution(object_attributes.root_directory)?; + join_directory_path(&root.path, &raw_name) + }; + if original_path.len() > 1 && original_path[1..].contains(r"\\") { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(( + Some(object_attributes), + Some(DirectoryName { original_path }), + )) + } + + fn insert_directory_handle( + &self, + directory: Arc>, + granted_access: DirectoryAccess, + ) -> Result { + self.insert_typed_handle::>( + DirectoryHandleObject { directory }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_directory_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_directory(directory: DirectoryHandleObject) { + drop(directory); + } + + pub(crate) fn sys_nt_create_directory_object( + &self, + directory_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + shadow_directory_handle: Handle, + flags: u32, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(directory_handle) { + return status; + } + if flags != 0 { + return NtStatus::INVALID_PARAMETER; + } + if !shadow_directory_handle.is_null() + && let Err(status) = self.directory_entry(shadow_directory_handle) + { + return status; + } + let (object_attributes, directory_name) = + match self.read_directory_object_attributes(object_attributes, false) { + Ok(value) => value, + Err(status) => return status, + }; + if let Some(object_attributes) = object_attributes + && ObjectAttributesFlags::from_bits_retain(object_attributes.attributes) + .contains(ObjectAttributesFlags::OPENLINK) + { + return NtStatus::INVALID_PARAMETER; + } + let granted_access = DirectoryAccess::from_desired_access(desired_access); + + if let Some(directory_name) = directory_name { + return self.process.object_manager.create_directory( + &directory_name.original_path, + |directory| { + let Some(object_attributes) = object_attributes else { + return NtStatus::INVALID_PARAMETER; + }; + if !ObjectAttributesFlags::from_bits_retain(object_attributes.attributes) + .contains(ObjectAttributesFlags::OPENIF) + { + return NtStatus::OBJECT_NAME_COLLISION; + } + let Ok(handle) = self.insert_directory_handle(directory, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if directory_handle.write_at_offset(0, handle).is_none() { + self.close_directory_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::OBJECT_NAME_EXISTS + }, + |directory| { + let Ok(handle) = self.insert_directory_handle(directory, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if directory_handle.write_at_offset(0, handle).is_none() { + self.close_directory_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + }, + ); + } + + let directory = Arc::new(ObjectNode::new_directory( + String::new(), + None, + String::new(), + )); + let Ok(handle) = self.insert_directory_handle(directory, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if directory_handle.write_at_offset(0, handle).is_none() { + self.close_directory_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_open_directory_object( + &self, + directory_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(directory_handle) { + return status; + } + let directory_name = match self.read_directory_object_attributes(object_attributes, true) { + Ok((Some(object_attributes), Some(directory_name))) => { + if ObjectAttributesFlags::from_bits_retain(object_attributes.attributes) + .contains(ObjectAttributesFlags::OPENLINK) + { + return NtStatus::INVALID_PARAMETER; + } + directory_name + } + Ok((_, None)) => return NtStatus::OBJECT_NAME_INVALID, + Ok((None, Some(_))) => return NtStatus::INVALID_PARAMETER, + Err(status) => return status, + }; + let directory = { + match self + .process + .object_manager + .resolve_directory(&directory_name.original_path) + { + Ok(directory) => directory, + Err(status) => return status, + } + }; + let Ok(handle) = self.insert_directory_handle( + directory, + DirectoryAccess::from_desired_access(desired_access), + ) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if directory_handle.write_at_offset(0, handle).is_none() { + self.close_directory_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + litebox_util_log::debug!( + object_name:% = directory_name.original_path.as_str(), + desired_access:% = format_args!("{desired_access:#x}"); + "Handled NtOpenDirectoryObject syscall" + ); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_directory_object( + &self, + params: DirectoryQueryParameters, + ) -> NtStatus { + let entry = match self.typed_handle_entry_with_access::>( + params.directory_handle, + DirectoryAccess::QUERY.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + let directory = entry.with_entry(|entry| Arc::clone(&entry.directory)); + let entries = match directory.children_snapshot() { + Ok(entries) => entries, + Err(status) => return status, + }; + let buffer_length = params.buffer_length as usize; + if let Err(status) = probe_guest_output_buffer::(params.buffer, buffer_length) { + return status; + } + + let start_index = if params.restart_scan != 0 { + 0 + } else { + let Some(context) = params.context.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + context as usize + }; + if start_index >= entries.len() { + let context = + u32::try_from(entries.len()).expect("directory entry count fits in ULONG"); + // Saturate the opaque resume cookie at end-of-directory so repeated + // continuation calls remain stable. + if params.context.write_at_offset(0, context).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = params.return_length + && return_length.write_at_offset(0, 0).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + return NtStatus::NO_MORE_ENTRIES; + } + + let end_for_required = if params.return_single_entry != 0 { + start_index + 1 + } else { + entries.len() + }; + let total_required = + match directory_query_required_size(&entries[start_index..end_for_required]) { + Ok(size) => size, + Err(status) => return status, + }; + let first_required = + match directory_query_required_size(core::slice::from_ref(&entries[start_index])) { + Ok(size) => size, + Err(status) => return status, + }; + if buffer_length < first_required { + if let Some(return_length) = params.return_length { + let required = u32::try_from(total_required).map_err(|_| NtStatus::NAME_TOO_LONG); + let Ok(required) = required else { + return NtStatus::NAME_TOO_LONG; + }; + if return_length.write_at_offset(0, required).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + } + return if params.return_single_entry != 0 { + NtStatus::BUFFER_TOO_SMALL + } else { + NtStatus::MORE_ENTRIES + }; + } + + let buffer_base = params.buffer.as_usize(); + let mut next_index = start_index; + let mut required_for_written = size_of::(); + while next_index < entries.len() { + let entry_size = match directory_record_size(&entries[next_index]) { + Ok(size) => size, + Err(status) => return status, + }; + if required_for_written + .checked_add(entry_size) + .is_none_or(|needed| needed > buffer_length) + { + break; + } + required_for_written += entry_size; + next_index += 1; + if params.return_single_entry != 0 { + break; + } + } + + if let Err(status) = write_directory_records::( + params.buffer, + buffer_base, + &entries[start_index..next_index], + ) { + return status; + } + let status = if next_index < entries.len() { + NtStatus::MORE_ENTRIES + } else { + NtStatus::SUCCESS + }; + + let context = u32::try_from(next_index).expect("directory entry count fits in ULONG"); + if params.context.write_at_offset(0, context).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = params.return_length { + let Ok(returned) = u32::try_from(total_required) else { + return NtStatus::NAME_TOO_LONG; + }; + if return_length.write_at_offset(0, returned).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + } + status + } +} + +pub(crate) fn seed_object_manager() +-> crate::WindowsObjectManager { + let object_manager = ObjectManager::new(); + for path in SEEDED_DIRECTORY_PATHS { + object_manager.seed_directory(path); + } + object_manager.seed_file_device( + r"\Device\HarddiskVolume1", + FileDeviceObject::Filesystem { + root_path: "/".to_string(), + }, + ); + object_manager.seed_file_device(r"\Device\ConDrv", FileDeviceObject::ConsoleDriver); + object_manager.seed_port(WINDOWS_API_PORT); + for (path, target) in SEEDED_SYMLINK_PATHS { + object_manager.seed_symlink(path, target); + } + object_manager +} + +#[cfg(test)] +mod tests { + use alloc::sync::Arc; + use core::mem::size_of; + + use litebox::platform::ThreadProvider; + use litebox::utils::TruncateExt as _; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::{ObjectAttributes, ObjectAttributesFlags}; + use crate::syscalls::section::{ + WINDOWS_SESSION_SHARED_SECTION_OBJECT, WINDOWS_SHARED_SECTION_OBJECT, + load_time_windows_shared_section, + }; + use crate::tests::{ + TestPlatform, const_ptr, mut_ptr, null_mut_ptr, object_attributes, test_task, + unicode_string, utf16_units, + }; + + const DIRECTORY_QUERY: u32 = 0x0000_0001; + const DIRECTORY_TRAVERSE: u32 = 0x0000_0002; + const DIRECTORY_ALL_ACCESS: u32 = 0x000f_000f; + + #[derive(Clone, Debug, Eq, PartialEq)] + struct ParsedDirectoryInformation { + name: String, + type_name: String, + } + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + fn read_u16(buffer: &[u8], offset: usize) -> u16 { + u16::from_le_bytes(buffer[offset..offset + 2].try_into().expect("u16 bytes")) + } + + fn read_usize(buffer: &[u8], offset: usize) -> usize { + usize::from_le_bytes( + buffer[offset..offset + size_of::()] + .try_into() + .expect("usize bytes"), + ) + } + + fn read_utf16_string( + buffer: &[u8], + buffer_base: usize, + address: usize, + length: usize, + ) -> String { + let offset = address + .checked_sub(buffer_base) + .expect("string buffer points into output buffer"); + assert!( + offset + .checked_add(length) + .is_some_and(|end| end <= buffer.len()), + "string buffer range stays inside output buffer" + ); + let units: alloc::vec::Vec = buffer[offset..offset + length] + .as_chunks::<2>() + .0 + .iter() + .map(|bytes| u16::from_le_bytes(*bytes)) + .collect(); + String::from_utf16_lossy(&units) + } + + fn read_directory_information(buffer: &[u8], offset: usize) -> ParsedDirectoryInformation { + let buffer_base = buffer.as_ptr() as usize; + let name_len = read_u16(buffer, offset) as usize; + let name_max = read_u16(buffer, offset + 2) as usize; + let name_buffer = read_usize(buffer, offset + 8); + let type_len = read_u16(buffer, offset + 16) as usize; + let type_max = read_u16(buffer, offset + 18) as usize; + let type_buffer = read_usize(buffer, offset + 24); + assert_eq!(name_max, name_len + size_of::()); + assert_eq!(type_max, type_len + size_of::()); + let name_offset = name_buffer + .checked_sub(buffer_base) + .expect("name buffer points into output buffer"); + let type_offset = type_buffer + .checked_sub(buffer_base) + .expect("type buffer points into output buffer"); + assert_eq!(read_u16(buffer, name_offset + name_len), 0); + assert_eq!(read_u16(buffer, type_offset + type_len), 0); + ParsedDirectoryInformation { + name: read_utf16_string(buffer, buffer_base, name_buffer, name_len), + type_name: read_utf16_string(buffer, buffer_base, type_buffer, type_len), + } + } + + fn assert_zero_directory_information(buffer: &[u8], offset: usize) { + assert_eq!(read_u16(buffer, offset), 0); + assert_eq!(read_u16(buffer, offset + 2), 0); + assert_eq!(read_usize(buffer, offset + 8), 0); + assert_eq!(read_u16(buffer, offset + 16), 0); + assert_eq!(read_u16(buffer, offset + 18), 0); + assert_eq!(read_usize(buffer, offset + 24), 0); + } + + fn create_named_directory( + task: &Task, + path: &str, + ) -> Handle { + let name_units = utf16_units(path); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut handle), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + handle + } + + fn open_named_directory(task: &Task, path: &str) -> Handle { + let name_units = utf16_units(path); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + handle + } + + fn object_attributes_with_root( + name: &UnicodeString, + root_directory: Handle, + attributes: u32, + ) -> ObjectAttributes { + ObjectAttributes { + root_directory, + ..object_attributes(name, attributes) + } + } + + fn expected_record_size(name: &str, type_name: &str) -> usize { + size_of::() + + name.encode_utf16().count() * size_of::() + + size_of::() + + type_name.encode_utf16().count() * size_of::() + + size_of::() + } + + fn expected_query_size(entries: &[(&str, &str)]) -> usize { + size_of::() + + entries + .iter() + .map(|(name, type_name)| expected_record_size(name, type_name)) + .sum::() + } + + #[test] + fn open_seeded_root_directory_succeeds() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn windows_shared_section_resolves_to_session_shared_section() { + run_with_test_platform_pointers(|| { + let object_manager = seed_object_manager::(); + let shared_section = load_time_windows_shared_section::(0x10000); + assert_eq!( + object_manager + .create_section(WINDOWS_SESSION_SHARED_SECTION_OBJECT, &shared_section,), + NtStatus::SUCCESS + ); + + let shortcut = object_manager + .resolve_symlink(WINDOWS_SHARED_SECTION_OBJECT, false) + .expect("Windows shared section shortcut is a symbolic link"); + assert_eq!( + shortcut.symlink_target().into_result(), + Ok(WINDOWS_SESSION_SHARED_SECTION_OBJECT.to_string()) + ); + let resolved = object_manager + .resolve_section(WINDOWS_SHARED_SECTION_OBJECT) + .expect("Windows shared section shortcut resolves to session section"); + assert!(Arc::ptr_eq(&resolved, &shared_section)); + }); + } + + #[test] + fn seeded_file_devices_resolve_through_object_manager() { + let object_manager = seed_object_manager::(); + + assert_eq!( + object_manager.resolve_file_device(r"\Device\HarddiskVolume1\Windows"), + Ok(( + FileDeviceObject::Filesystem { + root_path: "/".to_string(), + }, + "Windows".to_string(), + )) + ); + assert_eq!( + object_manager.resolve_file_device(r"\??\C:\Windows\System32"), + Ok(( + FileDeviceObject::Filesystem { + root_path: "/".to_string(), + }, + r"Windows\System32".to_string(), + )) + ); + assert_eq!( + object_manager.resolve_file_device(r"\SystemRoot\System32"), + Ok(( + FileDeviceObject::Filesystem { + root_path: "/".to_string(), + }, + r"Windows\System32".to_string(), + )) + ); + assert_eq!( + object_manager.resolve_file_device(r"\Device\ConDrv\Output"), + Ok((FileDeviceObject::ConsoleDriver, "Output".to_string())) + ); + } + + #[test] + fn open_directory_rejects_openlink_attribute() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\BaseNamedObjects"); + let name = unicode_string(&name_units); + let attrs = object_attributes( + &name, + (ObjectAttributesFlags::CASE_INSENSITIVE | ObjectAttributesFlags::OPENLINK).bits(), + ); + let mut handle = Handle::default(); + + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::default()); + }); + } + + #[test] + fn open_directory_distinguishes_missing_leaf_from_missing_parent() { + run_with_test_platform_pointers(|| { + let task = test_task(); + for (path, expected_status) in [ + ( + r"\BaseNamedObjects\DefinitelyMissingLiteBoxDir", + NtStatus::OBJECT_NAME_NOT_FOUND, + ), + ( + r"\KnownDlls\DefinitelyMissingLiteBoxDir", + NtStatus::OBJECT_NAME_NOT_FOUND, + ), + ( + r"\MissingParentLiteBox\Child", + NtStatus::OBJECT_PATH_NOT_FOUND, + ), + ( + r"\DefinitelyMissingLiteBoxDir", + NtStatus::OBJECT_NAME_NOT_FOUND, + ), + ] { + let name_units = utf16_units(path); + let name = unicode_string(&name_units); + let attrs = + object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + expected_status, + "unexpected status opening {path}", + ); + assert_eq!(handle, Handle::default()); + } + }); + } + + #[test] + fn open_section_rejects_empty_known_dlls_with_zeroed_output() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let known_dlls_units = utf16_units(r"\KnownDlls"); + let known_dlls_name = unicode_string(&known_dlls_units); + let known_dlls_attrs = object_attributes( + &known_dlls_name, + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + let mut known_dlls = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut known_dlls), + DIRECTORY_QUERY | DIRECTORY_TRAVERSE, + Some(const_ptr(&known_dlls_attrs)), + ), + NtStatus::SUCCESS + ); + let kernel32_units = utf16_units("KERNEL32.DLL"); + let kernel32 = unicode_string(&kernel32_units); + let attrs = object_attributes_with_root( + &kernel32, + known_dlls, + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + let mut handle = Handle::from_raw(0x5555_5555); + + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, Some(const_ptr(&attrs))), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!(handle, Handle::default()); + + let attrs = object_attributes_with_root( + &kernel32, + known_dlls, + (ObjectAttributesFlags::CASE_INSENSITIVE | ObjectAttributesFlags::OPENLINK).bits(), + ); + handle = Handle::from_raw(0x5555_5555); + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, Some(const_ptr(&attrs))), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!(handle, Handle::default()); + + let missing_parent_units = utf16_units(r"\MissingLiteBoxParent\KERNEL32.DLL"); + let missing_parent = unicode_string(&missing_parent_units); + let attrs = object_attributes( + &missing_parent, + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + handle = Handle::from_raw(0x5555_5555); + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, Some(const_ptr(&attrs))), + NtStatus::OBJECT_PATH_NOT_FOUND + ); + assert_eq!(handle, Handle::default()); + + let attrs = object_attributes( + &known_dlls_name, + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + handle = Handle::from_raw(0x5555_5555); + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, Some(const_ptr(&attrs))), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(handle, Handle::default()); + + let attrs = object_attributes_with_root( + &kernel32, + Handle::from_raw(0x1234), + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + handle = Handle::from_raw(0x5555_5555); + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, Some(const_ptr(&attrs))), + NtStatus::INVALID_HANDLE + ); + assert_eq!(handle, Handle::default()); + + handle = Handle::from_raw(0x5555_5555); + assert_eq!( + task.sys_nt_open_section(mut_ptr(&mut handle), 0x0d, None), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::default()); + + assert_eq!( + task.sys_nt_open_section(null_mut_ptr::(), 0x0d, Some(const_ptr(&attrs))), + NtStatus::ACCESS_VIOLATION + ); + + assert_eq!(task.sys_nt_close(known_dlls), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_directory_distinguishes_null_object_name_from_empty_name() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let root_units = utf16_units(r"\BaseNamedObjects"); + let root_name = unicode_string(&root_units); + let root_attrs = + object_attributes(&root_name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut root = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut root), + DIRECTORY_TRAVERSE | DIRECTORY_QUERY, + Some(const_ptr(&root_attrs)), + ), + NtStatus::SUCCESS + ); + + let null_name_with_root = ObjectAttributes { + length: size_of::().trunc(), + root_directory: root, + object_name: 0, + attributes: ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + security_descriptor: 0, + security_quality_of_service: 0, + }; + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut handle), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&null_name_with_root)), + Handle::default(), + 0, + ), + NtStatus::OBJECT_NAME_INVALID + ); + assert_eq!(handle, Handle::default()); + + let null_name_without_root = ObjectAttributes { + root_directory: Handle::default(), + ..null_name_with_root + }; + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut handle), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&null_name_without_root)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + assert_ne!(handle, Handle::default()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + + let empty_name_units: [u16; 0] = []; + let empty_name = unicode_string(&empty_name_units); + let empty_name_with_root = ObjectAttributes { + root_directory: root, + object_name: core::ptr::from_ref(&empty_name) as usize, + ..null_name_with_root + }; + handle = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut handle), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&empty_name_with_root)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + assert_ne!(handle, Handle::default()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(root), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_and_open_directory_relative_to_root_directory() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let root_units = utf16_units(r"\BaseNamedObjects"); + let root_name = unicode_string(&root_units); + let root_attrs = + object_attributes(&root_name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut root = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut root), + DIRECTORY_TRAVERSE | DIRECTORY_QUERY, + Some(const_ptr(&root_attrs)), + ), + NtStatus::SUCCESS + ); + + let child_units = utf16_units("LiteBoxDirectory"); + let child_name = unicode_string(&child_units); + let child_attrs = ObjectAttributes { + length: size_of::().trunc(), + root_directory: root, + object_name: core::ptr::from_ref(&child_name) as usize, + attributes: ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + security_descriptor: 0, + security_quality_of_service: 0, + }; + let mut created = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut created), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&child_attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut opened), + DIRECTORY_QUERY, + Some(const_ptr(&child_attrs)), + ), + NtStatus::SUCCESS + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(created), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(root), NtStatus::SUCCESS); + }); + } + + #[test] + fn directory_lookup_is_case_insensitive_and_case_preserving() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let mixed = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxCaseMixed"); + let lower_open = open_named_directory(&task, r"\basenamedobjects\liteboxcasemixed"); + let trailing_open = open_named_directory(&task, r"\BaseNamedObjects\LiteBoxCaseMixed\"); + let lower_created = + create_named_directory(&task, r"\BaseNamedObjects\liteboxcaselower"); + let upper_open = open_named_directory(&task, r"\BASENAMEDOBJECTS\LITEBOXCASELOWER"); + + let duplicate_units = utf16_units(r"\basenamedobjects\liteboxcasemixed\"); + let duplicate_name = unicode_string(&duplicate_units); + let duplicate_attrs = object_attributes( + &duplicate_name, + ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + ); + let mut duplicate = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut duplicate), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&duplicate_attrs)), + Handle::default(), + 0, + ), + NtStatus::OBJECT_NAME_COLLISION + ); + assert_eq!(duplicate, Handle::default()); + + let parent = open_named_directory(&task, r"\BaseNamedObjects"); + let mut buffer = [0u8; 512]; + let mut context = 0u32; + let mut return_length = 0u32; + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: parent, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 0, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + }), + NtStatus::SUCCESS + ); + + let first_record = read_directory_information(&buffer, 0); + assert_eq!( + first_record, + ParsedDirectoryInformation { + name: "liteboxcaselower".to_string(), + type_name: "Directory".to_string(), + } + ); + let second_offset = size_of::(); + let second_record = read_directory_information(&buffer, second_offset); + assert_eq!( + second_record, + ParsedDirectoryInformation { + name: "LiteBoxCaseMixed".to_string(), + type_name: "Directory".to_string(), + } + ); + assert_zero_directory_information( + &buffer, + second_offset + size_of::(), + ); + assert_eq!(context, 2); + assert_eq!( + return_length as usize, + expected_query_size(&[ + ("liteboxcaselower", "Directory"), + ("LiteBoxCaseMixed", "Directory") + ]) + ); + + assert_eq!(task.sys_nt_close(parent), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(upper_open), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(lower_created), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(trailing_open), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(lower_open), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(mixed), NtStatus::SUCCESS); + }); + } + + #[test] + fn relative_directory_name_requires_root_traverse_access() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let root_units = utf16_units(r"\BaseNamedObjects"); + let root_name = unicode_string(&root_units); + let root_attrs = + object_attributes(&root_name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut root = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut root), + DIRECTORY_QUERY, + Some(const_ptr(&root_attrs)), + ), + NtStatus::SUCCESS + ); + + let child_units = utf16_units("LiteBoxTraverseDenied"); + let child_name = unicode_string(&child_units); + let child_attrs = ObjectAttributes { + length: size_of::().trunc(), + root_directory: root, + object_name: core::ptr::from_ref(&child_name) as usize, + attributes: ObjectAttributesFlags::CASE_INSENSITIVE.bits(), + security_descriptor: 0, + security_quality_of_service: 0, + }; + let mut child = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut child), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&child_attrs)), + Handle::default(), + 0, + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(child, Handle::default()); + assert_eq!(task.sys_nt_close(root), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_nested_directory_after_parent_exists() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let parent_units = utf16_units(r"\BaseNamedObjects\LiteBoxTreeParent"); + let parent_name = unicode_string(&parent_units); + let parent_attrs = + object_attributes(&parent_name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut parent = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut parent), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&parent_attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + + let child_units = utf16_units(r"\BaseNamedObjects\LiteBoxTreeParent\LiteBoxTreeChild"); + let child_name = unicode_string(&child_units); + let child_attrs = + object_attributes(&child_name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut child = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut child), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&child_attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut opened), + DIRECTORY_QUERY, + Some(const_ptr(&child_attrs)), + ), + NtStatus::SUCCESS + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(child), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(parent), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_existing_directory_obeys_openif() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\BaseNamedObjects\LiteBoxOpenIfDirectory"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut first = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut first), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + + let mut collision = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut collision), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::OBJECT_NAME_COLLISION + ); + assert_eq!(collision, Handle::default()); + + let openif_attrs = ObjectAttributes { + attributes: (ObjectAttributesFlags::CASE_INSENSITIVE + | ObjectAttributesFlags::OPENIF) + .bits(), + ..attrs + }; + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut opened), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&openif_attrs)), + Handle::default(), + 0, + ), + NtStatus::OBJECT_NAME_EXISTS + ); + assert_ne!(opened, Handle::default()); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(first), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_empty_directory_reports_no_more_entries() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\BaseNamedObjects"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + + let mut buffer = [0xffu8; 32]; + let mut context = 99u32; + let mut return_length = u32::MAX; + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 0, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + },), + NtStatus::NO_MORE_ENTRIES + ); + assert_eq!(buffer[0], 0xff); + assert_eq!(context, 0); + assert_eq!(return_length, 0); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_directory_enumerates_children_in_stable_order() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let first = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxEnumB"); + let second = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxEnumA"); + let handle = open_named_directory(&task, r"\BaseNamedObjects"); + let mut buffer = [0u8; 512]; + let mut context = u32::MAX; + let mut return_length = 0u32; + + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 0, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + },), + NtStatus::SUCCESS + ); + + let first_record = read_directory_information(&buffer, 0); + assert_eq!( + first_record, + ParsedDirectoryInformation { + name: "LiteBoxEnumA".to_string(), + type_name: "Directory".to_string(), + } + ); + let second_offset = size_of::(); + let second_record = read_directory_information(&buffer, second_offset); + assert_eq!( + second_record, + ParsedDirectoryInformation { + name: "LiteBoxEnumB".to_string(), + type_name: "Directory".to_string(), + } + ); + assert_zero_directory_information( + &buffer, + second_offset + size_of::(), + ); + assert_eq!(context, 2); + assert_eq!( + return_length as usize, + expected_query_size(&[ + ("LiteBoxEnumA", "Directory"), + ("LiteBoxEnumB", "Directory") + ]) + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(second), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(first), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_directory_single_entry_uses_context_cookie() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let first = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxSingleA"); + let second = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxSingleB"); + let handle = open_named_directory(&task, r"\BaseNamedObjects"); + let mut buffer = [0u8; 256]; + let mut context = 123u32; + let mut return_length = 0u32; + + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 1, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + },), + NtStatus::MORE_ENTRIES + ); + assert_eq!(context, 1); + assert_eq!( + read_directory_information(&buffer, 0), + ParsedDirectoryInformation { + name: "LiteBoxSingleA".to_string(), + type_name: "Directory".to_string(), + } + ); + + buffer.fill(0); + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 1, + restart_scan: 0, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + },), + NtStatus::SUCCESS + ); + assert_eq!(context, 2); + assert_eq!( + read_directory_information(&buffer, 0), + ParsedDirectoryInformation { + name: "LiteBoxSingleB".to_string(), + type_name: "Directory".to_string(), + } + ); + assert_zero_directory_information(&buffer, size_of::()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(second), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(first), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_directory_too_small_reports_required_length_without_advancing_context() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let child = create_named_directory(&task, r"\BaseNamedObjects\LiteBoxSmall"); + let handle = open_named_directory(&task, r"\BaseNamedObjects"); + let mut buffer = [0xffu8; 8]; + let mut context = 99u32; + let mut return_length = 0u32; + + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer[0]), + buffer_length: buffer.len().trunc(), + return_single_entry: 0, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: Some(mut_ptr(&mut return_length)), + },), + NtStatus::MORE_ENTRIES + ); + assert_eq!(context, 99); + assert_eq!( + return_length as usize, + expected_query_size(&[("LiteBoxSmall", "Directory")]) + ); + assert_eq!(buffer, [0xffu8; 8]); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(child), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_requires_directory_query_access() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\BaseNamedObjects"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_TRAVERSE, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + let mut buffer = 0u8; + let mut context = 0u32; + assert_eq!( + task.sys_nt_query_directory_object(DirectoryQueryParameters { + directory_handle: handle, + buffer: mut_ptr(&mut buffer), + buffer_length: 1, + return_single_entry: 0, + restart_scan: 1, + context: mut_ptr(&mut context), + return_length: None, + },), + NtStatus::ACCESS_DENIED + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_probes_output_before_name_resolution() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\MissingParent\Child"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + + assert_eq!( + task.sys_nt_create_directory_object( + null_mut_ptr(), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_open_root_directory_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtOpenDirectoryObject( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + run_with_test_platform_pointers(|| { + let task = test_task(); + let name_units = utf16_units(r"\"); + let name = unicode_string(&name_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The object attributes and output handle point to live test + // stack values for the duration of the host ntdll call. + let host_status = unsafe { + NtOpenDirectoryObject(&raw mut host_handle, DIRECTORY_QUERY, &raw const attrs) + }; + if host_status == NtStatus::SUCCESS.as_raw() && !host_handle.is_null() { + // SAFETY: NtOpenDirectoryObject returned this non-null handle with + // STATUS_SUCCESS, so it is valid to close once here. + unsafe { + NtClose(host_handle); + } + } + + let mut litebox_handle = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut litebox_handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ) + .as_raw(), + host_status + ); + if litebox_handle != Handle::default() { + assert_eq!(task.sys_nt_close(litebox_handle), NtStatus::SUCCESS); + } + }); + } +} diff --git a/litebox_shim_windows/src/syscalls/process.rs b/litebox_shim_windows/src/syscalls/process.rs new file mode 100644 index 0000000000..2ac2fd3a4e --- /dev/null +++ b/litebox_shim_windows/src/syscalls/process.rs @@ -0,0 +1,1123 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use core::mem::{offset_of, size_of}; +use core::sync::atomic::Ordering; +use int_enum::IntEnum; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox::utils::TruncateExt; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::ThreadEnvironmentBlock; +use crate::syscalls::ProcessHandle; +use crate::{ConstPtr, MutPtr, ShimFS, ShimPlatform, Task}; + +const ACTIVE_PROCESS_EXIT_STATUS: i32 = 0x0000_0103; +const NORMAL_PROCESS_BASE_PRIORITY: i32 = 8; +pub(crate) const INITIAL_PROCESS_ID: usize = 1; +pub(crate) const INITIAL_THREAD_ID: usize = 1; +const GUEST_PARENT_PROCESS_ID: usize = 0; +const GUEST_PROCESS_AFFINITY_MASK: usize = 1; +const PROCESS_DEBUG_FLAGS_NO_DEBUGGER: u32 = 1; +const PROCESS_COOKIE: u32 = 0xdead_beef; +const TEB_TLS_SLOT_COUNT: usize = 64; + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum ProcessInformationClass { + BasicInformation = 0, + DebugPort = 7, + DefaultHardErrorMode = 12, + Wow64Information = 26, + DebugFlags = 31, + TlsInformation = 35, + Cookie = 36, + ConsoleHostProcess = 49, + ImageInformation = 53, + SchedulerSharedData = 112, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum ProcessTlsOperation { + ReplaceIndex = 0, + ReplaceVector = 1, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct ProcessTlsThreadDataFlags: u32 { + const OLD_DATA_WRITTEN = 0x2; + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessBasicInformation { + exit_status: i32, + _padding0: u32, + peb_base_address: usize, + affinity_mask: usize, + base_priority: i32, + _padding1: u32, + unique_process_id: usize, + inherited_from_unique_process_id: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessDefaultHardErrorMode { + default_hard_error_mode: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable)] +struct ProcessSchedulerSharedDataSlotInformation { + scheduler_shared_data_handle: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessTlsInformationHeader { + flags: u32, + operation_type: u32, + thread_data_count: u32, + tls_index: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessTlsInformationExtendedHeader { + header: ProcessTlsInformationHeader, + _reserved: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessTlsThreadDataSimple { + flags: u32, + _padding0: u32, + tls_data: usize, + _reserved: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessTlsThreadDataExtended { + flags: u32, + _padding0: u32, + new_tls_data: usize, + old_tls_data: usize, + _reserved: usize, +} + +enum ProcessTlsThreadData { + Simple(MutPtr), + Extended(MutPtr), +} + +impl ProcessTlsThreadData { + fn read_tls_data(&self) -> Option { + match self { + Self::Simple(ptr) => crate::read_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataSimple, tls_data), + ), + Self::Extended(ptr) => crate::read_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataExtended, new_tls_data), + ), + } + } + + fn write_tls_data(&self, value: usize) -> Option<()> { + match self { + Self::Simple(ptr) => crate::write_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataSimple, tls_data), + value, + ), + Self::Extended(ptr) => crate::write_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataExtended, old_tls_data), + value, + ), + } + } + + fn write_flags(&self, flags: ProcessTlsThreadDataFlags) -> Option<()> { + match self { + Self::Simple(ptr) => crate::write_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataSimple, flags), + flags.bits(), + ), + Self::Extended(ptr) => crate::write_field_at_offset::( + ptr.as_usize(), + offset_of!(ProcessTlsThreadDataExtended, flags), + flags.bits(), + ), + } + } +} + +#[derive(Clone, Copy, Debug)] +enum ProcessTlsLayout { + Simple, + Extended, +} + +impl ProcessTlsLayout { + fn detect(thread_data_count: u32, process_information_length: u32) -> Result { + let count = thread_data_count as usize; + let extended_len = size_of::() + .checked_add( + count + .checked_mul(size_of::()) + .ok_or(NtStatus::INFO_LENGTH_MISMATCH)?, + ) + .ok_or(NtStatus::INFO_LENGTH_MISMATCH)?; + let simple_len = size_of::() + .checked_add( + count + .checked_mul(size_of::()) + .ok_or(NtStatus::INFO_LENGTH_MISMATCH)?, + ) + .ok_or(NtStatus::INFO_LENGTH_MISMATCH)?; + let provided_len = process_information_length as usize; + + if provided_len >= extended_len { + Ok(Self::Extended) + } else if provided_len >= simple_len { + Ok(Self::Simple) + } else { + Err(NtStatus::INFO_LENGTH_MISMATCH) + } + } + + const fn header_size(self) -> usize { + match self { + Self::Simple => size_of::(), + Self::Extended => size_of::(), + } + } + + const fn entry_size(self) -> usize { + match self { + Self::Simple => size_of::(), + Self::Extended => size_of::(), + } + } + + const fn old_data_offset(self) -> usize { + match self { + Self::Simple => offset_of!(ProcessTlsThreadDataSimple, tls_data), + Self::Extended => offset_of!(ProcessTlsThreadDataExtended, old_tls_data), + } + } + + fn thread_data( + self, + base: MutPtr, + index: usize, + ) -> Option> { + match self { + Self::Simple => { + let arr = MutPtr::::from_usize( + base.as_usize().checked_add( + size_of::() + + index.checked_mul(size_of::())?, + )?, + ); + Some(ProcessTlsThreadData::Simple(arr)) + } + Self::Extended => { + let arr = MutPtr::::from_usize( + base.as_usize().checked_add( + size_of::() + + index.checked_mul(size_of::())?, + )?, + ); + Some(ProcessTlsThreadData::Extended(arr)) + } + } + } +} + +impl Task { + pub(crate) fn sys_nt_query_information_process( + &self, + process_handle: ProcessHandle, + process_information_class: u32, + process_information: MutPtr, + process_information_length: u32, + return_length: Option>, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + let Ok(process_information_class) = + ProcessInformationClass::try_from(process_information_class) + else { + litebox_util_log::debug!( + process_information_class = process_information_class; + "Unsupported NtQueryInformationProcess class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match process_information_class { + ProcessInformationClass::BasicInformation => Self::write_process_information( + process_information, + process_information_length, + return_length, + &self.process_basic_information(), + ), + ProcessInformationClass::DebugPort | ProcessInformationClass::Wow64Information => { + Self::write_process_information( + process_information, + process_information_length, + return_length, + &0usize, + ) + } + ProcessInformationClass::DebugFlags => Self::write_process_information( + process_information, + process_information_length, + return_length, + &PROCESS_DEBUG_FLAGS_NO_DEBUGGER, + ), + ProcessInformationClass::DefaultHardErrorMode => Self::write_process_information( + process_information, + process_information_length, + return_length, + &ProcessDefaultHardErrorMode { + default_hard_error_mode: self + .process + .default_hard_error_mode + .load(Ordering::Acquire), + }, + ), + ProcessInformationClass::Cookie => Self::write_process_information( + process_information, + process_information_length, + return_length, + &self.process.cookie, + ), + ProcessInformationClass::ConsoleHostProcess + | ProcessInformationClass::TlsInformation + | ProcessInformationClass::ImageInformation + | ProcessInformationClass::SchedulerSharedData => { + litebox_util_log::debug!( + process_information_class:? = process_information_class; + "Unsupported NtQueryInformationProcess class" + ); + NtStatus::INVALID_INFO_CLASS + } + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + process_information_class:? = process_information_class, + process_information_length = process_information_length; + "Handled NtQueryInformationProcess syscall" + ); + } + + status + } + + pub(crate) fn sys_nt_set_information_process( + &self, + process_handle: ProcessHandle, + process_information_class: u32, + process_information: MutPtr, + process_information_length: u32, + ) -> NtStatus { + let Ok(process_information_class) = + ProcessInformationClass::try_from(process_information_class) + else { + litebox_util_log::debug!( + process_information_class = process_information_class; + "Unsupported NtSetInformationProcess class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match process_information_class { + ProcessInformationClass::SchedulerSharedData => { + Self::set_process_scheduler_shared_data( + process_handle, + process_information, + process_information_length, + ) + } + ProcessInformationClass::TlsInformation => self.write_process_tls_information( + process_handle, + process_information, + process_information_length, + ), + // TODO: implement additional settable process information classes when a guest + // exercises them. + ProcessInformationClass::BasicInformation + | ProcessInformationClass::DebugPort + | ProcessInformationClass::DefaultHardErrorMode + | ProcessInformationClass::Wow64Information + | ProcessInformationClass::DebugFlags + | ProcessInformationClass::Cookie + | ProcessInformationClass::ConsoleHostProcess + | ProcessInformationClass::ImageInformation => { + litebox_util_log::debug!( + process_information_class:? = process_information_class; + "Unsupported NtSetInformationProcess class" + ); + NtStatus::INVALID_INFO_CLASS + } + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + process_information_class:? = process_information_class, + process_information_length = process_information_length; + "Handled NtSetInformationProcess syscall" + ); + } + + status + } + + fn write_process_information( + process_information: MutPtr, + process_information_length: u32, + return_length: Option>, + information: &T, + ) -> NtStatus { + let required_len = size_of::().trunc(); + if process_information_length < required_len { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if process_information + .write_slice_at_offset(0, information.as_bytes()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = return_length + && return_length.write_at_offset(0, required_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + fn set_process_scheduler_shared_data( + process_handle: ProcessHandle, + process_information: MutPtr, + process_information_length: u32, + ) -> NtStatus { + if process_information_length + < size_of::().trunc() + { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + let process_information = + ConstPtr::::from_usize( + process_information.as_usize(), + ); + if process_information.read_at_offset(0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + + // Host 25H2 returns SUCCESS after probing this struct even when the inner scheduler + // shared-data handle is null or bogus; LiteBox has no scheduler-shared-data object to bind. + NtStatus::SUCCESS + } + + fn write_process_tls_information( + &self, + process_handle: ProcessHandle, + process_information: MutPtr, + process_information_length: u32, + ) -> NtStatus { + if (process_information_length as usize) < size_of::() { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + let Some(header) = ConstPtr::::from_usize( + process_information.as_usize(), + ) + .read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let layout = + match ProcessTlsLayout::detect(header.thread_data_count, process_information_length) { + Ok(layout) => layout, + Err(status) => return status, + }; + + litebox_util_log::debug!( + operation_type = header.operation_type, + thread_data_count = header.thread_data_count, + tls_index = header.tls_index, + process_information_length, + layout_header_size = layout.header_size(), + layout_entry_size = layout.entry_size(), + layout_old_data_offset = layout.old_data_offset(); + "Handling ProcessTlsInformation" + ); + + if header.thread_data_count > 1 { + // TODO(multi-thread-tls): PROCESS_TLS_INFORMATION entries are positional per-thread + // data. The shim currently models only the active TEB, so handling multiple entries + // would corrupt prior TLS values by repeatedly writing one TEB's vector. + return NtStatus::NOT_SUPPORTED; + } + + match ProcessTlsOperation::try_from(header.operation_type) { + Ok(ProcessTlsOperation::ReplaceVector) => { + self.replace_tls_vector(process_information, header, layout) + } + Ok(ProcessTlsOperation::ReplaceIndex) => { + self.replace_tls_index(process_information, header, layout) + } + Err(_) => { + litebox_util_log::debug!( + operation_type = header.operation_type, + thread_data_count = header.thread_data_count, + tls_index = header.tls_index; + "Unsupported ProcessTlsInformation operation" + ); + NtStatus::INVALID_INFO_CLASS + } + } + } + + fn replace_tls_vector( + &self, + process_information: MutPtr, + header: ProcessTlsInformationHeader, + layout: ProcessTlsLayout, + ) -> NtStatus { + for index in 0..header.thread_data_count as usize { + let Some(thread_data) = layout.thread_data::(process_information, index) + else { + return NtStatus::INFO_LENGTH_MISMATCH; + }; + let Some(new_tls_data) = thread_data.read_tls_data() else { + return NtStatus::ACCESS_VIOLATION; + }; + // TODO(multi-thread-tls): read ith thread's tls + let teb = MutPtr::::from_usize(self.teb_address); + let Some(old_tls_data) = Self::read_teb_tls_pointer(teb) else { + return NtStatus::ACCESS_VIOLATION; + }; + + litebox_util_log::debug!( + thread_data_index = index, + new_tls_data:% = format_args!("{new_tls_data:#x}"), + old_tls_data:% = format_args!("{old_tls_data:#x}"); + "Replacing process TLS vector" + ); + + if let Err(status) = self.copy_initial_tls_slots(old_tls_data, new_tls_data) { + return status; + } + let old_tls_data_for_guest = self.guest_visible_old_tls_vector(old_tls_data); + + if thread_data.write_tls_data(old_tls_data_for_guest).is_none() + || Self::write_teb_tls_pointer(teb, new_tls_data).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if old_tls_data_for_guest == 0 + && thread_data + .write_flags(ProcessTlsThreadDataFlags::OLD_DATA_WRITTEN) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + } + + NtStatus::SUCCESS + } + + fn replace_tls_index( + &self, + process_information: MutPtr, + header: ProcessTlsInformationHeader, + layout: ProcessTlsLayout, + ) -> NtStatus { + let tls_index = header.tls_index as usize; + if tls_index >= TEB_TLS_SLOT_COUNT { + return NtStatus::INVALID_PARAMETER; + } + + for index in 0..header.thread_data_count as usize { + let Some(thread_data) = layout.thread_data::(process_information, index) + else { + return NtStatus::INFO_LENGTH_MISMATCH; + }; + let Some(new_tls_data) = thread_data.read_tls_data() else { + return NtStatus::ACCESS_VIOLATION; + }; + // TODO(multi-thread-tls): read ith thread's tls + let teb = ConstPtr::::from_usize(self.teb_address); + let Some(tls_array) = Self::read_teb_tls_pointer(teb) else { + return NtStatus::ACCESS_VIOLATION; + }; + if tls_array == 0 { + continue; + } + let tls_slots = MutPtr::::from_usize(tls_array); + let Some(old_tls_data) = tls_slots.read_at_offset(tls_index.cast_signed()) else { + return NtStatus::ACCESS_VIOLATION; + }; + + litebox_util_log::debug!( + thread_data_index = index, + tls_index, + tls_array:% = format_args!("{tls_array:#x}"), + new_tls_data:% = format_args!("{new_tls_data:#x}"), + old_tls_data:% = format_args!("{old_tls_data:#x}"); + "Replacing process TLS index" + ); + + if thread_data.write_tls_data(old_tls_data).is_none() + || tls_slots + .write_at_offset(tls_index.cast_signed(), new_tls_data) + .is_none() + || thread_data + .write_flags(ProcessTlsThreadDataFlags::OLD_DATA_WRITTEN) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + } + + NtStatus::SUCCESS + } + + fn guest_visible_old_tls_vector(&self, old_tls_data: usize) -> usize { + if old_tls_data > 0x10010 + && old_tls_data >= self.teb_address + && old_tls_data + < self + .teb_address + .saturating_add(size_of::()) + { + // The loader frees the returned old vector. The initial vector lives inside + // TEB.tls_slots, so report no heap-backed vector instead of exposing TEB memory. + 0 + } else { + old_tls_data + } + } + + fn copy_initial_tls_slots( + &self, + old_tls_data: usize, + new_tls_data: usize, + ) -> Result<(), NtStatus> { + if old_tls_data <= 0x10010 || new_tls_data <= 0x10010 { + return Ok(()); + } + let initial_tls_slots = self + .teb_address + .checked_add(offset_of!(ThreadEnvironmentBlock, tls_slots)) + .ok_or(NtStatus::INVALID_PARAMETER)?; + if old_tls_data != initial_tls_slots { + return Ok(()); + } + let old_tls_slots = ConstPtr::::from_usize(old_tls_data); + let new_tls_slots = MutPtr::::from_usize(new_tls_data); + for index in 0..TEB_TLS_SLOT_COUNT.cast_signed() { + let slot_value = old_tls_slots + .read_at_offset(index) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + new_tls_slots + .write_at_offset(index, slot_value) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + } + + Ok(()) + } + + fn read_teb_tls_pointer>( + teb: Ptr, + ) -> Option { + crate::read_field_at_offset::( + teb.as_usize(), + offset_of!(ThreadEnvironmentBlock, thread_local_storage_pointer), + ) + } + + fn write_teb_tls_pointer( + teb: MutPtr, + value: usize, + ) -> Option<()> { + crate::write_field_at_offset::( + teb.as_usize(), + offset_of!(ThreadEnvironmentBlock, thread_local_storage_pointer), + value, + ) + } + + fn process_basic_information(&self) -> ProcessBasicInformation { + ProcessBasicInformation { + exit_status: ACTIVE_PROCESS_EXIT_STATUS, + _padding0: 0, + peb_base_address: self.process.peb_address, + affinity_mask: GUEST_PROCESS_AFFINITY_MASK, + base_priority: NORMAL_PROCESS_BASE_PRIORITY, + _padding1: 0, + unique_process_id: INITIAL_PROCESS_ID, + inherited_from_unique_process_id: GUEST_PARENT_PROCESS_ID, + } + } +} + +pub(crate) const fn default_process_cookie() -> u32 { + // TODO: use CrngProvider to generate a random cookie + PROCESS_COOKIE +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{mut_byte_ptr, mut_ptr, null_mut_ptr}; + use litebox::platform::ThreadProvider; + + const RETURN_LENGTH_SENTINEL: u32 = 0xaaaa_aaaa; + + type TestPlatform = crate::tests::TestPlatform; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + #[test] + fn nt_query_information_process_validates_arguments() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut info = [0u8; size_of::()]; + let mut return_length = 0; + let basic_information_len: u32 = size_of::().trunc(); + + assert_eq!( + task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::BasicInformation as u32, + mut_byte_ptr(&mut info), + basic_information_len - 1, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!( + return_length, 0, + "ReactOS sets ReturnLength only after the exact-size check for this class; a host Windows probe shows the same result" + ); + + assert_eq!( + task.sys_nt_query_information_process( + ProcessHandle::from_raw(0x1234), + ProcessInformationClass::BasicInformation as u32, + mut_byte_ptr(&mut info), + basic_information_len, + None, + ), + NtStatus::INVALID_HANDLE + ); + + assert_eq!( + task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + 0xffff, + mut_byte_ptr(&mut info), + basic_information_len, + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::BasicInformation as u32, + null_mut_ptr::(), + basic_information_len, + None, + ), + NtStatus::ACCESS_VIOLATION + ); + + return_length = RETURN_LENGTH_SENTINEL; + assert_eq!( + task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::BasicInformation as u32, + null_mut_ptr::(), + basic_information_len, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::ACCESS_VIOLATION + ); + assert_eq!( + return_length, RETURN_LENGTH_SENTINEL, + "a host Windows probe leaves ReturnLength unchanged when ProcessInformation faults" + ); + }); + } + + #[test] + fn nt_set_information_process_scheduler_shared_data_validates_arguments() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut information = ProcessSchedulerSharedDataSlotInformation { + scheduler_shared_data_handle: 0, + }; + let information_len: u32 = + size_of::().trunc(); + let bad_handle = ProcessHandle::from_raw(0x1234); + + assert_eq!( + task.sys_nt_set_information_process( + bad_handle, + ProcessInformationClass::SchedulerSharedData as u32, + null_mut_ptr::(), + information_len - 1, + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + + assert_eq!( + task.sys_nt_set_information_process( + bad_handle, + 0xffff, + mut_byte_ptr(&mut information), + information_len - 1, + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + task.sys_nt_set_information_process( + bad_handle, + ProcessInformationClass::SchedulerSharedData as u32, + null_mut_ptr::(), + information_len, + ), + NtStatus::INVALID_HANDLE + ); + + assert_eq!( + task.sys_nt_set_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::SchedulerSharedData as u32, + null_mut_ptr::(), + information_len, + ), + NtStatus::ACCESS_VIOLATION + ); + + assert_eq!( + task.sys_nt_set_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::SchedulerSharedData as u32, + mut_byte_ptr(&mut information), + information_len, + ), + NtStatus::SUCCESS + ); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + mod host_fidelity { + use core::ffi::c_void; + + use super::*; + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtQueryInformationProcess( + process_handle: *mut c_void, + process_information_class: u32, + process_information: *mut c_void, + process_information_length: u32, + return_length: *mut u32, + ) -> i32; + fn NtSetInformationProcess( + process_handle: *mut c_void, + process_information_class: u32, + process_information: *const c_void, + process_information_length: u32, + ) -> i32; + } + + fn empty_basic_information() -> ProcessBasicInformation { + ProcessBasicInformation { + exit_status: 0, + _padding0: 0, + peb_base_address: 0, + affinity_mask: 0, + base_priority: 0, + _padding1: 0, + unique_process_id: 0, + inherited_from_unique_process_id: usize::MAX, + } + } + + fn host_nt_query_information_process( + process_information_class: ProcessInformationClass, + process_information: *mut c_void, + process_information_length: u32, + return_length: *mut u32, + ) -> NtStatus { + // SAFETY: The host ntdll call treats these as user-mode output pointers, probes them, + // and does not retain them. Tests pass either valid locals or null to observe NTSTATUS + // and output side effects. + let status = unsafe { + NtQueryInformationProcess( + usize::MAX as *mut c_void, + process_information_class as u32, + process_information, + process_information_length, + return_length, + ) + }; + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + fn host_nt_set_information_process( + process_handle: *mut c_void, + process_information_class: u32, + process_information: *const c_void, + process_information_length: u32, + ) -> NtStatus { + // SAFETY: The host ntdll call treats these as user-mode input pointers, probes them, + // and does not retain them. Tests pass either valid locals or null to observe NTSTATUS. + let status = unsafe { + NtSetInformationProcess( + process_handle, + process_information_class, + process_information, + process_information_length, + ) + }; + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + #[test] + fn nt_query_information_process_basic_length_mismatch_matches_host() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut host_info = empty_basic_information(); + let mut shim_info = empty_basic_information(); + let mut host_return_length = RETURN_LENGTH_SENTINEL; + let mut shim_return_length = RETURN_LENGTH_SENTINEL; + let basic_information_len: u32 = size_of::().trunc(); + let short_length = basic_information_len - 1; + + let host = host_nt_query_information_process( + ProcessInformationClass::BasicInformation, + core::ptr::addr_of_mut!(host_info).cast::(), + short_length, + core::ptr::addr_of_mut!(host_return_length), + ); + let shim = task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::BasicInformation as u32, + mut_byte_ptr(&mut shim_info), + short_length, + Some(mut_ptr(&mut shim_return_length)), + ); + + assert_eq!(shim, host); + assert_eq!(shim_return_length, host_return_length); + assert_eq!(shim_info.peb_base_address, 0); + }); + } + + #[test] + fn nt_query_information_process_invalid_output_leaves_return_length_unchanged() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut host_return_length = RETURN_LENGTH_SENTINEL; + let mut shim_return_length = RETURN_LENGTH_SENTINEL; + let basic_information_len: u32 = size_of::().trunc(); + + let host = host_nt_query_information_process( + ProcessInformationClass::BasicInformation, + core::ptr::null_mut(), + basic_information_len, + core::ptr::addr_of_mut!(host_return_length), + ); + let shim = task.sys_nt_query_information_process( + ProcessHandle::CURRENT, + ProcessInformationClass::BasicInformation as u32, + null_mut_ptr::(), + basic_information_len, + Some(mut_ptr(&mut shim_return_length)), + ); + + assert_eq!(shim, host); + assert_eq!(shim_return_length, host_return_length); + }); + } + + #[test] + fn nt_set_information_process_scheduler_shared_data_matches_host_statuses() { + run_with_test_platform_pointers(|| { + let mut null_information = ProcessSchedulerSharedDataSlotInformation { + scheduler_shared_data_handle: 0, + }; + let mut bogus_information = ProcessSchedulerSharedDataSlotInformation { + scheduler_shared_data_handle: 0x1234, + }; + let information_len: u32 = + size_of::().trunc(); + let current_process = usize::MAX as *mut c_void; + let bad_process = 0x1234usize as *mut c_void; + let scheduler_class = ProcessInformationClass::SchedulerSharedData as u32; + let bad_class = 0xffff; + + let supported_status = host_nt_set_information_process( + current_process, + scheduler_class, + core::ptr::from_ref(&null_information).cast::(), + information_len, + ); + + if supported_status != NtStatus::INVALID_INFO_CLASS { + assert_eq!(supported_status, NtStatus::SUCCESS); + + for ( + process_handle, + shim_process_handle, + process_information_class, + host_process_information, + shim_process_information, + process_information_length, + ) in [ + ( + current_process, + ProcessHandle::CURRENT, + scheduler_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len, + ), + ( + current_process, + ProcessHandle::CURRENT, + scheduler_class, + core::ptr::from_ref(&bogus_information).cast::(), + mut_byte_ptr(&mut bogus_information), + information_len, + ), + ( + current_process, + ProcessHandle::CURRENT, + scheduler_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len - 1, + ), + ( + current_process, + ProcessHandle::CURRENT, + scheduler_class, + core::ptr::null(), + null_mut_ptr::(), + information_len, + ), + ( + current_process, + ProcessHandle::CURRENT, + bad_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len, + ), + ( + bad_process, + ProcessHandle::from_raw(0x1234), + scheduler_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len, + ), + ( + bad_process, + ProcessHandle::from_raw(0x1234), + scheduler_class, + core::ptr::null(), + null_mut_ptr::(), + information_len - 1, + ), + ( + bad_process, + ProcessHandle::from_raw(0x1234), + bad_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len - 1, + ), + ( + bad_process, + ProcessHandle::from_raw(0x1234), + scheduler_class, + core::ptr::null(), + null_mut_ptr::(), + information_len, + ), + ( + current_process, + ProcessHandle::CURRENT, + scheduler_class, + core::ptr::null(), + null_mut_ptr::(), + information_len - 1, + ), + ( + current_process, + ProcessHandle::CURRENT, + bad_class, + core::ptr::from_ref(&null_information).cast::(), + mut_byte_ptr(&mut null_information), + information_len - 1, + ), + ] { + let host = host_nt_set_information_process( + process_handle, + process_information_class, + host_process_information, + process_information_length, + ); + let task = crate::tests::test_task(); + let shim = task.sys_nt_set_information_process( + shim_process_handle, + process_information_class, + shim_process_information, + process_information_length, + ); + + assert_eq!(shim, host); + } + } + }); + } + } +} diff --git a/litebox_shim_windows/src/syscalls/registry.rs b/litebox_shim_windows/src/syscalls/registry.rs new file mode 100644 index 0000000000..a3500c4614 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/registry.rs @@ -0,0 +1,1443 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows registry syscalls backed by a private file-system-shaped store (i.e., +//! a layered file system with in-mem and tar filesystems). +//! +//! Registry keys are represented as directories and values as files under each +//! key's `.values` directory: +//! +//! ```text +//! /registry/machine/system/currentcontrolset/control/nls/codepage/ +//! .values/ +//! acp +//! oemcp +//! maccp +//! ... +//! EUDCCodeRange/ +//! .values/ +//! 932 +//! ... +//! ... +//! ``` +//! +//! This is only an implementation detail: syscall handlers must expose registry +//! object semantics rather than file semantics. + +use core::marker::PhantomData; +use core::mem::{offset_of, size_of}; + +use alloc::string::String; +use alloc::vec; +use alloc::vec::Vec; + +use int_enum::IntEnum; +use litebox::LiteBox; +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry, TypedFd}; +use litebox::fs::errors::{ + FileStatusError, MkdirError, OpenError, PathError, ReadError, WriteError, +}; +use litebox::fs::{FileSystem as _, FileType, Mode, OFlags}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::syscalls::Handle; +use crate::{ConstPtr, MutPtr, ShimFS, Task, raw_handle_entry}; + +use crate::nt_types::{AccessMask, ObjectAttributes, UnicodeString, read_object_attributes}; + +type RegistryFileSystem = litebox::fs::layered::FileSystem< + Platform, + litebox::fs::in_mem::FileSystem, + litebox::fs::resolver::Resolver, +>; + +pub(crate) struct RegistryKeySubsystem(PhantomData); + +impl FdEnabledSubsystem for RegistryKeySubsystem { + type Entry = RegistryKeyObject; +} + +impl FdEnabledSubsystemEntry for RegistryKeyObject {} + +impl crate::WindowsHandleSubsystem + for RegistryKeySubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + RegistryKeyAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct RegistryKeyObject { + path: String, + fd: TypedFd>, +} + +pub(crate) struct RegistryStore { + fs: RegistryFileSystem, +} + +const VALUES_DIR_NAME: &str = ".values"; +const DEFAULT_CODE_PAGE_KEY: &str = + "\\Registry\\Machine\\System\\CurrentControlSet\\Control\\Nls\\CodePage"; +const DEFAULT_SESSION_MANAGER_KEY: &str = + "\\Registry\\Machine\\System\\CurrentControlSet\\Control\\Session Manager"; +const DEFAULT_SEGMENT_HEAP_KEY: &str = + "\\Registry\\Machine\\System\\CurrentControlSet\\Control\\Session Manager\\Segment Heap"; +const DEFAULT_IMAGE_FILE_EXECUTION_OPTIONS_KEY: &str = "\\Registry\\Machine\\Software\\Microsoft\\Windows NT\\CurrentVersion\\Image File Execution Options"; +const DEFAULT_ACP_VALUE: &[u8] = &[b'1', 0, b'2', 0, b'5', 0, b'2', 0, 0, 0]; +const DEFAULT_OEMCP_VALUE: &[u8] = &[b'4', 0, b'3', 0, b'7', 0, 0, 0]; +const DEFAULT_MACCP_VALUE: &[u8] = &[b'1', 0, b'0', 0, b'0', 0, b'0', 0, b'0', 0, 0, 0]; +const REGISTRY_VALUE_TYPE_SIZE: usize = size_of::(); + +bitflags::bitflags! { + /// Registry key `ACCESS_MASK` rights accepted by `NtOpenKey`/`NtCreateKey`. + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct RegistryKeyAccess: u32 { + const QUERY_VALUE = 0x0001; + const SET_VALUE = 0x0002; + const CREATE_SUB_KEY = 0x0004; + const ENUMERATE_SUB_KEYS = 0x0008; + const NOTIFY = 0x0010; + const CREATE_LINK = 0x0020; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() + | Self::QUERY_VALUE.bits() + | Self::ENUMERATE_SUB_KEYS.bits() + | Self::NOTIFY.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | Self::SET_VALUE.bits() + | Self::CREATE_SUB_KEY.bits(); + const EXECUTE = Self::READ.bits(); + const ALL_ACCESS = (AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::QUERY_VALUE.bits() + | Self::SET_VALUE.bits() + | Self::CREATE_SUB_KEY.bits() + | Self::ENUMERATE_SUB_KEYS.bits() + | Self::NOTIFY.bits() + | Self::CREATE_LINK.bits()) + & !AccessMask::SYNCHRONIZE.bits(); + + const FS_READ_ACCESS = Self::QUERY_VALUE.bits() + | Self::ENUMERATE_SUB_KEYS.bits() + | Self::NOTIFY.bits() + | AccessMask::GENERIC_READ.bits() + | AccessMask::GENERIC_EXECUTE.bits() + | AccessMask::GENERIC_ALL.bits(); + const FS_WRITE_ACCESS = Self::SET_VALUE.bits() + | Self::CREATE_SUB_KEY.bits() + | Self::CREATE_LINK.bits() + | AccessMask::DELETE.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits() + | AccessMask::GENERIC_WRITE.bits() + | AccessMask::GENERIC_ALL.bits(); + + const _ = !0; + } +} + +impl RegistryKeyAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +impl From for OFlags { + fn from(desired_access: RegistryKeyAccess) -> Self { + let wants_read = desired_access.intersects(RegistryKeyAccess::FS_READ_ACCESS); + let wants_write = desired_access.intersects(RegistryKeyAccess::FS_WRITE_ACCESS); + + let access = match (wants_read, wants_write) { + (true, true) => OFlags::RDWR, + (false, true) => OFlags::WRONLY, + _ => OFlags::RDONLY, + }; + access | OFlags::DIRECTORY + } +} + +/// System-defined `REG_*` value types stored in `KEY_VALUE_*_INFORMATION::Type`. +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum RegistryValueType { + /// `REG_NONE`: data with no particular type. + None = 0, + /// `REG_SZ`: a null-terminated Unicode string. + Sz = 1, + /// `REG_EXPAND_SZ`: a null-terminated Unicode string with unexpanded environment references. + ExpandSz = 2, + /// `REG_BINARY`: binary data in any form. + Binary = 3, + /// `REG_DWORD` / `REG_DWORD_LITTLE_ENDIAN`: a little-endian 4-byte value. + Dword = 4, + /// `REG_DWORD_BIG_ENDIAN`: a big-endian 4-byte value. + DwordBigEndian = 5, + /// `REG_LINK`: a Unicode string naming a symbolic link. + Link = 6, + /// `REG_MULTI_SZ`: null-terminated strings terminated by another zero. + MultiSz = 7, + /// `REG_RESOURCE_LIST`: a device driver's hardware resource list. + ResourceList = 8, + /// `REG_FULL_RESOURCE_DESCRIPTOR`: hardware resources used by a physical device. + FullResourceDescriptor = 9, + /// `REG_RESOURCE_REQUIREMENTS_LIST`: possible hardware resources for a device. + ResourceRequirementsList = 10, + /// `REG_QWORD` / `REG_QWORD_LITTLE_ENDIAN`: a little-endian 8-byte value. + Qword = 11, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum KeyValueInformationClass { + Basic = 0, + Full = 1, + Partial = 2, +} + +/// The `KEY_VALUE_BASIC_INFORMATION` structure defines a subset of the full +/// information available for a value entry of a registry key. +/// +/// The variable-length `Name` field follows this fixed-size header. +/// See . +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct KeyValueBasicInformation { + title_index: u32, + value_type: u32, + name_length: u32, + // Followed by a variable-length name. + name: [u8; 0], +} + +/// The `KEY_VALUE_FULL_INFORMATION` structure defines information available +/// for a value entry of a registry key. +/// +/// The variable-length `Name` field follows this fixed-size header. The value +/// data starts at `data_offset` after any alignment padding. +/// See . +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct KeyValueFullInformation { + title_index: u32, + value_type: u32, + data_offset: u32, + data_length: u32, + name_length: u32, + // Followed by a variable-length name and aligned value data. + name: [u8; 0], + // Followed by aligned value data. + // ... + // Data[u8; data_length]; +} + +/// The `KEY_VALUE_PARTIAL_INFORMATION` structure defines a subset of the value +/// information available for a value entry of a registry key. +/// +/// The variable-length `Data` field follows this fixed-size header. +/// See . +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct KeyValuePartialInformation { + title_index: u32, + value_type: u32, + data_length: u32, + // Followed by variable-length value data. + data: [u8; 0], +} + +struct RegistryValue { + value_type: RegistryValueType, + data: Vec, +} + +impl RegistryStore { + pub(crate) fn new(litebox: &LiteBox) -> Self { + let mut in_mem = litebox::fs::in_mem::FileSystem::new(litebox); + in_mem.with_root_privileges(|fs| { + for key in [ + DEFAULT_SESSION_MANAGER_KEY, + DEFAULT_SEGMENT_HEAP_KEY, + DEFAULT_IMAGE_FILE_EXECUTION_OPTIONS_KEY, + ] { + if let Err(status) = create_key_in_fs(fs, key) { + litebox_util_log::error!(key:% = key, status:? = status; "failed to initialize registry key"); + break; + } + } + for (name, value) in [ + ("ACP", DEFAULT_ACP_VALUE), + ("OEMCP", DEFAULT_OEMCP_VALUE), + ("MACCP", DEFAULT_MACCP_VALUE), + ] { + if let Err(status) = + write_value_in_fs(fs, DEFAULT_CODE_PAGE_KEY, name, RegistryValueType::Sz, value) + { + litebox_util_log::error!(name:% = name, status:? = status; "failed to initialize registry value"); + break; + } + } + }); + + let tar_ro = litebox::fs::resolver::Resolver::new( + litebox, + litebox::fs::composer::Composer::builder() + .mount("/", |allocator| { + litebox::fs::tar_ro::TarRo::new( + // TODO: Replace with tar file provided by the user + litebox::fs::tar_ro::EMPTY_TAR_FILE.into(), + allocator, + ) + }) + .build() + .unwrap(), + ); + let fs = litebox::fs::layered::FileSystem::new( + litebox, + in_mem, + tar_ro, + litebox::fs::layered::LayeringSemantics::LowerLayerReadOnly, + ); + + Self { fs } + } + + fn open_key( + &self, + path: &str, + desired_access: RegistryKeyAccess, + ) -> Result>, NtStatus> { + self.fs + .open(path, desired_access.into(), Mode::empty()) + .map_err(map_open_error) + } + + fn read_value_at_path( + &self, + key_path: &str, + value_name: &str, + ) -> Result { + let value_path = value_path(key_path, value_name)?; + let status = self + .fs + .file_status(&*value_path) + .map_err(map_file_status_error)?; + if status.file_type != FileType::RegularFile { + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + } + if status.size < REGISTRY_VALUE_TYPE_SIZE { + return Err(NtStatus::UNSUCCESSFUL); + } + + let fd = self + .fs + .open(&*value_path, OFlags::RDONLY, Mode::empty()) + .map_err(map_open_error)?; + let mut data = vec![0; status.size]; + let read = self + .fs + .read(&fd, &mut data, Some(0)) + .map_err(map_read_error)?; + let _ = self.fs.close(&fd); + if read != data.len() { + return Err(NtStatus::UNSUCCESSFUL); + } + + let value_type = RegistryValueType::try_from(u32::from_le_bytes( + data[..REGISTRY_VALUE_TYPE_SIZE] + .try_into() + .map_err(|_| NtStatus::UNSUCCESSFUL)?, + )) + .map_err(|_| NtStatus::UNSUCCESSFUL)?; + data.drain(..REGISTRY_VALUE_TYPE_SIZE); + + Ok(RegistryValue { value_type, data }) + } +} + +impl Task { + fn registry_key_entry( + &self, + handle: Handle, + ) -> Result>, NtStatus> { + raw_handle_entry::>( + &self.global.litebox, + &self.process.handles, + handle, + ) + .ok_or(NtStatus::INVALID_HANDLE) + } + + fn insert_registry_key_handle( + &self, + key: RegistryKeyObject, + granted_access: RegistryKeyAccess, + ) -> Result { + self.insert_typed_handle::>( + key, + granted_access.bits(), + |key| { + self.close_registry_key(key); + }, + ) + } + + pub(crate) fn close_registry_key_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, |key| { + self.close_registry_key(key); + }); + } + + pub(crate) fn close_registry_key(&self, key: RegistryKeyObject) { + let _ = self.global.registry.fs.close(&key.fd); + } + + pub(crate) fn sys_nt_open_key( + &self, + key_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + let Some(object_attributes) = object_attributes else { + return NtStatus::INVALID_PARAMETER; + }; + let object_attributes = match read_object_attributes::(object_attributes) { + Ok(object_attributes) => object_attributes, + Err(status) => return status, + }; + match self.do_nt_open_key(desired_access, object_attributes) { + Ok(handle) => { + if key_handle.write_at_offset(0, handle).is_none() { + self.close_registry_key_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + Err(status) => status, + } + } + + fn do_nt_open_key( + &self, + desired_access: u32, + object_attributes: ObjectAttributes, + ) -> Result { + if object_attributes.object_name == 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + + let object_name_ptr = + ConstPtr::::from_usize(object_attributes.object_name); + let object_name = object_name_ptr + .read_at_offset(0) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + let key_name = object_name.read_string::()?; + let path = if object_attributes.root_directory.is_null() || key_name.starts_with('\\') { + absolute_nt_key_name_to_fs_path(&key_name)? + } else { + let root_key = self.registry_key_entry(object_attributes.root_directory)?; + root_key + .with_entry(|root_key| relative_nt_key_name_to_fs_path(&root_key.path, &key_name))? + }; + + let desired_access = RegistryKeyAccess::from_desired_access(desired_access); + let fd = self + .global + .registry + .open_key(&path, desired_access) + .inspect_err(|status| { + if *status != NtStatus::OBJECT_NAME_NOT_FOUND { + litebox_util_log::debug!( + desired_access:? = desired_access, + root_directory:% = format_args!("{:#x}", object_attributes.root_directory.as_raw()), + name:% = key_name, + path:% = path, + status:? = status; + "NtOpenKey failed" + ); + } + })?; + self.insert_registry_key_handle(RegistryKeyObject { path, fd }, desired_access) + } + + pub(crate) fn sys_nt_query_value_key( + &self, + key_handle: Handle, + value_name: ConstPtr, + key_value_information_class: u32, + key_value_information: MutPtr, + length: u32, + result_length: MutPtr, + ) -> NtStatus { + let Some(value_name) = value_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Ok(key_value_information_class) = + KeyValueInformationClass::try_from(key_value_information_class) + else { + litebox_util_log::debug!( + key_value_information_class = key_value_information_class; + "Unsupported NtQueryValueKey class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + match self.do_nt_query_value_key( + key_handle, + value_name, + key_value_information_class, + key_value_information, + length, + result_length, + ) { + Ok(()) => NtStatus::SUCCESS, + Err(status) => status, + } + } + + fn do_nt_query_value_key( + &self, + key_handle: Handle, + value_name: UnicodeString, + key_value_information_class: KeyValueInformationClass, + key_value_information: MutPtr, + length: u32, + result_length: MutPtr, + ) -> Result<(), NtStatus> { + let key = self.typed_handle_entry_with_access::>( + key_handle, + RegistryKeyAccess::QUERY_VALUE.bits(), + )?; + let value_name = value_name.read_string::()?; + let value = key.with_entry(|key| { + // TODO: Open the value relative to `key.fd` once the FS has an openat-style API. + self.global + .registry + .read_value_at_path(&key.path, &value_name) + })?; + let name = utf16le(&value_name); + match key_value_information_class { + KeyValueInformationClass::Basic => { + let required_length = size_of::() + .checked_add(name.len()) + .ok_or(NtStatus::UNSUCCESSFUL)?; + let information = KeyValueBasicInformation { + title_index: 0, + value_type: value.value_type.into(), + name_length: name.len().trunc(), + name: [0u8; 0], + }; + write_query_result_length::(result_length, length, required_length)?; + write_query_information::( + key_value_information, + information.as_bytes(), + &[(offset_of!(KeyValueBasicInformation, name), name.as_slice())], + )?; + } + KeyValueInformationClass::Full => { + let name_end = offset_of!(KeyValueFullInformation, name) + .checked_add(name.len()) + .ok_or(NtStatus::UNSUCCESSFUL)?; + let data_offset = name_end + .checked_next_multiple_of(4) + .ok_or(NtStatus::UNSUCCESSFUL)?; + let required_length = data_offset + .checked_add(value.data.len()) + .ok_or(NtStatus::UNSUCCESSFUL)?; + write_query_result_length::(result_length, length, required_length)?; + let information = KeyValueFullInformation { + title_index: 0, + value_type: value.value_type.into(), + data_offset: data_offset.trunc(), + data_length: value.data.len().trunc(), + name_length: name.len().trunc(), + name: [0u8; 0], + }; + + write_query_information::( + key_value_information, + information.as_bytes(), + &[ + (offset_of!(KeyValueFullInformation, name), name.as_slice()), + (data_offset, value.data.as_slice()), + ], + )?; + } + KeyValueInformationClass::Partial => { + let required_length = size_of::() + .checked_add(value.data.len()) + .ok_or(NtStatus::UNSUCCESSFUL)?; + write_query_result_length::(result_length, length, required_length)?; + let information = KeyValuePartialInformation { + title_index: 0, + value_type: value.value_type.into(), + data_length: value.data.len().trunc(), + data: [0u8; 0], + }; + + write_query_information::( + key_value_information, + information.as_bytes(), + &[( + offset_of!(KeyValuePartialInformation, data), + value.data.as_slice(), + )], + )?; + } + } + + litebox_util_log::debug!( + handle:% = format_args!("{:#x}", key_handle.as_raw()), + value_name:% = value_name, + key_value_information_class:? = key_value_information_class, + length = length; + "Handled NtQueryValueKey syscall" + ); + + Ok(()) + } +} + +fn write_query_result_length( + result_length: MutPtr, + buffer_length: u32, + required_length: usize, +) -> Result<(), NtStatus> { + result_length + .write_at_offset(0, required_length.trunc()) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + if (buffer_length as usize) < required_length { + return Err(NtStatus::BUFFER_OVERFLOW); + } + Ok(()) +} + +fn write_query_information( + key_value_information: MutPtr, + header: &[u8], + trailing_slices: &[(usize, &[u8])], +) -> Result<(), NtStatus> { + key_value_information + .write_slice_at_offset(0, header) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + for &(offset, bytes) in trailing_slices { + key_value_information + .write_slice_at_offset(offset.cast_signed(), bytes) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + } + Ok(()) +} + +fn utf16le(value: &str) -> Vec { + let mut bytes = Vec::new(); + for code_unit in value.encode_utf16() { + bytes.extend_from_slice(&code_unit.to_le_bytes()); + } + bytes +} + +fn absolute_nt_key_name_to_fs_path(name: &str) -> Result { + if !name.starts_with('\\') { + return Err(NtStatus::INVALID_PARAMETER); + } + let mut path = String::from("/"); + append_registry_components(&mut path, name.trim_start_matches('\\'))?; + Ok(path) +} + +fn relative_nt_key_name_to_fs_path(root: &str, name: &str) -> Result { + if name.starts_with('\\') { + return absolute_nt_key_name_to_fs_path(name); + } + let mut path = String::from(root); + append_registry_components(&mut path, name)?; + Ok(path) +} + +fn append_registry_components(path: &mut String, name: &str) -> Result<(), NtStatus> { + if name.is_empty() { + return Err(NtStatus::INVALID_PARAMETER); + } + for component in name.split('\\') { + if !is_valid_key_component(component) { + return Err(NtStatus::INVALID_PARAMETER); + } + if !path.ends_with('/') { + path.push('/'); + } + path.push_str(&component.to_ascii_lowercase()); + } + Ok(()) +} + +fn is_valid_key_component(component: &str) -> bool { + !component.is_empty() + && component != "." + && component != ".." + && !component.eq_ignore_ascii_case(VALUES_DIR_NAME) + && !component.contains('/') +} + +fn write_value_in_fs( + fs: &FS, + key_nt_path: &str, + value_name: &str, + value_type: RegistryValueType, + value: &[u8], +) -> Result<(), NtStatus> { + let key_path = create_key_in_fs(fs, key_nt_path)?; + let value_path = value_path(&key_path, value_name)?; + let fd = fs + .open( + &*value_path, + OFlags::CREAT | OFlags::WRONLY | OFlags::TRUNC, + Mode::RUSR | Mode::WUSR | Mode::ROTH | Mode::WOTH, + ) + .map_err(map_open_error)?; + let written = fs + .write(&fd, &u32::from(value_type).to_le_bytes(), Some(0)) + .map_err(map_write_error)?; + if written != REGISTRY_VALUE_TYPE_SIZE { + return Err(NtStatus::DISK_FULL); + } + let written = fs + .write(&fd, value, Some(REGISTRY_VALUE_TYPE_SIZE)) + .map_err(map_write_error)?; + if written != value.len() { + return Err(NtStatus::DISK_FULL); + } + let _ = fs.close(&fd); + Ok(()) +} + +fn create_key_in_fs( + fs: &FS, + nt_path: &str, +) -> Result { + let path = absolute_nt_key_name_to_fs_path(nt_path)?; + create_key_path_in_fs(fs, &path)?; + Ok(path) +} + +fn create_key_path_in_fs(fs: &FS, path: &str) -> Result<(), NtStatus> { + let mut current = String::new(); + for component in path.trim_start_matches('/').split('/') { + if component.is_empty() { + continue; + } + current.push('/'); + current.push_str(component); + ensure_directory_in_fs(fs, ¤t)?; + + let mut values_dir = current.clone(); + values_dir.push('/'); + values_dir.push_str(VALUES_DIR_NAME); + ensure_directory_in_fs(fs, &values_dir)?; + } + Ok(()) +} + +fn ensure_directory_in_fs( + fs: &FS, + path: &str, +) -> Result<(), NtStatus> { + match fs.file_status(path) { + Ok(status) if status.file_type == FileType::Directory => Ok(()), + Ok(_) => Err(NtStatus::OBJECT_TYPE_MISMATCH), + Err(FileStatusError::PathError( + PathError::NoSuchFileOrDirectory | PathError::MissingComponent, + )) => match fs.mkdir( + path, + Mode::RUSR | Mode::WUSR | Mode::XUSR | Mode::ROTH | Mode::WOTH | Mode::XOTH, + ) { + Ok(()) | Err(MkdirError::AlreadyExists) => Ok(()), + Err(error) => Err(map_mkdir_error(error)), + }, + Err(FileStatusError::PathError(error)) => { + Err(map_path_error(error, NtStatus::OBJECT_NAME_NOT_FOUND)) + } + Err(_) => Err(NtStatus::UNSUCCESSFUL), + } +} + +fn value_path(key_path: &str, value_name: &str) -> Result { + if !is_valid_value_name(value_name) { + return Err(NtStatus::INVALID_PARAMETER); + } + + let mut path = String::from(key_path); + if !path.ends_with('/') { + path.push('/'); + } + path.push_str(VALUES_DIR_NAME); + path.push('/'); + path.push_str(&value_name.to_ascii_lowercase()); + Ok(path) +} + +fn is_valid_value_name(value_name: &str) -> bool { + !value_name.is_empty() + && value_name != "." + && value_name != ".." + && !value_name.contains('/') + && !value_name.contains('\\') +} + +fn map_open_error(error: OpenError) -> NtStatus { + match error { + OpenError::PathError(error) => map_path_error(error, NtStatus::OBJECT_NAME_NOT_FOUND), + OpenError::AccessNotAllowed | OpenError::NoWritePerms | OpenError::ReadOnlyFileSystem => { + NtStatus::ACCESS_DENIED + } + OpenError::AlreadyExists => NtStatus::OBJECT_NAME_COLLISION, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_file_status_error(error: FileStatusError) -> NtStatus { + match error { + FileStatusError::PathError(error) => map_path_error(error, NtStatus::OBJECT_NAME_NOT_FOUND), + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_mkdir_error(error: MkdirError) -> NtStatus { + match error { + MkdirError::AlreadyExists => NtStatus::OBJECT_NAME_COLLISION, + MkdirError::PathError(error) => map_path_error(error, NtStatus::OBJECT_PATH_NOT_FOUND), + MkdirError::NoWritePerms | MkdirError::ReadOnlyFileSystem => NtStatus::ACCESS_DENIED, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_path_error(error: PathError, not_found_status: NtStatus) -> NtStatus { + match error { + PathError::NoSuchFileOrDirectory | PathError::MissingComponent => not_found_status, + PathError::ComponentNotADirectory => NtStatus::NOT_A_DIRECTORY, + PathError::InvalidPathname => NtStatus::INVALID_PARAMETER, + PathError::NoSearchPerms { .. } => NtStatus::UNSUCCESSFUL, + } +} + +fn map_write_error(error: WriteError) -> NtStatus { + match error { + WriteError::NotForWriting => NtStatus::ACCESS_DENIED, + WriteError::NotAFile => NtStatus::OBJECT_TYPE_MISMATCH, + _ => NtStatus::UNSUCCESSFUL, + } +} + +fn map_read_error(error: ReadError) -> NtStatus { + match error { + ReadError::NotForReading => NtStatus::ACCESS_DENIED, + ReadError::NotAFile => NtStatus::OBJECT_TYPE_MISMATCH, + _ => NtStatus::UNSUCCESSFUL, + } +} + +#[cfg(test)] +mod tests { + use crate::tests::{ + TestFS, TestPlatform, const_ptr, mut_byte_ptr, mut_ptr, object_attributes, test_platform, + unicode_string, utf16_units as utf16, + }; + + use super::*; + use core::mem::size_of; + use litebox::LiteBox; + + extern crate std; + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const ERROR_ACCESS_DENIED: i32 = 5; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const ERROR_SUCCESS: i32 = 0; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const HKEY_CURRENT_USER: *mut core::ffi::c_void = 0xffffffff80000001usize as _; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const HKEY_LOCAL_MACHINE: *mut core::ffi::c_void = 0xffffffff80000002usize as _; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const HOST_CODE_PAGE_KEY: &str = "SYSTEM\\CurrentControlSet\\Control\\Nls\\CodePage"; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const HOST_ACCESS_TEST_KEY: &str = "Software\\LiteBoxRegistryAccessTest"; + + const KEY_VALUE_PARTIAL_INFORMATION_DATA_OFFSET: usize = + offset_of!(KeyValuePartialInformation, data); + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[allow(non_snake_case)] + #[link(name = "advapi32")] + unsafe extern "system" { + fn RegCreateKeyExW( + hKey: *mut core::ffi::c_void, + lpSubKey: *const u16, + Reserved: u32, + lpClass: *const u16, + dwOptions: u32, + samDesired: u32, + lpSecurityAttributes: *const core::ffi::c_void, + phkResult: *mut *mut core::ffi::c_void, + lpdwDisposition: *mut u32, + ) -> i32; + fn RegOpenKeyExW( + hKey: *mut core::ffi::c_void, + lpSubKey: *const u16, + ulOptions: u32, + samDesired: u32, + phkResult: *mut *mut core::ffi::c_void, + ) -> i32; + fn RegQueryValueExW( + hKey: *mut core::ffi::c_void, + lpValueName: *const u16, + lpReserved: *mut u32, + lpType: *mut u32, + lpData: *mut u8, + lpcbData: *mut u32, + ) -> i32; + fn RegSetValueExW( + hKey: *mut core::ffi::c_void, + lpValueName: *const u16, + Reserved: u32, + dwType: u32, + lpData: *const u8, + cbData: u32, + ) -> i32; + fn RegCloseKey(hKey: *mut core::ffi::c_void) -> i32; + fn RegDeleteTreeW(hKey: *mut core::ffi::c_void, lpSubKey: *const u16) -> i32; + } + + fn test_registry() -> (LiteBox, RegistryStore) { + let litebox = LiteBox::new(test_platform()); + let registry = RegistryStore::new(&litebox); + (litebox, registry) + } + + fn open_key( + task: &Task, + object_attributes: ObjectAttributes, + ) -> Result { + task.do_nt_open_key(RegistryKeyAccess::READ.bits(), object_attributes) + } + + fn open_code_page_key(task: &Task) -> Handle { + let code_page_name = utf16(DEFAULT_CODE_PAGE_KEY); + let code_page_name = unicode_string(&code_page_name); + let object_attributes = object_attributes(&code_page_name, 0); + open_key(task, object_attributes).expect("Failed to open code page key") + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn nul_terminated_utf16(value: &str) -> Vec { + let mut value = utf16(value); + value.push(0); + value + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_registry_value(key_path: &str, value_name: &str) -> RegistryValue { + let key_path = nul_terminated_utf16(key_path); + let value_name = nul_terminated_utf16(value_name); + let mut key = core::ptr::null_mut(); + // SAFETY: The key path is NUL-terminated, `phkResult` points to a live output + // slot, and `HKEY_LOCAL_MACHINE` is the documented predefined registry handle. + let status = unsafe { + RegOpenKeyExW( + HKEY_LOCAL_MACHINE, + key_path.as_ptr(), + 0, + RegistryKeyAccess::QUERY_VALUE.bits(), + &raw mut key, + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to open host registry key"); + + let mut value_type = 0; + let mut data_len = 0; + // SAFETY: The key handle was returned by `RegOpenKeyExW`, the value name is + // NUL-terminated, and the null data buffer requests the required byte length. + let status = unsafe { + RegQueryValueExW( + key, + value_name.as_ptr(), + core::ptr::null_mut(), + &raw mut value_type, + core::ptr::null_mut(), + &raw mut data_len, + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to size host registry value"); + + let mut data = vec![0; data_len as usize]; + // SAFETY: `data` has exactly the byte length returned by the sizing query, + // and all other pointers remain valid for the duration of the call. + let status = unsafe { + RegQueryValueExW( + key, + value_name.as_ptr(), + core::ptr::null_mut(), + &raw mut value_type, + data.as_mut_ptr(), + &raw mut data_len, + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to read host registry value"); + data.truncate(data_len as usize); + + // SAFETY: The key handle was returned by `RegOpenKeyExW` and has not been closed yet. + let status = unsafe { RegCloseKey(key) }; + assert_eq!(status, ERROR_SUCCESS, "failed to close host registry key"); + + RegistryValue { + value_type: RegistryValueType::try_from(value_type).expect("known registry value type"), + data, + } + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_query_value_with_set_only_access() -> i32 { + let key_path = nul_terminated_utf16(HOST_ACCESS_TEST_KEY); + let value_name = nul_terminated_utf16("Value"); + let mut key = core::ptr::null_mut(); + // SAFETY: The key path is NUL-terminated, output pointers are live slots, + // and `HKEY_CURRENT_USER` is the documented predefined registry handle. + let status = unsafe { + RegCreateKeyExW( + HKEY_CURRENT_USER, + key_path.as_ptr(), + 0, + core::ptr::null(), + 0, + RegistryKeyAccess::QUERY_VALUE.bits() | RegistryKeyAccess::SET_VALUE.bits(), + core::ptr::null(), + &raw mut key, + core::ptr::null_mut(), + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to create host test key"); + + let data = [b'x', 0, 0, 0]; + // SAFETY: The key handle was returned by `RegCreateKeyExW`, the value name + // is NUL-terminated, and `data` is valid for the specified byte length. + let status = unsafe { + RegSetValueExW( + key, + value_name.as_ptr(), + 0, + RegistryValueType::Sz.into(), + data.as_ptr(), + data.len().trunc(), + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to seed host test value"); + // SAFETY: The key handle was returned by `RegCreateKeyExW` and has not + // been closed yet. + let status = unsafe { RegCloseKey(key) }; + assert_eq!(status, ERROR_SUCCESS, "failed to close host test key"); + + // SAFETY: The key path is NUL-terminated, `phkResult` points to a live + // output slot, and `HKEY_CURRENT_USER` is the documented predefined handle. + let status = unsafe { + RegOpenKeyExW( + HKEY_CURRENT_USER, + key_path.as_ptr(), + 0, + RegistryKeyAccess::SET_VALUE.bits(), + &raw mut key, + ) + }; + assert_eq!(status, ERROR_SUCCESS, "failed to reopen host test key"); + + let mut value_type = 0; + let mut data_len = 0; + // SAFETY: The key handle was returned by `RegOpenKeyExW`, the value name + // is NUL-terminated, and the null data buffer requests the required length. + let query_status = unsafe { + RegQueryValueExW( + key, + value_name.as_ptr(), + core::ptr::null_mut(), + &raw mut value_type, + core::ptr::null_mut(), + &raw mut data_len, + ) + }; + + // SAFETY: The key handle was returned by `RegOpenKeyExW` and has not been closed yet. + let close_status = unsafe { RegCloseKey(key) }; + assert_eq!(close_status, ERROR_SUCCESS, "failed to close host test key"); + // SAFETY: The key path is NUL-terminated and rooted under the documented + // predefined `HKEY_CURRENT_USER` handle. + let delete_status = unsafe { RegDeleteTreeW(HKEY_CURRENT_USER, key_path.as_ptr()) }; + assert_eq!( + delete_status, ERROR_SUCCESS, + "failed to delete host test key" + ); + + query_status + } + + #[test] + fn registry_store_separates_values_from_subkeys() { + let (_litebox, registry) = test_registry(); + let key_path = absolute_nt_key_name_to_fs_path(DEFAULT_CODE_PAGE_KEY).unwrap(); + let value_path = value_path(&key_path, "ACP").unwrap(); + + assert_eq!( + registry.fs.file_status(&*value_path).unwrap().file_type, + FileType::RegularFile + ); + assert_eq!( + registry.fs.file_status(&*value_path).unwrap().size, + REGISTRY_VALUE_TYPE_SIZE + DEFAULT_ACP_VALUE.len() + ); + let value = registry.read_value_at_path(&key_path, "ACP").unwrap(); + assert_eq!(value.value_type, RegistryValueType::Sz); + assert_eq!(value.data, DEFAULT_ACP_VALUE); + + let values_dir = absolute_nt_key_name_to_fs_path( + "\\Registry\\Machine\\System\\CurrentControlSet\\Control\\Nls\\CodePage\\.values", + ); + assert_eq!(values_dir, Err(NtStatus::INVALID_PARAMETER)); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn registry_default_code_page_values_match_host() { + let task = crate::tests::test_task(); + let key_handle = open_code_page_key(&task); + + for name in ["ACP", "OEMCP", "MACCP"] { + let host_value = host_registry_value(HOST_CODE_PAGE_KEY, name); + let value_name = utf16(name); + let value_name = unicode_string(&value_name); + let mut information = [0u8; 64]; + let mut result_length = 0; + + assert!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .is_ok() + ); + + let information = &information[..(result_length as usize)]; + let (information, data) = + KeyValuePartialInformation::read_from_prefix(information).unwrap(); + + assert_eq!(host_value.value_type, RegistryValueType::Sz); + assert_eq!(information.value_type, host_value.value_type.into()); + assert_eq!(information.data_length as usize, host_value.data.len()); + assert_eq!(data, host_value.data.as_slice()); + } + } + + #[test] + fn nt_open_key_opens_existing_absolute_and_relative_keys() { + let task = crate::tests::test_task(); + let nls_name = utf16("\\Registry\\Machine\\System\\CurrentControlSet\\Control\\Nls"); + let nls_name = unicode_string(&nls_name); + let nls_object_attributes = object_attributes(&nls_name, 0); + let nls_handle = open_key(&task, nls_object_attributes).expect("Failed to open NLS key"); + assert_ne!(nls_handle, Handle::default()); + + let code_page_name = utf16("CodePage"); + let code_page_name = unicode_string(&code_page_name); + let mut code_page_object_attributes = object_attributes(&code_page_name, 0); + code_page_object_attributes.root_directory = nls_handle; + let code_page_handle = + open_key(&task, code_page_object_attributes).expect("Failed to open code page key"); + assert_ne!(code_page_handle, Handle::default()); + } + + #[test] + fn nt_open_key_reports_missing_absolute_key() { + let task = crate::tests::test_task(); + let name = utf16("\\Registry\\Machine\\Software\\Missing"); + let name = unicode_string(&name); + let object_attributes = object_attributes(&name, 0); + assert_eq!( + open_key(&task, object_attributes).unwrap_err(), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + } + + #[test] + fn nt_open_key_rejects_invalid_relative_root() { + let task = crate::tests::test_task(); + let name = utf16("Child"); + let name = unicode_string(&name); + let mut object_attributes = object_attributes(&name, 0); + object_attributes.root_directory = Handle::from_raw(0x1234); + assert_eq!( + open_key(&task, object_attributes).unwrap_err(), + NtStatus::INVALID_HANDLE + ); + } + + #[test] + fn nt_open_key_checks_backing_fs_permissions() { + let task = crate::tests::test_task(); + let private_key = "\\Registry\\Machine\\Software\\Private"; + let private_path = create_key_in_fs(&task.global.registry.fs, private_key).unwrap(); + task.global + .registry + .fs + .chmod(&*private_path, Mode::WUSR | Mode::XUSR) + .unwrap(); + + let private_name = utf16(private_key); + let private_name = unicode_string(&private_name); + let read_object_attributes = object_attributes(&private_name, 0); + assert_eq!( + open_key(&task, read_object_attributes).unwrap_err(), + NtStatus::ACCESS_DENIED + ); + + let private_name = utf16(private_key); + let private_name = unicode_string(&private_name); + let write_object_attributes = object_attributes(&private_name, 0); + let handle = task + .do_nt_open_key(RegistryKeyAccess::SET_VALUE.bits(), write_object_attributes) + .expect("write-only access should use write filesystem permissions"); + assert_ne!(handle, Handle::default()); + } + + #[test] + fn nt_close_removes_registry_key_handle() { + let task = crate::tests::test_task(); + let key_handle = open_code_page_key(&task); + let value_name = utf16("ACP"); + let value_name = unicode_string(&value_name); + let mut information = [0u8; 64]; + let mut result_length = 0; + + assert!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .is_ok() + ); + assert_eq!(task.sys_nt_close(key_handle), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(key_handle), NtStatus::INVALID_HANDLE); + assert_eq!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .unwrap_err(), + NtStatus::INVALID_HANDLE + ); + } + + #[test] + fn nt_query_value_key_reports_partial_information() { + let task = crate::tests::test_task(); + let key_handle = open_code_page_key(&task); + let value_name = utf16("ACP"); + let value_name = unicode_string(&value_name); + let mut information = [0u8; 64]; + let mut result_length = 0; + + assert!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .is_ok() + ); + + assert_eq!( + result_length as usize, + size_of::() + DEFAULT_ACP_VALUE.len() + ); + let information = &information[..(result_length as usize)]; + let (information, data) = + KeyValuePartialInformation::read_from_prefix(information).unwrap(); + assert_eq!(information.title_index, 0); + assert_eq!(information.value_type, RegistryValueType::Sz.into()); + assert_eq!(information.data_length, DEFAULT_ACP_VALUE.len().trunc()); + assert_eq!(data, DEFAULT_ACP_VALUE); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_value_key_without_query_access_matches_host() { + assert_eq!(host_query_value_with_set_only_access(), ERROR_ACCESS_DENIED); + + let task = crate::tests::test_task(); + let code_page_name = utf16(DEFAULT_CODE_PAGE_KEY); + let code_page_name = unicode_string(&code_page_name); + let object_attributes = object_attributes(&code_page_name, 0); + let key_handle = task + .do_nt_open_key(RegistryKeyAccess::SET_VALUE.bits(), object_attributes) + .expect("write-only open should succeed against the seeded registry store"); + let value_name = utf16("ACP"); + let value_name = unicode_string(&value_name); + let mut information = [0u8; 64]; + let mut result_length = 0; + + assert_eq!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .unwrap_err(), + NtStatus::ACCESS_DENIED + ); + } + + #[test] + fn nt_query_value_key_reports_basic_and_full_information() { + let task = crate::tests::test_task(); + let key_handle = open_code_page_key(&task); + let value_name = utf16("OEMCP"); + let value_name = unicode_string(&value_name); + let mut basic_information = [0u8; 64]; + let mut full_information = [0u8; 64]; + let mut result_length = 0; + + assert!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Basic, + mut_byte_ptr(&mut basic_information), + basic_information.len().trunc(), + mut_ptr(&mut result_length), + ) + .is_ok() + ); + let name = utf16le("OEMCP"); + assert_eq!( + result_length as usize, + size_of::() + name.len() + ); + let basic_information = &basic_information[..(result_length as usize)]; + let (basic_information, basic_name) = + KeyValueBasicInformation::read_from_prefix(basic_information).unwrap(); + assert_eq!(basic_information.title_index, 0); + assert_eq!(basic_information.value_type, RegistryValueType::Sz.into()); + assert_eq!(basic_information.name_length as usize, name.len()); + assert_eq!(basic_name, name.as_slice()); + + assert!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Full, + mut_byte_ptr(&mut full_information), + full_information.len().trunc(), + mut_ptr(&mut result_length), + ) + .is_ok() + ); + let full_information = &full_information[..(result_length as usize)]; + let (full_header, full_tail) = + KeyValueFullInformation::read_from_prefix(full_information).unwrap(); + let data_offset = full_header.data_offset as usize; + assert_eq!(full_header.title_index, 0); + assert_eq!(full_header.value_type, RegistryValueType::Sz.into()); + assert_eq!(full_header.data_length as usize, DEFAULT_OEMCP_VALUE.len()); + assert_eq!(full_header.name_length as usize, name.len()); + assert_eq!(&full_tail[..name.len()], name.as_slice()); + assert_eq!( + &full_information[data_offset..data_offset + DEFAULT_OEMCP_VALUE.len()], + DEFAULT_OEMCP_VALUE + ); + } + + #[test] + fn nt_query_value_key_rejects_invalid_arguments() { + let task = crate::tests::test_task(); + let key_handle = open_code_page_key(&task); + let value_name = utf16("ACP"); + let value_name = unicode_string(&value_name); + let missing_value_name = utf16("Missing"); + let missing_value_name = unicode_string(&missing_value_name); + let mut information = [0u8; 64]; + let mut short_information = [0u8; KEY_VALUE_PARTIAL_INFORMATION_DATA_OFFSET - 1]; + let mut result_length = 0; + + assert_eq!( + task.do_nt_query_value_key( + Handle::from_raw(0x1234), + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .unwrap_err(), + NtStatus::INVALID_HANDLE + ); + + assert_eq!( + task.do_nt_query_value_key( + key_handle, + missing_value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ) + .unwrap_err(), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + + assert_eq!( + task.sys_nt_query_value_key( + key_handle, + const_ptr(&value_name), + 0xffff, + mut_byte_ptr(&mut information), + information.len().trunc(), + mut_ptr(&mut result_length), + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + task.do_nt_query_value_key( + key_handle, + value_name, + KeyValueInformationClass::Partial, + mut_byte_ptr(&mut short_information), + short_information.len().trunc(), + mut_ptr(&mut result_length), + ) + .unwrap_err(), + NtStatus::BUFFER_OVERFLOW + ); + assert_eq!(result_length, 22); + } +} diff --git a/litebox_shim_windows/src/syscalls/section.rs b/litebox_shim_windows/src/syscalls/section.rs new file mode 100644 index 0000000000..775692d165 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/section.rs @@ -0,0 +1,1782 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::string::String; +use alloc::sync::Arc; +use core::marker::PhantomData; +use core::mem::size_of; +use core::sync::atomic::{AtomicBool, Ordering}; + +use int_enum::IntEnum; +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::mm::linux::{CreatePagesFlags, NonZeroPageSize}; +use litebox::platform::page_mgmt::MemoryRegionPermissions; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox_common_windows::nt_status::NtStatus; +use rangemap::RangeMap; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::{AccessMask, ObjectAttributes}; +use crate::syscalls::mm::{MemoryType, PageProtection, create_pages, parse_page_protection}; +use crate::syscalls::{Handle, ProcessHandle}; +use crate::{ConstPtr, MutPtr, PAGE_SIZE, ShimFS, ShimPlatform, Task, WindowsSectionView}; + +const VIEW_SHARE: u32 = 1; +const VIEW_UNMAP: u32 = 2; +const MEM_TOP_DOWN: u32 = 0x0010_0000; +const MEM_PHYSICAL: u32 = 0x0040_0000; +const MEM_DIFFERENT_IMAGE_BASE_OK: u32 = 0x0080_0000; +const SUPPORTED_MAP_ALLOCATION_TYPES: u32 = + MEM_TOP_DOWN | MEM_PHYSICAL | MEM_DIFFERENT_IMAGE_BASE_OK; +pub(crate) const WINDOWS_SHARED_SECTION_OBJECT: &str = r"\Windows\SharedSection"; +pub(crate) const WINDOWS_SESSION_SHARED_SECTION_OBJECT: &str = r"\Sessions\0\Windows\SharedSection"; +pub(crate) const WINDOWS_SHARED_SECTION_SIZE: usize = 0x1_0000; + +enum SectionBacking { + /// LiteBox lacks shared anonymous backing, so a pagefile section is + /// metadata-only until its single allowed view is mapped. Remap after unmap + /// is rejected instead of storing contents in shim memory or a file. + /// + /// Shared write-through backing across concurrent views is the deferred + /// capability (see `TODO(section-subsystem)`); until it lands, a single view + /// is the only observable-faithful case, which is why both the + /// second-concurrent-view and the remap-after-unmap rejects exist. They are + /// one missing feature, not two unrelated limitations. + Pagefile, + /// CSR shared section is created by kernel and shared across process. For now, + /// we create it in userland for a process during initialization, and thus the first + /// map request would return the pre-mapped address. Subsequent map requests would + /// be rejected as LiteBox lacks shared mapping support. + CsrSharedSection { + base: usize, + }, + ImageFile, +} + +pub(crate) struct SectionSubsystem(PhantomData); + +impl FdEnabledSubsystem for SectionSubsystem { + type Entry = SectionHandleObject; +} + +impl FdEnabledSubsystemEntry for SectionHandleObject {} + +impl crate::WindowsHandleSubsystem for SectionSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + SectionAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct SectionHandleObject { + section: Arc>, +} + +pub(crate) struct SectionObject { + fs_path: Option, + size: usize, + attributes: SectionAllocationAttributes, + protection: PageProtection, + backing: SectionBacking, + pagefile_view_active: AtomicBool, + _platform: PhantomData, +} + +pub(crate) struct MapViewOfSectionParameters { + pub(crate) section_handle: Handle, + pub(crate) process_handle: ProcessHandle, + pub(crate) base_address: MutPtr, + pub(crate) zero_bits: usize, + pub(crate) commit_size: usize, + pub(crate) section_offset: Option>, + pub(crate) view_size: MutPtr, + pub(crate) inherit_disposition: u32, + pub(crate) allocation_type: u32, + pub(crate) page_protection: u32, +} + +pub(super) struct MappedPagefileSectionView { + pub(super) base: usize, + pub(super) mapped_size: usize, + pub(super) view_size: usize, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct SectionAllocationAttributes: u32 { + const SEC_FILE = 0x0080_0000; + const SEC_IMAGE = 0x0100_0000; + const SEC_RESERVE = 0x0400_0000; + const SEC_COMMIT = 0x0800_0000; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct SectionAccess: u32 { + const QUERY = 0x0001; + const MAP_WRITE = 0x0002; + const MAP_READ = 0x0004; + const MAP_EXECUTE = 0x0008; + const EXTEND_SIZE = 0x0010; + const MAP_EXECUTE_EXPLICIT = 0x0020; + + const GENERIC_READ_EXPANSION = AccessMask::STANDARD_RIGHTS_READ.bits() + | Self::QUERY.bits() + | Self::MAP_READ.bits(); + const GENERIC_WRITE_EXPANSION = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | Self::MAP_WRITE.bits() + | Self::EXTEND_SIZE.bits(); + const GENERIC_EXECUTE_EXPANSION = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() + | Self::MAP_EXECUTE.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::QUERY.bits() + | Self::MAP_WRITE.bits() + | Self::MAP_READ.bits() + | Self::MAP_EXECUTE.bits() + | Self::EXTEND_SIZE.bits(); + const GENERIC_ALL = AccessMask::GENERIC_ALL.bits(); + const GENERIC_EXECUTE = AccessMask::GENERIC_EXECUTE.bits(); + const GENERIC_WRITE = AccessMask::GENERIC_WRITE.bits(); + const GENERIC_READ = AccessMask::GENERIC_READ.bits(); + + const _ = !0; + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct UnmapViewOfSectionFlags: u32 { + const _ = 0; + } +} + +impl SectionAccess { + fn from_desired_access(desired_access: u32) -> Self { + let mut access = Self::from_bits_retain(desired_access); + if access.contains(Self::GENERIC_READ) { + access.remove(Self::GENERIC_READ); + access.insert(Self::GENERIC_READ_EXPANSION); + } + if access.contains(Self::GENERIC_WRITE) { + access.remove(Self::GENERIC_WRITE); + access.insert(Self::GENERIC_WRITE_EXPANSION); + } + if access.contains(Self::GENERIC_EXECUTE) { + access.remove(Self::GENERIC_EXECUTE); + access.insert(Self::GENERIC_EXECUTE_EXPANSION); + } + if access.contains(Self::GENERIC_ALL) { + access.remove(Self::GENERIC_ALL); + access.insert(Self::ALL_ACCESS); + } + access + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum SectionInformationClass { + Basic = 0, + Image = 1, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SectionBasicInformation { + base_address: usize, + attributes: u32, + _padding: u32, + size: i64, +} + +const _: () = assert!(size_of::() == 24); + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SectionImageInformation { + transfer_address: usize, + zero_bits: u32, + _padding0: u32, + maximum_stack_size: usize, + committed_stack_size: usize, + subsystem_type: u32, + subsystem_minor_version: u16, + subsystem_major_version: u16, + gp_value: u32, + image_characteristics: u16, + dll_characteristics: u16, + machine: u16, + image_contains_code: u8, + image_flags: u8, + loader_flags: u32, + image_file_size: u32, + checksum: u32, +} + +const _: () = assert!(size_of::() == 64); + +impl Task { + fn insert_section_handle( + &self, + section: Arc>, + granted_access: SectionAccess, + ) -> Result { + self.insert_typed_handle::>( + SectionHandleObject { section }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_section_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_section(section: SectionHandleObject) { + drop(section); + } + + #[expect( + clippy::too_many_arguments, + reason = "NtCreateSection has seven ABI parameters; keeping ABI args explicit avoids reshuffling" + )] + pub(crate) fn sys_nt_create_section( + &self, + section_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + maximum_size: Option>, + section_page_protection: u32, + allocation_attributes: u32, + file_handle: Handle, + ) -> NtStatus { + // Host ntdll preserves the output handle for pre-creation validation failures such as a + // NULL MaximumSize pagefile section. + if let Err(status) = + crate::probe_guest_output_preserving_value::(section_handle) + { + return status; + } + let granted_access = SectionAccess::from_desired_access(desired_access); + if granted_access.is_empty() { + return NtStatus::ACCESS_DENIED; + } + let Some((protection, _)) = parse_page_protection(section_page_protection) else { + return NtStatus::INVALID_PAGE_PROTECTION; + }; + // NtCreateSection currently supports only pagefile-backed sections. File-backed image + // sections are synthesized by NtOpenSection for KnownDlls; accepting a file handle here + // requires section lifetime/sharing to be keyed by the underlying file object identity. + if !file_handle.is_null() { + litebox_util_log::debug!( + file_handle = file_handle.as_raw(), + allocation_attributes:% = format_args!("{allocation_attributes:#x}"), + section_page_protection:% = format_args!("{section_page_protection:#x}"), + desired_access:% = format_args!("{desired_access:#x}"); + "Unsupported file-backed NtCreateSection" + ); + return NtStatus::INVALID_HANDLE; + } + let allocation_attributes = + SectionAllocationAttributes::from_bits_retain(allocation_attributes); + let supported_create_attributes = + SectionAllocationAttributes::SEC_RESERVE | SectionAllocationAttributes::SEC_COMMIT; + if !allocation_attributes + .difference(supported_create_attributes) + .is_empty() + { + return NtStatus::INVALID_PARAMETER; + } + if !allocation_attributes.intersects(supported_create_attributes) { + return NtStatus::INVALID_PARAMETER; + } + + let Some(maximum_size) = maximum_size else { + return NtStatus::INVALID_PARAMETER_4; + }; + let maximum_size = match maximum_size.read_at_offset(0) { + Some(value) if value > 0 => value, + Some(_) => return NtStatus::INVALID_PARAMETER, + None => return NtStatus::ACCESS_VIOLATION, + }; + let Ok(size) = usize::try_from(maximum_size) else { + return NtStatus::SECTION_TOO_BIG; + }; + let Some(size) = size.checked_next_multiple_of(PAGE_SIZE) else { + return NtStatus::SECTION_TOO_BIG; + }; + if NonZeroPageSize::::new(size).is_none() { + return NtStatus::INVALID_PARAMETER; + } + + let name = match self.read_section_name(object_attributes) { + Ok(name) => name, + Err(status) => return status, + }; + let attributes = if allocation_attributes.contains(SectionAllocationAttributes::SEC_RESERVE) + { + SectionAllocationAttributes::SEC_RESERVE + } else { + SectionAllocationAttributes::SEC_COMMIT + }; + let section = Arc::new(SectionObject { + fs_path: None, + size, + attributes, + protection, + backing: SectionBacking::Pagefile, + pagefile_view_active: AtomicBool::new(false), + _platform: PhantomData, + }); + if let Some(name) = &name { + let status = self.process.object_manager.create_section(name, §ion); + if status != NtStatus::SUCCESS { + return status; + } + } + self.publish_section_handle(section_handle, section, granted_access) + } + + #[expect( + clippy::too_many_arguments, + reason = "NtCreateSectionEx extends NtCreateSection with two ABI parameters" + )] + pub(crate) fn sys_nt_create_section_ex( + &self, + section_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + maximum_size: Option>, + section_page_protection: u32, + allocation_attributes: u32, + file_handle: Handle, + extended_parameters: Option>, + extended_parameter_count: u32, + ) -> NtStatus { + if extended_parameters.is_some() || extended_parameter_count != 0 { + return NtStatus::INVALID_PARAMETER; + } + self.sys_nt_create_section( + section_handle, + desired_access, + object_attributes, + maximum_size, + section_page_protection, + allocation_attributes, + file_handle, + ) + } + + pub(crate) fn sys_nt_open_section( + &self, + section_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + // Host ntdll zeroes the output handle before resolving a missing section name. + if section_handle + .write_at_offset(0, Handle::default()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + let granted_access = SectionAccess::from_desired_access(desired_access); + if granted_access.is_empty() { + return NtStatus::ACCESS_DENIED; + } + if object_attributes.is_none() { + return NtStatus::INVALID_PARAMETER; + } + let name = match self.read_required_section_name(object_attributes) { + Ok(name) => name, + Err(status) => return status, + }; + match self.process.object_manager.resolve_section(&name) { + Ok(section) => { + return self.publish_section_handle(section_handle, section, granted_access); + } + Err(NtStatus::OBJECT_NAME_NOT_FOUND | NtStatus::OBJECT_PATH_NOT_FOUND) => {} + Err(status) => return status, + } + // TODO: Windows creates one image section per known DLL during boot and lets every process + // map the same section. LiteBox currently lacks a shared image section subsystem, so we create + // a new section for each process that opens a known DLL. + let Some(fs_path) = known_dll_section_fs_path(&name) else { + return section_missing_status( + self.process.object_manager.parent_directory_exists(&name), + ); + }; + litebox_util_log::debug!( + section_name:% = name, + fs_path:% = fs_path; + "NtOpenSection: creating section for KnownDlls image" + ); + let Ok(file_status) = self.fs.file_status(&fs_path) else { + return NtStatus::OBJECT_NAME_NOT_FOUND; + }; + let section = Arc::new(SectionObject { + fs_path: Some(fs_path), + size: file_status.size, + attributes: SectionAllocationAttributes::SEC_FILE + | SectionAllocationAttributes::SEC_IMAGE, + protection: PageProtection::PAGE_EXECUTE_WRITECOPY, + backing: SectionBacking::ImageFile, + pagefile_view_active: AtomicBool::new(false), + _platform: PhantomData, + }); + self.publish_section_handle(section_handle, section, granted_access) + } + + pub(crate) fn sys_nt_query_section( + &self, + section_handle: Handle, + section_information_class: u32, + section_information: MutPtr, + section_information_length: usize, + return_length: Option>, + ) -> NtStatus { + let Ok(information_class) = SectionInformationClass::try_from(section_information_class) + else { + return NtStatus::INVALID_INFO_CLASS; + }; + let entry = match self.typed_handle_entry_with_access::>( + section_handle, + SectionAccess::QUERY.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + let section = entry.with_entry(|entry| Arc::clone(&entry.section)); + match information_class { + SectionInformationClass::Basic => write_section_basic_information::( + §ion, + section_information, + section_information_length, + return_length, + ), + SectionInformationClass::Image => write_section_image_information::( + §ion, + Arc::clone(&self.fs), + section_information, + section_information_length, + return_length, + ), + } + } + + pub(crate) fn sys_nt_map_view_of_section( + &self, + request: MapViewOfSectionParameters, + ) -> NtStatus { + if !request.process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + let Some(base) = request.base_address.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(requested_view_size) = request.view_size.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let section_offset = match request.section_offset { + Some(section_offset) => match section_offset.read_at_offset(0) { + Some(value) if value >= 0 => usize::try_from(value).unwrap_or(usize::MAX), + Some(_) => return NtStatus::INVALID_PARAMETER, + None => return NtStatus::ACCESS_VIOLATION, + }, + None => 0, + }; + if base != 0 + || request.zero_bits != 0 + || request.commit_size != 0 + || !section_offset.is_multiple_of(PAGE_SIZE) + || !matches!(request.inherit_disposition, VIEW_SHARE | VIEW_UNMAP) + || request.allocation_type & !SUPPORTED_MAP_ALLOCATION_TYPES != 0 + { + return NtStatus::INVALID_PARAMETER; + } + let Some((page_protection, permissions)) = parse_page_protection(request.page_protection) + else { + return NtStatus::INVALID_PAGE_PROTECTION; + }; + let entry = match self.typed_handle_entry_with_access::>( + request.section_handle, + required_map_access(page_protection).bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + let section = entry.with_entry(|entry| Arc::clone(&entry.section)); + match section.backing { + SectionBacking::Pagefile => self.map_pagefile_section( + request, + §ion, + requested_view_size, + section_offset, + page_protection, + permissions, + ), + SectionBacking::CsrSharedSection { .. } => self.map_csr_shared_section( + request, + §ion, + requested_view_size, + section_offset, + page_protection, + ), + SectionBacking::ImageFile => self.map_image_section(request, §ion, page_protection), + } + } + + pub(crate) fn sys_nt_map_view_of_section_ex( + &self, + request: MapViewOfSectionParameters, + extended_parameters: Option>, + extended_parameter_count: u32, + ) -> NtStatus { + if extended_parameters.is_some() || extended_parameter_count != 0 { + // TODO(section-subsystem): model MEM_EXTENDED_PARAMETER address requirements. + return NtStatus::INVALID_PARAMETER; + } + self.sys_nt_map_view_of_section(request) + } + + pub(crate) fn sys_nt_unmap_view_of_section( + &self, + process_handle: ProcessHandle, + base_address: usize, + ) -> NtStatus { + if !process_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + let Some((view_base, view)) = self.remove_section_view_for_address(base_address) else { + return NtStatus::NOT_MAPPED_VIEW; + }; + let owns_pages = view.section.as_ref().is_none_or(|section| { + !matches!(section.backing, SectionBacking::CsrSharedSection { .. }) + }); + if owns_pages { + let ptr = MutPtr::::from_usize(view_base); + // SAFETY: Section views are tracked only after this shim successfully creates the pages; + // unmapping consumes the tracked view and removes the exact owned range. + if unsafe { self.global.page_manager.remove_pages(ptr, view.size) }.is_err() { + self.process.section_views.write().insert(view_base, view); + return NtStatus::UNABLE_TO_FREE_VM; + } + } + self.process.virtual_allocations.write().remove(&view_base); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_unmap_view_of_section_ex( + &self, + process_handle: ProcessHandle, + base_address: usize, + flags: u32, + ) -> NtStatus { + let flags = UnmapViewOfSectionFlags::from_bits_retain(flags); + if !flags.is_empty() { + return NtStatus::INVALID_PARAMETER; + } + self.sys_nt_unmap_view_of_section(process_handle, base_address) + } + + fn publish_section_handle( + &self, + section_handle: MutPtr, + section: Arc>, + granted_access: SectionAccess, + ) -> NtStatus { + let handle = match self.insert_section_handle(section, granted_access) { + Ok(handle) => handle, + Err(status) => return status, + }; + if section_handle.write_at_offset(0, handle).is_none() { + self.close_section_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + fn map_pagefile_section( + &self, + request: MapViewOfSectionParameters, + section: &Arc>, + requested_view_size: usize, + section_offset: usize, + page_protection: PageProtection, + permissions: MemoryRegionPermissions, + ) -> NtStatus { + let mapped_view = match self.map_pagefile_section_view( + section, + requested_view_size, + section_offset, + page_protection, + permissions, + ) { + Ok(mapped_view) => mapped_view, + Err(status) => return status, + }; + if request + .base_address + .write_at_offset(0, mapped_view.base) + .is_none() + || request + .view_size + .write_at_offset(0, mapped_view.view_size) + .is_none() + { + self.rollback_pagefile_section_view(mapped_view.base); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(super) fn map_client_port_section( + &self, + section_handle: Handle, + requested_view_size: usize, + ) -> Result { + let entry = self.typed_handle_entry_with_access::>( + section_handle, + (SectionAccess::MAP_READ | SectionAccess::MAP_WRITE).bits(), + )?; + let section = entry.with_entry(|entry| Arc::clone(&entry.section)); + let page_protection = PageProtection::PAGE_READWRITE; + let Some((_, permissions)) = parse_page_protection(page_protection.bits()) else { + return Err(NtStatus::INVALID_PAGE_PROTECTION); + }; + self.map_pagefile_section_view( + §ion, + requested_view_size, + 0, + page_protection, + permissions, + ) + } + + fn map_pagefile_section_view( + &self, + section: &Arc>, + requested_view_size: usize, + section_offset: usize, + page_protection: PageProtection, + permissions: MemoryRegionPermissions, + ) -> Result { + if section_offset > section.size { + return Err(NtStatus::INVALID_VIEW_SIZE); + } + let remaining = section.size - section_offset; + let view_size = if requested_view_size == 0 { + remaining + } else { + requested_view_size + }; + if view_size == 0 || view_size > remaining { + return Err(NtStatus::INVALID_VIEW_SIZE); + } + let mapped_size = view_size + .checked_next_multiple_of(PAGE_SIZE) + .ok_or(NtStatus::INVALID_VIEW_SIZE)?; + let length = + NonZeroPageSize::::new(mapped_size).ok_or(NtStatus::INVALID_VIEW_SIZE)?; + match section.backing { + SectionBacking::Pagefile => {} + SectionBacking::CsrSharedSection { .. } | SectionBacking::ImageFile => { + return Err(NtStatus::INVALID_FILE_FOR_SECTION); + } + } + if !pagefile_view_protection_is_compatible(section.protection, page_protection) { + litebox_util_log::debug!( + section_protection:% = format_args!("{:#x}", section.protection.bits()), + page_protection:% = format_args!("{:#x}", page_protection.bits()); + "Rejected pagefile section view protection incompatible with section protection" + ); + return Err(NtStatus::SECTION_PROTECTION); + } + if section.pagefile_view_active.swap(true, Ordering::AcqRel) { + litebox_util_log::debug!( + section_size = section.size, + requested_view_size, + section_offset; + "Rejected additional pagefile section view" + ); + // Host 25H2 allows repeated and simultaneous pagefile views. LiteBox + // returns NOT_SUPPORTED until PageManager has first-class shared + // anonymous backing that avoids kernel-side content storage. + return Err(NtStatus::NOT_SUPPORTED); + } + let mapping = create_pages( + &self.global.page_manager, + None, + length, + CreatePagesFlags::empty(), + permissions, + |_| Ok(0), + ) + .map_err(|_| { + section.pagefile_view_active.store(false, Ordering::Release); + NtStatus::NO_MEMORY + })?; + let base = mapping.as_usize(); + self.process.section_views.write().insert( + base, + WindowsSectionView { + size: mapped_size, + section_offset, + section: Some(Arc::clone(section)), + }, + ); + self.process.virtual_allocations.write().insert( + base, + crate::WindowsVirtualAllocation { + base, + size: mapped_size, + allocation_protect: section.protection, + type_: MemoryType::MEM_MAPPED, + pages: committed_pages(base, mapped_size, page_protection), + }, + ); + Ok(MappedPagefileSectionView { + base, + mapped_size, + view_size, + }) + } + + pub(super) fn rollback_pagefile_section_view(&self, base_address: usize) { + let Some((view_base, view)) = self.remove_section_view_for_address(base_address) else { + return; + }; + if let Some(section) = &view.section + && matches!(section.backing, SectionBacking::Pagefile) + { + section.pagefile_view_active.store(false, Ordering::Release); + } + let _ = remove_view_pages::(&self.global.page_manager, view_base, view.size); + self.process.virtual_allocations.write().remove(&view_base); + } + + fn map_csr_shared_section( + &self, + request: MapViewOfSectionParameters, + section: &Arc>, + requested_view_size: usize, + section_offset: usize, + page_protection: PageProtection, + ) -> NtStatus { + if section_offset != 0 { + return NtStatus::INVALID_VIEW_SIZE; + } + let view_size = if requested_view_size == 0 { + section.size + } else { + requested_view_size + }; + if view_size == 0 || view_size > section.size { + return NtStatus::INVALID_VIEW_SIZE; + } + let Some(mapped_size) = view_size.checked_next_multiple_of(PAGE_SIZE) else { + return NtStatus::INVALID_VIEW_SIZE; + }; + if mapped_size > WINDOWS_SHARED_SECTION_SIZE { + litebox_util_log::debug!( + section_size = section.size, + requested_view_size, + section_offset; + "Rejected CSR shared section view larger than host limit" + ); + return NtStatus::INVALID_VIEW_SIZE; + } + if !pagefile_view_protection_is_compatible(section.protection, page_protection) { + return NtStatus::SECTION_PROTECTION; + } + if section.pagefile_view_active.swap(true, Ordering::AcqRel) { + litebox_util_log::debug!( + section_size = section.size, + requested_view_size, + section_offset; + "Rejected additional CSR shared section view" + ); + return NtStatus::NOT_SUPPORTED; + } + // TODO: we just return the pre-mapped base address for now, but we should support mapping at a different base address in the future. + let base = match section.backing { + SectionBacking::CsrSharedSection { base } => base, + SectionBacking::Pagefile | SectionBacking::ImageFile => unreachable!(), + }; + if request.base_address.write_at_offset(0, base).is_none() + || request.view_size.write_at_offset(0, view_size).is_none() + { + section.pagefile_view_active.store(false, Ordering::Release); + return NtStatus::ACCESS_VIOLATION; + } + self.process.section_views.write().insert( + base, + WindowsSectionView { + size: mapped_size, + section_offset, + section: Some(Arc::clone(section)), + }, + ); + self.process.virtual_allocations.write().insert( + base, + crate::WindowsVirtualAllocation { + base, + size: mapped_size, + allocation_protect: section.protection, + type_: MemoryType::MEM_MAPPED, + // TODO(section-subsystem): honor per-view CSR protections only after + // the backing is no longer aliased by PEB direct-deref pointers. + pages: committed_pages(base, mapped_size, section.protection), + }, + ); + NtStatus::SUCCESS + } + + fn map_image_section( + &self, + request: MapViewOfSectionParameters, + section: &SectionObject, + page_protection: PageProtection, + ) -> NtStatus { + let Some(fs_path) = §ion.fs_path else { + return NtStatus::INVALID_FILE_FOR_SECTION; + }; + if required_map_access(page_protection).contains(SectionAccess::MAP_WRITE) { + litebox_util_log::debug!( + page_protection:% = format_args!("{:#x}", page_protection.bits()), + fs_path:% = fs_path; + "Rejected writable image section view" + ); + // Host 25H2 maps SEC_IMAGE with PAGE_READWRITE/PAGE_EXECUTE_READWRITE successfully + // (NtMapViewOfSection returns STATUS_IMAGE_NOT_AT_BASE in the probe). LiteBox rejects + // TODO(section-subsystem): allow this once image mappings support writable + // copy-on-write/shared image pages. + return NtStatus::SECTION_PROTECTION; + } + let mapping = match crate::loader::load_image_section( + self.global.platform, + Arc::clone(&self.fs), + fs_path, + &self.global.page_manager, + &self.process.virtual_allocations, + ) { + Ok(mapping) => mapping, + Err(crate::loader::WindowsLoadError::Access(_)) => { + return NtStatus::OBJECT_NAME_NOT_FOUND; + } + Err(crate::loader::WindowsLoadError::Load(_)) => return NtStatus::NO_MEMORY, + Err(_) => return NtStatus::INVALID_FILE_FOR_SECTION, + }; + if request + .base_address + .write_at_offset(0, mapping.base_addr) + .is_none() + || request + .view_size + .write_at_offset(0, mapping.image_size) + .is_none() + { + let _ = remove_view_pages::( + &self.global.page_manager, + mapping.base_addr, + mapping.mapping_size, + ); + self.process + .virtual_allocations + .write() + .remove(&mapping.base_addr); + return NtStatus::ACCESS_VIOLATION; + } + self.process.section_views.write().insert( + mapping.base_addr, + WindowsSectionView { + size: mapping.mapping_size, + section_offset: 0, + section: None, + }, + ); + NtStatus::SUCCESS + } + + fn remove_section_view_for_address( + &self, + base_address: usize, + ) -> Option<(usize, WindowsSectionView)> { + let mut views = self.process.section_views.write(); + let (&view_base, view) = views.range(..=base_address).next_back()?; + let view = view.clone(); + let view_end = view_base.checked_add(view.size)?; + if base_address < view_end { + views.remove(&view_base); + Some((view_base, view)) + } else { + None + } + } + + fn read_section_name( + &self, + object_attributes: Option>, + ) -> Result, NtStatus> { + let (_, directory_name) = + self.read_directory_object_attributes(object_attributes, false)?; + Ok(directory_name.map(|name| name.original_path)) + } + + fn read_required_section_name( + &self, + object_attributes: Option>, + ) -> Result { + let (_, Some(directory_name)) = + self.read_directory_object_attributes(object_attributes, true)? + else { + return Err(NtStatus::INVALID_PARAMETER); + }; + Ok(directory_name.original_path) + } +} + +fn section_missing_status(parent_exists: bool) -> NtStatus { + if parent_exists { + NtStatus::OBJECT_NAME_NOT_FOUND + } else { + NtStatus::OBJECT_PATH_NOT_FOUND + } +} + +fn known_dll_section_fs_path(object_path: &str) -> Option { + let (dll_name, fs_directory) = + if let Some(rest) = strip_case_insensitive_prefix(object_path, r"\KnownDlls\") { + (rest, "/Windows/System32/") + } else { + let rest = strip_case_insensitive_prefix(object_path, r"\KnownDlls32\")?; + (rest, "/Windows/SysWOW64/") + }; + if dll_name.contains(['\\', '/']) || !ends_with_ignore_ascii_case(dll_name, ".dll") { + return None; + } + let mut fs_path = String::from(fs_directory); + fs_path.push_str(&dll_name.to_ascii_lowercase()); + Some(fs_path) +} + +pub(crate) fn load_time_windows_shared_section( + base: usize, +) -> Arc> { + // CSRSS creates this named section for the CSR client/server contract. LiteBox + // synthesizes it from the same static server data shape used for the PEB CSR + // pointers instead of exposing a zeroed generic pagefile section. + Arc::new(SectionObject { + fs_path: None, + size: WINDOWS_SHARED_SECTION_SIZE, + attributes: SectionAllocationAttributes::SEC_COMMIT, + protection: PageProtection::PAGE_READWRITE, + backing: SectionBacking::CsrSharedSection { base }, + pagefile_view_active: AtomicBool::new(false), + _platform: PhantomData, + }) +} + +fn strip_case_insensitive_prefix<'a>(value: &'a str, prefix: &str) -> Option<&'a str> { + if value + .get(..prefix.len()) + .is_some_and(|head| head.eq_ignore_ascii_case(prefix)) + { + value.get(prefix.len()..) + } else { + None + } +} + +fn ends_with_ignore_ascii_case(value: &str, suffix: &str) -> bool { + value + .get(value.len().saturating_sub(suffix.len())..) + .is_some_and(|tail| tail.eq_ignore_ascii_case(suffix)) +} + +fn required_map_access(protection: PageProtection) -> SectionAccess { + let base = protection.bits() & PageProtection::BASE_MASK; + if base == PageProtection::PAGE_NOACCESS.bits() { + SectionAccess::MAP_READ + } else if matches!( + base, + value if value == PageProtection::PAGE_READWRITE.bits() + || value == PageProtection::PAGE_EXECUTE_READWRITE.bits() + ) { + SectionAccess::MAP_WRITE + } else if matches!( + base, + value if value == PageProtection::PAGE_EXECUTE.bits() + || value == PageProtection::PAGE_EXECUTE_READ.bits() + || value == PageProtection::PAGE_EXECUTE_WRITECOPY.bits() + ) { + SectionAccess::MAP_EXECUTE + } else { + SectionAccess::MAP_READ + } +} + +fn pagefile_view_protection_is_compatible( + section_protection: PageProtection, + view_protection: PageProtection, +) -> bool { + let view_base = view_protection.bits() & PageProtection::BASE_MASK; + if view_base == PageProtection::PAGE_NOACCESS.bits() { + return true; + } + + if page_protection_has_read(view_protection) && !page_protection_has_read(section_protection) { + return false; + } + if page_protection_has_direct_write(view_protection) + && !page_protection_has_direct_write(section_protection) + { + return false; + } + if page_protection_has_execute(view_protection) + && !page_protection_has_execute(section_protection) + { + return false; + } + true +} + +fn page_protection_has_read(protection: PageProtection) -> bool { + matches!( + protection.bits() & PageProtection::BASE_MASK, + value if value == PageProtection::PAGE_READONLY.bits() + || value == PageProtection::PAGE_READWRITE.bits() + || value == PageProtection::PAGE_WRITECOPY.bits() + || value == PageProtection::PAGE_EXECUTE_READ.bits() + || value == PageProtection::PAGE_EXECUTE_READWRITE.bits() + || value == PageProtection::PAGE_EXECUTE_WRITECOPY.bits() + ) +} + +fn page_protection_has_direct_write(protection: PageProtection) -> bool { + matches!( + protection.bits() & PageProtection::BASE_MASK, + value if value == PageProtection::PAGE_READWRITE.bits() + || value == PageProtection::PAGE_EXECUTE_READWRITE.bits() + ) +} + +fn page_protection_has_execute(protection: PageProtection) -> bool { + matches!( + protection.bits() & PageProtection::BASE_MASK, + value if value == PageProtection::PAGE_EXECUTE.bits() + || value == PageProtection::PAGE_EXECUTE_READ.bits() + || value == PageProtection::PAGE_EXECUTE_READWRITE.bits() + || value == PageProtection::PAGE_EXECUTE_WRITECOPY.bits() + ) +} + +fn write_section_basic_information( + section: &SectionObject, + section_information: MutPtr, + section_information_length: usize, + return_length: Option>, +) -> NtStatus { + let required_len = size_of::(); + if section_information_length < required_len { + return NtStatus::INFO_LENGTH_MISMATCH; + } + let Ok(size) = i64::try_from(section.size) else { + return NtStatus::SECTION_TOO_BIG; + }; + let info = SectionBasicInformation { + base_address: 0, + attributes: section.attributes.bits(), + _padding: 0, + size, + }; + let output = + MutPtr::::from_usize(section_information.as_usize()); + if output.write_at_offset(0, info).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = return_length + && return_length.write_at_offset(0, required_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS +} + +fn write_section_image_information( + section: &SectionObject, + fs: Arc, + section_information: MutPtr, + section_information_length: usize, + return_length: Option>, +) -> NtStatus { + if !matches!(section.backing, SectionBacking::ImageFile) { + return NtStatus::SECTION_NOT_IMAGE; + } + let required_len = size_of::(); + if section_information_length < required_len { + return NtStatus::INFO_LENGTH_MISMATCH; + } + let Some(fs_path) = §ion.fs_path else { + return NtStatus::INVALID_FILE_FOR_SECTION; + }; + let metadata = match crate::loader::image_section_metadata(fs, fs_path) { + Ok(metadata) => metadata, + Err(crate::loader::WindowsLoadError::Access(_)) => return NtStatus::OBJECT_NAME_NOT_FOUND, + Err(_) => return NtStatus::INVALID_FILE_FOR_SECTION, + }; + // Host ntdll reports ReturnLength=64 for SectionImageInformation on x64; the public + // winternl.h layout ends at CheckSum and has no trailing extension fields. + let info = SectionImageInformation { + transfer_address: metadata.transfer_address, + zero_bits: 0, + _padding0: 0, + maximum_stack_size: 0, + committed_stack_size: 0, + subsystem_type: metadata.subsystem, + subsystem_minor_version: metadata.subsystem_minor_version, + subsystem_major_version: metadata.subsystem_major_version, + gp_value: 0, + image_characteristics: metadata.image_characteristics, + dll_characteristics: metadata.dll_characteristics, + machine: metadata.machine, + image_contains_code: 1, + image_flags: 0, + loader_flags: 0, + image_file_size: metadata.file_size, + checksum: 0, + }; + let output = + MutPtr::::from_usize(section_information.as_usize()); + if output.write_at_offset(0, info).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(return_length) = return_length + && return_length.write_at_offset(0, required_len).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS +} + +fn committed_pages( + base: usize, + size: usize, + protect: PageProtection, +) -> RangeMap { + let mut pages = RangeMap::new(); + if let Some(end) = base.checked_add(size) { + pages.insert(base..end, protect); + } + pages +} + +fn remove_view_pages( + page_manager: &crate::WindowsPageManager, + base: usize, + size: usize, +) -> Result<(), ()> { + let ptr = MutPtr::::from_usize(base); + // SAFETY: The caller passes a section view range created by this module and not yet exposed, + // or a tracked view being rolled back after output write failure. + unsafe { page_manager.remove_pages(ptr, size) }.map_err(|_| ()) +} + +#[cfg(test)] +mod tests { + extern crate std; + + use core::mem::{size_of, size_of_val}; + + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::{ObjectAttributes, UnicodeString}; + use crate::syscalls::event::EventType; + use crate::tests::{TestFS, TestPlatform, const_ptr, mut_byte_ptr, mut_ptr, test_task}; + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const IMAGE_FILE_MACHINE_AMD64: u16 = 0x8664; + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + const IMAGE_SUBSYSTEM_WINDOWS_CUI: u32 = 3; + + fn wide(value: &str) -> alloc::vec::Vec { + value.encode_utf16().collect() + } + + fn unicode(value: &[u16]) -> UnicodeString { + UnicodeString { + length: u16::try_from(size_of_val(value)).unwrap(), + maximum_length: u16::try_from(size_of_val(value)).unwrap(), + padding_0: [0; 4], + buffer: value.as_ptr() as usize, + } + } + + fn object_attributes(name: &UnicodeString) -> ObjectAttributes { + ObjectAttributes { + length: u32::try_from(size_of::()).unwrap(), + root_directory: Handle::from_raw(0), + object_name: core::ptr::from_ref(name) as usize, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + } + } + + fn create_pagefile_section( + task: &Task, + access: u32, + size: i64, + protection: PageProtection, + ) -> Handle { + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut handle), + access, + None, + Some(const_ptr(&size)), + protection.bits(), + SectionAllocationAttributes::SEC_COMMIT.bits(), + Handle::default(), + ), + NtStatus::SUCCESS + ); + handle + } + + fn map_pagefile_section(task: &Task, handle: Handle) -> (usize, usize) { + let mut base = 0usize; + let mut view_size = 0usize; + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: handle, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_READWRITE.bits(), + }), + NtStatus::SUCCESS + ); + (base, view_size) + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_kernel32_image() -> std::vec::Vec { + let system_root = std::env::var_os("SystemRoot").expect("SystemRoot is set on Windows"); + std::fs::read( + std::path::PathBuf::from(system_root) + .join("System32") + .join("kernel32.dll"), + ) + .expect("host kernel32.dll is readable") + } + + #[test] + fn nt_create_section_creates_queryable_pagefile_section() { + let task = test_task(); + let handle = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2345, + PageProtection::PAGE_READWRITE, + ); + let mut info = SectionBasicInformation { + base_address: usize::MAX, + attributes: u32::MAX, + _padding: u32::MAX, + size: -1, + }; + let mut return_length = 0usize; + + assert_eq!( + task.sys_nt_query_section( + handle, + SectionInformationClass::Basic as u32, + mut_byte_ptr(&mut info), + size_of::(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + assert_eq!(return_length, size_of::()); + assert_eq!(info.base_address, 0); + assert_eq!( + info.attributes, + SectionAllocationAttributes::SEC_COMMIT.bits() + ); + assert_eq!(info.size, 0x3000); + + let mut too_small = [0xcc; size_of::() - 1]; + let too_small_len = too_small.len(); + return_length = 0x5555_5555; + // Host 25H2 leaves ReturnLength untouched on INFO_LENGTH_MISMATCH + // (Basic len=23 -> ret stays sentinel) and writes 0x18 only on success. + assert_eq!( + task.sys_nt_query_section( + handle, + SectionInformationClass::Basic as u32, + mut_byte_ptr(&mut too_small), + too_small_len, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(return_length, 0x5555_5555); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_section_image_information_uses_pe_headers() { + let image = host_kernel32_image(); + let task = + crate::tests::test_task_with_nls_files(&[("/Windows/System32/kernel32.dll", &image)]); + let name = wide(r"\KnownDlls\kernel32.dll"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_section( + mut_ptr(&mut handle), + SectionAccess::QUERY.bits(), + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + + let mut info = ::new_zeroed(); + let mut return_length = 0usize; + assert_eq!( + task.sys_nt_query_section( + handle, + SectionInformationClass::Image as u32, + mut_byte_ptr(&mut info), + size_of::(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + + assert_eq!(return_length, size_of::()); + assert_eq!(info.machine, IMAGE_FILE_MACHINE_AMD64); + assert_eq!(info.subsystem_type, IMAGE_SUBSYSTEM_WINDOWS_CUI); + assert_eq!(info.image_contains_code, 1); + assert_eq!(info.image_file_size, u32::try_from(image.len()).unwrap()); + + let mut too_small = [0xcc; size_of::() - 1]; + let too_small_len = too_small.len(); + return_length = 0x5555_5555; + // Host 25H2 leaves ReturnLength untouched on INFO_LENGTH_MISMATCH + // (Image len=63 -> ret stays sentinel) and writes 0x40 only on success. + assert_eq!( + task.sys_nt_query_section( + handle, + SectionInformationClass::Image as u32, + mut_byte_ptr(&mut too_small), + too_small_len, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(return_length, 0x5555_5555); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn image_section_rejects_writable_view_protection() { + let image = host_kernel32_image(); + let task = + crate::tests::test_task_with_nls_files(&[("/Windows/System32/kernel32.dll", &image)]); + let name = wide(r"\KnownDlls\kernel32.dll"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_section( + mut_ptr(&mut handle), + SectionAccess::ALL_ACCESS.bits(), + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + + let mut base = 0usize; + let mut view_size = 0usize; + // Host 25H2 maps SEC_IMAGE with PAGE_READWRITE successfully + // (NtMapViewOfSection returns STATUS_IMAGE_NOT_AT_BASE). LiteBox rejects writable image + // views until image mappings are backed by real shared image pages. + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: handle, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_READWRITE.bits(), + }), + NtStatus::SECTION_PROTECTION + ); + assert_eq!(base, 0); + assert_eq!(view_size, 0); + + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: handle, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_EXECUTE_READ.bits(), + }), + NtStatus::SUCCESS + ); + assert_ne!(base, 0); + assert_ne!(view_size, 0); + } + + #[test] + fn section_output_handles_follow_host_probe_contracts() { + let task = test_task(); + let name = wide(r"\KnownDlls\DefinitelyMissingLiteBoxProbe.dll"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let mut open_handle = Handle::from_raw(0x1111_2222); + assert_eq!( + task.sys_nt_open_section( + mut_ptr(&mut open_handle), + SectionAccess::QUERY.bits(), + Some(const_ptr(&attrs)), + ), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!(open_handle, Handle::default()); + + let mut create_handle = Handle::from_raw(0x3333_4444); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut create_handle), + SectionAccess::ALL_ACCESS.bits(), + None, + None, + PageProtection::PAGE_READWRITE.bits(), + SectionAllocationAttributes::SEC_COMMIT.bits(), + Handle::default(), + ), + NtStatus::INVALID_PARAMETER_4 + ); + assert_eq!(create_handle, Handle::from_raw(0x3333_4444)); + } + + #[test] + fn nt_map_view_of_section_maps_writable_pagefile_section() { + let task = test_task(); + let handle = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2000, + PageProtection::PAGE_READWRITE, + ); + let (base, view_size) = map_pagefile_section(&task, handle); + assert_ne!(base, 0); + assert_eq!(view_size, 0x2000); + + let mapped = MutPtr::::from_usize(base); + assert_eq!(mapped.read_at_offset(0), Some(0)); + assert!(mapped.write_at_offset(0, 0xfeed_cafe).is_some()); + assert_eq!(mapped.read_at_offset(0), Some(0xfeed_cafe)); + + assert_eq!( + task.sys_nt_unmap_view_of_section(ProcessHandle::CURRENT, base + 0x100), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_unmap_view_of_section(ProcessHandle::CURRENT, base), + NtStatus::NOT_MAPPED_VIEW + ); + } + + #[test] + fn pagefile_map_rejects_protection_incompatible_with_section_protection() { + let task = test_task(); + let readonly = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2000, + PageProtection::PAGE_READONLY, + ); + let execute = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2000, + PageProtection::PAGE_EXECUTE, + ); + let readwrite = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2000, + PageProtection::PAGE_READWRITE, + ); + + for (handle, page_protection) in [ + (readonly, PageProtection::PAGE_READWRITE), + (execute, PageProtection::PAGE_READONLY), + (readwrite, PageProtection::PAGE_EXECUTE_READ), + ] { + let mut base = 0usize; + let mut view_size = 0usize; + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: handle, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: page_protection.bits(), + }), + NtStatus::SECTION_PROTECTION + ); + assert_eq!(base, 0); + assert_eq!(view_size, 0); + } + } + + #[test] + fn pagefile_map_accepts_compatible_noaccess_and_copy_protections() { + let task = test_task(); + // Host 25H2 and ReactOS allow PAGE_WRITECOPY and PAGE_NOACCESS views of + // a PAGE_READONLY pagefile section. + for page_protection in [ + PageProtection::PAGE_WRITECOPY, + PageProtection::PAGE_NOACCESS, + ] { + let readonly = create_pagefile_section( + &task, + SectionAccess::ALL_ACCESS.bits(), + 0x2000, + PageProtection::PAGE_READONLY, + ); + let mut base = 0usize; + let mut view_size = 0usize; + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: readonly, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: page_protection.bits(), + }), + NtStatus::SUCCESS + ); + assert_ne!(base, 0); + assert_eq!(view_size, 0x2000); + assert_eq!( + task.sys_nt_unmap_view_of_section(ProcessHandle::CURRENT, base), + NtStatus::SUCCESS + ); + } + } + + #[test] + fn pagefile_noaccess_view_requires_map_read_access() { + let task = test_task(); + let handle = create_pagefile_section( + &task, + SectionAccess::QUERY.bits(), + 0x2000, + PageProtection::PAGE_READWRITE, + ); + let mut base = 0usize; + let mut view_size = 0usize; + + // Host 25H2 returns STATUS_ACCESS_DENIED for PAGE_NOACCESS maps unless + // the section handle has SECTION_MAP_READ. + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: handle, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut view_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_NOACCESS.bits(), + }), + NtStatus::ACCESS_DENIED + ); + assert_eq!(base, 0); + assert_eq!(view_size, 0); + } + + #[test] + fn pagefile_section_rejects_additional_views_across_handles_and_unmap() { + let task = test_task(); + let name = wide(r"\BaseNamedObjects\LiteBoxSingleViewSection"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let size = 0x2000i64; + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut handle), + SectionAccess::ALL_ACCESS.bits(), + Some(const_ptr(&attrs)), + Some(const_ptr(&size)), + PageProtection::PAGE_READWRITE.bits(), + SectionAllocationAttributes::SEC_COMMIT.bits(), + Handle::default(), + ), + NtStatus::SUCCESS + ); + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_open_section( + mut_ptr(&mut opened), + SectionAccess::ALL_ACCESS.bits(), + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + let (first_base, first_size) = map_pagefile_section(&task, handle); + assert_eq!(first_size, 0x2000); + + let mut second_base = 0usize; + let mut second_size = 0usize; + // Host 25H2 permits this second simultaneous pagefile view (STATUS_SUCCESS). LiteBox + // deliberately returns STATUS_NOT_SUPPORTED until shared anonymous backing exists. + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: opened, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut second_base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut second_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_READWRITE.bits(), + }), + NtStatus::NOT_SUPPORTED + ); + assert_eq!(second_base, 0); + assert_eq!(second_size, 0); + + assert_eq!( + task.sys_nt_unmap_view_of_section(ProcessHandle::CURRENT, first_base), + NtStatus::SUCCESS + ); + second_base = 0; + second_size = 0; + assert_eq!( + task.sys_nt_map_view_of_section(MapViewOfSectionParameters { + section_handle: opened, + process_handle: ProcessHandle::CURRENT, + base_address: mut_ptr(&mut second_base), + zero_bits: 0, + commit_size: 0, + section_offset: None, + view_size: mut_ptr(&mut second_size), + inherit_disposition: VIEW_SHARE, + allocation_type: 0, + page_protection: PageProtection::PAGE_READWRITE.bits(), + }), + NtStatus::NOT_SUPPORTED + ); + assert_eq!(second_base, 0); + assert_eq!(second_size, 0); + } + + #[test] + fn nt_open_section_opens_existing_named_pagefile_section() { + let task = test_task(); + let name = wide(r"\BaseNamedObjects\LiteBoxNamedSection"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let size = 0x1000i64; + let mut created = Handle::default(); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut created), + SectionAccess::ALL_ACCESS.bits(), + Some(const_ptr(&attrs)), + Some(const_ptr(&size)), + PageProtection::PAGE_READWRITE.bits(), + SectionAllocationAttributes::SEC_COMMIT.bits(), + Handle::default(), + ), + NtStatus::SUCCESS + ); + + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_open_section( + mut_ptr(&mut opened), + SectionAccess::QUERY.bits(), + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + assert_ne!(opened, Handle::default()); + assert_ne!(opened, created); + } + + #[test] + fn event_and_section_names_collide_in_object_namespace() { + let task = test_task(); + let name = wide(r"\BaseNamedObjects\LiteBoxSharedLeafName"); + let unicode = unicode(&name); + let attrs = object_attributes(&unicode); + let mut event = Handle::default(); + assert_eq!( + task.sys_nt_create_event( + mut_ptr(&mut event), + 0x001f_0003, + Some(const_ptr(&attrs)), + EventType::Notification as u32, + 0, + ), + NtStatus::SUCCESS + ); + + let size = 0x1000i64; + let mut section = Handle::from_raw(0xffff_ffff); + assert_eq!( + task.sys_nt_create_section( + mut_ptr(&mut section), + SectionAccess::ALL_ACCESS.bits(), + Some(const_ptr(&attrs)), + Some(const_ptr(&size)), + PageProtection::PAGE_READWRITE.bits(), + SectionAllocationAttributes::SEC_COMMIT.bits(), + Handle::default(), + ), + NtStatus::OBJECT_NAME_EXISTS + ); + assert_eq!(section, Handle::from_raw(0xffff_ffff)); + } +} diff --git a/litebox_shim_windows/src/syscalls/symlink.rs b/litebox_shim_windows/src/syscalls/symlink.rs new file mode 100644 index 0000000000..38b86d9755 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/symlink.rs @@ -0,0 +1,881 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT object-manager symbolic-link syscalls. + +use alloc::string::String; +use alloc::sync::Arc; +use alloc::vec::Vec; +use core::marker::PhantomData; +use core::mem::size_of; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::{AccessMask, ObjectAttributes, ObjectAttributesFlags, UnicodeString}; +use crate::syscalls::Handle; +use crate::syscalls::object_manager::{DirectoryName, ObjectNode}; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +const STANDARD_RIGHTS_REQUIRED: u32 = AccessMask::DELETE.bits() + | AccessMask::READ_CONTROL.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits(); + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct SymbolicLinkAccess: u32 { + const QUERY = 0x0001; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() | Self::QUERY.bits(); + const ALL_ACCESS = STANDARD_RIGHTS_REQUIRED | Self::QUERY.bits(); + + const _ = !0; + } +} + +impl SymbolicLinkAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +pub(crate) struct SymbolicLinkSubsystem(PhantomData); + +impl FdEnabledSubsystem for SymbolicLinkSubsystem { + type Entry = SymbolicLinkHandleObject; +} + +impl FdEnabledSubsystemEntry for SymbolicLinkHandleObject {} + +impl crate::WindowsHandleSubsystem + for SymbolicLinkSubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + SymbolicLinkAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct SymbolicLinkHandleObject { + link: Arc>, +} + +fn utf16_units(value: &str) -> Result, NtStatus> { + let units: Vec = value.encode_utf16().collect(); + if units + .len() + .checked_mul(size_of::()) + .is_none_or(|len| len > u16::MAX as usize) + { + return Err(NtStatus::NAME_TOO_LONG); + } + Ok(units) +} + +impl Task { + fn insert_symbolic_link_handle( + &self, + link: Arc>, + granted_access: SymbolicLinkAccess, + ) -> Result { + self.insert_typed_handle::>( + SymbolicLinkHandleObject { link }, + granted_access.bits(), + drop, + ) + } + + fn close_symbolic_link_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_symbolic_link(link: SymbolicLinkHandleObject) { + drop(link); + } + + pub(crate) fn sys_nt_create_symbolic_link_object( + &self, + link_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + link_target: ConstPtr, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(link_handle) { + return status; + } + let (object_attributes, link_name) = + match self.read_directory_object_attributes(object_attributes, true) { + Ok((Some(object_attributes), Some(link_name))) => (object_attributes, link_name), + Ok((_, None)) => return NtStatus::OBJECT_NAME_INVALID, + Ok((None, Some(_))) => return NtStatus::INVALID_PARAMETER, + Err(status) => return status, + }; + let target = match link_target.read_at_offset(0) { + Some(target) => match read_symbolic_link_target::(target) { + Ok(target) => target, + Err(status) => return status, + }, + None => return NtStatus::ACCESS_VIOLATION, + }; + let attributes = ObjectAttributesFlags::from_bits_retain(object_attributes.attributes); + if attributes.contains(ObjectAttributesFlags::OPENLINK) { + return NtStatus::INVALID_PARAMETER; + } + + // NT stores the target as an opaque string at creation time. Wine's + // create_symlink only copies the target and ReactOS leaves LinkTargetObject + // null; both defer namespace lookup until the link is traversed. + let granted_access = SymbolicLinkAccess::from_desired_access(desired_access); + self.create_symbolic_link( + link_handle, + granted_access, + link_name, + target, + attributes.contains(ObjectAttributesFlags::OPENIF), + ) + } + + fn create_symbolic_link( + &self, + link_handle: MutPtr, + granted_access: SymbolicLinkAccess, + link_name: DirectoryName, + target: String, + open_if: bool, + ) -> NtStatus { + self.process.object_manager.create_symlink( + &link_name.original_path, + target, + |link| { + if !open_if { + return NtStatus::OBJECT_NAME_COLLISION; + } + let Ok(handle) = self.insert_symbolic_link_handle(link, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if link_handle.write_at_offset(0, handle).is_none() { + self.close_symbolic_link_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::OBJECT_NAME_EXISTS + }, + |link| { + let Ok(handle) = self.insert_symbolic_link_handle(link, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if link_handle.write_at_offset(0, handle).is_none() { + self.close_symbolic_link_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + }, + ) + } + + pub(crate) fn sys_nt_open_symbolic_link_object( + &self, + link_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(link_handle) { + return status; + } + let link_name = match self.read_directory_object_attributes(object_attributes, true) { + Ok((Some(_), Some(link_name))) => link_name, + Ok((_, None)) => return NtStatus::OBJECT_NAME_INVALID, + Ok((None, Some(_))) => return NtStatus::INVALID_PARAMETER, + Err(status) => return status, + }; + let link = match self + .process + .object_manager + .resolve_symlink(&link_name.original_path, false) + { + Ok(link) => link, + Err(status) => return status, + }; + let Ok(handle) = self.insert_symbolic_link_handle( + link, + SymbolicLinkAccess::from_desired_access(desired_access), + ) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if link_handle.write_at_offset(0, handle).is_none() { + self.close_symbolic_link_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_symbolic_link_object( + &self, + link_handle: Handle, + link_target: MutPtr, + returned_length: Option>, + ) -> NtStatus { + let entry = match self.typed_handle_entry_with_access::>( + link_handle, + SymbolicLinkAccess::QUERY.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + if let Err(status) = probe_guest_output_preserving_value::(link_target) { + return status; + } + if let Some(returned_length) = returned_length + && let Err(status) = probe_guest_output_preserving_value::(returned_length) + { + return status; + } + + let target = match entry.with_entry(|entry| entry.link.symlink_target().into_result()) { + Ok(target) => target, + Err(status) => return status, + }; + let units = match utf16_units(&target) { + Ok(units) => units, + Err(status) => return status, + }; + let required_len = units + .len() + .checked_mul(size_of::()) + .ok_or(NtStatus::NAME_TOO_LONG); + let Ok(required_len) = required_len else { + return NtStatus::NAME_TOO_LONG; + }; + let required = match units + .len() + .checked_add(1) + .and_then(|units| units.checked_mul(size_of::())) + .and_then(|bytes| u32::try_from(bytes).ok()) + { + Some(required) if u16::try_from(required).is_ok() => required, + _ => return NtStatus::NAME_TOO_LONG, + }; + let Some(mut unicode) = link_target.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + + if let Some(returned_length) = returned_length + && returned_length.write_at_offset(0, required).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if required > u32::from(unicode.maximum_length) { + return NtStatus::BUFFER_TOO_SMALL; + } + if required != 0 && unicode.buffer == 0 { + return NtStatus::ACCESS_VIOLATION; + } + + let mut output_units = units; + output_units.push(0); + let target_buffer = MutPtr::::from_usize(unicode.buffer); + target_buffer + .write_slice_at_offset(0, &output_units) + .ok_or(NtStatus::ACCESS_VIOLATION) + .map_or_else( + |status| status, + |()| { + unicode.length = required_len.trunc(); + if link_target.write_at_offset(0, unicode).is_none() { + NtStatus::ACCESS_VIOLATION + } else { + NtStatus::SUCCESS + } + }, + ) + } +} + +fn read_symbolic_link_target( + target: UnicodeString, +) -> Result { + // ReactOS rounds odd MaximumLength down before validating this UNICODE_STRING; + // Wine's object-manager tests cover the zero MaximumLength rejection. + let maximum_length = target.maximum_length & !1u16; + if !target.length.is_multiple_of(2) || maximum_length < target.length || maximum_length == 0 { + return Err(NtStatus::INVALID_PARAMETER); + } + + let target = target.read_string::()?; + if target.is_empty() { + return Err(NtStatus::INVALID_PARAMETER); + } + Ok(target) +} + +#[cfg(test)] +mod tests { + use core::mem::size_of_val; + + use litebox::platform::ThreadProvider; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::{ObjectAttributes, ObjectAttributesFlags, UnicodeString}; + use crate::tests::{ + TestPlatform, const_ptr, mut_ptr, object_attributes, test_task, unicode_string, + utf16_units as test_utf16_units, + }; + + const SYMBOLIC_LINK_QUERY: u32 = 0x0000_0001; + const SYMBOLIC_LINK_ALL_ACCESS: u32 = 0x000f_0001; + const DIRECTORY_QUERY: u32 = 0x0000_0001; + const DIRECTORY_ALL_ACCESS: u32 = 0x000f_000f; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + fn link_target(value: &str) -> (Vec, UnicodeString) { + let units = test_utf16_units(value); + let unicode = unicode_string(&units); + (units, unicode) + } + + fn create_link( + task: &Task, + path: &str, + target: &str, + ) -> Handle { + let path_units = test_utf16_units(path); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let (_target_units, target) = link_target(target); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::SUCCESS + ); + handle + } + + fn create_directory(task: &Task, path: &str) -> Handle { + let path_units = test_utf16_units(path); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut handle), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + handle + } + + fn open_directory(task: &Task, path: &str) -> Handle { + let path_units = test_utf16_units(path); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_directory_object( + mut_ptr(&mut handle), + DIRECTORY_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + handle + } + + fn open_link(task: &Task, path: &str) -> Handle { + let path_units = test_utf16_units(path); + let name = unicode_string(&path_units); + let attrs = object_attributes( + &name, + (ObjectAttributesFlags::CASE_INSENSITIVE | ObjectAttributesFlags::OPENLINK).bits(), + ); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + handle + } + + fn open_link_without_openlink( + task: &Task, + path: &str, + ) -> Handle { + let path_units = test_utf16_units(path); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::SUCCESS + ); + handle + } + + fn query_link( + task: &Task, + handle: Handle, + output_units: &mut [u16], + ) -> (UnicodeString, u32) { + let mut target = UnicodeString { + length: u16::MAX, + maximum_length: size_of_val(output_units).trunc(), + padding_0: [0; 4], + buffer: output_units.as_mut_ptr() as usize, + }; + let original_buffer = target.buffer; + let original_maximum_length = target.maximum_length; + let mut returned_length = u32::MAX; + assert_eq!( + task.sys_nt_query_symbolic_link_object( + handle, + mut_ptr(&mut target), + Some(mut_ptr(&mut returned_length)), + ), + NtStatus::SUCCESS + ); + assert_eq!(target.buffer, original_buffer); + assert_eq!(target.maximum_length, original_maximum_length); + assert!(target.length as usize <= size_of_val(output_units)); + assert_eq!(output_units[target.length as usize / 2], 0); + (target, returned_length) + } + + #[test] + fn create_open_and_query_symbolic_link_round_trips_target() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let target = r"\BaseNamedObjects\LiteBoxTarget"; + let created = create_link(&task, r"\BaseNamedObjects\LiteBoxSymlink", target); + let opened = open_link(&task, r"\BaseNamedObjects\LiteBoxSymlink"); + let mut output = alloc::vec![0u16; target.encode_utf16().count() + 1]; + let (target, returned_length) = query_link(&task, opened, &mut output); + assert_eq!(returned_length, u32::from(target.length) + 2); + assert_eq!( + String::from_utf16_lossy(&output[..target.length as usize / 2]), + r"\BaseNamedObjects\LiteBoxTarget" + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(created), NtStatus::SUCCESS); + }); + } + + #[test] + fn predefined_known_dll_path_symbolic_link_matches_loader_contract() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let opened = open_link(&task, r"\KnownDlls\KnownDllPath"); + let target = r"C:\Windows\System32"; + let mut output = alloc::vec![0u16; target.encode_utf16().count() + 1]; + let (target_string, returned_length) = query_link(&task, opened, &mut output); + + assert_eq!(returned_length, u32::from(target_string.length) + 2); + assert_eq!( + String::from_utf16_lossy(&output[..target_string.length as usize / 2]), + target + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_symbolic_link_rejects_empty_target() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxEmptyTarget"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let empty_target = unicode_string(&[]); + let mut handle = Handle::default(); + + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&empty_target), + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::default()); + }); + } + + #[test] + fn create_symbolic_link_rejects_zero_target_maximum_length() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxZeroTargetMax"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let (_target_units, mut target) = link_target(r"\BaseNamedObjects\Target"); + let mut handle = Handle::default(); + target.maximum_length = 0; + + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::default()); + }); + } + + #[test] + fn create_symbolic_link_rejects_target_maximum_length_shorter_than_length() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxShortTargetMax"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let (_target_units, mut target) = link_target(r"\BaseNamedObjects\Target"); + let mut handle = Handle::default(); + target.maximum_length = target.length - 2; + + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::default()); + }); + } + + #[test] + fn create_symbolic_link_allows_odd_target_maximum_length_after_rounding() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxOddTargetMax"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let (_target_units, mut target) = link_target(r"\BaseNamedObjects\Target"); + let mut handle = Handle::default(); + target.maximum_length = target.length + 1; + + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::SUCCESS + ); + assert_ne!(handle, Handle::default()); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn open_symbolic_link_without_openlink_returns_final_link_itself() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let created = create_link( + &task, + r"\BaseNamedObjects\LiteBoxNoOpenLinkFinal", + r"\BaseNamedObjects\MissingTarget", + ); + let opened = + open_link_without_openlink(&task, r"\BaseNamedObjects\LiteBoxNoOpenLinkFinal"); + let mut output = [0u16; 64]; + let (target, _) = query_link(&task, opened, &mut output); + + assert_eq!( + String::from_utf16_lossy(&output[..target.length as usize / 2]), + r"\BaseNamedObjects\MissingTarget" + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(created), NtStatus::SUCCESS); + }); + } + + #[test] + fn open_symbolic_link_openlink_returns_final_link_itself() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let created = create_link( + &task, + r"\BaseNamedObjects\LiteBoxOpenLinkFinal", + r"\BaseNamedObjects\MissingTarget", + ); + let opened = open_link(&task, r"\BaseNamedObjects\LiteBoxOpenLinkFinal"); + let mut output = [0u16; 64]; + let (target, _) = query_link(&task, opened, &mut output); + + assert_eq!( + String::from_utf16_lossy(&output[..target.length as usize / 2]), + r"\BaseNamedObjects\MissingTarget" + ); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(created), NtStatus::SUCCESS); + }); + } + + #[test] + fn directory_open_follows_intermediate_symbolic_link() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let real = create_directory(&task, r"\BaseNamedObjects\LiteBoxRealDir"); + let child = create_directory(&task, r"\BaseNamedObjects\LiteBoxRealDir\Child"); + let link = create_link( + &task, + r"\BaseNamedObjects\LiteBoxDirLink", + r"\BaseNamedObjects\LiteBoxRealDir", + ); + + let opened = open_directory(&task, r"\BaseNamedObjects\LiteBoxDirLink\Child"); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(link), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(child), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(real), NtStatus::SUCCESS); + }); + } + + #[test] + fn directory_create_follows_symlinked_parent() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let real = create_directory(&task, r"\BaseNamedObjects\LiteBoxCreateRealDir"); + let link = create_link( + &task, + r"\BaseNamedObjects\LiteBoxCreateDirLink", + r"\BaseNamedObjects\LiteBoxCreateRealDir", + ); + let created = create_directory(&task, r"\BaseNamedObjects\LiteBoxCreateDirLink\Child"); + let opened = open_directory(&task, r"\BaseNamedObjects\LiteBoxCreateRealDir\Child"); + + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(created), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(link), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(real), NtStatus::SUCCESS); + }); + } + + #[test] + fn dos_device_style_symbolic_link_resolves_through_seeded_directory() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let real = create_directory(&task, r"\BaseNamedObjects\LiteBoxDriveTarget"); + let child = create_directory(&task, r"\BaseNamedObjects\LiteBoxDriveTarget\Child"); + let link = create_link(&task, r"\??\Z:", r"\BaseNamedObjects\LiteBoxDriveTarget"); + + let opened = open_directory(&task, r"\??\Z:\Child"); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(link), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(child), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(real), NtStatus::SUCCESS); + }); + } + + #[test] + fn open_symbolic_link_rejects_directory_type() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut handle = Handle::default(); + + assert_eq!( + task.sys_nt_open_symbolic_link_object( + mut_ptr(&mut handle), + SYMBOLIC_LINK_QUERY, + Some(const_ptr(&attrs)), + ), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(handle, Handle::default()); + }); + } + + #[test] + fn create_symbolic_link_obeys_collision_and_openif() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let first = create_link( + &task, + r"\BaseNamedObjects\LiteBoxOpenIfSymlink", + r"\BaseNamedObjects\Target", + ); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxOpenIfSymlink"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let (_target_units, target) = link_target(r"\BaseNamedObjects\Target"); + let mut collision = Handle::default(); + + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut collision), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::OBJECT_NAME_COLLISION + ); + assert_eq!(collision, Handle::default()); + + let openif_attrs = ObjectAttributes { + attributes: (ObjectAttributesFlags::CASE_INSENSITIVE + | ObjectAttributesFlags::OPENIF) + .bits(), + ..attrs + }; + let mut opened = Handle::default(); + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut opened), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&openif_attrs)), + const_ptr(&target), + ), + NtStatus::OBJECT_NAME_EXISTS + ); + assert_ne!(opened, Handle::default()); + assert_eq!(task.sys_nt_close(opened), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(first), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_symbolic_link_rejects_existing_directory_type() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let path_units = test_utf16_units(r"\BaseNamedObjects\LiteBoxSymlinkTypeDirectory"); + let name = unicode_string(&path_units); + let attrs = object_attributes(&name, ObjectAttributesFlags::CASE_INSENSITIVE.bits()); + let mut directory = Handle::default(); + assert_eq!( + task.sys_nt_create_directory_object( + mut_ptr(&mut directory), + DIRECTORY_ALL_ACCESS, + Some(const_ptr(&attrs)), + Handle::default(), + 0, + ), + NtStatus::SUCCESS + ); + + let (_target_units, target) = link_target(r"\BaseNamedObjects\Target"); + let mut link = Handle::default(); + assert_eq!( + task.sys_nt_create_symbolic_link_object( + mut_ptr(&mut link), + SYMBOLIC_LINK_ALL_ACCESS, + Some(const_ptr(&attrs)), + const_ptr(&target), + ), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(link, Handle::default()); + assert_eq!(task.sys_nt_close(directory), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_symbolic_link_reports_too_small_without_mutating_output() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let handle = create_link( + &task, + r"\BaseNamedObjects\LiteBoxSmallSymlink", + r"\BaseNamedObjects\LongTarget", + ); + let mut output = [0xeeeeu16; 2]; + let mut target = UnicodeString { + length: 0x1234, + maximum_length: size_of_val(&output).trunc(), + padding_0: [0; 4], + buffer: output.as_mut_ptr() as usize, + }; + let original = target; + let mut returned_length = 0; + + assert_eq!( + task.sys_nt_query_symbolic_link_object( + handle, + mut_ptr(&mut target), + Some(mut_ptr(&mut returned_length)), + ), + NtStatus::BUFFER_TOO_SMALL + ); + assert_eq!(target.length, original.length); + assert_eq!(target.maximum_length, original.maximum_length); + assert_eq!(target.buffer, original.buffer); + assert_eq!(output, [0xeeeeu16; 2]); + assert_eq!( + returned_length, + ((r"\BaseNamedObjects\LongTarget".encode_utf16().count() + 1) * 2).trunc() + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn query_symbolic_link_requires_space_for_trailing_nul() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let target = r"\BaseNamedObjects\ExactLengthTarget"; + let handle = create_link(&task, r"\BaseNamedObjects\LiteBoxExactSymlink", target); + let mut output = alloc::vec![0xeeeeu16; target.encode_utf16().count()]; + let mut target_string = UnicodeString { + length: 0x1234, + maximum_length: size_of_val(output.as_slice()).trunc(), + padding_0: [0; 4], + buffer: output.as_mut_ptr() as usize, + }; + let mut returned_length = 0; + + assert_eq!( + task.sys_nt_query_symbolic_link_object( + handle, + mut_ptr(&mut target_string), + Some(mut_ptr(&mut returned_length)), + ), + NtStatus::BUFFER_TOO_SMALL + ); + assert_eq!( + returned_length, + ((target.encode_utf16().count() + 1) * 2).trunc() + ); + assert!(output.iter().all(|unit| *unit == 0xeeee)); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } +} diff --git a/litebox_shim_windows/src/syscalls/sysinfo.rs b/litebox_shim_windows/src/syscalls/sysinfo.rs new file mode 100644 index 0000000000..84f65f8763 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/sysinfo.rs @@ -0,0 +1,1690 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::platform::{ + Instant as _, PageManagementProvider, RawConstPointer as _, RawMutPointer as _, +}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::GroupAffinity; +use crate::syscalls::mm::ALLOCATION_GRANULARITY; +use crate::{ConstPtr, MutPtr, PAGE_SIZE, ShimFS, ShimPlatform, Task}; + +const QPC_FREQUENCY_HZ: i64 = 1_000_000_000; +// These fixed values are deterministic sandbox answers: a default 15.625 ms +// timer tick, one synthetic processor, and a stable 4 GiB physical-memory view. +// They avoid leaking host topology while satisfying Windows CRT/environment +// probes that require plausible system-information success outputs. +const TIMER_RESOLUTION_100NS: u32 = 156_250; +const DEFAULT_PHYSICAL_PAGES: u32 = 1024 * 1024; +const NUMBER_OF_PROCESSORS: u8 = 1; +const PROCESSOR_AFFINITY_MASK: usize = (1usize << NUMBER_OF_PROCESSORS) - 1; +// SystemFlushInformation values observed from host ntdll on Windows 11 24H2. +// LiteBox keeps them fixed because they describe the synthetic CPU contract. +const SUPPORTED_FLUSH_METHODS: u32 = 0x7; +const SUPPORTED_FLUSH_PROCESSOR_FEATURES: u32 = 0x40; +const CACHE_UNIFIED: u32 = 0; +const DWORD_SIZE_U32: u32 = 4; +const NUMA_NODE_COUNT: usize = NUMBER_OF_PROCESSORS as usize; +const PROCESSOR_ARCHITECTURE_AMD64: u16 = 9; +const SYSTEM_VERIFIER_INFORMATION_LENGTH: u32 = 0x90; +const SYSTEM_VERIFIER_INFORMATION_LENGTH_USIZE: usize = 0x90; +const X64_SYSTEM_RANGE_START: usize = 0xffff_8000_0000_0000; + +pub(crate) const WINDOWS_TIME_ZONE_ID_INVALID: u32 = u32::MAX; +pub(crate) const WINDOWS_OS_MAJOR_VERSION: u16 = 10; +pub(crate) const WINDOWS_OS_MINOR_VERSION: u16 = 0; +pub(crate) const WINDOWS_OS_BUILD_NUMBER: u16 = 19041; +pub(crate) const WINDOWS_OS_PLATFORM_WIN32_NT: u32 = 2; +#[cfg(not(target_os = "windows"))] +pub(crate) const WINDOWS_NT_PRODUCT_WORKSTATION: u32 = 1; +pub(crate) const WINDOWS_DIRECTORY: &str = r"C:\Windows"; +pub(crate) const WINDOWS_SYSTEM_DIRECTORY: &str = r"C:\Windows\System32"; +pub(crate) const WINDOWS_NAMED_OBJECT_DIRECTORY: &str = r"\BaseNamedObjects"; + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum SystemInformationClass { + Basic = 0, + Processor = 1, + RangeStart = 50, + Verifier = 51, + NumaProcessorMap = 55, + EmulationBasic = 62, + LogicalProcessorAndGroup = 107, + Flush = 192, + HypervisorSharedPage = 197, + FeatureConfigurationSection = 211, + ProcessorFeaturesBitMap = 250, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum LogicalProcessorRelationship { + ProcessorCore = 0, + NumaNode = 1, + Cache = 2, + ProcessorPackage = 3, + Group = 4, + All = 0xffff, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemBasicInformation { + reserved: u32, + timer_resolution: u32, + page_size: u32, + number_of_physical_pages: u32, + lowest_physical_page_number: u32, + highest_physical_page_number: u32, + allocation_granularity: u32, + _padding0: u32, + minimum_user_mode_address: usize, + maximum_user_mode_address: usize, + active_processors_affinity_mask: usize, + number_of_processors: u8, + _padding1: [u8; 7], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemProcessorInformation { + processor_architecture: u16, + processor_level: u16, + processor_revision: u16, + maximum_processors: u16, + processor_feature_bits: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemNumaInformation { + highest_node_number: u32, + reserved: u32, + active_processors_group_affinity: [GroupAffinity; NUMA_NODE_COUNT], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemFlushInformation { + supported_flush_methods: u32, + processor_features: u32, + reserved: [u32; 6], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemHypervisorSharedPageInformation { + hypervisor_shared_user_va: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemRangeStartInformation { + system_range_start: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct SystemProcessorFeaturesBitMapInformation { + feature_bits: [u64; 2], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessorRelationship { + flags: u8, + efficiency_class: u8, + reserved: [u8; 20], + group_count: u16, + group_mask: [GroupAffinity; 1], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct NumaNodeRelationship { + node_number: u32, + reserved: [u8; 18], + group_count: u16, + group_mask: GroupAffinity, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct CacheRelationship { + level: u8, + associativity: u8, + line_size: u16, + cache_size: u32, + cache_type: u32, + reserved: [u8; 18], + group_count: u16, + group_mask: GroupAffinity, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessorGroupInfo { + maximum_processor_count: u8, + active_processor_count: u8, + reserved: [u8; 38], + active_processor_mask: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct GroupRelationship { + maximum_group_count: u16, + active_group_count: u16, + reserved: [u8; 20], + group_info: [ProcessorGroupInfo; 1], +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct ProcessorRelationshipInformation { + relationship: u32, + size: u32, + processor: ProcessorRelationship, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct NumaNodeRelationshipInformation { + relationship: u32, + size: u32, + numa_node: NumaNodeRelationship, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct CacheRelationshipInformation { + relationship: u32, + size: u32, + cache: CacheRelationship, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct GroupRelationshipInformation { + relationship: u32, + size: u32, + group: GroupRelationship, +} + +impl Task { + pub(crate) fn sys_nt_query_system_information( + system_information_class: u32, + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + ) -> NtStatus { + let Ok(system_information_class) = + SystemInformationClass::try_from(system_information_class) + else { + litebox_util_log::debug!( + system_information_class = system_information_class; + "Unsupported NtQuerySystemInformation class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match system_information_class { + SystemInformationClass::Basic | SystemInformationClass::EmulationBasic => { + Self::write_exact_system_information( + system_information, + system_information_length, + return_length, + &system_basic_information::(), + ) + } + SystemInformationClass::Processor => Self::write_system_information( + system_information, + system_information_length, + return_length, + &system_processor_information(), + ), + SystemInformationClass::RangeStart => Self::write_exact_system_information( + system_information, + system_information_length, + return_length, + &SystemRangeStartInformation { + system_range_start: X64_SYSTEM_RANGE_START, + }, + ), + SystemInformationClass::Verifier => Self::write_system_verifier_information( + system_information, + system_information_length, + return_length, + ), + SystemInformationClass::NumaProcessorMap => Self::write_numa_processor_map_information( + system_information, + system_information_length, + return_length, + ), + SystemInformationClass::Flush => Self::write_system_information( + system_information, + system_information_length, + return_length, + &system_flush_information(), + ), + SystemInformationClass::HypervisorSharedPage => Self::write_system_information( + system_information, + system_information_length, + return_length, + &SystemHypervisorSharedPageInformation { + hypervisor_shared_user_va: 0, + }, + ), + SystemInformationClass::ProcessorFeaturesBitMap => Self::write_system_information( + system_information, + system_information_length, + return_length, + &SystemProcessorFeaturesBitMapInformation { + feature_bits: [0; 2], + }, + ), + SystemInformationClass::LogicalProcessorAndGroup + | SystemInformationClass::FeatureConfigurationSection => NtStatus::INVALID_INFO_CLASS, + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + system_information_class:? = system_information_class, + system_information_length = system_information_length; + "Handled NtQuerySystemInformation syscall" + ); + } + + status + } + + pub(crate) fn sys_nt_query_system_information_ex( + system_information_class: u32, + input_buffer: Option>, + input_buffer_length: u32, + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + ) -> NtStatus { + if input_buffer_length < DWORD_SIZE_U32 { + return NtStatus::INVALID_PARAMETER; + } + let Some(input_buffer) = input_buffer else { + return NtStatus::INVALID_PARAMETER; + }; + + let Ok(system_information_class) = + SystemInformationClass::try_from(system_information_class) + else { + litebox_util_log::debug!( + system_information_class = system_information_class; + "Unsupported NtQuerySystemInformationEx class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match system_information_class { + SystemInformationClass::LogicalProcessorAndGroup => { + Self::write_logical_processor_and_group_information( + input_buffer, + system_information, + system_information_length, + return_length, + ) + } + // TODO: Windows returns section handles for this class. LiteBox does not yet model those + // NT section objects, so do not publish a fabricated success payload. + SystemInformationClass::FeatureConfigurationSection => NtStatus::INVALID_INFO_CLASS, + _ => { + litebox_util_log::debug!( + system_information_class:? = system_information_class; + "Unsupported NtQuerySystemInformationEx class" + ); + NtStatus::INVALID_INFO_CLASS + } + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + system_information_class:? = system_information_class, + system_information_length = system_information_length; + "Handled NtQuerySystemInformationEx syscall" + ); + } + + status + } + + fn write_logical_processor_and_group_information( + input_buffer: ConstPtr, + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + ) -> NtStatus { + let input_buffer = ConstPtr::::from_usize(input_buffer.as_usize()); + let Some(relationship) = input_buffer.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + + let Ok(relationship) = LogicalProcessorRelationship::try_from(relationship) else { + return NtStatus::UNSUCCESSFUL; + }; + + match relationship { + LogicalProcessorRelationship::ProcessorCore => Self::write_system_information( + system_information, + system_information_length, + return_length, + &processor_relationship_information(LogicalProcessorRelationship::ProcessorCore), + ), + LogicalProcessorRelationship::NumaNode => Self::write_system_information( + system_information, + system_information_length, + return_length, + &numa_node_relationship_information(), + ), + LogicalProcessorRelationship::Cache => Self::write_system_information( + system_information, + system_information_length, + return_length, + &cache_relationship_information(), + ), + LogicalProcessorRelationship::ProcessorPackage => Self::write_system_information( + system_information, + system_information_length, + return_length, + &processor_relationship_information(LogicalProcessorRelationship::ProcessorPackage), + ), + LogicalProcessorRelationship::Group => Self::write_system_information( + system_information, + system_information_length, + return_length, + &group_relationship_information(), + ), + LogicalProcessorRelationship::All => { + let core = + processor_relationship_information(LogicalProcessorRelationship::ProcessorCore); + let numa = numa_node_relationship_information(); + let cache = cache_relationship_information(); + let package = processor_relationship_information( + LogicalProcessorRelationship::ProcessorPackage, + ); + let group = group_relationship_information(); + let records = [ + core.as_bytes(), + numa.as_bytes(), + cache.as_bytes(), + package.as_bytes(), + group.as_bytes(), + ]; + let required_len = records.iter().try_fold(0u32, |total, record| { + total.checked_add(u32::try_from(record.len()).ok()?) + }); + let Some(required_len) = required_len else { + return NtStatus::INVALID_PARAMETER; + }; + + Self::write_sized_system_information( + system_information, + system_information_length, + return_length, + required_len, + move |system_information| { + let mut offset = 0; + for record in records { + system_information + .write_slice_at_offset(offset, record) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + offset = offset.wrapping_add_unsigned(record.len()); + } + Ok(()) + }, + ) + } + } + } + + fn write_system_information( + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + information: &T, + ) -> NtStatus { + let required_len = size_of::().trunc(); + Self::write_sized_system_information( + system_information, + system_information_length, + return_length, + required_len, + |system_information| { + system_information + .write_slice_at_offset(0, information.as_bytes()) + .ok_or(NtStatus::ACCESS_VIOLATION) + }, + ) + } + + fn write_exact_system_information( + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + information: &T, + ) -> NtStatus { + let required_len = size_of::().trunc(); + if system_information_length != required_len { + return Self::write_return_length_for_short_buffer(return_length, required_len); + } + + Self::write_system_information( + system_information, + system_information_length, + return_length, + information, + ) + } + + fn write_numa_processor_map_information( + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + ) -> NtStatus { + if system_information_length < DWORD_SIZE_U32 { + return Self::write_return_length_for_short_buffer(return_length, DWORD_SIZE_U32); + } + + if system_information_length < size_of::().trunc() { + let highest_node_number = + MutPtr::::from_usize(system_information.as_usize()); + if highest_node_number.write_at_offset(0, 0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if Self::write_return_length(return_length, DWORD_SIZE_U32).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + return NtStatus::SUCCESS; + } + + Self::write_system_information( + system_information, + system_information_length, + return_length, + &system_numa_information(), + ) + } + + fn write_system_verifier_information( + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + ) -> NtStatus { + if system_information_length < SYSTEM_VERIFIER_INFORMATION_LENGTH { + return Self::write_return_length_for_short_buffer( + return_length, + SYSTEM_VERIFIER_INFORMATION_LENGTH, + ); + } + + let verifier_information = [0u8; SYSTEM_VERIFIER_INFORMATION_LENGTH_USIZE]; + if system_information + .write_slice_at_offset(0, &verifier_information) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if Self::write_return_length(return_length, 0).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + fn write_sized_system_information( + system_information: MutPtr, + system_information_length: u32, + return_length: Option>, + required_len: u32, + write_payload: impl FnOnce(MutPtr) -> Result<(), NtStatus>, + ) -> NtStatus { + if system_information_length < required_len { + return Self::write_return_length_for_short_buffer(return_length, required_len); + } + if let Err(status) = write_payload(system_information) { + return status; + } + if Self::write_return_length(return_length, required_len).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::SUCCESS + } + + fn write_return_length_for_short_buffer( + return_length: Option>, + required_len: u32, + ) -> NtStatus { + if Self::write_return_length(return_length, required_len).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + + NtStatus::INFO_LENGTH_MISMATCH + } + + fn write_return_length( + return_length: Option>, + required_len: u32, + ) -> Result<(), NtStatus> { + if let Some(return_length) = return_length + && return_length.write_at_offset(0, required_len).is_none() + { + return Err(NtStatus::ACCESS_VIOLATION); + } + Ok(()) + } + + pub(crate) fn sys_nt_query_performance_counter( + &self, + performance_counter: MutPtr, + performance_frequency: Option>, + ) -> NtStatus { + let elapsed = self + .global + .platform + .now() + .duration_since(&self.global.qpc_boot_instant); + let ticks = duration_as_qpc_ticks(elapsed); + + if performance_counter.write_at_offset(0, ticks).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if let Some(performance_frequency) = performance_frequency + && performance_frequency + .write_at_offset(0, QPC_FREQUENCY_HZ) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + litebox_util_log::debug!( + performance_counter = ticks, + performance_frequency = QPC_FREQUENCY_HZ; + "Handled NtQueryPerformanceCounter syscall" + ); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_convert_between_auxiliary_counter_and_performance_counter( + _flag: u32, + source: ConstPtr, + _destination: MutPtr, + _conversion_error: Option>, + ) -> NtStatus { + if source.as_usize() == 0 { + return NtStatus::ACCESS_VIOLATION; + } + + // Wine reports auxiliary counter conversion as unsupported after validating the source. + NtStatus::NOT_SUPPORTED + } +} + +fn system_basic_information() -> SystemBasicInformation { + let maximum_user_mode_address = + >::TASK_ADDR_MAX.saturating_sub(1); + SystemBasicInformation { + reserved: 0, + timer_resolution: TIMER_RESOLUTION_100NS, + page_size: u32::try_from(PAGE_SIZE).expect("PAGE_SIZE fits in ULONG"), + number_of_physical_pages: DEFAULT_PHYSICAL_PAGES, + lowest_physical_page_number: 0, + highest_physical_page_number: DEFAULT_PHYSICAL_PAGES.saturating_sub(1), + allocation_granularity: ALLOCATION_GRANULARITY.trunc(), + _padding0: 0, + minimum_user_mode_address: >::TASK_ADDR_MIN, + maximum_user_mode_address, + active_processors_affinity_mask: PROCESSOR_AFFINITY_MASK, + number_of_processors: NUMBER_OF_PROCESSORS, + _padding1: [0; 7], + } +} + +fn system_processor_information() -> SystemProcessorInformation { + // TODO: x64 Windows reports AMD64 architecture with family/level 6 for modern + // x86-64 CPUs. The revision and feature bitmap are synthetic. + SystemProcessorInformation { + processor_architecture: PROCESSOR_ARCHITECTURE_AMD64, + processor_level: 6, + processor_revision: 0, + maximum_processors: u16::from(NUMBER_OF_PROCESSORS), + processor_feature_bits: 0, + } +} + +fn system_numa_information() -> SystemNumaInformation { + let mut active_processors_group_affinity = [GroupAffinity { + mask: 0, + group: 0, + reserved: [0; 3], + }; NUMA_NODE_COUNT]; + active_processors_group_affinity[0] = processor_group_affinity(); + + SystemNumaInformation { + highest_node_number: 0, + reserved: 0, + active_processors_group_affinity, + } +} + +fn system_flush_information() -> SystemFlushInformation { + SystemFlushInformation { + supported_flush_methods: SUPPORTED_FLUSH_METHODS, + processor_features: SUPPORTED_FLUSH_PROCESSOR_FEATURES, + reserved: [0; 6], + } +} + +fn processor_group_affinity() -> GroupAffinity { + GroupAffinity { + mask: PROCESSOR_AFFINITY_MASK, + group: 0, + reserved: [0; 3], + } +} + +fn processor_relationship_information( + relationship: LogicalProcessorRelationship, +) -> ProcessorRelationshipInformation { + ProcessorRelationshipInformation { + relationship: relationship as u32, + size: size_of::().trunc(), + processor: ProcessorRelationship { + flags: 0, + efficiency_class: 0, + reserved: [0; 20], + group_count: 1, + group_mask: [processor_group_affinity()], + }, + } +} + +fn numa_node_relationship_information() -> NumaNodeRelationshipInformation { + NumaNodeRelationshipInformation { + relationship: LogicalProcessorRelationship::NumaNode as u32, + size: size_of::().trunc(), + numa_node: NumaNodeRelationship { + node_number: 0, + reserved: [0; 18], + group_count: 1, + group_mask: processor_group_affinity(), + }, + } +} + +fn cache_relationship_information() -> CacheRelationshipInformation { + // Deliberate sandbox topology: one generic L1 unified cache for one synthetic processor. + // The relationship ABI follows WDK winnt.h; the field values avoid leaking host cache details. + CacheRelationshipInformation { + relationship: LogicalProcessorRelationship::Cache as u32, + size: size_of::().trunc(), + cache: CacheRelationship { + level: 1, + associativity: 0xff, + line_size: 64, + cache_size: 32 * 1024, + cache_type: CACHE_UNIFIED, + reserved: [0; 18], + group_count: 1, + group_mask: processor_group_affinity(), + }, + } +} + +fn group_relationship_information() -> GroupRelationshipInformation { + GroupRelationshipInformation { + relationship: LogicalProcessorRelationship::Group as u32, + size: size_of::().trunc(), + group: GroupRelationship { + maximum_group_count: 1, + active_group_count: 1, + reserved: [0; 20], + group_info: [ProcessorGroupInfo { + maximum_processor_count: NUMBER_OF_PROCESSORS, + active_processor_count: NUMBER_OF_PROCESSORS, + reserved: [0; 38], + active_processor_mask: PROCESSOR_AFFINITY_MASK, + }], + }, + } +} + +fn duration_as_qpc_ticks(duration: core::time::Duration) -> i64 { + i64::try_from(duration.as_nanos().min(i64::MAX as u128)).unwrap_or(i64::MAX) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{const_ptr, mut_byte_ptr, mut_ptr, null_const_ptr, null_mut_ptr}; + use core::time::Duration; + use litebox::platform::ThreadProvider; + + extern crate std; + + const QPC_SLEEP_DURATION: Duration = Duration::from_millis(25); + const QPC_SLEEP_TOLERANCE: Duration = Duration::from_millis(15); + + type TestPlatform = crate::tests::TestPlatform; + type TestTask = Task; + + const LOGICAL_PROCESSOR_ALL_INFORMATION_SIZE: usize = + size_of::() * 2 + + size_of::() + + size_of::() + + size_of::(); + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + unsafe extern "system" { + fn NtQuerySystemInformation( + system_information_class: u32, + system_information: *mut core::ffi::c_void, + system_information_length: u32, + return_length: *mut u32, + ) -> i32; + + fn NtQuerySystemInformationEx( + system_information_class: u32, + input_buffer: *const core::ffi::c_void, + input_buffer_length: u32, + system_information: *mut core::ffi::c_void, + system_information_length: u32, + return_length: *mut u32, + ) -> i32; + + fn NtQueryPerformanceCounter(counter: *mut i64, frequency: *mut i64) -> i32; + + fn NtConvertBetweenAuxiliaryCounterAndPerformanceCounter( + flag: u32, + source: *const u64, + destination: *mut u64, + conversion_error: *mut u64, + ) -> i32; + } + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + fn host_status(status: i32) -> NtStatus { + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + fn qpc_delta_nanos(start: i64, end: i64) -> u128 { + assert!(end >= start); + u128::try_from(end - start).unwrap() + } + + fn empty_basic_information() -> SystemBasicInformation { + SystemBasicInformation { + reserved: u32::MAX, + timer_resolution: 0, + page_size: 0, + number_of_physical_pages: 0, + lowest_physical_page_number: 0, + highest_physical_page_number: 0, + allocation_granularity: 0, + _padding0: 0, + minimum_user_mode_address: 0, + maximum_user_mode_address: 0, + active_processors_affinity_mask: 0, + number_of_processors: 0, + _padding1: [0; 7], + } + } + + fn const_byte_ptr(value: &T) -> ConstPtr { + ConstPtr::::from_usize(core::ptr::from_ref(value).cast::() as usize) + } + + #[test] + fn nt_query_system_information_ex_validates_query_input() { + run_with_test_platform_pointers(|| { + let relationship = LogicalProcessorRelationship::All as u32; + let mut output = [0u8; size_of::()]; + let mut return_length = 0; + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + None, + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(return_length, 0); + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(const_byte_ptr(&relationship)), + DWORD_SIZE_U32 - 1, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INVALID_PARAMETER + ); + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(null_const_ptr()), + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[test] + fn nt_query_system_information_ex_rejects_unsupported_classes() { + run_with_test_platform_pointers(|| { + let query = LogicalProcessorRelationship::All as u32; + let mut output = [0u8; size_of::()]; + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::Basic as u32, + Some(const_byte_ptr(&query)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::FeatureConfigurationSection as u32, + Some(const_byte_ptr(&query)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + u32::MAX, + None, + 0, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + None, + ), + NtStatus::INVALID_PARAMETER + ); + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + u32::MAX, + Some(const_byte_ptr(&query)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + u32::try_from(output.len()).unwrap(), + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + }); + } + + #[test] + fn nt_query_system_information_ex_reports_required_logical_processor_length() { + run_with_test_platform_pointers(|| { + let relationship = LogicalProcessorRelationship::All as u32; + let mut output = [0u8; 1]; + let mut return_length = 0; + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(const_byte_ptr(&relationship)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut output), + 0, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!( + return_length, + u32::try_from(LOGICAL_PROCESSOR_ALL_INFORMATION_SIZE).unwrap() + ); + }); + } + + #[test] + fn nt_query_system_information_reports_basic_information() { + run_with_test_platform_pointers(|| { + let mut info = empty_basic_information(); + let mut return_length = 0; + + assert_eq!( + TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + mut_byte_ptr(&mut info), + size_of::().trunc(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + + assert_eq!(return_length, size_of::().trunc()); + assert_eq!(info.page_size, u32::try_from(PAGE_SIZE).unwrap()); + assert_eq!( + info.allocation_granularity, + u32::try_from(ALLOCATION_GRANULARITY).unwrap() + ); + assert_eq!(info.number_of_processors, NUMBER_OF_PROCESSORS); + assert_eq!( + info.minimum_user_mode_address, + >::TASK_ADDR_MIN + ); + assert_eq!( + info.maximum_user_mode_address, + >::TASK_ADDR_MAX - 1 + ); + }); + } + + #[test] + fn nt_query_system_information_validates_class_and_buffer_length() { + run_with_test_platform_pointers(|| { + let mut info = [0u8; size_of::()]; + let mut return_length = 0; + let basic_len: u32 = size_of::().trunc(); + + assert_eq!( + TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + mut_byte_ptr(&mut info), + basic_len - 1, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(return_length, basic_len); + + assert_eq!( + TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + mut_byte_ptr(&mut info), + basic_len + 1, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + assert_eq!(return_length, basic_len); + + assert_eq!( + TestTask::sys_nt_query_system_information( + u32::MAX, + mut_byte_ptr(&mut info), + u32::try_from(info.len()).unwrap(), + None, + ), + NtStatus::INVALID_INFO_CLASS + ); + }); + } + + #[test] + fn nt_query_system_information_reports_partial_numa_processor_map() { + run_with_test_platform_pointers(|| { + let mut highest_node_number = u32::MAX; + let mut return_length = 0; + + assert_eq!( + TestTask::sys_nt_query_system_information( + SystemInformationClass::NumaProcessorMap as u32, + mut_byte_ptr(&mut highest_node_number), + DWORD_SIZE_U32, + Some(mut_ptr(&mut return_length)), + ), + NtStatus::SUCCESS + ); + assert_eq!(highest_node_number, 0); + assert_eq!(return_length, DWORD_SIZE_U32); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_basic_status_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_info = [0u8; size_of::()]; + let mut host_return_length = 0; + let mut guest_info = empty_basic_information(); + let mut guest_return_length = 0; + let information_length = size_of::().trunc(); + + // SAFETY: The output buffer and return-length pointer are valid locals and ntdll does + // not retain them. + let host_basic_status = unsafe { + host_status(NtQuerySystemInformation( + SystemInformationClass::Basic as u32, + host_info.as_mut_ptr().cast(), + information_length, + &raw mut host_return_length, + )) + }; + let guest_status = TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + mut_byte_ptr(&mut guest_info), + information_length, + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_status, host_basic_status); + assert_eq!(guest_return_length, host_return_length); + assert_eq!(guest_info.page_size, u32::try_from(PAGE_SIZE).unwrap()); + assert_eq!( + guest_info.allocation_granularity, + u32::try_from(ALLOCATION_GRANULARITY).unwrap() + ); + + let mut host_short_return_length = 0; + let mut guest_short_return_length = 0; + // SAFETY: Passing a short valid output buffer probes host ntdll's length handling; all + // pointers are valid local variables. + let host_short_status = unsafe { + host_status(NtQuerySystemInformation( + SystemInformationClass::Basic as u32, + host_info.as_mut_ptr().cast(), + information_length - 1, + &raw mut host_short_return_length, + )) + }; + let guest_short_status = TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + mut_byte_ptr(&mut guest_info), + information_length - 1, + Some(mut_ptr(&mut guest_short_return_length)), + ); + assert_eq!(guest_short_status, host_short_status); + assert_eq!(guest_short_return_length, host_short_return_length); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_fixed_class_lengths_match_host_ntdll() { + fn query_host( + class: SystemInformationClass, + output: &mut [u8], + return_length: &mut u32, + ) -> NtStatus { + // SAFETY: The output buffer and return-length pointer are valid locals and ntdll does + // not retain them. + unsafe { + host_status(NtQuerySystemInformation( + class as u32, + output.as_mut_ptr().cast(), + u32::try_from(output.len()).unwrap(), + return_length, + )) + } + } + + run_with_test_platform_pointers(|| { + let cases = [ + ( + SystemInformationClass::Processor, + size_of::(), + ), + ( + SystemInformationClass::RangeStart, + size_of::(), + ), + ( + SystemInformationClass::Verifier, + SYSTEM_VERIFIER_INFORMATION_LENGTH_USIZE, + ), + ( + SystemInformationClass::NumaProcessorMap, + size_of::(), + ), + ( + SystemInformationClass::EmulationBasic, + size_of::(), + ), + ( + SystemInformationClass::Flush, + size_of::(), + ), + ( + SystemInformationClass::HypervisorSharedPage, + size_of::(), + ), + ( + SystemInformationClass::ProcessorFeaturesBitMap, + size_of::(), + ), + ]; + + for (class, length) in cases { + let mut host_output = std::vec![0u8; length]; + let mut guest_output = std::vec![0u8; length]; + let mut host_return_length = 0; + let mut guest_return_length = 0; + let information_length = u32::try_from(length).unwrap(); + + let host_status = query_host(class, &mut host_output, &mut host_return_length); + let guest_status = TestTask::sys_nt_query_system_information( + class as u32, + mut_byte_ptr(&mut guest_output[0]), + information_length, + Some(mut_ptr(&mut guest_return_length)), + ); + + assert_eq!(guest_status, host_status, "{class:?}"); + assert_eq!(guest_return_length, host_return_length, "{class:?}"); + } + + let exact_length_cases = [ + ( + SystemInformationClass::Basic, + size_of::(), + ), + ( + SystemInformationClass::EmulationBasic, + size_of::(), + ), + ( + SystemInformationClass::RangeStart, + size_of::(), + ), + ]; + + for (class, length) in exact_length_cases { + let mut host_output = std::vec![0u8; length + 1]; + let mut guest_output = std::vec![0u8; length + 1]; + let mut host_return_length = 0; + let mut guest_return_length = 0; + let information_length = u32::try_from(length + 1).unwrap(); + + let host_status = query_host(class, &mut host_output, &mut host_return_length); + let guest_status = TestTask::sys_nt_query_system_information( + class as u32, + mut_byte_ptr(&mut guest_output[0]), + information_length, + Some(mut_ptr(&mut guest_return_length)), + ); + + assert_eq!(guest_status, host_status, "{class:?}"); + assert_eq!(guest_return_length, host_return_length, "{class:?}"); + } + }); + } + + #[test] + fn nt_query_system_information_does_not_publish_return_length_before_output_probe() { + run_with_test_platform_pointers(|| { + let mut return_length = u32::MAX; + + assert_eq!( + TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + null_mut_ptr(), + size_of::().trunc(), + Some(mut_ptr(&mut return_length)), + ), + NtStatus::ACCESS_VIOLATION + ); + assert_eq!(return_length, u32::MAX); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_null_output_return_length_order_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let mut host_return_length = u32::MAX; + let mut guest_return_length = u32::MAX; + let information_length = size_of::().trunc(); + + // SAFETY: This intentionally passes a null output buffer to probe host ntdll's + // NTSTATUS and return-length ordering; the return-length pointer is a valid local. + let host_status = unsafe { + host_status(NtQuerySystemInformation( + SystemInformationClass::Basic as u32, + core::ptr::null_mut(), + information_length, + &raw mut host_return_length, + )) + }; + let guest_status = TestTask::sys_nt_query_system_information( + SystemInformationClass::Basic as u32, + null_mut_ptr(), + information_length, + Some(mut_ptr(&mut guest_return_length)), + ); + + assert_eq!(guest_status, host_status); + assert_eq!(guest_return_length, host_return_length); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_ex_input_validation_order_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let query = LogicalProcessorRelationship::All as u32; + let mut host_output = [0u8; size_of::()]; + let mut guest_output = [0u8; size_of::()]; + let mut host_return_length = u32::MAX; + let mut guest_return_length = u32::MAX; + + // SAFETY: This intentionally passes a null input buffer to probe host ntdll's + // validation order. Output and return-length pointers are valid locals. + let host_null_status = unsafe { + host_status(NtQuerySystemInformationEx( + SystemInformationClass::Basic as u32, + core::ptr::null(), + 0, + host_output.as_mut_ptr().cast(), + u32::try_from(host_output.len()).unwrap(), + &raw mut host_return_length, + )) + }; + let guest_null_status = TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::Basic as u32, + None, + 0, + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_null_status, host_null_status); + assert_eq!(guest_null_status, NtStatus::INVALID_PARAMETER); + assert_eq!(guest_return_length, u32::MAX); + + // SAFETY: This uses a valid input DWORD and local output buffers to confirm that + // class validation still happens after the required input-buffer check succeeds. + let host_unknown_status = unsafe { + host_status(NtQuerySystemInformationEx( + u32::MAX, + core::ptr::from_ref(&query).cast(), + DWORD_SIZE_U32, + host_output.as_mut_ptr().cast(), + u32::try_from(host_output.len()).unwrap(), + &raw mut host_return_length, + )) + }; + let guest_unknown_status = TestTask::sys_nt_query_system_information_ex( + u32::MAX, + Some(const_byte_ptr(&query)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_unknown_status, host_unknown_status); + assert_eq!(guest_unknown_status, NtStatus::INVALID_INFO_CLASS); + assert_eq!(guest_return_length, u32::MAX); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_ex_logical_processor_status_matches_host_ntdll() { + fn query_host(relationship: u32, output: &mut [u8], return_length: &mut u32) -> NtStatus { + // SAFETY: This Windows-only test passes valid local input, output, and length pointers + // to ntdll and ntdll does not retain them. + unsafe { + host_status(NtQuerySystemInformationEx( + SystemInformationClass::LogicalProcessorAndGroup as u32, + core::ptr::from_ref(&relationship).cast(), + DWORD_SIZE_U32, + output.as_mut_ptr().cast(), + u32::try_from(output.len()).unwrap(), + return_length, + )) + } + } + + run_with_test_platform_pointers(|| { + let mut host_output = [0u8; 4096]; + let mut guest_output = [0u8; LOGICAL_PROCESSOR_ALL_INFORMATION_SIZE]; + let mut host_return_length = 0; + let mut guest_return_length = 0; + + let host_all_status = query_host( + LogicalProcessorRelationship::All as u32, + &mut host_output, + &mut host_return_length, + ); + let all_relationship = LogicalProcessorRelationship::All as u32; + let guest_all_status = TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(const_byte_ptr(&all_relationship)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_all_status, host_all_status); + assert_eq!(guest_all_status, NtStatus::SUCCESS); + assert!(host_return_length > 0); + assert_eq!( + guest_return_length, + u32::try_from(guest_output.len()).unwrap() + ); + + host_return_length = 0; + guest_return_length = 0; + let host_cache_status = query_host( + LogicalProcessorRelationship::Cache as u32, + &mut host_output, + &mut host_return_length, + ); + let cache_relationship = LogicalProcessorRelationship::Cache as u32; + let guest_cache_status = TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(const_byte_ptr(&cache_relationship)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_cache_status, host_cache_status); + assert_eq!(guest_cache_status, NtStatus::SUCCESS); + assert!(host_return_length >= size_of::().trunc()); + assert_eq!( + guest_return_length, + size_of::().trunc() + ); + + host_return_length = u32::MAX; + guest_return_length = u32::MAX; + let host_unknown_status = + query_host(u32::MAX, &mut host_output, &mut host_return_length); + let guest_unknown_status = TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::LogicalProcessorAndGroup as u32, + Some(const_byte_ptr(&u32::MAX)), + DWORD_SIZE_U32, + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ); + assert_eq!(guest_unknown_status, host_unknown_status); + assert_eq!(guest_return_length, u32::MAX); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_system_information_ex_feature_configuration_section_is_not_fabricated() { + run_with_test_platform_pointers(|| { + let request = [0u64; 4]; + let mut host_output = [0u8; 0x68]; + let mut guest_output = [0u8; 0x68]; + let mut host_return_length = 0; + let mut guest_return_length = u32::MAX; + + // SAFETY: This Windows-only test passes valid local input, output, and length pointers + // to ntdll and ntdll does not retain them. + let host_status = unsafe { + host_status(NtQuerySystemInformationEx( + SystemInformationClass::FeatureConfigurationSection as u32, + core::ptr::from_ref(&request).cast(), + u32::try_from(core::mem::size_of_val(&request)).unwrap(), + host_output.as_mut_ptr().cast(), + u32::try_from(host_output.len()).unwrap(), + &raw mut host_return_length, + )) + }; + if host_status == NtStatus::SUCCESS { + assert_eq!( + host_return_length, + u32::try_from(host_output.len()).unwrap() + ); + assert!(host_output.iter().any(|byte| *byte != 0)); + } + + assert_eq!( + TestTask::sys_nt_query_system_information_ex( + SystemInformationClass::FeatureConfigurationSection as u32, + Some(const_byte_ptr(&request)), + u32::try_from(core::mem::size_of_val(&request)).unwrap(), + mut_byte_ptr(&mut guest_output), + u32::try_from(guest_output.len()).unwrap(), + Some(mut_ptr(&mut guest_return_length)), + ), + NtStatus::INVALID_INFO_CLASS + ); + assert_eq!(guest_return_length, u32::MAX); + }); + } + + #[test] + fn nt_query_performance_counter_writes_monotonic_counter_and_frequency() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut first_counter = -1i64; + let mut second_counter = -1i64; + let mut frequency = 0i64; + + assert_eq!( + task.sys_nt_query_performance_counter( + mut_ptr(&mut first_counter), + Some(mut_ptr(&mut frequency)), + ), + NtStatus::SUCCESS + ); + assert_eq!(frequency, QPC_FREQUENCY_HZ); + assert!(first_counter >= 0); + + assert_eq!( + task.sys_nt_query_performance_counter(mut_ptr(&mut second_counter), None), + NtStatus::SUCCESS + ); + assert!(second_counter >= first_counter); + }); + } + + #[test] + fn nt_query_performance_counter_rejects_null_counter() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut frequency = 0i64; + + assert_eq!( + task.sys_nt_query_performance_counter( + null_mut_ptr(), + Some(mut_ptr(&mut frequency)), + ), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[test] + fn nt_convert_between_auxiliary_counter_and_performance_counter_is_not_supported() { + run_with_test_platform_pointers(|| { + let source = 0u64; + let mut destination = 0u64; + let mut conversion_error = 0u64; + + assert_eq!( + TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter( + 0, + null_const_ptr(), + mut_ptr(&mut destination), + Some(mut_ptr(&mut conversion_error)), + ), + NtStatus::ACCESS_VIOLATION + ); + assert_eq!( + TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter( + 0, + const_ptr(&source), + mut_ptr(&mut destination), + Some(mut_ptr(&mut conversion_error)), + ), + NtStatus::NOT_SUPPORTED + ); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_query_performance_counter_status_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut host_counter = 0i64; + let mut host_frequency = 0i64; + let mut guest_counter = 0i64; + let mut guest_frequency = 0i64; + + // SAFETY: This Windows-only test calls the process ntdll export with valid local + // output pointers and checks only the returned status and written scalar values. + let host_valid_status = unsafe { + host_status(NtQueryPerformanceCounter( + &raw mut host_counter, + &raw mut host_frequency, + )) + }; + let guest_status = task.sys_nt_query_performance_counter( + mut_ptr(&mut guest_counter), + Some(mut_ptr(&mut guest_frequency)), + ); + + assert_eq!(guest_status, host_valid_status); + assert!(guest_counter >= 0); + assert!(guest_frequency > 0); + assert!(host_counter >= 0); + assert!(host_frequency > 0); + + // SAFETY: Passing a null counter pointer intentionally probes host ntdll's invalid + // output behavior; the non-null frequency pointer is a valid local output. + let host_null_counter_status = unsafe { + host_status(NtQueryPerformanceCounter( + core::ptr::null_mut(), + &raw mut host_frequency, + )) + }; + let guest_null_counter_status = task.sys_nt_query_performance_counter( + null_mut_ptr(), + Some(mut_ptr(&mut guest_frequency)), + ); + assert_eq!(guest_null_counter_status, host_null_counter_status); + }); + } + + #[test] + fn nt_query_performance_counter_duration_tracks_sleep_duration() { + run_with_test_platform_pointers(|| { + let task = crate::tests::test_task(); + let mut guest_frequency = 0i64; + let mut guest_start = 0i64; + let mut guest_end = 0i64; + + let guest_start_status = task.sys_nt_query_performance_counter( + mut_ptr(&mut guest_start), + Some(mut_ptr(&mut guest_frequency)), + ); + + std::thread::sleep(QPC_SLEEP_DURATION); + + let guest_end_status = task.sys_nt_query_performance_counter( + mut_ptr(&mut guest_end), + Some(mut_ptr(&mut guest_frequency)), + ); + + assert_eq!(guest_start_status, NtStatus::SUCCESS); + assert_eq!(guest_end_status, NtStatus::SUCCESS); + assert_eq!(guest_frequency, QPC_FREQUENCY_HZ); + + let guest_duration_nanos = qpc_delta_nanos(guest_start, guest_end); + let minimum_duration_nanos = QPC_SLEEP_DURATION + .saturating_sub(QPC_SLEEP_TOLERANCE) + .as_nanos(); + let maximum_duration_nanos = QPC_SLEEP_DURATION + .saturating_add(QPC_SLEEP_TOLERANCE) + .as_nanos(); + + assert!( + guest_duration_nanos >= minimum_duration_nanos, + "guest duration {guest_duration_nanos}ns was shorter than requested sleep minus tolerance {minimum_duration_nanos}ns", + ); + assert!( + guest_duration_nanos <= maximum_duration_nanos, + "guest duration {guest_duration_nanos}ns was longer than requested sleep plus tolerance {maximum_duration_nanos}ns", + ); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn nt_convert_between_auxiliary_counter_status_matches_host_ntdll() { + run_with_test_platform_pointers(|| { + let source = 0u64; + let mut destination = 0u64; + let mut conversion_error = 0u64; + + // SAFETY: Passing a null source pointer intentionally probes host ntdll's invalid + // input behavior; the output pointers are valid local scalars for the duration. + let host_null_source_status = unsafe { + host_status(NtConvertBetweenAuxiliaryCounterAndPerformanceCounter( + 0, + core::ptr::null(), + &raw mut destination, + &raw mut conversion_error, + )) + }; + let guest_null_source_status = + TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter( + 0, + null_const_ptr(), + mut_ptr(&mut destination), + Some(mut_ptr(&mut conversion_error)), + ); + assert_eq!(guest_null_source_status, host_null_source_status); + + // SAFETY: All pointers passed to host ntdll point at local scalar variables that live + // for the whole call; the function does not retain them. + let host_valid_source_status = unsafe { + host_status(NtConvertBetweenAuxiliaryCounterAndPerformanceCounter( + 0, + &raw const source, + &raw mut destination, + &raw mut conversion_error, + )) + }; + let guest_valid_source_status = + TestTask::sys_nt_convert_between_auxiliary_counter_and_performance_counter( + 0, + const_ptr(&source), + mut_ptr(&mut destination), + Some(mut_ptr(&mut conversion_error)), + ); + assert_eq!(guest_valid_source_status, host_valid_source_status); + }); + } +} diff --git a/litebox_shim_windows/src/syscalls/thread.rs b/litebox_shim_windows/src/syscalls/thread.rs new file mode 100644 index 0000000000..2883801ab0 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/thread.rs @@ -0,0 +1,366 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use int_enum::IntEnum; +use litebox::platform::RawConstPointer as _; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable}; + +use crate::syscalls::{Handle, ThreadHandle}; +use crate::{ConstPtr, MutPtr, ShimFS, ShimPlatform, Task, probe_guest_output_preserving_value}; + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq, IntEnum)] +enum ThreadInformationClass { + SchedulerSharedDataSlot = 57, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable)] +struct ThreadSchedulerSharedDataSlotInformation { + action: u32, + _padding0: u32, + scheduler_shared_data_handle: usize, + slot: usize, +} + +impl Task { + pub(crate) fn sys_nt_set_information_thread( + thread_handle: ThreadHandle, + thread_information_class: u32, + thread_information: ConstPtr, + thread_information_length: u32, + ) -> NtStatus { + let Ok(thread_information_class) = + ThreadInformationClass::try_from(thread_information_class) + else { + litebox_util_log::debug!( + thread_information_class = thread_information_class; + "Unsupported NtSetInformationThread class" + ); + return NtStatus::INVALID_INFO_CLASS; + }; + + let status = match thread_information_class { + ThreadInformationClass::SchedulerSharedDataSlot => { + Self::set_thread_scheduler_shared_data_slot( + thread_handle, + thread_information, + thread_information_length, + ) + } + }; + + if status == NtStatus::SUCCESS { + litebox_util_log::debug!( + thread_information_class:? = thread_information_class, + thread_information_length = thread_information_length; + "Handled NtSetInformationThread syscall" + ); + } + + status + } + + fn set_thread_scheduler_shared_data_slot( + thread_handle: ThreadHandle, + thread_information: ConstPtr, + thread_information_length: u32, + ) -> NtStatus { + let thread_information = + ConstPtr::::from_usize( + thread_information.as_usize(), + ); + let Some(_thread_information) = thread_information.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if thread_information_length < size_of::().trunc() + { + return NtStatus::INFO_LENGTH_MISMATCH; + } + if !thread_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + // The scheduler-shared-data handle is never valid in the sandbox, matching the host + // current-thread path for the observed all-zero slot request. + NtStatus::INVALID_HANDLE + } + + pub(crate) fn sys_nt_open_thread_token( + thread_handle: ThreadHandle, + _desired_access: u32, + _open_as_self: u32, + token_handle: MutPtr, + ) -> NtStatus { + Self::open_thread_token(thread_handle, token_handle) + } + + pub(crate) fn sys_nt_open_thread_token_ex( + thread_handle: ThreadHandle, + _desired_access: u32, + _open_as_self: u32, + _handle_attributes: u32, + token_handle: MutPtr, + ) -> NtStatus { + // TODO: HandleAttributes is outcome-independent while the sandbox has no impersonation + // token. Once a real token subsystem exists it must be validated; host 25H2 returns + // STATUS_INVALID_PARAMETER for attrs=0xffffffff after ImpersonateSelf. + Self::open_thread_token(thread_handle, token_handle) + } + + fn open_thread_token( + thread_handle: ThreadHandle, + token_handle: MutPtr, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(token_handle) { + return status; + } + if !thread_handle.is_current() { + return NtStatus::INVALID_HANDLE; + } + + // A thread only has a token while it is actively impersonating (SetThreadToken / + // ImpersonateSelf). Sandbox threads never impersonate, so real host 25H2 returns + // STATUS_NO_TOKEN here as well: this is the host-faithful terminal answer, not a stub. + NtStatus::NO_TOKEN + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::null_const_ptr; + use litebox::platform::ThreadProvider; + + type TestPlatform = crate::tests::TestPlatform; + type TestTask = Task; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = crate::tests::test_platform(); + ::run_test_thread(f) + } + + fn const_byte_ptr(value: &T) -> ConstPtr { + ConstPtr::::from_usize(core::ptr::from_ref(value).cast::() as usize) + } + + #[test] + fn nt_set_information_thread_scheduler_shared_data_slot_validates_arguments() { + run_with_test_platform_pointers(|| { + let information = ThreadSchedulerSharedDataSlotInformation { + action: 0, + _padding0: 0, + scheduler_shared_data_handle: 0, + slot: 0, + }; + let information_len: u32 = + size_of::().trunc(); + let bad_handle = ThreadHandle::from_raw(0x1234); + + assert_eq!( + TestTask::sys_nt_set_information_thread( + bad_handle, + 0xffff, + null_const_ptr::(), + information_len - 1, + ), + NtStatus::INVALID_INFO_CLASS + ); + + assert_eq!( + TestTask::sys_nt_set_information_thread( + bad_handle, + ThreadInformationClass::SchedulerSharedDataSlot as u32, + null_const_ptr::(), + information_len, + ), + NtStatus::ACCESS_VIOLATION + ); + + assert_eq!( + TestTask::sys_nt_set_information_thread( + bad_handle, + ThreadInformationClass::SchedulerSharedDataSlot as u32, + const_byte_ptr(&information), + information_len - 1, + ), + NtStatus::INFO_LENGTH_MISMATCH + ); + + assert_eq!( + TestTask::sys_nt_set_information_thread( + bad_handle, + ThreadInformationClass::SchedulerSharedDataSlot as u32, + const_byte_ptr(&information), + information_len, + ), + NtStatus::INVALID_HANDLE + ); + + assert_eq!( + TestTask::sys_nt_set_information_thread( + ThreadHandle::CURRENT, + ThreadInformationClass::SchedulerSharedDataSlot as u32, + const_byte_ptr(&information), + information_len, + ), + NtStatus::INVALID_HANDLE + ); + }); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + mod host_fidelity { + use core::ffi::c_void; + + use super::*; + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtSetInformationThread( + thread_handle: *mut c_void, + thread_information_class: u32, + thread_information: *const c_void, + thread_information_length: u32, + ) -> i32; + } + + fn host_nt_set_information_thread( + thread_handle: *mut c_void, + thread_information_class: u32, + thread_information: *const c_void, + thread_information_length: u32, + ) -> NtStatus { + // SAFETY: The host ntdll call treats these as user-mode input pointers, probes them, + // and does not retain them. Tests pass either valid locals or null to observe NTSTATUS. + let status = unsafe { + NtSetInformationThread( + thread_handle, + thread_information_class, + thread_information, + thread_information_length, + ) + }; + NtStatus::from_raw(u32::from_ne_bytes(status.to_ne_bytes())) + } + + #[test] + fn nt_set_information_thread_scheduler_shared_data_slot_matches_host_statuses() { + run_with_test_platform_pointers(|| { + let information = ThreadSchedulerSharedDataSlotInformation { + action: 0, + _padding0: 0, + scheduler_shared_data_handle: 0, + slot: 0, + }; + let information_len: u32 = + size_of::().trunc(); + let current_thread = (usize::MAX - 1) as *mut c_void; + let bad_thread = 0x1234usize as *mut c_void; + let scheduler_class = ThreadInformationClass::SchedulerSharedDataSlot as u32; + let bad_class = 0xffff; + + if host_nt_set_information_thread( + current_thread, + scheduler_class, + core::ptr::from_ref(&information).cast::(), + information_len, + ) == NtStatus::INVALID_INFO_CLASS + { + return; + } + + for ( + thread_handle, + shim_thread_handle, + thread_information_class, + host_thread_information, + shim_thread_information, + thread_information_length, + ) in [ + ( + current_thread, + ThreadHandle::CURRENT, + scheduler_class, + core::ptr::from_ref(&information).cast::(), + const_byte_ptr(&information), + information_len, + ), + ( + current_thread, + ThreadHandle::CURRENT, + scheduler_class, + core::ptr::from_ref(&information).cast::(), + const_byte_ptr(&information), + information_len - 1, + ), + ( + current_thread, + ThreadHandle::CURRENT, + scheduler_class, + core::ptr::null(), + null_const_ptr::(), + information_len, + ), + ( + current_thread, + ThreadHandle::CURRENT, + bad_class, + core::ptr::null(), + null_const_ptr::(), + information_len, + ), + ( + bad_thread, + ThreadHandle::from_raw(0x1234), + scheduler_class, + core::ptr::from_ref(&information).cast::(), + const_byte_ptr(&information), + information_len, + ), + ( + bad_thread, + ThreadHandle::from_raw(0x1234), + scheduler_class, + core::ptr::null(), + null_const_ptr::(), + information_len, + ), + ( + bad_thread, + ThreadHandle::from_raw(0x1234), + scheduler_class, + core::ptr::from_ref(&information).cast::(), + const_byte_ptr(&information), + information_len - 1, + ), + ( + bad_thread, + ThreadHandle::from_raw(0x1234), + bad_class, + core::ptr::from_ref(&information).cast::(), + const_byte_ptr(&information), + information_len, + ), + ] { + let host = host_nt_set_information_thread( + thread_handle, + thread_information_class, + host_thread_information, + thread_information_length, + ); + let shim = TestTask::sys_nt_set_information_thread( + shim_thread_handle, + thread_information_class, + shim_thread_information, + thread_information_length, + ); + + assert_eq!(shim, host); + } + }); + } + } +} diff --git a/litebox_shim_windows/src/syscalls/timer.rs b/litebox_shim_windows/src/syscalls/timer.rs new file mode 100644 index 0000000000..0befe6ab04 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/timer.rs @@ -0,0 +1,518 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT timer syscalls. + +use alloc::sync::Arc; +use core::marker::PhantomData; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::{AccessMask, ObjectAttributes}; +use crate::syscalls::Handle; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +const TIMER2_ATTRIBUTE_IR_TIMER: u32 = 0x0000_0002; +const TIMER2_ATTRIBUTE_HIGH_RESOLUTION: u32 = 0x0000_0004; +const TIMER2_ATTRIBUTE_NO_WAKE: u32 = 0x0000_0008; +const TIMER2_ATTRIBUTE_NOTIFICATION: u32 = 0x8000_0000; +const TIMER2_ATTRIBUTE_KNOWN_MASK: u32 = TIMER2_ATTRIBUTE_IR_TIMER + | TIMER2_ATTRIBUTE_HIGH_RESOLUTION + | TIMER2_ATTRIBUTE_NO_WAKE + | TIMER2_ATTRIBUTE_NOTIFICATION; +const TIMER2_ATTRIBUTE_RESERVED_MASK: u32 = !TIMER2_ATTRIBUTE_KNOWN_MASK; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct TimerAccess: u32 { + const QUERY_STATE = 0x0001; + const MODIFY_STATE = 0x0002; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY_STATE.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() | Self::MODIFY_STATE.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() | AccessMask::SYNCHRONIZE.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::QUERY_STATE.bits() + | Self::MODIFY_STATE.bits(); + + const _ = !0; + } +} + +impl TimerAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct Timer2Attributes: u32 { + const HIGH_RESOLUTION = TIMER2_ATTRIBUTE_HIGH_RESOLUTION; + const NO_WAKE = TIMER2_ATTRIBUTE_NO_WAKE; + const NOTIFICATION = TIMER2_ATTRIBUTE_NOTIFICATION; + + const _ = !0; + } +} + +pub(crate) struct TimerSubsystem(PhantomData); + +impl FdEnabledSubsystem for TimerSubsystem { + type Entry = TimerHandleObject; +} + +impl FdEnabledSubsystemEntry for TimerHandleObject {} + +impl crate::WindowsHandleSubsystem for TimerSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + TimerAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct TimerHandleObject { + _timer: Arc>, +} + +pub(crate) struct TimerObject { + _attributes: Timer2Attributes, + _not_send_without_platform: PhantomData, +} + +pub(crate) struct TimerCreateParameters { + pub(crate) timer_handle: MutPtr, + pub(crate) timer_id: Option>, + pub(crate) object_attributes: Option>, + pub(crate) attributes: u32, + pub(crate) desired_access: u32, +} + +fn validate_timer2_before_output( + params: &TimerCreateParameters, +) -> Result<(), NtStatus> { + if params.object_attributes.is_some() { + return Err(NtStatus::INVALID_PARAMETER_3); + } + if params.attributes & TIMER2_ATTRIBUTE_RESERVED_MASK != 0 { + return Err(NtStatus::INVALID_PARAMETER_4); + } + if params.attributes & TIMER2_ATTRIBUTE_IR_TIMER == 0 && params.timer_id.is_some() { + return Err(NtStatus::INVALID_PARAMETER_2); + } + Ok(()) +} + +fn validate_timer2_after_output( + params: &TimerCreateParameters, +) -> Result<(), NtStatus> { + if params.attributes & TIMER2_ATTRIBUTE_IR_TIMER == 0 { + return Ok(()); + } + if params.timer_id.is_some() { + Err(NtStatus::ACCESS_DENIED) + } else { + Err(NtStatus::INVALID_PARAMETER) + } +} + +impl Task { + fn insert_timer_handle( + &self, + timer: Arc>, + granted_access: TimerAccess, + ) -> Result { + self.insert_typed_handle::>( + TimerHandleObject { _timer: timer }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_timer_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_timer(timer: TimerHandleObject) { + drop(timer); + } + + pub(crate) fn sys_nt_create_timer2(&self, params: TimerCreateParameters) -> NtStatus { + if let Err(status) = validate_timer2_before_output(¶ms) { + return status; + } + if let Err(status) = probe_guest_output_preserving_value::(params.timer_handle) + { + return status; + } + if let Err(status) = validate_timer2_after_output(¶ms) { + return status; + } + + let timer = Arc::new(TimerObject { + // TODO: store timer state once NtSetTimer2 schedules due times and waiters can + // observe expiration/signaling instead of only validating the handle shape. + _attributes: Timer2Attributes::from_bits_retain(params.attributes), + _not_send_without_platform: PhantomData, + }); + let granted_access = TimerAccess::from_desired_access(params.desired_access); + let Ok(handle) = self.insert_timer_handle(timer, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if params.timer_handle.write_at_offset(0, handle).is_none() { + self.close_timer_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_set_timer2( + &self, + timer_handle: Handle, + due_time: Option>, + period: Option>, + parameters: Option>, + ) -> NtStatus { + if let Err(status) = self.require_handle_access::>( + timer_handle, + TimerAccess::MODIFY_STATE.bits(), + ) { + return status; + } + let _due_time = match due_time { + Some(due_time) => match due_time.read_at_offset(0) { + Some(due_time) => Some(due_time), + None => return NtStatus::ACCESS_VIOLATION, + }, + None => None, + }; + let _period = match period { + Some(period) => match period.read_at_offset(0) { + Some(period) => Some(period), + None => return NtStatus::ACCESS_VIOLATION, + }, + None => None, + }; + + // TODO: parse T2_SET_PARAMETERS and model callbacks/tolerable delay when the timer + // object grows real scheduling and notification behavior. + let _ = parameters; + + // TODO: store due_time/period, transition the timer's signaled state, and notify + // waiters or associated wait-completion packets instead of returning a no-op success. + NtStatus::SUCCESS + } +} + +#[cfg(test)] +mod tests { + use core::mem::size_of; + + use litebox::platform::ThreadProvider; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::ObjectAttributes; + use crate::tests::{TestPlatform, const_ptr, mut_ptr, null_mut_ptr, test_platform, test_task}; + + const TIMER_ALL_ACCESS: u32 = 0x001f_0003; + + fn object_attributes_size() -> u32 { + u32::try_from(size_of::()).expect("OBJECT_ATTRIBUTES fits in ULONG") + } + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = test_platform(); + ::run_test_thread(f) + } + + fn create_timer2( + task: &Task, + handle: &mut Handle, + timer_id: Option>, + object_attributes: Option>, + attributes: u32, + ) -> NtStatus { + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: mut_ptr(handle), + timer_id, + object_attributes, + attributes, + desired_access: TIMER_ALL_ACCESS, + }) + } + + #[test] + fn set_timer2_accepts_created_timer() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let mut handle = Handle::default(); + let due_time = -10_000i64; + let period = 0i64; + + assert_eq!( + create_timer2(&task, &mut handle, None, None, 0), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_set_timer2( + handle, + Some(const_ptr(&due_time)), + Some(const_ptr(&period)), + None + ), + NtStatus::SUCCESS + ); + assert_eq!(task.sys_nt_close(handle), NtStatus::SUCCESS); + }); + } + + #[test] + fn create_rejects_object_attributes_before_output_pointer() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + assert_eq!( + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: null_mut_ptr(), + timer_id: None, + object_attributes: Some(const_ptr(&bad_length)), + attributes: 0, + desired_access: TIMER_ALL_ACCESS, + }), + NtStatus::INVALID_PARAMETER_3 + ); + }); + } + + #[test] + fn create_validates_reserved_bits_and_non_ir_timer_id_before_output_pointer() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let timer_id = 1u32; + + assert_eq!( + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: null_mut_ptr(), + timer_id: None, + object_attributes: None, + attributes: 1, + desired_access: TIMER_ALL_ACCESS, + }), + NtStatus::INVALID_PARAMETER_4 + ); + assert_eq!( + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: null_mut_ptr(), + timer_id: Some(const_ptr(&timer_id)), + object_attributes: None, + attributes: TIMER2_ATTRIBUTE_NOTIFICATION, + desired_access: TIMER_ALL_ACCESS, + }), + NtStatus::INVALID_PARAMETER_2 + ); + }); + } + + #[test] + fn create_probes_output_pointer_before_ir_timer_validation() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let timer_id = 1u32; + + assert_eq!( + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: null_mut_ptr(), + timer_id: None, + object_attributes: None, + attributes: TIMER2_ATTRIBUTE_IR_TIMER, + desired_access: TIMER_ALL_ACCESS, + }), + NtStatus::ACCESS_VIOLATION + ); + assert_eq!( + task.sys_nt_create_timer2(TimerCreateParameters { + timer_handle: null_mut_ptr(), + timer_id: Some(const_ptr(&timer_id)), + object_attributes: None, + attributes: TIMER2_ATTRIBUTE_IR_TIMER, + desired_access: TIMER_ALL_ACCESS, + }), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[test] + fn create_rejects_ir_timers_without_clobbering_output() { + let task = test_task(); + let timer_id = 1u32; + let mut handle = Handle::from_raw(usize::MAX); + + assert_eq!( + create_timer2(&task, &mut handle, None, None, TIMER2_ATTRIBUTE_IR_TIMER), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + assert_eq!( + create_timer2( + &task, + &mut handle, + Some(const_ptr(&timer_id)), + None, + TIMER2_ATTRIBUTE_IR_TIMER + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + } + + #[test] + fn create_rejections_do_not_clobber_output() { + let task = test_task(); + let timer_id = 1u32; + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + let valid_length = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + for (timer_id, object_attributes, attributes, expected_status) in [ + ( + None, + Some(const_ptr(&bad_length)), + 0, + NtStatus::INVALID_PARAMETER_3, + ), + ( + None, + Some(const_ptr(&valid_length)), + 0, + NtStatus::INVALID_PARAMETER_3, + ), + (None, None, 1, NtStatus::INVALID_PARAMETER_4), + ( + Some(const_ptr(&timer_id)), + None, + TIMER2_ATTRIBUTE_HIGH_RESOLUTION, + NtStatus::INVALID_PARAMETER_2, + ), + ] { + let mut handle = Handle::from_raw(usize::MAX); + assert_eq!( + create_timer2(&task, &mut handle, timer_id, object_attributes, attributes), + expected_status + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + } + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_create_timer2_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateTimer2( + handle: *mut *mut c_void, + timer_id: *const u32, + object_attributes: *const ObjectAttributes, + attributes: u32, + desired_access: u32, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + let task = test_task(); + let timer_id = 1u32; + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + for (timer_id, object_attributes, attributes) in [ + (None, None, 0), + (None, None, TIMER2_ATTRIBUTE_HIGH_RESOLUTION), + (None, None, TIMER2_ATTRIBUTE_NO_WAKE), + ( + None, + None, + TIMER2_ATTRIBUTE_HIGH_RESOLUTION | TIMER2_ATTRIBUTE_NO_WAKE, + ), + (None, None, TIMER2_ATTRIBUTE_NOTIFICATION), + ( + None, + None, + TIMER2_ATTRIBUTE_NOTIFICATION + | TIMER2_ATTRIBUTE_HIGH_RESOLUTION + | TIMER2_ATTRIBUTE_NO_WAKE, + ), + (None, None, 1), + (Some(&timer_id), None, TIMER2_ATTRIBUTE_HIGH_RESOLUTION), + (None, Some(&bad_length), 0), + (None, None, TIMER2_ATTRIBUTE_IR_TIMER), + (Some(&timer_id), None, TIMER2_ATTRIBUTE_IR_TIMER), + ] { + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, optional input pointers reference local values + // for the duration of the call, and successful host handles are closed below. + let host_status = unsafe { + NtCreateTimer2( + &raw mut host_handle, + timer_id.map_or(core::ptr::null(), core::ptr::from_ref), + object_attributes.map_or(core::ptr::null(), core::ptr::from_ref), + attributes, + TIMER_ALL_ACCESS, + ) + }; + if host_status == NtStatus::SUCCESS.as_raw() && !host_handle.is_null() { + // SAFETY: The handle was returned by NtCreateTimer2 in this test. + assert_eq!(unsafe { NtClose(host_handle) }, NtStatus::SUCCESS.as_raw()); + } + + let mut shim_handle = Handle::default(); + let shim_status = create_timer2( + &task, + &mut shim_handle, + timer_id.map(const_ptr), + object_attributes.map(const_ptr), + attributes, + ); + assert_eq!(shim_status.as_raw(), host_status); + if shim_status == NtStatus::SUCCESS { + assert!(!shim_handle.is_null()); + assert_eq!(task.sys_nt_close(shim_handle), NtStatus::SUCCESS); + } + } + } +} diff --git a/litebox_shim_windows/src/syscalls/token.rs b/litebox_shim_windows/src/syscalls/token.rs new file mode 100644 index 0000000000..48be754a54 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/token.rs @@ -0,0 +1,947 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT access-token syscalls. + +use alloc::boxed::Box; +use alloc::sync::Arc; +use alloc::vec::Vec; +use core::borrow::Borrow; +use core::mem::size_of; + +use int_enum::IntEnum; +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; +use zerocopy::{FromBytes, Immutable, IntoBytes}; + +use crate::nt_types::{AccessMask, Luid, UnicodeString}; +use crate::syscalls::{Handle, ProcessHandle}; +use crate::{ + ConstPtr, HandleAttributes, MutPtr, ShimFS, Task, WindowsHandleSubsystem, + probe_guest_output_buffer, probe_guest_output_preserving_value, +}; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct TokenAccess: u32 { + const ASSIGN_PRIMARY = 0x0001; + const DUPLICATE = 0x0002; + const IMPERSONATE = 0x0004; + const QUERY = 0x0008; + const QUERY_SOURCE = 0x0010; + const ADJUST_PRIVILEGES = 0x0020; + const ADJUST_GROUPS = 0x0040; + const ADJUST_DEFAULT = 0x0080; + const ADJUST_SESSION_ID = 0x0100; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() + | Self::ADJUST_PRIVILEGES.bits() + | Self::ADJUST_GROUPS.bits() + | Self::ADJUST_DEFAULT.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits(); + const ALL_ACCESS = AccessMask::DELETE.bits() + | AccessMask::READ_CONTROL.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits() + | Self::ASSIGN_PRIMARY.bits() + | Self::DUPLICATE.bits() + | Self::IMPERSONATE.bits() + | Self::QUERY.bits() + | Self::QUERY_SOURCE.bits() + | Self::ADJUST_PRIVILEGES.bits() + | Self::ADJUST_GROUPS.bits() + | Self::ADJUST_DEFAULT.bits() + | Self::ADJUST_SESSION_ID.bits(); + + const _ = !0; + } +} + +impl TokenAccess { + fn from_desired_access(desired_access: u32) -> Self { + let maximum_allowed = desired_access & AccessMask::MAXIMUM_ALLOWED.bits() != 0; + let explicit_access = desired_access & !AccessMask::MAXIMUM_ALLOWED.bits(); + let normalized = AccessMask::expand_generic_access( + explicit_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + ); + Self::from_bits_retain(if maximum_allowed { + normalized | Self::ALL_ACCESS.bits() + } else { + normalized + }) + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +pub(crate) enum TokenInformationClass { + User = 1, + Groups = 2, + Privileges = 3, + Owner = 4, + PrimaryGroup = 5, + DefaultDacl = 6, + Source = 7, + Type = 8, + ImpersonationLevel = 9, + Statistics = 10, + RestrictedSids = 11, + SessionId = 12, + GroupsAndPrivileges = 13, + SessionReference = 14, + SandBoxInert = 15, + AuditPolicy = 16, + Origin = 17, + ElevationType = 18, + LinkedToken = 19, + Elevation = 20, + HasRestrictions = 21, + AccessInformation = 22, + VirtualizationAllowed = 23, + VirtualizationEnabled = 24, + IntegrityLevel = 25, + UiAccess = 26, + MandatoryPolicy = 27, + LogonSid = 28, + IsAppContainer = 29, + Capabilities = 30, + AppContainerSid = 31, + AppContainerNumber = 32, + UserClaimAttributes = 33, + DeviceClaimAttributes = 34, + RestrictedUserClaimAttributes = 35, + RestrictedDeviceClaimAttributes = 36, + DeviceGroups = 37, + RestrictedDeviceGroups = 38, + SecurityAttributes = 39, + IsRestricted = 40, + ProcessTrustLevel = 41, + PrivateNameSpace = 42, + SingletonAttributes = 43, + BnoIsolation = 44, + ChildProcessFlags = 45, + IsLessPrivilegedAppContainer = 46, + IsSandboxed = 47, + IsAppSilo = 48, + LoggingInformation = 49, +} + +#[repr(u16)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum TokenSecurityAttributeValueType { + Invalid = 0x00, + Int64 = 0x01, + Uint64 = 0x02, + String = 0x03, + Fqbn = 0x04, + Sid = 0x05, + Boolean = 0x06, + OctetString = 0x10, +} + +#[repr(u16)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum TokenSecurityAttributesInformationVersion { + V1 = 1, +} + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + struct TokenSecurityAttributeFlags: u32 { + const NON_INHERITABLE = 0x0001; + const VALUE_CASE_SENSITIVE = 0x0002; + const USE_FOR_DENY_ONLY = 0x0004; + const DISABLED_BY_DEFAULT = 0x0008; + const DISABLED = 0x0010; + const MANDATORY = 0x0020; + const COMPARE_IGNORE = 0x0040; + const CUSTOM = 0xffff_0000; + } +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, FromBytes, Immutable, IntoBytes)] +struct TokenSecurityAttributeV1 { + name: UnicodeString, + value_type: u16, + reserved: u16, + flags: u32, + value_count: u32, + padding: u32, + values: usize, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +struct TokenSecurityAttributesInformation { + version: u16, + reserved: u16, + attribute_count: u32, + attribute_v1: usize, +} + +struct TokenSecurityAttribute { + name: Box<[u16]>, + value_type: TokenSecurityAttributeValueType, + flags: TokenSecurityAttributeFlags, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct SidAndAttributes { + sid: usize, + attributes: u32, + padding: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct TokenUser { + user: SidAndAttributes, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct Sid { + revision: u8, + sub_authority_count: u8, + identifier_authority: [u8; 6], + sub_authority: [u32; 1], +} + +#[repr(C, packed(4))] +#[derive(Immutable, IntoBytes)] +struct TokenUserInformation { + user: TokenUser, + sid: Sid, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct TokenPrivileges { + privilege_count: u32, +} + +#[repr(C)] +#[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] +pub(crate) struct TokenStatistics { + token_id: Luid, + authentication_id: Luid, + expiration_time: i64, + token_type: u32, + impersonation_level: u32, + dynamic_charged: u32, + dynamic_available: u32, + group_count: u32, + privilege_count: u32, + modified_id: Luid, +} + +const TOKEN_TYPE_PRIMARY: u32 = 1; +const SECURITY_ANONYMOUS: u32 = 0; + +// TODO(token-luid-allocation): Allocate these from sandbox-wide state once multiple token objects +// or token mutation are supported. +const PRIMARY_TOKEN_ID: Luid = Luid { + low_part: 1, + high_part: 0, +}; + +const PRIMARY_TOKEN_MODIFIED_ID: Luid = Luid { + low_part: 2, + high_part: 0, +}; + +const SYSTEM_LUID: Luid = Luid { + low_part: 0x3e7, + high_part: 0, +}; + +const LOCAL_SYSTEM_SID: Sid = Sid { + revision: 1, + sub_authority_count: 1, + identifier_authority: [0, 0, 0, 0, 0, 5], + sub_authority: [18], +}; + +pub(crate) struct TokenObject { + user: Sid, + statistics: TokenStatistics, + security_attributes: Box<[TokenSecurityAttribute]>, +} + +impl TokenObject { + pub(crate) fn primary() -> Self { + Self { + user: LOCAL_SYSTEM_SID, + statistics: TokenStatistics { + token_id: PRIMARY_TOKEN_ID, + authentication_id: SYSTEM_LUID, + expiration_time: i64::MAX, + token_type: TOKEN_TYPE_PRIMARY, + impersonation_level: SECURITY_ANONYMOUS, + dynamic_charged: 0, + dynamic_available: 0, + group_count: 0, + privilege_count: 0, + modified_id: PRIMARY_TOKEN_MODIFIED_ID, + }, + security_attributes: Box::new([]), + } + } +} + +pub(crate) struct TokenHandleObject { + token: Arc, +} + +pub(crate) struct TokenSubsystem; + +impl FdEnabledSubsystem for TokenSubsystem { + type Entry = TokenHandleObject; +} + +impl FdEnabledSubsystemEntry for TokenHandleObject {} + +impl WindowsHandleSubsystem for TokenSubsystem { + fn normalize_desired_access(desired_access: u32) -> u32 { + TokenAccess::from_desired_access(desired_access).bits() + } +} + +impl Task { + const CURRENT_PROCESS_TOKEN: Handle = Handle::from_raw(usize::MAX - 3); + const CURRENT_THREAD_TOKEN: Handle = Handle::from_raw(usize::MAX - 4); + const CURRENT_THREAD_EFFECTIVE_TOKEN: Handle = Handle::from_raw(usize::MAX - 5); + + pub(crate) fn sys_nt_open_process_token( + &self, + process_handle: ProcessHandle, + desired_access: u32, + token_handle: MutPtr, + ) -> NtStatus { + self.open_process_token(process_handle, desired_access, 0, token_handle) + } + + pub(crate) fn sys_nt_open_process_token_ex( + &self, + process_handle: ProcessHandle, + desired_access: u32, + handle_attributes: u32, + token_handle: MutPtr, + ) -> NtStatus { + self.open_process_token( + process_handle, + desired_access, + handle_attributes, + token_handle, + ) + } + + fn open_process_token( + &self, + process_handle: ProcessHandle, + desired_access: u32, + handle_attributes: u32, + token_handle: MutPtr, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(token_handle) { + return status; + } + let Some(attributes) = HandleAttributes::from_token_open_attributes(handle_attributes) + else { + return NtStatus::INVALID_PARAMETER; + }; + if !process_handle.is_current() { + // TODO(token-cross-process): Resolve real process handles once the sandbox supports + // multiple guest processes and per-process primary tokens. + return NtStatus::INVALID_HANDLE; + } + + let handle = match self.insert_typed_handle_with_attributes::( + TokenHandleObject { + token: self.process.token.clone(), + }, + TokenAccess::from_desired_access(desired_access).bits(), + attributes, + drop, + ) { + Ok(handle) => handle, + Err(status) => return status, + }; + if token_handle.write_at_offset(0, handle).is_none() { + self.close_token_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_information_token( + &self, + token_handle: Handle, + token_information_class: u32, + token_information: MutPtr, + token_information_length: u32, + return_length: MutPtr, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(return_length) { + return status; + } + let Ok(class) = TokenInformationClass::try_from(token_information_class) else { + return NtStatus::INVALID_INFO_CLASS; + }; + if let Err(status) = probe_guest_output_buffer::( + token_information, + token_information_length as usize, + ) { + return status; + } + + let entry = match self.typed_handle_entry_with_access::( + token_handle, + TokenAccess::QUERY.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + + match class { + TokenInformationClass::User => entry.with_entry(|entry| { + Self::write_token_information_value( + token_information, + token_information_length, + return_length, + || { + let sid_address = token_information + .as_usize() + .checked_add(size_of::()) + .ok_or(NtStatus::ACCESS_VIOLATION)?; + Ok(TokenUserInformation { + user: TokenUser { + user: SidAndAttributes { + sid: sid_address, + attributes: 0, + padding: 0, + }, + }, + sid: entry.token.user, + }) + }, + ) + }), + TokenInformationClass::Privileges => Self::write_token_information_value( + token_information, + token_information_length, + return_length, + || Ok(TokenPrivileges { privilege_count: 0 }), + ), + TokenInformationClass::Statistics => entry.with_entry(|entry| { + Self::write_token_information_value( + token_information, + token_information_length, + return_length, + || Ok(entry.token.statistics), + ) + }), + TokenInformationClass::SecurityAttributes => entry.with_entry(|entry| { + Self::write_token_security_attributes( + &entry.token.security_attributes, + token_information, + token_information_length, + return_length, + ) + }), + _ => { + // TODO(token-model): Add each information class when its backing token state is + // modeled; do not synthesize security-sensitive token data. + NtStatus::NOT_IMPLEMENTED + } + } + } + + pub(crate) fn sys_nt_query_security_attributes_token( + &self, + token_handle: Handle, + attributes: ConstPtr, + number_of_attributes: u32, + buffer: MutPtr, + length: u32, + return_length: MutPtr, + ) -> NtStatus { + if let Err(status) = probe_guest_output_preserving_value::(return_length) { + return status; + } + if number_of_attributes != 0 && attributes.as_usize() == 0 { + return NtStatus::INVALID_PARAMETER; + } + + let requested_names = + match Self::read_security_attribute_names(attributes, number_of_attributes) { + Ok(names) => names, + Err(status) => return status, + }; + + let query_attributes = |security_attributes: &[TokenSecurityAttribute]| { + let selected = + match Self::select_security_attributes(security_attributes, &requested_names) { + Ok(selected) => selected, + Err(status) => { + if return_length.write_at_offset(0, 0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + return status; + } + }; + Self::write_token_security_attributes(&selected, buffer, length, return_length) + }; + + if token_handle == Self::CURRENT_PROCESS_TOKEN + || token_handle == Self::CURRENT_THREAD_EFFECTIVE_TOKEN + { + return query_attributes(&self.process.token.security_attributes); + } + if token_handle == Self::CURRENT_THREAD_TOKEN { + return NtStatus::NO_TOKEN; + } + + let entry = match self.typed_handle_entry_with_access::( + token_handle, + TokenAccess::QUERY.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + entry.with_entry(|entry| query_attributes(&entry.token.security_attributes)) + } + + fn read_security_attribute_names( + attributes: ConstPtr, + number_of_attributes: u32, + ) -> Result, NtStatus> { + let mut names = Vec::new(); + for index in 0..number_of_attributes as usize { + let name = attributes + .read_at_offset(index.cast_signed()) + .ok_or(NtStatus::ACCESS_VIOLATION)? + .read_string::()?; + names.push(name); + } + Ok(names) + } + + fn select_security_attributes<'a>( + security_attributes: &'a [TokenSecurityAttribute], + requested_names: &[alloc::string::String], + ) -> Result, NtStatus> { + if requested_names.is_empty() { + return Ok(security_attributes.iter().collect()); + } + + let mut selected = Vec::new(); + for requested_name in requested_names { + // TODO(token-security-attribute-casefold): Use Windows invariant Unicode + // case-folding once non-ASCII attribute names are modeled. + let Some(attribute) = security_attributes.iter().find(|attribute| { + requested_name + .encode_utf16() + .eq(attribute.name.iter().copied()) + || requested_name.eq_ignore_ascii_case( + &alloc::string::String::from_utf16_lossy(&attribute.name), + ) + }) else { + return Err(NtStatus::NOT_FOUND); + }; + selected.push(attribute); + } + Ok(selected) + } + + fn write_token_security_attributes>( + security_attributes: &[S], + buffer: MutPtr, + length: u32, + return_length: MutPtr, + ) -> NtStatus { + let Some(attribute_bytes) = + size_of::().checked_mul(security_attributes.len()) + else { + return NtStatus::INVALID_PARAMETER; + }; + let Some(mut required_length) = + size_of::().checked_add(attribute_bytes) + else { + return NtStatus::INVALID_PARAMETER; + }; + for attribute in security_attributes { + let attribute = attribute.borrow(); + let Some(name_bytes) = attribute.name.len().checked_mul(size_of::()) else { + return NtStatus::INVALID_PARAMETER; + }; + required_length = match required_length.checked_add(name_bytes) { + Some(length) => length, + None => return NtStatus::INVALID_PARAMETER, + }; + } + let Ok(required_length_u32) = u32::try_from(required_length) else { + return NtStatus::INVALID_PARAMETER; + }; + if return_length + .write_at_offset(0, required_length_u32) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if length < required_length_u32 { + return NtStatus::BUFFER_TOO_SMALL; + } + if let Err(status) = probe_guest_output_buffer::(buffer, required_length) { + return status; + } + + let attribute_v1 = if security_attributes.is_empty() { + 0 + } else { + match buffer + .as_usize() + .checked_add(size_of::()) + { + Some(address) => address, + None => return NtStatus::ACCESS_VIOLATION, + } + }; + let information = TokenSecurityAttributesInformation { + version: TokenSecurityAttributesInformationVersion::V1 as u16, + reserved: 0, + attribute_count: security_attributes.len().trunc(), + attribute_v1, + }; + if buffer + .write_slice_at_offset(0, information.as_bytes()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + let mut name_offset = size_of::() + attribute_bytes; + let attribute_buffer = + MutPtr::::from_usize(attribute_v1); + for (index, attribute) in security_attributes.iter().enumerate() { + let attribute = attribute.borrow(); + let Some(name_length) = attribute.name.len().checked_mul(size_of::()) else { + return NtStatus::INVALID_PARAMETER; + }; + let Ok(name_length_u16) = u16::try_from(name_length) else { + return NtStatus::INVALID_PARAMETER; + }; + let Some(name_address) = buffer.as_usize().checked_add(name_offset) else { + return NtStatus::ACCESS_VIOLATION; + }; + let information = TokenSecurityAttributeV1 { + name: UnicodeString { + length: name_length_u16, + maximum_length: name_length_u16, + padding_0: [0; 4], + buffer: name_address, + }, + value_type: attribute.value_type as u16, + reserved: 0, + flags: attribute.flags.bits(), + value_count: 0, + padding: 0, + values: 0, + }; + if attribute_buffer + .write_at_offset(index.cast_signed(), information) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + if buffer + .write_slice_at_offset(name_offset.cast_signed(), attribute.name.as_bytes()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + name_offset += name_length; + } + + // TODO(token-security-attribute-values): Store and serialize typed V1 values when + // NtCreateTokenEx or NtSetInformationToken can populate token security attributes. + // TODO(token-security-attribute-set): Implement TokenSecurityAttributes mutation with + // SeTcbPrivilege enforcement when NtSetInformationToken is added. + // TODO(token-security-attribute-duplicate): Deep-copy attributes when NtDuplicateToken + // creates distinct token objects. + NtStatus::SUCCESS + } + + fn write_token_information_value( + token_information: MutPtr, + token_information_length: u32, + return_length: MutPtr, + build_information: impl FnOnce() -> Result, + ) -> NtStatus { + let required_length = size_of::().trunc(); + if return_length.write_at_offset(0, required_length).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + if token_information_length < required_length { + return NtStatus::BUFFER_TOO_SMALL; + } + let information = match build_information() { + Ok(information) => information, + Err(status) => return status, + }; + if token_information + .write_slice_at_offset(0, information.as_bytes()) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn close_token_handle(&self, handle: Handle) { + self.close_typed_handle::(handle, drop); + } + + pub(crate) fn close_token(token: TokenHandleObject) { + drop(token); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{ + const_ptr, mut_byte_ptr, mut_ptr, null_const_ptr, null_mut_ptr, test_task, unicode_string, + utf16_units, + }; + + #[repr(C)] + #[derive(Clone, Copy, Debug, Eq, FromBytes, Immutable, IntoBytes, PartialEq)] + struct TokenUserBuffer { + user: TokenUser, + sid: Sid, + padding: u32, + } + + #[test] + fn open_and_query_process_token_identity() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_process_token( + ProcessHandle::CURRENT, + TokenAccess::QUERY.bits(), + mut_ptr(&mut handle), + ), + NtStatus::SUCCESS + ); + + let mut required_length = 0; + assert_eq!( + task.sys_nt_query_information_token( + handle, + TokenInformationClass::User as u32, + null_mut_ptr(), + 0, + mut_ptr(&mut required_length), + ), + NtStatus::BUFFER_TOO_SMALL + ); + assert_eq!( + required_length as usize, + size_of::() + size_of::() + ); + + let mut output = TokenUserBuffer { + user: TokenUser { + user: SidAndAttributes { + sid: 0, + attributes: u32::MAX, + padding: 0, + }, + }, + sid: Sid { + revision: 0, + sub_authority_count: 0, + identifier_authority: [0; 6], + sub_authority: [0], + }, + padding: 0, + }; + assert_eq!( + task.sys_nt_query_information_token( + handle, + TokenInformationClass::User as u32, + mut_byte_ptr(&mut output), + required_length, + mut_ptr(&mut required_length), + ), + NtStatus::SUCCESS + ); + assert_eq!( + output.user.user.sid, + core::ptr::from_ref(&output.sid) as usize + ); + assert_eq!(output.user.user.attributes, 0); + assert_eq!(output.sid, LOCAL_SYSTEM_SID); + } + + #[test] + fn named_security_attribute_queries_are_case_insensitive() { + let mut task = test_task(); + let process = Arc::get_mut(&mut task.process).expect("test task must own its process"); + let token = Arc::get_mut(&mut process.token).expect("test process must own its token"); + token.security_attributes = alloc::vec![TokenSecurityAttribute { + name: utf16_units("LITEBOX://TestAttribute").into_boxed_slice(), + value_type: TokenSecurityAttributeValueType::Uint64, + flags: TokenSecurityAttributeFlags::MANDATORY, + }] + .into_boxed_slice(); + + let requested_name = utf16_units("litebox://testattribute"); + let requested_name = unicode_string(&requested_name); + let mut output = [0_u8; 128]; + let mut return_length = 0; + assert_eq!( + task.sys_nt_query_security_attributes_token( + Task::::CURRENT_PROCESS_TOKEN, + const_ptr(&requested_name), + 1, + mut_byte_ptr(&mut output), + output.len().trunc(), + mut_ptr(&mut return_length), + ), + NtStatus::SUCCESS + ); + + let output_address = output.as_ptr() as usize; + let information = TokenSecurityAttributesInformation::read_from_prefix(&output) + .expect("valid header") + .0; + assert_eq!(information.version, 1); + assert_eq!(information.attribute_count, 1); + assert_eq!( + information.attribute_v1, + output_address + size_of::() + ); + let attribute = TokenSecurityAttributeV1::read_from_prefix( + &output[size_of::()..], + ) + .expect("valid attribute") + .0; + assert_eq!( + attribute.value_type, + TokenSecurityAttributeValueType::Uint64 as u16 + ); + assert_eq!( + attribute.flags, + TokenSecurityAttributeFlags::MANDATORY.bits() + ); + assert_eq!(attribute.value_count, 0); + assert_eq!(attribute.values, 0); + assert_eq!( + attribute.name.buffer, + output_address + + size_of::() + + size_of::() + ); + } + + #[test] + fn named_security_attribute_query_reports_missing_names() { + let task = test_task(); + let requested_name = utf16_units("LITEBOX://Missing"); + let requested_name = unicode_string(&requested_name); + let mut return_length = u32::MAX; + + assert_eq!( + task.sys_nt_query_security_attributes_token( + Task::::CURRENT_PROCESS_TOKEN, + const_ptr(&requested_name), + 1, + null_mut_ptr(), + 0, + mut_ptr(&mut return_length), + ), + NtStatus::NOT_FOUND + ); + assert_eq!(return_length, 0); + } + + #[test] + fn query_security_attributes_enforces_query_access() { + let task = test_task(); + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_open_process_token( + ProcessHandle::CURRENT, + TokenAccess::DUPLICATE.bits(), + mut_ptr(&mut handle), + ), + NtStatus::SUCCESS + ); + let mut output = TokenSecurityAttributesInformation { + version: 0, + reserved: 0, + attribute_count: 0, + attribute_v1: 0, + }; + let mut return_length = 0; + + assert_eq!( + task.sys_nt_query_security_attributes_token( + handle, + null_const_ptr(), + 0, + mut_byte_ptr(&mut output), + size_of::().trunc(), + mut_ptr(&mut return_length), + ), + NtStatus::ACCESS_DENIED + ); + } + + #[test] + fn query_security_attributes_handles_token_pseudo_handles() { + let task = test_task(); + let mut output = TokenSecurityAttributesInformation { + version: 0, + reserved: 0, + attribute_count: 0, + attribute_v1: 0, + }; + let mut return_length = 0; + + assert_eq!( + task.sys_nt_query_security_attributes_token( + Task::::CURRENT_THREAD_TOKEN, + null_const_ptr(), + 0, + mut_byte_ptr(&mut output), + size_of::().trunc(), + mut_ptr(&mut return_length), + ), + NtStatus::NO_TOKEN + ); + assert_eq!( + task.sys_nt_query_security_attributes_token( + Task::::CURRENT_THREAD_EFFECTIVE_TOKEN, + null_const_ptr(), + 0, + mut_byte_ptr(&mut output), + size_of::().trunc(), + mut_ptr(&mut return_length), + ), + NtStatus::SUCCESS + ); + } +} diff --git a/litebox_shim_windows/src/syscalls/wait_completion_packet.rs b/litebox_shim_windows/src/syscalls/wait_completion_packet.rs new file mode 100644 index 0000000000..9b9d4da1fb --- /dev/null +++ b/litebox_shim_windows/src/syscalls/wait_completion_packet.rs @@ -0,0 +1,1404 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT wait completion packet syscalls. + +use alloc::sync::Arc; +use core::marker::PhantomData; + +use litebox::fd::{ErrRawIntFd, FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawMutPointer as _, RawPointerProvider}; +use litebox::sync::Mutex; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::{AccessMask, ObjectAttributes, read_object_attributes}; +use crate::syscalls::Handle; +use crate::syscalls::event::{EventHandleObject, EventSubsystem}; +use crate::syscalls::iocp::{IoCompletionAccess, IoCompletionSubsystem}; +use crate::syscalls::timer::TimerSubsystem; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +const STANDARD_RIGHTS_REQUIRED: u32 = AccessMask::DELETE.bits() + | AccessMask::READ_CONTROL.bits() + | AccessMask::WRITE_DAC.bits() + | AccessMask::WRITE_OWNER.bits(); + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct WaitCompletionPacketAccess: u32 { + const SET_STATE = 0x0001; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::SET_STATE.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() | Self::SET_STATE.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() | Self::SET_STATE.bits(); + const ALL_ACCESS = STANDARD_RIGHTS_REQUIRED | Self::SET_STATE.bits(); + + const _ = !0; + } +} + +impl WaitCompletionPacketAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +pub(crate) struct WaitCompletionPacketSubsystem(PhantomData); + +impl FdEnabledSubsystem for WaitCompletionPacketSubsystem { + type Entry = WaitCompletionPacketHandleObject; +} + +impl FdEnabledSubsystemEntry + for WaitCompletionPacketHandleObject +{ +} + +impl crate::WindowsHandleSubsystem + for WaitCompletionPacketSubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + WaitCompletionPacketAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct WaitCompletionPacketHandleObject { + packet: Arc>, +} + +pub(crate) struct WaitCompletionPacketObject { + association: Mutex>, + _not_send_without_platform: PhantomData, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +struct WaitCompletionPacketAssociation { + _key_context: usize, + _apc_context: usize, + _io_status: i32, + _io_status_information: usize, + already_signaled: bool, +} + +pub(crate) struct WaitCompletionPacketAssociateParameters { + pub(crate) wait_completion_packet_handle: Handle, + pub(crate) io_completion_handle: Handle, + pub(crate) target_object_handle: Handle, + pub(crate) key_context: usize, + pub(crate) apc_context: usize, + pub(crate) io_status: i32, + pub(crate) io_status_information: usize, + pub(crate) already_signaled: Option>, +} + +fn validate_wait_completion_packet_object_attributes( + object_attributes: Option>, +) -> Result<(), NtStatus> { + let Some(object_attributes) = object_attributes else { + return Ok(()); + }; + let object_attributes = read_object_attributes::(object_attributes)?; + if object_attributes.object_name == 0 && !object_attributes.root_directory.is_null() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(()) +} + +impl Task { + fn wait_completion_packet_entry( + &self, + handle: Handle, + ) -> Result>, NtStatus> + { + let Some(raw_fd) = handle.raw_fd() else { + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + }; + let typed = { + let handles = self.process.handles.read(); + match handles.fd_from_raw_integer::>(raw_fd) { + Ok(typed) => typed, + Err(ErrRawIntFd::NotFound | ErrRawIntFd::InvalidSubsystem) => { + return Err(NtStatus::OBJECT_TYPE_MISMATCH); + } + } + }; + self.require_typed_handle_access(&typed, WaitCompletionPacketAccess::SET_STATE.bits())?; + self.global + .litebox + .descriptor_table() + .entry_handle(&typed) + .ok_or(NtStatus::OBJECT_TYPE_MISMATCH) + } + + fn wait_completion_packet_entry_for_cancel( + &self, + handle: Handle, + ) -> Result>, NtStatus> + { + let Some(raw_fd) = handle.raw_fd() else { + return Err(NtStatus::INVALID_HANDLE); + }; + let typed = { + let handles = self.process.handles.read(); + match handles.fd_from_raw_integer::>(raw_fd) { + Ok(typed) => typed, + Err(ErrRawIntFd::NotFound) => return Err(NtStatus::INVALID_HANDLE), + Err(ErrRawIntFd::InvalidSubsystem) => return Err(NtStatus::OBJECT_TYPE_MISMATCH), + } + }; + self.require_typed_handle_access(&typed, WaitCompletionPacketAccess::SET_STATE.bits())?; + self.global + .litebox + .descriptor_table() + .entry_handle(&typed) + .ok_or(NtStatus::INVALID_HANDLE) + } + + fn validate_io_completion_for_wait_completion_packet( + &self, + handle: Handle, + ) -> Result<(), NtStatus> { + self.require_handle_access::>( + handle, + IoCompletionAccess::MODIFY_STATE.bits(), + ) + .map_err(|status| match status { + NtStatus::INVALID_HANDLE | NtStatus::OBJECT_TYPE_MISMATCH => { + NtStatus::OBJECT_TYPE_MISMATCH + } + status => status, + }) + } + + fn target_object_signaled_for_wait_completion_packet( + &self, + handle: Handle, + ) -> Result { + // TODO: support every waitable target object type that Windows accepts here; the + // current subset only recognizes event and timer handles. + let Some(raw_fd) = handle.raw_fd() else { + return Err(NtStatus::ACCESS_DENIED); + }; + if let Some(signaled) = self.event_signaled_for_wait_completion_packet(raw_fd)? { + return Ok(signaled); + } + if let Some(signaled) = self.timer_signaled_for_wait_completion_packet(raw_fd)? { + return Ok(signaled); + } + Err(NtStatus::INVALID_PARAMETER_3) + } + + fn event_signaled_for_wait_completion_packet( + &self, + raw_fd: usize, + ) -> Result, NtStatus> { + let typed = { + let handles = self.process.handles.read(); + match handles.fd_from_raw_integer::>(raw_fd) { + Ok(typed) => typed, + Err(ErrRawIntFd::NotFound) => return Err(NtStatus::ACCESS_DENIED), + Err(ErrRawIntFd::InvalidSubsystem) => return Ok(None), + } + }; + let Some(entry) = self.global.litebox.descriptor_table().entry_handle(&typed) else { + return Err(NtStatus::ACCESS_DENIED); + }; + self.require_typed_handle_access::>( + &typed, + AccessMask::SYNCHRONIZE.bits(), + )?; + Ok(Some(entry.with_entry(EventHandleObject::is_signaled))) + } + + fn timer_signaled_for_wait_completion_packet( + &self, + raw_fd: usize, + ) -> Result, NtStatus> { + let handle = Handle::from_raw_fd(raw_fd).ok_or(NtStatus::ACCESS_DENIED)?; + match self.require_handle_access::>( + handle, + AccessMask::SYNCHRONIZE.bits(), + ) { + Ok(()) => {} + Err(NtStatus::OBJECT_TYPE_MISMATCH) => return Ok(None), + Err(NtStatus::INVALID_HANDLE) => return Err(NtStatus::ACCESS_DENIED), + Err(status) => return Err(status), + } + // TODO: return the timer object's real signaled state after NtSetTimer2 models + // due-time expiration and periodic re-signaling. + Ok(Some(false)) + } + + fn insert_wait_completion_packet_handle( + &self, + packet: Arc>, + granted_access: WaitCompletionPacketAccess, + ) -> Result { + self.insert_typed_handle::>( + WaitCompletionPacketHandleObject { packet }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_wait_completion_packet_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_wait_completion_packet( + wait_completion_packet: WaitCompletionPacketHandleObject, + ) { + drop(wait_completion_packet); + } + + pub(crate) fn sys_nt_create_wait_completion_packet( + &self, + wait_completion_packet_handle: MutPtr, + desired_access: u32, + object_attributes: Option>, + ) -> NtStatus { + if let Err(status) = + probe_guest_output_preserving_value::(wait_completion_packet_handle) + { + return status; + } + if let Err(status) = + validate_wait_completion_packet_object_attributes::(object_attributes) + { + return status; + } + + let packet = Arc::new(WaitCompletionPacketObject { + association: Mutex::new(None), + _not_send_without_platform: PhantomData, + }); + let granted_access = WaitCompletionPacketAccess::from_desired_access(desired_access); + let Ok(handle) = self.insert_wait_completion_packet_handle(packet, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if wait_completion_packet_handle + .write_at_offset(0, handle) + .is_none() + { + self.close_wait_completion_packet_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_associate_wait_completion_packet( + &self, + params: WaitCompletionPacketAssociateParameters, + ) -> NtStatus { + let entry = match self.wait_completion_packet_entry(params.wait_completion_packet_handle) { + Ok(entry) => entry, + Err(status) => return status, + }; + let packet = entry.with_entry(|entry| entry.packet.clone()); + + if let Err(status) = + self.validate_io_completion_for_wait_completion_packet(params.io_completion_handle) + { + return status; + } + { + let association = packet.association.lock(); + if association.is_some() { + return NtStatus::INVALID_PARAMETER_1; + } + } + let already_signaled = match self + .target_object_signaled_for_wait_completion_packet(params.target_object_handle) + { + Ok(already_signaled) => already_signaled, + Err(status) => return status, + }; + if let Some(already_signaled_ptr) = params.already_signaled + && already_signaled_ptr + .write_at_offset(0, u8::from(already_signaled)) + .is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + + // TODO: link the packet into the target object's wait notification path and post to + // the associated IOCP when the target is already signaled or becomes signaled later. + let mut association = packet.association.lock(); + if association.is_some() { + return NtStatus::INVALID_PARAMETER_1; + } + *association = Some(WaitCompletionPacketAssociation { + _key_context: params.key_context, + _apc_context: params.apc_context, + _io_status: params.io_status, + _io_status_information: params.io_status_information, + already_signaled, + }); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_cancel_wait_completion_packet( + &self, + wait_completion_packet_handle: Handle, + remove_signaled_packet: u8, + ) -> NtStatus { + let entry = + match self.wait_completion_packet_entry_for_cancel(wait_completion_packet_handle) { + Ok(entry) => entry, + Err(status) => return status, + }; + let packet = entry.with_entry(|entry| Arc::clone(&entry.packet)); + + let mut association = packet.association.lock(); + let Some(current_association) = *association else { + return NtStatus::CANCELLED; + }; + if current_association.already_signaled && remove_signaled_packet == 0 { + return NtStatus::PENDING; + } + + // TODO: if a signaled packet has been posted to the IOCP, honor + // remove_signaled_packet by removing that queued completion packet. + *association = None; + NtStatus::SUCCESS + } +} + +#[cfg(test)] +mod tests { + use core::mem::size_of; + + use litebox::platform::{RawConstPointer as _, ThreadProvider}; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::ObjectAttributes; + use crate::tests::{TestPlatform, const_ptr, mut_ptr, test_platform, test_task}; + + const WAIT_COMPLETION_PACKET_SET_STATE: u32 = 0x0000_0001; + const WAIT_COMPLETION_PACKET_ALL_ACCESS: u32 = 0x000f_0001; + const IO_COMPLETION_QUERY_STATE: u32 = 0x0000_0001; + const IO_COMPLETION_ALL_ACCESS: u32 = 0x001f_0003; + const EVENT_QUERY_STATE: u32 = 0x0000_0001; + const EVENT_ALL_ACCESS: u32 = 0x001f_0003; + const SYNCHRONIZE: u32 = 0x0010_0000; + + fn object_attributes_size() -> u32 { + u32::try_from(size_of::()).expect("OBJECT_ATTRIBUTES fits in ULONG") + } + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = test_platform(); + ::run_test_thread(f) + } + + fn create_wait_completion_packet( + task: &Task, + handle: &mut Handle, + object_attributes: Option>, + ) -> NtStatus { + task.sys_nt_create_wait_completion_packet( + mut_ptr(handle), + WAIT_COMPLETION_PACKET_ALL_ACCESS, + object_attributes, + ) + } + + fn create_wait_completion_packet_with_access( + task: &Task, + handle: &mut Handle, + desired_access: u32, + ) -> NtStatus { + task.sys_nt_create_wait_completion_packet(mut_ptr(handle), desired_access, None) + } + + fn create_io_completion( + task: &Task, + handle: &mut Handle, + desired_access: u32, + ) -> NtStatus { + task.sys_nt_create_io_completion(mut_ptr(handle), desired_access, None, 0) + } + + fn create_event( + task: &Task, + handle: &mut Handle, + desired_access: u32, + initial_state: bool, + ) -> NtStatus { + task.sys_nt_create_event( + mut_ptr(handle), + desired_access, + None, + 0, + u8::from(initial_state), + ) + } + + fn create_timer( + task: &Task, + handle: &mut Handle, + desired_access: u32, + ) -> NtStatus { + task.sys_nt_create_timer2(crate::syscalls::timer::TimerCreateParameters { + timer_handle: mut_ptr(handle), + timer_id: None, + object_attributes: None, + attributes: 0, + desired_access, + }) + } + + fn associate_wait_completion_packet( + task: &Task, + packet: Handle, + io_completion: Handle, + target: Handle, + already_signaled: Option>, + ) -> NtStatus { + task.sys_nt_associate_wait_completion_packet(WaitCompletionPacketAssociateParameters { + wait_completion_packet_handle: packet, + io_completion_handle: io_completion, + target_object_handle: target, + key_context: 0x1111, + apc_context: 0x2222, + io_status: NtStatus::SUCCESS.as_raw(), + io_status_information: 0x3333, + already_signaled, + }) + } + + fn cancel_wait_completion_packet( + task: &Task, + packet: Handle, + remove_signaled_packet: bool, + ) -> NtStatus { + task.sys_nt_cancel_wait_completion_packet(packet, u8::from(remove_signaled_packet)) + } + + #[test] + fn create_validates_object_attributes_without_clobbering_output() { + let task = test_task(); + let mut handle = Handle::from_raw(usize::MAX); + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + assert_eq!( + create_wait_completion_packet(&task, &mut handle, Some(const_ptr(&bad_length))), + NtStatus::INVALID_PARAMETER + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + + let root_without_name = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::from_raw(4), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + assert_eq!( + create_wait_completion_packet(&task, &mut handle, Some(const_ptr(&root_without_name))), + NtStatus::OBJECT_NAME_INVALID + ); + assert_eq!(handle, Handle::from_raw(usize::MAX)); + } + + #[test] + fn associate_writes_signal_state_and_marks_packet_busy() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 0); + + already_signaled = 0xaa; + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::INVALID_PARAMETER_1 + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + Handle::from_raw(0x1234), + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::INVALID_PARAMETER_1 + ); + assert_eq!(already_signaled, 0xaa); + } + + #[test] + fn associate_reports_already_signaled_target() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, true), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 1); + } + + #[test] + fn associate_accepts_timer_target_as_unsignaled() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut timer = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_timer(&task, &mut timer, SYNCHRONIZE), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + timer, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 0); + } + + #[test] + fn associate_invalid_already_signaled_does_not_commit_association() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, true), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(MutPtr::::from_usize(1)), + ), + NtStatus::ACCESS_VIOLATION + ); + assert_eq!( + cancel_wait_completion_packet(&task, packet, true), + NtStatus::CANCELLED + ); + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 1); + }); + } + + #[test] + fn associate_enforces_native_observed_access_masks() { + let task = test_task(); + let mut packet = Handle::default(); + let mut packet_without_set_state = Handle::default(); + let mut io_completion = Handle::default(); + let mut io_completion_query_only = Handle::default(); + let mut event = Handle::default(); + let mut event_without_synchronize = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet_with_access( + &task, + &mut packet, + WAIT_COMPLETION_PACKET_SET_STATE, + ), + NtStatus::SUCCESS + ); + assert_eq!( + create_wait_completion_packet_with_access(&task, &mut packet_without_set_state, 0), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion( + &task, + &mut io_completion_query_only, + IO_COMPLETION_QUERY_STATE, + ), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, SYNCHRONIZE, false), + NtStatus::SUCCESS + ); + assert_eq!( + create_event( + &task, + &mut event_without_synchronize, + EVENT_QUERY_STATE, + false, + ), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet_without_set_state, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion_query_only, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event_without_synchronize, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 0); + } + + #[test] + fn associate_distinguishes_handle_errors_like_native_windows() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut target_io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut target_io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + Handle::from_raw(0x1234), + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + Handle::from_raw(0x1234), + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + Handle::from_raw(0x1234), + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + target_io_completion, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::INVALID_PARAMETER_3 + ); + assert_eq!(already_signaled, 0xaa); + } + + #[test] + fn cancel_clears_unsignaled_association_and_allows_reuse() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!( + cancel_wait_completion_packet(&task, packet, false), + NtStatus::SUCCESS + ); + assert_eq!( + cancel_wait_completion_packet(&task, packet, false), + NtStatus::CANCELLED + ); + + already_signaled = 0xaa; + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 0); + } + + #[test] + fn cancel_signaled_packet_obeys_remove_signaled_packet() { + let task = test_task(); + let mut packet = Handle::default(); + let mut io_completion = Handle::default(); + let mut event = Handle::default(); + let mut already_signaled = 0xaa; + + assert_eq!( + create_wait_completion_packet(&task, &mut packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_io_completion(&task, &mut io_completion, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, true), + NtStatus::SUCCESS + ); + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::SUCCESS + ); + assert_eq!(already_signaled, 1); + + assert_eq!( + cancel_wait_completion_packet(&task, packet, false), + NtStatus::PENDING + ); + already_signaled = 0xaa; + assert_eq!( + associate_wait_completion_packet( + &task, + packet, + io_completion, + event, + Some(mut_ptr(&mut already_signaled)), + ), + NtStatus::INVALID_PARAMETER_1 + ); + assert_eq!(already_signaled, 0xaa); + + assert_eq!( + cancel_wait_completion_packet(&task, packet, true), + NtStatus::SUCCESS + ); + assert_eq!( + cancel_wait_completion_packet(&task, packet, true), + NtStatus::CANCELLED + ); + } + + #[test] + fn cancel_distinguishes_handle_errors_and_requires_set_state() { + let task = test_task(); + let mut event = Handle::default(); + let mut packet_without_set_state = Handle::default(); + + assert_eq!( + create_event(&task, &mut event, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + assert_eq!( + create_wait_completion_packet_with_access(&task, &mut packet_without_set_state, 0), + NtStatus::SUCCESS + ); + + assert_eq!( + cancel_wait_completion_packet(&task, Handle::default(), false), + NtStatus::INVALID_HANDLE + ); + assert_eq!( + cancel_wait_completion_packet(&task, Handle::from_raw(0x1234), true), + NtStatus::INVALID_HANDLE + ); + assert_eq!( + cancel_wait_completion_packet(&task, event, false), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!( + cancel_wait_completion_packet(&task, packet_without_set_state, false), + NtStatus::ACCESS_DENIED + ); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_cancel_wait_completion_packet_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateWaitCompletionPacket( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + ) -> i32; + fn NtCancelWaitCompletionPacket(handle: *mut c_void, remove_signaled_packet: u8) + -> i32; + fn NtAssociateWaitCompletionPacket( + packet: *mut c_void, + io_completion: *mut c_void, + target: *mut c_void, + key_context: *mut c_void, + apc_context: *mut c_void, + io_status: i32, + io_status_information: usize, + already_signaled: *mut u8, + ) -> i32; + fn NtCreateIoCompletion( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + number_of_concurrent_threads: u32, + ) -> i32; + fn NtCreateEvent( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + event_type: u32, + initial_state: u8, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + unsafe fn close_host(handle: *mut c_void) { + if !handle.is_null() { + // SAFETY: The caller passes a live host handle returned by an NtCreate* call. + assert_eq!(unsafe { NtClose(handle) }, NtStatus::SUCCESS.as_raw()); + } + } + + let task = test_task(); + + // SAFETY: The null handle is an input-only value and no memory is dereferenced. + let host_null = unsafe { NtCancelWaitCompletionPacket(core::ptr::null_mut(), 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, Handle::default(), false).as_raw(), + host_null + ); + + let mut host_event = core::ptr::null_mut(); + // SAFETY: The output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateEvent( + &raw mut host_event, + EVENT_ALL_ACCESS, + core::ptr::null(), + 0, + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut shim_event = Handle::default(); + assert_eq!( + create_event(&task, &mut shim_event, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + // SAFETY: The host event handle is valid for the duration of this call. + let host_wrong_type = unsafe { NtCancelWaitCompletionPacket(host_event, 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_event, false).as_raw(), + host_wrong_type + ); + // SAFETY: The event handle was returned by NtCreateEvent in this test. + unsafe { close_host(host_event) }; + + let mut host_packet = core::ptr::null_mut(); + // SAFETY: The output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateWaitCompletionPacket( + &raw mut host_packet, + WAIT_COMPLETION_PACKET_ALL_ACCESS, + core::ptr::null(), + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut shim_packet = Handle::default(); + assert_eq!( + create_wait_completion_packet(&task, &mut shim_packet, None), + NtStatus::SUCCESS + ); + // SAFETY: The host packet handle is valid for the duration of this call. + let host_unassociated = unsafe { NtCancelWaitCompletionPacket(host_packet, 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_packet, false).as_raw(), + host_unassociated + ); + // SAFETY: The packet handle was returned by NtCreateWaitCompletionPacket in this test. + unsafe { close_host(host_packet) }; + + let mut host_packet_no_set = core::ptr::null_mut(); + // SAFETY: The output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateWaitCompletionPacket(&raw mut host_packet_no_set, 0, core::ptr::null()) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut shim_packet_no_set = Handle::default(); + assert_eq!( + create_wait_completion_packet_with_access(&task, &mut shim_packet_no_set, 0), + NtStatus::SUCCESS + ); + // SAFETY: The host packet handle is valid for the duration of this call. + let host_no_set = unsafe { NtCancelWaitCompletionPacket(host_packet_no_set, 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_packet_no_set, false).as_raw(), + host_no_set + ); + // SAFETY: The packet handle was returned by NtCreateWaitCompletionPacket in this test. + unsafe { close_host(host_packet_no_set) }; + + let mut host_iocp = core::ptr::null_mut(); + let mut host_assoc_packet = core::ptr::null_mut(); + let mut host_target = core::ptr::null_mut(); + // SAFETY: Output pointers are valid and successful handles are closed below. + assert_eq!( + unsafe { + NtCreateIoCompletion( + &raw mut host_iocp, + IO_COMPLETION_ALL_ACCESS, + core::ptr::null(), + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + // SAFETY: Output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateWaitCompletionPacket( + &raw mut host_assoc_packet, + WAIT_COMPLETION_PACKET_ALL_ACCESS, + core::ptr::null(), + ) + }, + NtStatus::SUCCESS.as_raw() + ); + // SAFETY: Output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateEvent( + &raw mut host_target, + EVENT_ALL_ACCESS, + core::ptr::null(), + 0, + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut host_already_signaled = 0xaa; + // SAFETY: All handles are valid and the output byte points to local storage. + assert_eq!( + unsafe { + NtAssociateWaitCompletionPacket( + host_assoc_packet, + host_iocp, + host_target, + core::ptr::null_mut(), + core::ptr::null_mut(), + 0, + 0, + &raw mut host_already_signaled, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut shim_iocp = Handle::default(); + let mut shim_assoc_packet = Handle::default(); + let mut shim_target = Handle::default(); + let mut shim_already_signaled = 0xaa; + assert_eq!( + create_io_completion(&task, &mut shim_iocp, IO_COMPLETION_ALL_ACCESS), + NtStatus::SUCCESS + ); + assert_eq!( + create_wait_completion_packet(&task, &mut shim_assoc_packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut shim_target, EVENT_ALL_ACCESS, false), + NtStatus::SUCCESS + ); + assert_eq!( + associate_wait_completion_packet( + &task, + shim_assoc_packet, + shim_iocp, + shim_target, + Some(mut_ptr(&mut shim_already_signaled)), + ), + NtStatus::SUCCESS + ); + // SAFETY: The host packet handle is valid for the duration of this call. + let host_cancel_unsignaled = unsafe { NtCancelWaitCompletionPacket(host_assoc_packet, 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_assoc_packet, false).as_raw(), + host_cancel_unsignaled + ); + // SAFETY: Handles were returned by NtCreate* calls in this test. + unsafe { + close_host(host_target); + close_host(host_assoc_packet); + close_host(host_iocp); + } + + let mut host_iocp = core::ptr::null_mut(); + let mut host_signaled_packet = core::ptr::null_mut(); + let mut host_signaled_event = core::ptr::null_mut(); + // SAFETY: Output pointers are valid and successful handles are closed below. + assert_eq!( + unsafe { + NtCreateIoCompletion( + &raw mut host_iocp, + IO_COMPLETION_ALL_ACCESS, + core::ptr::null(), + 0, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + // SAFETY: Output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateWaitCompletionPacket( + &raw mut host_signaled_packet, + WAIT_COMPLETION_PACKET_ALL_ACCESS, + core::ptr::null(), + ) + }, + NtStatus::SUCCESS.as_raw() + ); + // SAFETY: Output pointer is valid and the returned handle is closed below. + assert_eq!( + unsafe { + NtCreateEvent( + &raw mut host_signaled_event, + EVENT_ALL_ACCESS, + core::ptr::null(), + 0, + 1, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + host_already_signaled = 0xaa; + // SAFETY: All handles are valid and the output byte points to local storage. + assert_eq!( + unsafe { + NtAssociateWaitCompletionPacket( + host_signaled_packet, + host_iocp, + host_signaled_event, + core::ptr::null_mut(), + core::ptr::null_mut(), + 0, + 0, + &raw mut host_already_signaled, + ) + }, + NtStatus::SUCCESS.as_raw() + ); + let mut shim_signaled_packet = Handle::default(); + let mut shim_signaled_event = Handle::default(); + shim_already_signaled = 0xaa; + assert_eq!( + create_wait_completion_packet(&task, &mut shim_signaled_packet, None), + NtStatus::SUCCESS + ); + assert_eq!( + create_event(&task, &mut shim_signaled_event, EVENT_ALL_ACCESS, true), + NtStatus::SUCCESS + ); + assert_eq!( + associate_wait_completion_packet( + &task, + shim_signaled_packet, + shim_iocp, + shim_signaled_event, + Some(mut_ptr(&mut shim_already_signaled)), + ), + NtStatus::SUCCESS + ); + // SAFETY: The host packet handle is valid for the duration of these calls. + let host_cancel_signaled_pending = + unsafe { NtCancelWaitCompletionPacket(host_signaled_packet, 0) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_signaled_packet, false).as_raw(), + host_cancel_signaled_pending + ); + // SAFETY: The host packet handle is valid for the duration of this call. + let host_cancel_signaled_remove = + unsafe { NtCancelWaitCompletionPacket(host_signaled_packet, 1) }; + assert_eq!( + cancel_wait_completion_packet(&task, shim_signaled_packet, true).as_raw(), + host_cancel_signaled_remove + ); + // SAFETY: Handles were returned by NtCreate* calls in this test. + unsafe { + close_host(host_signaled_event); + close_host(host_signaled_packet); + close_host(host_iocp); + } + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_create_wait_completion_packet_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateWaitCompletionPacket( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + let task = test_task(); + + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + let root_without_name = ObjectAttributes { + length: object_attributes_size(), + root_directory: Handle::from_raw(4), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + + for object_attributes in [None, Some(&bad_length), Some(&root_without_name)] { + let mut host_handle = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, optional attributes reference local values for + // the duration of the call, and successful host handles are closed below. + let host_status = unsafe { + NtCreateWaitCompletionPacket( + &raw mut host_handle, + WAIT_COMPLETION_PACKET_ALL_ACCESS, + object_attributes.map_or(core::ptr::null(), core::ptr::from_ref), + ) + }; + if host_status == NtStatus::SUCCESS.as_raw() && !host_handle.is_null() { + // SAFETY: The handle was returned by NtCreateWaitCompletionPacket in this test. + assert_eq!(unsafe { NtClose(host_handle) }, NtStatus::SUCCESS.as_raw()); + } + + let mut shim_handle = Handle::from_raw(usize::MAX); + let shim_status = create_wait_completion_packet( + &task, + &mut shim_handle, + object_attributes.map(const_ptr), + ); + assert_eq!(shim_status.as_raw(), host_status); + if shim_status == NtStatus::SUCCESS { + assert!(!shim_handle.is_null()); + assert_eq!(task.sys_nt_close(shim_handle), NtStatus::SUCCESS); + } else { + assert_eq!(shim_handle, Handle::from_raw(usize::MAX)); + } + } + } +} diff --git a/litebox_shim_windows/src/syscalls/wnf.rs b/litebox_shim_windows/src/syscalls/wnf.rs new file mode 100644 index 0000000000..8b4f96db4c --- /dev/null +++ b/litebox_shim_windows/src/syscalls/wnf.rs @@ -0,0 +1,559 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +use alloc::collections::BTreeMap; +use alloc::vec::Vec; +use int_enum::IntEnum; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _}; +use litebox::utils::TruncateExt as _; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::Guid; +use crate::{ + ConstPtr, MutPtr, ShimFS, ShimPlatform, Task, probe_guest_output_buffer, + probe_guest_output_preserving_value, +}; + +const MAXIMUM_STATE_SIZE: u32 = 0x1000; +const STATE_NAME_XOR_KEY: u64 = 0x41c6_4e6d_a3bc_0074; +const MAXIMUM_UNIQUE_ID: u32 = 0x001f_ffff; +const STATE_NAME_INFORMATION_SIZE: u32 = 4; +const INITIAL_CHANGE_STAMP: u32 = 0; + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum WnfStateNameLifetime { + WellKnown = 0, + Permanent = 1, + Persistent = 2, + Temporary = 3, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum WnfDataScope { + System = 0, + Session = 1, + User = 2, + Process = 3, + Machine = 4, +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, IntEnum, PartialEq)] +enum WnfStateNameInformation { + Exists = 0, + SubscribersPresent = 1, + IsQuiescent = 2, +} + +#[derive(Clone)] +pub(crate) struct WnfStateData { + change_stamp: u32, + type_id: Option, + data: Vec, + maximum_state_size: u32, + lifetime: WnfStateNameLifetime, +} + +#[derive(Default)] +pub(crate) struct WnfStateStoreData { + next_unique_id: u32, + states: BTreeMap, +} + +pub(crate) type WnfStateStore = litebox::sync::RwLock; + +pub(crate) struct WnfCreateStateNameParameters { + pub(crate) state_name: MutPtr, + pub(crate) name_lifetime: u32, + pub(crate) data_scope: u32, + pub(crate) persist_data: u8, + pub(crate) type_id: Option>, + pub(crate) maximum_state_size: u32, + pub(crate) security_descriptor: ConstPtr, +} + +pub(crate) struct WnfUpdateStateDataParameters { + pub(crate) state_name: ConstPtr, + pub(crate) buffer: Option>, + pub(crate) buffer_size: u32, + pub(crate) type_id: Option>, + pub(crate) explicit_scope: Option>, + pub(crate) matching_change_stamp: u32, + pub(crate) check_stamp: i32, +} + +impl Task { + pub(crate) fn sys_nt_create_wnf_state_name( + &self, + params: WnfCreateStateNameParameters, + ) -> NtStatus { + if probe_guest_output_preserving_value::(params.state_name).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + let type_id = match read_type_id::(params.type_id) { + Ok(type_id) => type_id, + Err(status) => return status, + }; + if params.security_descriptor.read_at_offset(0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + let Ok(lifetime) = WnfStateNameLifetime::try_from(params.name_lifetime) else { + return NtStatus::INVALID_PARAMETER; + }; + let Ok(data_scope) = WnfDataScope::try_from(params.data_scope) else { + return NtStatus::INVALID_PARAMETER; + }; + if params.maximum_state_size > MAXIMUM_STATE_SIZE { + return NtStatus::INVALID_PARAMETER; + } + match lifetime { + WnfStateNameLifetime::WellKnown => return NtStatus::INVALID_PARAMETER, + WnfStateNameLifetime::Permanent | WnfStateNameLifetime::Persistent => { + // TODO(wnf-create-privilege): Allow privileged lifetimes once guest token + // privileges are modeled. + return NtStatus::PRIVILEGE_NOT_HELD; + } + WnfStateNameLifetime::Temporary => {} + } + if data_scope == WnfDataScope::Process || params.persist_data != 0 { + return NtStatus::INVALID_PARAMETER; + } + + // TODO(wnf-security-descriptor): Enforce the supplied DACL once guest tokens and WNF + // access checks are modeled. + let state_name = { + let mut store = self.global.wnf_states.write(); + let Some(unique_id) = store.next_unique_id.checked_add(1) else { + return NtStatus::NO_MEMORY; + }; + if unique_id > MAXIMUM_UNIQUE_ID { + return NtStatus::NO_MEMORY; + } + store.next_unique_id = unique_id; + let state_name = encode_state_name(lifetime, data_scope, false, unique_id); + let state = WnfStateData { + change_stamp: INITIAL_CHANGE_STAMP, + type_id, + data: Vec::new(), + maximum_state_size: params.maximum_state_size, + lifetime, + }; + // TODO(wnf-temporary-lifetime): Remove temporary names when their creating guest + // process exits once process lifecycle is modeled. + store.states.insert(state_name, state); + state_name + }; + if params.state_name.write_at_offset(0, state_name).is_none() { + let mut store = self.global.wnf_states.write(); + if store + .states + .get(&state_name) + .is_some_and(|current| current.change_stamp == INITIAL_CHANGE_STAMP) + { + store.states.remove(&state_name); + } + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_update_wnf_state_data( + &self, + params: WnfUpdateStateDataParameters, + ) -> NtStatus { + let Some(state_name) = params.state_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let type_id = match read_type_id::(params.type_id) { + Ok(type_id) => type_id, + Err(status) => return status, + }; + if params.explicit_scope.is_some() { + // TODO(wnf-explicit-scope): Key state data by the explicit SID once scoped WNF + // state access is modeled. + return NtStatus::INVALID_PARAMETER; + } + let data = if params.buffer_size == 0 { + Vec::new() + } else { + let Some(buffer) = params.buffer else { + return NtStatus::ACCESS_VIOLATION; + }; + let Some(data) = buffer.to_owned_slice(params.buffer_size as usize) else { + return NtStatus::ACCESS_VIOLATION; + }; + Vec::from(data) + }; + + let mut store = self.global.wnf_states.write(); + let Some(state) = store.states.get_mut(&state_name) else { + return NtStatus::OBJECT_NAME_NOT_FOUND; + }; + if !type_id_matches(state.type_id, type_id) || params.buffer_size > state.maximum_state_size + { + return NtStatus::INVALID_PARAMETER; + } + if params.check_stamp != 0 && params.matching_change_stamp != state.change_stamp { + return NtStatus::UNSUCCESSFUL; + } + state.change_stamp = state.change_stamp.wrapping_add(1); + state.data = data; + // TODO(wnf-notify): Deliver successful updates to subscribers when WNF subscriptions are + // modeled. + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_delete_wnf_state_data( + &self, + state_name: ConstPtr, + explicit_scope: Option>, + ) -> NtStatus { + let Some(state_name) = state_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + if explicit_scope.is_some() { + // TODO(wnf-explicit-scope): Delete only the selected SID-scoped data instance once + // scoped WNF state access is modeled. + return NtStatus::INVALID_PARAMETER; + } + let mut store = self.global.wnf_states.write(); + let Some(state) = store.states.get_mut(&state_name) else { + return NtStatus::OBJECT_NAME_NOT_FOUND; + }; + state.change_stamp = 0; + state.data.clear(); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_delete_wnf_state_name( + &self, + state_name: ConstPtr, + ) -> NtStatus { + let Some(state_name) = state_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let mut store = self.global.wnf_states.write(); + let Some(state) = store.states.get(&state_name) else { + return NtStatus::OBJECT_NAME_NOT_FOUND; + }; + if state.lifetime == WnfStateNameLifetime::WellKnown { + return NtStatus::INVALID_PARAMETER; + } + store.states.remove(&state_name); + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_query_wnf_state_name_information( + &self, + state_name: ConstPtr, + name_information_class: u32, + explicit_scope: Option>, + buffer: MutPtr, + buffer_size: u32, + ) -> NtStatus { + let Some(state_name) = state_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let Ok(information_class) = WnfStateNameInformation::try_from(name_information_class) + else { + return NtStatus::INVALID_INFO_CLASS; + }; + if buffer_size != STATE_NAME_INFORMATION_SIZE { + return NtStatus::INVALID_PARAMETER; + } + if explicit_scope.is_some() { + // TODO(wnf-explicit-scope): Resolve the selected SID-scoped state instance once scoped + // WNF state access is modeled. + return NtStatus::INVALID_PARAMETER; + } + if probe_guest_output_preserving_value::(buffer).is_err() { + return NtStatus::ACCESS_VIOLATION; + } + + let store = self.global.wnf_states.read(); + let exists = store.states.contains_key(&state_name); + let value = match information_class { + WnfStateNameInformation::Exists => u32::from(exists), + WnfStateNameInformation::SubscribersPresent => { + if !exists { + return NtStatus::OBJECT_NAME_NOT_FOUND; + } + // TODO(wnf-notify): Report registered subscribers once WNF subscriptions are + // modeled. + 0 + } + WnfStateNameInformation::IsQuiescent => { + if !exists { + return NtStatus::OBJECT_NAME_NOT_FOUND; + } + 1 + } + }; + buffer + .write_at_offset(0, value) + .map_or(NtStatus::ACCESS_VIOLATION, |()| NtStatus::SUCCESS) + } + + pub(crate) fn sys_nt_query_wnf_state_data( + &self, + state_name: ConstPtr, + type_id: Option>, + explicit_scope: Option>, + change_stamp: MutPtr, + buffer: MutPtr, + buffer_size: MutPtr, + ) -> NtStatus { + let Some(state_name) = state_name.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let type_id = match type_id { + Some(type_id) => match type_id.read_at_offset(0) { + Some(type_id) => Some(type_id), + None => return NtStatus::ACCESS_VIOLATION, + }, + None => None, + }; + let Some(available_size) = buffer_size.read_at_offset(0) else { + return NtStatus::ACCESS_VIOLATION; + }; + let outputs_valid = probe_guest_output_preserving_value::(change_stamp) + .is_ok() + && probe_guest_output_preserving_value::(buffer_size).is_ok() + && probe_guest_output_buffer::(buffer, available_size as usize).is_ok(); + if !outputs_valid { + return NtStatus::ACCESS_VIOLATION; + } + if explicit_scope.is_some() { + // TODO(wnf-explicit-scope): Key state data by the explicit SID once scoped WNF state + // creation and security checks are modeled. + litebox_util_log::debug!( + state_name:% = format_args!("{state_name:#x}"); + "Explicit-scope WNF state queries are not supported" + ); + return NtStatus::INVALID_PARAMETER; + } + + let state = { + let store = self.global.wnf_states.read(); + store.states.get(&state_name).cloned() + }; + let Some(state) = state else { + return NtStatus::OBJECT_NAME_NOT_FOUND; + }; + if !type_id_matches(state.type_id, type_id) { + return NtStatus::INVALID_PARAMETER; + } + + let required_size = state.data.len().trunc(); + let status = if available_size < required_size { + NtStatus::BUFFER_TOO_SMALL + } else { + if !state.data.is_empty() && buffer.write_slice_at_offset(0, &state.data).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + }; + if change_stamp + .write_at_offset(0, state.change_stamp) + .is_none() + || buffer_size.write_at_offset(0, required_size).is_none() + { + return NtStatus::ACCESS_VIOLATION; + } + status + } +} + +fn read_type_id( + type_id: Option>, +) -> Result, NtStatus> { + type_id + .map(|type_id| type_id.read_at_offset(0).ok_or(NtStatus::ACCESS_VIOLATION)) + .transpose() +} + +fn type_id_matches(expected: Option, supplied: Option) -> bool { + expected.is_none() + || matches!((expected, supplied), (Some(expected), Some(supplied)) if expected.data == supplied.data) +} + +fn encode_state_name( + lifetime: WnfStateNameLifetime, + data_scope: WnfDataScope, + persist_data: bool, + unique_id: u32, +) -> u64 { + let clear = 1 + | ((lifetime as u64) << 4) + | ((data_scope as u64) << 6) + | (u64::from(persist_data) << 10) + | (u64::from(unique_id) << 11); + clear ^ STATE_NAME_XOR_KEY +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::tests::{TestFS, TestPlatform, const_ptr, mut_byte_ptr, mut_ptr, test_task}; + + const SECURITY_DESCRIPTOR_REVISION: u8 = 1; + + fn create_state( + task: &Task, + type_id: Option, + maximum_state_size: u32, + ) -> u64 { + let mut state_name = 0; + assert_eq!( + task.sys_nt_create_wnf_state_name(WnfCreateStateNameParameters { + state_name: mut_ptr(&mut state_name), + name_lifetime: WnfStateNameLifetime::Temporary as u32, + data_scope: WnfDataScope::Machine as u32, + persist_data: 0, + type_id: type_id.as_ref().map(const_ptr), + maximum_state_size, + security_descriptor: const_ptr(&SECURITY_DESCRIPTOR_REVISION), + }), + NtStatus::SUCCESS + ); + state_name + } + + fn update_state( + task: &Task, + state_name: u64, + data: &[u8], + type_id: Option<&Guid>, + matching_change_stamp: u32, + check_stamp: i32, + ) -> NtStatus { + task.sys_nt_update_wnf_state_data(WnfUpdateStateDataParameters { + state_name: const_ptr(&state_name), + buffer: data.first().map(const_ptr), + buffer_size: u32::try_from(data.len()).expect("test payload length fits in u32"), + type_id: type_id.map(const_ptr), + explicit_scope: None, + matching_change_stamp, + check_stamp, + }) + } + + #[test] + fn delete_data_resets_state_and_delete_name_removes_it() { + let task = test_task(); + let state_name = create_state(&task, None, 4); + assert_eq!( + update_state(&task, state_name, &[1, 2], None, 0, 0), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_delete_wnf_state_data(const_ptr(&state_name), None), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_delete_wnf_state_data(const_ptr(&state_name), None), + NtStatus::SUCCESS + ); + + let mut change_stamp = 99; + let mut buffer = [0xaau8; 2]; + let mut buffer_size = 2; + assert_eq!( + task.sys_nt_query_wnf_state_data( + const_ptr(&state_name), + None, + None, + mut_ptr(&mut change_stamp), + mut_byte_ptr(&mut buffer), + mut_ptr(&mut buffer_size), + ), + NtStatus::SUCCESS + ); + assert_eq!(change_stamp, 0); + assert_eq!(buffer_size, 0); + assert_eq!(buffer, [0xaa; 2]); + + assert_eq!( + task.sys_nt_delete_wnf_state_name(const_ptr(&state_name)), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_delete_wnf_state_name(const_ptr(&state_name)), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!( + task.sys_nt_query_wnf_state_data( + const_ptr(&state_name), + None, + None, + mut_ptr(&mut change_stamp), + mut_byte_ptr(&mut buffer), + mut_ptr(&mut buffer_size), + ), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + } + + #[test] + fn state_name_information_reports_native_boolean_contract() { + let task = test_task(); + let state_name = create_state(&task, None, 4); + for (class, expected) in [ + (WnfStateNameInformation::Exists, 1), + (WnfStateNameInformation::SubscribersPresent, 0), + (WnfStateNameInformation::IsQuiescent, 1), + ] { + let mut value = u32::MAX; + assert_eq!( + task.sys_nt_query_wnf_state_name_information( + const_ptr(&state_name), + class as u32, + None, + mut_ptr(&mut value), + STATE_NAME_INFORMATION_SIZE, + ), + NtStatus::SUCCESS + ); + assert_eq!(value, expected); + } + + assert_eq!( + task.sys_nt_delete_wnf_state_name(const_ptr(&state_name)), + NtStatus::SUCCESS + ); + let mut value = u32::MAX; + assert_eq!( + task.sys_nt_query_wnf_state_name_information( + const_ptr(&state_name), + WnfStateNameInformation::Exists as u32, + None, + mut_ptr(&mut value), + STATE_NAME_INFORMATION_SIZE, + ), + NtStatus::SUCCESS + ); + assert_eq!(value, 0); + assert_eq!( + task.sys_nt_query_wnf_state_name_information( + const_ptr(&state_name), + WnfStateNameInformation::SubscribersPresent as u32, + None, + mut_ptr(&mut value), + STATE_NAME_INFORMATION_SIZE, + ), + NtStatus::OBJECT_NAME_NOT_FOUND + ); + assert_eq!( + task.sys_nt_query_wnf_state_name_information( + const_ptr(&state_name), + 3, + None, + mut_ptr(&mut value), + STATE_NAME_INFORMATION_SIZE, + ), + NtStatus::INVALID_INFO_CLASS + ); + } +} diff --git a/litebox_shim_windows/src/syscalls/worker_factory.rs b/litebox_shim_windows/src/syscalls/worker_factory.rs new file mode 100644 index 0000000000..7c986dd838 --- /dev/null +++ b/litebox_shim_windows/src/syscalls/worker_factory.rs @@ -0,0 +1,1056 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Windows NT worker factory syscalls. + +use alloc::sync::Arc; +use core::marker::PhantomData; +use core::mem::size_of; +use core::sync::atomic::{AtomicBool, AtomicU32, Ordering}; + +use litebox::fd::{FdEnabledSubsystem, FdEnabledSubsystemEntry}; +use litebox::platform::{RawConstPointer as _, RawMutPointer as _, RawPointerProvider}; +use litebox_common_windows::nt_status::NtStatus; + +use crate::nt_types::{AccessMask, ObjectAttributes, read_object_attributes}; +use crate::syscalls::iocp::{ + IoCompletionAccess, IoCompletionHandleObject, IoCompletionObject, IoCompletionSubsystem, +}; +use crate::syscalls::{Handle, ProcessHandle}; +use crate::{ConstPtr, MutPtr, ShimFS, Task, probe_guest_output_preserving_value}; + +bitflags::bitflags! { + #[derive(Clone, Copy, Debug, Eq, PartialEq)] + pub(crate) struct WorkerFactoryAccess: u32 { + const RELEASE_WORKER = 0x0001; + const WAIT = 0x0002; + const SET_INFORMATION = 0x0004; + const QUERY_INFORMATION = 0x0008; + const READY_WORKER = 0x0010; + const SHUTDOWN = 0x0020; + + const READ = AccessMask::STANDARD_RIGHTS_READ.bits() | Self::QUERY_INFORMATION.bits(); + const WRITE = AccessMask::STANDARD_RIGHTS_WRITE.bits() | Self::SET_INFORMATION.bits(); + const EXECUTE = AccessMask::STANDARD_RIGHTS_EXECUTE.bits() + | AccessMask::SYNCHRONIZE.bits() + | Self::WAIT.bits(); + const ALL_ACCESS = AccessMask::STANDARD_RIGHTS_ALL.bits() + | Self::RELEASE_WORKER.bits() + | Self::WAIT.bits() + | Self::SET_INFORMATION.bits() + | Self::QUERY_INFORMATION.bits() + | Self::READY_WORKER.bits() + | Self::SHUTDOWN.bits(); + + const _ = !0; + } +} + +impl WorkerFactoryAccess { + fn from_desired_access(desired_access: u32) -> Self { + Self::from_bits_retain(AccessMask::expand_generic_access( + desired_access, + Self::READ.bits(), + Self::WRITE.bits(), + Self::EXECUTE.bits(), + Self::ALL_ACCESS.bits(), + )) + } +} + +#[repr(u32)] +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +enum WorkerFactoryInformationClass { + BindingCount = 3, + ThreadMinimum = 4, + ThreadMaximum = 5, + ThreadSoftMaximum = 14, +} + +impl WorkerFactoryInformationClass { + fn from_raw(raw: u32) -> Result { + match raw { + 3 => Ok(Self::BindingCount), + 4 => Ok(Self::ThreadMinimum), + 5 => Ok(Self::ThreadMaximum), + 14 => Ok(Self::ThreadSoftMaximum), + _ => Err(NtStatus::INVALID_INFO_CLASS), + } + } +} + +pub(crate) struct WorkerFactorySubsystem(PhantomData); + +impl FdEnabledSubsystem for WorkerFactorySubsystem { + type Entry = WorkerFactoryHandleObject; +} + +impl FdEnabledSubsystemEntry + for WorkerFactoryHandleObject +{ +} + +impl crate::WindowsHandleSubsystem + for WorkerFactorySubsystem +{ + fn normalize_desired_access(desired_access: u32) -> u32 { + WorkerFactoryAccess::from_desired_access(desired_access).bits() + } +} + +pub(crate) struct WorkerFactoryHandleObject { + factory: Arc>, +} + +pub(crate) struct WorkerFactoryObject { + _completion_port: Arc>, + _start_routine: usize, + _start_parameter: usize, + binding_count: AtomicU32, + thread_minimum: AtomicU32, + thread_maximum: AtomicU32, + thread_soft_maximum: AtomicU32, + shutdown: AtomicBool, + _stack_reserve: usize, + _stack_commit: usize, +} + +pub(crate) struct WorkerFactoryCreateParameters { + pub(crate) worker_factory_handle: MutPtr, + pub(crate) desired_access: u32, + pub(crate) object_attributes: Option>, + pub(crate) completion_port_handle: Handle, + pub(crate) worker_process_handle: ProcessHandle, + pub(crate) start_routine: usize, + pub(crate) start_parameter: usize, + pub(crate) max_thread_count: u32, + pub(crate) stack_reserve: usize, + pub(crate) stack_commit: usize, +} + +fn validate_worker_factory_object_attributes( + object_attributes: Option>, +) -> Result<(), NtStatus> { + let Some(object_attributes) = object_attributes else { + return Ok(()); + }; + let object_attributes = read_object_attributes::(object_attributes)?; + if object_attributes.object_name == 0 && !object_attributes.root_directory.is_null() { + return Err(NtStatus::OBJECT_NAME_INVALID); + } + Ok(()) +} + +fn commit_worker_factory_shutdown( + factory: &WorkerFactoryObject, + pending_worker_count: MutPtr, +) -> NtStatus { + if pending_worker_count.write_at_offset(0, 0).is_none() { + return NtStatus::ACCESS_VIOLATION; + } + factory.shutdown.store(true, Ordering::Relaxed); + NtStatus::SUCCESS +} + +impl Task { + fn io_completion_port( + &self, + handle: Handle, + ) -> Result>, NtStatus> { + let entry = self.typed_handle_entry_with_access::>( + handle, + IoCompletionAccess::MODIFY_STATE.bits(), + )?; + Ok(entry.with_entry(IoCompletionHandleObject::port)) + } + + fn validate_worker_process_handle( + &self, + process_handle: ProcessHandle, + ) -> Result<(), NtStatus> { + if process_handle.is_current() { + return Ok(()); + } + let Some(raw_fd) = process_handle.as_handle().raw_fd() else { + return Err(NtStatus::INVALID_HANDLE); + }; + if self.process.handles.read().is_alive(raw_fd) { + Err(NtStatus::OBJECT_TYPE_MISMATCH) + } else { + Err(NtStatus::INVALID_HANDLE) + } + } + + fn insert_worker_factory_handle( + &self, + factory: Arc>, + granted_access: WorkerFactoryAccess, + ) -> Result { + self.insert_typed_handle::>( + WorkerFactoryHandleObject { factory }, + granted_access.bits(), + drop, + ) + } + + pub(crate) fn close_worker_factory_handle(&self, handle: Handle) { + self.close_typed_handle::>(handle, drop); + } + + pub(crate) fn close_worker_factory(worker_factory: WorkerFactoryHandleObject) { + drop(worker_factory); + } + + pub(crate) fn sys_nt_create_worker_factory( + &self, + params: WorkerFactoryCreateParameters, + ) -> NtStatus { + if let Err(status) = + probe_guest_output_preserving_value::(params.worker_factory_handle) + { + return status; + } + let completion_port = match self.io_completion_port(params.completion_port_handle) { + Ok(port) => port, + Err(status) => return status, + }; + if let Err(status) = self.validate_worker_process_handle(params.worker_process_handle) { + return status; + } + if let Err(status) = + validate_worker_factory_object_attributes::(params.object_attributes) + { + return status; + } + + let factory = Arc::new(WorkerFactoryObject { + // TODO: create and manage actual worker threads using start_routine/start_parameter + // once worker dispatch, NtWaitForWorkViaWorkerFactory, and + // NtReleaseWorkerFactoryWorker are implemented. + _completion_port: completion_port, + _start_routine: params.start_routine, + _start_parameter: params.start_parameter, + binding_count: AtomicU32::new(0), + thread_minimum: AtomicU32::new(0), + thread_maximum: AtomicU32::new(params.max_thread_count), + thread_soft_maximum: AtomicU32::new(params.max_thread_count), + shutdown: AtomicBool::new(false), + _stack_reserve: params.stack_reserve, + _stack_commit: params.stack_commit, + }); + let granted_access = WorkerFactoryAccess::from_desired_access(params.desired_access); + let Ok(handle) = self.insert_worker_factory_handle(factory, granted_access) else { + return NtStatus::QUOTA_EXCEEDED; + }; + if params + .worker_factory_handle + .write_at_offset(0, handle) + .is_none() + { + self.close_worker_factory_handle(handle); + return NtStatus::ACCESS_VIOLATION; + } + NtStatus::SUCCESS + } + + pub(crate) fn sys_nt_set_information_worker_factory( + &self, + handle: Handle, + information_class: u32, + information: ConstPtr, + information_length: u32, + ) -> NtStatus { + litebox_util_log::debug!( + information_class = information_class, + information_length = information_length; + "NtSetInformationWorkerFactory parameters" + ); + let Ok(information_class) = WorkerFactoryInformationClass::from_raw(information_class) + else { + return NtStatus::INVALID_INFO_CLASS; + }; + if information_length as usize != size_of::() { + return NtStatus::INFO_LENGTH_MISMATCH; + } + let Some(value_bytes) = information.to_owned_slice(size_of::()) else { + return NtStatus::ACCESS_VIOLATION; + }; + let value = u32::from_le_bytes( + value_bytes + .as_ref() + .try_into() + .expect("ULONG input is four bytes"), + ); + + let entry = match self.typed_handle_entry_with_access::>( + handle, + WorkerFactoryAccess::SET_INFORMATION.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + entry + .with_entry(|entry| { + // TODO: enforce these limits against real worker creation/drain behavior + // once worker threads are modeled; today they are only recorded. + match information_class { + WorkerFactoryInformationClass::BindingCount => { + // TODO: bind this to real worker/IOCP association state once worker + // factories track live bindings. + entry.factory.binding_count.store(value, Ordering::Relaxed); + } + WorkerFactoryInformationClass::ThreadMinimum => { + let maximum = entry.factory.thread_maximum.load(Ordering::Relaxed); + if value > maximum { + return Err(NtStatus::INVALID_PARAMETER); + } + entry.factory.thread_minimum.store(value, Ordering::Relaxed); + } + WorkerFactoryInformationClass::ThreadMaximum => { + let minimum = entry.factory.thread_minimum.load(Ordering::Relaxed); + if value < minimum { + return Err(NtStatus::INVALID_PARAMETER); + } + entry.factory.thread_maximum.store(value, Ordering::Relaxed); + } + WorkerFactoryInformationClass::ThreadSoftMaximum => { + let maximum = entry.factory.thread_maximum.load(Ordering::Relaxed); + if value > maximum { + return Err(NtStatus::INVALID_PARAMETER); + } + entry + .factory + .thread_soft_maximum + .store(value, Ordering::Relaxed); + } + } + Ok(()) + }) + .map_or_else(|status| status, |()| NtStatus::SUCCESS) + } + + pub(crate) fn sys_nt_shutdown_worker_factory( + &self, + handle: Handle, + pending_worker_count: MutPtr, + ) -> NtStatus { + if let Err(status) = + probe_guest_output_preserving_value::(pending_worker_count) + { + return status; + } + let entry = match self.typed_handle_entry_with_access::>( + handle, + WorkerFactoryAccess::SHUTDOWN.bits(), + ) { + Ok(entry) => entry, + Err(status) => return status, + }; + let factory = entry.with_entry(|entry| Arc::clone(&entry.factory)); + // TODO: report the actual pending worker count and wake/release workers once worker + // threads are modeled; the current subset has no workers to drain. + commit_worker_factory_shutdown(&factory, pending_worker_count) + } +} + +#[cfg(test)] +mod tests { + use core::mem::size_of; + + use litebox::platform::ThreadProvider; + use litebox::utils::TruncateExt as _; + use litebox_common_windows::nt_status::NtStatus; + + use super::*; + use crate::nt_types::ObjectAttributes; + use crate::tests::{TestFS, TestPlatform, mut_ptr, null_mut_ptr, test_platform, test_task}; + + const EVENT_ALL_ACCESS: u32 = 0x001f_0003; + const IO_COMPLETION_QUERY_STATE: u32 = 0x0000_0001; + const IO_COMPLETION_ALL_ACCESS: u32 = 0x001f_0003; + const WORKER_FACTORY_ALL_ACCESS: u32 = 0x001f_003f; + const WORKER_FACTORY_QUERY_INFORMATION: u32 = 0x0008; + const WORKER_FACTORY_SHUTDOWN: u32 = 0x0020; + const START_ROUTINE: usize = 0x1234_5678; + + fn run_with_test_platform_pointers(f: impl FnOnce() -> R) -> R { + let _ = test_platform(); + ::run_test_thread(f) + } + + fn create_io_completion_handle(task: &Task) -> Handle { + create_io_completion_handle_with_access(task, IO_COMPLETION_ALL_ACCESS) + } + + fn create_io_completion_handle_with_access( + task: &Task, + access: u32, + ) -> Handle { + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_io_completion(mut_ptr(&mut handle), access, None, 0), + NtStatus::SUCCESS + ); + handle + } + + fn create_worker_factory( + task: &Task, + worker_factory_handle: &mut Handle, + object_attributes: Option>, + completion_port_handle: Handle, + worker_process_handle: ProcessHandle, + ) -> NtStatus { + create_worker_factory_with_access( + task, + worker_factory_handle, + WORKER_FACTORY_ALL_ACCESS, + object_attributes, + completion_port_handle, + worker_process_handle, + ) + } + + fn create_worker_factory_with_access( + task: &Task, + worker_factory_handle: &mut Handle, + desired_access: u32, + object_attributes: Option>, + completion_port_handle: Handle, + worker_process_handle: ProcessHandle, + ) -> NtStatus { + task.sys_nt_create_worker_factory(WorkerFactoryCreateParameters { + worker_factory_handle: mut_ptr(worker_factory_handle), + desired_access, + object_attributes, + completion_port_handle, + worker_process_handle, + start_routine: START_ROUTINE, + start_parameter: 0, + max_thread_count: 1, + stack_reserve: 0, + stack_commit: 0, + }) + } + + fn information_ptr(value: &u32) -> ConstPtr { + ConstPtr::::from_usize(core::ptr::from_ref(value).cast::() as usize) + } + + #[test] + fn create_requires_modify_state_on_completion_port() { + let task = test_task(); + let io_completion = + create_io_completion_handle_with_access(&task, IO_COMPLETION_QUERY_STATE); + let mut worker_factory = Handle::from_raw(usize::MAX); + + assert_eq!( + create_worker_factory( + &task, + &mut worker_factory, + None, + io_completion, + ProcessHandle::CURRENT + ), + NtStatus::ACCESS_DENIED + ); + assert_eq!(worker_factory, Handle::from_raw(usize::MAX)); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + } + + #[test] + fn set_information_rejects_wrong_object_type_and_missing_access() { + let task = test_task(); + let io_completion = create_io_completion_handle(&task); + let value = 1; + let mut event = Handle::default(); + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut event), EVENT_ALL_ACCESS, None, 0, 0), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_set_information_worker_factory( + event, + WorkerFactoryInformationClass::ThreadMaximum as u32, + information_ptr(&value), + size_of::().trunc(), + ), + NtStatus::OBJECT_TYPE_MISMATCH + ); + + let mut worker_factory = Handle::default(); + assert_eq!( + create_worker_factory_with_access( + &task, + &mut worker_factory, + WORKER_FACTORY_QUERY_INFORMATION, + None, + io_completion, + ProcessHandle::CURRENT, + ), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_set_information_worker_factory( + worker_factory, + WorkerFactoryInformationClass::ThreadMaximum as u32, + information_ptr(&value), + size_of::().trunc(), + ), + NtStatus::ACCESS_DENIED + ); + + assert_eq!(task.sys_nt_close(worker_factory), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(event), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + } + + #[test] + fn shutdown_sets_pending_worker_count_to_zero() { + let task = test_task(); + let io_completion = create_io_completion_handle(&task); + let mut worker_factory = Handle::default(); + assert_eq!( + create_worker_factory( + &task, + &mut worker_factory, + None, + io_completion, + ProcessHandle::CURRENT, + ), + NtStatus::SUCCESS + ); + + let mut pending_worker_count = 7; + assert_eq!( + task.sys_nt_shutdown_worker_factory(worker_factory, mut_ptr(&mut pending_worker_count)), + NtStatus::SUCCESS + ); + assert_eq!(pending_worker_count, 0); + + assert_eq!(task.sys_nt_close(worker_factory), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + } + + #[test] + fn shutdown_output_fault_preserves_factory_state() { + run_with_test_platform_pointers(|| { + let task = test_task(); + let io_completion = create_io_completion_handle(&task); + let mut worker_factory = Handle::default(); + assert_eq!( + create_worker_factory( + &task, + &mut worker_factory, + None, + io_completion, + ProcessHandle::CURRENT, + ), + NtStatus::SUCCESS + ); + let factory = task + .typed_handle_entry::>(worker_factory) + .expect("worker factory handle is valid") + .with_entry(|entry| Arc::clone(&entry.factory)); + assert!(!factory.shutdown.load(Ordering::Relaxed)); + + assert_eq!( + task.sys_nt_shutdown_worker_factory(worker_factory, null_mut_ptr()), + NtStatus::ACCESS_VIOLATION + ); + assert!(!factory.shutdown.load(Ordering::Relaxed)); + + assert_eq!( + commit_worker_factory_shutdown(&factory, null_mut_ptr()), + NtStatus::ACCESS_VIOLATION + ); + assert!(!factory.shutdown.load(Ordering::Relaxed)); + + assert_eq!(task.sys_nt_close(worker_factory), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + }); + } + + #[test] + fn shutdown_validates_pending_worker_count_before_handle() { + run_with_test_platform_pointers(|| { + let task = test_task(); + + assert_eq!( + task.sys_nt_shutdown_worker_factory(Handle::default(), null_mut_ptr()), + NtStatus::ACCESS_VIOLATION + ); + }); + } + + #[test] + fn shutdown_rejects_wrong_object_type_and_missing_access() { + let task = test_task(); + let io_completion = create_io_completion_handle(&task); + let mut pending_worker_count = 1; + let mut event = Handle::default(); + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut event), EVENT_ALL_ACCESS, None, 0, 0), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_shutdown_worker_factory(event, mut_ptr(&mut pending_worker_count)), + NtStatus::OBJECT_TYPE_MISMATCH + ); + assert_eq!(pending_worker_count, 1); + + let mut worker_factory = Handle::default(); + assert_eq!( + create_worker_factory_with_access( + &task, + &mut worker_factory, + WORKER_FACTORY_QUERY_INFORMATION, + None, + io_completion, + ProcessHandle::CURRENT, + ), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_shutdown_worker_factory(worker_factory, mut_ptr(&mut pending_worker_count)), + NtStatus::ACCESS_DENIED + ); + assert_eq!(pending_worker_count, 1); + + assert_eq!(task.sys_nt_close(worker_factory), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(event), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + + let io_completion = create_io_completion_handle(&task); + let mut worker_factory = Handle::default(); + assert_eq!( + create_worker_factory_with_access( + &task, + &mut worker_factory, + WORKER_FACTORY_SHUTDOWN, + None, + io_completion, + ProcessHandle::CURRENT, + ), + NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_shutdown_worker_factory(worker_factory, mut_ptr(&mut pending_worker_count)), + NtStatus::SUCCESS + ); + + assert_eq!(task.sys_nt_close(worker_factory), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_create_worker_factory_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateEvent( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + event_type: u32, + initial_state: u8, + ) -> i32; + fn NtCreateIoCompletion( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + number_of_concurrent_threads: u32, + ) -> i32; + fn NtCreateWorkerFactory( + handle: *mut *mut c_void, + desired_access: u32, + object_attributes: *const ObjectAttributes, + completion_port_handle: *mut c_void, + worker_process_handle: *mut c_void, + start_routine: *mut c_void, + start_parameter: *mut c_void, + max_thread_count: u32, + stack_reserve: usize, + stack_commit: usize, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + let mut host_io_completion = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, attributes are null, and the handle is closed below. + let status = unsafe { + NtCreateIoCompletion( + &raw mut host_io_completion, + IO_COMPLETION_ALL_ACCESS, + core::ptr::null(), + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + + let task = test_task(); + let io_completion = create_io_completion_handle(&task); + + let mut host_worker_factory = core::ptr::null_mut(); + // SAFETY: All handles and pointers are valid for this status probe; the returned worker + // factory handle is closed before leaving the test. + let host_success = unsafe { + let status = NtCreateWorkerFactory( + &raw mut host_worker_factory, + WORKER_FACTORY_ALL_ACCESS, + core::ptr::null(), + host_io_completion, + usize::MAX as *mut c_void, + START_ROUTINE as *mut c_void, + core::ptr::null_mut(), + 1, + 0, + 0, + ); + if status == NtStatus::SUCCESS.as_raw() && !host_worker_factory.is_null() { + assert_eq!(NtClose(host_worker_factory), NtStatus::SUCCESS.as_raw()); + } + status + }; + + let mut shim_worker_factory = Handle::default(); + assert_eq!( + create_worker_factory( + &task, + &mut shim_worker_factory, + None, + io_completion, + ProcessHandle::CURRENT + ) + .as_raw(), + host_success + ); + assert!(!shim_worker_factory.is_null()); + assert_eq!(task.sys_nt_close(shim_worker_factory), NtStatus::SUCCESS); + + let mut host_query_io_completion = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, attributes are null, and the handle is closed below. + let status = unsafe { + NtCreateIoCompletion( + &raw mut host_query_io_completion, + IO_COMPLETION_QUERY_STATE, + core::ptr::null(), + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + let mut shim_query_io_completion = Handle::default(); + assert_eq!( + task.sys_nt_create_io_completion( + mut_ptr(&mut shim_query_io_completion), + IO_COMPLETION_QUERY_STATE, + None, + 0 + ), + NtStatus::SUCCESS + ); + + let mut host_query_worker_factory = core::ptr::null_mut(); + // SAFETY: All pointers are valid; the completion port intentionally lacks modify access to + // compare native access checking with the shim. + let host_query_only_completion = unsafe { + let status = NtCreateWorkerFactory( + &raw mut host_query_worker_factory, + WORKER_FACTORY_ALL_ACCESS, + core::ptr::null(), + host_query_io_completion, + usize::MAX as *mut c_void, + START_ROUTINE as *mut c_void, + core::ptr::null_mut(), + 1, + 0, + 0, + ); + if status == NtStatus::SUCCESS.as_raw() && !host_query_worker_factory.is_null() { + assert_eq!( + NtClose(host_query_worker_factory), + NtStatus::SUCCESS.as_raw() + ); + } + status + }; + let mut shim_query_worker_factory = Handle::default(); + assert_eq!( + create_worker_factory( + &task, + &mut shim_query_worker_factory, + None, + shim_query_io_completion, + ProcessHandle::CURRENT + ) + .as_raw(), + host_query_only_completion + ); + if !shim_query_worker_factory.is_null() { + assert_eq!( + task.sys_nt_close(shim_query_worker_factory), + NtStatus::SUCCESS + ); + } + assert_eq!( + task.sys_nt_close(shim_query_io_completion), + NtStatus::SUCCESS + ); + + let bad_length = ObjectAttributes { + length: 1, + root_directory: Handle::default(), + object_name: 0, + attributes: 0, + security_descriptor: 0, + security_quality_of_service: 0, + }; + // SAFETY: The host output and attributes pointers are valid locals; the bad length is the + // parameter being tested. + let host_bad_length = unsafe { + NtCreateWorkerFactory( + &raw mut host_worker_factory, + WORKER_FACTORY_ALL_ACCESS, + &raw const bad_length, + host_io_completion, + usize::MAX as *mut c_void, + START_ROUTINE as *mut c_void, + core::ptr::null_mut(), + 1, + 0, + 0, + ) + }; + assert_eq!( + create_worker_factory( + &task, + &mut shim_worker_factory, + Some(crate::tests::const_ptr(&bad_length)), + io_completion, + ProcessHandle::CURRENT + ) + .as_raw(), + host_bad_length + ); + + let mut host_event = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, attributes are null, and the handle is closed below. + let status = unsafe { + NtCreateEvent( + &raw mut host_event, + EVENT_ALL_ACCESS, + core::ptr::null(), + 0, + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + let mut shim_event = Handle::default(); + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut shim_event), EVENT_ALL_ACCESS, None, 0, 0), + NtStatus::SUCCESS + ); + + // SAFETY: The event handle is valid but intentionally has the wrong object type for the + // completion-port argument. + let host_wrong_completion_type = unsafe { + NtCreateWorkerFactory( + &raw mut host_worker_factory, + WORKER_FACTORY_ALL_ACCESS, + core::ptr::null(), + host_event, + usize::MAX as *mut c_void, + START_ROUTINE as *mut c_void, + core::ptr::null_mut(), + 1, + 0, + 0, + ) + }; + assert_eq!( + create_worker_factory( + &task, + &mut shim_worker_factory, + None, + shim_event, + ProcessHandle::CURRENT + ) + .as_raw(), + host_wrong_completion_type + ); + + // SAFETY: All non-process arguments are valid; the event handle is intentionally passed as + // the process handle to probe native type checking. + let host_wrong_process_type = unsafe { + NtCreateWorkerFactory( + &raw mut host_worker_factory, + WORKER_FACTORY_ALL_ACCESS, + core::ptr::null(), + host_io_completion, + host_event, + START_ROUTINE as *mut c_void, + core::ptr::null_mut(), + 1, + 0, + 0, + ) + }; + assert_eq!( + create_worker_factory( + &task, + &mut shim_worker_factory, + None, + io_completion, + ProcessHandle::from_raw(shim_event.as_raw()) + ) + .as_raw(), + host_wrong_process_type + ); + + assert_eq!(task.sys_nt_close(shim_event), NtStatus::SUCCESS); + assert_eq!(task.sys_nt_close(io_completion), NtStatus::SUCCESS); + // SAFETY: Handles were created successfully above and have not yet been closed. + unsafe { + assert_eq!(NtClose(host_event), NtStatus::SUCCESS.as_raw()); + assert_eq!(NtClose(host_io_completion), NtStatus::SUCCESS.as_raw()); + assert_eq!( + NtClose(host_query_io_completion), + NtStatus::SUCCESS.as_raw() + ); + } + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_set_worker_factory_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtSetInformationWorkerFactory( + handle: *mut c_void, + worker_factory_information_class: u32, + worker_factory_information: *const c_void, + worker_factory_information_length: u32, + ) -> i32; + } + + let task = test_task(); + let value = 1; + + for (handle, class, info, length) in [ + ( + core::ptr::null_mut(), + WorkerFactoryInformationClass::ThreadMaximum as u32, + (&raw const value).cast(), + size_of::().trunc(), + ), + // This causes a crash + // ( + // core::ptr::null_mut(), + // WorkerFactoryInformationClass::ThreadMaximum as u32, + // core::ptr::null(), + // size_of::().trunc(), + // ), + ( + core::ptr::null_mut(), + 16, + (&raw const value).cast(), + size_of::().trunc(), + ), + ( + core::ptr::null_mut(), + WorkerFactoryInformationClass::ThreadMaximum as u32, + (&raw const value).cast(), + 0, + ), + ] { + // SAFETY: These probes intentionally use an invalid handle; any non-null input pointer + // points to a live local and no native worker factory can be started. + let host_status = unsafe { NtSetInformationWorkerFactory(handle, class, info, length) }; + assert_eq!( + task.sys_nt_set_information_worker_factory( + Handle::default(), + class, + if info.is_null() { + crate::tests::null_const_ptr() + } else { + information_ptr(&value) + }, + length, + ) + .as_raw(), + host_status + ); + } + } + + #[cfg(all(target_os = "windows", target_arch = "x86_64"))] + #[test] + fn host_set_worker_factory_wrong_type_status_fidelity() { + use core::ffi::c_void; + + unsafe extern "system" { + fn NtCreateEvent( + handle: *mut *mut c_void, + access: u32, + attributes: *const ObjectAttributes, + event_type: u32, + initial_state: u8, + ) -> i32; + fn NtSetInformationWorkerFactory( + handle: *mut c_void, + worker_factory_information_class: u32, + worker_factory_information: *const c_void, + worker_factory_information_length: u32, + ) -> i32; + fn NtClose(handle: *mut c_void) -> i32; + } + + let task = test_task(); + let value = 1; + let mut host_event = core::ptr::null_mut(); + // SAFETY: The output pointer is valid, attributes are null, and the handle is closed below. + let status = unsafe { + NtCreateEvent( + &raw mut host_event, + EVENT_ALL_ACCESS, + core::ptr::null(), + 0, + 0, + ) + }; + assert_eq!(status, NtStatus::SUCCESS.as_raw()); + let mut shim_event = Handle::default(); + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut shim_event), EVENT_ALL_ACCESS, None, 0, 0), + NtStatus::SUCCESS + ); + + // SAFETY: The event handle is valid but intentionally has the wrong object type. + let host_wrong_type = unsafe { + NtSetInformationWorkerFactory( + host_event, + WorkerFactoryInformationClass::ThreadMaximum as u32, + (&raw const value).cast(), + size_of::().trunc(), + ) + }; + assert_eq!( + task.sys_nt_set_information_worker_factory( + shim_event, + WorkerFactoryInformationClass::ThreadMaximum as u32, + information_ptr(&value), + size_of::().trunc(), + ) + .as_raw(), + host_wrong_type + ); + + assert_eq!(task.sys_nt_close(shim_event), NtStatus::SUCCESS); + // SAFETY: The host event handle was created successfully above and has not yet been closed. + unsafe { + assert_eq!(NtClose(host_event), NtStatus::SUCCESS.as_raw()); + } + } +} diff --git a/litebox_shim_windows/src/tests.rs b/litebox_shim_windows/src/tests.rs new file mode 100644 index 0000000000..38a2250f55 --- /dev/null +++ b/litebox_shim_windows/src/tests.rs @@ -0,0 +1,448 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +extern crate std; + +use alloc::sync::Arc; +use alloc::vec::Vec; +use core::mem::size_of; +use litebox::LiteBox; +use litebox::fs::{FileSystem as _, Mode, OFlags}; +use litebox::platform::RawConstPointer as _; +use litebox::utils::TruncateExt as _; + +use crate::nt_types::{ObjectAttributes, UnicodeString}; +use crate::syscalls::Handle; +use crate::{ConstPtr, DefaultFS, MutPtr, Process, Task, WindowsShim}; + +#[cfg(target_os = "linux")] +pub(crate) type TestPlatform = litebox_platform_linux_userland::LinuxUserland; +#[cfg(target_os = "windows")] +pub(crate) type TestPlatform = litebox_platform_windows_userland::WindowsUserland; +pub(crate) type TestFS = DefaultFS; + +pub(crate) fn const_ptr(value: &T) -> ConstPtr { + ConstPtr::::from_usize(core::ptr::from_ref(value).cast::() as usize) +} + +pub(crate) fn mut_ptr( + value: &mut T, +) -> MutPtr { + MutPtr::::from_usize(core::ptr::from_mut(value).cast::() as usize) +} + +pub(crate) fn mut_byte_ptr(value: &mut T) -> MutPtr { + MutPtr::::from_usize(core::ptr::from_mut(value).cast::() as usize) +} + +pub(crate) fn null_const_ptr() -> ConstPtr { + ConstPtr::::from_usize(0) +} + +pub(crate) fn null_mut_ptr() -> MutPtr +{ + MutPtr::::from_usize(0) +} + +pub(crate) fn unicode_string(units: &[u16]) -> UnicodeString { + let byte_len = core::mem::size_of_val(units).trunc(); + UnicodeString { + length: byte_len, + maximum_length: byte_len, + padding_0: [0; 4], + buffer: units.as_ptr() as usize, + } +} + +pub(crate) fn utf16_units(value: &str) -> Vec { + value.encode_utf16().collect() +} + +pub(crate) fn object_attributes(name: &UnicodeString, attributes: u32) -> ObjectAttributes { + ObjectAttributes { + length: size_of::().trunc(), + root_directory: Handle::default(), + object_name: core::ptr::from_ref(name) as usize, + attributes, + security_descriptor: 0, + security_quality_of_service: 0, + } +} + +pub(crate) fn test_platform() -> &'static TestPlatform { + static PLATFORM: std::sync::OnceLock<&'static TestPlatform> = std::sync::OnceLock::new(); + PLATFORM.get_or_init(|| { + #[cfg(target_os = "linux")] + let platform = TestPlatform::new(None); + + #[cfg(target_os = "windows")] + let platform = TestPlatform::new(); + + platform + }) +} + +fn map_csr_server_shared_memory( + page_manager: &crate::WindowsPageManager, +) -> Option { + let length = litebox::mm::linux::NonZeroPageSize::new( + crate::syscalls::section::WINDOWS_SHARED_SECTION_SIZE, + )?; + // SAFETY: address selection is left to the page manager, so this cannot replace a mapping. + unsafe { + page_manager.create_writable_pages( + None, + length, + litebox::mm::linux::CreatePagesFlags::empty(), + |_| Ok(0), + ) + } + .map(|mapping| mapping.as_usize()) + .ok() +} + +pub(crate) fn test_task() -> Task { + test_task_with_nls_files(&[]) +} + +pub(crate) fn test_task_with_nls_files(nls_files: &[(&str, &[u8])]) -> Task { + let platform = test_platform(); + let litebox = LiteBox::new(platform); + let mut in_mem = litebox::fs::in_mem::FileSystem::new(&litebox); + in_mem.with_root_privileges(|fs| { + fs.mkdir( + "/tmp", + litebox::fs::Mode::RWXU | litebox::fs::Mode::RWXG | litebox::fs::Mode::RWXO, + ) + .expect("/tmp creation cannot fail on a fresh in-memory file system"); + fs.chown("/tmp", Some(1000), Some(1000)) + .expect("/tmp chown cannot fail on a fresh in-memory file system"); + + if !nls_files.is_empty() { + fs.mkdir("/Windows", Mode::RWXU | Mode::RWXG | Mode::RWXO) + .expect("/Windows creation cannot fail on a fresh in-memory file system"); + fs.mkdir("/Windows/System32", Mode::RWXU | Mode::RWXG | Mode::RWXO) + .expect("/Windows/System32 creation cannot fail on a fresh in-memory file system"); + fs.mkdir( + "/Windows/Globalization", + Mode::RWXU | Mode::RWXG | Mode::RWXO, + ) + .expect("/Windows/Globalization creation cannot fail on a fresh in-memory file system"); + fs.mkdir( + "/Windows/Globalization/Sorting", + Mode::RWXU | Mode::RWXG | Mode::RWXO, + ) + .expect("/Windows/Globalization/Sorting creation cannot fail on a fresh in-memory file system"); + } + for (path, bytes) in nls_files { + let fd = fs + .open( + *path, + OFlags::WRONLY | OFlags::CREAT, + Mode::RUSR | Mode::WUSR | Mode::RGRP | Mode::ROTH, + ) + .expect("NLS fixture creation should succeed"); + fs.write(&fd, bytes, Some(0)) + .expect("NLS fixture write should succeed"); + fs.close(&fd).expect("NLS fixture close should succeed"); + } + }); + let shim_builder = crate::WindowsShimBuilder::::new(platform); + let fs = Arc::new(shim_builder.default_fs(in_mem, litebox::fs::tar_ro::EMPTY_TAR_FILE.into())); + let shim = shim_builder.build(); + let WindowsShim(global) = shim; + + let windows_shared_section_base = map_csr_server_shared_memory(&global.page_manager) + .expect("mapping shared memory should succeed"); + let windows_shared_section = + crate::syscalls::section::load_time_windows_shared_section(windows_shared_section_base); + + Task { + global, + process: Arc::new(Process::default(None, windows_shared_section)), + fs, + entry_point: 0, + stack_top: 0, + context: 0, + teb_address: 0, + } +} + +const EVENT_MODIFY_STATE: u32 = 0x0002; +const SYNCHRONIZE: u32 = 0x0010_0000; +const DUPLICATE_CLOSE_SOURCE: u32 = 0x0000_0001; +const DUPLICATE_SAME_ACCESS: u32 = 0x0000_0002; + +fn create_event(task: &Task, desired_access: u32) -> Handle { + let mut handle = Handle::default(); + assert_eq!( + task.sys_nt_create_event(mut_ptr(&mut handle), desired_access, None, 0, 0,), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + handle +} + +#[test] +fn nt_duplicate_object_preserves_identity_with_independent_access() { + let task = test_task(); + let source = create_event(&task, SYNCHRONIZE); + let mut duplicate = Handle::default(); + + assert_eq!( + task.sys_nt_duplicate_object( + crate::syscalls::ProcessHandle::CURRENT, + source, + crate::syscalls::ProcessHandle::CURRENT, + Some(mut_ptr(&mut duplicate)), + EVENT_MODIFY_STATE, + 0, + 0, + ), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_ne!(source, duplicate); + assert_eq!( + task.sys_nt_set_event(source, None), + litebox_common_windows::nt_status::NtStatus::ACCESS_DENIED + ); + assert_eq!( + task.sys_nt_set_event(duplicate, None), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_close(source), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_set_event(duplicate, None), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_close(duplicate), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); +} + +#[test] +fn nt_duplicate_object_can_atomically_replace_the_source_handle() { + let task = test_task(); + let source = create_event(&task, EVENT_MODIFY_STATE); + let mut duplicate = Handle::default(); + + assert_eq!( + task.sys_nt_duplicate_object( + crate::syscalls::ProcessHandle::CURRENT, + source, + crate::syscalls::ProcessHandle::CURRENT, + Some(mut_ptr(&mut duplicate)), + 0, + 0, + DUPLICATE_CLOSE_SOURCE | DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_close(source), + litebox_common_windows::nt_status::NtStatus::INVALID_HANDLE + ); + assert_eq!( + task.sys_nt_set_event(duplicate, None), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); + assert_eq!( + task.sys_nt_close(duplicate), + litebox_common_windows::nt_status::NtStatus::SUCCESS + ); +} + +#[test] +fn nt_duplicate_object_closes_source_even_when_duplication_fails() { + let task = test_task(); + let source = create_event(&task, EVENT_MODIFY_STATE); + let mut duplicate = Handle::from_raw(0x7777); + + assert_eq!( + task.sys_nt_duplicate_object( + crate::syscalls::ProcessHandle::CURRENT, + source, + crate::syscalls::ProcessHandle::from_raw(0x1234), + Some(mut_ptr(&mut duplicate)), + 0, + 0, + DUPLICATE_CLOSE_SOURCE | DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::INVALID_HANDLE + ); + assert_eq!( + task.sys_nt_close(source), + litebox_common_windows::nt_status::NtStatus::INVALID_HANDLE + ); + assert!(duplicate.is_null()); +} + +#[cfg(target_os = "windows")] +#[test] +fn host_nt_duplicate_object_failure_and_access_matrix() { + use core::ffi::c_void; + + #[link(name = "kernel32")] + unsafe extern "system" { + fn CreateEventW( + event_attributes: *const c_void, + manual_reset: i32, + initial_state: i32, + name: *const u16, + ) -> *mut c_void; + } + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtClose(handle: *mut c_void) -> i32; + fn NtSetEvent(handle: *mut c_void, previous_state: *mut i32) -> i32; + fn NtDuplicateObject( + source_process_handle: *mut c_void, + source_handle: *mut c_void, + target_process_handle: *mut c_void, + target_handle: *mut c_void, + desired_access: u32, + handle_attributes: u32, + options: u32, + ) -> i32; + } + + // SAFETY: All pointers are either documented pseudo-handles, null, or valid local outputs. + unsafe { + let source = CreateEventW(core::ptr::null(), 0, 0, core::ptr::null()); + assert!(!source.is_null()); + let mut duplicate: *mut c_void = core::ptr::null_mut(); + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + source, + 0x1234usize as *mut c_void, + (&raw mut duplicate).cast(), + 0, + 0, + DUPLICATE_CLOSE_SOURCE | DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::INVALID_HANDLE.as_raw() + ); + assert_eq!( + NtClose(source), + litebox_common_windows::nt_status::NtStatus::INVALID_HANDLE.as_raw() + ); + assert!(duplicate.is_null()); + + let source = CreateEventW(core::ptr::null(), 0, 0, core::ptr::null()); + assert!(!source.is_null()); + let mut duplicate = usize::MAX as *mut c_void; + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + source, + core::ptr::null_mut(), + (&raw mut duplicate).cast(), + 0, + 0, + DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::INVALID_PARAMETER.as_raw() + ); + assert!(duplicate.is_null()); + assert_eq!( + NtClose(source), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + + let source = CreateEventW(core::ptr::null(), 0, 0, core::ptr::null()); + assert!(!source.is_null()); + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + source, + usize::MAX as *mut c_void, + core::ptr::null_mut(), + 0, + 0, + DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtClose(source), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + + let source = CreateEventW(core::ptr::null(), 0, 0, core::ptr::null()); + assert!(!source.is_null()); + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + source, + usize::MAX as *mut c_void, + core::ptr::dangling_mut::(), + 0, + 0, + DUPLICATE_CLOSE_SOURCE | DUPLICATE_SAME_ACCESS, + ), + litebox_common_windows::nt_status::NtStatus::ACCESS_VIOLATION.as_raw() + ); + assert_eq!( + NtSetEvent(source, core::ptr::null_mut()), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtClose(source), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + + let source = CreateEventW(core::ptr::null(), 0, 0, core::ptr::null()); + assert!(!source.is_null()); + let mut reduced: *mut c_void = core::ptr::null_mut(); + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + source, + usize::MAX as *mut c_void, + (&raw mut reduced).cast(), + SYNCHRONIZE, + 0, + 0, + ), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + let mut expanded: *mut c_void = core::ptr::null_mut(); + assert_eq!( + NtDuplicateObject( + usize::MAX as *mut c_void, + reduced, + usize::MAX as *mut c_void, + (&raw mut expanded).cast(), + EVENT_MODIFY_STATE, + 0, + 0, + ), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtSetEvent(reduced, core::ptr::null_mut()), + litebox_common_windows::nt_status::NtStatus::ACCESS_DENIED.as_raw() + ); + assert_eq!( + NtSetEvent(expanded, core::ptr::null_mut()), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtClose(source), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtClose(reduced), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + assert_eq!( + NtClose(expanded), + litebox_common_windows::nt_status::NtStatus::SUCCESS.as_raw() + ); + } +} diff --git a/litebox_syscall_rewriter/Cargo.toml b/litebox_syscall_rewriter/Cargo.toml index 644eb55a81..d8488fd758 100644 --- a/litebox_syscall_rewriter/Cargo.toml +++ b/litebox_syscall_rewriter/Cargo.toml @@ -11,7 +11,8 @@ clap = ["dep:clap"] [dependencies] iced-x86 = { version = "1.21", default-features = false, features = ["no_std", "decoder", "encoder", "instr_info"] } -object = { version = "0.36.7", default-features = false, features = ["elf", "read_core"] } +litebox_common_windows = { path = "../litebox_common_windows/", version = "0.1.0" } +object = { version = "0.36.7", default-features = false, features = ["elf", "pe", "read_core"] } thiserror = { version = "2.0.6", default-features = false } zerocopy = { version = "0.8", default-features = false, features = ["derive"] } diff --git a/litebox_syscall_rewriter/src/arm64.rs b/litebox_syscall_rewriter/src/arm64.rs new file mode 100644 index 0000000000..51c6368e41 --- /dev/null +++ b/litebox_syscall_rewriter/src/arm64.rs @@ -0,0 +1,2029 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! AArch64 (ARM64) syscall rewriting support for Linux ELF binaries. +//! +//! Every AArch64 instruction is 4 bytes including a direct branch (`B imm26`) +//! with a ±128MB range. This lets us replace a single instruction with +//! a branch into the trampoline without instruction borrowing. +//! +//! The trampoline is placed just past the highest mapped segment, so every +//! site-to-gate branch points forward. A site farther than the `B imm26` +//! ±128MB reach from its gate cannot redirect; it is replaced with a sentinel +//! `BRK #TRAP_BRK_IMM` and reported as a trapped site. Any trapped site makes +//! the rewrite incomplete, so the ELF-level caller rejects the binary with +//! `Error::UnpatchableSyscalls`, mirroring the x86-64 unpatchable-syscall path. +//! Executing the `BRK` raises a synchronous debug exception, so a trapped site +//! faults the guest rather than letting the unpatched instruction escape to the +//! host kernel. Recognizing the `TRAP_BRK_IMM` immediate in the runtime — to +//! attribute the trap to the rewriter rather than a guest breakpoint — is +//! planned but not yet implemented. +//! +//! ### Assumption: executable sections contain only instructions +//! +//! The patch scan walks each executable section word-by-word and treats every +//! 4-byte word that matches the `SVC`/`MSR TPIDR_EL0`/`MRS TPIDR_EL0` bit +//! patterns as that instruction. It does **not** distinguish inline data — literal +//! pools or jump tables embedded in `.text` — from code, because a fixed-width +//! decode cannot tell a data word from an instruction with the same bits. In +//! practice this is safe: default AArch64 codegen places constants in `.rodata`, +//! not `.text`, and the odds of an unrelated data word colliding with these +//! patterns are tiny. A binary that stores such a word inside an executable section +//! would have it rewritten; bounding the scan to symbol-defined function ranges +//! (via `STT_FUNC` extents) would remove the assumption (TODO). +//! +//! Two kinds of access are involved, three forms of instruction gated: +//! +//! * `SVC #imm` — the syscall instruction (any immediate; Linux ignores it). +//! Replaced with a branch to a per-site *SVC gate* that records the return +//! address and falls through to the shared SVC handler, a thin shim that +//! tail-jumps to the syscall callback. +//! * `MSR TPIDR_EL0, Xn` — a write to the thread pointer. Replaced with a branch +//! to a per-site *MSR gate* that stores the guest value into the guest +//! thread-pointer slot at `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]`. +//! * `MRS Xd, TPIDR_EL0` — a read of the thread pointer. Replaced with a branch +//! to a per-site *MRS gate* that loads the guest value from the same slot. +//! `MRS XZR, TPIDR_EL0` is a discarded read and is left native. +//! +//! ## Thread-pointer virtualization +//! +//! The host owns the hardware `TPIDR_EL0` as a per-thread anchor; the guest's +//! logical thread pointer is a host-managed memory slot at `[TPIDR_EL0 + +//! GUEST_TPIDR_OFFSET]`. Every gated guest read/write of the thread pointer +//! addresses that slot with a scaled `LDR`/`STR` off the anchor: +//! +//! * the MSR gate reads the anchor (`MRS X16, TPIDR_EL0`) and stores the guest +//! value into the slot; +//! * the MRS gate reads the anchor and loads the guest value from the slot. +//! +//! This mirrors the x64 model: the host keeps the native thread-pointer anchor, +//! the guest is statically relegated off it, and the gates emit nothing +//! TLS-related to the callback. +//! +//! ## Gate scratch storage and the stack invariant +//! +//! `SVC` and `MSR TPIDR_EL0` clobber no general-purpose registers, so a gate has +//! no free scratch register on entry. The SVC and MSR gates therefore spill their +//! scratch registers (and, for SVC, the computed return address the callback +//! reads back) to a frame carved out of the guest stack with `SUB SP, SP, #frame` +//! / `ADD SP, SP, #frame`. The MRS gate normally needs no frame: it reuses its own +//! destination register as scratch and never touches the stack. It does take one +//! when the host reads its guest thread-pointer offset at run time +//! (`GuestTpAddressing::RuntimeSlot`), because holding that offset costs a second +//! register that the guest may still be using — X16/X17 are reserved for linker +//! veneers, not dead at an arbitrary instruction the way they are at an `SVC`. +//! +//! Consequently those gates **require `SP` to hold a valid, writable, +//! 16-byte-aligned stack at the patched site** — the same condition the kernel +//! relies on when it writes a signal frame below `SP`, and which every conforming +//! AArch64 caller already satisfies at a syscall boundary. The gate decrements +//! `SP` before storing, so nothing (signal delivery included) writes into the +//! frame while it is live; there is no red-zone hazard. A site reached with `SP` +//! pointing at unmapped or guard memory would fault where the native instruction +//! would not. AArch64 offers no cheaper alternative: with no segment-relative +//! store (unlike x86's `gs:`-relative spill) and no free register, reaching any +//! runtime-owned scratch area would itself require first clobbering an unsaved +//! guest register to materialize a base pointer. +//! +//! ## Trampoline layout (Linux) +//! +//! ```text +//! Offset 0: [8 bytes] syscall callback address (filled at load time) +//! Offset 8: [8 bytes] shared SVC handler (LDR X16,; BR X16) +//! Offset 16: per-site gates (SVC: 24 bytes, MSR: 36 bytes, MRS: 12 bytes) +//! ``` +//! +//! A binary with **no** patch sites gets no trampoline at all: the rewriter +//! appends only a size-0 sentinel header (matching the x86-64 path), recording +//! that the image was checked and needs no redirection. Signal returns are +//! handled by the runtime (see "Signal returns" below). +//! +//! The offset-0 callback address is **filled in by the loader/runtime, not by +//! this crate.** The emitted trampolines are therefore *not runnable as-is*: a +//! loader must write the syscall-callback address at offset 0 before any guest +//! `SVC` reaches a gate. (`callback` may be passed to [`hook_syscalls_aarch64`] +//! to prefill offset 0.) The callback reads host TLS from `TPIDR_EL0` and the +//! guest thread pointer from `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]` itself. +//! +//! ## Signal returns +//! +//! This crate emits no sigreturn gate; `rt_sigreturn` is handled by the runtime. +//! The runtime installs its own sigreturn trampoline address into the signal +//! frame's return slot; because that is an absolute address (not a `B`), a +//! single runtime-owned gate is reachable from any guest regardless of the +//! ±128MB branch range, so no per-binary gate is required. +//! +//! ## Runtime contract +//! +//! Per thread, the runtime sets the hardware `TPIDR_EL0` to the host anchor and +//! reserves the guest thread-pointer slot at `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]`. +//! No new callback ABI is introduced: the callback reaches host TLS through +//! `TPIDR_EL0` directly. Multi-threaded correctness depends only on the runtime +//! keeping the anchor valid and the slot reachable for every thread it starts; +//! no process-global table is involved, so concurrent threads never contend. +//! +//! ## Host-OS scope +//! +//! This module fully virtualizes the guest thread pointer against a stable +//! per-thread host anchor register. The model is host-OS-agnostic; only the +//! choice of anchor register varies per host, selected by [`Host`] (a gate names +//! its anchor through [`Host::anchor_read`]). On a Linux host the anchor is +//! `TPIDR_EL0` itself: the kernel preserves it across host execution, so the host +//! can keep its own value there as the anchor while the guest thread pointer +//! lives in the slot beside it. The instruction encoders and gate framing here +//! are host-agnostic; see [`Host`] for the per-host anchor registers and what +//! each additional host requires. + +use alloc::format; +use alloc::vec::Vec; + +use crate::{Error, Result, TextSectionInfo, checked_add_u64}; + +// ============================================================ +// Constants +// ============================================================ + +/// `SVC #0` (supervisor call) — the canonical syscall instruction. +const SVC_0: u32 = 0xD400_0001; + +/// Mask/match for *any* `SVC #imm16`. Linux dispatches every `SVC64` exception +/// to the syscall handler regardless of the immediate (the syscall number comes +/// from `x8`), so all immediates are rewritten, not just `svc #0`. The `imm16` +/// field occupies bits \[20:5]; masking it out leaves bits \[4:0] = `0b00001`, +/// which distinguishes `SVC` from `HVC` (`…0b10`) and `SMC` (`…0b11`). +const SVC_OPCODE_MASK: u32 = 0xFFE0_001F; +const SVC_OPCODE_BITS: u32 = SVC_0; + +/// Mask/match for `MSR TPIDR_EL0, Xt` (`0xD51BD04t`, the low 5 bits select Xt). +const MSR_TPIDR_EL0_MASK: u32 = 0xFFFF_FFE0; +const MSR_TPIDR_EL0_BITS: u32 = 0xD51B_D040; + +/// Mask/match for `MRS Xd, TPIDR_EL0` (`0xD53BD04d`, the low 5 bits select Xd). +const MRS_TPIDR_EL0_MASK: u32 = 0xFFFF_FFE0; +const MRS_TPIDR_EL0_BITS: u32 = 0xD53B_D040; + +/// `MRS Xd, TPIDRRO_EL0` (`0xD53BD06d`) — [`Host::MacOs`]'s anchor read. +/// `TPIDRRO_EL0` differs from `TPIDR_EL0` only in the `op2` system-register +/// field (3 instead of 2), which is bit 5 of the instruction word, so this is +/// exactly [`MRS_TPIDR_EL0_BITS`] with that bit set. Confirmed against a real +/// toolchain (`cc`/`otool -tvV` on an Apple M3 Pro): `mrs x9, TPIDR_EL0` and +/// `mrs x9, TPIDRRO_EL0` assemble to `0xD53BD049` and `0xD53BD069` +/// respectively. This is the EL0-*read-only* companion register, which Darwin +/// keeps pointing at the current pthread's thread-specific-data base; it is an +/// emitted-gate anchor only, never a scanned patch pattern (a Linux guest +/// image has no reason to contain it, and rewriting one would be wrong +/// anyway). +const MRS_TPIDRRO_EL0_BITS: u32 = 0xD53B_D060; + +/// `BRK` immediate planted at a patch site whose gate lies outside the `B` +/// instruction's ±128MB reach. Executing the site raises a synchronous debug +/// exception (`SIGTRAP`) carrying this immediate, faulting the guest rather than +/// letting the unpatched instruction escape to the host kernel; the site is also +/// reported as a trapped site so the ELF-level caller can reject the binary. +/// +/// Recognizing this immediate in the runtime — to attribute the trap to the +/// rewriter rather than a guest breakpoint — is planned but not yet implemented. +const TRAP_BRK_IMM: u16 = 0xB10B; + +// --- Register operands used by the emitted gates/handlers --- +// +// X16/X17 are the intra-procedure scratch registers (IP0/IP1), and register +// number 31 names the stack pointer in a base-register position. + +/// First scratch register (IP0). +const X16: u8 = 16; +/// Second scratch register (IP1). +const X17: u8 = 17; +/// Stack pointer (encoded as register 31 in a base-register field). +const SP: u8 = 31; +/// Zero register (register 31 in a transfer-register field, where it reads as +/// zero / discards writes — distinct from `SP`'s base-register meaning). +const XZR: u8 = 31; + +// --- Guest thread-pointer virtualization --- +// +// The host owns the hardware `TPIDR_EL0` as a per-thread anchor; the guest's +// logical thread pointer is a memory slot the runtime reserves at a fixed byte +// offset from that anchor. Every gated guest read/write of the thread pointer +// addresses the slot with a scaled `LDR`/`STR` off `TPIDR_EL0`. + +/// Byte offset from the host anchor in `TPIDR_EL0` at which the runtime reserves +/// this thread's guest thread-pointer slot. Every guest read/write of the thread +/// pointer is virtualized to `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]` via a scaled +/// `LDR`/`STR`. +/// +/// Fixed ABI offset: the runtime points `TPIDR_EL0` at a per-thread block whose +/// `guest_tp` field sits just past the AArch64 variant-1 16-byte TCB header, so a +/// stray "deref `TPIDR_EL0` as a TCB" cannot mistake the guest pointer for the +/// dtv slot. Because the scaled immediate is baked into statically rewritten +/// binaries, this value is part of the rewriter/runtime ABI and must match the +/// runtime's block layout. +const GUEST_TPIDR_OFFSET: u16 = 16; + +/// [`Host::MacOs`]'s counterpart of [`GUEST_TPIDR_OFFSET`]: the byte offset +/// from the macOS anchor (`TPIDRRO_EL0`) of the guest thread-pointer slot. +/// +/// On Darwin the anchor register is *kernel-owned*: `TPIDRRO_EL0` points at +/// the current pthread's thread-specific-data array (TSD slot `N` lives at +/// `[TPIDRRO_EL0 + N*8]`, with no low-bit masking on arm64 — verified against +/// xnu's `libsyscall/os/tsd.h` `_os_tsd_get_base`), so the runtime cannot +/// point it at a block of its own the way the Linux runtime does with +/// `TPIDR_EL0`. Instead the guest thread-pointer slot is a pthread TSD slot: +/// index [`MACOS_GUEST_TPIDR_TSD_SLOT`], i.e. byte offset `256 * 8`. +/// +/// Slot 256 is the *first dynamic key* on macOS (`pthread_key_create` hands +/// out keys 256..768 there; 0..255 are reserved static keys — values verified +/// against apple-oss-distributions/libpthread `pthread_tsd.c` and +/// `types_internal.h`). The runtime owns it by calling `pthread_key_create` +/// during platform init, before anything else in the process creates a +/// dynamic key, and verifying it was handed exactly this slot; reading and +/// writing the slot directly off `TPIDRRO_EL0` is then the same "direct TSD" +/// fast path libSystem's own `errno` accessor and WebKit's `FastTLS` +/// (`_pthread_getspecific_direct`) use. Like [`GUEST_TPIDR_OFFSET`], this is +/// rewriter/runtime ABI: statically rewritten binaries bake the scaled +/// immediate in. +const GUEST_TPIDR_OFFSET_MACOS: u16 = MACOS_GUEST_TPIDR_TSD_SLOT * 8; + +/// The pthread TSD slot index backing [`GUEST_TPIDR_OFFSET_MACOS`]. Exported +/// (via the crate root) so the macOS runtime reserves exactly the slot the +/// emitted gates address, rather than the two ever drifting apart. +pub(crate) const MACOS_GUEST_TPIDR_TSD_SLOT: u16 = 256; + +// --- SVC gate stack frame --- +// +// The SVC gate touches only X16, so it needs a minimal 16-byte frame: one slot +// for the saved guest X16 and one for the computed post-SVC return address. + +/// SVC gate frame size (`SUB/ADD SP, SP, #SVC_FRAME_BYTES`). 16-byte aligned. +const SVC_FRAME_BYTES: u16 = 16; +/// Saved guest X16. +const SVC_FRAME_OFF_X16: u16 = 0; +/// Computed post-SVC return address. +const SVC_FRAME_OFF_RETADDR: u16 = 8; + +// --- MSR gate stack frame --- +// +// The MSR gate spills X16/X17 (one `STP`/`LDP` pair) and stages the captured +// guest value so the source register needs no special-casing. + +/// MSR gate frame size (`SUB/ADD SP, SP, #MSR_FRAME_BYTES`). 16-byte aligned. +const MSR_FRAME_BYTES: u16 = 32; +/// Saved X16 (and, +8, X17 via the `STP`/`LDP` pair). +const MSR_FRAME_OFF_X16: u16 = 0; +/// Captured guest thread-pointer value, staged while all guest registers are +/// still pristine. +const MSR_FRAME_OFF_VALUE: u16 = 16; + +// --- MRS gate stack frame --- +// +// Only the `GuestTpAddressing::RuntimeSlot` form needs a frame, to borrow one +// scratch register for the loaded offset. The baked form touches only `Rd` and +// emits no frame at all. + +/// MRS gate frame size (`SUB/ADD SP, SP, #MRS_FRAME_BYTES`). 16-byte aligned: +/// AArch64 requires `SP` to stay 16-byte aligned on every access that uses it as +/// a base, so 16 is the smallest legal frame even though one register is spilled. +const MRS_FRAME_BYTES: u16 = 16; +/// Saved scratch register. +const MRS_FRAME_OFF_SCRATCH: u16 = 0; + +// --- Trampoline layout offsets (all in bytes) --- + +/// Callback address slot. +const HEADER_CALLBACK_OFFSET: usize = 0; + +/// Guest thread-pointer *byte offset* slot, read by [`Host::MacOs`] gates. +/// +/// Holds the byte offset from the host anchor at which the runtime keeps the +/// guest thread pointer — never the thread pointer itself. A `Host::MacOs` gate +/// loads this word and addresses `[TPIDRRO_EL0 + offset]`, so the number the +/// gates use is settled when the image is loaded rather than when it is +/// packaged. That is the whole point of the slot: the TSD key a macOS runtime +/// gets from `pthread_key_create` is a property of the runner binary's own +/// startup sequence, which the ahead-of-time rewriter cannot know. +/// +/// A thread-pointer *value* could not live here. The loader maps the trampoline +/// writable, fills the header, then flips it to read+execute +/// (`litebox_common_linux`'s `load_trampoline`), so nothing may write this word +/// again once a guest runs — and one word cannot serve two threads anyway. An +/// offset is constant for the process; the per-thread part comes from +/// `TPIDRRO_EL0`, which is already per-thread. +/// +/// The rewriter seeds it with [`GUEST_TPIDR_OFFSET_MACOS`], so an image whose +/// loader does not fill the slot keeps the exact behaviour of the earlier +/// baked-immediate gates instead of addressing offset zero — which would be +/// pthread TSD slot 0, live libpthread state. +pub(crate) const HEADER_GUEST_TP_OFFSET_MACOS: usize = HEADER_CALLBACK_OFFSET + 8; + +/// Shared SVC handler, placed just past the trampoline header slots. Per-site gates +/// follow it and are each appended dynamically, so this shared prologue is the +/// only fixed-offset region the emitters reference. +const SHARED_SVC_HANDLER_OFFSET: usize = HEADER_GUEST_TP_OFFSET_MACOS + 8; + +// ============================================================ +// Instruction encoders +// +// Each encoder returns the 32-bit little-endian instruction word. Encoders that +// can fail range checks return `Option`; callers convert `None` into an +// `Error::AddressOverflow` with context. +// +// Each encoder ORs an [`Opcode`] base with its shifted, masked operands. The +// `IMM*_MASK` values isolate the immediate fields shared by several encoders. +// ============================================================ + +/// 26-bit `imm26` branch-offset field (`B`/`BL`), bits \[25:0]. +const IMM26_MASK: u32 = 0x03FF_FFFF; +/// 19-bit `imm19` offset field (`B.cond`/`LDR`-literal/`ADRP` immhi), bits \[18:0]. +const IMM19_MASK: u32 = 0x0007_FFFF; + +/// Base opcode of an emitted instruction: every fixed bit set with all operand +/// fields zeroed. An encoder selects a variant and ORs in its operands via +/// [`Opcode::bits`]. (`MRS TPIDR_EL0` is encoded from [`Opcode::MrsTpidrEl0`], +/// whose bits equal [`MRS_TPIDR_EL0_BITS`] — the scan-detection pattern in +/// [`find_patch_sites`].) +#[repr(u32)] +#[derive(Clone, Copy)] +enum Opcode { + B = 0x1400_0000, + LdrLiteral = 0x5800_0000, + Adrp = 0x9000_0000, + Br = 0xD61F_0000, + SubImm = 0xD100_0000, + AddImm = 0x9100_0000, + StrUimm = 0xF900_0000, + LdrUimm = 0xF940_0000, + /// `LDR Xt, [Xn, Xm]` — register offset, `option=LSL`, `S=0`, so `Xm` is a + /// byte offset rather than an element index. The base word already carries + /// bit 21 and `bits[11:10]=0b10`; those two fields are what select the + /// register-offset class at all, and clearing either one decodes as an + /// `LDUR` with a garbage immediate rather than failing. + LdrReg = 0xF860_6800, + /// `ADD Xd, Xn, Xm` — shifted register, `LSL #0`. + AddReg = 0x8B00_0000, + Stp = 0xA900_0000, + Ldp = 0xA940_0000, + MrsTpidrEl0 = MRS_TPIDR_EL0_BITS, + /// [`Host::MacOs`]'s anchor read. + MrsTpidrroEl0 = MRS_TPIDRRO_EL0_BITS, + Brk = 0xD420_0000, +} + +impl Opcode { + /// The base opcode word, for ORing in operand fields. + const fn bits(self) -> u32 { + self as u32 + } +} + +// --- Shared instruction-format encoders --- +// +// Several instructions share one field layout and differ only by opcode, so +// each layout is encoded once here and selected by an `Opcode`. [`Insn::encode`] +// dispatches each variant to its format here; every range check lives in exactly +// one place per format. + +/// `op | imm26` — PC-relative branch (`B`/`BL`), ±128MB, 4-byte aligned. +fn branch_imm26(op: Opcode, offset: i64) -> Option { + if offset % 4 != 0 { + return None; + } + let imm26 = i32::try_from(offset >> 2).ok()?; + if !(-(1 << 25)..(1 << 25)).contains(&imm26) { + return None; + } + Some(op.bits() | (imm26.cast_unsigned() & IMM26_MASK)) +} + +/// `op | imm19<<5 | low` — PC-relative imm19 form (`B.cond`/`LDR`-literal), ±1MB, +/// 4-byte aligned. `low` is the instruction's 5-bit \[4:0] field: `Rt`, or the +/// condition code for `B.cond`. +fn pcrel_imm19(op: Opcode, offset: i64, low: u32) -> Option { + if offset % 4 != 0 { + return None; + } + let imm19 = i32::try_from(offset >> 2).ok()?; + if !(-(1 << 18)..(1 << 18)).contains(&imm19) { + return None; + } + Some(op.bits() | ((imm19.cast_unsigned() & IMM19_MASK) << 5) | low) +} + +/// `op | rn<<5` — instruction whose only operand is a register in the `Rn` field +/// (`BR`/`RET`). +fn reg_in_rn(op: Opcode, rn: u8) -> u32 { + op.bits() | (u32::from(rn) << 5) +} + +/// `op | imm12<<10 | rn<<5 | rd` — 12-bit-immediate add/sub form +/// (`ADD`/`SUB`/`ADDS`). The caller supplies an already-scaled `imm12`. +fn data_imm12(op: Opcode, rd: u8, rn: u8, imm12: u16) -> Option { + if imm12 >= (1 << 12) { + return None; + } + Some(op.bits() | (u32::from(imm12) << 10) | (u32::from(rn) << 5) | u32::from(rd)) +} + +/// `op | imm12<<10 | rn<<5 | rt` — unsigned scaled (×8) 64-bit load/store +/// (`STR`/`LDR [Xn, #imm]`). `imm_bytes` must be a multiple of 8. +fn ldst_uimm12(op: Opcode, rt: u8, rn: u8, imm_bytes: u16) -> Option { + if !imm_bytes.is_multiple_of(8) { + return None; + } + let imm12 = imm_bytes / 8; + if imm12 >= (1 << 12) { + return None; + } + Some(op.bits() | (u32::from(imm12) << 10) | (u32::from(rn) << 5) | u32::from(rt)) +} + +/// `op | rm<<16 | rn<<5 | rt` — register-offset 64-bit load +/// (`LDR Xt, [Xn, Xm]`). Every fixed field, including the `option`, `S` and +/// bit-21 bits that select the register-offset class, is already part of `op`. +fn ldst_reg_offset(op: Opcode, rt: u8, rn: u8, rm: u8) -> u32 { + op.bits() | (u32::from(rm) << 16) | (u32::from(rn) << 5) | u32::from(rt) +} + +/// `op | rm<<16 | rn<<5 | rd` — three-register data-processing form +/// (`ADD Xd, Xn, Xm`). +fn data_reg(op: Opcode, rd: u8, rn: u8, rm: u8) -> u32 { + op.bits() | (u32::from(rm) << 16) | (u32::from(rn) << 5) | u32::from(rd) +} + +/// `op | imm7<<15 | rt2<<10 | rn<<5 | rt` — signed scaled (×8) 64-bit load/store +/// pair (`STP`/`LDP`). `imm_bytes` must be a multiple of 8 within ±512 bytes. +fn ldst_pair(op: Opcode, rt: u8, rt2: u8, rn: u8, imm_bytes: i16) -> Option { + if imm_bytes % 8 != 0 { + return None; + } + let imm7 = imm_bytes / 8; + if !(-64..=63).contains(&imm7) { + return None; + } + let imm7_u = u32::from(imm7.cast_unsigned() & 0x7F); + Some(op.bits() | (imm7_u << 15) | (u32::from(rt2) << 10) | (u32::from(rn) << 5) | u32::from(rt)) +} + +/// `base | rt` — system-register move (`MRS`/`MSR`); `base` already encodes the +/// system register and transfer direction. +fn sysreg_move(base: u32, rt: u8) -> u32 { + base | u32::from(rt) +} + +/// A single AArch64 instruction emitted into a trampoline, described by its +/// mnemonic and operands. [`Insn::encode`] produces the 32-bit little-endian +/// word; range-checked forms return `None` when an operand is out of range. +/// +/// Register operands are register numbers (`X16`, `SP`, ...). This enum, with +/// the format helpers above, is the only place instruction bit layouts live; +/// the gate emitters build `Insn` values and never touch raw opcodes. +#[derive(Clone, Copy)] +enum Insn { + /// `B` (unconditional branch), PC-relative, ±128MB, 4-byte aligned. + B(i64), + /// `ADRP Xd, #page_off` — page-relative address, ±4GB (in 4KB pages). + Adrp { rd: u8, page_off: i64 }, + /// `LDR Xt, ` (PC-relative literal load), ±1MB, 4-byte aligned. + LdrLiteral { rt: u8, off: i64 }, + /// `BR Xn` (branch to register). + Br(u8), + /// `SUB SP, SP, #imm12`. + SubSp(u16), + /// `ADD SP, SP, #imm12`. + AddSp(u16), + /// `ADD Xd, Xn, #imm12`. + AddImm { rd: u8, rn: u8, imm12: u16 }, + /// `STR Xt, [Xn, #imm_bytes]` (unsigned scaled; `imm_bytes` multiple of 8). + StrUimm { rt: u8, rn: u8, imm_bytes: u16 }, + /// `LDR Xt, [Xn, #imm_bytes]` (unsigned scaled; `imm_bytes` multiple of 8). + LdrUimm { rt: u8, rn: u8, imm_bytes: u16 }, + /// `LDR Xt, [Xn, Xm]` — register offset, unscaled, so `Xm` is a byte offset. + LdrReg { rt: u8, rn: u8, rm: u8 }, + /// `ADD Xd, Xn, Xm`. + AddReg { rd: u8, rn: u8, rm: u8 }, + /// `STP Xt, Xt2, [Xn, #imm_bytes]` (signed scaled; `imm_bytes` multiple of 8). + Stp { + rt: u8, + rt2: u8, + rn: u8, + imm_bytes: i16, + }, + /// `LDP Xt, Xt2, [Xn, #imm_bytes]` (signed scaled; `imm_bytes` multiple of 8). + Ldp { + rt: u8, + rt2: u8, + rn: u8, + imm_bytes: i16, + }, + /// `MRS Xt, TPIDR_EL0` (read thread pointer). + MrsTpidrEl0(u8), + /// `MRS Xt, TPIDRRO_EL0` (read the EL0-read-only thread register — the + /// Darwin pthread TSD base; [`Host::MacOs`]'s anchor read). + MrsTpidrroEl0(u8), + /// `BRK #imm16` — software breakpoint raising a synchronous debug exception. + Brk(u16), +} + +impl Insn { + /// Encode to a 32-bit little-endian instruction word, or `None` if an + /// operand is outside the instruction's encodable range. + fn encode(self) -> Option { + match self { + Insn::B(off) => branch_imm26(Opcode::B, off), + Insn::Adrp { rd, page_off } => { + let imm = i32::try_from(page_off).ok()?; + if !(-(1 << 20)..(1 << 20)).contains(&imm) { + return None; + } + let imm = imm.cast_unsigned(); + let immlo = (imm & 0x3) << 29; + let immhi = ((imm >> 2) & IMM19_MASK) << 5; + Some(Opcode::Adrp.bits() | immlo | immhi | u32::from(rd)) + } + Insn::LdrLiteral { rt, off } => pcrel_imm19(Opcode::LdrLiteral, off, u32::from(rt)), + Insn::Br(rn) => Some(reg_in_rn(Opcode::Br, rn)), + Insn::SubSp(imm12) => data_imm12(Opcode::SubImm, SP, SP, imm12), + Insn::AddSp(imm12) => data_imm12(Opcode::AddImm, SP, SP, imm12), + Insn::AddImm { rd, rn, imm12 } => data_imm12(Opcode::AddImm, rd, rn, imm12), + Insn::StrUimm { rt, rn, imm_bytes } => ldst_uimm12(Opcode::StrUimm, rt, rn, imm_bytes), + Insn::LdrUimm { rt, rn, imm_bytes } => ldst_uimm12(Opcode::LdrUimm, rt, rn, imm_bytes), + Insn::LdrReg { rt, rn, rm } => Some(ldst_reg_offset(Opcode::LdrReg, rt, rn, rm)), + Insn::AddReg { rd, rn, rm } => Some(data_reg(Opcode::AddReg, rd, rn, rm)), + Insn::Stp { + rt, + rt2, + rn, + imm_bytes, + } => ldst_pair(Opcode::Stp, rt, rt2, rn, imm_bytes), + Insn::Ldp { + rt, + rt2, + rn, + imm_bytes, + } => ldst_pair(Opcode::Ldp, rt, rt2, rn, imm_bytes), + Insn::MrsTpidrEl0(rt) => Some(sysreg_move(Opcode::MrsTpidrEl0.bits(), rt)), + Insn::MrsTpidrroEl0(rt) => Some(sysreg_move(Opcode::MrsTpidrroEl0.bits(), rt)), + Insn::Brk(imm) => Some(Opcode::Brk.bits() | (u32::from(imm) << 5)), + } + } +} + +// ============================================================ +// Host anchor selection +// ============================================================ + +/// The host OS the rewritten guest runs under. +/// +/// The guest thread pointer is virtualized the same way on every host; only the +/// *anchor register* a gate reads to reach the host's per-thread block varies. +/// [`Host`] selects that register, so a gate names the anchor through +/// `Host::anchor_read` rather than hardcoding a system register. Adding a host +/// is a new variant plus its anchor-read arm. +/// +/// A host OS beyond these two needs a different stable anchor register (a +/// future variant supplying its own `anchor_read` and `guest_tp_offset`): +/// +/// * **Linux-on-Windows** (Windows on ARM64): Windows does not preserve +/// `TPIDR_EL0` across context switches and reserves `x18` as the TEB pointer +/// (always valid). The TEB is the stable anchor: the per-thread TLS state is +/// reached through a TEB TLS slot, and `TPIDR_EL0` (plus guest `x18`, where the +/// guest uses it) is virtualized against that. +#[derive(Clone, Copy)] +pub enum Host { + /// Linux host. The kernel preserves `TPIDR_EL0` across host execution, so the + /// host keeps its anchor there and the guest thread-pointer slot lives beside + /// it; the anchor read is `MRS Xd, TPIDR_EL0`. + Linux, + /// macOS host (Apple Silicon). Confirmed on real hardware (Apple M3 Pro, + /// macOS 26.3.1): XNU clobbers `TPIDR_EL0` across both a voluntary + /// context switch and a signal-handler invocation — not merely leaves it + /// stale, but overwrites it with its own per-thread value — so it cannot + /// anchor the guest thread pointer. The anchor is the EL0-read-only + /// `TPIDRRO_EL0` (confirmed stable across a reschedule and distinct per + /// thread on the same hardware), which Darwin keeps pointing at the + /// current pthread's TSD base. The runtime cannot repoint a read-only, + /// kernel-owned register at a block of its own the way the Linux runtime + /// does with `TPIDR_EL0` — addressing a fixed raw offset into the pthread + /// structure it points at would corrupt live libpthread state — so the + /// guest thread-pointer slot is instead a *runtime-reserved pthread TSD + /// slot*, addressed at `GUEST_TPIDR_OFFSET_MACOS`; see that constant + /// for the verified TSD layout and how the runtime reserves the slot. + /// + /// Caveat, deliberate and documented rather than silently wrong: XNU also + /// zeroes `x18` on every return to EL0 for ordinary (non-Rosetta, + /// non-entitled) processes, so a guest binary that uses `x18` as a live + /// general-purpose register cannot run correctly under this host. `x18` + /// uses cannot be found reliably without full disassembly (any operand + /// field of any instruction can name it), so they are not scanned or + /// gated; guests should be built with `-ffixed-x18`. Linux AArch64 + /// distro binaries generally treat `x18` as allocatable, so this is a + /// real restriction, not a formality. + MacOs, +} + +impl Host { + /// The instruction a gate uses to read this host's per-thread anchor into + /// `rd`. + fn anchor_read(self, rd: u8) -> Insn { + match self { + Host::Linux => Insn::MrsTpidrEl0(rd), + Host::MacOs => Insn::MrsTpidrroEl0(rd), + } + } + + /// How a gate on this host reaches the guest thread-pointer slot. + fn guest_tp_addressing(self) -> GuestTpAddressing { + match self { + Host::Linux => GuestTpAddressing::Baked(GUEST_TPIDR_OFFSET), + Host::MacOs => GuestTpAddressing::RuntimeSlot, + } + } +} + +/// How a gate obtains the byte offset from the host anchor to the guest +/// thread-pointer slot. +/// +/// The hosts differ in *when* that number is knowable, not in what it means, so +/// the choice is spelled out here rather than left implicit in each emitter. +#[derive(Clone, Copy)] +enum GuestTpAddressing { + /// Rewriter/runtime ABI, fixed at packaging time and baked into each gate's + /// scaled `LDR`/`STR` immediate. Linux qualifies: its runtime owns + /// `TPIDR_EL0` outright and puts the slot at a compile-time offset beside its + /// own block. + Baked(u16), + /// Read at run time from [`HEADER_GUEST_TP_OFFSET_MACOS`], because the + /// packaging-time rewriter cannot know it. macOS qualifies: the anchor is + /// kernel-owned and the slot is a pthread TSD key whose number depends on the + /// runner binary's own startup sequence. + RuntimeSlot, +} + +// ============================================================ +// Patch-site scanning +// ============================================================ + +/// A located instruction to rewrite. +struct PatchSite { + /// Byte offset of the instruction within the ELF file image. + file_offset: usize, + /// Virtual address of the instruction. + vaddr: u64, + kind: PatchKind, +} + +/// The kind of instruction at a [`PatchSite`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum PatchKind { + /// `SVC #imm` for any immediate. Linux dispatches every `SVC64` to the + /// syscall handler regardless of the immediate (the number comes from `x8`), + /// so the immediate is not significant and is not recorded. + Svc, + /// `MSR TPIDR_EL0, Xt`; the `u8` is the source register (0-31). + MsrTpidr(u8), + /// `MRS Xd, TPIDR_EL0`; the `u8` is the destination register (0-30). + MrsTpidr(u8), +} + +/// Scan all executable sections for `SVC #imm`, `MSR TPIDR_EL0` (thread-pointer +/// writes), and `MRS TPIDR_EL0` (thread-pointer reads). AArch64 instructions are +/// always 4-byte aligned, so we step in 4-byte units. Returns sites in ascending +/// file order. +/// +/// `MRS XZR, TPIDR_EL0` is a discarded read (register 31 as an `LDR` base would +/// mean `SP`), so it is left native; every other `MRS Xd, TPIDR_EL0` is gated. +fn find_patch_sites(sections: &[TextSectionInfo], buf: &[u8]) -> Result> { + let mut sites = Vec::new(); + + for section in sections { + let start = usize::try_from(section.file_offset) + .map_err(|_| Error::ParseError("section file offset too large".into()))?; + let size = usize::try_from(section.size) + .map_err(|_| Error::ParseError("section size too large".into()))?; + let end = start + .checked_add(size) + .filter(|&e| e <= buf.len()) + .ok_or_else(|| Error::ParseError("section extends beyond file".into()))?; + let section_data = &buf[start..end]; + + for i in (0..section_data.len()).step_by(4) { + if i + 4 > section_data.len() { + break; + } + let insn = u32::from_le_bytes(section_data[i..i + 4].try_into().unwrap()); + let kind = if (insn & SVC_OPCODE_MASK) == SVC_OPCODE_BITS { + PatchKind::Svc + } else if (insn & MSR_TPIDR_EL0_MASK) == MSR_TPIDR_EL0_BITS { + PatchKind::MsrTpidr((insn & 0x1F) as u8) + } else if (insn & MRS_TPIDR_EL0_MASK) == MRS_TPIDR_EL0_BITS { + let rd = (insn & 0x1F) as u8; + // `MRS XZR, TPIDR_EL0` discards its result (a no-op read); gating + // it would mean using register 31 as an `LDR` base (= SP), so + // leave it native. + if rd == XZR { + continue; + } + PatchKind::MrsTpidr(rd) + } else { + continue; + }; + sites.push(PatchSite { + file_offset: start + i, + vaddr: checked_add_u64(section.vaddr, i as u64, "patch site")?, + kind, + }); + } + } + + Ok(sites) +} + +// ============================================================ +// Main hooking entry point +// ============================================================ + +/// Outcome of rewriting one AArch64 image's patch sites. +pub(crate) struct HookOutcome { + /// Trampoline blob the caller appends after the ELF (page-aligned). + pub trampoline: Vec, + /// Virtual addresses of patch sites that could not be redirected to their + /// gate — the inbound `B` or one of the gate's own branches fell outside the + /// branch's ±128MB range — and were replaced with a trap instead of a + /// redirect. A non-empty list means the rewrite is incomplete: those sites + /// fault at runtime rather than entering the trampoline. + pub trapped_sites: Vec, +} + +/// Hook all `SVC #imm`, `MSR TPIDR_EL0` writes, and `MRS TPIDR_EL0` reads in an +/// AArch64 ELF image. (`MRS XZR, TPIDR_EL0` is a discarded read and is left +/// native — see the module docs.) +/// +/// `buf` is patched in place; the returned [`HookOutcome::trampoline`] is the +/// blob that the caller appends after the ELF (page-aligned). +/// `trampoline_base_addr` is the virtual address the trampoline will be mapped +/// at; `callback` is the absolute address stored in the callback slot (0 if the +/// loader fills it in later). +/// +/// Returns `Ok(None)` when the image contains no patch sites: no trampoline is +/// needed and the caller emits a size-0 sentinel header instead (matching the +/// x86-64 path). Signal returns are handled by the runtime — not a per-binary +/// gate — so a syscall-free binary needs no trampoline at all. +/// +/// Otherwise returns `Ok(Some(outcome))`. A site whose inbound `B` cannot reach +/// its gate, or whose gate cannot branch back within the `B` instruction's +/// ±128MB reach, cannot be redirected; it is replaced with a trap and listed in +/// [`HookOutcome::trapped_sites`] so the caller can reject the incomplete +/// rewrite, mirroring the x86-64 unpatchable-syscall path. +pub(crate) fn hook_syscalls_aarch64( + buf: &mut [u8], + text_sections: &[TextSectionInfo], + trampoline_base_addr: u64, + callback: u64, + host: Host, +) -> Result> { + let sites = find_patch_sites(text_sections, buf)?; + + if sites.is_empty() { + // No patch sites: nothing to redirect, so no trampoline is + // emitted. The caller writes a size-0 sentinel header instead. + return Ok(None); + } + + let mut trampoline_data: Vec = Vec::new(); + emit_shared_prologue(&mut trampoline_data, trampoline_base_addr, callback)?; + + let mut trapped_sites: Vec = Vec::new(); + + for site in &sites { + let gate_offset = trampoline_data.len(); + let gate_vaddr = + checked_add_u64(trampoline_base_addr, gate_offset as u64, "trampoline gate")?; + + // A site is redirected to its gate with a single in-place `B` (±128MB + // forward reach), and each gate branches back to `site + 4` (the SVC gate + // also reaches its shared handler). The gate's return branch spans a wider + // displacement than the inbound one, so the inbound branch encoding is + // necessary but not sufficient: the gate is built only when the inbound + // branch fits, and a gate whose own branches are out of range reports + // `GateBuild::Unreachable` and appends nothing. If either the inbound + // branch or the gate is unreachable, replace the site with the sentinel + // trap, record it as unpatchable, and emit no gate. + let b_offset = gate_vaddr + .cast_signed() + .saturating_sub(site.vaddr.cast_signed()); + let inbound = Insn::B(b_offset).encode(); + + let build = if inbound.is_some() { + match site.kind { + PatchKind::Svc => emit_svc_gate( + &mut trampoline_data, + gate_offset, + trampoline_base_addr, + site, + )?, + PatchKind::MsrTpidr(rt) => emit_msr_gate( + &mut trampoline_data, + gate_offset, + trampoline_base_addr, + site, + rt, + host, + )?, + PatchKind::MrsTpidr(rd) => emit_mrs_gate( + &mut trampoline_data, + gate_offset, + trampoline_base_addr, + site, + rd, + host, + )?, + } + } else { + GateBuild::Unreachable + }; + + if let (Some(b_insn), GateBuild::Emitted) = (inbound, build) { + // Replace the original instruction with `B `. + buf[site.file_offset..site.file_offset + 4].copy_from_slice(&b_insn.to_le_bytes()); + } else { + let brk = Insn::Brk(TRAP_BRK_IMM) + .encode() + .expect("BRK always encodes"); + buf[site.file_offset..site.file_offset + 4].copy_from_slice(&brk.to_le_bytes()); + trapped_sites.push(site.vaddr); + } + } + + Ok(Some(HookOutcome { + trampoline: trampoline_data, + trapped_sites, + })) +} + +/// Emit the header slot and the shared SVC handler — the fixed-size shared +/// prologue that per-site gates follow. +fn emit_shared_prologue( + trampoline_data: &mut Vec, + trampoline_base_addr: u64, + callback: u64, +) -> Result<()> { + // Offset 0: callback address. + trampoline_data.extend_from_slice(&callback.to_le_bytes()); + + // Offset 8: guest thread-pointer byte offset, seeded with the packaging-time + // default so a loader that does not fill it leaves the gates behaving exactly + // as the earlier baked-immediate ones did. Zero would mean pthread TSD slot 0 + // on macOS, which is live libpthread state. + trampoline_data.extend_from_slice(&u64::from(GUEST_TPIDR_OFFSET_MACOS).to_le_bytes()); + + emit_shared_svc_handler( + trampoline_data, + SHARED_SVC_HANDLER_OFFSET, + trampoline_base_addr, + )?; + + Ok(()) +} + +// ============================================================ +// SVC gate + shared SVC handler +// ============================================================ + +/// Whether a gate was fully emitted or could not be placed within reach. +/// +/// A gate redirects back to the guest (and, for the SVC gate, out to the shared +/// handler) with PC-relative branches. When any of those branches is out of +/// range the gate emits nothing and reports [`GateBuild::Unreachable`], leaving +/// the trampoline blob untouched so the caller can trap the originating site. +enum GateBuild { + Emitted, + Unreachable, +} + +/// Per-site SVC gate (6 instructions, 24 bytes, 16-byte frame). +/// +/// Saves only X16 (already a scratch register), computes the post-SVC return +/// address into X16, records it on the frame, then branches to the shared SVC +/// handler. Guest X17/X18/LR and NZCV are untouched; the callback finds the +/// post-SVC return address at `[SP, #8]` and restores X16 from `[SP, #0]`. +/// +/// Frame layout (relative to the decremented SP): `[0]=X16 [8]=return_addr`. +/// Requires `SP` to address a valid writable stack at the site (see the module +/// docs, "Gate scratch storage and the stack invariant"). +fn emit_svc_gate( + trampoline_data: &mut Vec, + gate_offset: usize, + trampoline_base_addr: u64, + site: &PatchSite, +) -> Result { + let gate_vaddr = checked_add_u64(trampoline_base_addr, gate_offset as u64, "SVC gate")?; + let mut asm = Asm::new(gate_vaddr); + + // SUB SP, SP, #16 ; STR X16, [SP] — save the guest X16. + asm.emit(Insn::SubSp(SVC_FRAME_BYTES)); + asm.emit(Insn::StrUimm { + rt: X16, + rn: SP, + imm_bytes: SVC_FRAME_OFF_X16, + }); + + // ADRP X16, ; ADD X16, X16, # — post-SVC return + // address. + let return_addr = checked_add_u64(site.vaddr, 4, "SVC return")?; + if !asm.adrp(X16, return_addr)? { + return Ok(GateBuild::Unreachable); + } + let page_lo = u16::try_from(return_addr & 0xFFF).expect("masked to 12 bits"); + asm.emit(Insn::AddImm { + rd: X16, + rn: X16, + imm12: page_lo, + }); + + // STR X16, [SP, #8] — record the return address. + asm.emit(Insn::StrUimm { + rt: X16, + rn: SP, + imm_bytes: SVC_FRAME_OFF_RETADDR, + }); + + // B . + let handler_vaddr = checked_add_u64( + trampoline_base_addr, + SHARED_SVC_HANDLER_OFFSET as u64, + "SVC handler", + )?; + if !asm.branch_to(handler_vaddr)? { + return Ok(GateBuild::Unreachable); + } + + trampoline_data.extend_from_slice(&asm.finish()); + Ok(GateBuild::Emitted) +} + +/// Shared SVC handler (2 instructions, 8 bytes). +/// +/// A thin shim that conveys nothing TLS-related: it loads the syscall-callback +/// pointer from the trampoline header and tail-jumps to it. The callback reads +/// host TLS from `TPIDR_EL0` (the host anchor) and the guest thread pointer from +/// `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]` itself, so the handler carries no TLS state. +/// +/// Nothing in the handler clobbers NZCV, so the guest's pre-svc flags reach the +/// callback unchanged with no save/restore. +fn emit_shared_svc_handler( + trampoline_data: &mut Vec, + handler_offset: usize, + trampoline_base_addr: u64, +) -> Result<()> { + let handler_vaddr = + checked_add_u64(trampoline_base_addr, handler_offset as u64, "SVC handler")?; + let callback_vaddr = checked_add_u64( + trampoline_base_addr, + HEADER_CALLBACK_OFFSET as u64, + "callback slot", + )?; + let mut asm = Asm::new(handler_vaddr); + + // LDR X16, =callback ; BR X16. Nothing here clobbers NZCV, so the guest's + // pre-svc flags reach the callback unchanged with no save/restore. + asm.ldr_literal(X16, callback_vaddr)?; + asm.emit(Insn::Br(X16)); + + trampoline_data.extend_from_slice(&asm.finish()); + Ok(()) +} + +// ============================================================ +// MSR + MRS gates +// ============================================================ + +/// Per-site MSR gate (9 instructions, 36 bytes, 32-byte frame). +/// +/// Virtualizes a guest `MSR TPIDR_EL0, Xn` write. The hardware register holds the +/// host anchor, so the gate stores the guest value into the guest thread-pointer +/// slot at `[TPIDR_EL0 + GUEST_TPIDR_OFFSET]`: +/// +/// 1. spill X16/X17 and capture the guest value `Xn` to the frame while all guest +/// registers are still pristine (so `Xn` needs no special-casing, even when it +/// is one of the scratch registers just spilled (X16/X17) or XZR); +/// 2. `MRS X16, TPIDR_EL0` reads the host anchor; +/// 3. reload the captured value into X17 and `STR X17, [X16, #GUEST_TPIDR_OFFSET]` +/// stores it into the guest thread-pointer slot; +/// 4. restore X16/X17 and branch back to the instruction after the original MSR. +/// +/// The slot is always reachable because `TPIDR_EL0` is the host anchor the +/// runtime keeps valid, so a guest value of `0` (XZR) is an ordinary store — never +/// a fault. +/// +/// The X16/X17 spill and the captured value use a guest-stack frame, so this gate +/// requires `SP` to address a valid writable stack at the site (see the module +/// docs, "Gate scratch storage and the stack invariant"). +/// +/// `MSR TPIDR_EL0` does not touch the condition flags and the gate uses only +/// plain loads/stores and `B` (never `BL`), so NZCV and X30 reach the guest +/// unchanged with no save/restore. +fn emit_msr_gate( + trampoline_data: &mut Vec, + gate_offset: usize, + trampoline_base_addr: u64, + site: &PatchSite, + rt: u8, + host: Host, +) -> Result { + let gate_vaddr = checked_add_u64(trampoline_base_addr, gate_offset as u64, "MSR gate")?; + let mut asm = Asm::new(gate_vaddr); + + // SUB SP, SP, #32 ; STP X16, X17, [SP] — spill the gate's scratch registers. + asm.emit(Insn::SubSp(MSR_FRAME_BYTES)); + asm.emit(Insn::Stp { + rt: X16, + rt2: X17, + rn: SP, + imm_bytes: MSR_FRAME_OFF_X16.cast_signed(), + }); + + // STR Xn, [SP, #16] — capture the guest value while all guest registers are + // still pristine, so Xn needs no special-casing even when it is one of the + // scratch registers just spilled (X16/X17) or XZR. + asm.emit(Insn::StrUimm { + rt, + rn: SP, + imm_bytes: MSR_FRAME_OFF_VALUE, + }); + + // Put the slot's address in X16, then store the captured guest value through + // it. Under `Baked` the address is anchor + immediate, so the immediate rides + // on the store itself; under `RuntimeSlot` the offset is a loaded value, so it + // has to be folded into the base first — a register-offset store would need + // three live registers (value, base, offset) and this gate has only X16/X17. + match host.guest_tp_addressing() { + GuestTpAddressing::Baked(offset) => { + // MRS X16, ; LDR X17, [SP, #16] ; STR X17, [X16, #offset]. + asm.emit(host.anchor_read(X16)); + asm.emit(Insn::LdrUimm { + rt: X17, + rn: SP, + imm_bytes: MSR_FRAME_OFF_VALUE, + }); + asm.emit(Insn::StrUimm { + rt: X17, + rn: X16, + imm_bytes: offset, + }); + } + GuestTpAddressing::RuntimeSlot => { + // LDR X16, ; MRS X17, ; ADD X16, X17, X16 + // ; LDR X17, [SP, #16] ; STR X17, [X16]. + let slot_vaddr = checked_add_u64( + trampoline_base_addr, + HEADER_GUEST_TP_OFFSET_MACOS as u64, + "MSR guest-TP offset slot", + )?; + if !asm.ldr_literal_reachable(X16, slot_vaddr)? { + return Ok(GateBuild::Unreachable); + } + asm.emit(host.anchor_read(X17)); + asm.emit(Insn::AddReg { + rd: X16, + rn: X17, + rm: X16, + }); + asm.emit(Insn::LdrUimm { + rt: X17, + rn: SP, + imm_bytes: MSR_FRAME_OFF_VALUE, + }); + asm.emit(Insn::StrUimm { + rt: X17, + rn: X16, + imm_bytes: 0, + }); + } + } + + // Restore: LDP X16, X17, [SP] ; ADD SP, SP, #32. + asm.emit(Insn::Ldp { + rt: X16, + rt2: X17, + rn: SP, + imm_bytes: MSR_FRAME_OFF_X16.cast_signed(), + }); + asm.emit(Insn::AddSp(MSR_FRAME_BYTES)); + + // B . + if !asm.branch_to(checked_add_u64(site.vaddr, 4, "MSR return")?)? { + return Ok(GateBuild::Unreachable); + } + + trampoline_data.extend_from_slice(&asm.finish()); + Ok(GateBuild::Emitted) +} + +/// Per-site MRS gate (3 instructions baked, 8 with a runtime offset). +/// +/// Virtualizes a guest `MRS Xd, TPIDR_EL0` read. The hardware register holds the +/// host anchor, so the gate reads the anchor and then loads the guest thread +/// pointer from its slot. +/// +/// Under [`GuestTpAddressing::Baked`] the offset is an immediate, so `Xd` — which +/// the guest's own `MRS` was about to overwrite anyway — is the only register +/// touched and no frame is needed: +/// `MRS Xd, TPIDR_EL0 ; LDR Xd, [Xd, #GUEST_TPIDR_OFFSET] ; B `. +/// +/// Under [`GuestTpAddressing::RuntimeSlot`] the offset arrives in a register, so +/// the gate needs a second one and must give it back untouched: an `MRS` site is +/// an ordinary instruction in the middle of a function, not a call boundary, and +/// X16/X17 are only reserved for linker veneers — a compiler is free to keep a +/// live value in either across it. The gate therefore spills its scratch to a +/// 16-byte frame, which makes this variant share the SVC and MSR gates' +/// requirement that `SP` address a valid writable stack at the site (see the +/// module docs, "Gate scratch storage and the stack invariant"). +fn emit_mrs_gate( + trampoline_data: &mut Vec, + gate_offset: usize, + trampoline_base_addr: u64, + site: &PatchSite, + rd: u8, + host: Host, +) -> Result { + let gate_vaddr = checked_add_u64(trampoline_base_addr, gate_offset as u64, "MRS gate")?; + let mut asm = Asm::new(gate_vaddr); + + match host.guest_tp_addressing() { + GuestTpAddressing::Baked(offset) => { + asm.emit(host.anchor_read(rd)); + asm.emit(Insn::LdrUimm { + rt: rd, + rn: rd, + imm_bytes: offset, + }); + } + GuestTpAddressing::RuntimeSlot => { + // The scratch register must not be `rd`: the anchor read lands in `rd` + // and would destroy a loaded offset held there. + let scratch = if rd == X16 { X17 } else { X16 }; + + let slot_vaddr = checked_add_u64( + trampoline_base_addr, + HEADER_GUEST_TP_OFFSET_MACOS as u64, + "MRS guest-TP offset slot", + )?; + + // SUB SP, SP, #16 ; STR Xs, [SP] — spill the borrowed scratch. + asm.emit(Insn::SubSp(MRS_FRAME_BYTES)); + asm.emit(Insn::StrUimm { + rt: scratch, + rn: SP, + imm_bytes: MRS_FRAME_OFF_SCRATCH, + }); + + // LDR Xs, ; MRS Xd, ; LDR Xd, [Xd, Xs]. + if !asm.ldr_literal_reachable(scratch, slot_vaddr)? { + return Ok(GateBuild::Unreachable); + } + asm.emit(host.anchor_read(rd)); + asm.emit(Insn::LdrReg { + rt: rd, + rn: rd, + rm: scratch, + }); + + // LDR Xs, [SP] ; ADD SP, SP, #16 — hand the scratch back. + asm.emit(Insn::LdrUimm { + rt: scratch, + rn: SP, + imm_bytes: MRS_FRAME_OFF_SCRATCH, + }); + asm.emit(Insn::AddSp(MRS_FRAME_BYTES)); + } + } + + if !asm.branch_to(checked_add_u64(site.vaddr, 4, "MRS return")?)? { + return Ok(GateBuild::Unreachable); + } + trampoline_data.extend_from_slice(&asm.finish()); + Ok(GateBuild::Emitted) +} + +// ============================================================ +// Small helpers +// ============================================================ + +/// A position-tracking assembler for one trampoline fragment (a gate or a shared +/// handler). It owns the emitted words and the base virtual address of the first +/// word, so the current vaddr — [`Asm::here`] — is always known without manual +/// instruction counting. +/// +/// Branches and loads to an absolute target ([`Asm::branch_to`], +/// [`Asm::ldr_literal`], [`Asm::ldr_literal_reachable`], [`Asm::adrp`]) resolve +/// immediately against [`Asm::here`]. Everything a *per-site gate* emits +/// ([`Asm::branch_to`], [`Asm::ldr_literal_reachable`], [`Asm::adrp`]) reports an +/// out-of-range target by emitting nothing and returning `false`, so the caller +/// can trap that one site and keep rewriting; [`Asm::ldr_literal`], used by the +/// fixed prologue, instead errors, since a prologue that cannot be placed is +/// fatal. A gate reaching for the fatal variant would turn one unreachable site +/// into a failed rewrite of the whole image. +struct Asm { + base_vaddr: u64, + code: Vec, +} + +impl Asm { + fn new(base_vaddr: u64) -> Self { + Asm { + base_vaddr, + code: Vec::new(), + } + } + + /// Virtual address of the next instruction to be emitted. + fn here(&self) -> Result { + checked_add_u64( + self.base_vaddr, + self.code.len() as u64, + "trampoline gate next-instruction", + ) + } + + /// Append a raw little-endian word. + fn push_word(&mut self, word: u32) { + self.code.extend_from_slice(&word.to_le_bytes()); + } + + /// Append a fixed-operand instruction. Every operand at the call sites is a + /// compile-time-known register or frame offset, so encoding cannot fail; a + /// `None` would be a rewriter bug rather than an unencodable program. + fn emit(&mut self, insn: Insn) { + let word = insn.encode().expect("statically valid instruction"); + self.push_word(word); + } + + /// `B ` — unconditional branch to an absolute address. Returns + /// whether the target was within the branch's ±128MB reach: an out-of-range + /// target emits nothing and yields `false`, so the caller can trap the + /// originating site instead of failing the whole rewrite. + fn branch_to(&mut self, target_vaddr: u64) -> Result { + let offset = self.delta_to(target_vaddr)?; + let Some(word) = Insn::B(offset).encode() else { + return Ok(false); + }; + self.push_word(word); + Ok(true) + } + + /// `LDR Xt, ` — PC-relative literal load that reports reach instead + /// of failing. An out-of-range target emits nothing and yields `false`, so a + /// per-site gate can trap its own site and leave the rest of the rewrite + /// intact. [`Asm::ldr_literal`] is the fatal variant, for the fixed prologue. + fn ldr_literal_reachable(&mut self, rt: u8, target_vaddr: u64) -> Result { + let offset = self.delta_to(target_vaddr)?; + let Some(word) = (Insn::LdrLiteral { rt, off: offset }).encode() else { + return Ok(false); + }; + self.push_word(word); + Ok(true) + } + + /// `LDR Xt, =target` — PC-relative literal load of an absolute address. + fn ldr_literal(&mut self, rt: u8, target_vaddr: u64) -> Result<()> { + let offset = self.delta_to(target_vaddr)?; + let word = Insn::LdrLiteral { rt, off: offset } + .encode() + .ok_or_else(|| { + Error::AddressOverflow(format!("LDR literal offset {offset:#x} out of ±1MB range")) + })?; + self.push_word(word); + Ok(()) + } + + /// `ADRP Xd, ` — page-relative address of an absolute target. + /// Returns whether the target's page was within ADRP's ±4GB reach (see + /// [`Asm::branch_to`] for the out-of-range contract). + fn adrp(&mut self, rd: u8, target_vaddr: u64) -> Result { + let here = self.here()?; + let page_off = (target_vaddr & !0xFFF) + .cast_signed() + .saturating_sub((here & !0xFFF).cast_signed()) + >> 12; + let Some(word) = Insn::Adrp { rd, page_off }.encode() else { + return Ok(false); + }; + self.push_word(word); + Ok(true) + } + + /// Signed byte distance from [`Asm::here`] to `target_vaddr`. The subtraction + /// saturates so a pathological address can't overflow it; a distance the + /// branch can't encode is rejected by the encoder's range check at the call + /// site, with the saturated value reported for diagnostics. + fn delta_to(&self, target_vaddr: u64) -> Result { + Ok(target_vaddr + .cast_signed() + .saturating_sub(self.here()?.cast_signed())) + } + + /// Return the emitted bytes. + fn finish(self) -> Vec { + self.code + } +} + +#[cfg(test)] +mod tests { + use super::*; + use alloc::vec; + + // Gate and shared-handler sizes. The emitters append each gate dynamically + // (`gate_offset = trampoline_data.len()`), so these sizes drive no emission; + // the tests use them to slice individual gates out of the trampoline blob and + // to assert its total length. `GATES_START_OFFSET` is the fixed shared-prologue + // size that the per-site gates follow. + const SVC_GATE_INSNS: usize = 6; + const SVC_GATE_SIZE: usize = SVC_GATE_INSNS * 4; + const SHARED_SVC_HANDLER_INSNS: usize = 2; + const SHARED_SVC_HANDLER_SIZE: usize = SHARED_SVC_HANDLER_INSNS * 4; + const MSR_GATE_INSNS: usize = 9; + const MSR_GATE_SIZE: usize = MSR_GATE_INSNS * 4; + const MRS_GATE_INSNS: usize = 3; + const MRS_GATE_SIZE: usize = MRS_GATE_INSNS * 4; + // `GuestTpAddressing::RuntimeSlot` sizes: the MSR gate gains the offset load + // and the fold into the base; the MRS gate gains those plus its spill pair. + const MSR_GATE_INSNS_RUNTIME_SLOT: usize = MSR_GATE_INSNS + 2; + const MSR_GATE_SIZE_RUNTIME_SLOT: usize = MSR_GATE_INSNS_RUNTIME_SLOT * 4; + const MRS_GATE_INSNS_RUNTIME_SLOT: usize = MRS_GATE_INSNS + 5; + const MRS_GATE_SIZE_RUNTIME_SLOT: usize = MRS_GATE_INSNS_RUNTIME_SLOT * 4; + const GATES_START_OFFSET: usize = SHARED_SVC_HANDLER_OFFSET + SHARED_SVC_HANDLER_SIZE; + + /// Top-6 opcode bits, isolating the `B`/`BL` major opcode for read-back checks. + const OPCODE_TOP6_MASK: u32 = 0xFC00_0000; + + fn word_at(data: &[u8], byte_off: usize) -> u32 { + u32::from_le_bytes(data[byte_off..byte_off + 4].try_into().unwrap()) + } + + /// `MSR TPIDR_EL0, Xrt` guest instruction word (the low 5 bits select Xrt). + /// The rewriter only matches/scans this form; it never emits it, so the + /// encoder lives only here for building test inputs. + fn msr_tpidr_el0(rt: u8) -> u32 { + MSR_TPIDR_EL0_BITS | u32::from(rt) + } + + /// Helper: emit just the shared SVC handler and return its instruction words. + fn shared_svc_handler_words() -> vec::Vec { + let mut buf = vec::Vec::new(); + emit_shared_svc_handler(&mut buf, 0, 0x1000).unwrap(); + buf.as_chunks::<4>() + .0 + .iter() + .map(|w| u32::from_le_bytes(*w)) + .collect() + } + + #[test] + fn svc_handler_jumps_to_callback_without_tls() { + let words = shared_svc_handler_words(); + assert_eq!(words.len(), SHARED_SVC_HANDLER_INSNS); + // The handler conveys nothing TLS-related: it loads the callback pointer + // and tail-jumps. The callback reads host TLS from TPIDR_EL0 itself. + assert_eq!( + words[1], + Insn::Br(X16).encode().unwrap(), + "handler ends in BR X16" + ); + // No MRS TPIDR_EL0 anywhere in the handler. + assert!( + !words + .iter() + .any(|&w| w & MRS_TPIDR_EL0_MASK == MRS_TPIDR_EL0_BITS) + ); + } + + #[test] + fn encoders_match_known_words() { + // `B #0`. + assert_eq!(Insn::B(0).encode().unwrap(), 0x1400_0000); + // `B #4` advances one instruction. + assert_eq!(Insn::B(4).encode().unwrap(), 0x1400_0001); + // `B #-4` is the all-ones imm26. + assert_eq!(Insn::B(-4).encode().unwrap(), 0x17FF_FFFF); + // `BR X16`. + assert_eq!(Insn::Br(16).encode().unwrap(), 0xD61F_0200); + // TPIDR_EL0 accessor. + assert_eq!(Insn::MrsTpidrEl0(9).encode().unwrap(), 0xD53B_D049); + // TPIDRRO_EL0 accessor (`Host::MacOs`'s anchor read). Cross-checked + // against a real toolchain: `cc -c` + `otool -tvV` on an Apple M3 Pro + // assembles `mrs x9, tpidrro_el0` to this exact word. + assert_eq!(Insn::MrsTpidrroEl0(9).encode().unwrap(), 0xD53B_D069); + // `MSR TPIDR_EL0, X9` guest word (scanned, never emitted). + assert_eq!(msr_tpidr_el0(9), 0xD51B_D049); + // Register-offset load and three-register add, cross-checked against a + // real toolchain (`clang -arch arm64` + `otool -t` on an Apple M3 Pro): + // `ldr x0,[x1,x2]` and `add x0,x1,x2`. + // + // These two words are worth pinning exactly. The register-offset class is + // selected by bit 21 together with `bits[11:10]=0b10`; drop either and the + // word silently decodes as an `LDUR` with an unrelated immediate rather + // than failing to encode, so a mask-based check can pass on an encoding + // that addresses the wrong memory. + assert_eq!( + Insn::LdrReg { + rt: 0, + rn: 1, + rm: 2 + } + .encode() + .unwrap(), + 0xF862_6820 + ); + assert_eq!( + Insn::AddReg { + rd: 0, + rn: 1, + rm: 2 + } + .encode() + .unwrap(), + 0x8B02_0020 + ); + // Scaled (×8) 64-bit load/store: `ldr x9,[x9,#16]` / `str x17,[x16,#16]`. + assert_eq!( + Insn::LdrUimm { + rt: 9, + rn: 9, + imm_bytes: 16 + } + .encode() + .unwrap(), + 0xF940_0929 + ); + assert_eq!( + Insn::StrUimm { + rt: 17, + rn: 16, + imm_bytes: 16 + } + .encode() + .unwrap(), + 0xF900_0A11 + ); + // The guest thread-pointer slot lives at the fixed ABI offset + // GuestThreadBlock::guest_tp; pin both the value and the emitted word. + assert_eq!(GUEST_TPIDR_OFFSET, 16); + // Slot access at GUEST_TPIDR_OFFSET: `ldr x9,[x9,#16]` / `str x17,[x16,#16]`. + assert_eq!( + Insn::LdrUimm { + rt: 9, + rn: 9, + imm_bytes: GUEST_TPIDR_OFFSET + } + .encode() + .unwrap(), + 0xF940_0929 + ); + assert_eq!( + Insn::StrUimm { + rt: 17, + rn: 16, + imm_bytes: GUEST_TPIDR_OFFSET + } + .encode() + .unwrap(), + 0xF900_0A11 + ); + // `BRK #0xB10B`, the trap that replaces an out-of-range patch site. + assert_eq!(Insn::Brk(TRAP_BRK_IMM).encode().unwrap(), 0xD436_2160); + } + + #[test] + fn encoder_range_checks() { + assert!(Insn::B(2).encode().is_none()); // not 4-aligned + assert!(Insn::B(1 << 27).encode().is_none()); // out of ±128MB + assert!( + Insn::StrUimm { + rt: 0, + rn: 0, + imm_bytes: 4 + } + .encode() + .is_none() + ); // not 8-scaled + } + + #[test] + fn asm_ldr_literal_computes_pc_relative_offset() { + let mut asm = Asm::new(0x1000); + asm.emit(Insn::Br(30)); // [0] at 0x1000 + asm.ldr_literal(16, 0x1010).unwrap(); // [1] at 0x1004, target 0x1010 => +0xC + let code = asm.finish(); + assert_eq!( + word_at(&code, 4), + Insn::LdrLiteral { rt: 16, off: 0xC }.encode().unwrap() + ); + } + + /// Build a one-section image whose section data == the supplied words and + /// run the hooker. Returns `(patched_section, trampoline)`. Panics if the + /// input has no patch sites (use [`hook_words_opt`] for that case). + fn hook_words(words: &[u32], base: u64, tramp_base: u64) -> (Vec, Vec) { + let (patched, outcome) = hook_words_opt(words, base, tramp_base); + ( + patched, + outcome + .expect("expected a trampoline (input has patch sites)") + .trampoline, + ) + } + + /// Like [`hook_words`] but returns the raw `Option` outcome so callers can + /// assert the "no patch sites" (`None`) sentinel and trapped-site cases. + fn hook_words_opt(words: &[u32], base: u64, tramp_base: u64) -> (Vec, Option) { + hook_words_host(words, base, tramp_base, Host::Linux) + } + + /// Like [`hook_words_opt`] with an explicit [`Host`]. + fn hook_words_host( + words: &[u32], + base: u64, + tramp_base: u64, + host: Host, + ) -> (Vec, Option) { + let mut buf = Vec::new(); + for w in words { + buf.extend_from_slice(&w.to_le_bytes()); + } + let sections = vec![TextSectionInfo { + vaddr: base, + file_offset: 0, + size: buf.len() as u64, + }]; + let outcome = hook_syscalls_aarch64(&mut buf, §ions, tramp_base, 0, host).unwrap(); + (buf, outcome) + } + + #[test] + fn no_patch_sites_emit_no_trampoline() { + // No patch sites: a NOP-only section yields no trampoline at all, so the + // caller emits a size-0 sentinel (matching the x86-64 path). + let (_patched, tramp) = hook_words_opt(&[0xD503_201F], 0x1000, 0x100000); + assert!(tramp.is_none()); + } + + #[test] + fn svc_is_replaced_with_branch_into_gate() { + let base = 0x1000; + let tramp_base = 0x200000; + let (patched, tramp) = hook_words(&[SVC_0], base, tramp_base); + + // The SVC word became a `B`. + let patched_word = word_at(&patched, 0); + assert_eq!( + patched_word & OPCODE_TOP6_MASK, + Opcode::B.bits(), + "expected B opcode" + ); + + // It targets the first per-site gate at GATES_START_OFFSET. + let imm26 = i64::from(patched_word & IMM26_MASK); + let disp = imm26 << 2; // positive here + let target = base + disp.cast_unsigned(); + assert_eq!(target, tramp_base + GATES_START_OFFSET as u64); + + // The gate's first instruction is SUB SP, SP, #SVC_FRAME_BYTES. + assert_eq!( + word_at(&tramp, GATES_START_OFFSET), + Insn::SubSp(SVC_FRAME_BYTES).encode().unwrap() + ); + // Total = prologue + one SVC gate. + assert_eq!(tramp.len(), GATES_START_OFFSET + SVC_GATE_SIZE); + } + + #[test] + fn svc_with_nonzero_immediate_is_also_rewritten() { + // Linux dispatches every `SVC64` exception to the syscall handler + // regardless of the immediate (the syscall number comes from x8), so + // `svc #imm` with imm != 0 must be rewritten too. imm16 occupies bits + // [20:5], so `svc #1` is `SVC_0 | (1 << 5)`. + let svc_imm1 = SVC_0 | (1 << 5); + let (patched, tramp) = hook_words(&[svc_imm1], 0x1000, 0x200000); + // The SVC word became a `B` into the gate. + assert_eq!(word_at(&patched, 0) & OPCODE_TOP6_MASK, Opcode::B.bits()); + assert_eq!(tramp.len(), GATES_START_OFFSET + SVC_GATE_SIZE); + } + + #[test] + fn site_beyond_branch_range_is_trapped() { + // The trampoline sits 256MB above the section, past the `B` instruction's + // ±128MB reach, so the site cannot branch into its gate. It is replaced + // with the sentinel `BRK`, surfaced as a trapped site, and no gate is + // emitted for it, leaving the trampoline at the prologue-only size. + let (patched, outcome) = hook_words_opt(&[SVC_0], 0x1000, 0x1000_0000); + let outcome = outcome.expect("expected a trampoline (input has patch sites)"); + assert_eq!( + word_at(&patched, 0), + Insn::Brk(TRAP_BRK_IMM).encode().unwrap() + ); + assert_eq!(outcome.trapped_sites, vec![0x1000]); + assert_eq!(outcome.trampoline.len(), GATES_START_OFFSET); + } + + #[test] + fn msr_gate_return_branch_out_of_range_is_trapped() { + // Boundary window where the site can reach its gate but the gate cannot + // reach back. The MSR gate's return `B` is its last instruction, at + // `gate + 32`, branching to `site + 4`; its displacement magnitude is + // `b_offset + 28`, larger than the inbound `B`'s `b_offset`. Placing the + // gate at the maximum encodable forward offset (`2^27 - 4`) makes the + // inbound branch encode while the return needs `-(2^27 + 24)`, just past + // the `-2^27` reach. The site must still be trapped gracefully — replaced + // with `BRK` and surfaced through `trapped_sites` — not error out. + let base = 0x1000u64; + let max_fwd = (1u64 << 27) - 4; // largest 4-aligned forward `B` offset + let tramp_base = base + max_fwd - GATES_START_OFFSET as u64; + let (patched, outcome) = hook_words_opt(&[msr_tpidr_el0(5)], base, tramp_base); + let outcome = outcome.expect("expected a trampoline (input has patch sites)"); + assert_eq!( + word_at(&patched, 0), + Insn::Brk(TRAP_BRK_IMM).encode().unwrap() + ); + assert_eq!(outcome.trapped_sites, vec![base]); + // No gate emitted for the trapped site: prologue-only trampoline. + assert_eq!(outcome.trampoline.len(), GATES_START_OFFSET); + } + + #[test] + fn hvc_and_smc_are_not_treated_as_svc() { + // `HVC #0` (…02) and `SMC #0` (…03) share the SVC opcode base but differ + // in bits [1:0]; they must not be rewritten as syscalls. + let hvc_0 = 0xD400_0002u32; + let smc_0 = 0xD400_0003u32; + let (_p, tramp) = hook_words_opt(&[hvc_0, smc_0], 0x1000, 0x200000); + assert!(tramp.is_none(), "HVC/SMC must not be matched as SVC"); + } + + #[test] + fn msr_and_mrs_both_get_gates() { + let base = 0x1000; + let tramp_base = 0x300000; + // MSR TPIDR_EL0, X5 then MRS X9, TPIDR_EL0. + let words = [msr_tpidr_el0(5), Insn::MrsTpidrEl0(9).encode().unwrap()]; + let (patched, tramp) = hook_words(&words, base, tramp_base); + // Both the write and the read are rewritten to a branch into their gate. + assert_eq!(word_at(&patched, 0) & OPCODE_TOP6_MASK, Opcode::B.bits()); + assert_eq!(word_at(&patched, 4) & OPCODE_TOP6_MASK, Opcode::B.bits()); + // Trampoline = prologue + one MSR gate + one MRS gate. + assert_eq!( + tramp.len(), + GATES_START_OFFSET + MSR_GATE_SIZE + MRS_GATE_SIZE + ); + } + + #[test] + fn macos_host_anchors_gates_on_tpidrro_el0() { + // Same input as `msr_and_mrs_both_get_gates`, but hooked for + // `Host::MacOs`: both gates' anchor-read instruction must be + // `MRS X16, TPIDRRO_EL0`, not `TPIDR_EL0` -- real hardware (see this + // module's `Host` doc comment) confirmed `TPIDR_EL0` does not survive + // a host context switch on macOS. + let base = 0x1000; + let tramp_base = 0x300000; + let words = [msr_tpidr_el0(5), Insn::MrsTpidrEl0(9).encode().unwrap()]; + let mut buf = Vec::new(); + for w in words { + buf.extend_from_slice(&w.to_le_bytes()); + } + let sections = vec![TextSectionInfo { + vaddr: base, + file_offset: 0, + size: buf.len() as u64, + }]; + let outcome = hook_syscalls_aarch64(&mut buf, §ions, tramp_base, 0, Host::MacOs) + .unwrap() + .expect("expected a trampoline (input has patch sites)"); + let tramp = outcome.trampoline; + + // MSR gate: SubSp, Stp, StrUimm, LdrLiteral, , ... -- + // anchor is the 5th instruction (index 4). It lands in X17 rather than + // X16 because X16 already holds the offset loaded from the header slot. + let msr_anchor_off = GATES_START_OFFSET + 4 * 4; + assert_eq!( + word_at(&tramp, msr_anchor_off), + Insn::MrsTpidrroEl0(X17).encode().unwrap() + ); + assert_ne!( + word_at(&tramp, msr_anchor_off), + Insn::MrsTpidrEl0(X17).encode().unwrap() + ); + + // MRS gate: SubSp, StrUimm, LdrLiteral, , ... -- anchor is the + // 4th instruction (index 3), read directly into the guest's own + // destination register (X9 here, from `Insn::MrsTpidrEl0(9)` above). + let mrs_anchor_off = GATES_START_OFFSET + MSR_GATE_SIZE_RUNTIME_SLOT + 3 * 4; + assert_eq!( + word_at(&tramp, mrs_anchor_off), + Insn::MrsTpidrroEl0(9).encode().unwrap() + ); + assert_ne!( + word_at(&tramp, mrs_anchor_off), + Insn::MrsTpidrEl0(9).encode().unwrap() + ); + } + + #[test] + fn mrs_with_xzr_dest_is_left_native() { + // `MRS XZR, TPIDR_EL0` reads-and-discards; it must not be rewritten. + let mrs_xzr = Insn::MrsTpidrEl0(31).encode().unwrap(); + let (_p, tramp) = hook_words_opt(&[mrs_xzr], 0x1000, 0x200000); + assert!(tramp.is_none(), "MRS XZR, TPIDR_EL0 must be left native"); + } + + #[test] + fn msr_gate_stores_guest_value_to_slot_for_any_register() { + const BL_TOP6: u32 = 0x9400_0000; + for n in [5u8, 16, 17, 30, 31] { + let (_p, tramp) = hook_words(&[msr_tpidr_el0(n)], 0x1000, 0x500000); + let gate = &tramp[GATES_START_OFFSET..GATES_START_OFFSET + MSR_GATE_SIZE]; + // Self-contained: never BL out. + assert!( + (0..MSR_GATE_INSNS).all(|i| word_at(gate, i * 4) & OPCODE_TOP6_MASK != BL_TOP6) + ); + // Capture the guest value while pristine: STR Xn, [SP, #16]. + let capture = Insn::StrUimm { + rt: n, + rn: SP, + imm_bytes: MSR_FRAME_OFF_VALUE, + } + .encode() + .unwrap(); + let cap_i = (0..MSR_GATE_INSNS) + .find(|&i| word_at(gate, i * 4) == capture) + .expect("MSR gate must capture the guest value (incl. XZR=0) while pristine"); + // Read the host anchor: MRS X16, TPIDR_EL0. + let anchor = Insn::MrsTpidrEl0(X16).encode().unwrap(); + let anc_i = (0..MSR_GATE_INSNS) + .find(|&i| word_at(gate, i * 4) == anchor) + .expect("MSR gate must read the host anchor"); + // Store to the slot: STR X17, [X16, #GUEST_TPIDR_OFFSET]. + let store = Insn::StrUimm { + rt: X17, + rn: X16, + imm_bytes: GUEST_TPIDR_OFFSET, + } + .encode() + .unwrap(); + let st_i = (0..MSR_GATE_INSNS) + .find(|&i| word_at(gate, i * 4) == store) + .expect("MSR gate must store the guest value into its slot"); + assert!( + cap_i < anc_i && anc_i < st_i, + "capture -> anchor -> store order" + ); + // Ends in B back to the guest (not the last word being the store). + assert_eq!( + word_at(gate, (MSR_GATE_INSNS - 1) * 4) & OPCODE_TOP6_MASK, + Opcode::B.bits() + ); + } + } + + #[test] + fn mrs_gate_loads_guest_tp_from_slot() { + for d in [5u8, 16, 17, 30] { + let (_p, tramp) = + hook_words(&[Insn::MrsTpidrEl0(d).encode().unwrap()], 0x1000, 0x400000); + let gate = &tramp[GATES_START_OFFSET..GATES_START_OFFSET + MRS_GATE_SIZE]; + assert_eq!(word_at(gate, 0), Insn::MrsTpidrEl0(d).encode().unwrap()); + assert_eq!( + word_at(gate, 4), + Insn::LdrUimm { + rt: d, + rn: d, + imm_bytes: GUEST_TPIDR_OFFSET + } + .encode() + .unwrap() + ); + assert_eq!(word_at(gate, 8) & OPCODE_TOP6_MASK, Opcode::B.bits()); + } + } + + #[test] + fn macos_host_gates_anchor_on_tpidrro_at_tsd_slot() { + // Under `Host::MacOs` the same guest instructions produce gates that + // anchor on `TPIDRRO_EL0` (Darwin's pthread TSD base) and address the + // guest thread-pointer slot at the reserved TSD offset — never the + // Linux anchor or block offset. + let base = 0x1000; + let tramp_base = 0x300000; + let words = [msr_tpidr_el0(5), Insn::MrsTpidrEl0(9).encode().unwrap()]; + let (patched, outcome) = hook_words_host(&words, base, tramp_base, Host::MacOs); + let tramp = outcome.expect("both sites must be gated").trampoline; + assert_eq!(word_at(&patched, 0) & OPCODE_TOP6_MASK, Opcode::B.bits()); + assert_eq!(word_at(&patched, 4) & OPCODE_TOP6_MASK, Opcode::B.bits()); + + // The offset the gates use is no longer an immediate: it is read from the + // header slot, so neither gate may contain the baked macOS offset and the + // slot itself must carry it as the packaging-time seed. + let seeded = u64::from_le_bytes( + tramp[HEADER_GUEST_TP_OFFSET_MACOS..HEADER_GUEST_TP_OFFSET_MACOS + 8] + .try_into() + .unwrap(), + ); + assert_eq!( + seeded, + u64::from(GUEST_TPIDR_OFFSET_MACOS), + "the header slot must be seeded with the packaging-time default" + ); + + // MSR gate: LdrLiteral(X16), MRS X17 anchor, ADD X16, X17, X16, + // LDR X17, [SP,#16], STR X17, [X16]. The Linux anchor never appears. + let msr_gate = &tramp[GATES_START_OFFSET..GATES_START_OFFSET + MSR_GATE_SIZE_RUNTIME_SLOT]; + let anchor = Insn::MrsTpidrroEl0(X17).encode().unwrap(); + assert!( + (0..MSR_GATE_INSNS_RUNTIME_SLOT).any(|i| word_at(msr_gate, i * 4) == anchor), + "MSR gate must read TPIDRRO_EL0 as the anchor" + ); + let fold = Insn::AddReg { + rd: X16, + rn: X17, + rm: X16, + } + .encode() + .unwrap(); + assert!( + (0..MSR_GATE_INSNS_RUNTIME_SLOT).any(|i| word_at(msr_gate, i * 4) == fold), + "MSR gate must fold the loaded offset into the anchor" + ); + let store = Insn::StrUimm { + rt: X17, + rn: X16, + imm_bytes: 0, + } + .encode() + .unwrap(); + assert!( + (0..MSR_GATE_INSNS_RUNTIME_SLOT).any(|i| word_at(msr_gate, i * 4) == store), + "MSR gate must store the guest value through the folded slot address" + ); + let baked_store = Insn::StrUimm { + rt: X17, + rn: X16, + imm_bytes: GUEST_TPIDR_OFFSET_MACOS, + } + .encode() + .unwrap(); + assert!( + (0..MSR_GATE_INSNS_RUNTIME_SLOT).all(|i| word_at(msr_gate, i * 4) != baked_store), + "the baked offset must not survive in a runtime-slot gate" + ); + for linux_anchor in [ + Insn::MrsTpidrEl0(X16).encode().unwrap(), + Insn::MrsTpidrEl0(X17).encode().unwrap(), + ] { + assert!( + (0..MSR_GATE_INSNS_RUNTIME_SLOT).all(|i| word_at(msr_gate, i * 4) != linux_anchor), + "the Linux anchor read must not appear in a macOS gate" + ); + } + + // MRS gate: SUB SP ; STR X16,[SP] ; LDR X16, ; MRS X9, TPIDRRO_EL0 + // ; LDR X9,[X9,X16] ; LDR X16,[SP] ; ADD SP ; B. + let mrs_gate = + &tramp[GATES_START_OFFSET + MSR_GATE_SIZE_RUNTIME_SLOT..][..MRS_GATE_SIZE_RUNTIME_SLOT]; + assert_eq!( + word_at(mrs_gate, 0), + Insn::SubSp(MRS_FRAME_BYTES).encode().unwrap() + ); + assert_eq!( + word_at(mrs_gate, 4), + Insn::StrUimm { + rt: X16, + rn: SP, + imm_bytes: MRS_FRAME_OFF_SCRATCH + } + .encode() + .unwrap() + ); + assert_eq!( + word_at(mrs_gate, 12), + Insn::MrsTpidrroEl0(9).encode().unwrap() + ); + assert_eq!( + word_at(mrs_gate, 16), + Insn::LdrReg { + rt: 9, + rn: 9, + rm: X16 + } + .encode() + .unwrap() + ); + assert_eq!( + word_at(mrs_gate, 20), + Insn::LdrUimm { + rt: X16, + rn: SP, + imm_bytes: MRS_FRAME_OFF_SCRATCH + } + .encode() + .unwrap(), + "the borrowed scratch register must be handed back" + ); + assert_eq!( + word_at(mrs_gate, 24), + Insn::AddSp(MRS_FRAME_BYTES).encode().unwrap() + ); + assert_eq!(word_at(mrs_gate, 28) & OPCODE_TOP6_MASK, Opcode::B.bits()); + } + + #[test] + fn macos_mrs_gate_scratch_never_collides_with_the_destination() { + // The anchor read lands in the guest's own destination register, so the + // borrowed scratch holding the loaded offset must never be that register + // -- otherwise the anchor would overwrite the offset before it is used. + for d in [5u8, 16, 17, 30] { + let (_p, outcome) = hook_words_host( + &[Insn::MrsTpidrEl0(d).encode().unwrap()], + 0x1000, + 0x300000, + Host::MacOs, + ); + let tramp = outcome.expect("site must be gated").trampoline; + let gate = &tramp[GATES_START_OFFSET..][..MRS_GATE_SIZE_RUNTIME_SLOT]; + let scratch = if d == X16 { X17 } else { X16 }; + assert_ne!(scratch, d, "scratch must differ from the destination"); + assert_eq!( + word_at(gate, 4), + Insn::StrUimm { + rt: scratch, + rn: SP, + imm_bytes: MRS_FRAME_OFF_SCRATCH + } + .encode() + .unwrap() + ); + assert_eq!( + word_at(gate, 16), + Insn::LdrReg { + rt: d, + rn: d, + rm: scratch + } + .encode() + .unwrap() + ); + } + } + + #[test] + fn macos_tsd_slot_offset_is_stable_abi_and_encodable() { + // Slot 256 is the first dynamic pthread key on macOS; the byte offset + // is baked into statically rewritten binaries, so it must never move, + // and it must stay encodable as a scaled LDR/STR immediate. + assert_eq!(MACOS_GUEST_TPIDR_TSD_SLOT, 256); + assert_eq!(GUEST_TPIDR_OFFSET_MACOS, 2048); + assert!( + Insn::LdrUimm { + rt: 0, + rn: 0, + imm_bytes: GUEST_TPIDR_OFFSET_MACOS + } + .encode() + .is_some() + ); + } + + #[test] + fn svc_gate_saves_only_x16_and_records_return() { + let base = 0x1000; + let tramp_base = 0x600000; + let (_p, tramp) = hook_words(&[SVC_0], base, tramp_base); + let gate = &tramp[GATES_START_OFFSET..GATES_START_OFFSET + SVC_GATE_SIZE]; + // SUB SP,#16 ; STR X16,[SP] ; ADRP X16,.. ; ADD X16,X16,#.. ; STR X16,[SP,#8] ; B + assert_eq!( + word_at(gate, 0), + Insn::SubSp(SVC_FRAME_BYTES).encode().unwrap() + ); + assert_eq!( + word_at(gate, 4), + Insn::StrUimm { + rt: X16, + rn: SP, + imm_bytes: SVC_FRAME_OFF_X16 + } + .encode() + .unwrap() + ); + assert_eq!( + word_at(gate, 16), + Insn::StrUimm { + rt: X16, + rn: SP, + imm_bytes: SVC_FRAME_OFF_RETADDR + } + .encode() + .unwrap() + ); + // Self-contained tail branch to the shared handler (B, never BL). + assert_eq!(word_at(gate, 20) & OPCODE_TOP6_MASK, Opcode::B.bits()); + assert_eq!(tramp.len(), GATES_START_OFFSET + SVC_GATE_SIZE); + } +} diff --git a/litebox_syscall_rewriter/src/lib.rs b/litebox_syscall_rewriter/src/lib.rs index 9dcc10c28f..59fd171f0c 100644 --- a/litebox_syscall_rewriter/src/lib.rs +++ b/litebox_syscall_rewriter/src/lib.rs @@ -1,7 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. -//! Rewrite ELF files to hook syscalls +//! Rewrite binaries for LiteBox execution. //! //! This crate sets up a trampoline point for every `syscall` instruction in its input binary, //! allowing for conveniently taking control of a binary without ptrace/systrap/seccomp/... @@ -12,18 +12,34 @@ //! However, as an explicit goal, it is intended to provide low-overhead hooking of syscalls, //! without needing to undergo a user-kernel transition. //! -//! This crate currently only supports x86-64 (i.e., amd64) ELFs. +//! This crate currently supports x86-64 ELFs for syscall hooking and x86-64 PEs for syscall +//! hooking plus rewriting Windows TEB accesses from GS segment overrides to FS segment overrides. +//! +//! It also supports AArch64 ELFs, for Linux guests on Linux or macOS hosts (see [`Host`]), and +//! rewrites `SVC #imm` syscalls plus both directions of guest thread-pointer access +//! (`MSR TPIDR_EL0` writes and `MRS TPIDR_EL0` reads): the host owns a per-thread anchor register +//! and the guest thread pointer is fully virtualized to a host-managed memory slot reached through +//! it. Which register anchors the host varies by host OS -- [`hook_syscalls_in_elf`] defaults to +//! [`Host::Linux`]; [`hook_syscalls_in_elf_for_host`] selects explicitly. See the `arm64` module +//! for details, including a caveat on `Host::MacOs` guests that use `x18`. #![cfg_attr(not(feature = "std"), no_std)] extern crate alloc; -use alloc::collections::BTreeSet; +mod arm64; + +pub use arm64::Host; + +use alloc::collections::{BTreeMap, BTreeSet}; use alloc::format; use alloc::string::{String, ToString}; use alloc::vec; use alloc::vec::Vec; +use litebox_common_windows::NtSysno; +use object::pe::{IMAGE_SCN_CNT_CODE, IMAGE_SCN_MEM_EXECUTE}; use object::read::elf::{ElfFile, ProgramHeader as _}; +use object::read::pe::{ImageNtHeaders as _, ImageOptionalHeader as _, PeFile64}; use object::read::{Object as _, ObjectSection as _}; use thiserror::Error; use zerocopy::{FromBytes, Immutable, IntoBytes}; @@ -73,6 +89,51 @@ const BUN_FOOTER_MARKER: &[u8] = b"\n---- Bun! ----\n"; /// This is checked by the loader to verify that the trampoline is valid. pub const TRAMPOLINE_MAGIC: &[u8; 8] = b"LITEBOX0"; +/// The pthread thread-specific-data slot index at which the macOS runtime +/// keeps each thread's guest thread pointer, as addressed by every gate a +/// [`Host::MacOs`] rewrite emits (`[TPIDRRO_EL0 + slot * 8]`). The runtime +/// reserves exactly this slot with `pthread_key_create` at platform init; see +/// the `arm64` module's `GUEST_TPIDR_OFFSET_MACOS` docs for the verified +/// Darwin TSD layout this rests on. +pub const MACOS_GUEST_TPIDR_TSD_SLOT: u16 = arm64::MACOS_GUEST_TPIDR_TSD_SLOT; + +/// Byte offset, within an emitted AArch64 trampoline, of the word holding the +/// guest thread-pointer byte offset. +/// +/// A loader writes the offset its runtime actually reserved into this word +/// before making the trampoline executable; gates emitted for a host that reads +/// the offset at run time load it from here. Exported so a loader names the same +/// slot the emitter wrote rather than repeating the number. +pub const TRAMPOLINE_GUEST_TP_SLOT_OFFSET: usize = arm64::HEADER_GUEST_TP_OFFSET_MACOS; + +/// Rewrite a supported binary for LiteBox. +/// +/// ELF64 inputs are passed through [`hook_syscalls_in_elf`] (the +/// [`Host::Linux`] anchor). PE64 inputs have executable-section GS segment +/// overrides rewritten to FS and `syscall` instructions redirected through a +/// LiteBox trampoline footer. +/// +/// Use [`rewrite_binary_for_host`] to target an AArch64 ELF input at a +/// non-Linux host. +pub fn rewrite_binary(input_binary: &[u8], trampoline: Option) -> Result> { + rewrite_binary_for_host(input_binary, trampoline, Host::Linux) +} + +/// As [`rewrite_binary`], but selects the AArch64 host anchor explicitly +/// instead of defaulting to [`Host::Linux`] -- see [`hook_syscalls_in_elf_for_host`]. +/// Ignored for PE64 input, which has no per-host anchor concept. +pub fn rewrite_binary_for_host( + input_binary: &[u8], + trampoline: Option, + host: Host, +) -> Result> { + if is_pe_binary(input_binary) { + rewrite_pe_for_litebox(input_binary, trampoline) + } else { + hook_syscalls_in_elf_for_host(input_binary, trampoline, host) + } +} + /// Trampoline header for 64-bit: 8 (magic) + 8 (file_offset) + 8 (vaddr) + 8 (size) = 32 bytes #[repr(C, packed)] #[derive(FromBytes, IntoBytes, Immutable)] @@ -83,7 +144,7 @@ struct TrampolineHeader64 { trampoline_size: u64, } -/// Metadata about an executable section, extracted from the read-only ELF parse. +/// Metadata about an executable section, extracted from a read-only object parse. struct TextSectionInfo { /// Virtual address of the section vaddr: u64, @@ -93,6 +154,17 @@ struct TextSectionInfo { size: u64, } +struct SyscallPatchResult { + found_syscall: bool, + skipped_addrs: Vec, +} + +/// Limit on how far backward from a `syscall` we look for the `mov eax, imm32` +/// that loads its sysno. A real NT stub always sets `eax` within a handful of +/// instructions of the `syscall`; the bound keeps us from rewriting some +/// unrelated `mov eax` that happens to share an immediate value with a sysno. +const NT_SYSNO_REWRITE_LOOKBACK: usize = 16; + /// Update the `input_binary` with a call to `trampoline` instead of any `syscall` instructions. /// /// The `trampoline` must be an absolute address if specified; if unspecified, it will be set to @@ -108,20 +180,45 @@ struct TextSectionInfo { /// - trampoline virtual address (8 bytes) /// - trampoline size (8 bytes) /// -/// This layout allows loaders to read just the last 32 bytes to get the metadata. Even when -/// there is no syscall instruction in the binary, the rewriter still appends the header and the initial -/// syscall-entry placeholder so the loader/audit path can tell the binary was processed. +/// This layout allows loaders to read just the last 32 bytes to get the metadata. +/// +/// When there is nothing to patch, both architectures append only a 32-byte +/// header carrying a `trampoline_size = 0` *sentinel* (no trampoline body), so a +/// loader can distinguish "processed, nothing to patch" from "never processed"; +/// no instructions are rewritten in that case. +/// +/// AArch64 differs in one way: it also rewrites guest thread-pointer accesses +/// (`MSR TPIDR_EL0` writes and `MRS TPIDR_EL0` reads), so a binary containing one +/// is patched (and gets a non-empty trampoline) even when it has no syscall +/// (`SVC`) instructions at all. (See the `arm64` module docs.) /// /// Returns the rewritten binary. Binaries that cannot or do not need to be /// patched (relocatable objects, non-ELF files, already-hooked binaries, -/// binaries without executable sections or syscall instructions) are returned -/// unchanged — these are not errors. +/// binaries without executable sections) are returned unchanged — these are +/// not errors. See the per-architecture behavior above. /// /// Returns `Err` for genuinely broken inputs (corrupt ELF, unsupported /// executables like Bun, arithmetic overflow) and for binaries that contain -/// syscall instructions that could not be patched (replaced with `icebp; hlt` -/// so they trap instead of escaping to the host kernel). +/// patch sites that could not be redirected. An unpatchable site is replaced +/// with a trapping instruction so it faults instead of escaping to the host +/// kernel: `icebp; hlt` on x86-64, and `BRK` on AArch64 (where a patch site is +/// an `SVC`, `MSR TPIDR_EL0`, or `MRS TPIDR_EL0` instruction). +/// +/// AArch64 gates against [`Host::Linux`]'s anchor (`TPIDR_EL0`); use +/// [`hook_syscalls_in_elf_for_host`] to target a different host. pub fn hook_syscalls_in_elf(input_binary: &[u8], trampoline: Option) -> Result> { + hook_syscalls_in_elf_for_host(input_binary, trampoline, Host::Linux) +} + +/// As [`hook_syscalls_in_elf`], but selects the AArch64 host anchor explicitly +/// instead of defaulting to [`Host::Linux`]. A binary rewritten for one host +/// will not run correctly under another. Ignored for x86-64 input, which has +/// no per-host anchor concept (and no macOS host at all, by design). +pub fn hook_syscalls_in_elf_for_host( + input_binary: &[u8], + trampoline: Option, + host: Host, +) -> Result> { if input_binary.ends_with(BUN_FOOTER_MARKER) { return Err(Error::UnsupportedExecutable( "Bun-packaged executable".into(), @@ -131,9 +228,16 @@ pub fn hook_syscalls_in_elf(input_binary: &[u8], trampoline: Option) -> Res // Relocatable object files (.o) must not be patched: they are linker // input, not executable code. Rewriting instructions or appending // trampoline data would corrupt the object file for the linker. - // Check the ELF e_type field (bytes 16..18) before doing any work. + // Check the ELF e_type field (bytes 16..18) before doing any work. The + // encoding of multi-byte fields is selected by e_ident[EI_DATA] (byte 5), + // so decode e_type in that endianness rather than assuming little-endian. if input_binary.len() >= 18 { - let e_type = u16::from_le_bytes([input_binary[16], input_binary[17]]); + let e_type_bytes = [input_binary[16], input_binary[17]]; + let e_type = if input_binary[5] == object::elf::ELFDATA2MSB { + u16::from_be_bytes(e_type_bytes) + } else { + u16::from_le_bytes(e_type_bytes) + }; if e_type == object::elf::ET_REL { return Ok(input_binary.to_vec()); } @@ -153,11 +257,15 @@ pub fn hook_syscalls_in_elf(input_binary: &[u8], trampoline: Option) -> Res fixup_phdr_alignment(buf); // Parse the ELF and extract all metadata we need, then drop the borrow so we can mutate buf. - let (arch, text_sections, control_transfer_targets, trampoline_base_addr) = { + let (arch, text_sections, trampoline_base_addr) = { let file = object::File::parse(&*buf).map_err(|e| Error::ParseError(e.to_string()))?; let arch = match file { - object::File::Elf64(_) => Arch::X86_64, + object::File::Elf64(_) => match file.architecture() { + object::Architecture::X86_64 => Arch::X86_64, + object::Architecture::Aarch64 => Arch::Aarch64, + _ => return Ok(input_binary.to_vec()), + }, _ => return Ok(input_binary.to_vec()), }; @@ -172,40 +280,483 @@ pub fn hook_syscalls_in_elf(input_binary: &[u8], trampoline: Option) -> Res return Ok(input_binary.to_vec()); } - let control_transfer_targets = get_control_transfer_targets(arch, &*buf, &text_sections)?; + let trampoline_base_addr = find_addr_for_trampoline_code(&file, arch.trampoline_align())?; + + (arch, text_sections, trampoline_base_addr) + }; + + // AArch64 uses a fully separate rewriting strategy (single-instruction + // branch replacement, no instruction borrowing). Dispatch to it before any + // x86-only work (iced-x86 decoding would misinterpret AArch64 bytes). + // See the `arm64` module docs. + if arch == Arch::Aarch64 { + return hook_aarch64_elf( + input_binary, + buf, + &text_sections, + trampoline_base_addr, + trampoline.unwrap_or(0), + host, + ); + } + + if !matches!(host, Host::Linux) { + return Err(Error::UnsupportedExecutable( + "x86-64 guests only run under a Linux host".into(), + )); + } + + let control_transfer_targets = get_control_transfer_targets(arch, &*buf, &text_sections)?; + let mut trampoline_data = Vec::from(trampoline.unwrap_or(0).to_le_bytes()); + let patch_result = patch_syscalls_in_sections( + arch, + buf, + &text_sections, + &control_transfer_targets, + trampoline_base_addr, + trampoline_base_addr, + &mut trampoline_data, + )?; + + if !patch_result.found_syscall { + let mut out = input_binary.to_vec(); + let header = TrampolineHeader64 { + magic: *TRAMPOLINE_MAGIC, + file_offset: 0, + vaddr: 0, + trampoline_size: 0, + }; + out.extend_from_slice(header.as_bytes()); + return Ok(out); + } + + // Build output: [patched ELF][padding to page boundary][trampoline code][header] + let mut out = buf.to_vec(); + append_trampoline_footer( + &mut out, + &mut trampoline_data, + trampoline_base_addr, + false, + Arch::X86_64.trampoline_align(), + ); - let trampoline_base_addr = find_addr_for_trampoline_code(&file)?; + if !patch_result.skipped_addrs.is_empty() { + return Err(Error::UnpatchableSyscalls(format!( + "{} unpatchable syscall instruction(s) at {skipped_addrs:?}", + patch_result.skipped_addrs.len(), + skipped_addrs = patch_result.skipped_addrs, + ))); + } + Ok(out) +} + +/// Rewrite an x86-64 PE for LiteBox's current Windows shim. +/// +/// The PE file layout is preserved, but executable-section GS segment overrides +/// are rewritten to FS and `syscall` instructions are redirected through a +/// LiteBox trampoline appended as a file overlay. The Windows shim loader maps +/// that overlay by reading the footer this function appends. +pub fn rewrite_pe_for_litebox(input_binary: &[u8], trampoline: Option) -> Result> { + if is_already_hooked(input_binary, Arch::X86_64) { + return Ok(input_binary.to_vec()); + } + + let mut backing = vec![0u64; input_binary.len().div_ceil(8)]; + let buf: &mut [u8] = zerocopy::IntoBytes::as_mut_bytes(backing.as_mut_slice()); + buf[..input_binary.len()].copy_from_slice(input_binary); + let buf = &mut buf[..input_binary.len()]; + + let (text_sections, sysno_map, trampoline_base_rva, trampoline_base_addr) = { + let pe = PeFile64::parse(&*buf).map_err(|e| Error::ParseError(e.to_string()))?; + let optional_header = pe.nt_headers().optional_header(); + let size_of_image = u64::from(optional_header.size_of_image()); + let trampoline_base_rva = + checked_add_u64(size_of_image, 0xfff, "PE trampoline base")? & !0xfff; + let trampoline_base_addr = checked_add_u64( + optional_header.image_base(), + trampoline_base_rva, + "PE trampoline virtual address", + )?; + let file = object::File::parse(&*buf).map_err(|e| Error::ParseError(e.to_string()))?; + match file { + object::File::Pe64(_) if file.architecture() == object::Architecture::X86_64 => {} + _ => return Ok(input_binary.to_vec()), + } + + let text_sections = match pe_text_sections(&file) { + Ok(sections) => sections, + Err(InternalError::NoTextSectionFound) => return Ok(input_binary.to_vec()), + Err(InternalError::Public(e)) => return Err(e), + Err(e) => unreachable!("unexpected internal error: {e:?}"), + }; + let sysno_map = pe_ntdll_sysno_map(&file, buf, &text_sections)?; ( - arch, text_sections, - control_transfer_targets, + sysno_map, + trampoline_base_rva, trampoline_base_addr, ) }; - // Build the trampoline code (without header - header goes at the end) - // The code starts with the syscall entry point placeholder (8 bytes for x86-64) - let mut trampoline_data = vec![]; - let trampoline = trampoline.unwrap_or(0); - trampoline_data.extend_from_slice(&trampoline.to_le_bytes()); - // Patch syscalls in-place in buf + for section in &text_sections { + let section_data = section_slice_mut(buf, section)?; + rewrite_gs_to_fs_in_section(Arch::X86_64, section.vaddr, section_data)?; + } + let control_transfer_targets = get_control_transfer_targets(Arch::X86_64, buf, &text_sections)?; + rewrite_nt_sysnos_in_sections( + Arch::X86_64, + buf, + &text_sections, + &sysno_map, + &control_transfer_targets, + )?; + + let mut trampoline_data = Vec::from(trampoline.unwrap_or(0).to_le_bytes()); + // Windows ntdll packs some syscall stubs too tightly for the generic + // five-byte jump patcher; keep that PE-specific shape out of the generic path. + let patched_dense_windows_stubs = patch_dense_windows_syscall_stubs_in_sections( + Arch::X86_64, + buf, + &text_sections, + &control_transfer_targets, + trampoline_base_addr, + trampoline_base_addr, + &mut trampoline_data, + )?; + let patch_result = patch_syscalls_in_sections( + Arch::X86_64, + buf, + &text_sections, + &control_transfer_targets, + trampoline_base_addr, + trampoline_base_addr, + &mut trampoline_data, + )?; + + if !patched_dense_windows_stubs && !patch_result.found_syscall { + return Ok(buf.to_vec()); + } + + let mut out = buf.to_vec(); + append_trampoline_footer( + &mut out, + &mut trampoline_data, + trampoline_base_rva, + true, + Arch::X86_64.trampoline_align(), + ); + + if !patch_result.skipped_addrs.is_empty() { + return Err(Error::UnpatchableSyscalls(format!( + "{} unpatchable syscall instruction(s) at {skipped_addrs:?}", + patch_result.skipped_addrs.len(), + skipped_addrs = patch_result.skipped_addrs, + ))); + } + + Ok(out) +} + +fn is_pe_binary(input_binary: &[u8]) -> bool { + if input_binary.len() < 0x40 || &input_binary[..2] != b"MZ" { + return false; + } + let pe_offset = u32::from_le_bytes(input_binary[0x3c..0x40].try_into().unwrap()) as usize; + input_binary + .get(pe_offset..pe_offset.saturating_add(4)) + .is_some_and(|magic| magic == b"PE\0\0") +} + +fn pe_text_sections( + file: &object::File<'_>, +) -> core::result::Result, InternalError> { + let text_sections: Vec<_> = file + .sections() + .filter_map(|section| { + let object::SectionFlags::Coff { characteristics } = section.flags() else { + return None; + }; + if characteristics & IMAGE_SCN_CNT_CODE == 0 { + return None; + } + if characteristics & IMAGE_SCN_MEM_EXECUTE == 0 { + return None; + } + let (file_offset, size) = section.file_range()?; + Some(TextSectionInfo { + vaddr: section.address(), + file_offset, + size, + }) + }) + .collect(); + if text_sections.is_empty() { + return Err(InternalError::NoTextSectionFound); + } + Ok(text_sections) +} + +/// For ntdll-like PEs, walks `Nt*` exports of `file`, reads the build-specific +/// sysno each stub loads into `eax`, and maps it to the stable LiteBox +/// [`NtSysno`] for that name. `Nt*` and `Zw*` always share sysno numbering +/// inside ntdll, so a map keyed on the build-specific number lets a later pass +/// rewrite both flavors (and any internal ntdll helpers that issue the same +/// syscall inline) uniformly. +fn pe_ntdll_sysno_map( + file: &object::File<'_>, + buf: &[u8], + text_sections: &[TextSectionInfo], +) -> Result> { + let mut map = BTreeMap::new(); + let mut exports_ntdll_loader_entrypoint = false; + + for export in file + .exports() + .map_err(|e| Error::ParseError(e.to_string()))? + { + let Ok(name) = core::str::from_utf8(export.name()) else { + continue; + }; + exports_ntdll_loader_entrypoint |= name == "LdrInitializeThunk"; + + let Some(sysno) = NtSysno::from_export_name(name) else { + continue; + }; + + let addr = export.address(); + let Some(section) = text_sections.iter().find(|s| { + s.vaddr + .checked_add(s.size) + .is_some_and(|end| addr >= s.vaddr && addr < end) + }) else { + continue; + }; + + let section_data = section_slice(buf, section)?; + let stub_offset = usize::try_from(addr - section.vaddr) + .map_err(|_| Error::ParseError("export offset out of range".into()))?; + if let Some(build_sysno) = read_nt_stub_sysno(section_data, stub_offset) { + map.insert(build_sysno, sysno); + } + } + + if !exports_ntdll_loader_entrypoint { + return Ok(BTreeMap::new()); + } + + Ok(map) +} + +/// Reads the `mov eax, imm32` immediate that precedes a `syscall` instruction +/// within the first 32 bytes of an NT syscall stub starting at `stub_offset`. +/// Returns `None` if the bytes do not match the expected stub shape. +fn read_nt_stub_sysno(section_data: &[u8], stub_offset: usize) -> Option { + let stub = section_data.get(stub_offset..)?; + let stub_len = stub.len().min(32); + let syscall_offset = stub[..stub_len] + .windows(2) + .position(|bytes| bytes == [0x0f, 0x05])?; + let mov_eax_offset = stub[..syscall_offset] + .windows(5) + .position(|bytes| bytes[0] == 0xb8)?; + let imm = u32::from_le_bytes( + stub[mov_eax_offset + 1..mov_eax_offset + 5] + .try_into() + .ok()?, + ); + Some(imm) +} + +fn rewrite_nt_sysnos_in_sections( + arch: Arch, + buf: &mut [u8], + text_sections: &[TextSectionInfo], + sysno_map: &BTreeMap, + control_transfer_targets: &BTreeSet, +) -> Result { + if sysno_map.is_empty() { + return Ok(0); + } + let mut rewritten = 0; + for section in text_sections { + let section_data = section_slice_mut(buf, section)?; + rewritten += rewrite_nt_sysnos_in_section( + arch, + section.vaddr, + section_data, + sysno_map, + control_transfer_targets, + )?; + } + Ok(rewritten) +} + +/// For every `syscall` in `section_data`, looks backward up to +/// [`NT_SYSNO_REWRITE_LOOKBACK`] instructions for the closest `mov r32, imm32` +/// that targets `eax`. If the immediate is a known build-specific sysno from +/// `sysno_map`, rewrites it in place to the stable LiteBox sysno. +/// +/// The backward walk stops at any unconditional control transfer (`jmp`, +/// `ret`, indirect branch, exception), at any instruction that is itself a +/// control-transfer target, and at any earlier write to `eax`. Conditional +/// branches are walked through, because the canonical NT stub has a `test +/// [...], 1; jne +3; syscall` sequence where execution reaches `syscall` by +/// falling through `jne`. Syscalls that are themselves jump targets are +/// skipped entirely — there's no way to know which `mov eax` the jumping code +/// arrived with. +fn rewrite_nt_sysnos_in_section( + arch: Arch, + section_base_addr: u64, + section_data: &mut [u8], + sysno_map: &BTreeMap, + control_transfer_targets: &BTreeSet, +) -> Result { + let instructions = decode_section_instructions(arch, section_data, section_base_addr)?; + let mut info_factory = iced_x86::InstructionInfoFactory::new(); + let mut rewritten = 0; + + for (i, inst) in instructions.iter().enumerate() { + if inst.code() != iced_x86::Code::Syscall { + continue; + } + if control_transfer_targets.contains(&inst.ip()) { + continue; + } + let lookback_start = i.saturating_sub(NT_SYSNO_REWRITE_LOOKBACK); + for j in (lookback_start..i).rev() { + let prev = &instructions[j]; + // A `jne`/`je`/etc. between `mov eax, sysno` and `syscall` is normal + // (the canonical NT stub has `test ...; jne +3; syscall`), so we + // keep walking through conditional branches and calls — they fall + // through to the next instruction in the common case. We only stop + // at unconditional transfers that prove the linear chain from + // `prev → next → ... → syscall` was never the execution path. + if matches!( + prev.flow_control(), + iced_x86::FlowControl::UnconditionalBranch + | iced_x86::FlowControl::IndirectBranch + | iced_x86::FlowControl::Call + | iced_x86::FlowControl::IndirectCall + | iced_x86::FlowControl::Return + | iced_x86::FlowControl::Exception + ) { + break; + } + if prev.code() == iced_x86::Code::Mov_r32_imm32 + && prev.op0_register() == iced_x86::Register::EAX + { + if let Some(&sysno) = sysno_map.get(&prev.immediate32()) { + let inst_offset = usize::try_from(prev.ip() - section_base_addr) + .map_err(|_| Error::ParseError("instruction offset out of range".into()))?; + // `Mov_r32_imm32` always encodes the 32-bit immediate as the + // last four bytes of the instruction, regardless of any REX + // prefix in front of the opcode. + let imm_end = inst_offset + .checked_add(prev.len()) + .ok_or_else(|| Error::AddressOverflow("mov eax end".into()))?; + let imm_start = imm_end + .checked_sub(4) + .ok_or_else(|| Error::ParseError("mov eax length < 4".into()))?; + section_data[imm_start..imm_end].copy_from_slice(&sysno.as_raw().to_le_bytes()); + rewritten += 1; + } + break; + } + if instruction_writes_eax(&mut info_factory, prev) { + break; + } + if control_transfer_targets.contains(&prev.ip()) { + break; + } + } + } + + Ok(rewritten) +} + +/// Returns `true` if `inst` writes (or partially writes) the `eax` register +/// family — `eax`, `rax`, `ax`, `al`, `ah` (including implicit writes such as +/// `cpuid`/`mul`/`div`/`cdq`). Used by the sysno rewriter to detect an EAX +/// clobber between a stale `mov eax, K` and a downstream `syscall`, so a +/// sequence like `mov eax, K; xor eax, eax; syscall` does not mis-rewrite `K` +/// as a sysno load. +fn instruction_writes_eax( + info_factory: &mut iced_x86::InstructionInfoFactory, + inst: &iced_x86::Instruction, +) -> bool { + use iced_x86::{OpAccess, Register}; + for used in info_factory.info(inst).used_registers() { + if !matches!( + used.access(), + OpAccess::Write | OpAccess::ReadWrite | OpAccess::CondWrite | OpAccess::ReadCondWrite + ) { + continue; + } + if matches!( + used.register(), + Register::EAX | Register::RAX | Register::AX | Register::AL | Register::AH + ) { + return true; + } + } + false +} + +fn rewrite_gs_to_fs_in_section( + arch: Arch, + section_base_addr: u64, + section_data: &mut [u8], +) -> Result { + let instructions = decode_section_instructions(arch, section_data, section_base_addr)?; + let mut rewritten = 0; + + for instruction in &instructions { + if instruction.memory_segment() != iced_x86::Register::GS { + continue; + } + + let offset = usize::try_from(instruction.ip() - section_base_addr).unwrap(); + let instruction_bytes = &mut section_data[offset..offset + instruction.len()]; + let Some(segment_prefix) = instruction_bytes.iter_mut().find(|byte| **byte == 0x65) else { + return Err(Error::DisassemblyFailure(format!( + "GS memory operand at {:#x} has no GS segment prefix", + instruction.ip() + ))); + }; + *segment_prefix = 0x64; + rewritten += 1; + } + + Ok(rewritten) +} + +fn patch_syscalls_in_sections( + arch: Arch, + buf: &mut [u8], + text_sections: &[TextSectionInfo], + control_transfer_targets: &BTreeSet, + trampoline_base_addr: u64, + syscall_entry_addr: u64, + trampoline_data: &mut Vec, +) -> Result { + let mut found_syscall = false; let mut skipped_addrs = Vec::new(); - let mut syscall_insns_found = false; - for s in &text_sections { - let section_data = section_slice_mut(buf, s)?; + + for section in text_sections { + let section_data = section_slice_mut(buf, section)?; match hook_syscalls_in_section( arch, - &control_transfer_targets, - s.vaddr, + control_transfer_targets, + section.vaddr, section_data, trampoline_base_addr, - trampoline_base_addr, // entry point is at offset 0 of trampoline - &mut trampoline_data, + syscall_entry_addr, + trampoline_data, ) { Ok(addrs) => { + found_syscall = true; skipped_addrs.extend(addrs); - syscall_insns_found = true; } Err(InternalError::NoSyscallInstructionsFound) => {} Err(InternalError::Public(e)) => return Err(e), @@ -213,53 +764,134 @@ pub fn hook_syscalls_in_elf(input_binary: &[u8], trampoline: Option) -> Res } } - if !syscall_insns_found { - // No syscall instructions found. Append a header-only marker so the - // loader can distinguish "checked by rewriter, nothing to patch" from - // "never processed." The trampoline_size=0 sentinel tells the loader - // to skip trampoline mapping entirely. - // Use the original input (not `buf`) to avoid emitting the phdr - // alignment fixup that is only needed for the `object` crate parser. - let mut out = input_binary.to_vec(); - let header = TrampolineHeader64 { - magic: *TRAMPOLINE_MAGIC, - file_offset: 0, - vaddr: 0, - trampoline_size: 0, - }; - out.extend_from_slice(header.as_bytes()); - return Ok(out); + Ok(SyscallPatchResult { + found_syscall, + skipped_addrs, + }) +} + +fn patch_dense_windows_syscall_stubs_in_sections( + arch: Arch, + buf: &mut [u8], + text_sections: &[TextSectionInfo], + control_transfer_targets: &BTreeSet, + trampoline_base_addr: u64, + syscall_entry_addr: u64, + trampoline_data: &mut Vec, +) -> Result { + let mut patched_any = false; + + for section in text_sections { + let section_data = section_slice_mut(buf, section)?; + let instructions = decode_section_instructions(arch, section_data, section.vaddr)?; + for (i, inst) in instructions.iter().enumerate() { + if inst.code() != iced_x86::Code::Syscall { + continue; + } + + patched_any |= patch_dense_windows_syscall_stub( + control_transfer_targets, + section.vaddr, + section_data, + trampoline_base_addr, + syscall_entry_addr, + trampoline_data, + &instructions, + i, + )?; + } } - // Build output: [patched ELF][padding to page boundary][trampoline code][header] - let mut out = buf.to_vec(); - let remain = out.len() % 0x1000; - out.extend_from_slice(&vec![0; if remain == 0 { 0 } else { 0x1000 - remain }]); + Ok(patched_any) +} + +fn append_trampoline_footer( + out: &mut Vec, + trampoline_data: &mut Vec, + header_vaddr: u64, + align_trampoline_size: bool, + align: u64, +) { + // The file offset has to carry the same alignment as the virtual address: + // the loader maps the trampoline straight out of the file at that offset, + // and a page-granular file mapping cannot start part-way into a page. + let align = usize::try_from(align).expect("trampoline alignment fits a pointer"); + let remain = out.len() % align; + out.extend_from_slice(&vec![0; if remain == 0 { 0 } else { align - remain }]); - // Calculate file offset where trampoline code starts let trampoline_file_offset = out.len() as u64; + if align_trampoline_size { + let trampoline_size = trampoline_data.len().next_multiple_of(align); + trampoline_data.extend_from_slice(&vec![0; trampoline_size - trampoline_data.len()]); + } let trampoline_size = trampoline_data.len(); + out.extend_from_slice(trampoline_data); - // Append trampoline code - out.extend_from_slice(&trampoline_data); - - // Build the header (goes at the end of the file) - // The entry point placeholder is at offset 0 of the trampoline code, not in the header. let header = TrampolineHeader64 { magic: *TRAMPOLINE_MAGIC, file_offset: trampoline_file_offset, - vaddr: trampoline_base_addr, + vaddr: header_vaddr, trampoline_size: trampoline_size as u64, }; out.extend_from_slice(header.as_bytes()); - if !skipped_addrs.is_empty() { +} + +/// Rewrite an AArch64 ELF, appending the trampoline and trailing header. +/// +/// `input_binary` is the original, unmodified ELF; `buf` is the mutable copy +/// (patched in place by the arm64 module). `callback` is the absolute address +/// stored in the trampoline's callback slot (0 when the loader fills it in +/// later). +/// +/// Like the x86-64 path, a binary with no patch sites is emitted as the +/// original bytes followed by a size-0 trampoline sentinel header (the arm64 +/// module signals this by returning `None`). Otherwise the output layout is +/// `[patched ELF][padding to page boundary][trampoline code][header]`. +fn hook_aarch64_elf( + input_binary: &[u8], + buf: &mut [u8], + text_sections: &[TextSectionInfo], + trampoline_base_addr: u64, + callback: u64, + host: Host, +) -> Result> { + let Some(outcome) = + arm64::hook_syscalls_aarch64(buf, text_sections, trampoline_base_addr, callback, host)? + else { + // No patch sites: emit the original binary with a size-0 sentinel + // header so the loader knows there is no trampoline to map. + let mut out = input_binary.to_vec(); + let header = TrampolineHeader64 { + magic: *TRAMPOLINE_MAGIC, + file_offset: 0, + vaddr: 0, + trampoline_size: 0, + }; + out.extend_from_slice(header.as_bytes()); + return Ok(out); + }; + + // Build output: [patched ELF][padding to page boundary][trampoline][header]. + let mut trampoline_data = outcome.trampoline; + let mut out = buf.to_vec(); + append_trampoline_footer( + &mut out, + &mut trampoline_data, + trampoline_base_addr, + false, + Arch::Aarch64.trampoline_align(), + ); + + if !outcome.trapped_sites.is_empty() { return Err(Error::UnpatchableSyscalls(format!( - "{} unpatchable syscall instruction(s) at {skipped_addrs:?}", - skipped_addrs.len(), + "{} unpatchable instruction(s) (SVC / MSR / MRS TPIDR_EL0) at {trapped:?}", + outcome.trapped_sites.len(), + trapped = outcome.trapped_sites, ))); } Ok(out) } + /// (private) Get metadata for executable sections fn text_sections( file: &object::File<'_>, @@ -296,7 +928,7 @@ fn text_sections( /// Check if the binary is already hooked by looking for TRAMPOLINE_MAGIC at the end of the file. fn is_already_hooked(input_binary: &[u8], arch: Arch) -> bool { let header_size = match arch { - Arch::X86_64 => size_of::(), + Arch::X86_64 | Arch::Aarch64 => size_of::(), }; if input_binary.len() < header_size { @@ -315,8 +947,9 @@ fn is_already_hooked(input_binary: &[u8], arch: Arch) -> bool { (header.file_offset, header.vaddr, header.trampoline_size); if trampoline_size == 0 { - // Size=0 sentinel: the rewriter processed this binary but found no - // syscall instructions. It is already hooked (nothing to do). + // Size=0 sentinel: the rewriter processed this binary but found nothing + // to patch — no syscall instructions, and on AArch64 no `MSR`/`MRS + // TPIDR_EL0` accesses either. It is already hooked (nothing to do). return true; } if file_offset % 0x1000 != 0 { @@ -325,7 +958,7 @@ fn is_already_hooked(input_binary: &[u8], arch: Arch) -> bool { if vaddr % 0x1000 != 0 { return false; } - if file_offset + trampoline_size != header_start as u64 { + if file_offset.checked_add(trampoline_size) != Some(header_start as u64) { return false; } @@ -335,6 +968,28 @@ fn is_already_hooked(input_binary: &[u8], arch: Arch) -> bool { #[derive(PartialEq, Eq, Clone, Copy, Debug, Hash)] enum Arch { X86_64, + Aarch64, +} + +impl Arch { + /// Alignment for the appended trampoline's virtual address, file offset and + /// size. + /// + /// The loader maps the trampoline as its own page-granular mapping and + /// rejects a header whose `vaddr` is not aligned to the *host's* page size, + /// so this has to satisfy every host the image might be loaded on, not the + /// one that rewrote it. x86-64 pages are always 4 KiB. AArch64's are not: + /// Apple Silicon uses 16 KiB, and Linux can be built for 16 KiB or 64 KiB, + /// so a 4 KiB-aligned trampoline is unloadable on most of them. 64 KiB + /// covers all three, and is the maximum page size AArch64 ELF images are + /// conventionally linked for anyway (see `docs/macos.md`), so it costs + /// address space that the layout already assumed. + const fn trampoline_align(self) -> u64 { + match self { + Arch::X86_64 => 0x1000, + Arch::Aarch64 => 0x1_0000, + } + } } /// (private) Hook all syscalls in `section`, possibly extending `trampoline_data` to do so. @@ -362,6 +1017,7 @@ fn hook_syscalls_in_section( continue; } } + Arch::Aarch64 => unreachable!("AArch64 uses the arm64 module, not iced-x86"), } found_any = true; @@ -658,6 +1314,131 @@ fn replace_with_trap( } } +#[allow(clippy::too_many_arguments)] +fn patch_dense_windows_syscall_stub( + control_transfer_targets: &BTreeSet, + section_base_addr: u64, + section_data: &mut [u8], + trampoline_base_addr: u64, + syscall_entry_addr: u64, + trampoline_data: &mut Vec, + instructions: &[iced_x86::Instruction], + inst_index: usize, +) -> Result { + if inst_index < 2 { + return Ok(false); + } + + let test_inst = &instructions[inst_index - 2]; + let jne_inst = &instructions[inst_index - 1]; + let syscall_inst = &instructions[inst_index]; + + if !is_dense_windows_syscall_stub_sequence(test_inst, jne_inst, section_base_addr, section_data) + { + return Ok(false); + } + + let stub_addr = test_inst.ip(); + let fallback_addr = checked_add_u64( + jne_inst.ip(), + DENSE_WINDOWS_SYSCALL_STUB_TAIL_FALLBACK_OFFSET as u64, + "dense Windows syscall fallback address", + )?; + let stub_end_addr = checked_add_u64( + jne_inst.ip(), + DENSE_WINDOWS_SYSCALL_STUB_TAIL.len() as u64, + "dense Windows syscall stub end address", + )?; + if control_transfer_targets + .iter() + .any(|target| (stub_addr..stub_end_addr).contains(target) && *target != fallback_addr) + { + return Ok(false); + } + + let target_addr = checked_add_u64( + trampoline_base_addr, + trampoline_data.len() as u64, + "dense Windows syscall trampoline target", + )?; + + let return_addr = syscall_inst.next_ip(); + let jmp_back_base = checked_add_u64( + trampoline_base_addr, + trampoline_data.len() as u64 + 7, + "dense Windows syscall trampoline return base", + )?; + // lea rcx, [rip + disp32] + trampoline_data.extend_from_slice(&[0x48, 0x8D, 0x0D]); + trampoline_data.extend_from_slice(&rel32_bytes( + return_addr, + jmp_back_base, + "dense Windows syscall trampoline return", + )?); + + // jmp qword ptr [rip + disp32] + trampoline_data.extend_from_slice(&[0xFF, 0x25]); + let entry_base = checked_add_u64( + trampoline_base_addr, + trampoline_data.len() as u64 + 4, + "dense Windows syscall trampoline entry base", + )?; + trampoline_data.extend_from_slice(&rel32_bytes( + syscall_entry_addr, + entry_base, + "dense Windows syscall trampoline entry", + )?); + + let stub_offset = usize::try_from(stub_addr - section_base_addr).unwrap(); + section_data[stub_offset] = 0xe9; + let patch_base = checked_add_u64(stub_addr, 5, "dense Windows syscall patch jump base")?; + section_data[stub_offset + 1..stub_offset + 5].copy_from_slice(&rel32_bytes( + target_addr, + patch_base, + "dense Windows syscall patch jump", + )?); + + let syscall_end_offset = usize::try_from(syscall_inst.next_ip() - section_base_addr).unwrap(); + for byte in &mut section_data[stub_offset + 5..syscall_end_offset] { + *byte = 0x90; + } + + let fallback_offset = usize::try_from(fallback_addr - section_base_addr).unwrap(); + section_data[fallback_offset] = 0xeb; + section_data[fallback_offset + 1] = + i8::try_from(i128::from(stub_addr) - i128::from(fallback_addr) - 2) + .map_err(|_| { + Error::AddressOverflow("dense Windows syscall fallback jump out of range".into()) + })? + .to_ne_bytes()[0]; + + Ok(true) +} + +fn is_dense_windows_syscall_stub_sequence( + test_inst: &iced_x86::Instruction, + jne_inst: &iced_x86::Instruction, + section_base_addr: u64, + section_data: &[u8], +) -> bool { + if !matches!( + test_inst.code(), + iced_x86::Code::Test_rm8_imm8 | iced_x86::Code::Test_rm8_imm8_F6r1 + ) || test_inst.immediate8() != 1 + { + return false; + } + + let Ok(tail_offset) = usize::try_from(jne_inst.ip() - section_base_addr) else { + return false; + }; + let Some(tail_end) = tail_offset.checked_add(DENSE_WINDOWS_SYSCALL_STUB_TAIL.len()) else { + return false; + }; + + section_data.get(tail_offset..tail_end) == Some(DENSE_WINDOWS_SYSCALL_STUB_TAIL) +} + fn checked_add_u64(base: u64, addend: u64, context: &'static str) -> Result { base.checked_add(addend) .ok_or_else(|| Error::AddressOverflow(format!("{context} address overflow"))) @@ -734,7 +1515,7 @@ pub fn trap_all_syscalls_in_code(code: &mut [u8], code_vaddr: u64) -> Result) -> Result { +fn find_addr_for_trampoline_code(file: &object::File<'_>, align: u64) -> Result { // Find the highest virtual address among all PT_LOAD segments let max_virtual_addr = match file { object::File::Elf64(elf) => max_load_segment_end(elf), @@ -742,8 +1523,7 @@ fn find_addr_for_trampoline_code(file: &object::File<'_>) -> Result { } .ok_or_else(|| Error::ParseError("no PT_LOAD segments found".into()))?; - // Round up to the nearest page (assume 0x1000 page size) - checked_add_u64(max_virtual_addr, 0xFFF, "trampoline base").map(|addr| addr & !0xFFF) + checked_add_u64(max_virtual_addr, align - 1, "trampoline base").map(|addr| addr & !(align - 1)) } /// Returns the highest `p_vaddr + p_memsz` among all `PT_LOAD` segments. @@ -784,6 +1564,9 @@ fn get_control_transfer_targets( const MAX_X86_INSTRUCTION_LEN: usize = 15; const CHUNK_OVERLAP_LEN: usize = MAX_X86_INSTRUCTION_LEN - 1; const TARGET_DECODE_CHUNK_LEN: usize = 8 * 1024 * 1024; +// jne +3; syscall; ret; int 0x2e; ret +const DENSE_WINDOWS_SYSCALL_STUB_TAIL: &[u8] = &[0x75, 0x03, 0x0f, 0x05, 0xc3, 0xcd, 0x2e, 0xc3]; +const DENSE_WINDOWS_SYSCALL_STUB_TAIL_FALLBACK_OFFSET: usize = 5; fn bytes_until_next_4g_boundary(ptr: *const u8) -> usize { let low = (ptr as u64) & 0xFFFF_FFFF; @@ -802,6 +1585,7 @@ fn decode_section_instructions( ) -> Result> { let bitness = match arch { Arch::X86_64 => 64, + Arch::Aarch64 => unreachable!("AArch64 uses the arm64 module, not iced-x86"), }; let mut instructions = Vec::new(); @@ -1016,7 +1800,8 @@ fn hook_syscall_and_after( // any RIP-relative memory operands for the new location. let syscall_inst_end = syscall_inst.next_ip(); let postsyscall_bytes = if syscall_inst_end < replace_end { - let postsyscall_target = target_addr + preamble_len; + let postsyscall_target = + checked_add_u64(target_addr, preamble_len, "post-syscall trampoline target")?; match reencode_instructions( &instructions[(inst_index + 1)..replace_end_idx], postsyscall_target, @@ -1086,6 +1871,168 @@ fn hook_syscall_and_after( mod tests { use super::*; + #[test] + fn aarch64_out_of_range_site_is_rejected_as_unpatchable() { + // A trampoline mapped 256MB above the text is outside the site's ±128MB + // branch reach, so the `SVC` is trapped and the rewrite is rejected, + // mirroring the x86-64 unpatchable-syscall contract. + let mut buf = 0xD400_0001u32.to_le_bytes().to_vec(); // SVC #0 + let input = buf.clone(); + let sections = vec![TextSectionInfo { + vaddr: 0x1000, + file_offset: 0, + size: buf.len() as u64, + }]; + let err = + hook_aarch64_elf(&input, &mut buf, §ions, 0x1000_0000, 0, Host::Linux).unwrap_err(); + assert!( + matches!(err, Error::UnpatchableSyscalls(_)), + "expected UnpatchableSyscalls, got {err:?}" + ); + } + + const NT_STUB_BUILD_SYSNO: u32 = 0x1234; + + fn nt_stub_bytes() -> [u8; 24] { + [ + 0x4c, 0x8b, 0xd1, // mov r10, rcx + 0xb8, 0x34, 0x12, 0x00, 0x00, // mov eax, 0x1234 + 0xf6, 0x04, 0x25, 0x08, 0x03, 0xfe, 0x7f, 0x01, // test byte ptr [...], 1 + 0x75, 0x03, // jne +3 + 0x0f, 0x05, // syscall + 0xc3, // ret + 0xcd, 0x2e, // int 2e + 0xc3, // ret + ] + } + + #[test] + fn read_nt_stub_sysno_extracts_build_specific_imm32() { + let stub = nt_stub_bytes(); + assert_eq!(read_nt_stub_sysno(&stub, 0), Some(NT_STUB_BUILD_SYSNO)); + } + + #[test] + fn read_nt_stub_sysno_rejects_stub_without_syscall() { + let stub = [0xb8, 0x34, 0x12, 0x00, 0x00, 0xc3]; + assert_eq!(read_nt_stub_sysno(&stub, 0), None); + } + + #[test] + fn rewrite_replaces_mov_eax_before_syscall() { + let mut stub = nt_stub_bytes(); + let mut map = BTreeMap::new(); + map.insert(NT_STUB_BUILD_SYSNO, NtSysno::NtTerminateProcess); + let targets = BTreeSet::new(); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut stub, &map, &targets).unwrap(); + assert_eq!(rewritten, 1); + assert_eq!( + &stub[4..8], + &NtSysno::NtTerminateProcess.as_raw().to_le_bytes(), + ); + } + + #[test] + fn rewrite_covers_zw_alias_with_same_build_sysno() { + // Two stubs back-to-back sharing the same build-specific sysno, the way + // ntdll's Nt* / Zw* pair often look when emitted as separate stubs. + let mut section = Vec::new(); + section.extend_from_slice(&nt_stub_bytes()); + section.extend_from_slice(&nt_stub_bytes()); + + let mut map = BTreeMap::new(); + map.insert(NT_STUB_BUILD_SYSNO, NtSysno::NtTerminateProcess); + let targets = BTreeSet::new(); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut section, &map, &targets).unwrap(); + assert_eq!(rewritten, 2); + let expected = NtSysno::NtTerminateProcess.as_raw().to_le_bytes(); + assert_eq!(§ion[4..8], &expected); + assert_eq!( + §ion[nt_stub_bytes().len() + 4..nt_stub_bytes().len() + 8], + &expected + ); + } + + #[test] + fn rewrite_leaves_mov_eax_with_unknown_imm_alone() { + let mut stub = nt_stub_bytes(); + let map: BTreeMap = BTreeMap::new(); + let targets = BTreeSet::new(); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut stub, &map, &targets).unwrap(); + assert_eq!(rewritten, 0); + assert_eq!(&stub[4..8], &NT_STUB_BUILD_SYSNO.to_le_bytes()); + } + + #[test] + fn rewrite_skips_when_eax_is_clobbered_before_syscall() { + // `mov eax, K; xor eax, eax; syscall`. The mov's K matches a known + // build sysno, but the xor zeroes eax before the syscall — so K is not + // the sysno that feeds the syscall and must not be rewritten. + let mut section: Vec = vec![ + 0xb8, 0x34, 0x12, 0x00, 0x00, // mov eax, 0x1234 + 0x31, 0xc0, // xor eax, eax + 0x0f, 0x05, // syscall + ]; + + let mut map = BTreeMap::new(); + map.insert(NT_STUB_BUILD_SYSNO, NtSysno::NtTerminateProcess); + let targets = BTreeSet::new(); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut section, &map, &targets).unwrap(); + assert_eq!(rewritten, 0); + assert_eq!(§ion[1..5], &NT_STUB_BUILD_SYSNO.to_le_bytes()); + } + + #[test] + fn rewrite_does_not_cross_basic_block_boundary() { + // `mov eax, K; ret; syscall`. The mov's K matches + // a known sysno but lives in a previous function (separated by `ret`); + // the syscall is reached by control flow that never touched that mov. + let mut section: Vec = vec![ + 0xb8, 0x34, 0x12, 0x00, 0x00, // mov eax, 0x1234 (in prior function) + 0xc3, // ret (block boundary) + 0x0f, 0x05, // syscall (next function) + ]; + + let mut map = BTreeMap::new(); + map.insert(NT_STUB_BUILD_SYSNO, NtSysno::NtTerminateProcess); + let targets = BTreeSet::new(); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut section, &map, &targets).unwrap(); + assert_eq!(rewritten, 0); + assert_eq!(§ion[1..5], &NT_STUB_BUILD_SYSNO.to_le_bytes()); + } + + #[test] + fn rewrite_skips_syscall_that_is_jump_target() { + // `mov eax, K; syscall` where the syscall is jumped to from elsewhere. + // We can't trust that the preceding mov is what set eax for callers that + // arrived via the jump. + let syscall_offset: u64 = 5; + let mut section: Vec = vec![ + 0xb8, 0x34, 0x12, 0x00, 0x00, // mov eax, 0x1234 (offset 0..5) + 0x0f, 0x05, // syscall (offset 5..7) + ]; + + let mut map = BTreeMap::new(); + map.insert(NT_STUB_BUILD_SYSNO, NtSysno::NtTerminateProcess); + let mut targets = BTreeSet::new(); + targets.insert(syscall_offset); + + let rewritten = + rewrite_nt_sysnos_in_section(Arch::X86_64, 0, &mut section, &map, &targets).unwrap(); + assert_eq!(rewritten, 0); + assert_eq!(§ion[1..5], &NT_STUB_BUILD_SYSNO.to_le_bytes()); + } + #[cfg(target_pointer_width = "64")] #[test] #[ignore = "allocates over 4GiB to reproduce the iced-x86 host-pointer bug without mmap"] diff --git a/litebox_syscall_rewriter/src/main.rs b/litebox_syscall_rewriter/src/main.rs index 7ef8eef14c..17599b152c 100644 --- a/litebox_syscall_rewriter/src/main.rs +++ b/litebox_syscall_rewriter/src/main.rs @@ -4,16 +4,43 @@ //! Runner for [`litebox_syscall_rewriter`] use clap::Parser; +use clap::ValueEnum; use std::io::Read as _; use std::io::Write as _; #[cfg(unix)] use std::os::unix::fs::{MetadataExt as _, PermissionsExt as _}; use std::path::PathBuf; -/// Rewrite ELF files to hook syscalls +/// The AArch64 host anchor to rewrite an ELF's gates against -- mirrors +/// [`litebox_syscall_rewriter::Host`], which is not itself `ValueEnum` (it +/// lives in a `no_std` crate). Ignored for x86-64/PE input. Getting this +/// wrong is not cosmetic: a `Linux`-anchored rewrite run on a macOS guest +/// reads a live thread-pointer value from the wrong register (`TPIDR_EL0`, +/// which the host does not preserve) and crashes the guest on its first +/// reschedule, misleadingly far from the actual cause -- see +/// [`litebox_syscall_rewriter::Host::MacOs`]'s own doc comment. +#[derive(Clone, Copy, Debug, ValueEnum)] +enum HostArg { + /// The host preserves `TPIDR_EL0` across a context switch. + Linux, + /// The host is macOS/Darwin (Apple Silicon): `TPIDR_EL0` does not survive + /// a context switch, so gates anchor on `TPIDRRO_EL0` instead. + Macos, +} + +impl From for litebox_syscall_rewriter::Host { + fn from(value: HostArg) -> Self { + match value { + HostArg::Linux => litebox_syscall_rewriter::Host::Linux, + HostArg::Macos => litebox_syscall_rewriter::Host::MacOs, + } + } +} + +/// Rewrite ELF files to hook syscalls, or PE files to hook syscalls and change GS TEB accesses to FS. #[derive(Parser, Debug)] struct CliArgs { - /// Path to input ELF binary + /// Path to input binary input_binary: PathBuf, /// Path to output the generated binary (default = .hooked) #[arg(short = 'o', long = "output")] @@ -21,6 +48,11 @@ struct CliArgs { /// Absolute address to set in the trampoline (default = 0) #[arg(long)] trampoline_addr: Option, + /// AArch64 ELF host to anchor the rewritten gates against (ignored for + /// x86-64/PE input). Defaults to `linux`; pass `macos` when the rewritten + /// binary will run under a macOS-hosted LiteBox runner instead. + #[arg(long, value_enum, default_value_t = HostArg::Linux)] + host: HostArg, } fn copy_file_permissions( @@ -47,9 +79,10 @@ fn main() -> anyhow::Result<()> { let mut input_binary = std::fs::File::open(&cli_args.input_binary)?; let mut input_binary_bytes = vec![]; input_binary.read_to_end(&mut input_binary_bytes)?; - let output_binary = litebox_syscall_rewriter::hook_syscalls_in_elf( + let output_binary = litebox_syscall_rewriter::rewrite_binary_for_host( &input_binary_bytes, cli_args.trampoline_addr, + cli_args.host.into(), )?; let output_path = cli_args.output_binary.unwrap_or_else(|| { cli_args.input_binary.with_file_name( diff --git a/litebox_syscall_rewriter/tests/aarch64_tests.rs b/litebox_syscall_rewriter/tests/aarch64_tests.rs new file mode 100644 index 0000000000..6ab3779547 --- /dev/null +++ b/litebox_syscall_rewriter/tests/aarch64_tests.rs @@ -0,0 +1,202 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +//! Integration tests for the AArch64 (Linux) rewriter, exercised through the +//! public [`hook_syscalls_in_elf`] entry point. +//! +//! These assert byte-level invariants rather than an objdump snapshot: an +//! aarch64 objdump is not reliably available on the (x86) test host, and the +//! emitted trampoline is a clean reimplementation whose exact bytes differ from +//! the reference implementation. + +// Deliberate, range-checked casts on a 64-bit host throughout this test. +#![allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)] + +use litebox_syscall_rewriter::{TRAMPOLINE_MAGIC, hook_syscalls_in_elf}; + +const HELLO_AARCH64: &[u8] = include_bytes!("hello-aarch64"); + +/// `SVC #0`. +const SVC_0: u32 = 0xD400_0001; + +/// `MSR TPIDR_EL0, Xt` / `MRS Xd, TPIDR_EL0`: the low 5 bits select the register, +/// so mask them off to match the opcode. +const TPIDR_REG_MASK: u32 = 0xFFFF_FFE0; +const MSR_TPIDR_BITS: u32 = 0xD51B_D040; +const MRS_TPIDR_BITS: u32 = 0xD53B_D040; + +fn read_u16(data: &[u8], off: usize) -> u16 { + u16::from_le_bytes(data[off..off + 2].try_into().unwrap()) +} +fn read_u32(data: &[u8], off: usize) -> u32 { + u32::from_le_bytes(data[off..off + 4].try_into().unwrap()) +} +fn read_u64(data: &[u8], off: usize) -> u64 { + u64::from_le_bytes(data[off..off + 8].try_into().unwrap()) +} + +/// Minimal ELF64 section-header walk: returns `(file_offset, vaddr, size)` for +/// every executable (`SHF_EXECINSTR`) `PROGBITS` section. +fn exec_sections(data: &[u8]) -> Vec<(usize, u64, usize)> { + let e_shoff = read_u64(data, 40) as usize; + let e_shentsize = read_u16(data, 58) as usize; + let e_shnum = read_u16(data, 60) as usize; + let mut out = Vec::new(); + for i in 0..e_shnum { + let base = e_shoff + i * e_shentsize; + let sh_type = read_u32(data, base + 4); + let sh_flags = read_u64(data, base + 8); + let sh_addr = read_u64(data, base + 16); + let sh_offset = read_u64(data, base + 24) as usize; + let sh_size = read_u64(data, base + 32) as usize; + // SHT_PROGBITS = 1, SHF_EXECINSTR = 0x4. + if sh_type == 1 && (sh_flags & 0x4) != 0 { + out.push((sh_offset, sh_addr, sh_size)); + } + } + out +} + +/// File offsets and virtual addresses of every `SVC #0` in the executable +/// sections of `data`. +fn svc_sites(data: &[u8]) -> Vec<(usize, u64)> { + let mut sites = Vec::new(); + for (file_off, vaddr, size) in exec_sections(data) { + let mut i = 0; + while i + 4 <= size { + if read_u32(data, file_off + i) == SVC_0 { + sites.push((file_off + i, vaddr + i as u64)); + } + i += 4; + } + } + sites +} + +/// File offset of the first instruction in `data`'s executable sections whose +/// bits satisfy `(insn & mask) == bits`, if any. +fn first_site(data: &[u8], mask: u32, bits: u32) -> Option { + for (file_off, _vaddr, size) in exec_sections(data) { + let mut i = 0; + while i + 4 <= size { + if read_u32(data, file_off + i) & mask == bits { + return Some(file_off + i); + } + i += 4; + } + } + None +} + +/// Decode the trailing [`TrampolineHeader64`]: `(file_offset, vaddr, size)`. +fn trampoline_header(out: &[u8]) -> (u64, u64, u64) { + let header = &out[out.len() - 32..]; + assert_eq!(&header[..8], TRAMPOLINE_MAGIC, "trampoline magic mismatch"); + ( + read_u64(header, 8), + read_u64(header, 16), + read_u64(header, 24), + ) +} + +#[test] +fn aarch64_hello_world_is_hooked() { + let original_sites = svc_sites(HELLO_AARCH64); + assert_eq!(original_sites.len(), 3, "expected 3 SVC sites in fixture"); + + let callback = 0xDEAD_0000u64; + let out = hook_syscalls_in_elf(HELLO_AARCH64, Some(callback)).unwrap(); + + // Output grew: original (patched, same length) + padding + trampoline + header. + assert!(out.len() > HELLO_AARCH64.len()); + + // --- Trailing header invariants --- + let (file_offset, vaddr, size) = trampoline_header(&out); + assert!( + size != 0, + "fixture has SVC sites, so a trampoline is emitted" + ); + assert_eq!( + file_offset % 0x1000, + 0, + "trampoline file offset page-aligned" + ); + assert_eq!(vaddr % 0x1000, 0, "trampoline vaddr page-aligned"); + assert_eq!( + file_offset + size, + (out.len() - 32) as u64, + "trampoline must end right before the 32-byte header" + ); + + // --- Trampoline prologue invariants --- + let tramp = &out[file_offset as usize..(file_offset + size) as usize]; + // Offset 0: callback slot holds the value we passed in. + assert_eq!(read_u64(tramp, 0), callback, "callback slot"); + // Offset 8: the guest thread-pointer byte offset, seeded by the rewriter so + // that a loader which does not fill it leaves the gates behaving as the + // earlier baked-immediate ones did rather than addressing offset zero. + assert_eq!( + read_u64(tramp, 8), + u64::from(litebox_syscall_rewriter::MACOS_GUEST_TPIDR_TSD_SLOT) * 8, + "guest thread-pointer offset slot" + ); + // Offset 16: the shared SVC handler — LDR X16,; BR X16. + assert_eq!( + read_u32(tramp, 16), + 0x58FF_FF90, + "LDR X16, (pcrel -16)" + ); + assert_eq!(read_u32(tramp, 20), 0xD61F_0200, "BR X16"); + + // --- Every SVC became a branch into the trampoline region --- + let tramp_range = vaddr..(vaddr + size); + for (file_off, site_vaddr) in &original_sites { + let word = read_u32(&out, *file_off); + assert_eq!( + word & 0xFC00_0000, + 0x1400_0000, + "SVC at {site_vaddr:#x} should be rewritten to B" + ); + // Reconstruct the branch target and confirm it lands in the trampoline. + let imm26 = i64::from(word & 0x03FF_FFFF); + // Sign-extend the 26-bit immediate, then scale by 4. + let disp = (imm26 << 38) >> 38 << 2; + let target = site_vaddr.wrapping_add(disp as u64); + assert!( + tramp_range.contains(&target), + "branch target {target:#x} not in trampoline range {tramp_range:?}" + ); + } + + // --- Thread-pointer handling --- + // The `MSR TPIDR_EL0` write is virtualized: rewritten to a branch into the + // trampoline's MSR gate. + let msr_off = first_site(HELLO_AARCH64, TPIDR_REG_MASK, MSR_TPIDR_BITS) + .expect("fixture has an MSR TPIDR_EL0 write"); + assert_eq!( + read_u32(&out, msr_off) & 0xFC00_0000, + 0x1400_0000, + "MSR TPIDR_EL0 should be rewritten to B" + ); + + // The `MRS TPIDR_EL0` read is virtualized: rewritten to a branch into the + // MRS gate. + let mrs_off = first_site(HELLO_AARCH64, TPIDR_REG_MASK, MRS_TPIDR_BITS) + .expect("fixture has an MRS TPIDR_EL0 read"); + assert_eq!( + read_u32(&out, mrs_off) & 0xFC00_0000, + 0x1400_0000, + "MRS TPIDR_EL0 should be rewritten to B" + ); +} + +#[test] +fn aarch64_rehooking_is_idempotent() { + let out = hook_syscalls_in_elf(HELLO_AARCH64, Some(0)).unwrap(); + // Running the rewriter on an already-hooked binary returns it unchanged. + let again = hook_syscalls_in_elf(&out, Some(0)).unwrap(); + assert_eq!( + again, out, + "already-hooked binary must be returned unchanged" + ); +} diff --git a/litebox_syscall_rewriter/tests/hello-aarch64 b/litebox_syscall_rewriter/tests/hello-aarch64 new file mode 100644 index 0000000000..e9fce87199 Binary files /dev/null and b/litebox_syscall_rewriter/tests/hello-aarch64 differ diff --git a/litebox_syscall_rewriter/tests/snapshot_tests.rs b/litebox_syscall_rewriter/tests/snapshot_tests.rs index 1efe0be826..fb01ef4167 100644 --- a/litebox_syscall_rewriter/tests/snapshot_tests.rs +++ b/litebox_syscall_rewriter/tests/snapshot_tests.rs @@ -1,7 +1,7 @@ // Copyright (c) Microsoft Corporation. // Licensed under the MIT license. -fn objdump(binary: &[u8]) -> String { +fn objdump(objdump_cmd: &str, binary: &[u8]) -> String { use std::io::Write; use std::process::Command; use tempfile::NamedTempFile; @@ -11,18 +11,42 @@ fn objdump(binary: &[u8]) -> String { temp_file.write_all(binary).unwrap(); // Run objdump on the temporary file and capture the output - let output = Command::new("objdump") + let output = Command::new(objdump_cmd) .arg("-d") .arg(temp_file.path()) .output() .unwrap(); - String::from_utf8_lossy(&output.stdout) + let mut lines = String::from_utf8_lossy(&output.stdout) .lines() .filter(|l| !l.contains("/tmp/")) .map(|line| normalize_objdump_line(line, trampoline_range.as_ref())) - .collect::>() - .join("\n") + .collect::>(); + let first_content = lines + .iter() + .position(|line| !line.is_empty()) + .unwrap_or(lines.len()); + lines.drain(..first_content); + lines.join("\n") +} + +/// Return the first objdump-like command that exists on the host from +/// `candidates`, or `None` if none are available. +fn find_objdump(candidates: &[&str]) -> Option { + use std::process::Command; + candidates + .iter() + .find(|cmd| { + // The committed snapshots are GNU-objdump renderings; LLVM's + // objdump (macOS /usr/bin/objdump, llvm-objdump) formats operands, + // byte grouping, and the file header differently (including the + // nondeterministic temp path), so anything non-GNU can only + // produce format noise. Skip rather than fail on such hosts. + Command::new(cmd).arg("--version").output().is_ok_and(|o| { + o.status.success() && String::from_utf8_lossy(&o.stdout).contains("GNU objdump") + }) + }) + .map(|cmd| (*cmd).to_owned()) } fn trampoline_range(binary: &[u8]) -> Option> { @@ -42,40 +66,77 @@ fn trampoline_range(binary: &[u8]) -> Option> { } fn normalize_objdump_line(line: &str, trampoline_range: Option<&std::ops::Range>) -> String { - let Some(trampoline_range) = trampoline_range else { - return line.trim_end().to_owned(); - }; let Some((address, rest)) = line.split_once(':') else { return line.trim_end().to_owned(); }; let tokens: Vec<_> = rest.split_whitespace().collect(); - let Some((mnemonic_idx, mnemonic)) = tokens + + // A control-transfer into the trampoline appears as a branch mnemonic + // (`jmp` on x86, `b`/`bl` on AArch64) followed by an absolute target. When + // that target lands in the trampoline region, render it relative to the + // trampoline base so the snapshot is independent of the trampoline's exact + // address. Other branches (and same-mnemonic branches that stay in the + // original code) are left untouched. + if let Some(trampoline_range) = trampoline_range { + for (i, token) in tokens.iter().enumerate() { + if !matches!(*token, "jmp" | "b" | "bl") { + continue; + } + if let Some(target) = tokens + .get(i + 1) + .and_then(|t| u64::from_str_radix(t.trim_start_matches("0x"), 16).ok()) + && trampoline_range.contains(&target) + { + let offset = target - trampoline_range.start; + return format!("{address}:\t"); + } + } + } + + // GNU and LLVM objdump differ in whitespace, capitalization, comments, + // and some numeric formatting. Keep snapshots focused on instructions + // rather than the disassembler that happened to be available. + let code_len = tokens .iter() - .enumerate() - .find(|(_, token)| !token.chars().all(|ch| ch.is_ascii_hexdigit())) - else { - return line.trim_end().to_owned(); - }; - if *mnemonic == "jmp" - && let Some(target) = tokens - .get(mnemonic_idx + 1) - .and_then(|token| u64::from_str_radix(token.trim_start_matches("0x"), 16).ok()) - && trampoline_range.contains(&target) - { - let offset = target - trampoline_range.start; - return format!("{address}:\t"); + .take_while(|token| { + matches!(token.len(), 2 | 8) && token.bytes().all(|byte| byte.is_ascii_hexdigit()) + }) + .count(); + if code_len != 0 && code_len < tokens.len() { + let machine_code = tokens[..code_len].join(" ").to_ascii_lowercase(); + let instruction = tokens[code_len..] + .iter() + .take_while(|token| !matches!(**token, "//" | "#")) + .map(|token| { + let token = token.to_ascii_lowercase(); + if token == "#0" { + "#0x0".to_owned() + } else if let Some(value) = token.strip_prefix("0x") + && value.bytes().all(|byte| byte.is_ascii_hexdigit()) + { + value.to_owned() + } else { + token + } + }) + .collect::>() + .join(" ") + .replace(", ", ","); + return format!("{address}:\t{machine_code}\t{instruction}"); } + line.trim_end().to_owned() } const HELLO_INPUT_64: &[u8] = include_bytes!("hello"); +const HELLO_INPUT_AARCH64: &[u8] = include_bytes!("hello-aarch64"); -fn run_snapshot_test(input: &[u8], snapshot: &str) { +fn run_snapshot_test(objdump_cmd: &str, input: &[u8], snapshot: &str) { let output = litebox_syscall_rewriter::hook_syscalls_in_elf(input, None).unwrap(); let diff = similar::udiff::unified_diff( similar::Algorithm::Myers, - &objdump(input), - &objdump(&output), + &objdump(objdump_cmd, input), + &objdump(objdump_cmd, &output), 3, Some(("original", "rewritten")), ); @@ -85,5 +146,35 @@ fn run_snapshot_test(input: &[u8], snapshot: &str) { #[test] fn snapshot_test_hello_world_x86_64() { - run_snapshot_test(HELLO_INPUT_64, "hello-diff"); + // Skip (rather than fail) where no GNU objdump exists: macOS' + // /usr/bin/objdump is LLVM and renders GNU-format-incompatible output + // (see `find_objdump`). + let Some(objdump_cmd) = find_objdump(&["x86_64-linux-gnu-objdump", "objdump"]) else { + eprintln!("skipping snapshot_test_hello_world_x86_64: no GNU objdump (install binutils)"); + return; + }; + run_snapshot_test(&objdump_cmd, HELLO_INPUT_64, "hello-diff"); +} + +#[test] +fn snapshot_test_hello_world_aarch64() { + // The `hello-aarch64` fixture exercises every rewrite path: an `MSR + // TPIDR_EL0` write (→ branch into an MSR gate), an `MRS TPIDR_EL0` read + // (→ branch into an MRS gate), and several `SVC #0`s. Only `MRS XZR, + // TPIDR_EL0` is left native, and the fixture has none. + // objdump only disassembles the original `.text`, so the diff captures the + // call-site rewriting, not the appended trampoline's gate internals. + // + // The host objdump usually cannot disassemble AArch64; a GNU cross + // objdump is required (see `find_objdump` for why LLVM's is excluded). + // Skip (rather than fail) when none is installed, so x86-only and macOS + // dev environments still pass. + let Some(objdump_cmd) = find_objdump(&["aarch64-linux-gnu-objdump"]) else { + eprintln!( + "skipping snapshot_test_hello_world_aarch64: no AArch64-capable GNU objdump \ + (install binutils-aarch64-linux-gnu)" + ); + return; + }; + run_snapshot_test(&objdump_cmd, HELLO_INPUT_AARCH64, "hello-aarch64-diff"); } diff --git a/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-aarch64-diff.snap b/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-aarch64-diff.snap new file mode 100644 index 0000000000..14da9636c1 --- /dev/null +++ b/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-aarch64-diff.snap @@ -0,0 +1,29 @@ +--- +source: litebox_syscall_rewriter/tests/snapshot_tests.rs +expression: diff +--- +--- original ++++ rewritten +@@ -1,15 +1,15 @@ + Disassembly of section .text: + + 0000000000400110 <_start>: +- 400110: d51bd045 msr tpidr_el0,x5 +- 400114: d53bd049 mrs x9,tpidr_el0 ++ 400110: ++ 400114: + 400118: d2800808 mov x8,#0x40 + 40011c: d2800020 mov x0,#0x1 + 400120: 910003e1 mov x1,sp + 400124: d28001c2 mov x2,#0xe +- 400128: d4000001 svc #0x0 ++ 400128: + 40012c: d2801588 mov x8,#0xac +- 400130: d4000001 svc #0x0 ++ 400130: + 400134: d2800ba8 mov x8,#0x5d + 400138: d2800000 mov x0,#0x0 +- 40013c: d4000001 svc #0x0 +\ No newline at end of file ++ 40013c: +\ No newline at end of file diff --git a/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-diff.snap b/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-diff.snap index 5c41cdaaad..9f91e9b7fa 100644 --- a/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-diff.snap +++ b/litebox_syscall_rewriter/tests/snapshots/snapshot_tests__hello-diff.snap @@ -4,1288 +4,1288 @@ expression: diff --- --- original +++ rewritten -@@ -131,8 +131,9 @@ - 401217: 48 c7 85 50 ff ff ff movq $0x20,-0xb0(%rbp) +@@ -128,8 +128,9 @@ + 401217: 48 c7 85 50 ff ff ff movq $0x20,-0xb0(%rbp) 40121e: 20 00 00 00 - 401222: bf 01 00 00 00 mov $0x1,%edi -- 401227: b8 0e 00 00 00 mov $0xe,%eax -- 40122c: 0f 05 syscall + 401222: bf 01 00 00 00 mov $0x1,%edi +- 401227: b8 0e 00 00 00 mov $0xe,%eax +- 40122c: 0f 05 syscall + 401227: -+ 40122c: 90 nop -+ 40122d: 90 nop - 40122e: 8b 05 0c fc 0a 00 mov 0xafc0c(%rip),%eax # 4b0e40 - 401234: 83 f8 01 cmp $0x1,%eax - 401237: 75 77 jne 4012b0 -@@ -1133,9 +1134,8 @@ - 401e6c: 74 12 je 401e80 <__libc_start_call_main+0x90> - 401e6e: ba 3c 00 00 00 mov $0x3c,%edx - 401e73: 0f 1f 44 00 00 nopl 0x0(%rax,%rax,1) -- 401e78: 31 ff xor %edi,%edi -- 401e7a: 89 d0 mov %edx,%eax -- 401e7c: 0f 05 syscall ++ 40122c: 90 nop ++ 40122d: 90 nop + 40122e: 8b 05 0c fc 0a 00 mov 0xafc0c(%rip),%eax + 401234: 83 f8 01 cmp $0x1,%eax + 401237: 75 77 jne 4012b0 +@@ -1130,9 +1131,8 @@ + 401e6c: 74 12 je 401e80 <__libc_start_call_main+0x90> + 401e6e: ba 3c 00 00 00 mov $0x3c,%edx + 401e73: 0f 1f 44 00 00 nopl 0x0(%rax,%rax,1) +- 401e78: 31 ff xor %edi,%edi +- 401e7a: 89 d0 mov %edx,%eax +- 401e7c: 0f 05 syscall + 401e78: -+ 401e7d: 90 nop - 401e7e: eb f8 jmp 401e78 <__libc_start_call_main+0x88> - 401e80: 31 c0 xor %eax,%eax - 401e82: eb d4 jmp 401e58 <__libc_start_call_main+0x68> -@@ -3117,8 +3117,9 @@ - 403ed9: 74 11 je 403eec <__libc_start_main+0x13c> - 403edb: be 01 00 00 00 mov $0x1,%esi - 403ee0: bf 01 50 00 00 mov $0x5001,%edi -- 403ee5: b8 9e 00 00 00 mov $0x9e,%eax -- 403eea: 0f 05 syscall ++ 401e7d: 90 nop + 401e7e: eb f8 jmp 401e78 <__libc_start_call_main+0x88> + 401e80: 31 c0 xor %eax,%eax + 401e82: eb d4 jmp 401e58 <__libc_start_call_main+0x68> +@@ -3114,8 +3114,9 @@ + 403ed9: 74 11 je 403eec <__libc_start_main+0x13c> + 403edb: be 01 00 00 00 mov $0x1,%esi + 403ee0: bf 01 50 00 00 mov $0x5001,%edi +- 403ee5: b8 9e 00 00 00 mov $0x9e,%eax +- 403eea: 0f 05 syscall + 403ee5: -+ 403eea: 90 nop -+ 403eeb: 90 nop - 403eec: 44 89 ef mov %r13d,%edi - 403eef: e8 9c d4 01 00 call 421390 <_dl_cet_setup_features> - 403ef4: 48 8b 15 0d 52 0a 00 mov 0xa520d(%rip),%rdx # 4a9108 <_dl_random> -@@ -3441,18 +3442,22 @@ - 4043c5: 48 89 46 08 mov %rax,0x8(%rsi) - 4043c9: b8 9e 00 00 00 mov $0x9e,%eax - 4043ce: 48 89 36 mov %rsi,(%rsi) -- 4043d1: 48 89 76 10 mov %rsi,0x10(%rsi) -- 4043d5: 0f 05 syscall ++ 403eea: 90 nop ++ 403eeb: 90 nop + 403eec: 44 89 ef mov %r13d,%edi + 403eef: e8 9c d4 01 00 call 421390 <_dl_cet_setup_features> + 403ef4: 48 8b 15 0d 52 0a 00 mov 0xa520d(%rip),%rdx +@@ -3438,18 +3439,22 @@ + 4043c5: 48 89 46 08 mov %rax,0x8(%rsi) + 4043c9: b8 9e 00 00 00 mov $0x9e,%eax + 4043ce: 48 89 36 mov %rsi,(%rsi) +- 4043d1: 48 89 76 10 mov %rsi,0x10(%rsi) +- 4043d5: 0f 05 syscall + 4043d1: -+ 4043d6: 90 nop - 4043d7: 85 c0 test %eax,%eax - 4043d9: 74 24 je 4043ff <__libc_setup_tls+0x1df> - 4043db: ba 2d 00 00 00 mov $0x2d,%edx - 4043e0: bf 02 00 00 00 mov $0x2,%edi - 4043e5: b8 01 00 00 00 mov $0x1,%eax -- 4043ea: 48 8d 35 c7 d1 07 00 lea 0x7d1c7(%rip),%rsi # 4815b8 -- 4043f1: 0f 05 syscall ++ 4043d6: 90 nop + 4043d7: 85 c0 test %eax,%eax + 4043d9: 74 24 je 4043ff <__libc_setup_tls+0x1df> + 4043db: ba 2d 00 00 00 mov $0x2d,%edx + 4043e0: bf 02 00 00 00 mov $0x2,%edi + 4043e5: b8 01 00 00 00 mov $0x1,%eax +- 4043ea: 48 8d 35 c7 d1 07 00 lea 0x7d1c7(%rip),%rsi +- 4043f1: 0f 05 syscall + 4043ea: -+ 4043ef: 90 nop -+ 4043f0: 90 nop -+ 4043f1: 90 nop -+ 4043f2: 90 nop - 4043f3: bf 7f 00 00 00 mov $0x7f,%edi -- 4043f8: b8 e7 00 00 00 mov $0xe7,%eax -- 4043fd: 0f 05 syscall ++ 4043ef: 90 nop ++ 4043f0: 90 nop ++ 4043f1: 90 nop ++ 4043f2: 90 nop + 4043f3: bf 7f 00 00 00 mov $0x7f,%edi +- 4043f8: b8 e7 00 00 00 mov $0xe7,%eax +- 4043fd: 0f 05 syscall + 4043f8: -+ 4043fd: 90 nop -+ 4043fe: 90 nop - 4043ff: e8 dc ba 01 00 call 41fee0 <__tls_init_tp> - 404404: 48 8b 45 c8 mov -0x38(%rbp),%rax - 404408: 4d 89 ae 78 04 00 00 mov %r13,0x478(%r14) -@@ -3492,11 +3497,15 @@ - 4044b0: ba 2d 00 00 00 mov $0x2d,%edx - 4044b5: bf 02 00 00 00 mov $0x2,%edi - 4044ba: b8 01 00 00 00 mov $0x1,%eax -- 4044bf: 48 8d 35 f2 d0 07 00 lea 0x7d0f2(%rip),%rsi # 4815b8 -- 4044c6: 0f 05 syscall ++ 4043fd: 90 nop ++ 4043fe: 90 nop + 4043ff: e8 dc ba 01 00 call 41fee0 <__tls_init_tp> + 404404: 48 8b 45 c8 mov -0x38(%rbp),%rax + 404408: 4d 89 ae 78 04 00 00 mov %r13,0x478(%r14) +@@ -3489,11 +3494,15 @@ + 4044b0: ba 2d 00 00 00 mov $0x2d,%edx + 4044b5: bf 02 00 00 00 mov $0x2,%edi + 4044ba: b8 01 00 00 00 mov $0x1,%eax +- 4044bf: 48 8d 35 f2 d0 07 00 lea 0x7d0f2(%rip),%rsi +- 4044c6: 0f 05 syscall + 4044bf: -+ 4044c4: 90 nop -+ 4044c5: 90 nop -+ 4044c6: 90 nop -+ 4044c7: 90 nop - 4044c8: bf 7f 00 00 00 mov $0x7f,%edi -- 4044cd: b8 e7 00 00 00 mov $0xe7,%eax -- 4044d2: 0f 05 syscall ++ 4044c4: 90 nop ++ 4044c5: 90 nop ++ 4044c6: 90 nop ++ 4044c7: 90 nop + 4044c8: bf 7f 00 00 00 mov $0x7f,%edi +- 4044cd: b8 e7 00 00 00 mov $0xe7,%eax +- 4044d2: 0f 05 syscall + 4044cd: -+ 4044d2: 90 nop -+ 4044d3: 90 nop - 4044d4: e9 70 fe ff ff jmp 404349 <__libc_setup_tls+0x129> - 4044d9: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) ++ 4044d2: 90 nop ++ 4044d3: 90 nop + 4044d4: e9 70 fe ff ff jmp 404349 <__libc_setup_tls+0x129> + 4044d9: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) -@@ -9234,8 +9243,7 @@ - 40a3dc: 0f 1f 40 00 nopl 0x0(%rax) - 40a3e0: 48 8b b5 f0 fe ff ff mov -0x110(%rbp),%rsi - 40a3e7: bf 02 00 00 00 mov $0x2,%edi -- 40a3ec: 44 89 c8 mov %r9d,%eax -- 40a3ef: 0f 05 syscall +@@ -9231,8 +9240,7 @@ + 40a3dc: 0f 1f 40 00 nopl 0x0(%rax) + 40a3e0: 48 8b b5 f0 fe ff ff mov -0x110(%rbp),%rsi + 40a3e7: bf 02 00 00 00 mov $0x2,%edi +- 40a3ec: 44 89 c8 mov %r9d,%eax +- 40a3ef: 0f 05 syscall + 40a3ec: - 40a3f1: 48 83 f8 fc cmp $0xfffffffffffffffc,%rax - 40a3f5: 74 e9 je 40a3e0 <__libc_message_impl+0x150> - 40a3f7: 45 31 c9 xor %r9d,%r9d -@@ -9372,8 +9380,9 @@ - 40a5c7: 45 31 d2 xor %r10d,%r10d - 40a5ca: ba 02 00 00 00 mov $0x2,%edx - 40a5cf: be 80 00 00 00 mov $0x80,%esi -- 40a5d4: b8 ca 00 00 00 mov $0xca,%eax -- 40a5d9: 0f 05 syscall + 40a3f1: 48 83 f8 fc cmp $0xfffffffffffffffc,%rax + 40a3f5: 74 e9 je 40a3e0 <__libc_message_impl+0x150> + 40a3f7: 45 31 c9 xor %r9d,%r9d +@@ -9369,8 +9377,9 @@ + 40a5c7: 45 31 d2 xor %r10d,%r10d + 40a5ca: ba 02 00 00 00 mov $0x2,%edx + 40a5cf: be 80 00 00 00 mov $0x80,%esi +- 40a5d4: b8 ca 00 00 00 mov $0xca,%eax +- 40a5d9: 0f 05 syscall + 40a5d4: -+ 40a5d9: 90 nop -+ 40a5da: 90 nop - 40a5db: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 40a5e1: 76 d8 jbe 40a5bb <__lll_lock_wait_private+0xb> - 40a5e3: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -9405,8 +9414,8 @@ - 40a62d: 45 31 d2 xor %r10d,%r10d - 40a630: ba 02 00 00 00 mov $0x2,%edx - 40a635: b8 ca 00 00 00 mov $0xca,%eax -- 40a63a: 40 80 f6 80 xor $0x80,%sil -- 40a63e: 0f 05 syscall ++ 40a5d9: 90 nop ++ 40a5da: 90 nop + 40a5db: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 40a5e1: 76 d8 jbe 40a5bb <__lll_lock_wait_private+0xb> + 40a5e3: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -9402,8 +9411,8 @@ + 40a62d: 45 31 d2 xor %r10d,%r10d + 40a630: ba 02 00 00 00 mov $0x2,%edx + 40a635: b8 ca 00 00 00 mov $0xca,%eax +- 40a63a: 40 80 f6 80 xor $0x80,%sil +- 40a63e: 0f 05 syscall + 40a63a: -+ 40a63f: 90 nop - 40a640: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 40a646: 76 d6 jbe 40a61e <__lll_lock_wait+0xe> - 40a648: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -9426,8 +9435,9 @@ - 40a674: 45 31 d2 xor %r10d,%r10d - 40a677: ba 01 00 00 00 mov $0x1,%edx - 40a67c: be 81 00 00 00 mov $0x81,%esi -- 40a681: b8 ca 00 00 00 mov $0xca,%eax -- 40a686: 0f 05 syscall ++ 40a63f: 90 nop + 40a640: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 40a646: 76 d6 jbe 40a61e <__lll_lock_wait+0xe> + 40a648: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -9423,8 +9432,9 @@ + 40a674: 45 31 d2 xor %r10d,%r10d + 40a677: ba 01 00 00 00 mov $0x1,%edx + 40a67c: be 81 00 00 00 mov $0x81,%esi +- 40a681: b8 ca 00 00 00 mov $0xca,%eax +- 40a686: 0f 05 syscall + 40a681: -+ 40a686: 90 nop -+ 40a687: 90 nop - 40a688: c3 ret - 40a689: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) ++ 40a686: 90 nop ++ 40a687: 90 nop + 40a688: c3 ret + 40a689: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) -@@ -9436,8 +9446,9 @@ - 40a694: 40 80 f6 81 xor $0x81,%sil - 40a698: 45 31 d2 xor %r10d,%r10d - 40a69b: ba 01 00 00 00 mov $0x1,%edx -- 40a6a0: b8 ca 00 00 00 mov $0xca,%eax -- 40a6a5: 0f 05 syscall +@@ -9433,8 +9443,9 @@ + 40a694: 40 80 f6 81 xor $0x81,%sil + 40a698: 45 31 d2 xor %r10d,%r10d + 40a69b: ba 01 00 00 00 mov $0x1,%edx +- 40a6a0: b8 ca 00 00 00 mov $0xca,%eax +- 40a6a5: 0f 05 syscall + 40a6a0: -+ 40a6a5: 90 nop -+ 40a6a6: 90 nop - 40a6a7: c3 ret - 40a6a8: 0f 1f 84 00 00 00 00 nopl 0x0(%rax,%rax,1) ++ 40a6a5: 90 nop ++ 40a6a6: 90 nop + 40a6a7: c3 ret + 40a6a8: 0f 1f 84 00 00 00 00 nopl 0x0(%rax,%rax,1) 40a6af: 00 -@@ -10840,8 +10851,9 @@ - 40bbd5: 48 89 45 e8 mov %rax,-0x18(%rbp) - 40bbd9: 31 c0 xor %eax,%eax - 40bbdb: c6 05 3e 4c 0a 00 01 movb $0x1,0xa4c3e(%rip) # 4b0820 <__malloc_initialized> -- 40bbe2: b8 3e 01 00 00 mov $0x13e,%eax -- 40bbe7: 0f 05 syscall +@@ -10837,8 +10848,9 @@ + 40bbd5: 48 89 45 e8 mov %rax,-0x18(%rbp) + 40bbd9: 31 c0 xor %eax,%eax + 40bbdb: c6 05 3e 4c 0a 00 01 movb $0x1,0xa4c3e(%rip) +- 40bbe2: b8 3e 01 00 00 mov $0x13e,%eax +- 40bbe7: 0f 05 syscall + 40bbe2: -+ 40bbe7: 90 nop -+ 40bbe8: 90 nop - 40bbe9: 48 8d 5d d0 lea -0x30(%rbp),%rbx - 40bbed: 48 83 f8 08 cmp $0x8,%rax - 40bbf1: 74 4e je 40bc41 -@@ -23532,8 +23544,9 @@ - 4181dc: 5d pop %rbp - 4181dd: c3 ret - 4181de: 66 90 xchg %ax,%ax -- 4181e0: b8 e4 00 00 00 mov $0xe4,%eax -- 4181e5: 0f 05 syscall ++ 40bbe7: 90 nop ++ 40bbe8: 90 nop + 40bbe9: 48 8d 5d d0 lea -0x30(%rbp),%rbx + 40bbed: 48 83 f8 08 cmp $0x8,%rax + 40bbf1: 74 4e je 40bc41 +@@ -23529,8 +23541,9 @@ + 4181dc: 5d pop %rbp + 4181dd: c3 ret + 4181de: 66 90 xchg %ax,%ax +- 4181e0: b8 e4 00 00 00 mov $0xe4,%eax +- 4181e5: 0f 05 syscall + 4181e0: -+ 4181e5: 90 nop -+ 4181e6: 90 nop - 4181e7: 85 c0 test %eax,%eax - 4181e9: 75 1d jne 418208 <__clock_gettime+0x48> - 4181eb: 31 c0 xor %eax,%eax -@@ -23566,8 +23579,10 @@ - 418242: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) - 418248: f4 hlt - 418249: 89 d0 mov %edx,%eax -- 41824b: 0f 05 syscall -- 41824d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax ++ 4181e5: 90 nop ++ 4181e6: 90 nop + 4181e7: 85 c0 test %eax,%eax + 4181e9: 75 1d jne 418208 <__clock_gettime+0x48> + 4181eb: 31 c0 xor %eax,%eax +@@ -23563,8 +23576,10 @@ + 418242: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) + 418248: f4 hlt + 418249: 89 d0 mov %edx,%eax +- 41824b: 0f 05 syscall +- 41824d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 41824b: -+ 418250: 90 nop -+ 418251: 90 nop -+ 418252: 90 nop - 418253: 76 f3 jbe 418248 <_exit+0x18> - 418255: f7 d8 neg %eax - 418257: 64 89 06 mov %eax,%fs:(%rsi) -@@ -23576,8 +23591,9 @@ ++ 418250: 90 nop ++ 418251: 90 nop ++ 418252: 90 nop + 418253: 76 f3 jbe 418248 <_exit+0x18> + 418255: f7 d8 neg %eax + 418257: 64 89 06 mov %eax,%fs:(%rsi) +@@ -23573,8 +23588,9 @@ 0000000000418260 <__fstat>: - 418260: f3 0f 1e fa endbr64 -- 418264: b8 05 00 00 00 mov $0x5,%eax -- 418269: 0f 05 syscall + 418260: f3 0f 1e fa endbr64 +- 418264: b8 05 00 00 00 mov $0x5,%eax +- 418269: 0f 05 syscall + 418264: -+ 418269: 90 nop -+ 41826a: 90 nop - 41826b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 418271: 77 05 ja 418278 <__fstat+0x18> - 418273: c3 ret -@@ -23591,8 +23607,9 @@ ++ 418269: 90 nop ++ 41826a: 90 nop + 41826b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 418271: 77 05 ja 418278 <__fstat+0x18> + 418273: c3 ret +@@ -23588,8 +23604,9 @@ 0000000000418290 <__close_nocancel>: - 418290: f3 0f 1e fa endbr64 -- 418294: b8 03 00 00 00 mov $0x3,%eax -- 418299: 0f 05 syscall + 418290: f3 0f 1e fa endbr64 +- 418294: b8 03 00 00 00 mov $0x3,%eax +- 418299: 0f 05 syscall + 418294: -+ 418299: 90 nop -+ 41829a: 90 nop - 41829b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 4182a1: 77 05 ja 4182a8 <__close_nocancel+0x18> - 4182a3: c3 ret -@@ -23621,8 +23638,9 @@ - 4182f2: 48 89 45 c0 mov %rax,-0x40(%rbp) - 4182f6: 83 fe 09 cmp $0x9,%esi - 4182f9: 74 25 je 418320 <__fcntl64_nocancel+0x60> -- 4182fb: b8 48 00 00 00 mov $0x48,%eax -- 418300: 0f 05 syscall ++ 418299: 90 nop ++ 41829a: 90 nop + 41829b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 4182a1: 77 05 ja 4182a8 <__close_nocancel+0x18> + 4182a3: c3 ret +@@ -23618,8 +23635,9 @@ + 4182f2: 48 89 45 c0 mov %rax,-0x40(%rbp) + 4182f6: 83 fe 09 cmp $0x9,%esi + 4182f9: 74 25 je 418320 <__fcntl64_nocancel+0x60> +- 4182fb: b8 48 00 00 00 mov $0x48,%eax +- 418300: 0f 05 syscall + 4182fb: -+ 418300: 90 nop -+ 418301: 90 nop - 418302: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 418308: 77 3e ja 418348 <__fcntl64_nocancel+0x88> - 41830a: 48 8b 55 c8 mov -0x38(%rbp),%rdx -@@ -23634,8 +23652,9 @@ - 41831b: 0f 1f 44 00 00 nopl 0x0(%rax,%rax,1) - 418320: 48 8d 55 a8 lea -0x58(%rbp),%rdx - 418324: be 10 00 00 00 mov $0x10,%esi -- 418329: b8 48 00 00 00 mov $0x48,%eax -- 41832e: 0f 05 syscall ++ 418300: 90 nop ++ 418301: 90 nop + 418302: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 418308: 77 3e ja 418348 <__fcntl64_nocancel+0x88> + 41830a: 48 8b 55 c8 mov -0x38(%rbp),%rdx +@@ -23631,8 +23649,9 @@ + 41831b: 0f 1f 44 00 00 nopl 0x0(%rax,%rax,1) + 418320: 48 8d 55 a8 lea -0x58(%rbp),%rdx + 418324: be 10 00 00 00 mov $0x10,%esi +- 418329: b8 48 00 00 00 mov $0x48,%eax +- 41832e: 0f 05 syscall + 418329: -+ 41832e: 90 nop -+ 41832f: 90 nop - 418330: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 418335: 77 11 ja 418348 <__fcntl64_nocancel+0x88> - 418337: 83 7d a8 02 cmpl $0x2,-0x58(%rbp) -@@ -23662,8 +23681,9 @@ - 418379: 31 c0 xor %eax,%eax - 41837b: 83 fe 09 cmp $0x9,%esi - 41837e: 74 20 je 4183a0 <__fcntl64_nocancel_adjusted+0x40> -- 418380: b8 48 00 00 00 mov $0x48,%eax -- 418385: 0f 05 syscall ++ 41832e: 90 nop ++ 41832f: 90 nop + 418330: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 418335: 77 11 ja 418348 <__fcntl64_nocancel+0x88> + 418337: 83 7d a8 02 cmpl $0x2,-0x58(%rbp) +@@ -23659,8 +23678,9 @@ + 418379: 31 c0 xor %eax,%eax + 41837b: 83 fe 09 cmp $0x9,%esi + 41837e: 74 20 je 4183a0 <__fcntl64_nocancel_adjusted+0x40> +- 418380: b8 48 00 00 00 mov $0x48,%eax +- 418385: 0f 05 syscall + 418380: -+ 418385: 90 nop -+ 418386: 90 nop - 418387: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 41838d: 77 39 ja 4183c8 <__fcntl64_nocancel_adjusted+0x68> - 41838f: 48 8b 55 f8 mov -0x8(%rbp),%rdx -@@ -23674,8 +23694,9 @@ - 41839f: c3 ret - 4183a0: 48 8d 55 f0 lea -0x10(%rbp),%rdx - 4183a4: be 10 00 00 00 mov $0x10,%esi -- 4183a9: b8 48 00 00 00 mov $0x48,%eax -- 4183ae: 0f 05 syscall ++ 418385: 90 nop ++ 418386: 90 nop + 418387: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 41838d: 77 39 ja 4183c8 <__fcntl64_nocancel_adjusted+0x68> + 41838f: 48 8b 55 f8 mov -0x8(%rbp),%rdx +@@ -23671,8 +23691,9 @@ + 41839f: c3 ret + 4183a0: 48 8d 55 f0 lea -0x10(%rbp),%rdx + 4183a4: be 10 00 00 00 mov $0x10,%esi +- 4183a9: b8 48 00 00 00 mov $0x48,%eax +- 4183ae: 0f 05 syscall + 4183a9: -+ 4183ae: 90 nop -+ 4183af: 90 nop - 4183b0: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 4183b5: 77 11 ja 4183c8 <__fcntl64_nocancel_adjusted+0x68> - 4183b7: 83 7d f0 02 cmpl $0x2,-0x10(%rbp) -@@ -23711,8 +23732,9 @@ - 418413: 89 f2 mov %esi,%edx - 418415: b8 01 01 00 00 mov $0x101,%eax - 41841a: 48 89 fe mov %rdi,%rsi -- 41841d: bf 9c ff ff ff mov $0xffffff9c,%edi -- 418422: 0f 05 syscall ++ 4183ae: 90 nop ++ 4183af: 90 nop + 4183b0: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 4183b5: 77 11 ja 4183c8 <__fcntl64_nocancel_adjusted+0x68> + 4183b7: 83 7d f0 02 cmpl $0x2,-0x10(%rbp) +@@ -23708,8 +23729,9 @@ + 418413: 89 f2 mov %esi,%edx + 418415: b8 01 01 00 00 mov $0x101,%eax + 41841a: 48 89 fe mov %rdi,%rsi +- 41841d: bf 9c ff ff ff mov $0xffffff9c,%edi +- 418422: 0f 05 syscall + 41841d: -+ 418422: 90 nop -+ 418423: 90 nop - 418424: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 41842a: 77 34 ja 418460 <__open64_nocancel+0x80> - 41842c: 48 8b 55 c8 mov -0x38(%rbp),%rdx -@@ -23740,9 +23762,10 @@ ++ 418422: 90 nop ++ 418423: 90 nop + 418424: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 41842a: 77 34 ja 418460 <__open64_nocancel+0x80> + 41842c: 48 8b 55 c8 mov -0x38(%rbp),%rdx +@@ -23737,9 +23759,10 @@ 41847f: 00 0000000000418480 <__read_nocancel>: -- 418480: f3 0f 1e fa endbr64 -- 418484: 31 c0 xor %eax,%eax -- 418486: 0f 05 syscall +- 418480: f3 0f 1e fa endbr64 +- 418484: 31 c0 xor %eax,%eax +- 418486: 0f 05 syscall + 418480: -+ 418485: 90 nop -+ 418486: 90 nop -+ 418487: 90 nop - 418488: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 41848e: 77 08 ja 418498 <__read_nocancel+0x18> - 418490: c3 ret -@@ -23756,8 +23779,9 @@ ++ 418485: 90 nop ++ 418486: 90 nop ++ 418487: 90 nop + 418488: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 41848e: 77 08 ja 418498 <__read_nocancel+0x18> + 418490: c3 ret +@@ -23753,8 +23776,9 @@ 00000000004184b0 <__brk>: - 4184b0: f3 0f 1e fa endbr64 -- 4184b4: b8 0c 00 00 00 mov $0xc,%eax -- 4184b9: 0f 05 syscall + 4184b0: f3 0f 1e fa endbr64 +- 4184b4: b8 0c 00 00 00 mov $0xc,%eax +- 4184b9: 0f 05 syscall + 4184b4: -+ 4184b9: 90 nop -+ 4184ba: 90 nop - 4184bb: 48 89 05 96 83 09 00 mov %rax,0x98396(%rip) # 4b0858 <__curbrk> - 4184c2: 48 39 f8 cmp %rdi,%rax - 4184c5: 72 09 jb 4184d0 <__brk+0x20> -@@ -23922,8 +23946,9 @@ - 418714: 48 89 45 f8 mov %rax,-0x8(%rbp) - 418718: 31 c0 xor %eax,%eax - 41871a: 48 8d 95 f0 ef ff ff lea -0x1010(%rbp),%rdx -- 418721: b8 cc 00 00 00 mov $0xcc,%eax -- 418726: 0f 05 syscall ++ 4184b9: 90 nop ++ 4184ba: 90 nop + 4184bb: 48 89 05 96 83 09 00 mov %rax,0x98396(%rip) + 4184c2: 48 39 f8 cmp %rdi,%rax + 4184c5: 72 09 jb 4184d0 <__brk+0x20> +@@ -23919,8 +23943,9 @@ + 418714: 48 89 45 f8 mov %rax,-0x8(%rbp) + 418718: 31 c0 xor %eax,%eax + 41871a: 48 8d 95 f0 ef ff ff lea -0x1010(%rbp),%rdx +- 418721: b8 cc 00 00 00 mov $0xcc,%eax +- 418726: 0f 05 syscall + 418721: -+ 418726: 90 nop -+ 418727: 90 nop - 418728: 85 c0 test %eax,%eax - 41872a: 7f 24 jg 418750 <__get_nprocs_sched+0x60> - 41872c: 83 f8 ea cmp $0xffffffea,%eax -@@ -24231,8 +24256,9 @@ ++ 418726: 90 nop ++ 418727: 90 nop + 418728: 85 c0 test %eax,%eax + 41872a: 7f 24 jg 418750 <__get_nprocs_sched+0x60> + 41872c: 83 f8 ea cmp $0xffffffea,%eax +@@ -24228,8 +24253,9 @@ 0000000000418b40 <__madvise>: - 418b40: f3 0f 1e fa endbr64 -- 418b44: b8 1c 00 00 00 mov $0x1c,%eax -- 418b49: 0f 05 syscall + 418b40: f3 0f 1e fa endbr64 +- 418b44: b8 1c 00 00 00 mov $0x1c,%eax +- 418b49: 0f 05 syscall + 418b44: -+ 418b49: 90 nop -+ 418b4a: 90 nop - 418b4b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 418b51: 73 01 jae 418b54 <__madvise+0x14> - 418b53: c3 ret -@@ -24259,8 +24285,9 @@ - 418b8d: 74 41 je 418bd0 <__mmap64+0x60> - 418b8f: 45 89 e2 mov %r12d,%r10d - 418b92: 48 89 df mov %rbx,%rdi -- 418b95: b8 09 00 00 00 mov $0x9,%eax -- 418b9a: 0f 05 syscall ++ 418b49: 90 nop ++ 418b4a: 90 nop + 418b4b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 418b51: 73 01 jae 418b54 <__madvise+0x14> + 418b53: c3 ret +@@ -24256,8 +24282,9 @@ + 418b8d: 74 41 je 418bd0 <__mmap64+0x60> + 418b8f: 45 89 e2 mov %r12d,%r10d + 418b92: 48 89 df mov %rbx,%rdi +- 418b95: b8 09 00 00 00 mov $0x9,%eax +- 418b9a: 0f 05 syscall + 418b95: -+ 418b9a: 90 nop -+ 418b9b: 90 nop - 418b9c: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 418ba2: 77 6c ja 418c10 <__mmap64+0xa0> - 418ba4: 5b pop %rbx -@@ -24284,8 +24311,8 @@ - 418be8: 45 89 e2 mov %r12d,%r10d - 418beb: 31 ff xor %edi,%edi - 418bed: b8 09 00 00 00 mov $0x9,%eax -- 418bf2: 41 83 ca 40 or $0x40,%r10d -- 418bf6: 0f 05 syscall ++ 418b9a: 90 nop ++ 418b9b: 90 nop + 418b9c: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 418ba2: 77 6c ja 418c10 <__mmap64+0xa0> + 418ba4: 5b pop %rbx +@@ -24281,8 +24308,8 @@ + 418be8: 45 89 e2 mov %r12d,%r10d + 418beb: 31 ff xor %edi,%edi + 418bed: b8 09 00 00 00 mov $0x9,%eax +- 418bf2: 41 83 ca 40 or $0x40,%r10d +- 418bf6: 0f 05 syscall + 418bf2: -+ 418bf7: 90 nop - 418bf8: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 418bfe: 76 a4 jbe 418ba4 <__mmap64+0x34> - 418c00: 48 c7 c1 c0 ff ff ff mov $0xffffffffffffffc0,%rcx -@@ -24303,8 +24330,9 @@ ++ 418bf7: 90 nop + 418bf8: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 418bfe: 76 a4 jbe 418ba4 <__mmap64+0x34> + 418c00: 48 c7 c1 c0 ff ff ff mov $0xffffffffffffffc0,%rcx +@@ -24300,8 +24327,9 @@ 0000000000418c30 <__mprotect>: - 418c30: f3 0f 1e fa endbr64 -- 418c34: b8 0a 00 00 00 mov $0xa,%eax -- 418c39: 0f 05 syscall + 418c30: f3 0f 1e fa endbr64 +- 418c34: b8 0a 00 00 00 mov $0xa,%eax +- 418c39: 0f 05 syscall + 418c34: -+ 418c39: 90 nop -+ 418c3a: 90 nop - 418c3b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 418c41: 73 01 jae 418c44 <__mprotect+0x14> - 418c43: c3 ret -@@ -24319,8 +24347,9 @@ ++ 418c39: 90 nop ++ 418c3a: 90 nop + 418c3b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 418c41: 73 01 jae 418c44 <__mprotect+0x14> + 418c43: c3 ret +@@ -24316,8 +24344,9 @@ 0000000000418c60 <__munmap>: - 418c60: f3 0f 1e fa endbr64 -- 418c64: b8 0b 00 00 00 mov $0xb,%eax -- 418c69: 0f 05 syscall + 418c60: f3 0f 1e fa endbr64 +- 418c64: b8 0b 00 00 00 mov $0xb,%eax +- 418c69: 0f 05 syscall + 418c64: -+ 418c69: 90 nop -+ 418c6a: 90 nop - 418c6b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 418c71: 73 01 jae 418c74 <__munmap+0x14> - 418c73: c3 ret -@@ -24396,8 +24425,9 @@ - 418d42: 83 e1 02 and $0x2,%ecx - 418d45: 75 29 jne 418d70 <__mremap+0x50> - 418d47: 45 31 c0 xor %r8d,%r8d -- 418d4a: b8 19 00 00 00 mov $0x19,%eax -- 418d4f: 0f 05 syscall ++ 418c69: 90 nop ++ 418c6a: 90 nop + 418c6b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 418c71: 73 01 jae 418c74 <__munmap+0x14> + 418c73: c3 ret +@@ -24393,8 +24422,9 @@ + 418d42: 83 e1 02 and $0x2,%ecx + 418d45: 75 29 jne 418d70 <__mremap+0x50> + 418d47: 45 31 c0 xor %r8d,%r8d +- 418d4a: b8 19 00 00 00 mov $0x19,%eax +- 418d4f: 0f 05 syscall + 418d4a: -+ 418d4f: 90 nop -+ 418d50: 90 nop - 418d51: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 418d57: 77 37 ja 418d90 <__mremap+0x70> - 418d59: 48 8b 55 c8 mov -0x38(%rbp),%rdx -@@ -24463,8 +24493,9 @@ - 418e1e: 48 89 da mov %rbx,%rdx - 418e21: 31 f6 xor %esi,%esi - 418e23: bf 41 4d 56 53 mov $0x53564d41,%edi -- 418e28: b8 9d 00 00 00 mov $0x9d,%eax -- 418e2d: 0f 05 syscall ++ 418d4f: 90 nop ++ 418d50: 90 nop + 418d51: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 418d57: 77 37 ja 418d90 <__mremap+0x70> + 418d59: 48 8b 55 c8 mov -0x38(%rbp),%rdx +@@ -24460,8 +24490,9 @@ + 418e1e: 48 89 da mov %rbx,%rdx + 418e21: 31 f6 xor %esi,%esi + 418e23: bf 41 4d 56 53 mov $0x53564d41,%edi +- 418e28: b8 9d 00 00 00 mov $0x9d,%eax +- 418e2d: 0f 05 syscall + 418e28: -+ 418e2d: 90 nop -+ 418e2e: 90 nop - 418e2f: 83 f8 ea cmp $0xffffffea,%eax - 418e32: 75 a4 jne 418dd8 <__set_vma_name+0x28> - 418e34: c7 05 5e 1c 09 00 00 movl $0x0,0x91c5e(%rip) # 4aaa9c -@@ -24477,8 +24508,9 @@ ++ 418e2d: 90 nop ++ 418e2e: 90 nop + 418e2f: 83 f8 ea cmp $0xffffffea,%eax + 418e32: 75 a4 jne 418dd8 <__set_vma_name+0x28> + 418e34: c7 05 5e 1c 09 00 00 movl $0x0,0x91c5e(%rip) +@@ -24474,8 +24505,9 @@ 0000000000418e50 <__sysinfo>: - 418e50: f3 0f 1e fa endbr64 -- 418e54: b8 63 00 00 00 mov $0x63,%eax -- 418e59: 0f 05 syscall + 418e50: f3 0f 1e fa endbr64 +- 418e54: b8 63 00 00 00 mov $0x63,%eax +- 418e59: 0f 05 syscall + 418e54: -+ 418e59: 90 nop -+ 418e5a: 90 nop - 418e5b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 418e61: 73 01 jae 418e64 <__sysinfo+0x14> - 418e63: c3 ret -@@ -29948,8 +29980,7 @@ - 41e488: b8 0b 01 00 00 mov $0x10b,%eax - 41e48d: 48 8d 35 0d 17 06 00 lea 0x6170d(%rip),%rsi # 47fba1 <__PRETTY_FUNCTION__.20+0x37e> - 41e494: 48 8d 9d e0 ef ff ff lea -0x1020(%rbp),%rbx -- 41e49b: 48 89 da mov %rbx,%rdx -- 41e49e: 0f 05 syscall ++ 418e59: 90 nop ++ 418e5a: 90 nop + 418e5b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 418e61: 73 01 jae 418e64 <__sysinfo+0x14> + 418e63: c3 ret +@@ -29945,8 +29977,7 @@ + 41e488: b8 0b 01 00 00 mov $0x10b,%eax + 41e48d: 48 8d 35 0d 17 06 00 lea 0x6170d(%rip),%rsi + 41e494: 48 8d 9d e0 ef ff ff lea -0x1020(%rbp),%rbx +- 41e49b: 48 89 da mov %rbx,%rdx +- 41e49e: 0f 05 syscall + 41e49b: - 41e4a0: 85 c0 test %eax,%eax - 41e4a2: 7e 5c jle 41e500 <_dl_get_origin+0xa0> - 41e4a4: 0f b6 95 e0 ef ff ff movzbl -0x1020(%rbp),%edx -@@ -30115,8 +30146,9 @@ - 41e6d2: 8b bd a8 f6 ff ff mov -0x958(%rbp),%edi - 41e6d8: 48 63 d3 movslq %ebx,%rdx - 41e6db: 48 8d b5 d0 f6 ff ff lea -0x930(%rbp),%rsi -- 41e6e2: b8 14 00 00 00 mov $0x14,%eax -- 41e6e7: 0f 05 syscall + 41e4a0: 85 c0 test %eax,%eax + 41e4a2: 7e 5c jle 41e500 <_dl_get_origin+0xa0> + 41e4a4: 0f b6 95 e0 ef ff ff movzbl -0x1020(%rbp),%edx +@@ -30112,8 +30143,9 @@ + 41e6d2: 8b bd a8 f6 ff ff mov -0x958(%rbp),%edi + 41e6d8: 48 63 d3 movslq %ebx,%rdx + 41e6db: 48 8d b5 d0 f6 ff ff lea -0x930(%rbp),%rsi +- 41e6e2: b8 14 00 00 00 mov $0x14,%eax +- 41e6e7: 0f 05 syscall + 41e6e2: -+ 41e6e7: 90 nop -+ 41e6e8: 90 nop - 41e6e9: 48 81 c4 38 09 00 00 add $0x938,%rsp - 41e6f0: 5b pop %rbx - 41e6f1: 41 5c pop %r12 -@@ -31674,8 +31706,9 @@ - 41ff19: 48 89 42 08 mov %rax,0x8(%rdx) - 41ff1d: 48 89 05 ec 09 09 00 mov %rax,0x909ec(%rip) # 4b0910 <_dl_stack_user> - 41ff24: 48 8d bb d0 02 00 00 lea 0x2d0(%rbx),%rdi -- 41ff2b: b8 da 00 00 00 mov $0xda,%eax -- 41ff30: 0f 05 syscall ++ 41e6e7: 90 nop ++ 41e6e8: 90 nop + 41e6e9: 48 81 c4 38 09 00 00 add $0x938,%rsp + 41e6f0: 5b pop %rbx + 41e6f1: 41 5c pop %r12 +@@ -31671,8 +31703,9 @@ + 41ff19: 48 89 42 08 mov %rax,0x8(%rdx) + 41ff1d: 48 89 05 ec 09 09 00 mov %rax,0x909ec(%rip) + 41ff24: 48 8d bb d0 02 00 00 lea 0x2d0(%rbx),%rdi +- 41ff2b: b8 da 00 00 00 mov $0xda,%eax +- 41ff30: 0f 05 syscall + 41ff2b: -+ 41ff30: 90 nop -+ 41ff31: 90 nop - 41ff32: 89 83 d0 02 00 00 mov %eax,0x2d0(%rbx) - 41ff38: 48 8d 83 10 03 00 00 lea 0x310(%rbx),%rax - 41ff3f: 64 48 89 04 25 10 05 mov %rax,%fs:0x510 -@@ -31692,8 +31725,11 @@ - 41ff77: b8 11 01 00 00 mov $0x111,%eax - 41ff7c: 66 48 0f 6e c7 movq %rdi,%xmm0 - 41ff81: 66 0f 6c c0 punpcklqdq %xmm0,%xmm0 -- 41ff85: 0f 11 83 d8 02 00 00 movups %xmm0,0x2d8(%rbx) -- 41ff8c: 0f 05 syscall ++ 41ff30: 90 nop ++ 41ff31: 90 nop + 41ff32: 89 83 d0 02 00 00 mov %eax,0x2d0(%rbx) + 41ff38: 48 8d 83 10 03 00 00 lea 0x310(%rbx),%rax + 41ff3f: 64 48 89 04 25 10 05 mov %rax,%fs:0x510 +@@ -31689,8 +31722,11 @@ + 41ff77: b8 11 01 00 00 mov $0x111,%eax + 41ff7c: 66 48 0f 6e c7 movq %rdi,%xmm0 + 41ff81: 66 0f 6c c0 punpcklqdq %xmm0,%xmm0 +- 41ff85: 0f 11 83 d8 02 00 00 movups %xmm0,0x2d8(%rbx) +- 41ff8c: 0f 05 syscall + 41ff85: -+ 41ff8a: 90 nop -+ 41ff8b: 90 nop -+ 41ff8c: 90 nop -+ 41ff8d: 90 nop - 41ff8e: 31 d2 xor %edx,%edx - 41ff90: 48 8d 75 ec lea -0x14(%rbp),%rsi - 41ff94: bf 28 00 00 00 mov $0x28,%edi -@@ -31719,8 +31755,9 @@ - 41ffed: 31 d2 xor %edx,%edx - 41ffef: be 20 00 00 00 mov $0x20,%esi - 41fff4: 48 89 df mov %rbx,%rdi -- 41fff7: b8 4e 01 00 00 mov $0x14e,%eax -- 41fffc: 0f 05 syscall ++ 41ff8a: 90 nop ++ 41ff8b: 90 nop ++ 41ff8c: 90 nop ++ 41ff8d: 90 nop + 41ff8e: 31 d2 xor %edx,%edx + 41ff90: 48 8d 75 ec lea -0x14(%rbp),%rsi + 41ff94: bf 28 00 00 00 mov $0x28,%edi +@@ -31716,8 +31752,9 @@ + 41ffed: 31 d2 xor %edx,%edx + 41ffef: be 20 00 00 00 mov $0x20,%esi + 41fff4: 48 89 df mov %rbx,%rdi +- 41fff7: b8 4e 01 00 00 mov $0x14e,%eax +- 41fffc: 0f 05 syscall + 41fff7: -+ 41fffc: 90 nop -+ 41fffd: 90 nop - 41fffe: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 420003: 77 a7 ja 41ffac <__tls_init_tp+0xcc> - 420005: c7 05 11 7b 08 00 20 movl $0x20,0x87b11(%rip) # 4a7b20 <__rseq_size> -@@ -33086,8 +33123,9 @@ - 421339: 0f 84 d9 fe ff ff je 421218 <_dl_cet_open_check+0x158> - 42133f: be 01 00 00 00 mov $0x1,%esi - 421344: bf 02 50 00 00 mov $0x5002,%edi -- 421349: b8 9e 00 00 00 mov $0x9e,%eax -- 42134e: 0f 05 syscall ++ 41fffc: 90 nop ++ 41fffd: 90 nop + 41fffe: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 420003: 77 a7 ja 41ffac <__tls_init_tp+0xcc> + 420005: c7 05 11 7b 08 00 20 movl $0x20,0x87b11(%rip) +@@ -33083,8 +33120,9 @@ + 421339: 0f 84 d9 fe ff ff je 421218 <_dl_cet_open_check+0x158> + 42133f: be 01 00 00 00 mov $0x1,%esi + 421344: bf 02 50 00 00 mov $0x5002,%edi +- 421349: b8 9e 00 00 00 mov $0x9e,%eax +- 42134e: 0f 05 syscall + 421349: -+ 42134e: 90 nop -+ 42134f: 90 nop - 421350: 89 c7 mov %eax,%edi - 421352: 85 c0 test %eax,%eax - 421354: 75 24 jne 42137a <_dl_cet_open_check+0x2ba> -@@ -33117,8 +33155,8 @@ - 42139e: bf 05 50 00 00 mov $0x5005,%edi - 4213a3: 89 d0 mov %edx,%eax - 4213a5: 48 89 e5 mov %rsp,%rbp -- 4213a8: 48 8d 75 f8 lea -0x8(%rbp),%rsi -- 4213ac: 0f 05 syscall ++ 42134e: 90 nop ++ 42134f: 90 nop + 421350: 89 c7 mov %eax,%edi + 421352: 85 c0 test %eax,%eax + 421354: 75 24 jne 42137a <_dl_cet_open_check+0x2ba> +@@ -33114,8 +33152,8 @@ + 42139e: bf 05 50 00 00 mov $0x5005,%edi + 4213a3: 89 d0 mov %edx,%eax + 4213a5: 48 89 e5 mov %rsp,%rbp +- 4213a8: 48 8d 75 f8 lea -0x8(%rbp),%rsi +- 4213ac: 0f 05 syscall + 4213a8: -+ 4213ad: 90 nop - 4213ae: 48 85 c0 test %rax,%rax - 4213b1: 74 15 je 4213c8 <_dl_cet_setup_features+0x38> - 4213b3: 31 c0 xor %eax,%eax -@@ -33141,9 +33179,11 @@ - 4213ec: a8 0c test $0xc,%al - 4213ee: 74 10 je 421400 <_dl_cet_setup_features+0x70> - 4213f0: 48 c7 c6 ff ff ff ff mov $0xffffffffffffffff,%rsi -- 4213f7: bf 03 50 00 00 mov $0x5003,%edi -- 4213fc: 89 d0 mov %edx,%eax -- 4213fe: 0f 05 syscall ++ 4213ad: 90 nop + 4213ae: 48 85 c0 test %rax,%rax + 4213b1: 74 15 je 4213c8 <_dl_cet_setup_features+0x38> + 4213b3: 31 c0 xor %eax,%eax +@@ -33138,9 +33176,11 @@ + 4213ec: a8 0c test $0xc,%al + 4213ee: 74 10 je 421400 <_dl_cet_setup_features+0x70> + 4213f0: 48 c7 c6 ff ff ff ff mov $0xffffffffffffffff,%rsi +- 4213f7: bf 03 50 00 00 mov $0x5003,%edi +- 4213fc: 89 d0 mov %edx,%eax +- 4213fe: 0f 05 syscall + 4213f7: -+ 4213fc: 90 nop -+ 4213fd: 90 nop -+ 4213fe: 90 nop -+ 4213ff: 90 nop - 421400: b8 02 00 00 00 mov $0x2,%eax - 421405: eb ae jmp 4213b5 <_dl_cet_setup_features+0x25> - 421407: 66 0f 1f 84 00 00 00 nopw 0x0(%rax,%rax,1) -@@ -33172,13 +33212,13 @@ - 421446: 66 2e 0f 1f 84 00 00 cs nopw 0x0(%rax,%rax,1) ++ 4213fc: 90 nop ++ 4213fd: 90 nop ++ 4213fe: 90 nop ++ 4213ff: 90 nop + 421400: b8 02 00 00 00 mov $0x2,%eax + 421405: eb ae jmp 4213b5 <_dl_cet_setup_features+0x25> + 421407: 66 0f 1f 84 00 00 00 nopw 0x0(%rax,%rax,1) +@@ -33169,13 +33209,13 @@ + 421446: 66 2e 0f 1f 84 00 00 cs nopw 0x0(%rax,%rax,1) 42144d: 00 00 00 - 421450: be 0c 00 00 00 mov $0xc,%esi -- 421455: 31 ff xor %edi,%edi -- 421457: 89 f0 mov %esi,%eax -- 421459: 0f 05 syscall + 421450: be 0c 00 00 00 mov $0xc,%esi +- 421455: 31 ff xor %edi,%edi +- 421457: 89 f0 mov %esi,%eax +- 421459: 0f 05 syscall + 421455: -+ 42145a: 90 nop - 42145b: 48 89 c2 mov %rax,%rdx -- 42145e: 48 8d 3c 18 lea (%rax,%rbx,1),%rdi -- 421462: 89 f0 mov %esi,%eax -- 421464: 0f 05 syscall ++ 42145a: 90 nop + 42145b: 48 89 c2 mov %rax,%rdx +- 42145e: 48 8d 3c 18 lea (%rax,%rbx,1),%rdi +- 421462: 89 f0 mov %esi,%eax +- 421464: 0f 05 syscall + 42145e: -+ 421463: 90 nop -+ 421464: 90 nop -+ 421465: 90 nop - 421466: 48 39 c2 cmp %rax,%rdx - 421469: 75 cd jne 421438 <_dl_early_allocate+0x28> - 42146b: 45 31 c9 xor %r9d,%r9d -@@ -33187,8 +33227,9 @@ - 421479: 31 ff xor %edi,%edi - 42147b: 41 ba 22 00 00 00 mov $0x22,%r10d - 421481: 48 89 de mov %rbx,%rsi -- 421484: b8 09 00 00 00 mov $0x9,%eax -- 421489: 0f 05 syscall ++ 421463: 90 nop ++ 421464: 90 nop ++ 421465: 90 nop + 421466: 48 39 c2 cmp %rax,%rdx + 421469: 75 cd jne 421438 <_dl_early_allocate+0x28> + 42146b: 45 31 c9 xor %r9d,%r9d +@@ -33184,8 +33224,9 @@ + 421479: 31 ff xor %edi,%edi + 42147b: 41 ba 22 00 00 00 mov $0x22,%r10d + 421481: 48 89 de mov %rbx,%rsi +- 421484: b8 09 00 00 00 mov $0x9,%eax +- 421489: 0f 05 syscall + 421484: -+ 421489: 90 nop -+ 42148a: 90 nop - 42148b: 31 d2 xor %edx,%edx - 42148d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 421493: 48 8b 5d f8 mov -0x8(%rbp),%rbx -@@ -69741,8 +69782,9 @@ - 444c0d: 41 ba 08 00 00 00 mov $0x8,%r10d - 444c13: 4c 89 f2 mov %r14,%rdx - 444c16: 48 8d 35 b3 0a 04 00 lea 0x40ab3(%rip),%rsi # 4856d0 -- 444c1d: b8 0e 00 00 00 mov $0xe,%eax -- 444c22: 0f 05 syscall ++ 421489: 90 nop ++ 42148a: 90 nop + 42148b: 31 d2 xor %edx,%edx + 42148d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 421493: 48 8b 5d f8 mov -0x8(%rbp),%rbx +@@ -69738,8 +69779,9 @@ + 444c0d: 41 ba 08 00 00 00 mov $0x8,%r10d + 444c13: 4c 89 f2 mov %r14,%rdx + 444c16: 48 8d 35 b3 0a 04 00 lea 0x40ab3(%rip),%rsi +- 444c1d: b8 0e 00 00 00 mov $0xe,%eax +- 444c22: 0f 05 syscall + 444c1d: -+ 444c22: 90 nop -+ 444c23: 90 nop - 444c24: 31 c0 xor %eax,%eax - 444c26: 4c 8d a3 04 09 00 00 lea 0x904(%rbx),%r12 - 444c2d: ba 01 00 00 00 mov $0x1,%edx -@@ -69759,8 +69801,9 @@ - 444c5e: 31 d2 xor %edx,%edx - 444c60: 4c 89 f6 mov %r14,%rsi - 444c63: bf 02 00 00 00 mov $0x2,%edi -- 444c68: b8 0e 00 00 00 mov $0xe,%eax -- 444c6d: 0f 05 syscall ++ 444c22: 90 nop ++ 444c23: 90 nop + 444c24: 31 c0 xor %eax,%eax + 444c26: 4c 8d a3 04 09 00 00 lea 0x904(%rbx),%r12 + 444c2d: ba 01 00 00 00 mov $0x1,%edx +@@ -69756,8 +69798,9 @@ + 444c5e: 31 d2 xor %edx,%edx + 444c60: 4c 89 f6 mov %r14,%rsi + 444c63: bf 02 00 00 00 mov $0x2,%edi +- 444c68: b8 0e 00 00 00 mov $0xe,%eax +- 444c6d: 0f 05 syscall + 444c68: -+ 444c6d: 90 nop -+ 444c6e: 90 nop - 444c6f: 48 8b 45 d8 mov -0x28(%rbp),%rax - 444c73: 64 48 2b 04 25 28 00 sub %fs:0x28,%rax ++ 444c6d: 90 nop ++ 444c6e: 90 nop + 444c6f: 48 8b 45 d8 mov -0x28(%rbp),%rax + 444c73: 64 48 2b 04 25 28 00 sub %fs:0x28,%rax 444c7a: 00 00 -@@ -69779,23 +69822,26 @@ - 444ca3: 44 89 ea mov %r13d,%edx - 444ca6: 89 c7 mov %eax,%edi - 444ca8: 89 de mov %ebx,%esi -- 444caa: b8 ea 00 00 00 mov $0xea,%eax -- 444caf: 0f 05 syscall +@@ -69776,23 +69819,26 @@ + 444ca3: 44 89 ea mov %r13d,%edx + 444ca6: 89 c7 mov %eax,%edi + 444ca8: 89 de mov %ebx,%esi +- 444caa: b8 ea 00 00 00 mov $0xea,%eax +- 444caf: 0f 05 syscall + 444caa: -+ 444caf: 90 nop -+ 444cb0: 90 nop - 444cb1: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 444cb6: 76 8f jbe 444c47 <__pthread_kill_internal+0x77> - 444cb8: 89 c3 mov %eax,%ebx - 444cba: f7 db neg %ebx - 444cbc: eb 8b jmp 444c49 <__pthread_kill_internal+0x79> - 444cbe: 66 90 xchg %ax,%ax -- 444cc0: b8 ba 00 00 00 mov $0xba,%eax -- 444cc5: 0f 05 syscall ++ 444caf: 90 nop ++ 444cb0: 90 nop + 444cb1: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 444cb6: 76 8f jbe 444c47 <__pthread_kill_internal+0x77> + 444cb8: 89 c3 mov %eax,%ebx + 444cba: f7 db neg %ebx + 444cbc: eb 8b jmp 444c49 <__pthread_kill_internal+0x79> + 444cbe: 66 90 xchg %ax,%ax +- 444cc0: b8 ba 00 00 00 mov $0xba,%eax +- 444cc5: 0f 05 syscall + 444cc0: -+ 444cc5: 90 nop -+ 444cc6: 90 nop - 444cc7: 89 c3 mov %eax,%ebx - 444cc9: e8 82 6e 01 00 call 45bb50 <__getpid> - 444cce: 44 89 ea mov %r13d,%edx - 444cd1: 89 de mov %ebx,%esi - 444cd3: 89 c7 mov %eax,%edi -- 444cd5: b8 ea 00 00 00 mov $0xea,%eax -- 444cda: 0f 05 syscall ++ 444cc5: 90 nop ++ 444cc6: 90 nop + 444cc7: 89 c3 mov %eax,%ebx + 444cc9: e8 82 6e 01 00 call 45bb50 <__getpid> + 444cce: 44 89 ea mov %r13d,%edx + 444cd1: 89 de mov %ebx,%esi + 444cd3: 89 c7 mov %eax,%edi +- 444cd5: b8 ea 00 00 00 mov $0xea,%eax +- 444cda: 0f 05 syscall + 444cd5: -+ 444cda: 90 nop -+ 444cdb: 90 nop - 444cdc: 89 c3 mov %eax,%ebx - 444cde: f7 db neg %ebx - 444ce0: 3d 00 f0 ff ff cmp $0xfffff000,%eax -@@ -69843,8 +69889,11 @@ - 444d71: 31 ff xor %edi,%edi - 444d73: b8 0e 00 00 00 mov $0xe,%eax - 444d78: 4c 89 fa mov %r15,%rdx -- 444d7b: 48 8d 35 4e 09 04 00 lea 0x4094e(%rip),%rsi # 4856d0 -- 444d82: 0f 05 syscall ++ 444cda: 90 nop ++ 444cdb: 90 nop + 444cdc: 89 c3 mov %eax,%ebx + 444cde: f7 db neg %ebx + 444ce0: 3d 00 f0 ff ff cmp $0xfffff000,%eax +@@ -69840,8 +69886,11 @@ + 444d71: 31 ff xor %edi,%edi + 444d73: b8 0e 00 00 00 mov $0xe,%eax + 444d78: 4c 89 fa mov %r15,%rdx +- 444d7b: 48 8d 35 4e 09 04 00 lea 0x4094e(%rip),%rsi +- 444d82: 0f 05 syscall + 444d7b: -+ 444d80: 90 nop -+ 444d81: 90 nop -+ 444d82: 90 nop -+ 444d83: 90 nop - 444d84: 31 c0 xor %eax,%eax - 444d86: 4c 8d ab 04 09 00 00 lea 0x904(%rbx),%r13 - 444d8d: ba 01 00 00 00 mov $0x1,%edx -@@ -69861,8 +69910,9 @@ - 444dbf: 31 d2 xor %edx,%edx - 444dc1: 4c 89 fe mov %r15,%rsi - 444dc4: bf 02 00 00 00 mov $0x2,%edi -- 444dc9: b8 0e 00 00 00 mov $0xe,%eax -- 444dce: 0f 05 syscall ++ 444d80: 90 nop ++ 444d81: 90 nop ++ 444d82: 90 nop ++ 444d83: 90 nop + 444d84: 31 c0 xor %eax,%eax + 444d86: 4c 8d ab 04 09 00 00 lea 0x904(%rbx),%r13 + 444d8d: ba 01 00 00 00 mov $0x1,%edx +@@ -69858,8 +69907,9 @@ + 444dbf: 31 d2 xor %edx,%edx + 444dc1: 4c 89 fe mov %r15,%rsi + 444dc4: bf 02 00 00 00 mov $0x2,%edi +- 444dc9: b8 0e 00 00 00 mov $0xe,%eax +- 444dce: 0f 05 syscall + 444dc9: -+ 444dce: 90 nop -+ 444dcf: 90 nop - 444dd0: 48 8b 45 c8 mov -0x38(%rbp),%rax - 444dd4: 64 48 2b 04 25 28 00 sub %fs:0x28,%rax ++ 444dce: 90 nop ++ 444dcf: 90 nop + 444dd0: 48 8b 45 c8 mov -0x38(%rbp),%rax + 444dd4: 64 48 2b 04 25 28 00 sub %fs:0x28,%rax 444ddb: 00 00 -@@ -69882,22 +69932,25 @@ - 444e03: 44 89 e2 mov %r12d,%edx - 444e06: 89 c7 mov %eax,%edi - 444e08: 89 de mov %ebx,%esi -- 444e0a: b8 ea 00 00 00 mov $0xea,%eax -- 444e0f: 0f 05 syscall +@@ -69879,22 +69929,25 @@ + 444e03: 44 89 e2 mov %r12d,%edx + 444e06: 89 c7 mov %eax,%edi + 444e08: 89 de mov %ebx,%esi +- 444e0a: b8 ea 00 00 00 mov $0xea,%eax +- 444e0f: 0f 05 syscall + 444e0a: -+ 444e0f: 90 nop -+ 444e10: 90 nop - 444e11: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 444e16: 76 8f jbe 444da7 <__pthread_kill+0x87> - 444e18: 41 89 c6 mov %eax,%r14d - 444e1b: 41 f7 de neg %r14d - 444e1e: eb 8a jmp 444daa <__pthread_kill+0x8a> -- 444e20: b8 ba 00 00 00 mov $0xba,%eax -- 444e25: 0f 05 syscall ++ 444e0f: 90 nop ++ 444e10: 90 nop + 444e11: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 444e16: 76 8f jbe 444da7 <__pthread_kill+0x87> + 444e18: 41 89 c6 mov %eax,%r14d + 444e1b: 41 f7 de neg %r14d + 444e1e: eb 8a jmp 444daa <__pthread_kill+0x8a> +- 444e20: b8 ba 00 00 00 mov $0xba,%eax +- 444e25: 0f 05 syscall + 444e20: -+ 444e25: 90 nop -+ 444e26: 90 nop - 444e27: 89 c3 mov %eax,%ebx - 444e29: e8 22 6d 01 00 call 45bb50 <__getpid> - 444e2e: 44 89 e2 mov %r12d,%edx - 444e31: 89 de mov %ebx,%esi - 444e33: 89 c7 mov %eax,%edi -- 444e35: b8 ea 00 00 00 mov $0xea,%eax -- 444e3a: 0f 05 syscall ++ 444e25: 90 nop ++ 444e26: 90 nop + 444e27: 89 c3 mov %eax,%ebx + 444e29: e8 22 6d 01 00 call 45bb50 <__getpid> + 444e2e: 44 89 e2 mov %r12d,%edx + 444e31: 89 de mov %ebx,%esi + 444e33: 89 c7 mov %eax,%edi +- 444e35: b8 ea 00 00 00 mov $0xea,%eax +- 444e3a: 0f 05 syscall + 444e35: -+ 444e3a: 90 nop -+ 444e3b: 90 nop - 444e3c: 41 89 c6 mov %eax,%r14d - 444e3f: 41 f7 de neg %r14d - 444e42: 3d 00 f0 ff ff cmp $0xfffff000,%eax -@@ -70102,8 +70155,10 @@ - 445101: 48 89 df mov %rbx,%rdi - 445104: 44 89 f0 mov %r14d,%eax - 445107: f7 d6 not %esi -- 445109: 81 e6 80 00 00 00 and $0x80,%esi -- 44510f: 0f 05 syscall ++ 444e3a: 90 nop ++ 444e3b: 90 nop + 444e3c: 41 89 c6 mov %eax,%r14d + 444e3f: 41 f7 de neg %r14d + 444e42: 3d 00 f0 ff ff cmp $0xfffff000,%eax +@@ -70099,8 +70152,10 @@ + 445101: 48 89 df mov %rbx,%rdi + 445104: 44 89 f0 mov %r14d,%eax + 445107: f7 d6 not %esi +- 445109: 81 e6 80 00 00 00 and $0x80,%esi +- 44510f: 0f 05 syscall + 445109: -+ 44510e: 90 nop -+ 44510f: 90 nop -+ 445110: 90 nop - 445111: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 445117: 76 b7 jbe 4450d0 <__pthread_mutex_lock_full+0x1a0> - 445119: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -70220,8 +70275,9 @@ - 4452df: 45 31 d2 xor %r10d,%r10d - 4452e2: 31 f6 xor %esi,%esi - 4452e4: 48 89 df mov %rbx,%rdi -- 4452e7: b8 ca 00 00 00 mov $0xca,%eax -- 4452ec: 0f 05 syscall ++ 44510e: 90 nop ++ 44510f: 90 nop ++ 445110: 90 nop + 445111: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 445117: 76 b7 jbe 4450d0 <__pthread_mutex_lock_full+0x1a0> + 445119: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -70217,8 +70272,9 @@ + 4452df: 45 31 d2 xor %r10d,%r10d + 4452e2: 31 f6 xor %esi,%esi + 4452e4: 48 89 df mov %rbx,%rdi +- 4452e7: b8 ca 00 00 00 mov $0xca,%eax +- 4452ec: 0f 05 syscall + 4452e7: -+ 4452ec: 90 nop -+ 4452ed: 90 nop - 4452ee: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 4452f4: 0f 87 4e 02 00 00 ja 445548 <__pthread_mutex_lock_full+0x618> - 4452fa: 8b 13 mov (%rbx),%edx -@@ -70339,8 +70395,9 @@ - 4454fa: 31 d2 xor %edx,%edx - 4454fc: 48 89 df mov %rbx,%rdi - 4454ff: be 07 00 00 00 mov $0x7,%esi -- 445504: b8 ca 00 00 00 mov $0xca,%eax -- 445509: 0f 05 syscall ++ 4452ec: 90 nop ++ 4452ed: 90 nop + 4452ee: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 4452f4: 0f 87 4e 02 00 00 ja 445548 <__pthread_mutex_lock_full+0x618> + 4452fa: 8b 13 mov (%rbx),%edx +@@ -70336,8 +70392,9 @@ + 4454fa: 31 d2 xor %edx,%edx + 4454fc: 48 89 df mov %rbx,%rdi + 4454ff: be 07 00 00 00 mov $0x7,%esi +- 445504: b8 ca 00 00 00 mov $0xca,%eax +- 445509: 0f 05 syscall + 445504: -+ 445509: 90 nop -+ 44550a: 90 nop - 44550b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 445511: 0f 86 71 ff ff ff jbe 445488 <__pthread_mutex_lock_full+0x558> - 445517: 83 f8 92 cmp $0xffffff92,%eax -@@ -70720,8 +70777,8 @@ - 445aa1: 4c 89 c7 mov %r8,%rdi - 445aa4: b8 ca 00 00 00 mov $0xca,%eax - 445aa9: 81 e6 80 00 00 00 and $0x80,%esi -- 445aaf: 40 80 f6 81 xor $0x81,%sil -- 445ab3: 0f 05 syscall ++ 445509: 90 nop ++ 44550a: 90 nop + 44550b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 445511: 0f 86 71 ff ff ff jbe 445488 <__pthread_mutex_lock_full+0x558> + 445517: 83 f8 92 cmp $0xffffff92,%eax +@@ -70717,8 +70774,8 @@ + 445aa1: 4c 89 c7 mov %r8,%rdi + 445aa4: b8 ca 00 00 00 mov $0xca,%eax + 445aa9: 81 e6 80 00 00 00 and $0x80,%esi +- 445aaf: 40 80 f6 81 xor $0x81,%sil +- 445ab3: 0f 05 syscall + 445aaf: -+ 445ab4: 90 nop - 445ab5: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 445abb: 0f 87 0e 02 00 00 ja 445ccf <__pthread_mutex_unlock_full+0x3bf> - 445ac1: 90 nop -@@ -70863,8 +70920,9 @@ - 445cf3: ba 01 00 00 00 mov $0x1,%edx - 445cf8: be 01 00 00 00 mov $0x1,%esi - 445cfd: 4c 89 c7 mov %r8,%rdi -- 445d00: b8 ca 00 00 00 mov $0xca,%eax -- 445d05: 0f 05 syscall ++ 445ab4: 90 nop + 445ab5: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 445abb: 0f 87 0e 02 00 00 ja 445ccf <__pthread_mutex_unlock_full+0x3bf> + 445ac1: 90 nop +@@ -70860,8 +70917,9 @@ + 445cf3: ba 01 00 00 00 mov $0x1,%edx + 445cf8: be 01 00 00 00 mov $0x1,%esi + 445cfd: 4c 89 c7 mov %r8,%rdi +- 445d00: b8 ca 00 00 00 mov $0xca,%eax +- 445d05: 0f 05 syscall + 445d00: -+ 445d05: 90 nop -+ 445d06: 90 nop - 445d07: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 445d0d: 0f 86 36 fd ff ff jbe 445a49 <__pthread_mutex_unlock_full+0x139> - 445d13: 83 c0 16 add $0x16,%eax -@@ -70875,8 +70933,9 @@ - 445d24: 45 31 d2 xor %r10d,%r10d - 445d27: 31 d2 xor %edx,%edx - 445d29: 4c 89 c7 mov %r8,%rdi -- 445d2c: b8 ca 00 00 00 mov $0xca,%eax -- 445d31: 0f 05 syscall ++ 445d05: 90 nop ++ 445d06: 90 nop + 445d07: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 445d0d: 0f 86 36 fd ff ff jbe 445a49 <__pthread_mutex_unlock_full+0x139> + 445d13: 83 c0 16 add $0x16,%eax +@@ -70872,8 +70930,9 @@ + 445d24: 45 31 d2 xor %r10d,%r10d + 445d27: 31 d2 xor %edx,%edx + 445d29: 4c 89 c7 mov %r8,%rdi +- 445d2c: b8 ca 00 00 00 mov $0xca,%eax +- 445d31: 0f 05 syscall + 445d2c: -+ 445d31: 90 nop -+ 445d32: 90 nop - 445d33: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 445d39: 0f 86 f8 fd ff ff jbe 445b37 <__pthread_mutex_unlock_full+0x227> - 445d3f: 83 f8 92 cmp $0xffffff92,%eax -@@ -71093,8 +71152,9 @@ - 446007: 45 31 d2 xor %r10d,%r10d - 44600a: be 80 00 00 00 mov $0x80,%esi - 44600f: 48 89 df mov %rbx,%rdi -- 446012: b8 ca 00 00 00 mov $0xca,%eax -- 446017: 0f 05 syscall ++ 445d31: 90 nop ++ 445d32: 90 nop + 445d33: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 445d39: 0f 86 f8 fd ff ff jbe 445b37 <__pthread_mutex_unlock_full+0x227> + 445d3f: 83 f8 92 cmp $0xffffff92,%eax +@@ -71090,8 +71149,9 @@ + 446007: 45 31 d2 xor %r10d,%r10d + 44600a: be 80 00 00 00 mov $0x80,%esi + 44600f: 48 89 df mov %rbx,%rdi +- 446012: b8 ca 00 00 00 mov $0xca,%eax +- 446017: 0f 05 syscall + 446012: -+ 446017: 90 nop -+ 446018: 90 nop - 446019: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 44601f: 76 a1 jbe 445fc2 <__pthread_once_slow+0x22> - 446021: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -71130,8 +71190,9 @@ - 4460a5: be 81 00 00 00 mov $0x81,%esi - 4460aa: c7 03 02 00 00 00 movl $0x2,(%rbx) - 4460b0: 48 89 df mov %rbx,%rdi -- 4460b3: b8 ca 00 00 00 mov $0xca,%eax -- 4460b8: 0f 05 syscall ++ 446017: 90 nop ++ 446018: 90 nop + 446019: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 44601f: 76 a1 jbe 445fc2 <__pthread_once_slow+0x22> + 446021: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -71127,8 +71187,9 @@ + 4460a5: be 81 00 00 00 mov $0x81,%esi + 4460aa: c7 03 02 00 00 00 movl $0x2,(%rbx) + 4460b0: 48 89 df mov %rbx,%rdi +- 4460b3: b8 ca 00 00 00 mov $0xca,%eax +- 4460b8: 0f 05 syscall + 4460b3: -+ 4460b8: 90 nop -+ 4460b9: 90 nop - 4460ba: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 4460c0: 0f 86 02 ff ff ff jbe 445fc8 <__pthread_once_slow+0x28> - 4460c6: 83 c0 16 add $0x16,%eax -@@ -71173,8 +71234,9 @@ - 44613a: 45 31 d2 xor %r10d,%r10d - 44613d: ba ff ff ff 7f mov $0x7fffffff,%edx - 446142: be 81 00 00 00 mov $0x81,%esi -- 446147: b8 ca 00 00 00 mov $0xca,%eax -- 44614c: 0f 05 syscall ++ 4460b8: 90 nop ++ 4460b9: 90 nop + 4460ba: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 4460c0: 0f 86 02 ff ff ff jbe 445fc8 <__pthread_once_slow+0x28> + 4460c6: 83 c0 16 add $0x16,%eax +@@ -71170,8 +71231,9 @@ + 44613a: 45 31 d2 xor %r10d,%r10d + 44613d: ba ff ff ff 7f mov $0x7fffffff,%edx + 446142: be 81 00 00 00 mov $0x81,%esi +- 446147: b8 ca 00 00 00 mov $0xca,%eax +- 44614c: 0f 05 syscall + 446147: -+ 44614c: 90 nop -+ 44614d: 90 nop - 44614e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446154: 77 0a ja 446160 - 446156: c3 ret -@@ -71316,8 +71378,8 @@ - 4462dc: 40 0f 95 c6 setne %sil - 4462e0: 45 31 d2 xor %r10d,%r10d - 4462e3: c1 e6 07 shl $0x7,%esi -- 4462e6: 40 80 f6 81 xor $0x81,%sil -- 4462ea: 0f 05 syscall ++ 44614c: 90 nop ++ 44614d: 90 nop + 44614e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446154: 77 0a ja 446160 + 446156: c3 ret +@@ -71313,8 +71375,8 @@ + 4462dc: 40 0f 95 c6 setne %sil + 4462e0: 45 31 d2 xor %r10d,%r10d + 4462e3: c1 e6 07 shl $0x7,%esi +- 4462e6: 40 80 f6 81 xor $0x81,%sil +- 4462ea: 0f 05 syscall + 4462e6: -+ 4462eb: 90 nop - 4462ec: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 4462f2: 0f 86 2e ff ff ff jbe 446226 <___pthread_rwlock_rdlock+0x46> - 4462f8: 83 c0 16 add $0x16,%eax -@@ -71420,8 +71482,9 @@ - 44642f: ba ff ff ff 7f mov $0x7fffffff,%edx - 446434: 4c 89 c7 mov %r8,%rdi - 446437: 40 80 f6 81 xor $0x81,%sil -- 44643b: b8 ca 00 00 00 mov $0xca,%eax -- 446440: 0f 05 syscall ++ 4462eb: 90 nop + 4462ec: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 4462f2: 0f 86 2e ff ff ff jbe 446226 <___pthread_rwlock_rdlock+0x46> + 4462f8: 83 c0 16 add $0x16,%eax +@@ -71417,8 +71479,9 @@ + 44642f: ba ff ff ff 7f mov $0x7fffffff,%edx + 446434: 4c 89 c7 mov %r8,%rdi + 446437: 40 80 f6 81 xor $0x81,%sil +- 44643b: b8 ca 00 00 00 mov $0xca,%eax +- 446440: 0f 05 syscall + 44643b: -+ 446440: 90 nop -+ 446441: 90 nop - 446442: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446448: 0f 87 da 00 00 00 ja 446528 <___pthread_rwlock_unlock+0x158> - 44644e: 5b pop %rbx -@@ -71446,8 +71509,8 @@ - 446482: 45 31 d2 xor %r10d,%r10d - 446485: ba ff ff ff 7f mov $0x7fffffff,%edx - 44648a: b8 ca 00 00 00 mov $0xca,%eax -- 44648f: 40 80 f6 81 xor $0x81,%sil -- 446493: 0f 05 syscall ++ 446440: 90 nop ++ 446441: 90 nop + 446442: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446448: 0f 87 da 00 00 00 ja 446528 <___pthread_rwlock_unlock+0x158> + 44644e: 5b pop %rbx +@@ -71443,8 +71506,8 @@ + 446482: 45 31 d2 xor %r10d,%r10d + 446485: ba ff ff ff 7f mov $0x7fffffff,%edx + 44648a: b8 ca 00 00 00 mov $0xca,%eax +- 44648f: 40 80 f6 81 xor $0x81,%sil +- 446493: 0f 05 syscall + 44648f: -+ 446494: 90 nop - 446495: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 44649b: 76 83 jbe 446420 <___pthread_rwlock_unlock+0x50> - 44649d: 83 c0 16 add $0x16,%eax -@@ -71487,8 +71550,9 @@ - 446509: ba 01 00 00 00 mov $0x1,%edx - 44650e: 48 89 df mov %rbx,%rdi - 446511: 40 80 f6 81 xor $0x81,%sil -- 446515: b8 ca 00 00 00 mov $0xca,%eax -- 44651a: 0f 05 syscall ++ 446494: 90 nop + 446495: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 44649b: 76 83 jbe 446420 <___pthread_rwlock_unlock+0x50> + 44649d: 83 c0 16 add $0x16,%eax +@@ -71484,8 +71547,9 @@ + 446509: ba 01 00 00 00 mov $0x1,%edx + 44650e: 48 89 df mov %rbx,%rdi + 446511: 40 80 f6 81 xor $0x81,%sil +- 446515: b8 ca 00 00 00 mov $0xca,%eax +- 44651a: 0f 05 syscall + 446515: -+ 44651a: 90 nop -+ 44651b: 90 nop - 44651c: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446522: 0f 86 26 ff ff ff jbe 44644e <___pthread_rwlock_unlock+0x7e> - 446528: 83 c0 16 add $0x16,%eax -@@ -71515,8 +71579,8 @@ - 44656f: 45 31 d2 xor %r10d,%r10d - 446572: ba ff ff ff 7f mov $0x7fffffff,%edx - 446577: b8 ca 00 00 00 mov $0xca,%eax -- 44657c: 40 80 f6 81 xor $0x81,%sil -- 446580: 0f 05 syscall ++ 44651a: 90 nop ++ 44651b: 90 nop + 44651c: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446522: 0f 86 26 ff ff ff jbe 44644e <___pthread_rwlock_unlock+0x7e> + 446528: 83 c0 16 add $0x16,%eax +@@ -71512,8 +71576,8 @@ + 44656f: 45 31 d2 xor %r10d,%r10d + 446572: ba ff ff ff 7f mov $0x7fffffff,%edx + 446577: b8 ca 00 00 00 mov $0xca,%eax +- 44657c: 40 80 f6 81 xor $0x81,%sil +- 446580: 0f 05 syscall + 44657c: -+ 446581: 90 nop - 446582: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446588: 0f 86 6c ff ff ff jbe 4464fa <___pthread_rwlock_unlock+0x12a> - 44658e: 83 c0 16 add $0x16,%eax -@@ -71736,8 +71800,9 @@ - 44684d: ba 01 00 00 00 mov $0x1,%edx - 446852: 4c 89 e7 mov %r12,%rdi - 446855: 40 80 f6 81 xor $0x81,%sil -- 446859: b8 ca 00 00 00 mov $0xca,%eax -- 44685e: 0f 05 syscall ++ 446581: 90 nop + 446582: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446588: 0f 86 6c ff ff ff jbe 4464fa <___pthread_rwlock_unlock+0x12a> + 44658e: 83 c0 16 add $0x16,%eax +@@ -71733,8 +71797,9 @@ + 44684d: ba 01 00 00 00 mov $0x1,%edx + 446852: 4c 89 e7 mov %r12,%rdi + 446855: 40 80 f6 81 xor $0x81,%sil +- 446859: b8 ca 00 00 00 mov $0xca,%eax +- 44685e: 0f 05 syscall + 446859: -+ 44685e: 90 nop -+ 44685f: 90 nop - 446860: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446866: 0f 87 f9 00 00 00 ja 446965 <___pthread_rwlock_wrlock+0x3c5> - 44686c: 41 83 e0 04 and $0x4,%r8d -@@ -71747,8 +71812,9 @@ - 446878: ba ff ff ff 7f mov $0x7fffffff,%edx - 44687d: 48 89 df mov %rbx,%rdi - 446880: 40 80 f6 81 xor $0x81,%sil -- 446884: b8 ca 00 00 00 mov $0xca,%eax -- 446889: 0f 05 syscall ++ 44685e: 90 nop ++ 44685f: 90 nop + 446860: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446866: 0f 87 f9 00 00 00 ja 446965 <___pthread_rwlock_wrlock+0x3c5> + 44686c: 41 83 e0 04 and $0x4,%r8d +@@ -71744,8 +71809,9 @@ + 446878: ba ff ff ff 7f mov $0x7fffffff,%edx + 44687d: 48 89 df mov %rbx,%rdi + 446880: 40 80 f6 81 xor $0x81,%sil +- 446884: b8 ca 00 00 00 mov $0xca,%eax +- 446889: 0f 05 syscall + 446884: -+ 446889: 90 nop -+ 44688a: 90 nop - 44688b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446891: 0f 87 c1 00 00 00 ja 446958 <___pthread_rwlock_wrlock+0x3b8> - 446897: 41 b8 6e 00 00 00 mov $0x6e,%r8d -@@ -71786,8 +71852,9 @@ - 44691c: ba 01 00 00 00 mov $0x1,%edx - 446921: 4c 89 e7 mov %r12,%rdi - 446924: 40 80 f6 81 xor $0x81,%sil -- 446928: b8 ca 00 00 00 mov $0xca,%eax -- 44692d: 0f 05 syscall ++ 446889: 90 nop ++ 44688a: 90 nop + 44688b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446891: 0f 87 c1 00 00 00 ja 446958 <___pthread_rwlock_wrlock+0x3b8> + 446897: 41 b8 6e 00 00 00 mov $0x6e,%r8d +@@ -71783,8 +71849,9 @@ + 44691c: ba 01 00 00 00 mov $0x1,%edx + 446921: 4c 89 e7 mov %r12,%rdi + 446924: 40 80 f6 81 xor $0x81,%sil +- 446928: b8 ca 00 00 00 mov $0xca,%eax +- 44692d: 0f 05 syscall + 446928: -+ 44692d: 90 nop -+ 44692e: 90 nop - 44692f: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 446935: 0f 86 e8 fc ff ff jbe 446623 <___pthread_rwlock_wrlock+0x83> - 44693b: 83 c0 16 add $0x16,%eax -@@ -71852,8 +71919,9 @@ - 446a05: 48 89 f0 mov %rsi,%rax - 446a08: 48 89 c6 mov %rax,%rsi - 446a0b: 41 ba 08 00 00 00 mov $0x8,%r10d -- 446a11: b8 0e 00 00 00 mov $0xe,%eax -- 446a16: 0f 05 syscall ++ 44692d: 90 nop ++ 44692e: 90 nop + 44692f: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 446935: 0f 86 e8 fc ff ff jbe 446623 <___pthread_rwlock_wrlock+0x83> + 44693b: 83 c0 16 add $0x16,%eax +@@ -71849,8 +71916,9 @@ + 446a05: 48 89 f0 mov %rsi,%rax + 446a08: 48 89 c6 mov %rax,%rsi + 446a0b: 41 ba 08 00 00 00 mov $0x8,%r10d +- 446a11: b8 0e 00 00 00 mov $0xe,%eax +- 446a16: 0f 05 syscall + 446a11: -+ 446a16: 90 nop -+ 446a17: 90 nop - 446a18: 89 c2 mov %eax,%edx - 446a1a: f7 da neg %edx - 446a1c: 3d 00 f0 ff ff cmp $0xfffff000,%eax -@@ -93239,8 +93307,9 @@ - 45ba24: b8 ff ff ff 7f mov $0x7fffffff,%eax - 45ba29: 48 39 c2 cmp %rax,%rdx - 45ba2c: 48 0f 47 d0 cmova %rax,%rdx -- 45ba30: b8 d9 00 00 00 mov $0xd9,%eax -- 45ba35: 0f 05 syscall ++ 446a16: 90 nop ++ 446a17: 90 nop + 446a18: 89 c2 mov %eax,%edx + 446a1a: f7 da neg %edx + 446a1c: 3d 00 f0 ff ff cmp $0xfffff000,%eax +@@ -93236,8 +93304,9 @@ + 45ba24: b8 ff ff ff 7f mov $0x7fffffff,%eax + 45ba29: 48 39 c2 cmp %rax,%rdx + 45ba2c: 48 0f 47 d0 cmova %rax,%rdx +- 45ba30: b8 d9 00 00 00 mov $0xd9,%eax +- 45ba35: 0f 05 syscall + 45ba30: -+ 45ba35: 90 nop -+ 45ba36: 90 nop - 45ba37: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45ba3d: 77 01 ja 45ba40 <__getdents+0x20> - 45ba3f: c3 ret -@@ -93332,8 +93401,9 @@ ++ 45ba35: 90 nop ++ 45ba36: 90 nop + 45ba37: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45ba3d: 77 01 ja 45ba40 <__getdents+0x20> + 45ba3f: c3 ret +@@ -93329,8 +93398,9 @@ 000000000045bb50 <__getpid>: - 45bb50: f3 0f 1e fa endbr64 -- 45bb54: b8 27 00 00 00 mov $0x27,%eax -- 45bb59: 0f 05 syscall + 45bb50: f3 0f 1e fa endbr64 +- 45bb54: b8 27 00 00 00 mov $0x27,%eax +- 45bb59: 0f 05 syscall + 45bb54: -+ 45bb59: 90 nop -+ 45bb5a: 90 nop - 45bb5b: c3 ret - 45bb5c: 0f 1f 40 00 nopl 0x0(%rax) ++ 45bb59: 90 nop ++ 45bb5a: 90 nop + 45bb5b: c3 ret + 45bb5c: 0f 1f 40 00 nopl 0x0(%rax) -@@ -93365,8 +93435,9 @@ +@@ -93362,8 +93432,9 @@ 000000000045bba0 <__sched_getparam>: - 45bba0: f3 0f 1e fa endbr64 -- 45bba4: b8 8f 00 00 00 mov $0x8f,%eax -- 45bba9: 0f 05 syscall + 45bba0: f3 0f 1e fa endbr64 +- 45bba4: b8 8f 00 00 00 mov $0x8f,%eax +- 45bba9: 0f 05 syscall + 45bba4: -+ 45bba9: 90 nop -+ 45bbaa: 90 nop - 45bbab: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 45bbb1: 73 01 jae 45bbb4 <__sched_getparam+0x14> - 45bbb3: c3 ret -@@ -93381,8 +93452,9 @@ ++ 45bba9: 90 nop ++ 45bbaa: 90 nop + 45bbab: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 45bbb1: 73 01 jae 45bbb4 <__sched_getparam+0x14> + 45bbb3: c3 ret +@@ -93378,8 +93449,9 @@ 000000000045bbd0 <__sched_getscheduler>: - 45bbd0: f3 0f 1e fa endbr64 -- 45bbd4: b8 91 00 00 00 mov $0x91,%eax -- 45bbd9: 0f 05 syscall + 45bbd0: f3 0f 1e fa endbr64 +- 45bbd4: b8 91 00 00 00 mov $0x91,%eax +- 45bbd9: 0f 05 syscall + 45bbd4: -+ 45bbd9: 90 nop -+ 45bbda: 90 nop - 45bbdb: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 45bbe1: 73 01 jae 45bbe4 <__sched_getscheduler+0x14> - 45bbe3: c3 ret -@@ -93397,8 +93469,9 @@ ++ 45bbd9: 90 nop ++ 45bbda: 90 nop + 45bbdb: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 45bbe1: 73 01 jae 45bbe4 <__sched_getscheduler+0x14> + 45bbe3: c3 ret +@@ -93394,8 +93466,9 @@ 000000000045bc00 <__sched_get_priority_max>: - 45bc00: f3 0f 1e fa endbr64 -- 45bc04: b8 92 00 00 00 mov $0x92,%eax -- 45bc09: 0f 05 syscall + 45bc00: f3 0f 1e fa endbr64 +- 45bc04: b8 92 00 00 00 mov $0x92,%eax +- 45bc09: 0f 05 syscall + 45bc04: -+ 45bc09: 90 nop -+ 45bc0a: 90 nop - 45bc0b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 45bc11: 73 01 jae 45bc14 <__sched_get_priority_max+0x14> - 45bc13: c3 ret -@@ -93413,8 +93486,9 @@ ++ 45bc09: 90 nop ++ 45bc0a: 90 nop + 45bc0b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 45bc11: 73 01 jae 45bc14 <__sched_get_priority_max+0x14> + 45bc13: c3 ret +@@ -93410,8 +93483,9 @@ 000000000045bc30 <__sched_get_priority_min>: - 45bc30: f3 0f 1e fa endbr64 -- 45bc34: b8 93 00 00 00 mov $0x93,%eax -- 45bc39: 0f 05 syscall + 45bc30: f3 0f 1e fa endbr64 +- 45bc34: b8 93 00 00 00 mov $0x93,%eax +- 45bc39: 0f 05 syscall + 45bc34: -+ 45bc39: 90 nop -+ 45bc3a: 90 nop - 45bc3b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 45bc41: 73 01 jae 45bc44 <__sched_get_priority_min+0x14> - 45bc43: c3 ret -@@ -93429,8 +93503,9 @@ ++ 45bc39: 90 nop ++ 45bc3a: 90 nop + 45bc3b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 45bc41: 73 01 jae 45bc44 <__sched_get_priority_min+0x14> + 45bc43: c3 ret +@@ -93426,8 +93500,9 @@ 000000000045bc60 <__sched_setscheduler>: - 45bc60: f3 0f 1e fa endbr64 -- 45bc64: b8 90 00 00 00 mov $0x90,%eax -- 45bc69: 0f 05 syscall + 45bc60: f3 0f 1e fa endbr64 +- 45bc64: b8 90 00 00 00 mov $0x90,%eax +- 45bc69: 0f 05 syscall + 45bc64: -+ 45bc69: 90 nop -+ 45bc6a: 90 nop - 45bc6b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax - 45bc71: 73 01 jae 45bc74 <__sched_setscheduler+0x14> - 45bc73: c3 ret -@@ -93477,8 +93552,9 @@ - 45bd00: 48 89 85 08 ff ff ff mov %rax,-0xf8(%rbp) - 45bd07: 0f 84 21 04 00 00 je 45c12e <__getcwd+0x49e> - 45bd0d: 48 8b bd 08 ff ff ff mov -0xf8(%rbp),%rdi -- 45bd14: b8 4f 00 00 00 mov $0x4f,%eax -- 45bd19: 0f 05 syscall ++ 45bc69: 90 nop ++ 45bc6a: 90 nop + 45bc6b: 48 3d 01 f0 ff ff cmp $0xfffffffffffff001,%rax + 45bc71: 73 01 jae 45bc74 <__sched_setscheduler+0x14> + 45bc73: c3 ret +@@ -93474,8 +93549,9 @@ + 45bd00: 48 89 85 08 ff ff ff mov %rax,-0xf8(%rbp) + 45bd07: 0f 84 21 04 00 00 je 45c12e <__getcwd+0x49e> + 45bd0d: 48 8b bd 08 ff ff ff mov -0xf8(%rbp),%rdi +- 45bd14: b8 4f 00 00 00 mov $0x4f,%eax +- 45bd19: 0f 05 syscall + 45bd14: -+ 45bd19: 90 nop -+ 45bd1a: 90 nop - 45bd1b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45bd21: 0f 87 85 05 00 00 ja 45c2ac <__getcwd+0x61c> - 45bd27: 85 c0 test %eax,%eax -@@ -93914,8 +93990,9 @@ ++ 45bd19: 90 nop ++ 45bd1a: 90 nop + 45bd1b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45bd21: 0f 87 85 05 00 00 ja 45c2ac <__getcwd+0x61c> + 45bd27: 85 c0 test %eax,%eax +@@ -93911,8 +93987,9 @@ 000000000045c510 <__libc_lseek>: - 45c510: f3 0f 1e fa endbr64 -- 45c514: b8 08 00 00 00 mov $0x8,%eax -- 45c519: 0f 05 syscall + 45c510: f3 0f 1e fa endbr64 +- 45c514: b8 08 00 00 00 mov $0x8,%eax +- 45c519: 0f 05 syscall + 45c514: -+ 45c519: 90 nop -+ 45c51a: 90 nop - 45c51b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c521: 77 05 ja 45c528 <__libc_lseek+0x18> - 45c523: c3 ret -@@ -93962,8 +94039,9 @@ - 45c5a4: 89 da mov %ebx,%edx - 45c5a6: 4c 89 e6 mov %r12,%rsi - 45c5a9: bf 9c ff ff ff mov $0xffffff9c,%edi -- 45c5ae: b8 01 01 00 00 mov $0x101,%eax -- 45c5b3: 0f 05 syscall ++ 45c519: 90 nop ++ 45c51a: 90 nop + 45c51b: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c521: 77 05 ja 45c528 <__libc_lseek+0x18> + 45c523: c3 ret +@@ -93959,8 +94036,9 @@ + 45c5a4: 89 da mov %ebx,%edx + 45c5a6: 4c 89 e6 mov %r12,%rsi + 45c5a9: bf 9c ff ff ff mov $0xffffff9c,%edi +- 45c5ae: b8 01 01 00 00 mov $0x101,%eax +- 45c5b3: 0f 05 syscall + 45c5ae: -+ 45c5b3: 90 nop -+ 45c5b4: 90 nop - 45c5b5: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c5bb: 0f 87 7f 00 00 00 ja 45c640 <__libc_open+0xe0> - 45c5c1: 48 8b 55 b8 mov -0x48(%rbp),%rdx -@@ -93991,8 +94069,9 @@ - 45c613: 4c 89 e6 mov %r12,%rsi - 45c616: 41 89 c0 mov %eax,%r8d - 45c619: bf 9c ff ff ff mov $0xffffff9c,%edi -- 45c61e: b8 01 01 00 00 mov $0x101,%eax -- 45c623: 0f 05 syscall ++ 45c5b3: 90 nop ++ 45c5b4: 90 nop + 45c5b5: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c5bb: 0f 87 7f 00 00 00 ja 45c640 <__libc_open+0xe0> + 45c5c1: 48 8b 55 b8 mov -0x48(%rbp),%rdx +@@ -93988,8 +94066,9 @@ + 45c613: 4c 89 e6 mov %r12,%rsi + 45c616: 41 89 c0 mov %eax,%r8d + 45c619: bf 9c ff ff ff mov $0xffffff9c,%edi +- 45c61e: b8 01 01 00 00 mov $0x101,%eax +- 45c623: 0f 05 syscall + 45c61e: -+ 45c623: 90 nop -+ 45c624: 90 nop - 45c625: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c62b: 77 33 ja 45c660 <__libc_open+0x100> - 45c62d: 44 89 c7 mov %r8d,%edi -@@ -94036,8 +94115,9 @@ - 45c6b0: 74 36 je 45c6e8 <__libc_openat64+0x68> - 45c6b2: 80 3d df e3 04 00 00 cmpb $0x0,0x4e3df(%rip) # 4aaa98 <__libc_single_threaded> - 45c6b9: 74 51 je 45c70c <__libc_openat64+0x8c> -- 45c6bb: b8 01 01 00 00 mov $0x101,%eax -- 45c6c0: 0f 05 syscall ++ 45c623: 90 nop ++ 45c624: 90 nop + 45c625: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c62b: 77 33 ja 45c660 <__libc_open+0x100> + 45c62d: 44 89 c7 mov %r8d,%edi +@@ -94033,8 +94112,9 @@ + 45c6b0: 74 36 je 45c6e8 <__libc_openat64+0x68> + 45c6b2: 80 3d df e3 04 00 00 cmpb $0x0,0x4e3df(%rip) + 45c6b9: 74 51 je 45c70c <__libc_openat64+0x8c> +- 45c6bb: b8 01 01 00 00 mov $0x101,%eax +- 45c6c0: 0f 05 syscall + 45c6bb: -+ 45c6c0: 90 nop -+ 45c6c1: 90 nop - 45c6c2: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c6c8: 0f 87 8a 00 00 00 ja 45c758 <__libc_openat64+0xd8> - 45c6ce: 48 8b 55 c8 mov -0x38(%rbp),%rdx -@@ -94065,8 +94145,9 @@ - 45c726: 41 89 c0 mov %eax,%r8d - 45c729: 48 8b 75 a0 mov -0x60(%rbp),%rsi - 45c72d: 8b 7d a8 mov -0x58(%rbp),%edi -- 45c730: b8 01 01 00 00 mov $0x101,%eax -- 45c735: 0f 05 syscall ++ 45c6c0: 90 nop ++ 45c6c1: 90 nop + 45c6c2: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c6c8: 0f 87 8a 00 00 00 ja 45c758 <__libc_openat64+0xd8> + 45c6ce: 48 8b 55 c8 mov -0x38(%rbp),%rdx +@@ -94062,8 +94142,9 @@ + 45c726: 41 89 c0 mov %eax,%r8d + 45c729: 48 8b 75 a0 mov -0x60(%rbp),%rsi + 45c72d: 8b 7d a8 mov -0x58(%rbp),%edi +- 45c730: b8 01 01 00 00 mov $0x101,%eax +- 45c735: 0f 05 syscall + 45c730: -+ 45c735: 90 nop -+ 45c736: 90 nop - 45c737: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c73d: 77 31 ja 45c770 <__libc_openat64+0xf0> - 45c73f: 44 89 c7 mov %r8d,%edi -@@ -94095,8 +94176,10 @@ - 45c794: 80 3d fd e2 04 00 00 cmpb $0x0,0x4e2fd(%rip) # 4aaa98 <__libc_single_threaded> - 45c79b: 74 13 je 45c7b0 <__libc_read+0x20> - 45c79d: 31 c0 xor %eax,%eax -- 45c79f: 0f 05 syscall -- 45c7a1: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax ++ 45c735: 90 nop ++ 45c736: 90 nop + 45c737: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c73d: 77 31 ja 45c770 <__libc_openat64+0xf0> + 45c73f: 44 89 c7 mov %r8d,%edi +@@ -94092,8 +94173,10 @@ + 45c794: 80 3d fd e2 04 00 00 cmpb $0x0,0x4e2fd(%rip) + 45c79b: 74 13 je 45c7b0 <__libc_read+0x20> + 45c79d: 31 c0 xor %eax,%eax +- 45c79f: 0f 05 syscall +- 45c7a1: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c79f: -+ 45c7a4: 90 nop -+ 45c7a5: 90 nop -+ 45c7a6: 90 nop - 45c7a7: 77 4f ja 45c7f8 <__libc_read+0x68> - 45c7a9: c3 ret - 45c7aa: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) -@@ -94110,9 +94193,9 @@ - 45c7c8: 48 8b 55 e8 mov -0x18(%rbp),%rdx - 45c7cc: 48 8b 75 f0 mov -0x10(%rbp),%rsi - 45c7d0: 41 89 c0 mov %eax,%r8d -- 45c7d3: 8b 7d f8 mov -0x8(%rbp),%edi -- 45c7d6: 31 c0 xor %eax,%eax -- 45c7d8: 0f 05 syscall ++ 45c7a4: 90 nop ++ 45c7a5: 90 nop ++ 45c7a6: 90 nop + 45c7a7: 77 4f ja 45c7f8 <__libc_read+0x68> + 45c7a9: c3 ret + 45c7aa: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) +@@ -94107,9 +94190,9 @@ + 45c7c8: 48 8b 55 e8 mov -0x18(%rbp),%rdx + 45c7cc: 48 8b 75 f0 mov -0x10(%rbp),%rsi + 45c7d0: 41 89 c0 mov %eax,%r8d +- 45c7d3: 8b 7d f8 mov -0x8(%rbp),%edi +- 45c7d6: 31 c0 xor %eax,%eax +- 45c7d8: 0f 05 syscall + 45c7d3: -+ 45c7d8: 90 nop -+ 45c7d9: 90 nop - 45c7da: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c7e0: 77 2e ja 45c810 <__libc_read+0x80> - 45c7e2: 44 89 c7 mov %r8d,%edi -@@ -94151,8 +94234,9 @@ - 45c850: f3 0f 1e fa endbr64 - 45c854: 80 3d 3d e2 04 00 00 cmpb $0x0,0x4e23d(%rip) # 4aaa98 <__libc_single_threaded> - 45c85b: 74 13 je 45c870 <__libc_write+0x20> -- 45c85d: b8 01 00 00 00 mov $0x1,%eax -- 45c862: 0f 05 syscall ++ 45c7d8: 90 nop ++ 45c7d9: 90 nop + 45c7da: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c7e0: 77 2e ja 45c810 <__libc_read+0x80> + 45c7e2: 44 89 c7 mov %r8d,%edi +@@ -94148,8 +94231,9 @@ + 45c850: f3 0f 1e fa endbr64 + 45c854: 80 3d 3d e2 04 00 00 cmpb $0x0,0x4e23d(%rip) + 45c85b: 74 13 je 45c870 <__libc_write+0x20> +- 45c85d: b8 01 00 00 00 mov $0x1,%eax +- 45c862: 0f 05 syscall + 45c85d: -+ 45c862: 90 nop -+ 45c863: 90 nop - 45c864: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c86a: 77 54 ja 45c8c0 <__libc_write+0x70> - 45c86c: c3 ret -@@ -94168,8 +94252,9 @@ - 45c88c: 48 8b 75 f0 mov -0x10(%rbp),%rsi - 45c890: 41 89 c0 mov %eax,%r8d - 45c893: 8b 7d f8 mov -0x8(%rbp),%edi -- 45c896: b8 01 00 00 00 mov $0x1,%eax -- 45c89b: 0f 05 syscall ++ 45c862: 90 nop ++ 45c863: 90 nop + 45c864: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c86a: 77 54 ja 45c8c0 <__libc_write+0x70> + 45c86c: c3 ret +@@ -94165,8 +94249,9 @@ + 45c88c: 48 8b 75 f0 mov -0x10(%rbp),%rsi + 45c890: 41 89 c0 mov %eax,%r8d + 45c893: 8b 7d f8 mov -0x8(%rbp),%edi +- 45c896: b8 01 00 00 00 mov $0x1,%eax +- 45c89b: 0f 05 syscall + 45c896: -+ 45c89b: 90 nop -+ 45c89c: 90 nop - 45c89d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c8a3: 77 33 ja 45c8d8 <__libc_write+0x88> - 45c8a5: 44 89 c7 mov %r8d,%edi -@@ -94210,8 +94295,9 @@ - 45c919: f7 d0 not %eax - 45c91b: a9 00 00 41 00 test $0x410000,%eax - 45c920: 74 26 je 45c948 <__openat64_nocancel+0x58> -- 45c922: b8 01 01 00 00 mov $0x101,%eax -- 45c927: 0f 05 syscall ++ 45c89b: 90 nop ++ 45c89c: 90 nop + 45c89d: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c8a3: 77 33 ja 45c8d8 <__libc_write+0x88> + 45c8a5: 44 89 c7 mov %r8d,%edi +@@ -94207,8 +94292,9 @@ + 45c919: f7 d0 not %eax + 45c91b: a9 00 00 41 00 test $0x410000,%eax + 45c920: 74 26 je 45c948 <__openat64_nocancel+0x58> +- 45c922: b8 01 01 00 00 mov $0x101,%eax +- 45c927: 0f 05 syscall + 45c922: -+ 45c927: 90 nop -+ 45c928: 90 nop - 45c929: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c92f: 77 37 ja 45c968 <__openat64_nocancel+0x78> - 45c931: 48 8b 55 c8 mov -0x38(%rbp),%rdx -@@ -94239,8 +94325,9 @@ ++ 45c927: 90 nop ++ 45c928: 90 nop + 45c929: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c92f: 77 37 ja 45c968 <__openat64_nocancel+0x78> + 45c931: 48 8b 55 c8 mov -0x38(%rbp),%rdx +@@ -94236,8 +94322,9 @@ 000000000045c980 <__pread64_nocancel>: - 45c980: f3 0f 1e fa endbr64 - 45c984: 49 89 ca mov %rcx,%r10 -- 45c987: b8 11 00 00 00 mov $0x11,%eax -- 45c98c: 0f 05 syscall + 45c980: f3 0f 1e fa endbr64 + 45c984: 49 89 ca mov %rcx,%r10 +- 45c987: b8 11 00 00 00 mov $0x11,%eax +- 45c98c: 0f 05 syscall + 45c987: -+ 45c98c: 90 nop -+ 45c98d: 90 nop - 45c98e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c994: 77 0a ja 45c9a0 <__pread64_nocancel+0x20> - 45c996: c3 ret -@@ -94257,8 +94344,9 @@ ++ 45c98c: 90 nop ++ 45c98d: 90 nop + 45c98e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c994: 77 0a ja 45c9a0 <__pread64_nocancel+0x20> + 45c996: c3 ret +@@ -94254,8 +94341,9 @@ 000000000045c9c0 <__write_nocancel>: - 45c9c0: f3 0f 1e fa endbr64 -- 45c9c4: b8 01 00 00 00 mov $0x1,%eax -- 45c9c9: 0f 05 syscall + 45c9c0: f3 0f 1e fa endbr64 +- 45c9c4: b8 01 00 00 00 mov $0x1,%eax +- 45c9c9: 0f 05 syscall + 45c9c4: -+ 45c9c9: 90 nop -+ 45c9ca: 90 nop - 45c9cb: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45c9d1: 77 05 ja 45c9d8 <__write_nocancel+0x18> - 45c9d3: c3 ret -@@ -94282,8 +94370,9 @@ - 45ca0d: 48 89 45 f8 mov %rax,-0x8(%rbp) - 45ca11: 31 c0 xor %eax,%eax - 45ca13: 48 8d 55 d0 lea -0x30(%rbp),%rdx -- 45ca17: b8 10 00 00 00 mov $0x10,%eax -- 45ca1c: 0f 05 syscall ++ 45c9c9: 90 nop ++ 45c9ca: 90 nop + 45c9cb: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45c9d1: 77 05 ja 45c9d8 <__write_nocancel+0x18> + 45c9d3: c3 ret +@@ -94279,8 +94367,9 @@ + 45ca0d: 48 89 45 f8 mov %rax,-0x8(%rbp) + 45ca11: 31 c0 xor %eax,%eax + 45ca13: 48 8d 55 d0 lea -0x30(%rbp),%rdx +- 45ca17: b8 10 00 00 00 mov $0x10,%eax +- 45ca1c: 0f 05 syscall + 45ca17: -+ 45ca1c: 90 nop -+ 45ca1d: 90 nop - 45ca1e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45ca24: 77 6a ja 45ca90 <__tcgetattr+0xa0> - 45ca26: 89 c2 mov %eax,%edx -@@ -94329,9 +94418,11 @@ - 45cab4: 49 89 f2 mov %rsi,%r10 - 45cab7: 31 d2 xor %edx,%edx - 45cab9: 89 fe mov %edi,%esi -- 45cabb: b8 2e 01 00 00 mov $0x12e,%eax -- 45cac0: 31 ff xor %edi,%edi -- 45cac2: 0f 05 syscall ++ 45ca1c: 90 nop ++ 45ca1d: 90 nop + 45ca1e: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45ca24: 77 6a ja 45ca90 <__tcgetattr+0xa0> + 45ca26: 89 c2 mov %eax,%edx +@@ -94326,9 +94415,11 @@ + 45cab4: 49 89 f2 mov %rsi,%r10 + 45cab7: 31 d2 xor %edx,%edx + 45cab9: 89 fe mov %edi,%esi +- 45cabb: b8 2e 01 00 00 mov $0x12e,%eax +- 45cac0: 31 ff xor %edi,%edi +- 45cac2: 0f 05 syscall + 45cabb: -+ 45cac0: 90 nop -+ 45cac1: 90 nop -+ 45cac2: 90 nop -+ 45cac3: 90 nop - 45cac4: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 45caca: 77 04 ja 45cad0 <__GI___getrlimit+0x20> - 45cacc: c3 ret -@@ -97928,8 +98019,9 @@ - 45ff97: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax ++ 45cac0: 90 nop ++ 45cac1: 90 nop ++ 45cac2: 90 nop ++ 45cac3: 90 nop + 45cac4: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 45caca: 77 04 ja 45cad0 <__gi___getrlimit+0x20> + 45cacc: c3 ret +@@ -97925,8 +98016,9 @@ + 45ff97: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax 45ff9e: 00 00 - 45ffa0: 48 8d 78 1c lea 0x1c(%rax),%rdi -- 45ffa4: b8 ca 00 00 00 mov $0xca,%eax -- 45ffa9: 0f 05 syscall + 45ffa0: 48 8d 78 1c lea 0x1c(%rax),%rdi +- 45ffa4: b8 ca 00 00 00 mov $0xca,%eax +- 45ffa9: 0f 05 syscall + 45ffa4: -+ 45ffa9: 90 nop -+ 45ffaa: 90 nop - 45ffab: 48 8d 3d 6e ab 04 00 lea 0x4ab6e(%rip),%rdi # 4aab20 <_dl_load_lock> - 45ffb2: 44 89 8d 44 ff ff ff mov %r9d,-0xbc(%rbp) - 45ffb9: 4c 89 85 48 ff ff ff mov %r8,-0xb8(%rbp) -@@ -100864,8 +100956,7 @@ - 463062: 45 31 d2 xor %r10d,%r10d - 463065: ba 02 00 00 00 mov $0x2,%edx - 46306a: be 80 00 00 00 mov $0x80,%esi -- 46306f: 44 89 c8 mov %r9d,%eax -- 463072: 0f 05 syscall ++ 45ffa9: 90 nop ++ 45ffaa: 90 nop + 45ffab: 48 8d 3d 6e ab 04 00 lea 0x4ab6e(%rip),%rdi + 45ffb2: 44 89 8d 44 ff ff ff mov %r9d,-0xbc(%rbp) + 45ffb9: 4c 89 85 48 ff ff ff mov %r8,-0xb8(%rbp) +@@ -100861,8 +100953,7 @@ + 463062: 45 31 d2 xor %r10d,%r10d + 463065: ba 02 00 00 00 mov $0x2,%edx + 46306a: be 80 00 00 00 mov $0x80,%esi +- 46306f: 44 89 c8 mov %r9d,%eax +- 463072: 0f 05 syscall + 46306f: - 463074: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 46307a: 76 dc jbe 463058 <__thread_gscope_wait+0x88> - 46307c: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -100904,8 +100995,7 @@ - 463102: 45 31 d2 xor %r10d,%r10d - 463105: ba 02 00 00 00 mov $0x2,%edx - 46310a: be 80 00 00 00 mov $0x80,%esi -- 46310f: 44 89 c8 mov %r9d,%eax -- 463112: 0f 05 syscall + 463074: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 46307a: 76 dc jbe 463058 <__thread_gscope_wait+0x88> + 46307c: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -100901,8 +100992,7 @@ + 463102: 45 31 d2 xor %r10d,%r10d + 463105: ba 02 00 00 00 mov $0x2,%edx + 46310a: be 80 00 00 00 mov $0x80,%esi +- 46310f: 44 89 c8 mov %r9d,%eax +- 463112: 0f 05 syscall + 46310f: - 463114: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 46311a: 76 dc jbe 4630f8 <__thread_gscope_wait+0x128> - 46311c: 83 f8 f5 cmp $0xfffffff5,%eax -@@ -104731,8 +104821,11 @@ - 4669cc: 0f 1f 40 00 nopl 0x0(%rax) + 463114: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 46311a: 76 dc jbe 4630f8 <__thread_gscope_wait+0x128> + 46311c: 83 f8 f5 cmp $0xfffffff5,%eax +@@ -104728,8 +104818,11 @@ + 4669cc: 0f 1f 40 00 nopl 0x0(%rax) 00000000004669d0 <__restore_rt>: -- 4669d0: 48 c7 c0 0f 00 00 00 mov $0xf,%rax -- 4669d7: 0f 05 syscall +- 4669d0: 48 c7 c0 0f 00 00 00 mov $0xf,%rax +- 4669d7: 0f 05 syscall + 4669d0: -+ 4669d5: 90 nop -+ 4669d6: 90 nop -+ 4669d7: 90 nop -+ 4669d8: 90 nop - 4669d9: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) ++ 4669d5: 90 nop ++ 4669d6: 90 nop ++ 4669d7: 90 nop ++ 4669d8: 90 nop + 4669d9: 0f 1f 80 00 00 00 00 nopl 0x0(%rax) 00000000004669e0 <__libc_sigaction>: -@@ -104776,8 +104869,9 @@ - 466a9f: 0f 11 b5 38 ff ff ff movups %xmm6,-0xc8(%rbp) - 466aa6: 0f 11 bd 48 ff ff ff movups %xmm7,-0xb8(%rbp) - 466aad: 41 ba 08 00 00 00 mov $0x8,%r10d -- 466ab3: b8 0d 00 00 00 mov $0xd,%eax -- 466ab8: 0f 05 syscall +@@ -104773,8 +104866,9 @@ + 466a9f: 0f 11 b5 38 ff ff ff movups %xmm6,-0xc8(%rbp) + 466aa6: 0f 11 bd 48 ff ff ff movups %xmm7,-0xb8(%rbp) + 466aad: 41 ba 08 00 00 00 mov $0x8,%r10d +- 466ab3: b8 0d 00 00 00 mov $0xd,%eax +- 466ab8: 0f 05 syscall + 466ab3: -+ 466ab8: 90 nop -+ 466ab9: 90 nop - 466aba: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 466ac0: 0f 87 ba 00 00 00 ja 466b80 <__libc_sigaction+0x1a0> - 466ac6: 89 c2 mov %eax,%edx -@@ -111544,8 +111638,7 @@ - 46cb11: 45 31 d2 xor %r10d,%r10d - 46cb14: 89 ca mov %ecx,%edx - 46cb16: be 80 00 00 00 mov $0x80,%esi -- 46cb1b: 44 89 c0 mov %r8d,%eax -- 46cb1e: 0f 05 syscall ++ 466ab8: 90 nop ++ 466ab9: 90 nop + 466aba: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 466ac0: 0f 87 ba 00 00 00 ja 466b80 <__libc_sigaction+0x1a0> + 466ac6: 89 c2 mov %eax,%edx +@@ -111541,8 +111635,7 @@ + 46cb11: 45 31 d2 xor %r10d,%r10d + 46cb14: 89 ca mov %ecx,%edx + 46cb16: be 80 00 00 00 mov $0x80,%esi +- 46cb1b: 44 89 c0 mov %r8d,%eax +- 46cb1e: 0f 05 syscall + 46cb1b: - 46cb20: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax - 46cb26: 77 0d ja 46cb35 <__pthread_disable_asynccancel+0x65> - 46cb28: 8b 0f mov (%rdi),%ecx -@@ -111701,8 +111794,7 @@ - 46ccd5: 81 f6 00 01 00 00 xor $0x100,%esi - 46ccdb: 40 80 ce 89 or $0x89,%sil - 46ccdf: 44 31 c6 xor %r8d,%esi -- 46cce2: 45 31 c0 xor %r8d,%r8d -- 46cce5: 0f 05 syscall + 46cb20: 48 3d 00 f0 ff ff cmp $0xfffffffffffff000,%rax + 46cb26: 77 0d ja 46cb35 <__pthread_disable_asynccancel+0x65> + 46cb28: 8b 0f mov (%rdi),%ecx +@@ -111698,8 +111791,7 @@ + 46ccd5: 81 f6 00 01 00 00 xor $0x100,%esi + 46ccdb: 40 80 ce 89 or $0x89,%sil + 46ccdf: 44 31 c6 xor %r8d,%esi +- 46cce2: 45 31 c0 xor %r8d,%r8d +- 46cce5: 0f 05 syscall + 46cce2: - 46cce7: 85 c0 test %eax,%eax - 46cce9: 7f 27 jg 46cd12 <__futex_abstimed_wait64+0x62> - 46cceb: 83 f8 ea cmp $0xffffffea,%eax -@@ -111756,8 +111848,9 @@ - 46cd83: 45 31 c0 xor %r8d,%r8d - 46cd86: 49 89 ca mov %rcx,%r10 - 46cd89: 44 89 e2 mov %r12d,%edx -- 46cd8c: b8 ca 00 00 00 mov $0xca,%eax -- 46cd91: 0f 05 syscall + 46cce7: 85 c0 test %eax,%eax + 46cce9: 7f 27 jg 46cd12 <__futex_abstimed_wait64+0x62> + 46cceb: 83 f8 ea cmp $0xffffffea,%eax +@@ -111753,8 +111845,9 @@ + 46cd83: 45 31 c0 xor %r8d,%r8d + 46cd86: 49 89 ca mov %rcx,%r10 + 46cd89: 44 89 e2 mov %r12d,%edx +- 46cd8c: b8 ca 00 00 00 mov $0xca,%eax +- 46cd91: 0f 05 syscall + 46cd8c: -+ 46cd91: 90 nop -+ 46cd92: 90 nop - 46cd93: 48 89 c3 mov %rax,%rbx - 46cd96: 89 d8 mov %ebx,%eax - 46cd98: 85 db test %ebx,%ebx -@@ -111802,8 +111895,9 @@ - 46ce0d: 48 8b 7d d0 mov -0x30(%rbp),%rdi - 46ce11: 41 b9 ff ff ff ff mov $0xffffffff,%r9d - 46ce17: 44 89 e2 mov %r12d,%edx -- 46ce1a: b8 ca 00 00 00 mov $0xca,%eax -- 46ce1f: 0f 05 syscall ++ 46cd91: 90 nop ++ 46cd92: 90 nop + 46cd93: 48 89 c3 mov %rax,%rbx + 46cd96: 89 d8 mov %ebx,%eax + 46cd98: 85 db test %ebx,%ebx +@@ -111799,8 +111892,9 @@ + 46ce0d: 48 8b 7d d0 mov -0x30(%rbp),%rdi + 46ce11: 41 b9 ff ff ff ff mov $0xffffffff,%r9d + 46ce17: 44 89 e2 mov %r12d,%edx +- 46ce1a: b8 ca 00 00 00 mov $0xca,%eax +- 46ce1f: 0f 05 syscall + 46ce1a: -+ 46ce1f: 90 nop -+ 46ce20: 90 nop - 46ce21: 44 89 ef mov %r13d,%edi - 46ce24: 48 89 c3 mov %rax,%rbx - 46ce27: e8 a4 fc ff ff call 46cad0 <__pthread_disable_asynccancel> -@@ -111827,8 +111921,9 @@ - 46ce66: 48 85 d2 test %rdx,%rdx - 46ce69: 0f 45 f1 cmovne %ecx,%esi - 46ce6c: 31 d2 xor %edx,%edx -- 46ce6e: b8 ca 00 00 00 mov $0xca,%eax -- 46ce73: 0f 05 syscall ++ 46ce1f: 90 nop ++ 46ce20: 90 nop + 46ce21: 44 89 ef mov %r13d,%edi + 46ce24: 48 89 c3 mov %rax,%rbx + 46ce27: e8 a4 fc ff ff call 46cad0 <__pthread_disable_asynccancel> +@@ -111824,8 +111918,9 @@ + 46ce66: 48 85 d2 test %rdx,%rdx + 46ce69: 0f 45 f1 cmovne %ecx,%esi + 46ce6c: 31 d2 xor %edx,%edx +- 46ce6e: b8 ca 00 00 00 mov $0xca,%eax +- 46ce73: 0f 05 syscall + 46ce6e: -+ 46ce73: 90 nop -+ 46ce74: 90 nop - 46ce75: 83 f8 da cmp $0xffffffda,%eax - 46ce78: 74 26 je 46cea0 <__futex_lock_pi64+0x50> - 46ce7a: 83 f8 92 cmp $0xffffff92,%eax -@@ -114436,8 +114531,9 @@ ++ 46ce73: 90 nop ++ 46ce74: 90 nop + 46ce75: 83 f8 da cmp $0xffffffda,%eax + 46ce78: 74 26 je 46cea0 <__futex_lock_pi64+0x50> + 46ce7a: 83 f8 92 cmp $0xffffff92,%eax +@@ -114433,8 +114528,9 @@ 000000000046f340 <__GI___fstatat>: - 46f340: f3 0f 1e fa endbr64 - 46f344: 41 89 ca mov %ecx,%r10d -- 46f347: b8 06 01 00 00 mov $0x106,%eax -- 46f34c: 0f 05 syscall + 46f340: f3 0f 1e fa endbr64 + 46f344: 41 89 ca mov %ecx,%r10d +- 46f347: b8 06 01 00 00 mov $0x106,%eax +- 46f34c: 0f 05 syscall + 46f347: -+ 46f34c: 90 nop -+ 46f34d: 90 nop - 46f34e: 3d 00 f0 ff ff cmp $0xfffff000,%eax - 46f353: 77 0b ja 46f360 <__GI___fstatat+0x20> - 46f355: 31 c0 xor %eax,%eax -@@ -117945,8 +118041,9 @@ - 47296c: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax ++ 46f34c: 90 nop ++ 46f34d: 90 nop + 46f34e: 3d 00 f0 ff ff cmp $0xfffff000,%eax + 46f353: 77 0b ja 46f360 <__gi___fstatat+0x20> + 46f355: 31 c0 xor %eax,%eax +@@ -117942,8 +118038,9 @@ + 47296c: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax 472973: 00 00 - 472975: 48 8d 78 1c lea 0x1c(%rax),%rdi -- 472979: b8 ca 00 00 00 mov $0xca,%eax -- 47297e: 0f 05 syscall + 472975: 48 8d 78 1c lea 0x1c(%rax),%rdi +- 472979: b8 ca 00 00 00 mov $0xca,%eax +- 47297e: 0f 05 syscall + 472979: -+ 47297e: 90 nop -+ 47297f: 90 nop - 472980: eb 8c jmp 47290e <_dl_fixup+0x10e> - 472982: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) - 472988: 31 c0 xor %eax,%eax -@@ -122353,8 +122450,9 @@ - 476c07: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax ++ 47297e: 90 nop ++ 47297f: 90 nop + 472980: eb 8c jmp 47290e <_dl_fixup+0x10e> + 472982: 66 0f 1f 44 00 00 nopw 0x0(%rax,%rax,1) + 472988: 31 c0 xor %eax,%eax +@@ -122350,8 +122447,9 @@ + 476c07: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax 476c0e: 00 00 - 476c10: 48 8d 78 1c lea 0x1c(%rax),%rdi -- 476c14: b8 ca 00 00 00 mov $0xca,%eax -- 476c19: 0f 05 syscall + 476c10: 48 8d 78 1c lea 0x1c(%rax),%rdi +- 476c14: b8 ca 00 00 00 mov $0xca,%eax +- 476c19: 0f 05 syscall + 476c14: -+ 476c19: 90 nop -+ 476c1a: 90 nop - 476c1b: 48 83 7d 98 00 cmpq $0x0,-0x68(%rbp) - 476c20: 48 8b 4d b0 mov -0x50(%rbp),%rcx - 476c24: 0f 84 ae fd ff ff je 4769d8 <_dl_vsym+0xb8> -@@ -122513,8 +122611,9 @@ - 476e41: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax ++ 476c19: 90 nop ++ 476c1a: 90 nop + 476c1b: 48 83 7d 98 00 cmpq $0x0,-0x68(%rbp) + 476c20: 48 8b 4d b0 mov -0x50(%rbp),%rcx + 476c24: 0f 84 ae fd ff ff je 4769d8 <_dl_vsym+0xb8> +@@ -122510,8 +122608,9 @@ + 476e41: 64 48 8b 04 25 10 00 mov %fs:0x10,%rax 476e48: 00 00 - 476e4a: 48 8d 78 1c lea 0x1c(%rax),%rdi -- 476e4e: b8 ca 00 00 00 mov $0xca,%eax -- 476e53: 0f 05 syscall + 476e4a: 48 8d 78 1c lea 0x1c(%rax),%rdi +- 476e4e: b8 ca 00 00 00 mov $0xca,%eax +- 476e53: 0f 05 syscall + 476e4e: -+ 476e53: 90 nop -+ 476e54: 90 nop - 476e55: 48 83 7d 98 00 cmpq $0x0,-0x68(%rbp) - 476e5a: 48 8b 4d b0 mov -0x50(%rbp),%rcx - 476e5e: 0f 84 3c fe ff ff je 476ca0 <_dl_sym+0x60> ++ 476e53: 90 nop ++ 476e54: 90 nop + 476e55: 48 83 7d 98 00 cmpq $0x0,-0x68(%rbp) + 476e5a: 48 8b 4d b0 mov -0x50(%rbp),%rcx + 476e5e: 0f 84 3c fe ff ff je 476ca0 <_dl_sym+0x60> diff --git a/npm/README.md b/npm/README.md new file mode 100644 index 0000000000..6848f49a07 --- /dev/null +++ b/npm/README.md @@ -0,0 +1,90 @@ +# @openclew/litebox + +Boot an interactive Linux shell inside [LiteBox](https://github.com/dywongcloud/litebox), +a userspace syscall-translation sandbox. + +```sh +npx @openclew/litebox +``` + +## What this actually does + +LiteBox runs unmodified Linux programs by translating their syscalls, rather than +by emulating instructions or booting a VM. Guest code executes natively on your +CPU. This package: + +1. downloads a **pinned** source revision of LiteBox, +2. builds the runner and packager for your host with `cargo`, +3. packages a guest root filesystem from a public OCI image, and +4. starts a shell inside it, attached to your terminal. + +The first run takes a few minutes. Everything is cached per source revision +afterwards; `npx @openclew/litebox --where` prints the cache directory. + +## Requirements + +- **Node.js 18+** +- **A Rust toolchain** (`cargo`, `rustc`) — +- **Network access on first run**, for the source archive and the guest image + +This package builds from source instead of shipping prebuilt binaries. That is a +deliberate trade: prebuilt binaries would mean publishing five platform/arch +combinations that cannot all be tested, and a binary nobody has ever executed is +not a nicer user experience than a build. + +## Usage modes + +The pinned revision implements `fork(2)`, so the interactive shell can launch +external commands. Direct execution remains useful for scripts and automation +because the guest program's output goes straight to the host process: + +```sh +npx @openclew/litebox +npx @openclew/litebox -- /bin/busybox cat /etc/alpine-release +npx @openclew/litebox -- /bin/busybox ls -l /etc +``` + +## Platform support + +Stated at the level it has actually been verified, not at the level the code +implies: + +| Host | Status | Notes | +|---|---|---| +| macOS arm64 | **verified** | Developed and tested here | +| macOS x64 | builds, unverified | The macOS platform is aarch64-only in places | +| Linux x64 / arm64 | builds, unverified | Two known-failing tests upstream | +| Windows x64 / arm64 | builds, unverified | Guest networking is an unimplemented stub; console input incomplete | + +"Builds, unverified" means the runner exists and is expected to compile, but no +guest has been run there by us. It may work. Reports welcome. + +On macOS arm64, LiteBox rewrites Linux guests' use of `x18`, allowing Node.js and +its child processes to run. A separate Node shutdown issue can print `pure virtual +method called` and discard buffered `console.log` output; use `fs.writeSync` when +the final output must be observed synchronously. + +## Usage + +```sh +npx @openclew/litebox # interactive shell +npx @openclew/litebox -- /bin/busybox uname -a # run one command +npx @openclew/litebox --image public.ecr.aws/docker/library/node:alpine +``` + +| Option | Meaning | +|---|---| +| `--image ` | Guest OCI image (public registries only) | +| `--shell ` | Guest shell to start | +| `--rev ` | Build a specific source revision | +| `--rebuild` | Rebuild even if cached | +| `--refresh-image` | Re-package the guest image even if cached | +| `--where` | Print the cache directory | +| `-q, --quiet` | Suppress progress output | + +`LITEBOX_SRC=/path/to/checkout` builds from a local tree instead of downloading. +`LITEBOX_CACHE_DIR` overrides the cache location. + +## License + +MIT. Copyright (c) Microsoft Corporation. diff --git a/npm/bin/litebox.js b/npm/bin/litebox.js new file mode 100644 index 0000000000..6f51f519ed --- /dev/null +++ b/npm/bin/litebox.js @@ -0,0 +1,155 @@ +#!/usr/bin/env node +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +const { spawn } = require('child_process'); +const { detect, PINNED_REV } = require('../lib/platform'); +const { resolveSource } = require('../lib/source'); +const { buildBinaries } = require('../lib/build'); +const { ensureImage, DEFAULT_IMAGE } = require('../lib/image'); +const { revDir } = require('../lib/cache'); + +const USAGE = `litebox -- boot an interactive Linux shell inside LiteBox + +Usage: + npx @openclew/litebox [options] [-- [args...]] + +Options: + --image Guest OCI image (default: ${DEFAULT_IMAGE}) + --shell Guest shell to start (default: /bin/busybox sh) + --rev Build a specific source revision (default: pinned) + --rebuild Rebuild the binaries even if cached + --refresh-image Re-package the guest image even if cached + --where Print the cache directory and exit + -q, --quiet Suppress progress output + -h, --help Show this help + -V, --version Show version + +Examples: + npx @openclew/litebox # interactive shell + npx @openclew/litebox -- /bin/busybox uname -a + npx @openclew/litebox --image public.ecr.aws/docker/library/node:alpine + +First run downloads a pinned source revision and builds it with cargo, which +takes a few minutes. Later runs reuse the cache. A Rust toolchain is required. +`; + +function parseArgs(argv) { + const opts = { + image: DEFAULT_IMAGE, + shell: null, + rev: PINNED_REV, + rebuild: false, + refreshImage: false, + quiet: false, + where: false, + command: null, + }; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + if (a === '--') { + opts.command = argv.slice(i + 1); + break; + } else if (a === '--image') opts.image = argv[++i]; + else if (a === '--shell') opts.shell = argv[++i]; + else if (a === '--rev') { + const rev = argv[++i]; + if (!/^[0-9a-f]{40}$/i.test(rev || '')) { + return { error: '--rev requires a full 40-character hexadecimal commit SHA.' }; + } + opts.rev = rev.toLowerCase(); + } + else if (a === '--rebuild') opts.rebuild = true; + else if (a === '--refresh-image') opts.refreshImage = true; + else if (a === '--where') opts.where = true; + else if (a === '-q' || a === '--quiet') opts.quiet = true; + else if (a === '-h' || a === '--help') { + process.stdout.write(USAGE); + process.exit(0); + } else if (a === '-V' || a === '--version') { + process.stdout.write(`${require('../package.json').version}\n`); + process.exit(0); + } else { + throw new Error(`unknown option: ${a}\n\n${USAGE}`); + } + } + if (opts.command && opts.command.length === 0) opts.command = null; + return opts; +} + +async function main() { + const opts = parseArgs(process.argv.slice(2)); + if (opts.error) { + process.stderr.write(`litebox: ${opts.error}\n`); + return 1; + } + const plat = detect(); + + if (opts.where) { + process.stdout.write(revDir(opts.rev) + '\n'); + return 0; + } + + // Progress goes to stderr so `-- ` output on stdout stays clean and + // pipeable on the host side. + const log = opts.quiet ? () => {} : (m) => process.stderr.write(`litebox: ${m}\n`); + + if (!plat.supported) { + throw new Error( + `unsupported host platform: ${plat.os}/${plat.arch}\n` + + 'LiteBox has runners for macOS, Linux and Windows.' + ); + } + if (plat.support.status !== 'verified') { + log(`warning: ${plat.key} is "${plat.support.status}" -- ${plat.support.note}`); + } + + const srcDir = await resolveSource(opts.rev, log); + const { runner, packager } = buildBinaries({ + rev: opts.rev, + srcDir, + plat, + rebuild: opts.rebuild, + log, + }); + const imageTar = await ensureImage({ + rev: opts.rev, + packager, + image: opts.image, + refresh: opts.refreshImage, + log, + }); + + const guestArgv = opts.command + ? opts.command + : [opts.shell || '/bin/busybox', ...(opts.shell ? [] : ['sh'])]; + + if (!opts.command) log(`starting ${guestArgv.join(' ')}`); + + // `inherit` hands the guest the real terminal, which is what makes this an + // interactive session rather than a pipe: the shim's terminal support reads + // the host tty directly, including raw mode when the guest asks for it. + const child = spawn(runner, ['--initial-files', imageTar, '--', ...guestArgv], { + stdio: 'inherit', + }); + + return await new Promise((resolve) => { + child.on('error', (e) => { + process.stderr.write(`litebox: could not start the runner: ${e.message}\n`); + resolve(70); + }); + // A guest killed by a signal is reported the way a shell reports it, so + // `echo $?` after a guest segfault reads the same as it would on Linux. + child.on('exit', (code, signal) => resolve(signal ? 128 + (require('os').constants.signals[signal] || 0) : code)); + }); +} + +main().then( + (code) => process.exit(code), + (err) => { + process.stderr.write(`litebox: ${err && err.message ? err.message : err}\n`); + process.exit(1); + } +); diff --git a/npm/in/prd-resolve/fable-1787265011-20.txt b/npm/in/prd-resolve/fable-1787265011-20.txt new file mode 100644 index 0000000000..3f10a8fc6e --- /dev/null +++ b/npm/in/prd-resolve/fable-1787265011-20.txt @@ -0,0 +1 @@ +{"SESSION_ID":"fable-1787265011","id":"linux-terminal-set-action-ioctls","resolution":"Landed in 4a62d24: TCSETSW/TCSETSF decode to TerminalSetAction Now/Drain/Flush through IoctlArg::TCSETS, shim, and StdioProvider::set_terminal_raw_mode_with_action; macOS maps to TCSANOW/TCSADRAIN/TCSAFLUSH and Flush additionally discards StdinPump ring bytes. Verified live: stty -echo/echo toggles real pty echo; a no_std guest issuing TCSETSF after pty input was buffered reads EAGAIN while a no-flush control reads 14 bytes."} diff --git a/npm/in/prd-resolve/fable-1787265011-21.txt b/npm/in/prd-resolve/fable-1787265011-21.txt new file mode 100644 index 0000000000..e1033c5b9a --- /dev/null +++ b/npm/in/prd-resolve/fable-1787265011-21.txt @@ -0,0 +1 @@ +{"SESSION_ID":"fable-1787265011","id":"npm-release-refresh-user-guidance","resolution":"Landed in 4a62d24: SHELL_CAVEAT removed from npm/bin/litebox.js, README fork-limitation section replaced with working usage modes, x18 note updated to describe the shipped rewrite. npm 0.1.1 pins rev 4815891 which contains fork and the x18 rewriter."} diff --git a/npm/in/prd-resolve/fable-1787265011-22.txt b/npm/in/prd-resolve/fable-1787265011-22.txt new file mode 100644 index 0000000000..4cf9e93a5d --- /dev/null +++ b/npm/in/prd-resolve/fable-1787265011-22.txt @@ -0,0 +1 @@ +{"SESSION_ID":"fable-1787265011","id":"npm-revision-input-containment","resolution":"Landed in 4a62d24: --rev now rejects anything but a full 40-char hex commit SHA (lowercased before use). Verified live: --rev deadbeef exits 1 with a clear error; a full SHA is accepted."} diff --git a/npm/in/prd-resolve/fable-1787265011-23.txt b/npm/in/prd-resolve/fable-1787265011-23.txt new file mode 100644 index 0000000000..01a4b50e56 --- /dev/null +++ b/npm/in/prd-resolve/fable-1787265011-23.txt @@ -0,0 +1 @@ +{"SESSION_ID":"fable-1787265011","id":"npm-pin-current-source-revision","resolution":"Landed in 4a62d24: PINNED_REV updated to 4815891d363351e8bab675736d19b8b7cff16fad, the Node-enabled revision (fork, x18 rewrite, AF_NETLINK). npm version bumped to 0.1.1."} diff --git a/npm/lib/build.js b/npm/lib/build.js new file mode 100644 index 0000000000..effa6606ba --- /dev/null +++ b/npm/lib/build.js @@ -0,0 +1,104 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +const fs = require('fs'); +const path = require('path'); +const { spawnSync } = require('child_process'); +const { revDir, ensureDir } = require('./cache'); + +const PACKAGER = 'litebox_packager'; + +function have(cmd) { + const probe = spawnSync(cmd, ['--version'], { stdio: 'ignore' }); + return !probe.error && probe.status === 0; +} + +/// Rust is a hard prerequisite rather than something this package vendors. +/// +/// Shipping prebuilt binaries would mean shipping five platform/arch +/// combinations that cannot all be tested; building from a pinned source +/// revision on the user's own machine is the honest alternative, and it costs +/// one dependency the user can see and audit. +function requireToolchain() { + if (have('cargo') && have('rustc')) return; + throw new Error( + 'A Rust toolchain is required to build LiteBox.\n\n' + + ' Install it with: curl --proto =https --tlsv1.2 -sSf https://sh.rustup.rs | sh\n' + + ' (Windows: https://rustup.rs)\n\n' + + 'This package builds from a pinned source revision rather than shipping prebuilt\n' + + 'binaries, so that every supported platform runs code built for it on the machine\n' + + 'it will run on.' + ); +} + +/// Darwin enforces W^X, so the runner maps guest code `MAP_JIT` and needs the +/// `com.apple.security.cs.allow-jit` entitlement to write to those pages. An +/// ad-hoc signature (`-`) is enough; this is not distribution signing and needs +/// no Apple Developer account. +function codesignForJit(binary, workDir, log) { + const plist = path.join(workDir, 'litebox.entitlements'); + fs.writeFileSync( + plist, + '\n' + + '\n' + + '\n\n' + + ' com.apple.security.cs.allow-jit\n \n' + + '\n\n' + ); + const res = spawnSync('codesign', ['--sign', '-', '--entitlements', plist, '--force', binary], { + encoding: 'utf8', + }); + if (res.error || res.status !== 0) { + throw new Error( + `codesign failed for ${binary}: ${(res.stderr || res.error || '').toString().trim()}\n` + + 'Without the JIT entitlement the guest cannot execute on macOS.' + ); + } + log('signed runner with the JIT entitlement'); +} + +/// Build the runner and packager for this host, caching per revision. +function buildBinaries({ rev, srcDir, plat, rebuild, log }) { + const outDir = ensureDir(path.join(revDir(rev), 'bin')); + const runnerOut = path.join(outDir, plat.runner + plat.exeSuffix); + const packagerOut = path.join(outDir, PACKAGER + plat.exeSuffix); + + if (!rebuild && fs.existsSync(runnerOut) && fs.existsSync(packagerOut)) { + log('using cached binaries'); + return { runner: runnerOut, packager: packagerOut }; + } + + requireToolchain(); + log(`building ${plat.runner} and ${PACKAGER} (first run; this takes a few minutes)`); + + // A dedicated target directory keeps this out of the user's own build + // artifacts if they happen to be pointing LITEBOX_SRC at a real checkout. + const targetDir = path.join(revDir(rev), 'target'); + const res = spawnSync( + 'cargo', + ['build', '--release', '-p', plat.runner, '-p', PACKAGER, '--target-dir', targetDir], + { cwd: srcDir, stdio: 'inherit' } + ); + if (res.error || res.status !== 0) { + throw new Error( + `cargo build failed for ${plat.key}.\n` + + `Support status for this platform is "${plat.support.status}": ${plat.support.note}` + ); + } + + for (const [built, dest] of [ + [path.join(targetDir, 'release', plat.runner + plat.exeSuffix), runnerOut], + [path.join(targetDir, 'release', PACKAGER + plat.exeSuffix), packagerOut], + ]) { + if (!fs.existsSync(built)) throw new Error(`cargo reported success but ${built} is missing.`); + fs.copyFileSync(built, dest); + fs.chmodSync(dest, 0o755); + } + + if (plat.needsJitCodesign) codesignForJit(runnerOut, revDir(rev), log); + return { runner: runnerOut, packager: packagerOut }; +} + +module.exports = { buildBinaries, requireToolchain }; diff --git a/npm/lib/cache.js b/npm/lib/cache.js new file mode 100644 index 0000000000..abe108589e --- /dev/null +++ b/npm/lib/cache.js @@ -0,0 +1,37 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +const os = require('os'); +const path = require('path'); +const fs = require('fs'); + +/// Everything this package produces is derived, reproducible from `PINNED_REV`, +/// and potentially large (a Rust target directory and packaged guest images), so +/// it belongs in a cache directory rather than beside the installed module: an +/// `npx` invocation may install into a throwaway directory, and re-downloading +/// and rebuilding on every run would make the tool unusable. +function cacheRoot() { + if (process.env.LITEBOX_CACHE_DIR) return path.resolve(process.env.LITEBOX_CACHE_DIR); + if (process.platform === 'win32') { + return path.join(process.env.LOCALAPPDATA || path.join(os.homedir(), 'AppData', 'Local'), 'litebox'); + } + if (process.platform === 'darwin') { + return path.join(os.homedir(), 'Library', 'Caches', 'litebox'); + } + return path.join(process.env.XDG_CACHE_HOME || path.join(os.homedir(), '.cache'), 'litebox'); +} + +/// Keyed by revision so a package upgrade never silently reuses binaries built +/// from a different source tree. +function revDir(rev) { + return path.join(cacheRoot(), rev); +} + +function ensureDir(p) { + fs.mkdirSync(p, { recursive: true }); + return p; +} + +module.exports = { cacheRoot, revDir, ensureDir }; diff --git a/npm/lib/image.js b/npm/lib/image.js new file mode 100644 index 0000000000..6f7f0ab416 --- /dev/null +++ b/npm/lib/image.js @@ -0,0 +1,86 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +const fs = require('fs'); +const path = require('path'); +const crypto = require('crypto'); +const { spawnSync } = require('child_process'); +const { revDir, ensureDir } = require('./cache'); + +/// Docker Hub's anonymous-pull endpoint is not reachable from every network, and +/// the packager supports only public registries, so the default points at the +/// AWS public mirror of the same image. +const DEFAULT_IMAGE = 'public.ecr.aws/docker/library/alpine:latest'; + +/// Public mirrors rate-limit anonymous pulls per client, and a first run that +/// happens to land inside a limit window is otherwise indistinguishable from a +/// broken install. Retrying is correct here specifically because the failure is +/// transient and the request is idempotent; the cap keeps a genuinely blocked +/// network from hanging the command. +const PULL_ATTEMPTS = 3; +const BACKOFF_MS = [4000, 12000]; + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); + +function looksRateLimited(text) { + return /rate exceeded|too many requests|\b429\b|toomanyrequests/i.test(text || ''); +} + +/// Package a guest root filesystem, caching per (revision, image reference). +/// +/// The packager rewrites every executable ELF in the image for this host's +/// syscall-gate flavour, so a packaged tar is specific to the revision that +/// produced it and must not be shared across revisions. +async function ensureImage({ rev, packager, image, refresh, log }) { + const imagesDir = ensureDir(path.join(revDir(rev), 'images')); + const key = crypto.createHash('sha256').update(image).digest('hex').slice(0, 16); + const tarPath = path.join(imagesDir, `${key}.tar`); + + if (!refresh && fs.existsSync(tarPath)) { + log(`using cached guest image (${image})`); + return tarPath; + } + + log(`packaging guest image ${image} (first time for this image)`); + const partial = tarPath + '.partial'; + let lastOutput = ''; + + for (let attempt = 1; attempt <= PULL_ATTEMPTS; attempt++) { + fs.rmSync(partial, { force: true }); + // Captured rather than inherited so the rate-limit case can be recognised + // and explained; the packager's own output is echoed on the final failure. + const res = spawnSync(packager, ['--oci-image', image, '-o', partial], { encoding: 'utf8' }); + if (!res.error && res.status === 0) { + fs.renameSync(partial, tarPath); + return tarPath; + } + lastOutput = `${res.stdout || ''}${res.stderr || ''}${res.error ? res.error.message : ''}`; + if (attempt < PULL_ATTEMPTS && looksRateLimited(lastOutput)) { + const wait = BACKOFF_MS[attempt - 1]; + log(`registry rate-limited the pull; retrying in ${wait / 1000}s (${attempt}/${PULL_ATTEMPTS - 1})`); + await sleep(wait); + continue; + } + break; + } + + fs.rmSync(partial, { force: true }); + const rateLimited = looksRateLimited(lastOutput); + throw new Error( + `could not package the guest image "${image}".\n\n` + + lastOutput.trim() + + '\n\n' + + (rateLimited + ? 'The registry rate-limited anonymous pulls, which is transient and not a\n' + + 'problem with your install. The build is already cached, so simply running the\n' + + 'command again in a minute usually succeeds. You can also pass a different\n' + + 'image with --image.' + : 'Only public registries are supported, and network access is required the first\n' + + 'time an image is used. Docker Hub anonymous pulls are blocked on some networks;\n' + + `the default (${DEFAULT_IMAGE}) is a public mirror that usually works.`) + ); +} + +module.exports = { ensureImage, DEFAULT_IMAGE }; diff --git a/npm/lib/platform.js b/npm/lib/platform.js new file mode 100644 index 0000000000..22c3916b67 --- /dev/null +++ b/npm/lib/platform.js @@ -0,0 +1,68 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +/// The commit this package builds. Pinned rather than tracking a branch so a +/// given npm version always produces the same binaries. +const PINNED_REV = '4815891d363351e8bab675736d19b8b7cff16fad'; + +/// Which runner crate serves a Linux guest on each host OS. +/// +/// LiteBox is a syscall-translation layer, not an instruction emulator: guest +/// instructions execute natively. The guest architecture is therefore always the +/// host architecture, and a host/guest arch mismatch is not something a runner +/// can paper over. +const RUNNERS = { + darwin: 'litebox_runner_linux_on_macos_userland', + linux: 'litebox_runner_linux_userland', + win32: 'litebox_runner_linux_on_windows_userland', +}; + +/// Support status per (os, arch), stated at the level it has actually been +/// verified rather than at the level the crate list implies. +/// +/// `verified` means a real guest was run on that exact host and arch and its +/// output observed. `builds-unverified` means the runner crate exists and is +/// expected to compile, but no guest has been run there by this package's +/// authors -- it may work, and it may not. Being honest here is the difference +/// between a user filing a useful bug and concluding the whole thing is broken. +const SUPPORT = { + 'darwin/arm64': { status: 'verified', note: 'Apple Silicon; developed and tested here.' }, + 'darwin/x64': { + status: 'builds-unverified', + note: 'Intel Mac. The macOS platform is aarch64-only in places; expect build or runtime failures.', + }, + 'linux/x64': { status: 'builds-unverified', note: 'Two known-failing tests in this crate upstream.' }, + 'linux/arm64': { status: 'builds-unverified', note: 'Two known-failing tests in this crate upstream.' }, + 'win32/x64': { + status: 'builds-unverified', + note: 'Guest networking is an unimplemented stub on Windows, and console input is incomplete.', + }, + 'win32/arm64': { + status: 'builds-unverified', + note: 'Guest networking is an unimplemented stub on Windows, and console input is incomplete.', + }, +}; + +function detect() { + const os = process.platform; + const arch = process.arch; + const key = `${os}/${arch}`; + const runner = RUNNERS[os]; + return { + os, + arch, + key, + runner, + supported: Boolean(runner), + support: SUPPORT[key] || { status: 'unknown', note: 'No support information recorded.' }, + /// Only macOS needs the guest's executable pages signed for JIT: Darwin + /// enforces W^X, so the runner maps guest code `MAP_JIT` and must carry the + /// `com.apple.security.cs.allow-jit` entitlement to write to it. + needsJitCodesign: os === 'darwin', + exeSuffix: os === 'win32' ? '.exe' : '', + }; +} + +module.exports = { detect, PINNED_REV, RUNNERS, SUPPORT }; diff --git a/npm/lib/source.js b/npm/lib/source.js new file mode 100644 index 0000000000..3922bc058d --- /dev/null +++ b/npm/lib/source.js @@ -0,0 +1,71 @@ +// Copyright (c) Microsoft Corporation. +// Licensed under the MIT license. + +'use strict'; + +const fs = require('fs'); +const path = require('path'); +const { spawnSync } = require('child_process'); +const { revDir, ensureDir } = require('./cache'); + +const TARBALL = (rev) => `https://codeload.github.com/dywongcloud/litebox/tar.gz/${rev}`; + +/// A marker written only after extraction fully succeeds, so an interrupted +/// download can never be mistaken for a usable tree on the next run. +const STAMP = '.litebox-source-complete'; + +/// Resolve the source tree to build from. +/// +/// `LITEBOX_SRC` short-circuits everything and points at a working checkout. +/// That exists for two reasons: developing this package without a network round +/// trip, and letting someone build a revision other than the pinned one without +/// republishing. +async function resolveSource(rev, log) { + if (process.env.LITEBOX_SRC) { + const local = path.resolve(process.env.LITEBOX_SRC); + if (!fs.existsSync(path.join(local, 'Cargo.toml'))) { + throw new Error(`LITEBOX_SRC=${local} does not look like a litebox checkout (no Cargo.toml).`); + } + log(`using local source tree: ${local}`); + return local; + } + + const dest = path.join(revDir(rev), 'src'); + if (fs.existsSync(path.join(dest, STAMP))) return dest; + + ensureDir(path.dirname(dest)); + const url = TARBALL(rev); + log(`downloading source ${rev.slice(0, 12)} ...`); + + const res = await fetch(url); + if (!res.ok) { + throw new Error(`could not download source from ${url} (HTTP ${res.status}).`); + } + const tarPath = path.join(revDir(rev), 'source.tar.gz'); + fs.writeFileSync(tarPath, Buffer.from(await res.arrayBuffer())); + + // Extract into a staging directory first, then rename: a half-extracted tree + // that happens to contain Cargo.toml would otherwise look buildable. + const staging = path.join(revDir(rev), 'src.partial'); + fs.rmSync(staging, { recursive: true, force: true }); + ensureDir(staging); + + // `tar` is present on macOS and Linux, and Windows 10+ ships bsdtar as + // `tar.exe`. `--strip-components=1` drops GitHub's `-/` wrapper. + const untar = spawnSync('tar', ['-xzf', tarPath, '--strip-components=1', '-C', staging], { + stdio: 'inherit', + }); + if (untar.error || untar.status !== 0) { + throw new Error( + 'could not extract the source archive. A `tar` command is required ' + + '(present by default on macOS, Linux, and Windows 10 and later).' + ); + } + fs.writeFileSync(path.join(staging, STAMP), rev); + fs.rmSync(dest, { recursive: true, force: true }); + fs.renameSync(staging, dest); + fs.rmSync(tarPath, { force: true }); + return dest; +} + +module.exports = { resolveSource }; diff --git a/npm/package.json b/npm/package.json new file mode 100644 index 0000000000..0c25a12931 --- /dev/null +++ b/npm/package.json @@ -0,0 +1,34 @@ +{ + "name": "@openclew/litebox", + "version": "0.1.1", + "description": "Boot an interactive Linux shell inside LiteBox, a userspace syscall-translation sandbox. Builds from a pinned source revision on first run.", + "bin": { + "litebox": "bin/litebox.js" + }, + "type": "commonjs", + "engines": { + "node": ">=18" + }, + "files": [ + "bin", + "lib", + "README.md" + ], + "keywords": [ + "litebox", + "sandbox", + "syscall", + "linux", + "container" + ], + "license": "MIT", + "repository": { + "type": "git", + "url": "git+https://github.com/dywongcloud/litebox.git", + "directory": "npm" + }, + "homepage": "https://github.com/dywongcloud/litebox#readme", + "bugs": { + "url": "https://github.com/dywongcloud/litebox/issues" + } +} diff --git a/vendor/tar-no-std-0.3.5/.editorconfig b/vendor/tar-no-std-0.3.5/.editorconfig new file mode 100644 index 0000000000..cebe5f4844 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.editorconfig @@ -0,0 +1,15 @@ +# top-most EditorConfig file +root = true + +# Unix-style newlines with a newline ending every file +[*] +charset = utf-8 +end_of_line = lf +insert_final_newline = true +indent_style = space +indent_size = 4 +trim_trailing_whitespace = true +max_line_length = 80 + +[{*.toml,*.yml}] +indent_size = 2 diff --git a/vendor/tar-no-std-0.3.5/.github/FUNDING.yml b/vendor/tar-no-std-0.3.5/.github/FUNDING.yml new file mode 100644 index 0000000000..1d2ce3e610 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.github/FUNDING.yml @@ -0,0 +1,3 @@ +# These are supported funding model platforms + +github: phip1611 diff --git a/vendor/tar-no-std-0.3.5/.github/dependabot.yml b/vendor/tar-no-std-0.3.5/.github/dependabot.yml new file mode 100644 index 0000000000..3e8e7a928b --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.github/dependabot.yml @@ -0,0 +1,15 @@ +version: 2 +updates: + - package-ecosystem: cargo + directory: "/" + schedule: + interval: monthly + open-pull-requests-limit: 10 + ignore: + - dependency-name: "*" + update-types: [ "version-update:semver-patch" ] + - package-ecosystem: github-actions + directory: "/" + schedule: + interval: monthly + open-pull-requests-limit: 10 diff --git a/vendor/tar-no-std-0.3.5/.github/workflows/qa.yml b/vendor/tar-no-std-0.3.5/.github/workflows/qa.yml new file mode 100644 index 0000000000..86215c090e --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.github/workflows/qa.yml @@ -0,0 +1,12 @@ +name: QA + +on: [ push, pull_request, merge_group ] + +jobs: + spellcheck: + name: Spellcheck + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + # Executes "typos ." + - uses: crate-ci/typos@v1.34.0 diff --git a/vendor/tar-no-std-0.3.5/.github/workflows/rust.yml b/vendor/tar-no-std-0.3.5/.github/workflows/rust.yml new file mode 100644 index 0000000000..f8c9a917a0 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.github/workflows/rust.yml @@ -0,0 +1,82 @@ +name: Build + +on: [ push, pull_request, merge_group ] + +env: + CARGO_TERM_COLOR: always + +jobs: + build: + runs-on: "${{ matrix.runs-on }}" + strategy: + matrix: + runs-on: + - windows-latest + - ubuntu-latest + rust: + - stable + - nightly + - 1.76.0 # MSVR + steps: + - uses: actions/checkout@v4 + - name: Setup Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + toolchain: "${{ matrix.rust }}" + - uses: Swatinem/rust-cache@v2 + with: + key: "${{ matrix.runs-on }}-${{ matrix.rust }}" + - name: Build + run: cargo build --all-targets --verbose --features alloc + # use some arbitrary no_std target + - name: Install no_std target thumbv7em-none-eabihf + run: rustup target add thumbv7em-none-eabihf + - name: Build (no_std) + run: cargo build --verbose --target thumbv7em-none-eabihf --features alloc + - name: Run tests + run: cargo test --verbose --features alloc + + miri: + runs-on: "${{ matrix.runs-on }}" + needs: + # Logical dependency and wait for cache to be present + - build + strategy: + matrix: + runs-on: + - ubuntu-latest + rust: + - nightly + steps: + - uses: actions/checkout@v4 + - name: Setup Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + toolchain: "${{ matrix.rust }}" + - uses: Swatinem/rust-cache@v2 + with: + key: "${{ matrix.runs-on }}-${{ matrix.rust }}" + - run: rustup component add miri + - run: cargo miri test --tests + + style_checks: + runs-on: ubuntu-latest + strategy: + matrix: + rust: + - stable + steps: + - uses: actions/checkout@v4 + - name: Setup Rust toolchain + uses: dtolnay/rust-toolchain@stable + with: + toolchain: "${{ matrix.rust }}" + - uses: Swatinem/rust-cache@v2 + with: + key: "${{ matrix.runs-on }}-${{ matrix.rust }}" + - name: Rustfmt + run: cargo fmt -- --check + - name: Clippy + run: cargo clippy --features alloc + - name: Rustdoc + run: cargo doc --no-deps --document-private-items --features alloc diff --git a/vendor/tar-no-std-0.3.5/.gitignore b/vendor/tar-no-std-0.3.5/.gitignore new file mode 100644 index 0000000000..0b42d2ddfd --- /dev/null +++ b/vendor/tar-no-std-0.3.5/.gitignore @@ -0,0 +1 @@ +/target diff --git a/vendor/tar-no-std-0.3.5/CHANGELOG.md b/vendor/tar-no-std-0.3.5/CHANGELOG.md new file mode 100644 index 0000000000..fdad19f0da --- /dev/null +++ b/vendor/tar-no-std-0.3.5/CHANGELOG.md @@ -0,0 +1,52 @@ +# Unreleased + +# v0.3.5 (2025-08-08) + +- Increased lifetime of `TarArchiveRef::entries` +- Dropped dependency on `memchr` + +# v0.3.4 (2025-05-13) + +- Fixed a bug when data fills an entire block + +# v0.3.3 (2025-03-20) + +- Added `ArchiveEntry::posix_header()` to get metadata for an entry + +# v0.3.2 (2024-08-02) + +- `TarArchive::entries` is now `#[must_use]` + +# v0.3.1 (2024-05-03) + +- More sanity checks with malformed Tar archives. + +# v0.3.0 (2024-05-03) + +- MSRV is now 1.76 stable +- added support for more Tar archives + - 256 character long filename support (prefix + name) + - add support for space terminated numbers + - non-null terminated names + - iterate over directories: read regular files from directories + - more info: +- `TarArchive[Ref]::new` now returns a result +- added `unstable` feature with enhanced functionality for `nightly` compilers + - error types implement `core::error::Error` +- various bug fixes and code improvements +- better error reporting / less panics + +Special thanks to the following external contributors or helpers: + +- https://github.com/thenhnn: provide me with a bunch of Tar archives coming + from a fuzzer +- https://github.com/schnoberts1 implemented 256 character long filenames (ustar + Tar format) + +# v0.2.0 (2023-04-11) + +- MSRV is 1.60.0 +- bitflags bump: 1.x -> 2.x +- few internal code improvements (less possible panics) +- `Mode::to_flags` now returns a Result +- Feature `all` was removed. Use `alloc` instead. diff --git a/vendor/tar-no-std-0.3.5/Cargo.toml b/vendor/tar-no-std-0.3.5/Cargo.toml new file mode 100644 index 0000000000..41ba922ec5 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/Cargo.toml @@ -0,0 +1,45 @@ +[package] +name = "tar-no-std" +description = """ +Library to read Tar archives (by GNU Tar) in `no_std` contexts with zero allocations. +The crate is simple and only supports reading of "basic" archives, therefore no extensions, such +as GNU Longname. The maximum supported file name length is 256 characters excluding the NULL-byte +(using the tar name/prefix longname implementation).The maximum supported file size is 8GiB. +Directories are supported, but only regular fields are yielded in iteration. +""" +version = "0.3.5" +edition = "2021" +keywords = ["tar", "tarball", "archive"] +categories = ["data-structures", "no-std", "parser-implementations"] +readme = "README.md" +license = "MIT" +homepage = "https://github.com/phip1611/tar-no-std" +repository = "https://github.com/phip1611/tar-no-std" +documentation = "https://docs.rs/tar-no-std" +rust-version = "1.76.0" +exclude = [ + "tests" +] + +# required because "env_logger" uses "log" but with dependency to std. +resolver = "2" + +[features] +default = [] +alloc = [] +unstable = [] # requires nightly + +[[example]] +name = "alloc_feature" +required-features = ["alloc"] + +[dependencies] +bitflags = "2.5" +log = { version = "~0.4", default-features = false } +num-traits = { version = "~0.2", default-features = false } + +[dev-dependencies] +env_logger = "0.11" + +[package.metadata.docs.rs] +all-features = true diff --git a/vendor/tar-no-std-0.3.5/LICENSE b/vendor/tar-no-std-0.3.5/LICENSE new file mode 100644 index 0000000000..cad0a68680 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/vendor/tar-no-std-0.3.5/README.md b/vendor/tar-no-std-0.3.5/README.md new file mode 100644 index 0000000000..3d46f8be92 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/README.md @@ -0,0 +1,65 @@ +# `tar-no-std` - Parse Tar Archives (Tarballs) + +_Due to historical reasons, there are several formats of Tar archives. All of +them are based on the same principles, but have some subtle differences that +often make them incompatible with each other._ [(reference)](https://www.gnu.org/software/tar/manual/html_section/Formats.html) + +Library to read Tar archives in `no_std` environments with zero allocations. If +you have a standard environment and need full feature support, I recommend the +use of instead. + +## Limitations + +This crate is simple and focuses on reading files and their content from a Tar +archive. Historic basic Tar and ustar [formats](https://www.gnu.org/software/tar/manual/html_section/Formats.html) +are supported. Other formats may work, but likely without all supported +features. GNU Extensions such as sparse files, incremental archives, and long +filename extension are not supported. + +The maximum supported file name length is 256 characters excluding the +NULL-byte (using the Tar name/prefix longname implementation of ustar). The +maximum supported file size is 8GiB. Directories are supported, but only regular +fields are yielded in iteration. The path is reflected in their file name. + +## Use Case + +This library is useful, if you write a kernel or a similar low-level +application, which needs "a bunch of files" from an archive (like an +"init ramdisk"). The Tar file could for example come as a Multiboot2 boot module +provided by the bootloader. + +## Example + +```rust +use tar_no_std::TarArchiveRef; + +fn main() { + // init a logger (optional) + std::env::set_var("RUST_LOG", "trace"); + env_logger::init(); + + // also works in no_std environment (except the println!, of course) + let archive = include_bytes!("../tests/gnu_tar_default.tar"); + let archive = TarArchiveRef::new(archive).unwrap(); + // Vec needs an allocator of course, but the library itself doesn't need one + let entries = archive.entries().collect::>(); + println!("{:#?}", entries); +} +``` + +## Cargo Feature + +This crate allows the usage of the additional Cargo build time feature `alloc`. +When this is active, the crate also provides the type `TarArchive`, which owns +the data on the heap. The `unstable` feature provides additional convenience +only available on the nightly channel. + +## Compression (`tar.gz`) + +If your Tar file is compressed, e.g. by `.tar.gz`/`gzip`, you need to uncompress +the bytes first (e.g. by a *gzip* library). Afterwards, this crate can read the +Tar archive format from the uncompressed bytes. + +## MSRV + +The MSRV is 1.76.0 stable. diff --git a/vendor/tar-no-std-0.3.5/build.sh b/vendor/tar-no-std-0.3.5/build.sh new file mode 100644 index 0000000000..bab461eeb6 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/build.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash + +cargo build --all-targets --verbose --features alloc +# use some random no_std target +rustup target add thumbv7em-none-eabihf +cargo build --verbose --target thumbv7em-none-eabihf --features alloc +cargo test --verbose --features alloc + +cargo fmt -- --check +cargo +1.60.0 clippy --features alloc +cargo +1.60.0 doc --no-deps --document-private-items --features alloc diff --git a/vendor/tar-no-std-0.3.5/examples/alloc_feature.rs b/vendor/tar-no-std-0.3.5/examples/alloc_feature.rs new file mode 100644 index 0000000000..21b1571c6e --- /dev/null +++ b/vendor/tar-no-std-0.3.5/examples/alloc_feature.rs @@ -0,0 +1,45 @@ +/* +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +*/ + +use tar_no_std::TarArchive; + +fn main() { + // log: not mandatory + std::env::set_var("RUST_LOG", "trace"); + std::env::set_var("RUST_LOG_STYLE", "always"); + env_logger::init(); + + // also works in no_std environment (except the println!, of course) + let archive = include_bytes!("../tests/gnu_tar_default.tar"); + let archive_heap_owned = archive.to_vec().into_boxed_slice(); + let archive = TarArchive::new(archive_heap_owned).unwrap(); + // Vec needs an allocator of course, but the library itself doesn't need one + let entries = archive.entries().collect::>(); + println!("{:#?}", entries); + println!("content of last file:"); + println!( + "{:#?}", + entries[2].data_as_str().expect("Should be valid UTF-8") + ); +} diff --git a/vendor/tar-no-std-0.3.5/examples/minimal.rs b/vendor/tar-no-std-0.3.5/examples/minimal.rs new file mode 100644 index 0000000000..48c6e7be3b --- /dev/null +++ b/vendor/tar-no-std-0.3.5/examples/minimal.rs @@ -0,0 +1,43 @@ +/* +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +*/ +use tar_no_std::TarArchiveRef; + +fn main() { + // log: not mandatory + std::env::set_var("RUST_LOG", "trace"); + std::env::set_var("RUST_LOG_STYLE", "always"); + env_logger::init(); + + // also works in no_std environment (except the println!, of course) + let archive = include_bytes!("../tests/gnu_tar_default.tar"); + let archive = TarArchiveRef::new(archive).unwrap(); + // Vec needs an allocator of course, but the library itself doesn't need one + let entries = archive.entries().collect::>(); + println!("{:#?}", entries); + println!("content of last file:"); + println!( + "{:#?}", + entries[2].data_as_str().expect("Should be valid UTF-8") + ); +} diff --git a/vendor/tar-no-std-0.3.5/src/archive.rs b/vendor/tar-no-std-0.3.5/src/archive.rs new file mode 100644 index 0000000000..3261492834 --- /dev/null +++ b/vendor/tar-no-std-0.3.5/src/archive.rs @@ -0,0 +1,713 @@ +/* +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +*/ +//! Module for [`TarArchiveRef`]. If the `alloc`-feature is enabled, this crate +//! also exports `TarArchive`, which owns data on the heap. + +use crate::header::PosixHeader; +use crate::tar_format_types::TarFormatString; +use crate::{BLOCKSIZE, POSIX_1003_MAX_FILENAME_LEN}; +#[cfg(feature = "alloc")] +use alloc::boxed::Box; +use core::fmt::{Debug, Display, Formatter}; +use core::str::Utf8Error; +use log::{error, warn}; + +/// Minimum amount of blocks that an archive must have to be considered sane. +/// - one header block +/// - two terminating zero blocks +pub const MIN_BLOCK_COUNT: usize = 3; + +/// Describes an entry in an archive. +/// Currently only supports files but no directories. +pub struct ArchiveEntry<'a> { + filename: TarFormatString, + data: &'a [u8], + size: usize, + posix_header: &'a PosixHeader, +} + +#[allow(unused)] +impl<'a> ArchiveEntry<'a> { + const fn new( + filename: TarFormatString, + data: &'a [u8], + posix_header: &'a PosixHeader, + ) -> Self { + ArchiveEntry { + filename, + data, + size: data.len(), + posix_header, + } + } + + /// Filename of the entry with a maximum of 100 characters (including the + /// terminating NULL-byte). + #[must_use] + pub const fn filename(&self) -> TarFormatString<{ POSIX_1003_MAX_FILENAME_LEN }> { + self.filename + } + + /// Data of the file. + #[must_use] + pub const fn data(&self) -> &'a [u8] { + self.data + } + + /// Data of the file as string slice, if data is valid UTF-8. + /// + /// # Errors + /// Returns a [`Utf8Error`] error for invalid strings. + #[allow(clippy::missing_const_for_fn)] + pub fn data_as_str(&self) -> Result<&'a str, Utf8Error> { + core::str::from_utf8(self.data) + } + + /// Filesize in bytes. + #[must_use] + pub const fn size(&self) -> usize { + self.size + } + + /// Returns the [`PosixHeader`] for the entry. + #[must_use] + pub const fn posix_header(&self) -> &PosixHeader { + self.posix_header + } +} + +impl Debug for ArchiveEntry<'_> { + fn fmt(&self, f: &mut Formatter<'_>) -> core::fmt::Result { + f.debug_struct("ArchiveEntry") + .field("filename", &self.filename().as_str()) + .field("size", &self.size()) + .field("data", &"") + .finish() + } +} + +/// The data is corrupt and doesn't present a valid Tar archive. Reasons for +/// that are: +/// - the data is empty +/// - the data is not a multiple of 512 (the BLOCKSIZE) +/// - the data is not at least [`MIN_BLOCK_COUNT`] blocks long +#[derive(Copy, Clone, Debug, PartialEq, Eq)] +pub struct CorruptDataError; + +impl Display for CorruptDataError { + fn fmt(&self, f: &mut Formatter<'_>) -> core::fmt::Result { + Debug::fmt(self, f) + } +} + +#[cfg(feature = "unstable")] +impl core::error::Error for CorruptDataError {} + +/// Type that owns bytes on the heap, that represents a Tar archive. +/// Unlike [`TarArchiveRef`], this type takes ownership of the data. +/// +/// This is only available with the `alloc` feature of this crate. +#[cfg(feature = "alloc")] +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct TarArchive { + data: Box<[u8]>, +} + +#[cfg(feature = "alloc")] +impl TarArchive { + /// Creates a new archive wrapper type. The provided byte array is + /// interpreted as bytes in Tar archive format. + /// + /// Returns an error, if the sanity checks report problems. + pub fn new(data: Box<[u8]>) -> Result { + TarArchiveRef::validate(&data).map(|_| Self { data }) + } + + /// Iterates over all entries of the Tar archive. + /// Returns items of type [`ArchiveEntry`]. + /// See also [`ArchiveEntryIterator`]. + #[must_use] + pub fn entries(&self) -> ArchiveEntryIterator<'_> { + ArchiveEntryIterator::new(self.data.as_ref()) + } +} + +#[cfg(feature = "alloc")] +impl From> for TarArchive { + fn from(data: Box<[u8]>) -> Self { + Self::new(data).unwrap() + } +} + +#[cfg(feature = "alloc")] +impl From for Box<[u8]> { + fn from(ar: TarArchive) -> Self { + ar.data + } +} + +/// Wrapper type around bytes, which represents a Tar archive. To iterate the +/// entries, use [`TarArchiveRef::entries`]. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct TarArchiveRef<'a> { + data: &'a [u8], +} + +#[allow(unused)] +impl<'a> TarArchiveRef<'a> { + /// Creates a new archive wrapper type. The provided byte array is + /// interpreted as bytes in Tar archive format. + /// + /// # Errors + /// Returns an [`CorruptDataError`], if the sanity checks fail. + pub fn new(data: &'a [u8]) -> Result { + Self::validate(data).map(|()| Self { data }) + } + + fn validate(data: &'a [u8]) -> Result<(), CorruptDataError> { + let is_malformed = (data.len() % BLOCKSIZE) != 0; + let has_min_block_count = data.len() / BLOCKSIZE >= MIN_BLOCK_COUNT; + (!data.is_empty() && !is_malformed && has_min_block_count) + .then_some(()) + .ok_or(CorruptDataError) + } + + /// Creates an [`ArchiveEntryIterator`]. + #[must_use] + pub fn entries(&self) -> ArchiveEntryIterator<'a> { + ArchiveEntryIterator::new(self.data) + } +} + +/// Iterates over the headers of the Tar archive. +#[derive(Debug)] +pub struct ArchiveHeaderIterator<'a> { + archive_data: &'a [u8], + next_hdr_block_index: usize, +} + +impl<'a> ArchiveHeaderIterator<'a> { + /// Creates a new iterator. + /// + /// # Panics + /// Panics if the slice is zero or not a multiple of `BLOCKSIZE`. + #[must_use] + pub fn new(archive: &'a [u8]) -> Self { + assert!(!archive.is_empty()); + assert_eq!(archive.len() % BLOCKSIZE, 0); + Self { + archive_data: archive, + next_hdr_block_index: 0, + } + } + + /// Parse the memory at the given block as [`PosixHeader`]. + fn block_as_header(&self, block_index: usize) -> &'a PosixHeader { + unsafe { + self.archive_data + .as_ptr() + .add(block_index * BLOCKSIZE) + .cast::() + .as_ref() + .unwrap() + } + } +} + +type BlockIndex = usize; + +impl<'a> Iterator for ArchiveHeaderIterator<'a> { + type Item = (BlockIndex, &'a PosixHeader); + + /// Returns the next header. Internally, it updates the necessary data + /// structures to not read the same header multiple times. + /// + /// This returns `None` if either no further headers are found or if a + /// header can't be parsed. + fn next(&mut self) -> Option { + let total_block_count = self.archive_data.len() / BLOCKSIZE; + if self.next_hdr_block_index >= total_block_count { + warn!("Invalid block index. Probably the Tar is corrupt: an header had an invalid payload size"); + return None; + } + + let hdr = self.block_as_header(self.next_hdr_block_index); + let block_index = self.next_hdr_block_index; + + // Start at next block on next iteration. + self.next_hdr_block_index += 1; + + // A fully zeroed block is the archive's end-of-archive marker (or, if this archive is + // corrupt, mid-stream junk); either way it has no typeflag or size to parse. Bail out + // before attempting to interpret it as a header: a NUL typeflag byte otherwise parses + // as `AREGTYPE` (regular file) per the tar spec's own "NUL means old-style regular + // file" rule, which would then fail to parse the (also all-zero) size field and abort + // iteration here -- silently truncating every real entry that follows in the archive, + // instead of letting the caller's own is_zero_block()-based end-of-archive detection + // run. + if hdr.is_zero_block() { + return Some((block_index, hdr)); + } + + // We only update the block index for types that have a payload. + // In directory entries, for example, the size field has other + // semantics. See spec. + if let Ok(typeflag) = hdr.typeflag.try_to_type_flag() { + if typeflag.is_regular_file() { + let payload_block_count = hdr + .payload_block_count() + .inspect_err(|e| { + log::error!("Unparsable size ({e:?}) in header {hdr:#?}"); + }) + .ok()?; + self.next_hdr_block_index += payload_block_count; + } + } + + Some((block_index, hdr)) + } +} + +impl ExactSizeIterator for ArchiveEntryIterator<'_> {} + +/// Iterator over the files of the archive. +/// +/// Only regular files are supported, but not directories, links, or other +/// special types ([`crate::TypeFlag`]). The full path to files is reflected +/// in their file name. +#[derive(Debug)] +pub struct ArchiveEntryIterator<'a>(ArchiveHeaderIterator<'a>); + +impl<'a> ArchiveEntryIterator<'a> { + fn new(archive: &'a [u8]) -> Self { + Self(ArchiveHeaderIterator::new(archive)) + } + + fn next_hdr(&mut self) -> Option<(BlockIndex, &'a PosixHeader)> { + self.0.next() + } +} + +impl<'a> Iterator for ArchiveEntryIterator<'a> { + type Item = ArchiveEntry<'a>; + + fn next(&mut self) -> Option { + let (mut block_index, mut hdr) = self.next_hdr()?; + + // Ignore directory entries, i.e. yield only regular files. Works as + // filenames in tarballs are fully specified, e.g. dirA/dirB/file1 + while !hdr + .typeflag + .try_to_type_flag() + .inspect_err(|e| error!("Invalid TypeFlag: {e:?}")) + .ok()? + .is_regular_file() + { + warn!( + "Skipping entry of type {:?} (not supported yet)", + hdr.typeflag + ); + + // Update properties. + (block_index, hdr) = self.next_hdr()?; + } + + // check if we found end of archive (two zero blocks) + if hdr.is_zero_block() { + if self.next_hdr()?.1.is_zero_block() { + // found end + return None; + } + + panic!("should never have a missing double zero block: is the Tar archive corrupt?"); + } + + let payload_size: usize = hdr + .size + .as_number() + .inspect_err(|e| error!("Can't parse the file size from the header. {e:#?}")) + .ok()?; + + let idx_first_data_block = block_index + 1; + let idx_begin = idx_first_data_block * BLOCKSIZE; + let idx_end_exclusive = idx_begin + payload_size; + + // This doesn't subtract with overflow as we ensured a minimum size in + // the constructor. + let max_data_end_index_exclusive = self.0.archive_data.len() - 2 * BLOCKSIZE; + if idx_end_exclusive > max_data_end_index_exclusive { + warn!("Invalid Tar. The size of the payload ({payload_size}) is larger than what is valid"); + return None; + } + + let file_bytes = &self.0.archive_data[idx_begin..idx_end_exclusive]; + + let mut filename = + TarFormatString::::new([0; POSIX_1003_MAX_FILENAME_LEN]); + + // POXIS_1003 long filename check + // https://docs.scinet.utoronto.ca/index.php/(POSIX_1003.1_USTAR) + if ( + hdr.magic.as_str(), + hdr.version.as_str(), + hdr.prefix.is_empty(), + ) == (Ok("ustar"), Ok("00"), false) + { + filename.append(&hdr.prefix); + filename.append(&TarFormatString::<1>::new([b'/'])); + } + filename.append(&hdr.name); + Some(ArchiveEntry::new(filename, file_bytes, hdr)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::TarFormatOctal; + use std::vec::Vec; + + #[test] + #[rustfmt::skip] + fn test_constructor_returns_error() { + assert_eq!(TarArchiveRef::new(&[0]), Err(CorruptDataError)); + assert_eq!(TarArchiveRef::new(&[]), Err(CorruptDataError)); + assert!(TarArchiveRef::new(&[0; BLOCKSIZE * MIN_BLOCK_COUNT]).is_ok()); + + #[cfg(feature = "alloc")] + { + assert_eq!(TarArchive::new(vec![].into_boxed_slice()), Err(CorruptDataError)); + assert_eq!(TarArchive::new(vec![0].into_boxed_slice()), Err(CorruptDataError)); + assert!(TarArchive::new(vec![0; BLOCKSIZE * MIN_BLOCK_COUNT].into_boxed_slice()).is_ok()); + }; + } + + #[test] + fn test_header_iterator() { + let archive = include_bytes!("../tests/gnu_tar_default.tar"); + let iter = ArchiveHeaderIterator::new(archive); + let names = iter + .map(|(_i, hdr)| hdr.name.as_str().unwrap()) + .collect::>(); + + assert_eq!( + names.as_slice(), + &[ + "bye_world_513b.txt", + "hello_world_513b.txt", + "hello_world.txt", + ] + ) + } + + /// The test here is that no panics occur. + #[test] + fn test_print_archive_headers() { + let data = include_bytes!("../tests/gnu_tar_default.tar"); + + let iter = ArchiveHeaderIterator::new(data); + let entries = iter.map(|(_, hdr)| hdr).collect::>(); + println!("{:#?}", entries); + } + + /// The test here is that no panics occur. + #[test] + fn test_print_archive_list() { + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_default.tar")).unwrap(); + let entries = archive.entries().collect::>(); + println!("{:#?}", entries); + } + + /// Tests various weird (= invalid, corrupt) tarballs that are bundled + /// within this file. The tarball(s) originate from a fuzzing process from a + /// GitHub contributor [0]. + /// + /// The test succeeds if no panics occur. + /// + /// [0] https://github.com/phip1611/tar-no-std/issues/12#issuecomment-2092632090 + #[test] + fn test_weird_fuzzing_tarballs() { + /*std::env::set_var("RUST_LOG", "trace"); + std::env::set_var("RUST_LOG_STYLE", "always"); + env_logger::init();*/ + + let main_tarball = + TarArchiveRef::new(include_bytes!("../tests/weird_fuzzing_tarballs.tar")).unwrap(); + + let mut all_entries = vec![]; + for tarball in main_tarball.entries() { + let tarball = TarArchiveRef::new(tarball.data()).unwrap(); + for entry in tarball.entries() { + all_entries.push(entry.filename()); + } + } + + // Test succeeds if this works without a panic. + for entry in all_entries { + eprintln!("\"{entry:?}\","); + } + } + + /// Tests to read the entries from existing archives in various Tar flavors. + #[test] + fn test_archive_entries() { + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_default.tar")).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_gnu.tar")).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_oldgnu.tar")).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + + // UNSUPPORTED. Uses extensions. + /*let archive = TarArchive::new(include_bytes!("../tests/gnu_tar_pax.tar")); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries);*/ + + // UNSUPPORTED. Uses extensions. + /*let archive = TarArchive::new(include_bytes!("../tests/gnu_tar_posix.tar")); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries);*/ + + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_ustar.tar")).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + + let archive = TarArchiveRef::new(include_bytes!("../tests/gnu_tar_v7.tar")).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + } + + /// Tests to read the entries from an existing tarball with a directory in it + #[test] + fn test_archive_with_long_dir_entries() { + // tarball created with: + // $ cd tests; gtar --format=ustar -cf gnu_tar_ustar_long.tar 012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678 01234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234/ABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJ + let archive = + TarArchiveRef::new(include_bytes!("../tests/gnu_tar_ustar_long.tar")).unwrap(); + let entries = archive.entries().collect::>(); + + assert_eq!(entries.len(), 2); + // Maximum length of a directory and name when the directory itself is tar'd + assert_entry_content(&entries[0], "012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678/ABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJ", 7); + // Maximum length of a directory and name when only the file is tar'd. + assert_entry_content(&entries[1], "01234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789012345678901234/ABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJABCDEFGHIJ", 7); + } + + #[test] + fn test_archive_with_deep_dir_entries() { + // tarball created with: + // $ cd tests; gtar --format=ustar -cf gnu_tar_ustar_deep.tar 0123456789 + let archive = + TarArchiveRef::new(include_bytes!("../tests/gnu_tar_ustar_deep.tar")).unwrap(); + let entries = archive.entries().collect::>(); + + assert_eq!(entries.len(), 1); + assert_entry_content(&entries[0], "0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/0123456789/empty", 0); + } + + #[test] + fn test_default_archive_with_dir_entries() { + // tarball created with: + // $ gtar -cf tests/gnu_tar_default_with_dir.tar --exclude '*.tar' --exclude '012345678*' tests + let archive = + TarArchiveRef::new(include_bytes!("../tests/gnu_tar_default_with_dir.tar")).unwrap(); + let entries = archive.entries().collect::>(); + + assert_archive_with_dir_content(&entries); + } + + #[test] + fn test_ustar_archive_with_dir_entries() { + // tarball created with: + // $(osx) tar -cf tests/mac_tar_ustar_with_dir.tar --format=ustar --exclude '*.tar' --exclude '012345678*' tests + let archive = + TarArchiveRef::new(include_bytes!("../tests/mac_tar_ustar_with_dir.tar")).unwrap(); + let entries = archive.entries().collect::>(); + + assert_archive_with_dir_content(&entries); + } + + #[test] + fn test_data_fills_entire_block() { + // header, data block, 2 zero blocks + let mut data = [0_u8; 4 * BLOCKSIZE]; + + // Fill payload: We have a full block + { + data[BLOCKSIZE..BLOCKSIZE * 2].fill(0xff); + } + + // Write header + { + let hdr = unsafe { data.as_mut_ptr().cast::().as_mut().unwrap() }; + let blocksize_octal = "1000\0\0\0\0\0\0\0\0" /* BLOCKSIZE */; + let blocksize_octal_bytes: [u8; 12] = { + let mut val = [0; 12]; + val.copy_from_slice(blocksize_octal.as_bytes()); + val + }; + hdr.size = TarFormatOctal::new(blocksize_octal_bytes); + } + let archive = TarArchiveRef::new(data.as_slice()).unwrap(); + let entries = archive.entries().collect::>(); + assert_eq!(entries.len(), 1); + assert!(entries[0].data.iter().all(|&v| v == 0xff)); + } + + /// Like [`test_archive_entries`] but with additional `alloc` functionality. + #[cfg(feature = "alloc")] + #[test] + fn test_archive_entries_alloc() { + let data = include_bytes!("../tests/gnu_tar_default.tar") + .to_vec() + .into_boxed_slice(); + let archive = TarArchive::new(data.clone()).unwrap(); + let entries = archive.entries().collect::>(); + assert_archive_content(&entries); + + // Test that the archive can be transformed into owned heap data. + assert_eq!(data, archive.into()); + } + + /// Test that the entry's contents match the expected content. + fn assert_entry_content(entry: &ArchiveEntry, filename: &str, size: usize) { + assert_eq!(entry.filename().as_str(), Ok(filename)); + assert_eq!(entry.size(), size); + assert_eq!(entry.data().len(), size); + } + + /// Tests that the parsed archive matches the expected order. The tarballs + /// the tests directory were created once by me with files in the order + /// specified in this test. + fn assert_archive_content(entries: &[ArchiveEntry]) { + use crate::ModeFlags; + let permissions = ModeFlags::OwnerRead + | ModeFlags::OwnerWrite + | ModeFlags::OwnerExec + | ModeFlags::GroupRead + | ModeFlags::GroupWrite + | ModeFlags::GroupExec + | ModeFlags::OthersRead + | ModeFlags::OthersWrite + | ModeFlags::OthersExec; + let rw_rw_r__ = ModeFlags::OwnerRead + | ModeFlags::OwnerWrite + | ModeFlags::GroupRead + | ModeFlags::GroupWrite + | ModeFlags::OthersRead; + // Rust complains otherwise, but this is intentionally written this way. + #[allow(non_snake_case)] + let rw_r__r__ = ModeFlags::OwnerRead + | ModeFlags::OwnerWrite + | ModeFlags::GroupRead + | ModeFlags::OthersRead; + + assert_eq!(entries.len(), 3); + + assert_entry_content(&entries[0], "bye_world_513b.txt", 513); + assert_eq!( + entries[0].data_as_str().expect("Should be valid UTF-8"), + // .replace: Ensure that the test also works on Windows + include_str!("../tests/bye_world_513b.txt").replace("\r\n", "\n") + ); + assert_eq!( + entries[0] + .posix_header() + .mode + .to_flags() + .unwrap() + .intersection(permissions), + rw_rw_r__ + ); + + // Test that an entry that needs two 512 byte data blocks is read + // properly. + assert_entry_content(&entries[1], "hello_world_513b.txt", 513); + assert_eq!( + entries[1].data_as_str().expect("Should be valid UTF-8"), + // .replace: Ensure that the test also works on Windows + include_str!("../tests/hello_world_513b.txt").replace("\r\n", "\n") + ); + assert_eq!( + entries[1] + .posix_header() + .mode + .to_flags() + .unwrap() + .intersection(permissions), + rw_rw_r__ + ); + + assert_entry_content(&entries[2], "hello_world.txt", 12); + assert_eq!( + entries[2].data_as_str().expect("Should be valid UTF-8"), + "Hello World\n", + "file content must match" + ); + assert_eq!( + entries[2] + .posix_header() + .mode + .to_flags() + .unwrap() + .intersection(permissions), + rw_r__r__ + ); + } + + /// Tests that the parsed archive matches the expected order and the filename includes + /// the directory name. The tarballs the tests directory were created once by me with files + /// in the order specified in this test. + fn assert_archive_with_dir_content(entries: &[ArchiveEntry]) { + assert_eq!(entries.len(), 3); + + assert_entry_content(&entries[0], "tests/hello_world.txt", 12); + assert_eq!( + entries[0].data_as_str().expect("Should be valid UTF-8"), + "Hello World\n", + "file content must match" + ); + + // Test that an entry that needs two 512 byte data blocks is read + // properly. + assert_entry_content(&entries[1], "tests/bye_world_513b.txt", 513); + assert_eq!( + entries[1].data_as_str().expect("Should be valid UTF-8"), + // .replace: Ensure that the test also works on Windows + include_str!("../tests/bye_world_513b.txt").replace("\r\n", "\n") + ); + + assert_entry_content(&entries[2], "tests/hello_world_513b.txt", 513); + assert_eq!( + entries[2].data_as_str().expect("Should be valid UTF-8"), + // .replace: Ensure that the test also works on Windows + include_str!("../tests/hello_world_513b.txt").replace("\r\n", "\n") + ); + } +} diff --git a/vendor/tar-no-std-0.3.5/src/header.rs b/vendor/tar-no-std-0.3.5/src/header.rs new file mode 100644 index 0000000000..301feb176a --- /dev/null +++ b/vendor/tar-no-std-0.3.5/src/header.rs @@ -0,0 +1,403 @@ +/* +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +*/ +//! TAR header definition taken from . +//! A Tar-archive is a collection of 512-byte sized blocks. Unfortunately there are several +//! TAR-like archive specifications. An Overview can be found here: +//! +//! +//! This library focuses on extracting files from the GNU Tar format. + +#![allow(non_upper_case_globals)] + +use crate::{TarFormatDecimal, TarFormatOctal, TarFormatString, BLOCKSIZE, NAME_LEN, PREFIX_LEN}; +use core::fmt::{Debug, Display, Formatter}; +use core::num::ParseIntError; + +/// Errors that may happen when parsing the [`ModeFlags`]. +#[derive(Debug)] +pub enum ModeError { + ParseInt(ParseIntError), + IllegalMode, +} + +/// Wrapper around the UNIX file permissions given in octal ASCII. +#[derive(Copy, Clone, PartialEq, Eq)] +#[repr(transparent)] +pub struct Mode(TarFormatOctal<8>); + +impl Mode { + /// Parses the [`ModeFlags`] from the mode string. + /// + /// # Errors + /// Returns [`ModeError`] for invalid values. + pub fn to_flags(self) -> Result { + let bits = self.0.as_number::().map_err(ModeError::ParseInt)?; + ModeFlags::from_bits(bits).ok_or(ModeError::IllegalMode) + } +} + +impl Debug for Mode { + fn fmt(&self, f: &mut Formatter<'_>) -> core::fmt::Result { + Debug::fmt(&self.to_flags(), f) + } +} + +#[derive(Copy, Clone, Debug, PartialOrd, PartialEq, Eq)] +pub struct InvalidTypeFlagError(u8); + +impl Display for InvalidTypeFlagError { + fn fmt(&self, f: &mut Formatter<'_>) -> core::fmt::Result { + f.write_fmt(format_args!("{:x} is not a valid TypeFlag", self.0)) + } +} + +#[cfg(feature = "unstable")] +impl core::error::Error for InvalidTypeFlagError {} + +#[derive(Copy, Clone, PartialOrd, PartialEq, Eq)] +pub struct TypeFlagRaw(u8); + +impl TypeFlagRaw { + /// Tries to parse the underlying value as [`TypeFlag`]. This fails if the + /// Tar file is corrupt and the type is invalid. + /// + /// # Errors + /// Returns [`InvalidTypeFlagError`] for invalid values. + pub fn try_to_type_flag(self) -> Result { + TypeFlag::try_from(self) + } +} + +impl Debug for TypeFlagRaw { + fn fmt(&self, f: &mut Formatter<'_>) -> core::fmt::Result { + Debug::fmt(&self.try_to_type_flag(), f) + } +} + +/// Describes the kind of payload, that follows after a +/// [`PosixHeader`]. The properties of this payload are +/// described inside the header. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +#[repr(u8)] +#[allow(unused)] +pub enum TypeFlag { + /// Represents a regular file. In order to be compatible with older versions of tar, a typeflag + /// value of AREGTYPE should be silently recognized as a regular file. New archives should be + /// created using REGTYPE. Also, for backward compatibility, tar treats a regular file whose + /// name ends with a slash as a directory. + REGTYPE = b'0', + /// Represents a regular file. In order to be compatible with older versions of tar, a typeflag + /// value of AREGTYPE should be silently recognized as a regular file. New archives should be + /// created using REGTYPE. Also, for backward compatibility, tar treats a regular file whose + /// name ends with a slash as a directory. + AREGTYPE = b'\0', + /// This flag represents a file linked to another file, of any type, previously archived. Such + /// files are identified in Unix by each file having the same device and inode number. The + /// linked-to name is specified in the linkname field with a trailing null. + LINK = b'1', + /// This represents a symbolic link to another file. The linked-to name is specified in the + /// linkname field with a trailing null. + SYMTYPE = b'2', + /// Represents character special files and block special files respectively. In this case the + /// devmajor and devminor fields will contain the major and minor device numbers respectively. + /// Operating systems may map the device specifications to their own local specification, or + /// may ignore the entry. + CHRTYPE = b'3', + /// Represents character special files and block special files respectively. In this case the + /// devmajor and devminor fields will contain the major and minor device numbers respectively. + /// Operating systems may map the device specifications to their own local specification, or + /// may ignore the entry. + BLKTYPE = b'4', + /// This flag specifies a directory or sub-directory. The directory name in the name field + /// should end with a slash. On systems where disk allocation is performed on a directory + /// basis, the size field will contain the maximum number of bytes (which may be rounded to + /// the nearest disk block allocation unit) which the directory may hold. A size field of zero + /// indicates no such limiting. Systems which do not support limiting in this manner should + /// ignore the size field. + DIRTYPE = b'5', + /// This specifies a FIFO special file. Note that the archiving of a FIFO file archives the + /// existence of this file and not its contents. + FIFOTYPE = b'6', + /// This specifies a contiguous file, which is the same as a normal file except that, in + /// operating systems which support it, all its space is allocated contiguously on the disk. + /// Operating systems which do not allow contiguous allocation should silently treat this type + /// as a normal file. + CONTTYPE = b'7', + /// Extended header referring to the next file in the archive + XHDTYPE = b'x', + /// Global extended header + XGLTYPE = b'g', +} + +impl TypeFlag { + /// Whether we have a regular file. + #[must_use] + pub fn is_regular_file(self) -> bool { + // Equivalent. See spec. + self == Self::AREGTYPE || self == Self::REGTYPE + } +} + +impl TryFrom for TypeFlag { + type Error = InvalidTypeFlagError; + + fn try_from(value: TypeFlagRaw) -> Result { + match value.0 { + b'0' => Ok(Self::REGTYPE), + b'\0' => Ok(Self::AREGTYPE), + b'1' => Ok(Self::LINK), + b'2' => Ok(Self::SYMTYPE), + b'3' => Ok(Self::CHRTYPE), + b'4' => Ok(Self::BLKTYPE), + b'5' => Ok(Self::DIRTYPE), + b'6' => Ok(Self::FIFOTYPE), + b'7' => Ok(Self::CONTTYPE), + b'x' => Ok(Self::XHDTYPE), + b'g' => Ok(Self::XGLTYPE), + e => Err(InvalidTypeFlagError(e)), + } + } +} + +bitflags::bitflags! { + /// UNIX file permissions in octal format. + #[repr(transparent)] + #[derive(Debug, Clone, Copy, PartialEq, Eq)] + pub struct ModeFlags: u64 { + /// Set UID on execution. + const SetUID = 0o4000; + /// Set GID on execution. + const SetGID = 0o2000; + /// Reserved. + const TSVTX = 0o1000; + /// Owner read. + const OwnerRead = 0o400; + /// Owner write. + const OwnerWrite = 0o200; + /// Owner execute. + const OwnerExec = 0o100; + /// Group read. + const GroupRead = 0o040; + /// Group write. + const GroupWrite = 0o020; + /// Group execute. + const GroupExec = 0o010; + /// Others read. + const OthersRead = 0o004; + /// Others read. + const OthersWrite = 0o002; + /// Others execute. + const OthersExec = 0o001; + } +} + +/// Header of the TAR format as specified by POSIX (POSIX 1003.1-1990). +/// +/// "New" GNU Tar versions use this archive format by default. +/// (). +/// +/// Each file is started by such a header, that describes the size and +/// the file name. After that, the file content stands in chunks of 512 bytes. +/// The number of bytes can be derived from the file size. +/// +/// This is also mostly compatible with the "Ustar"-header and the "GNU format". +/// Because this library mainly targets the filename, the data, and basic +/// metadata, we don't need advanced checks for specific extensions. +#[derive(Debug, Copy, Clone, PartialEq, Eq)] +#[repr(C, packed)] +pub struct PosixHeader { + pub name: TarFormatString, + pub mode: Mode, + pub uid: TarFormatOctal<8>, + pub gid: TarFormatOctal<8>, + // confusing; size is stored as ASCII string + pub size: TarFormatOctal<12>, + pub mtime: TarFormatDecimal<12>, + pub cksum: TarFormatOctal<8>, + pub typeflag: TypeFlagRaw, + /// Name. There is always a null byte, therefore + /// the max len is 99. + pub linkname: TarFormatString, + pub magic: TarFormatString<6>, + pub version: TarFormatString<2>, + /// Username. There is always a null byte, therefore + /// the max len is N-1. + pub uname: TarFormatString<32>, + /// Groupname. There is always a null byte, therefore + /// the max len is N-1. + pub gname: TarFormatString<32>, + pub dev_major: TarFormatOctal<8>, + pub dev_minor: TarFormatOctal<8>, + pub prefix: TarFormatString, + // padding => to BLOCKSIZE bytes + pub _pad: [u8; 12], +} + +impl PosixHeader { + /// Returns the number of blocks that are required to read the whole file + /// content. Returns an error, if the file size can't be parsed from the + /// header. + /// + /// # Errors + /// Returns a [`ParseIntError`] error if the size can't be parsed. + pub fn payload_block_count(&self) -> Result { + let parsed_size = self.size.as_number::()?; + Ok(parsed_size.div_ceil(BLOCKSIZE)) + } + + /// A Tar archive is terminated, if an end-of-archive entry, which consists + /// of two 512 blocks of zero bytes, is found. + #[must_use] + pub fn is_zero_block(&self) -> bool { + let ptr = core::ptr::addr_of!(*self); + let ptr = ptr.cast::(); + + let self_bytes = unsafe { core::slice::from_raw_parts(ptr, BLOCKSIZE) }; + self_bytes.iter().filter(|x| **x == 0).count() == BLOCKSIZE + } +} + +#[cfg(test)] +mod tests { + use crate::header::{PosixHeader, TypeFlag}; + use crate::BLOCKSIZE; + use std::mem::size_of; + + /// Returns the PosixHeader at the beginning of the Tar archive. + fn bytes_to_archive(tar_archive_data: &[u8]) -> &PosixHeader { + unsafe { (tar_archive_data.as_ptr() as *const PosixHeader).as_ref() }.unwrap() + } + + #[test] + fn test_display_header() { + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_default.tar")); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + println!("{:#?}'", archive); + } + + #[test] + fn test_payload_block_count() { + // first file is "bye_world_513b.txt" => we expect two data blocks + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_default.tar")); + assert_eq!(archive.payload_block_count(), Ok(2)); + } + + #[test] + fn test_show_tar_header_magics() { + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_default.tar")); + println!( + "default: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_gnu.tar")); + println!( + "gnu: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_oldgnu.tar")); + println!( + "oldgnu: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_pax.tar")); + println!( + "pax: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_posix.tar")); + println!( + "posix: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_ustar.tar")); + println!( + "ustar: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_v7.tar")); + println!( + "v7: magic='{:?}', version='{:?}'", + archive.magic, archive.version + ); + } + + #[test] + fn test_parse_tar_header_filename() { + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_default.tar")); + assert_eq!( + archive.typeflag.try_to_type_flag(), + Ok(TypeFlag::REGTYPE), + "the first entry is a regular file!" + ); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_gnu.tar")); + assert_eq!( + archive.typeflag.try_to_type_flag(), + Ok(TypeFlag::REGTYPE), + "the first entry is a regular file!" + ); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_oldgnu.tar")); + assert_eq!( + archive.typeflag.try_to_type_flag(), + Ok(TypeFlag::REGTYPE), + "the first entry is a regular file!" + ); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + + /* UNSUPPORTED YET. Uses extensions.. + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_pax.tar")); + assert_eq!(archive.typeflag, TypeFlag::REGTYPE, "the first entry is a regular file!"); + assert_eq!(archive.name.as_string().as_str(), "bye_world_513b.txt"); */ + + /* UNSUPPORTED YET. Uses extensions. + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_posix.tar")); + unsupported extension XHDTYPE assert_eq!(archive.typeflag, TypeFlag::REGTYPE, "the first entry is a regular file!"); + assert_eq!(archive.name.as_string().as_str(), "bye_world_513b.txt"); */ + + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_ustar.tar")); + assert_eq!( + archive.typeflag.try_to_type_flag(), + Ok(TypeFlag::REGTYPE), + "the first entry is a regular file!" + ); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + + let archive = bytes_to_archive(include_bytes!("../tests/gnu_tar_v7.tar")); + // ARegType: legacy + assert_eq!( + archive.typeflag.try_to_type_flag(), + Ok(TypeFlag::AREGTYPE), + "the first entry is a regular file!" + ); + assert_eq!(archive.name.as_str(), Ok("bye_world_513b.txt")); + } + + #[test] + fn test_size() { + assert_eq!(BLOCKSIZE, size_of::()); + } +} diff --git a/vendor/tar-no-std-0.3.5/src/lib.rs b/vendor/tar-no-std-0.3.5/src/lib.rs new file mode 100644 index 0000000000..e45fd7b3cc --- /dev/null +++ b/vendor/tar-no-std-0.3.5/src/lib.rs @@ -0,0 +1,134 @@ +/* +MIT License + +Copyright (c) 2025 Philipp Schuster + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. +*/ +//! # `tar-no-std` - Parse Tar Archives (Tarballs) +//! +//! _Due to historical reasons, there are several formats of Tar archives. All of +//! them are based on the same principles, but have some subtle differences that +//! often make them incompatible with each other._ [(reference)](https://www.gnu.org/software/tar/manual/html_section/Formats.html) +//! +//! Library to read Tar archives in `no_std` environments with zero allocations. If +//! you have a standard environment and need full feature support, I recommend the +//! use of instead. +//! +//! ## TL;DR +//! +//! Look at the [`TarArchiveRef`] type. +//! +//! ## Limitations +//! +//! This crate is simple and focuses on reading files and their content from a Tar +//! archive. Historic basic Tar and ustar [formats](https://www.gnu.org/software/tar/manual/html_section/Formats.html) +//! are supported. Other formats may work, but likely without all supported +//! features. GNU Extensions such as sparse files, incremental archives, and +//! long filename extension are not supported. +//! +//! The maximum supported file name length is 256 characters excluding the +//! NULL-byte (using the Tar name/prefix longname implementation of ustar). The +//! maximum supported file size is 8GiB. Directories are supported, but only regular +//! fields are yielded in iteration. The path is reflected in their file name. +//! +//! ## Use Case +//! +//! This library is useful, if you write a kernel or a similar low-level +//! application, which needs "a bunch of files" from an archive (like an +//! "init ramdisk"). The Tar file could for example come as a Multiboot2 boot module +//! provided by the bootloader. +//! +//! ## Example +//! +//! ```rust +//! use tar_no_std::TarArchiveRef; +//! +//! // init a logger (optional) +//! std::env::set_var("RUST_LOG", "trace"); +//! env_logger::init(); +//! +//! // also works in no_std environment (except the println!, of course) +//! let archive = include_bytes!("../tests/gnu_tar_default.tar"); +//! let archive = TarArchiveRef::new(archive).unwrap(); +//! // Vec needs an allocator of course, but the library itself doesn't need one +//! let entries = archive.entries().collect::>(); +//! println!("{:#?}", entries); +//! ``` +//! +//! ## Cargo Feature +//! +//! This crate allows the usage of the additional Cargo build time feature `alloc`. +//! When this is active, the crate also provides the type `TarArchive`, which owns +//! the data on the heap. The `unstable` feature provides additional convenience +//! only available on the nightly channel. +//! +//! ## Compression (`tar.gz`) +//! +//! If your Tar file is compressed, e.g. by `.tar.gz`/`gzip`, you need to uncompress +//! the bytes first (e.g. by a *gzip* library). Afterwards, this crate can read the +//! Tar archive format from the uncompressed bytes. +//! +//! ## MSRV +//! +//! The MSRV is 1.76.0 stable. + +#![cfg_attr(feature = "unstable", feature(error_in_core))] +#![cfg_attr(not(test), no_std)] +#![deny( + clippy::all, + clippy::cargo, + clippy::nursery, + clippy::must_use_candidate, + // clippy::restriction, + // clippy::pedantic +)] +// now allow a few rules which are denied by the above statement +// --> they are ridiculous and not necessary +#![allow( + clippy::suboptimal_flops, + clippy::redundant_pub_crate, + clippy::fallible_impl_from +)] +#![deny(missing_debug_implementations)] +#![deny(rustdoc::all)] + +#[cfg_attr(test, macro_use)] +#[cfg(test)] +extern crate std; + +#[cfg(feature = "alloc")] +extern crate alloc; + +/// Each Archive Entry (either Header or Data Block) is a block of 512 bytes. +const BLOCKSIZE: usize = 512; +/// Maximum filename length of the base Tar format including the terminating NULL-byte. +const NAME_LEN: usize = 100; +/// Maximum long filename length of the base Tar format including the prefix +const POSIX_1003_MAX_FILENAME_LEN: usize = 256; +/// Maximum length of the prefix in Posix tar format +const PREFIX_LEN: usize = 155; + +mod archive; +mod header; +mod tar_format_types; + +pub use archive::*; +pub use header::*; +pub use tar_format_types::*; diff --git a/vendor/tar-no-std-0.3.5/src/tar_format_types.rs b/vendor/tar-no-std-0.3.5/src/tar_format_types.rs new file mode 100644 index 0000000000..92540fe87d --- /dev/null +++ b/vendor/tar-no-std-0.3.5/src/tar_format_types.rs @@ -0,0 +1,327 @@ +#![allow(unused_imports)] + +use core::fmt::{Debug, Formatter}; +use core::num::ParseIntError; +use core::ptr::copy_nonoverlapping; +use core::str::{from_utf8, Utf8Error}; +use num_traits::Num; + +/// Base type for strings embedded in a Tar header. The length depends on the +/// context. The returned string is likely to be UTF-8/ASCII, which is verified +/// by getters, such as [`TarFormatString::as_str`]. +/// +/// An optionally null terminated string. The contents are either: +/// 1. A fully populated string with no null termination or +/// 2. A partially populated string where the unused bytes are zero. +#[derive(Copy, Clone, PartialEq, Eq)] +#[repr(C)] +pub struct TarFormatString { + bytes: [u8; N], +} + +/// A Tar format string is a fixed length byte array containing UTF-8 bytes. +/// This string will be null terminated if it doesn't fill the entire array. +impl TarFormatString { + /// Constructor. + /// + /// # Panics + /// Panics of `N` is zero, i.e., the underlying array has no length. + #[must_use] + pub const fn new(bytes: [u8; N]) -> Self { + assert!(N > 0, "array should have at least one element"); + Self { bytes } + } + + /// True if the is string empty (ignoring NULL bytes). + #[must_use] + pub const fn is_empty(&self) -> bool { + self.bytes[0] == 0 + } + + /// Returns the length of the payload in bytes. This is either the full + /// capacity `N` or the data until the first NULL byte. + #[must_use] + pub fn size(&self) -> usize { + self.bytes.iter().position(|&byte| byte == 0).unwrap_or(N) + } + + /// Returns a str ref without terminating or intermediate NULL bytes. The + /// string is truncated at the first NULL byte, in case not the full length + /// was used. + /// + /// # Errors + /// Returns a [`Utf8Error`] error for invalid strings. + pub fn as_str(&self) -> Result<&str, Utf8Error> { + from_utf8(&self.bytes[0..self.size()]) + } + + /// Wrapper around [`Self::as_str`] that stops as soon as the first space + /// is found. This is necessary to properly parse certain Tar-style encoded + /// numbers. Some ustar implementations pad spaces which prevents the proper + /// parsing as number. + /// + /// # Errors + /// Returns a [`Utf8Error`] error for invalid strings. + pub fn as_str_until_first_space(&self) -> Result<&str, Utf8Error> { + from_utf8(&self.bytes[0..self.size()]).map(|str| { + let end_index_exclusive = str.find(' ').unwrap_or(str.len()); + &str[0..end_index_exclusive] + }) + } + + /// Append to end of string. + /// + /// # Panics + /// Panics if there is not enough capacity. + pub fn append(&mut self, other: &TarFormatString) { + let resulting_length = self.size() + other.size(); + + assert!(resulting_length <= N, "Result to long for capacity {N}"); + + unsafe { + let dst = self.bytes.as_mut_ptr().add(self.size()); + let src = other.bytes.as_ptr(); + copy_nonoverlapping(src, dst, other.size()); + } + + if resulting_length < N { + self.bytes[resulting_length] = 0; + } + } +} + +impl Debug for TarFormatString { + fn fmt(&self, f: &mut Formatter) -> core::fmt::Result { + let sub_array = &self.bytes[0..self.size()]; + write!( + f, + "str='{:?}',byte_usage={}/{}", + from_utf8(sub_array), + self.size(), + N + ) + } +} + +/// A number with a specified base. Trailing spaces in the string are ignored. +#[derive(Copy, Clone, PartialEq, Eq)] +#[repr(C)] +pub struct TarFormatNumber(TarFormatString); + +/// An octal number. Trailing spaces in the string are ignored. +#[derive(Copy, Clone, PartialEq, Eq)] +#[repr(C)] +pub struct TarFormatOctal(TarFormatNumber); + +#[cfg(test)] +impl TarFormatOctal { + #[must_use] + pub const fn new(bytes: [u8; N]) -> Self { + Self(TarFormatNumber::::new(bytes)) + } +} + +/// A decimal number. Trailing spaces in the string are ignored. +#[derive(Copy, Clone, PartialEq, Eq)] +#[repr(C)] +pub struct TarFormatDecimal(TarFormatNumber); + +impl TarFormatNumber { + #[cfg(test)] + const fn new(bytes: [u8; N]) -> Self { + Self(TarFormatString:: { bytes }) + } + + /// Interprets the underlying value as a number of the specified type using + /// its respective radix. + /// + /// # Errors + /// + /// Returns an error if the underlying value cannot be parsed as a number + /// of the specified type and respective radix. + pub fn as_number(&self) -> core::result::Result + where + T: num_traits::Num, + { + let str = self.0.as_str_until_first_space().unwrap_or("0"); + T::from_str_radix(str, R) + } + + /// Returns the underlying [`TarFormatString`]. + #[must_use] + pub const fn as_inner(&self) -> &TarFormatString { + &self.0 + } +} + +impl Debug for TarFormatNumber { + fn fmt(&self, f: &mut Formatter) -> core::fmt::Result { + let sub_array = &self.0.bytes[0..self.0.size()]; + match self.as_number::() { + Err(msg) => write!(f, "{} [{}]", msg, from_utf8(sub_array).unwrap()), + Ok(val) => write!(f, "{} [{}]", val, from_utf8(sub_array).unwrap()), + } + } +} + +impl Debug for TarFormatOctal { + fn fmt(&self, f: &mut Formatter) -> core::fmt::Result { + self.0.fmt(f) + } +} + +impl Debug for TarFormatDecimal { + fn fmt(&self, f: &mut Formatter) -> core::fmt::Result { + self.0.fmt(f) + } +} + +impl TarFormatDecimal { + /// Interprets the underlying value as a number of the specified type using + /// its respective radix. + /// + /// # Errors + /// + /// Returns an error if the underlying value cannot be parsed as a number + /// of the specified type and respective radix. + pub fn as_number(&self) -> core::result::Result + where + T: num_traits::Num, + { + self.0.as_number::() + } + + /// Returns the underlying [`TarFormatString`]. + #[must_use] + pub const fn as_inner(&self) -> &TarFormatString { + self.0.as_inner() + } +} + +impl TarFormatOctal { + /// Interprets the underlying value as a number of the specified type using + /// its respective radix. + /// + /// # Errors + /// + /// Returns an error if the underlying value cannot be parsed as a number + /// of the specified type and respective radix. + pub fn as_number(&self) -> core::result::Result + where + T: num_traits::Num, + { + self.0.as_number::() + } + + /// Returns the underlying [`TarFormatString`]. + #[must_use] + pub const fn as_inner(&self) -> &TarFormatString { + self.0.as_inner() + } +} + +#[cfg(test)] +mod tar_format_string_tests { + use super::TarFormatString; + + use core::mem::size_of_val; + + #[test] + fn test_empty_string() { + let empty = TarFormatString::new([0]); + assert_eq!(size_of_val(&empty), 1); + assert!(empty.is_empty()); + assert_eq!(empty.size(), 0); + assert_eq!(empty.as_str(), Ok("")); + } + + #[test] + fn test_one_byte_string() { + let s = TarFormatString::new([b'A']); + assert_eq!(size_of_val(&s), 1); + assert!(!s.is_empty()); + assert_eq!(s.size(), 1); + assert_eq!(s.as_str(), Ok("A")); + } + + #[test] + fn test_two_byte_string_nul_terminated() { + let s = TarFormatString::new([b'A', 0, b'B']); + assert_eq!(size_of_val(&s), 3); + assert!(!s.is_empty()); + assert_eq!(s.size(), 1); + assert_eq!(s.as_str(), Ok("A")); + } + + #[test] + fn test_str_until_first_space() { + let s = TarFormatString::new([b'A', b'B', b' ', b'X', 0]); + assert_eq!(size_of_val(&s), 5); + assert!(!s.is_empty()); + assert_eq!(s.size(), 4); + assert_eq!(s.as_str(), Ok("AB X")); + assert_eq!(s.as_str_until_first_space(), Ok("AB")); + } + + #[test] + #[allow(clippy::cognitive_complexity)] + fn test_append() { + let mut s = TarFormatString::new([0; 20]); + + // When adding a zero terminated string with one byte of zero + s.append(&TarFormatString::new([0])); + // Then the result is no change + assert_eq!(size_of_val(&s), 20); + assert!(s.is_empty()); + assert_eq!(s.size(), 0); + assert_eq!(s.as_str(), Ok("")); + + // When adding ABC + s.append(&TarFormatString::new([b'A', b'B', b'C'])); + // Then the string contains the additional 3 chars + assert_eq!(size_of_val(&s), 20); + assert!(!s.is_empty()); + assert_eq!(s.size(), 3); + assert_eq!(s.as_str(), Ok("ABC")); + + s.append(&TarFormatString::new([b'D', b'E', b'F'])); + // Then the string contains the additional 3 chars + assert_eq!(size_of_val(&s), 20); + assert!(!s.is_empty()); + assert_eq!(s.size(), 6); + assert_eq!(s.as_str(), Ok("ABCDEF")); + + s.append(&TarFormatString::new([b'A'; 12])); + // Then the string contains the additional 12 chars + assert_eq!(size_of_val(&s), 20); + assert!(!s.is_empty()); + assert_eq!(s.size(), 18); + assert_eq!(s.as_str(), Ok("ABCDEFAAAAAAAAAAAA")); + + s.append(&TarFormatString::new([b'A'; 1])); + // Then the string contains the additional 1 chars + assert_eq!(size_of_val(&s), 20); + assert!(!s.is_empty()); + assert_eq!(s.size(), 19); + assert_eq!(s.as_str(), Ok("ABCDEFAAAAAAAAAAAAA")); + + s.append(&TarFormatString::new([b'Z'; 1])); + // Then the string contains the additional 1 char, is full and not null terminated + assert_eq!(size_of_val(&s), 20); + assert!(!s.is_empty()); + assert_eq!(s.size(), 20); + assert_eq!(s.as_str(), Ok("ABCDEFAAAAAAAAAAAAAZ")); + } +} + +#[cfg(test)] +mod tar_format_number_tests { + use crate::{TarFormatDecimal, TarFormatNumber, TarFormatString}; + + #[test] + fn test_as_number_with_space_in_string() { + let str = [b'0', b'1', b'0', b' ', 0]; + let str = TarFormatNumber::<5, 10>::new(str); + assert_eq!(str.as_number::(), Ok(10)); + } +}