diff --git a/Cargo.lock b/Cargo.lock index e285699..2eb3cb9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,6 +2,12 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + [[package]] name = "aho-corasick" version = "1.1.5" @@ -11,6 +17,129 @@ dependencies = [ "memchr", ] +[[package]] +name = "anstream" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "824a212faf96e9acacdbd09febd34438f8f711fb84e09a8916013cd7815ca28d" +dependencies = [ + "anstyle", + "anstyle-parse", + "anstyle-query", + "anstyle-wincon", + "colorchoice", + "is_terminal_polyfill", + "utf8parse", +] + +[[package]] +name = "anstyle" +version = "1.0.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "940b3a0ca603d1eade50a4846a2afffd5ef57a9feac2c0e2ec2e14f9ead76000" + +[[package]] +name = "anstyle-parse" +version = "1.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "52ce7f38b242319f7cabaa6813055467063ecdc9d355bbb4ce0c68908cd8130e" +dependencies = [ + "utf8parse", +] + +[[package]] +name = "anstyle-query" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "40c48f72fd53cd289104fc64099abca73db4166ad86ea0b4341abe65af83dadc" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "anstyle-wincon" +version = "3.0.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "291e6a250ff86cd4a820112fb8898808a366d8f9f58ce16d1f538353ad55747d" +dependencies = [ + "anstyle", + "once_cell_polyfill", + "windows-sys", +] + +[[package]] +name = "anyhow" +version = "1.0.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" + +[[package]] +name = "bitflags" +version = "2.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3ded4057c258ba199e2d26386d3af3780957ecaee6c4ef4041c6b4b8b97c0b06" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + +[[package]] +name = "cfg-if" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600" + +[[package]] +name = "clap" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "aa8876b300ab35ba921adea3dfd70157a46249b33f95c9084ae5709785478946" +dependencies = [ + "clap_builder", + "clap_derive", +] + +[[package]] +name = "clap_builder" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ec0797fb7aeb1406c84efac526901f7ec3ead2124f946b494e72879d4b54704d" +dependencies = [ + "anstream", + "anstyle", + "clap_lex", + "strsim", +] + +[[package]] +name = "clap_derive" +version = "4.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9c751b79415d4e559e3d1fcf128e09e720eb673a06d26cf6f392d37d75b66e0" +dependencies = [ + "heck", + "proc-macro2", + "quote", + "syn 3.0.6", +] + +[[package]] +name = "clap_lex" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c133bc6a41be0d194c306b5506d15e6feeea7b1d6604bd3f8310dfb2ca96486" + +[[package]] +name = "colorchoice" +version = "1.0.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1d07550c9036bf2ae0c684c4297d503f838287c83c53686d05370d0e139ae570" + [[package]] name = "count-calls" version = "0.0.0" @@ -19,18 +148,163 @@ dependencies = [ "mirth-build", ] +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crc32fast" +version = "1.5.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "crossbeam-deque" +version = "0.8.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a" +dependencies = [ + "crossbeam-epoch", + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-epoch" +version = "0.9.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d" +dependencies = [ + "crossbeam-utils", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + +[[package]] +name = "either" +version = "1.19.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e9c71c2167ca323c882b99918929403426e2373ea17242ff5653e0d5e1058be" + [[package]] name = "equivalent" version = "1.0.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "877a4ace8713b0bcf2a4e7eec82529c029f1d0619886d18145fea96c3ffe5c0f" +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "filetime" +version = "0.2.29" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5c287a33c7f0a620c38e641e7f60827713987b3c0f26e8ddc9462cc69cf75759" +dependencies = [ + "cfg-if", + "libc", +] + +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide", + "zlib-rs", +] + +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "899def5c37c4fd7b2664648c28120ecec138e4d395b459e5ca34f9cce2dd77fd" +dependencies = [ + "cfg-if", + "libc", + "r-efi 5.3.0", + "wasip2", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi 6.0.0", +] + [[package]] name = "hashbrown" version = "0.17.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "ed5909b6e89a2db4456e54cd5f673791d7eca6732202bbf2a9cc504fe2f9b84a" +[[package]] +name = "heck" +version = "0.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea" + [[package]] name = "indexmap" version = "2.14.2" @@ -41,12 +315,46 @@ dependencies = [ "hashbrown", ] +[[package]] +name = "is_terminal_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "libc" +version = "0.2.190" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ce5d3ddc6d3fa000eb1536d85e147bfe31aacaba692ed6a876f95cb7c855be78" + +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + [[package]] name = "memchr" version = "2.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "mirth" version = "0.0.0" @@ -65,6 +373,28 @@ dependencies = [ "regex", ] +[[package]] +name = "mirth-lab" +version = "0.0.0" +dependencies = [ + "anyhow", + "clap", + "flate2", + "libc", + "mirth-rewrite", + "rand", + "rayon", + "regex", + "serde", + "serde_json", + "sha2", + "similar", + "tar", + "tempfile", + "wait-timeout", + "walkdir", +] + [[package]] name = "mirth-rewrite" version = "0.0.0" @@ -88,6 +418,27 @@ dependencies = [ "toml", ] +[[package]] +name = "once_cell" +version = "1.21.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" + +[[package]] +name = "once_cell_polyfill" +version = "1.70.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" + +[[package]] +name = "ppv-lite86" +version = "0.2.21" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85eae3c4ed2f50dcfe72643da4befc30deadb458a9b590d720cde2f2b1e97da9" +dependencies = [ + "zerocopy", +] + [[package]] name = "proc-macro2" version = "1.0.107" @@ -106,6 +457,67 @@ dependencies = [ "proc-macro2", ] +[[package]] +name = "r-efi" +version = "5.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "69cdb34c158ceb288df11e18b4bd39de994f6657d83847bdffdbd7f346754b0f" + +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rand" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41" +dependencies = [ + "rand_chacha", + "rand_core", +] + +[[package]] +name = "rand_chacha" +version = "0.9.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d3022b5f1df60f26e1ffddd6c66e8aa15de382ae63b3a0c1bfc0e4d3e3f325cb" +dependencies = [ + "ppv-lite86", + "rand_core", +] + +[[package]] +name = "rand_core" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "76afc826de14238e6e8c374ddcc1fa19e374fd8dd986b0d2af0d02377261d83c" +dependencies = [ + "getrandom 0.3.4", +] + +[[package]] +name = "rayon" +version = "1.12.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fb39b166781f92d482534ef4b4b1b2568f42613b53e5b6c160e24cfbfa30926d" +dependencies = [ + "either", + "rayon-core", +] + +[[package]] +name = "rayon-core" +version = "1.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "22e18b0f0062d30d4230b2e85ff77fdfe4326feb054b9783a3460d8435c8ab91" +dependencies = [ + "crossbeam-deque", + "crossbeam-utils", +] + [[package]] name = "regex" version = "1.13.1" @@ -135,6 +547,28 @@ version = "0.8.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d6f6ff9a378485b298a5286656da665ba74413d36db0979633275d2e708145d4" +[[package]] +name = "rustix" +version = "1.1.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "891efababe418670775f199f0d233d84843c227a0949a883ce15b37c78d6629d" +dependencies = [ + "bitflags", + "errno", + "libc", + "linux-raw-sys", + "windows-sys", +] + +[[package]] +name = "same-file" +version = "1.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "93fc1dc3aaa9bfed95e02e6eadabb4baf7e3078b0bd1b4d7b6b0b68378900502" +dependencies = [ + "winapi-util", +] + [[package]] name = "serde" version = "1.0.229" @@ -165,6 +599,19 @@ dependencies = [ "syn 3.0.6", ] +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + [[package]] name = "serde_spanned" version = "1.1.1" @@ -174,6 +621,35 @@ dependencies = [ "serde_core", ] +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + +[[package]] +name = "similar" +version = "2.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bbbb5d9659141646ae647b42fe094daf6c6192d1620870b449d9557f748b2daa" + +[[package]] +name = "strsim" +version = "0.11.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7da8b5736845d9f2fcb837ea5d9e2628564b3b043a70948a3f0b778838c5fb4f" + [[package]] name = "syn" version = "2.0.119" @@ -196,6 +672,30 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "tar" +version = "0.4.46" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3f6221d9a6003c78398e3b239969f352578258df48c8eb051caadae0015bc840" +dependencies = [ + "filetime", + "libc", + "xattr", +] + +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom 0.4.3", + "once_cell", + "rustix", + "windows-sys", +] + [[package]] name = "toml" version = "0.9.12+spec-1.1.0" @@ -241,12 +741,82 @@ version = "2.1.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "86a801b3cea342a06d468c8710662aa29e5e05e4f5c0d62f00bbb7f2ad7941c2" +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "unicode-ident" version = "1.0.26" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" +[[package]] +name = "utf8parse" +version = "0.2.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" + +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + +[[package]] +name = "wait-timeout" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09ac3b126d3914f9849036f826e054cbabdc8519970b8998ddaf3b5bd3c65f11" +dependencies = [ + "libc", +] + +[[package]] +name = "walkdir" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29790946404f91d9c5d06f9874efddea1dc06c5efe94541a7d6863108e3a5e4b" +dependencies = [ + "same-file", + "winapi-util", +] + +[[package]] +name = "wasip2" +version = "1.0.4+wasi-0.2.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67efb37e106e55ce722a510d6b5f9c17f083e5fc79afc2badeb12cc313d9487" +dependencies = [ + "wit-bindgen", +] + +[[package]] +name = "winapi-util" +version = "0.1.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" +dependencies = [ + "windows-sys", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.61.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ae137229bcbd6cdf0f7b80a31df61766145077ddf49416a728b02cb3921ff3fc" +dependencies = [ + "windows-link", +] + [[package]] name = "winnow" version = "0.7.15" @@ -258,3 +828,51 @@ name = "winnow" version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "23b97319f7b8343df12cc98938e5c3eb436064524c8d2b4e30a1d3a36eecdf81" + +[[package]] +name = "wit-bindgen" +version = "0.57.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1ebf944e87a7c253233ad6766e082e3cd714b5d03812acc24c318f549614536e" + +[[package]] +name = "xattr" +version = "1.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32e45ad4206f6d2479085147f02bc2ef834ac85886624a23575ae137c8aa8156" +dependencies = [ + "libc", + "rustix", +] + +[[package]] +name = "zerocopy" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879fb705ce98c32e41ebdb970fbe1204f8492423b314c6ab0354c3e7b5542866" +dependencies = [ + "zerocopy-derive", +] + +[[package]] +name = "zerocopy-derive" +version = "0.8.61" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "708882a28301d604fa039cc7727607a98d04c4b86dc76ec9cc683f805d709759" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + +[[package]] +name = "zlib-rs" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml index ca464a3..79d3b50 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -7,6 +7,7 @@ members = [ "crates/mirth-runtime", "crates/mirth-watch", "crates/mirth-rewrite", + "crates/mirth-lab", "examples/count-calls", ] exclude = ["fixtures"] diff --git a/README.md b/README.md index ae16d3d..75f672c 100644 --- a/README.md +++ b/README.md @@ -56,9 +56,9 @@ rustc/check.sh chain # check fixtures/chain rustc/edits.sh chain # apply, check and revert each edit in rustc/edits EDITS=regressions rustc/edits.sh chain # the same for the past bugs in rustc/regressions rustc/hunt.sh wide # repeated threaded builds, and P6 for each of fixtures/wide/edits -rustc/fuzz.py --rustc --fixture fixtures/sink --work # random edits, P6 on each -rustc/replay.py --rustc --repo --work # a crate's history, P6 per commit -rustc/audit-options.py --rustc --source --crate fixtures/audit/lib.rs # untracked options +mirth-lab fuzz --rustc --fixture fixtures/sink --work # random edits, P6 on each +mirth-lab replay --rustc --repo --work # a crate's history, P6 per commit +mirth-lab audit-options --rustc --source --crate fixtures/audit/lib.rs # untracked options ``` The build takes about an hour on 16 cores. `rustc/rmeta.toml` says what is diff --git a/crates/mirth-lab/Cargo.toml b/crates/mirth-lab/Cargo.toml new file mode 100644 index 0000000..1c00609 --- /dev/null +++ b/crates/mirth-lab/Cargo.toml @@ -0,0 +1,29 @@ +[package] +name = "mirth-lab" +description = "Checks (oracles) over rustc: differentials, metamorphic rewrites, invariants, sweeps over its test suites" +version.workspace = true +edition.workspace = true +license.workspace = true +publish.workspace = true + +[[bin]] +name = "mirth-lab" +path = "src/main.rs" + +[dependencies] +anyhow = "1" +clap = { version = "4.6", features = ["derive"] } +rayon = "1.12" +rand = "0.9" +regex = "1" +serde = { version = "1", features = ["derive"] } +serde_json = "1" +similar = "2" +sha2 = "0.10" +tempfile = "3" +walkdir = "2" +wait-timeout = "0.2" +libc = "0.2" +flate2 = "1" +tar = "0.4" +mirth-rewrite = { path = "../mirth-rewrite" } diff --git a/crates/mirth-lab/src/artifacts.rs b/crates/mirth-lab/src/artifacts.rs new file mode 100644 index 0000000..431b0a0 --- /dev/null +++ b/crates/mirth-lab/src/artifacts.rs @@ -0,0 +1,199 @@ +//! Compiler outputs in a form two builds can be compared in: archives taken apart, session +//! suffixes in member names and link metadata removed, everything hashed. + +use std::collections::BTreeMap; +use std::path::Path; +use std::sync::LazyLock; + +use regex::bytes::Regex; +use sha2::{Digest, Sha256}; + +/// `.<7 chars>.rcgu.o`: the per-session suffix of codegen-unit objects. +static SESSION: LazyLock = LazyLock::new(|| Regex::new(r"\.[0-9a-z]{7}(\.rcgu\.(?:o|dwo))").unwrap()); + +pub fn sha256(data: &[u8]) -> String { + let mut h = Sha256::new(); + h.update(data); + h.finalize().iter().map(|b| format!("{b:02x}")).collect() +} + +/// The members of a Unix ar archive (GNU format), names with session suffixes removed. +pub fn ar_members(data: &[u8]) -> BTreeMap> { + let mut members = BTreeMap::new(); + if !data.starts_with(b"!\n") { + members.insert("".into(), data.to_vec()); + return members; + } + let (mut names, mut pos): (&[u8], usize) = (b"", 8); + while pos + 60 <= data.len() { + let header = &data[pos..pos + 60]; + let raw_name = trim_end(&header[..16]); + let size: usize = std::str::from_utf8(&header[48..58]).ok().and_then(|s| s.trim().parse().ok()).unwrap_or(0); + let end = (pos + 60 + size).min(data.len()); + let body = &data[pos + 60..end]; + pos += 60 + size + (size & 1); + if raw_name == b"//" { + names = body; + continue; + } + if raw_name == b"/" || raw_name == b"/SYM64/" { + continue; + } + let mut name: Vec = raw_name.to_vec(); + if name.len() > 1 && name[0] == b'/' && name[1..].iter().all(u8::is_ascii_digit) { + let offset: usize = std::str::from_utf8(&name[1..]).unwrap().parse().unwrap_or(0); + let rest = names.get(offset..).unwrap_or(b""); + let len = rest.windows(2).position(|w| w == b"/\n").unwrap_or(rest.len()); + name = rest[..len].to_vec(); + } + while name.last() == Some(&b'/') { + name.pop(); + } + let name = SESSION.replace_all(&name, &b"$1"[..]); + members.insert(String::from_utf8_lossy(&name).into_owned(), body.to_vec()); + } + members +} + +fn trim_end(b: &[u8]) -> &[u8] { + let n = b.iter().rposition(|c| !c.is_ascii_whitespace()).map_or(0, |i| i + 1); + &b[..n] +} + +/// An rlib as {member: digest}, session suffixes removed from names and contents. +pub fn normalized_rlib(path: &Path) -> BTreeMap { + let data = std::fs::read(path).unwrap_or_default(); + ar_members(&data) + .into_iter() + .map(|(name, body)| (name, sha256(&SESSION.replace_all(&body, &b"$1"[..])))) + .collect() +} + +/// Every file in `dir` as {name: digest}; rlibs are normalized member by member. +pub fn digest_dir(dir: &Path) -> BTreeMap { + let mut out = BTreeMap::new(); + let Ok(entries) = std::fs::read_dir(dir) else { return out }; + for e in entries.flatten() { + let p = e.path(); + if !p.is_file() { + continue; + } + let name = e.file_name().to_string_lossy().into_owned(); + let digest = if p.extension().is_some_and(|x| x == "rlib") { + serde_json::to_string(&normalized_rlib(&p)).unwrap_or_default() + } else { + sha256(&std::fs::read(&p).unwrap_or_default()) + }; + out.insert(name, digest); + } + out +} + +/// What a Cargo build produced for packages built from a path, from the JSON messages of +/// `cargo build --message-format=json-render-diagnostics`: every .rmeta and executable by +/// digest, every .rlib member by member, every rendered diagnostic counted per crate. Paths are +/// relative to the target directory. +#[derive(Default, Debug, PartialEq, serde::Serialize)] +pub struct Collected { + pub rmeta: BTreeMap, + pub rlib: BTreeMap>, + pub exe: BTreeMap, + pub diag: BTreeMap<(String, String), usize>, +} + +pub fn collect(stdout: &str, target: &Path) -> Collected { + let mut found = Collected::default(); + let rel = |f: &str| Path::new(f).strip_prefix(target).map_or(f.to_owned(), |r| r.display().to_string()); + for line in stdout.lines() { + let Ok(msg) = serde_json::from_str::(line) else { continue }; + if !msg["package_id"].as_str().unwrap_or("").contains("path+file") { + continue; + } + match msg["reason"].as_str() { + Some("compiler-message") => { + let m = &msg["message"]; + let text = m["rendered"].as_str().filter(|t| !t.is_empty()).or(m["message"].as_str()).unwrap_or(""); + let krate = msg["target"]["name"].as_str().unwrap_or("").to_owned(); + *found.diag.entry((krate, text.to_owned())).or_default() += 1; + } + Some("compiler-artifact") => { + for f in msg["filenames"].as_array().into_iter().flatten().filter_map(|f| f.as_str()) { + if f.ends_with(".rmeta") { + found.rmeta.insert(rel(f), sha256(&std::fs::read(f).unwrap_or_default())); + } else if f.ends_with(".rlib") { + found.rlib.insert(rel(f), normalized_rlib(Path::new(f))); + } + } + if let Some(f) = msg["executable"].as_str() { + let data = std::fs::read(f).unwrap_or_default(); + found.exe.insert(rel(f), sha256(&SESSION.replace_all(&data, &b"$1"[..]))); + } + } + _ => {} + } + } + found +} + +/// {kind: [what differs]} for the kinds that differ between two collections. +pub fn compare(a: &Collected, b: &Collected) -> BTreeMap<&'static str, Vec> { + fn keys<'a, V: PartialEq>(a: &'a BTreeMap, b: &'a BTreeMap) -> Vec { + let all: std::collections::BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + all.into_iter().filter(|k| a.get(*k) != b.get(*k)).cloned().collect() + } + let mut out = BTreeMap::new(); + for (kind, x, y) in [("rmeta", &a.rmeta, &b.rmeta), ("exe", &a.exe, &b.exe)] { + let diff = keys(x, y); + if !diff.is_empty() { + out.insert(kind, diff); + } + } + let empty = BTreeMap::new(); + let rlibs: Vec = keys(&a.rlib, &b.rlib) + .into_iter() + .map(|rel| { + let members = keys(a.rlib.get(&rel).unwrap_or(&empty), b.rlib.get(&rel).unwrap_or(&empty)); + let more = if members.len() > 5 { " …" } else { "" }; + format!("{rel}: {}{more}", members[..members.len().min(5)].join(", ")) + }) + .collect(); + if !rlibs.is_empty() { + out.insert("rlib", rlibs); + } + if a.diag != b.diag { + let only = |x: &BTreeMap<(String, String), usize>, y: &BTreeMap<(String, String), usize>, which: &str| -> Vec { + x.iter() + .filter(|(k, n)| **n > y.get(*k).copied().unwrap_or(0)) + .map(|((c, t), _)| format!("{c} only in the {which}: {:?}", t.chars().take(200).collect::())) + .collect() + }; + let mut d = only(&a.diag, &b.diag, "first"); + d.extend(only(&b.diag, &a.diag, "second")); + out.insert("diag", d); + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn session_suffixes_removed() { + let mut ar = b"!\n".to_vec(); + let body = b"refers to t.abc1234.rcgu.o"; + ar.extend(format!("{:<16}{:<32}{:<10}`\n", "t.abc1234.rcgu.o/", "", body.len()).as_bytes()); + ar.extend(body); + let m = ar_members(&ar); + assert!(m.contains_key("t.rcgu.o"), "{m:?}"); + } + + #[test] + fn diagnostics_compared_by_count() { + let line = |t: &str| format!(r#"{{"reason":"compiler-message","package_id":"path+file:///x#a@0.1.0","target":{{"name":"a"}},"message":{{"rendered":"{t}"}}}}"#); + let a = collect(&format!("{}\n{}", line("w"), line("w")), Path::new("/t")); + let b = collect(&line("w"), Path::new("/t")); + assert_eq!(compare(&a, &b)["diag"], vec![r#"a only in the first: "w""#]); + assert!(compare(&a, &a).is_empty()); + } +} diff --git a/crates/mirth-lab/src/cargo.rs b/crates/mirth-lab/src/cargo.rs new file mode 100644 index 0000000..345196f --- /dev/null +++ b/crates/mirth-lab/src/cargo.rs @@ -0,0 +1,205 @@ +//! Incremental Cargo builds for the fuzzers: Cargo's JSON messages, what the compiler's own +//! checks print (RUSTC_VERIFY_REUSE, RUSTC_REPORT_UNTRACKED), a process-group timeout, and tree +//! copies that keep modification times. What a build produced is compared by +//! `artifacts::collect` and `artifacts::compare`. + +use std::collections::BTreeSet; +use std::io::Read; +use std::os::unix::process::CommandExt; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; +use std::sync::LazyLock; +use std::time::Duration; + +use regex::Regex; +use serde::Deserialize; +use wait_timeout::ChildExt; + +#[derive(Deserialize)] +pub struct Message { + #[serde(default)] + pub reason: String, + #[serde(default)] + pub package_id: String, + #[serde(default)] + pub target: Option, + #[serde(default)] + pub filenames: Vec, + #[serde(default)] + pub executable: Option, + #[serde(default)] + pub fresh: bool, + #[serde(default)] + pub message: Option, +} + +#[derive(Deserialize)] +pub struct Target { + pub name: String, + #[serde(default)] + pub kind: Vec, +} + +impl Message { + pub fn target_name(&self) -> &str { + self.target.as_ref().map_or("", |t| t.name.as_str()) + } + pub fn from_path(&self) -> bool { + self.package_id.contains("path+file") + } +} + +/// Cargo's JSON messages in `stdout`, skipping lines that are not. +pub fn messages(stdout: &str) -> impl Iterator + '_ { + stdout.lines().filter_map(|l| serde_json::from_str(l).ok()) +} + +/// `f` relative to `target` when it is inside it. +pub fn relative(f: &str, target: &Path) -> String { + Path::new(f).strip_prefix(target).map_or_else(|_| f.to_owned(), |p| p.to_string_lossy().into_owned()) +} + +/// What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found stale: +/// `query `, `metadata`, `codegen unit` or `allocation sharing `, once each. +pub fn reuse_checks(log: &str) -> Vec { + static QUERY: LazyLock = LazyLock::new(|| Regex::new(r"query `(\w+)`").unwrap()); + static MODE: LazyLock = LazyLock::new(|| Regex::new(r"TypingModeEqWrapper\((\w+)\)").unwrap()); + let mut found = BTreeSet::new(); + for line in log.lines() { + if line.starts_with("rustc-verify-reuse: query `") { + found.insert(format!("query {}", line.split('`').nth(1).unwrap_or(""))); + } else if line.starts_with("rustc-verify-reuse: metadata") { + found.insert("metadata".to_owned()); + } else if line.starts_with("rustc-verify-reuse: codegen unit") { + found.insert("codegen unit".to_owned()); + } else if line.starts_with("rustc-verify-reuse: allocation shared differently") { + // Named by the two queries and typing modes, so that a new pattern is kept apart + // from a known one. + let mut parts: Vec = QUERY.captures_iter(line).map(|c| c[1].to_owned()).collect(); + let modes: BTreeSet = MODE.captures_iter(line).map(|c| c[1].to_owned()).collect(); + parts.extend(modes); + found.insert(format!("allocation sharing {}", parts.join(" / "))); + } + } + found.into_iter().collect() +} + +/// Reads of untracked state the compiler reported (docs/hunt/report-untracked.patch). +pub fn untracked_reads(log: &str) -> BTreeSet { + log.lines().filter(|l| l.starts_with("rustc-untracked-read:")).map(|l| l.trim().to_owned()).collect() +} + +/// Adds new reports of untracked reads to `path`, which lists each once. +pub fn note_untracked(path: &Path, lines: &BTreeSet) { + let known: BTreeSet = std::fs::read_to_string(path).unwrap_or_default().lines().map(str::to_owned).collect(); + let new: String = lines.difference(&known).map(|l| format!("{l}\n")).collect(); + if !new.is_empty() { + use std::io::Write; + if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open(path) { + let _ = f.write_all(new.as_bytes()); + } + } +} + +/// Whether a compiler log shows a crash. +pub fn is_ice(log: &str) -> bool { + log.contains("internal compiler error") || log.contains("the compiler unexpectedly panicked") +} + +/// How a build command ended. +pub struct Run { + pub ok: bool, + pub stdout: String, + pub stderr: String, + /// Killed at the timeout while a rustc was still running. + pub hang: bool, +} + +/// Run `cmd` in its own process group; after `timeout` the group is killed, and the run counts +/// as a hang if a rustc was still running in it (a looping build script is the fixture's +/// problem, not the compiler's). +pub fn run_group(mut cmd: Command, timeout: Option) -> std::io::Result { + cmd.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()).process_group(0); + let mut child = cmd.spawn()?; + let pid = child.id() as i32; + let (mut out, mut err) = (child.stdout.take().expect("piped"), child.stderr.take().expect("piped")); + let out_thread = std::thread::spawn(move || { + let mut s = Vec::new(); + let _ = out.read_to_end(&mut s); + s + }); + let err_thread = std::thread::spawn(move || { + let mut s = Vec::new(); + let _ = err.read_to_end(&mut s); + s + }); + let status = match timeout { + Some(t) => child.wait_timeout(t)?, + None => Some(child.wait()?), + }; + let mut note = String::new(); + let (ok, hang) = match status { + Some(s) => (s.success(), false), + None => { + let ps = Command::new("ps").args(["-o", "args=", "-g", &pid.to_string()]).output(); + let group = ps.map(|o| String::from_utf8_lossy(&o.stdout).into_owned()).unwrap_or_default(); + let hang = group.lines().filter_map(|l| l.split_whitespace().next()).any(|a| a.ends_with("/rustc")); + // SAFETY: signalling the process group we created. + unsafe { libc::kill(-pid, libc::SIGKILL) }; + let _ = child.wait(); + note = format!("\nkilled after {}s; still running:\n{group}", timeout.unwrap().as_secs()); + (false, hang) + } + }; + let stdout = String::from_utf8_lossy(&out_thread.join().unwrap_or_default()).into_owned(); + let mut stderr = String::from_utf8_lossy(&err_thread.join().unwrap_or_default()).into_owned(); + stderr.push_str(¬e); + Ok(Run { ok, stdout, stderr, hang }) +} + +/// The last `n` characters of `s`. +pub fn tail(s: &str, n: usize) -> &str { + let count = s.chars().count(); + if count <= n { + return s; + } + let skip = s.char_indices().nth(count - n).map_or(0, |(i, _)| i); + &s[skip..] +} + +/// Copy a tree, skipping entries named in `ignore` at any depth, keeping modification times +/// (Cargo's freshness depends on them). Symbolic links are copied as links with `symlinks`, +/// followed otherwise. +pub fn copy_tree(from: &Path, to: &Path, ignore: &[&str], symlinks: bool) -> std::io::Result<()> { + let walk = walkdir::WalkDir::new(from).follow_links(!symlinks).into_iter().filter_entry(|e| { + e.depth() == 0 || !ignore.iter().any(|i| e.file_name() == *i) + }); + for e in walk { + let e = e.map_err(std::io::Error::other)?; + let dest: PathBuf = to.join(e.path().strip_prefix(from).unwrap()); + let ft = e.file_type(); + if ft.is_symlink() { + std::os::unix::fs::symlink(std::fs::read_link(e.path())?, &dest)?; + } else if ft.is_dir() { + std::fs::create_dir_all(&dest)?; + } else { + std::fs::copy(e.path(), &dest)?; + let meta = e.metadata().map_err(std::io::Error::other)?; + let times = std::fs::FileTimes::new().set_modified(meta.modified()?).set_accessed(meta.accessed()?); + std::fs::File::options().write(true).open(&dest)?.set_times(times)?; + } + } + Ok(()) +} + +/// Free bytes on the file system holding `path`. +pub fn free_bytes(path: &Path) -> u64 { + use std::os::unix::ffi::OsStrExt; + let Ok(c) = std::ffi::CString::new(path.as_os_str().as_bytes()) else { return u64::MAX }; + // SAFETY: statvfs is plain data, filled by the call for a valid C string. + let mut s: libc::statvfs = unsafe { std::mem::zeroed() }; + if unsafe { libc::statvfs(c.as_ptr(), &mut s) } != 0 { + return u64::MAX; + } + s.f_bavail as u64 * s.f_frsize as u64 +} diff --git a/crates/mirth-lab/src/coverage.rs b/crates/mirth-lab/src/coverage.rs new file mode 100644 index 0000000..b724668 --- /dev/null +++ b/crates/mirth-lab/src/coverage.rs @@ -0,0 +1,170 @@ +//! What a coverage-instrumented compiler (rustc/coverage.toml) writes: the site tables of its +//! build (`.sites`) and the `V ` lines each process run with MIRTH_OUT logs. + +use std::collections::BTreeSet; +use std::io::{self, Write}; +use std::path::{Path, PathBuf}; + +use serde_json::ser::Formatter; + +/// An instrumented function: a `cover` site. +pub struct Function { + pub site: String, + pub krate: String, + pub path: String, + pub span: String, +} + +/// The files of `dir` with extension `ext`, not recursing. +pub fn files_with(dir: &Path, ext: &str) -> Vec { + let Ok(entries) = std::fs::read_dir(dir) else { return Vec::new() }; + entries.flatten().map(|e| e.path()).filter(|p| p.extension().is_some_and(|x| x == ext)).collect() +} + +fn read_lossy(path: &Path) -> String { + String::from_utf8_lossy(&std::fs::read(path).unwrap_or_default()).into_owned() +} + +/// Every `cover` site of the tables in `dir`. +pub fn functions(dir: &Path) -> Vec { + let mut out = Vec::new(); + for table in files_with(dir, "sites") { + for line in read_lossy(&table).lines() { + let f: Vec<&str> = line.split('\t').collect(); + if f.len() >= 7 && f[1] == "cover" { + out.push(Function { site: f[0].into(), krate: f[3].into(), path: f[4].into(), span: f[6].into() }); + } + } + } + out +} + +/// The sites a log reached. +pub fn log_hits(text: &str) -> impl Iterator { + text.lines().filter_map(|l| l.strip_prefix("V\t")) +} + +/// The sites the logs in `dir` reached, and how many logs there were. +pub fn hits(dir: &Path) -> (BTreeSet, usize) { + let logs = files_with(dir, "log"); + let mut out = BTreeSet::new(); + for log in &logs { + out.extend(log_hits(&read_lossy(log)).map(str::to_owned)); + } + (out, logs.len()) +} + +/// Python's json.dumps spacing (`, ` and `: `) and escaping (non-ASCII as `\\uXXXX`), on one +/// line or, with `indent`, as `json.dumps(…, indent=n)` writes it. +struct Python<'a> { + indent: Option<&'a [u8]>, + depth: usize, + has_value: bool, +} + +impl Python<'_> { + fn newline(&self, w: &mut W) -> io::Result<()> { + if let Some(indent) = self.indent { + w.write_all(b"\n")?; + for _ in 0..self.depth { + w.write_all(indent)?; + } + } + Ok(()) + } + fn open(&mut self, w: &mut W, c: &[u8]) -> io::Result<()> { + self.depth += 1; + self.has_value = false; + w.write_all(c) + } + fn close(&mut self, w: &mut W, c: &[u8]) -> io::Result<()> { + self.depth -= 1; + if self.has_value { + self.newline(w)?; + } + self.has_value = true; + w.write_all(c) + } + fn item(&mut self, w: &mut W, first: bool) -> io::Result<()> { + match (first, self.indent.is_some()) { + (true, _) => {} + (false, true) => w.write_all(b",")?, + (false, false) => w.write_all(b", ")?, + } + self.newline(w) + } +} + +impl Formatter for Python<'_> { + fn begin_array(&mut self, w: &mut W) -> io::Result<()> { + self.open(w, b"[") + } + fn end_array(&mut self, w: &mut W) -> io::Result<()> { + self.close(w, b"]") + } + fn begin_array_value(&mut self, w: &mut W, first: bool) -> io::Result<()> { + self.item(w, first) + } + fn end_array_value(&mut self, _: &mut W) -> io::Result<()> { + self.has_value = true; + Ok(()) + } + fn begin_object(&mut self, w: &mut W) -> io::Result<()> { + self.open(w, b"{") + } + fn end_object(&mut self, w: &mut W) -> io::Result<()> { + self.close(w, b"}") + } + fn begin_object_key(&mut self, w: &mut W, first: bool) -> io::Result<()> { + self.item(w, first) + } + fn begin_object_value(&mut self, w: &mut W) -> io::Result<()> { + w.write_all(b": ") + } + fn end_object_value(&mut self, _: &mut W) -> io::Result<()> { + self.has_value = true; + Ok(()) + } + fn write_string_fragment(&mut self, w: &mut W, fragment: &str) -> io::Result<()> { + for c in fragment.chars() { + if c.is_ascii() { + w.write_all(&[c as u8])?; + } else { + for unit in c.encode_utf16(&mut [0; 2]) { + write!(w, "\\u{unit:04x}")?; + } + } + } + Ok(()) + } +} + +fn python_json(value: &impl serde::Serialize, indent: Option<&[u8]>) -> String { + let mut buf = Vec::new(); + let mut ser = serde_json::Serializer::with_formatter(&mut buf, Python { indent, depth: 0, has_value: false }); + value.serialize(&mut ser).expect("serializable"); + String::from_utf8(buf).expect("utf-8") +} + +/// One line of JSON as Python's `json.dumps(value)` writes it. +pub fn to_json_line(value: &impl serde::Serialize) -> String { + python_json(value, None) +} + +/// JSON as Python's `json.dumps(value, indent=n)` writes it. +pub fn to_json_indent(value: &impl serde::Serialize, n: usize) -> String { + python_json(value, Some(&b" "[..n])) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn python_spacing() { + let v = serde_json::json!({"a": [1, 2], "b": {}, "c": "é", "d": []}); + assert_eq!(to_json_line(&v), r#"{"a": [1, 2], "b": {}, "c": "\u00e9", "d": []}"#); + assert_eq!(to_json_indent(&v, 1), "{\n \"a\": [\n 1,\n 2\n ],\n \"b\": {},\n \"c\": \"\\u00e9\",\n \"d\": []\n}"); + assert_eq!(to_json_indent(&serde_json::json!({"t": [0]}), 0), "{\n\"t\": [\n0\n]\n}"); + } +} diff --git a/crates/mirth-lab/src/driver.rs b/crates/mirth-lab/src/driver.rs new file mode 100644 index 0000000..673f6bf --- /dev/null +++ b/crates/mirth-lab/src/driver.rs @@ -0,0 +1,166 @@ +//! Running a check over many tests: a thread pool, one JSON line of results per test, findings +//! written to their own directories, and the frontier loop's options (pause at the first +//! finding, recheck only the tests with findings, leave known findings out). + +use std::collections::BTreeSet; +use std::fs::{self, File, OpenOptions}; +use std::io::{BufWriter, Write}; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::Mutex; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; + +use rayon::prelude::*; +use serde::Serialize; +use serde_json::Value; + +use crate::uitest::Test; + +/// The options every sweep over UI tests takes. +#[derive(clap::Args, Clone, Debug)] +pub struct Sweep { + /// The rustc's test directory (`/tests/ui`). + #[arg(long)] + pub tests: PathBuf, + /// Where results, findings and scratch files go. + #[arg(long)] + pub work: PathBuf, + /// Only tests whose path contains this. + #[arg(long)] + pub only: Option, + /// A file of test paths to leave out (known findings). + #[arg(long)] + pub known: Option, + #[arg(long, default_value_t = 8)] + pub jobs: usize, + /// Stop starting new tests at the first finding, and exit 3 (the frontier loop). + #[arg(long)] + pub pause_on_finding: bool, + /// Run only the tests that have findings under /findings. + #[arg(long)] + pub recheck: bool, +} + +impl Sweep { + pub fn scratch(&self) -> PathBuf { + let s = self.work.join("scratch"); + let _ = fs::create_dir_all(&s); + s + } + + /// Filter `tests` by --only, --known and --recheck. + pub fn select(&self, tests: Vec) -> Vec { + let known: BTreeSet = self + .known + .as_ref() + .and_then(|k| fs::read_to_string(k).ok()) + .map(|t| t.split_whitespace().map(str::to_owned).collect()) + .unwrap_or_default(); + let wanted = self.recheck.then(|| findings_tests(&self.work)); + tests + .into_iter() + .filter(|t| { + !known.contains(&t.rel) + && self.only.as_ref().is_none_or(|o| t.rel.contains(o.as_str())) + && wanted.as_ref().is_none_or(|w| w.contains(&t.rel)) + }) + .collect() + } +} + +/// The tests named by /findings/*/finding.json. +pub fn findings_tests(work: &Path) -> BTreeSet { + let mut out = BTreeSet::new(); + if let Ok(dir) = fs::read_dir(work.join("findings")) { + for entry in dir.flatten() { + if let Ok(text) = fs::read_to_string(entry.path().join("finding.json")) + && let Ok(v) = serde_json::from_str::(&text) + && let Some(t) = v.get("test").and_then(Value::as_str) + { + out.insert(t.to_owned()); + } + } + } + out +} + +/// What a check reports for one test. +pub trait Record: Serialize + Send { + /// The findings, one line each (empty when there are none). + fn findings(&self) -> Vec; + fn test(&self) -> &str; +} + +/// Write a finding's directory: the test source, extra files, and `finding.json`. +pub fn write_finding(work: &Path, test: &Test, files: &[(String, Vec)], detail: &impl Serialize) { + let dir = work.join("findings").join(test.rel.replace('/', "__")); + let _ = fs::remove_dir_all(&dir); + if fs::create_dir_all(&dir).is_err() { + return; + } + let _ = fs::copy(&test.path, dir.join(test.file_name())); + for (name, bytes) in files { + let _ = fs::write(dir.join(name), bytes); + } + let mut value = serde_json::to_value(detail).unwrap_or(Value::Null); + if let Value::Object(map) = &mut value { + map.insert("test".into(), Value::String(test.rel.clone())); + map.insert("flags".into(), serde_json::json!(test.flags)); + map.insert("edition".into(), serde_json::json!(test.edition)); + } + let _ = fs::write(dir.join("finding.json"), serde_json::to_string_pretty(&value).unwrap_or_default()); +} + +/// Run `check` over `items` on `jobs` threads; results go to /results.jsonl (appended). +/// Returns exit 3 if paused at a finding, 0 otherwise. +pub fn drive(items: &[T], sweep: &Sweep, check: impl Fn(&T) -> R + Sync) -> ExitCode { + let _ = fs::create_dir_all(&sweep.work); + let file = OpenOptions::new().create(true).append(true).open(sweep.work.join("results.jsonl")); + let out: Mutex>> = Mutex::new(file.ok().map(BufWriter::new)); + let stop = AtomicBool::new(false); + let done = AtomicUsize::new(0); + let with_findings = AtomicUsize::new(0); + let total = items.len(); + // Large stacks: in-process parsers (syn in the rewrites) recurse as deep as a test nests. + let pool = rayon::ThreadPoolBuilder::new() + .num_threads(sweep.jobs.max(1)) + .stack_size(256 << 20) + .build() + .expect("thread pool"); + pool.install(|| { + items.par_iter().for_each(|item| { + if stop.load(Ordering::Relaxed) { + return; + } + let record = check(item); + let findings = record.findings(); + if let Ok(mut guard) = out.lock() + && let Some(w) = guard.as_mut() + { + let _ = serde_json::to_writer(&mut *w, &record); + let _ = w.write_all(b"\n"); + let _ = w.flush(); + } + if !findings.is_empty() { + with_findings.fetch_add(1, Ordering::Relaxed); + println!("FINDING {}: {:?}", record.test(), findings); + if sweep.pause_on_finding { + stop.store(true, Ordering::Relaxed); + } + } + let n = done.fetch_add(1, Ordering::Relaxed) + 1; + if n % 100 == 0 { + println!("{n}/{total} done, {} with findings", with_findings.load(Ordering::Relaxed)); + } + }) + }); + let n = done.load(Ordering::Relaxed); + let f = with_findings.load(Ordering::Relaxed); + println!("{n} tests, {f} with findings"); + if f > 0 && sweep.pause_on_finding { ExitCode::from(3) } else { ExitCode::SUCCESS } +} + +/// A per-test scratch directory, removed when dropped. +pub fn scratch_dir(sweep: &Sweep) -> tempfile::TempDir { + tempfile::Builder::new().prefix("t").tempdir_in(sweep.scratch()).expect("scratch directory") +} diff --git a/crates/mirth-lab/src/lib.rs b/crates/mirth-lab/src/lib.rs new file mode 100644 index 0000000..ca3d0e3 --- /dev/null +++ b/crates/mirth-lab/src/lib.rs @@ -0,0 +1,12 @@ +//! Checks (oracles) over rustc: the shared parts. Each check is a subcommand of the +//! `mirth-lab` binary, in `src/tools/`; docs/checks.md says what each looks for and found. + +pub mod artifacts; +pub mod cargo; +pub mod coverage; +pub mod driver; +pub mod miri; +pub mod mutations; +pub mod normalize; +pub mod rustc; +pub mod uitest; diff --git a/crates/mirth-lab/src/main.rs b/crates/mirth-lab/src/main.rs new file mode 100644 index 0000000..7cdf0ad --- /dev/null +++ b/crates/mirth-lab/src/main.rs @@ -0,0 +1,161 @@ +//! `mirth-lab …`: the checks over rustc. Each subcommand's module documents what it +//! looks for; docs/checks.md has what each found. + +use std::process::ExitCode; + +use clap::{Parser, Subcommand}; + +mod tools { + pub mod abi_diff; + pub mod audit_options; + pub mod callgraph; + pub mod coverage; + pub mod coverage_compact; + pub mod coverage_flags; + pub mod coverage_generators; + pub mod crash_diff; + pub mod diag_check; + pub mod fuzz; + pub mod fuzz_replay; + pub mod flag_fuzz; + pub mod flag_min; + pub mod flag_model; + pub mod flag_rows; + pub mod flag_universe; + pub mod flag_walk; + pub mod gate_check; + pub mod grammar_coverage; + pub mod instr_check; + pub mod miri_diff; + pub mod opt_diff; + pub mod release_diff; + pub mod replay; + pub mod repro_diff; + pub mod rewrite_diff; + pub mod scale_check; + pub mod solver_diff; + pub mod suggest_diff; + pub mod ui_coverage; + pub mod ui_fuzz; + pub mod xlink; +} + +#[derive(Parser)] +#[command(name = "mirth-lab", about = "Checks (oracles) over rustc")] +struct Cli { + #[command(subcommand)] + check: Check, +} + +#[derive(Subcommand)] +enum Check { + /// Behavior must not depend on optimization (opt levels, MIR opt levels, LTO, Cranelift). + OptDiff(tools::opt_diff::Args), + /// The old and new trait solvers, and NLL and Polonius, must agree. + SolverDiff(tools::solver_diff::Args), + /// rustc's internal checks (debug assertions, MIR validation) on every UI test. + CrashDiff(tools::crash_diff::Args), + /// Invariants of every diagnostic (no internal debug output, spans in bounds). + DiagCheck(tools::diag_check::Args), + /// Accepted safe programs are UB-free under Miri; MIR optimizations and native code agree with Miri. + MiriDiff(tools::miri_diff::Args), + /// Meaning-preserving rewrites (generic-wrap, alias, reorder, unused) keep the verdict. + RewriteDiff(tools::rewrite_diff::Args), + /// Machine-applicable suggestions, applied one at a time, keep the program compiling. + SuggestDiff(tools::suggest_diff::Args), + /// Outputs depend only on inputs: repeat, other directory, threads, decoy libraries. + ReproDiff(tools::repro_diff::Args), + /// Nothing unstable is usable from stable code (attributes, library items). + GateCheck(tools::gate_check::Args), + /// PGO and coverage instrumentation round trips. + InstrCheck(tools::instr_check::Args), + /// Real crates accepted by one toolchain are accepted by the next, in comparable time. + ReleaseDiff(tools::release_diff::Args), + /// Every target builds core and alloc and links a program with no undefined symbols. + Xlink(tools::xlink::Args), + /// Compile time, memory, frames and future sizes grow about linearly with program size. + ScaleCheck(tools::scale_check::Args), + /// rustc's extern "C" lowering matches clang's for random C signatures, per target. + AbiDiff(tools::abi_diff::Args), + /// Incremental rebuilds of UI tests after random edits match clean builds. + UiFuzz(tools::ui_fuzz::Args), + /// Reachability over the compiler's call graph; coverage of what can run, and gap lists. + Callgraph(tools::callgraph::Args), + /// rustc's -C and -Z options: domains, values and pairs accepted, covering-array sizes. + FlagUniverse(tools::flag_universe::Args), + /// A PICT model of the option universe (optionally of transitions, for a Cargo build). + FlagModel(tools::flag_model::Args), + /// Compile a trivial crate per PICT row; count the rejected rows by first error. + FlagRows(tools::flag_rows::Args), + /// Option transitions between incremental sessions: the rebuild must match a clean build. + FlagWalk(tools::flag_walk::Args), + /// Delta-debug the options of a flag walk's failing rows to a minimal set per error. + FlagMin(tools::flag_min::Args), + /// The fuzzer under the option configurations of a PICT table. + FlagFuzz(tools::flag_fuzz::Args), + /// [UNTRACKED] options must not change what incremental compilation reuses. + AuditOptions(tools::audit_options::Args), + /// Which of the compiler's functions ran, per crate and file, from coverage logs. + Coverage(tools::coverage::Args), + /// Fold coverage logs into a running union as they finish, and delete them. + CoverageCompact(tools::coverage_compact::Args), + /// Compiler runs the test suites hardly make (prints, every target, links, dumps), for coverage. + CoverageGenerators(tools::coverage_generators::Args), + /// Which alternatives and tokens of Ur's Rust grammar a fixture's sources use. + GrammarCoverage(tools::grammar_coverage::Args), + /// The functions each UI test reaches beyond a baseline; a small set reaching the most. + UiCoverage(tools::ui_coverage::Args), + /// Random edits to a fixture; each incremental rebuild must match a clean build (P6). + Fuzz(tools::fuzz::Args), + /// Replay a fuzz finding's edits and compare the last incremental build with a clean one. + FuzzReplay(tools::fuzz_replay::Args), + /// A crate's git history through incremental builds, each compared with a clean build. + Replay(tools::replay::Args), + /// Coverage of the compiler across option configurations (a fixture per PICT transitions row). + CoverageFlags(tools::coverage_flags::Args), +} + +fn main() -> ExitCode { + let cli = Cli::parse(); + let result = match cli.check { + Check::OptDiff(a) => tools::opt_diff::run(a), + Check::SolverDiff(a) => tools::solver_diff::run(a), + Check::CrashDiff(a) => tools::crash_diff::run(a), + Check::DiagCheck(a) => tools::diag_check::run(a), + Check::MiriDiff(a) => tools::miri_diff::run(a), + Check::RewriteDiff(a) => tools::rewrite_diff::run(a), + Check::SuggestDiff(a) => tools::suggest_diff::run(a), + Check::ReproDiff(a) => tools::repro_diff::run(a), + Check::GateCheck(a) => tools::gate_check::run(a), + Check::InstrCheck(a) => tools::instr_check::run(a), + Check::ReleaseDiff(a) => tools::release_diff::run(a), + Check::Xlink(a) => tools::xlink::run(a), + Check::ScaleCheck(a) => tools::scale_check::run(a), + Check::AbiDiff(a) => tools::abi_diff::run(a), + Check::UiFuzz(a) => tools::ui_fuzz::run(a), + Check::Callgraph(a) => tools::callgraph::run(a), + Check::FlagUniverse(a) => tools::flag_universe::run(a), + Check::FlagModel(a) => tools::flag_model::run(a), + Check::FlagRows(a) => tools::flag_rows::run(a), + Check::FlagWalk(a) => tools::flag_walk::run(a), + Check::FlagMin(a) => tools::flag_min::run(a), + Check::FlagFuzz(a) => tools::flag_fuzz::run(a), + Check::AuditOptions(a) => tools::audit_options::run(a), + Check::Coverage(a) => tools::coverage::run(a), + Check::CoverageCompact(a) => tools::coverage_compact::run(a), + Check::CoverageGenerators(a) => tools::coverage_generators::run(a), + Check::GrammarCoverage(a) => tools::grammar_coverage::run(a), + Check::UiCoverage(a) => tools::ui_coverage::run(a), + Check::Fuzz(a) => tools::fuzz::run(a), + Check::FuzzReplay(a) => tools::fuzz_replay::run(a), + Check::Replay(a) => tools::replay::run(a), + Check::CoverageFlags(a) => tools::coverage_flags::run(a), + }; + match result { + Ok(code) => code, + Err(e) => { + eprintln!("mirth-lab: {e:#}"); + ExitCode::from(2) + } + } +} diff --git a/crates/mirth-lab/src/miri.rs b/crates/mirth-lab/src/miri.rs new file mode 100644 index 0000000..4811531 --- /dev/null +++ b/crates/mirth-lab/src/miri.rs @@ -0,0 +1,82 @@ +//! Interpreting a test with Miri (the pinned nightly's, with the sysroot `cargo miri setup` +//! makes). + +use std::path::{Path, PathBuf}; +use std::process::Command; +use std::time::Duration; + +use serde::Serialize; + +use crate::rustc::{Exit, run_command}; + +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum MiriStatus { + /// Ran to the end, whatever the exit code. + Ok, + Ub, + Unsupported, + Error, + Ice, + Timeout, +} + +#[derive(Clone, Debug, Serialize)] +pub struct MiriRun { + pub status: MiriStatus, + pub exit: Exit, + pub stdout: String, + pub stderr: String, +} + +pub struct Miri { + pub binary: PathBuf, + pub sysroot: PathBuf, +} + +impl Miri { + /// The pinned toolchain's Miri and the default `cargo miri setup` sysroot. + pub fn pinned(toolchain: &str) -> Miri { + let home = PathBuf::from(std::env::var("HOME").unwrap_or_default()); + Miri { + binary: home.join(format!(".rustup/toolchains/{toolchain}-x86_64-unknown-linux-gnu/bin/miri")), + sysroot: home.join(".cache/miri"), + } + } + + pub fn run(&self, source: &Path, flags: &[String], edition: &str, extra: &[String], timeout: u64, cwd: &Path) -> MiriRun { + let mut cmd = Command::new(&self.binary); + cmd.arg("--sysroot") + .arg(&self.sysroot) + .arg(source) + .args(["--edition", edition]) + .args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .args(["-Zmiri-disable-isolation", "-Zmiri-deterministic-floats"]) + .args(flags) + .args(extra) + .current_dir(cwd) + .env("RUSTC_BOOTSTRAP", "1") + .env("RUST_BACKTRACE", "0"); + let done = match run_command(cmd, Duration::from_secs(timeout)) { + Ok(d) => d, + Err(e) => { + return MiriRun { status: MiriStatus::Error, exit: Exit::Code(-1), stdout: String::new(), stderr: e.to_string() }; + } + }; + let err = done.stderr_text(); + let status = if done.exit == Exit::Timeout { + MiriStatus::Timeout + } else if err.contains("Undefined Behavior:") { + MiriStatus::Ub + } else if err.contains("unsupported operation") || err.contains("can't call foreign function") { + MiriStatus::Unsupported + } else if crate::rustc::is_ice(&err) { + MiriStatus::Ice + } else if done.exit == Exit::Code(1) && !err.contains("panicked") && err.lines().any(|l| l.starts_with("error")) { + MiriStatus::Error + } else { + MiriStatus::Ok + }; + MiriRun { status, exit: done.exit.clone(), stdout: done.stdout_text(), stderr: err } + } +} diff --git a/crates/mirth-lab/src/mutations.rs b/crates/mirth-lab/src/mutations.rs new file mode 100644 index 0000000..46749e5 --- /dev/null +++ b/crates/mirth-lab/src/mutations.rs @@ -0,0 +1,306 @@ +//! Random mechanical edits to Rust source, shared by the fuzzers. +//! +//! Each edit takes the file's text, a random generator and a counter, and returns the new text, +//! or None when it does not apply. EDITS lists them with weights. + +use std::sync::LazyLock; + +use rand::distr::Distribution; +use rand::distr::weighted::WeightedIndex; +use rand::rngs::StdRng; +use rand::seq::{IndexedRandom, SliceRandom}; +use rand::{Rng, SeedableRng}; +use regex::Regex; + +pub type Edit = fn(&str, &mut StdRng, usize) -> Option; + +/// A generator seeded by a name (a test path), so a run can be repeated. +pub fn seeded(name: &str) -> StdRng { + let digest = crate::artifacts::sha256(name.as_bytes()); + StdRng::seed_from_u64(u64::from_str_radix(&digest[..16], 16).unwrap()) +} + +/// A random edit, by weight. +pub fn pick(rng: &mut StdRng) -> (&'static str, Edit) { + static WEIGHTS: LazyLock> = LazyLock::new(|| WeightedIndex::new(EDITS.iter().map(|e| e.2)).unwrap()); + let (name, f, _) = EDITS[WEIGHTS.sample(rng)]; + (name, f) +} + +static INDENT: LazyLock = LazyLock::new(|| Regex::new(r"^\s*").unwrap()); + +fn indent(line: &str) -> &str { + INDENT.find(line).map_or("", |m| m.as_str()) +} + +/// Top-level items: blank-line separated, continuation blocks merged. +pub fn blocks(text: &str) -> Vec { + let mut out: Vec = Vec::new(); + for b in text.split("\n\n") { + let continues = b.chars().next().is_some_and(char::is_whitespace) || b.starts_with('}') || b.starts_with("where"); + match out.last_mut() { + Some(last) if continues => { + last.push_str("\n\n"); + last.push_str(b); + } + _ => out.push(b.to_owned()), + } + } + out +} + +fn lines(text: &str) -> Vec { + text.split('\n').map(str::to_owned).collect() +} + +pub fn comment_line(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut l = lines(text); + let i = rng.random_range(0..=l.len()); + let ind = indent(l.get(i).map_or("", String::as_str)).to_owned(); + l.insert(i, format!("{ind}// fuzz {n}")); + Some(l.join("\n")) +} + +pub fn blank_line(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let i = rng.random_range(0..=l.len()); + l.insert(i, String::new()); + Some(l.join("\n")) +} + +pub fn remove_comment(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()) + .filter(|&i| { + let t = l[i].trim(); + t.starts_with("//") && !t.starts_with("//!") + }) + .collect(); + let i = *idx.choose(rng)?; + l.remove(i); + Some(l.join("\n")) +} + +pub fn indent_line(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| !l[i].trim().is_empty()).collect(); + let i = *idx.choose(rng)?; + l[i] = format!(" {}", l[i]); + Some(l.join("\n")) +} + +pub fn swap_items(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let i = rng.random_range(1..b.len() - 1); + b.swap(i, i + 1); + Some(b.join("\n\n")) +} + +pub fn move_item_to_end(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + let item = b.remove(rng.random_range(1..b.len())); + b.push(item.trim_end_matches('\n').to_owned()); + Some(b.join("\n\n") + "\n") +} + +pub fn delete_item(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut b = blocks(text); + if b.len() < 3 { + return None; + } + b.remove(rng.random_range(1..b.len())); + Some(b.join("\n\n")) +} + +static FN_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^(pub(\([^)]*\))? )?(const )?(async )?fn \w+").unwrap()); +static FN_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\bfn (\w+)").unwrap()); + +pub fn duplicate_fn(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut b = blocks(text); + let fns: Vec = (0..b.len()).filter(|&i| FN_ITEM.is_match(&b[i])).collect(); + let i = *fns.choose(rng)?; + let copy = FN_NAME.replacen(&b[i], 1, format!("fn ${{1}}_fuzz{n}")).into_owned(); + b.insert(i + 1, copy); + Some(b.join("\n\n")) +} + +const ADDITIONS: &[&str] = &[ + "fn fuzz_private_{n}() -> u32 { {n} }", + "pub fn fuzz_public_{n}(x: u32) -> u32 { x.wrapping_mul({n}) }", + "#[inline]\npub fn fuzz_inline_{n}(x: &T) -> (T, u32) { (x.clone(), {n}) }", + "pub const FUZZ_{n}: &str = \"fuzz {n}\";", + "pub static FUZZ_STATIC_{n}: [u8; 3] = [{n} as u8, 1, 2];", + "#[derive(Debug, Clone, PartialEq)]\npub struct Fuzz{n} { pub items: [T; N], pub tag: &'static str }", + "pub enum FuzzEnum{n} { A(u32), B { x: i64 }, C }", + "pub trait FuzzTrait{n} { fn go(&self) -> impl Sized; const K: u32 = {n}; }", + "pub async fn fuzz_async_{n}() -> u32 { {n} }", + "pub type FuzzAlias{n} = Vec<(T, u32)>;", + "macro_rules! fuzz_macro_{n} { ($e:expr) => { $e + {n} }; }", + "pub mod fuzz_mod_{n} { pub fn inner() -> &'static str { \"{n}\" } }", +]; + +pub fn add_item(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut b = blocks(text); + let i = rng.random_range(1..=b.len()); + b.insert(i, ADDITIONS.choose(rng).unwrap().replace("{n}", &n.to_string())); + Some(b.join("\n\n")) +} + +fn word_or(c: Option, extra: char) -> bool { + c.is_some_and(|c| c.is_alphanumeric() || c == '_' || c == extra) +} + +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +pub fn int_literal(text: &str, rng: &mut StdRng, _: usize) -> Option { + // A whole run of digits, not touching a word character or a dot on either side. + let ms: Vec = DIGITS + .find_iter(text) + .filter(|m| !word_or(text[..m.start()].chars().next_back(), '.') && !word_or(text[m.end()..].chars().next(), '.')) + .collect(); + let m = ms.choose(rng)?; + let bumped = m.as_str().parse::().map_or_else(|_| format!("{}1", m.as_str()), |v| (v + 1).to_string()); + Some(format!("{}{bumped}{}", &text[..m.start()], &text[m.end()..])) +} + +static STRING: LazyLock = LazyLock::new(|| Regex::new(r#""([^"\\\n]*)""#).unwrap()); + +pub fn str_literal(text: &str, rng: &mut StdRng, _: usize) -> Option { + // A string literal not preceded by a word character or a backslash; after a rejected + // opening quote the search resumes one character later, as a lookbehind would. + let mut ms = Vec::new(); + let mut pos = 0; + while let Some(c) = STRING.captures_at(text, pos) { + let m = c.get(0).unwrap(); + if word_or(text[..m.start()].chars().next_back(), '\\') { + pos = m.start() + 1; + } else { + ms.push((m.start(), m.end(), c[1].to_owned())); + pos = m.end(); + } + } + let (s, e, body) = ms.choose(rng)?; + Some(format!("{}\"{body}~\"{}", &text[..*s], &text[*e..])) +} + +static FN_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(const )?fn ").unwrap()); + +pub fn toggle_inline(text: &str, rng: &mut StdRng, _: usize) -> Option { + let mut l = lines(text); + let inl: Vec = (0..l.len()).filter(|&i| matches!(l[i].trim(), "#[inline]" | "#[inline(never)]" | "#[inline(always)]")).collect(); + let fns: Vec = (0..l.len()).filter(|&i| FN_LINE.is_match(&l[i])).collect(); + if !inl.is_empty() && rng.random::() < 0.5 { + let i = *inl.choose(rng).unwrap(); + l.remove(i); + } else if !fns.is_empty() { + let i = *fns.choose(rng).unwrap(); + let attr = ["#[inline]", "#[inline(never)]", "#[cold]", "#[must_use]"].choose(rng).unwrap(); + let ind = indent(&l[i]).to_owned(); + l.insert(i, format!("{ind}{attr}")); + } else { + return None; + } + Some(l.join("\n")) +} + +static ITEM_LINE: LazyLock = + LazyLock::new(|| Regex::new(r"^\s*(pub(\([^)]*\))? )?(fn|struct|enum|trait|const|static|type|mod) ").unwrap()); + +pub fn doc_comment(text: &str, rng: &mut StdRng, n: usize) -> Option { + let mut l = lines(text); + let idx: Vec = (0..l.len()).filter(|&i| ITEM_LINE.is_match(&l[i])).collect(); + let i = *idx.choose(rng)?; + let ind = indent(&l[i]).to_owned(); + l.insert(i, format!("{ind}/// Fuzz doc {n}, see [`Vec`].")); + Some(l.join("\n")) +} + +static DERIVE: LazyLock = LazyLock::new(|| Regex::new(r"#\[derive\(([^)]*)\)\]").unwrap()); + +pub fn reorder_derive(text: &str, rng: &mut StdRng, _: usize) -> Option { + let ms: Vec = DERIVE.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let mut names: Vec<&str> = c[1].split(',').map(str::trim).filter(|x| !x.is_empty()).collect(); + if names.len() < 2 { + return None; + } + names.shuffle(rng); + let m = c.get(0).unwrap(); + Some(format!("{}#[derive({})]{}", &text[..m.start()], names.join(", "), &text[m.end()..])) +} + +static PUB_ITEM: LazyLock = LazyLock::new(|| Regex::new(r"\bpub (fn|struct|enum|const|static|trait|mod|type) ").unwrap()); + +pub fn narrow_visibility(text: &str, rng: &mut StdRng, _: usize) -> Option { + let ms: Vec = PUB_ITEM.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let m = c.get(0).unwrap(); + Some(format!("{}pub(crate) {} {}", &text[..m.start()], &c[1], &text[m.end()..])) +} + +static LET: LazyLock = LazyLock::new(|| Regex::new(r"\blet (mut )?([a-z_][a-z0-9_]*)\b").unwrap()); + +pub fn rename_local(text: &str, rng: &mut StdRng, n: usize) -> Option { + let ms: Vec = LET.captures_iter(text).collect(); + let c = ms.choose(rng)?; + let name = &c[2]; + if name == "_" { + return None; + } + let m = c.get(0).unwrap(); + // Rename from the binding to the end of the enclosing top-level block. + let end = text[m.end()..].find("\n}\n").map_or(text.len(), |i| m.end() + i); + let re = Regex::new(&format!(r"\b{}\b", regex::escape(name))).unwrap(); + let body = re.replace_all(&text[m.start()..end], format!("{name}_f{n}").as_str()).into_owned(); + Some(format!("{}{body}{}", &text[..m.start()], &text[end..])) +} + +pub const EDITS: &[(&str, Edit, u32)] = &[ + ("comment_line", comment_line, 10), + ("blank_line", blank_line, 6), + ("remove_comment", remove_comment, 3), + ("indent_line", indent_line, 4), + ("swap_items", swap_items, 6), + ("move_item_to_end", move_item_to_end, 3), + ("delete_item", delete_item, 2), + ("duplicate_fn", duplicate_fn, 4), + ("add_item", add_item, 8), + ("int_literal", int_literal, 6), + ("str_literal", str_literal, 4), + ("toggle_inline", toggle_inline, 4), + ("doc_comment", doc_comment, 4), + ("reorder_derive", reorder_derive, 2), + ("narrow_visibility", narrow_visibility, 2), + ("rename_local", rename_local, 3), +]; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn literals() { + let mut rng = seeded("t"); + assert_eq!(int_literal("let x = 12; y.0; a1", &mut rng, 0).unwrap(), "let x = 13; y.0; a1"); + assert_eq!(str_literal(r#"r"a" + "b""#, &mut rng, 0).unwrap(), r#"r"a" + "b~""#); + assert!(int_literal("x1 y.2", &mut rng, 0).is_none()); + } + + #[test] + fn blocks_merge_continuations() { + assert_eq!(blocks("fn a() {\n\n x\n}\n\nfn b() {}"), vec!["fn a() {\n\n x\n}", "fn b() {}"]); + } + + #[test] + fn rename() { + let mut rng = seeded("t"); + let t = "fn f() {\n let x = 1;\n x + 1\n}\nfn g() { x }\n"; + assert_eq!(rename_local(t, &mut rng, 3).unwrap(), "fn f() {\n let x_f3 = 1;\n x_f3 + 1\n}\nfn g() { x }\n"); + } +} diff --git a/crates/mirth-lab/src/normalize.rs b/crates/mirth-lab/src/normalize.rs new file mode 100644 index 0000000..bb85aa9 --- /dev/null +++ b/crates/mirth-lab/src/normalize.rs @@ -0,0 +1,53 @@ +//! What two runs of a program may print differently without the program behaving differently. + +use std::sync::LazyLock; + +use regex::Regex; + +static THREAD_ID: LazyLock = LazyLock::new(|| Regex::new(r"(thread '[^']*') \(\d+\)").unwrap()); +static STD_PATH: LazyLock = + LazyLock::new(|| Regex::new(r"\S*/lib/rustlib/src/rust/library/|/rustc/[0-9a-f]+/library/").unwrap()); +static REGISTRY: LazyLock = + LazyLock::new(|| Regex::new(r"\S*/registry/(src/)?[^/\s]+/([^/\s]+-\d[^/\s]*)/").unwrap()); +static TIMING: LazyLock = LazyLock::new(|| Regex::new(r"finished in \d+\.\d+s").unwrap()); + +/// Panic messages name the thread with its OS id; toolchains print std's and dependencies' paths +/// differently (in full with rust-src, remapped, relative); backtrace hints come and go. +pub fn stderr(text: &str) -> String { + let lines: Vec<&str> = text + .lines() + .filter(|l| { + !l.starts_with("note: run with `RUST_BACKTRACE") + && !l.starts_with("note: Some details are omitted") + && !l.starts_with("note: in Miri, you may have to set `MIRIFLAGS") + }) + .collect(); + let t = THREAD_ID.replace_all(&lines.join("\n"), "$1").into_owned(); + let t = STD_PATH.replace_all(&t, "library/").into_owned(); + REGISTRY.replace_all(&t, "/$2/").into_owned() +} + +/// The test harness prints how long tests took, and its result lines in completion order. +pub fn stdout(text: &str) -> String { + let t = TIMING.replace_all(text, "finished in …s").into_owned(); + let is_result = |l: &str| l.starts_with("test ") && l.contains(" ... "); + let mut results: Vec<&str> = t.split('\n').filter(|l| is_result(l)).collect(); + results.sort_unstable(); + let mut it = results.into_iter(); + t.split('\n').map(|l| if is_result(l) { it.next().unwrap_or(l) } else { l }).collect::>().join("\n") +} + +#[cfg(test)] +mod tests { + #[test] + fn thread_ids_and_paths() { + let s = super::stderr("thread 'main' (909942) panicked at /h/.rustup/x/lib/rustlib/src/rust/library/core/src/a.rs:1:2:"); + assert_eq!(s, "thread 'main' panicked at library/core/src/a.rs:1:2:"); + } + + #[test] + fn harness_order() { + let s = super::stdout("test b ... ok\ntest a ... ok\nfinished in 0.12s"); + assert_eq!(s, "test a ... ok\ntest b ... ok\nfinished in …s"); + } +} diff --git a/crates/mirth-lab/src/rustc.rs b/crates/mirth-lab/src/rustc.rs new file mode 100644 index 0000000..8952f3c --- /dev/null +++ b/crates/mirth-lab/src/rustc.rs @@ -0,0 +1,315 @@ +//! Running rustc (and the programs it builds, and Miri) with timeouts, and reading its JSON +//! diagnostics into types. + +use std::io::Read; +use std::path::{Path, PathBuf}; +use std::process::{Command, Stdio}; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; +use wait_timeout::ChildExt; + +/// How a command ended. +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Exit { + Code(i32), + Signal(i32), + Timeout, +} + +/// What a finished command printed. +#[derive(Clone, Debug)] +pub struct Finished { + pub exit: Exit, + pub stdout: Vec, + pub stderr: Vec, +} + +impl Finished { + pub fn success(&self) -> bool { + self.exit == Exit::Code(0) + } + pub fn stderr_text(&self) -> String { + String::from_utf8_lossy(&self.stderr).into_owned() + } + pub fn stdout_text(&self) -> String { + String::from_utf8_lossy(&self.stdout).into_owned() + } +} + +/// Run `cmd` with stdin closed, its output captured, killed after `timeout`. +pub fn run_command(mut cmd: Command, timeout: Duration) -> std::io::Result { + cmd.stdin(Stdio::null()).stdout(Stdio::piped()).stderr(Stdio::piped()); + // A binary just written can be "busy" when another thread forked while it was open for + // writing (the child holds the descriptor until it execs): retry for a while. + let mut tries = 0; + let mut child = loop { + match cmd.spawn() { + Err(e) if e.raw_os_error() == Some(26) && tries < 100 => { + tries += 1; + std::thread::sleep(Duration::from_millis(20)); + } + other => break other?, + } + }; + // Read both pipes on threads, so a chatty child cannot block on a full pipe. + let mut out = child.stdout.take().expect("piped"); + let mut err = child.stderr.take().expect("piped"); + let out_thread = std::thread::spawn(move || { + let mut v = Vec::new(); + let _ = out.read_to_end(&mut v); + v + }); + let err_thread = std::thread::spawn(move || { + let mut v = Vec::new(); + let _ = err.read_to_end(&mut v); + v + }); + let exit = match child.wait_timeout(timeout)? { + Some(status) => match status.code() { + Some(code) => Exit::Code(code), + None => { + use std::os::unix::process::ExitStatusExt; + Exit::Signal(status.signal().unwrap_or(0)) + } + }, + None => { + let _ = child.kill(); + let _ = child.wait(); + Exit::Timeout + } + }; + Ok(Finished { + exit, + stdout: out_thread.join().unwrap_or_default(), + stderr: err_thread.join().unwrap_or_default(), + }) +} + +pub fn is_ice(stderr: &str) -> bool { + stderr.contains("internal compiler error") + || stderr.contains("the compiler unexpectedly panicked") + || stderr.contains("rustc interrupted by SIG") +} + +/// The outcome of a compilation. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)] +#[serde(rename_all = "lowercase")] +pub enum Status { + Ok, + Error, + Ice, + Timeout, +} + +/// A compilation's result: its status, what it printed, and the program if it built one. +#[derive(Clone, Debug)] +pub struct Compiled { + pub status: Status, + pub stderr: String, + pub binary: Option, +} + +/// One rustc invocation, the way the checks compile a test. +pub struct Compile<'a> { + pub rustc: &'a Path, + pub source: &'a Path, + pub out_dir: &'a Path, + pub flags: &'a [String], + pub edition: &'a str, + pub extra: Vec, + /// `link`, `metadata`, `link,metadata`, ... + pub emit: &'a str, + pub json: bool, + pub timeout: Duration, + /// Run with `RUSTC_BOOTSTRAP=1` (the default; a stable user's view needs it off). + pub bootstrap: bool, + /// Name the output `/prog` with `-o` (the default); off when the extra options say + /// where outputs go (`--out-dir`). + pub name_output: bool, +} + +impl<'a> Compile<'a> { + pub fn new(rustc: &'a Path, source: &'a Path, out_dir: &'a Path, flags: &'a [String], edition: &'a str) -> Self { + Compile { + rustc, + source, + out_dir, + flags, + edition, + extra: Vec::new(), + emit: "link", + json: false, + timeout: Duration::from_secs(300), + bootstrap: true, + name_output: true, + } + } + + pub fn extra, S: Into>(mut self, extra: I) -> Self { + self.extra.extend(extra.into_iter().map(Into::into)); + self + } + + pub fn emit(mut self, emit: &'a str) -> Self { + self.emit = emit; + self + } + + pub fn json(mut self) -> Self { + self.json = true; + self + } + + pub fn unnamed_output(mut self) -> Self { + self.name_output = false; + self + } + + pub fn stable(mut self) -> Self { + self.bootstrap = false; + self + } + + pub fn timeout(mut self, secs: u64) -> Self { + self.timeout = Duration::from_secs(secs); + self + } + + pub fn run(self) -> Compiled { + let _ = std::fs::create_dir_all(self.out_dir); + let binary = self.out_dir.join("prog"); + let mut cmd = Command::new(self.rustc); + cmd.arg(self.source) + .args(["--edition", self.edition]) + .arg(format!("--emit={}", self.emit)); + if self.name_output { + cmd.arg("-o").arg(&binary); + } + if self.bootstrap { + cmd.args(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]); + } + cmd + .arg(if self.json { "--error-format=json" } else { "--error-format=short" }) + .args(self.flags) + .args(&self.extra) + .current_dir(self.out_dir) + .env("RUST_BACKTRACE", "0"); + if self.bootstrap { + cmd.env("RUSTC_BOOTSTRAP", "1"); + } else { + cmd.env_remove("RUSTC_BOOTSTRAP"); + } + let done = match run_command(cmd, self.timeout) { + Ok(d) => d, + Err(e) => { + return Compiled { status: Status::Error, stderr: format!("spawning rustc: {e}"), binary: None }; + } + }; + let stderr = done.stderr_text(); + let status = match done.exit { + Exit::Timeout => Status::Timeout, + _ if is_ice(&stderr) => Status::Ice, + Exit::Code(0) => Status::Ok, + _ => Status::Error, + }; + let binary = (status == Status::Ok && self.emit.contains("link") && binary.exists()).then_some(binary); + Compiled { status, stderr, binary } + } +} + +/// What running a built program observed. +#[derive(Clone, Debug, PartialEq, Eq, Serialize)] +pub struct Observed { + pub exit: Exit, + pub stdout: String, + pub stderr: String, +} + +/// Run a program built by a check, from its own directory. +pub fn observe(binary: &Path, timeout_secs: u64, env: &[(&str, &str)]) -> Observed { + let mut cmd = Command::new(binary); + cmd.current_dir(binary.parent().unwrap_or(Path::new("."))).env("RUST_BACKTRACE", "0"); + for (k, v) in env { + cmd.env(k, v); + } + match run_command(cmd, Duration::from_secs(timeout_secs)) { + Ok(d) => Observed { exit: d.exit.clone(), stdout: d.stdout_text(), stderr: d.stderr_text() }, + Err(e) => Observed { exit: Exit::Code(-1), stdout: String::new(), stderr: format!("spawning: {e}") }, + } +} + +/// One diagnostic of `--error-format=json`. +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Diagnostic { + pub message: String, + #[serde(default)] + pub code: Option, + pub level: String, + #[serde(default)] + pub spans: Vec, + #[serde(default)] + pub children: Vec, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Code { + pub code: String, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct Span { + pub file_name: String, + pub byte_start: usize, + pub byte_end: usize, + pub line_start: usize, + pub line_end: usize, + #[serde(default)] + pub is_primary: bool, + #[serde(default)] + pub label: Option, + #[serde(default)] + pub suggested_replacement: Option, + #[serde(default)] + pub suggestion_applicability: Option, +} + +impl Diagnostic { + pub fn code(&self) -> &str { + self.code.as_ref().map_or("", |c| c.code.as_str()) + } + + /// This diagnostic and its children, depth first. + pub fn walk(&self) -> Vec<&Diagnostic> { + let mut out = vec![self]; + for c in &self.children { + out.extend(c.walk()); + } + out + } + + pub fn primary(&self) -> Option<&Span> { + self.spans.iter().find(|s| s.is_primary) + } +} + +/// The diagnostics in rustc's JSON stderr (lines that are not JSON are skipped). +pub fn diagnostics(stderr: &str) -> Vec { + stderr.lines().filter_map(|l| serde_json::from_str(l).ok()).collect() +} + +/// The error codes in human-readable stderr, sorted and deduplicated. +pub fn error_codes(stderr: &str) -> Vec { + static RE: std::sync::LazyLock = + std::sync::LazyLock::new(|| regex::Regex::new(r"error\[(E\d{4})\]").unwrap()); + let mut codes: Vec = RE.captures_iter(stderr).map(|c| c[1].to_owned()).collect(); + codes.sort(); + codes.dedup(); + codes +} + +/// The first line of stderr that starts an error, for reports. +pub fn first_error(stderr: &str) -> String { + stderr.lines().find(|l| l.contains("error")).unwrap_or("").chars().take(300).collect() +} diff --git a/crates/mirth-lab/src/tools/abi_diff.rs b/crates/mirth-lab/src/tools/abi_diff.rs new file mode 100644 index 0000000..1100ce8 --- /dev/null +++ b/crates/mirth-lab/src/tools/abi_diff.rs @@ -0,0 +1,644 @@ +//! ABI differential: rustc's `extern "C"` must lower a signature the way clang lowers the same +//! C signature, on every target both support. +//! +//! Generates random C signatures (bool, integers of each width, float, double, pointers, +//! `__int128` on 64-bit targets, and repr(C) structs, unions and arrays of them, nested, packed +//! or over-aligned), writes each as a C function (clang, the target's LLVM triple, CPU and +//! features) and a Rust `#[no_mangle] extern "C" fn` (rustc against minicore, no sysroot), and +//! compares the two LLVM IR signatures parameter by parameter after first-class aggregates are +//! flattened. A difference in register class, extension, inreg/byval/sret, byval alignment, +//! parameter count or calling convention is a finding; representation-only differences are +//! notes. Verified equivalences per architecture, known bugs (#163911, findings 19 and 20) and +//! two differences this host cannot decide are labelled. + +use std::collections::BTreeMap; +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (for tests/auxiliary/minicore.rs). + #[arg(long)] + rust: PathBuf, + #[arg(long)] + work: PathBuf, + /// Comma-separated targets (default: the main tier 1 and 2 targets). + #[arg(long)] + targets: Option, + /// Every target rustc knows (the non-main ones still show representation differences). + #[arg(long)] + all: bool, + #[arg(long, default_value_t = 200)] + count: usize, + #[arg(long, default_value_t = 1)] + seed: u64, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long, default_value = "clang")] + clang: String, +} + +const MAIN: &[&str] = &[ + "x86_64-unknown-linux-gnu", "x86_64-unknown-linux-musl", "x86_64-pc-windows-msvc", "x86_64-pc-windows-gnu", + "x86_64-apple-darwin", "i686-unknown-linux-gnu", "i686-pc-windows-msvc", "aarch64-unknown-linux-gnu", + "aarch64-apple-darwin", "aarch64-pc-windows-msvc", "aarch64-unknown-linux-musl", "armv7-unknown-linux-gnueabihf", + "arm-unknown-linux-gnueabi", "thumbv7em-none-eabihf", "riscv64gc-unknown-linux-gnu", "riscv32imac-unknown-none-elf", + "loongarch64-unknown-linux-gnu", "powerpc64le-unknown-linux-gnu", "s390x-unknown-linux-gnu", "wasm32-unknown-unknown", + "wasm32-wasip1", +]; + +// ---- generation ---- + +/// splitmix64: a small deterministic generator, so a seed names a program. +struct Rng(u64); +impl Rng { + fn next(&mut self) -> u64 { + self.0 = self.0.wrapping_add(0x9e37_79b9_7f4a_7c15); + let mut z = self.0; + z = (z ^ (z >> 30)).wrapping_mul(0xbf58_476d_1ce4_e5b9); + z = (z ^ (z >> 27)).wrapping_mul(0x94d0_49bb_1331_11eb); + z ^ (z >> 31) + } + fn float(&mut self) -> f64 { + (self.next() >> 11) as f64 / (1u64 << 53) as f64 + } + fn below(&mut self, n: usize) -> usize { + (self.next() % n as u64) as usize + } + fn pick<'a, T>(&mut self, v: &'a [T]) -> &'a T { + &v[self.below(v.len())] + } +} + +const SCALARS: &[(&str, &str)] = &[ + ("_Bool", "bool"), ("signed char", "i8"), ("unsigned char", "u8"), ("short", "i16"), ("unsigned short", "u16"), + ("int", "i32"), ("unsigned int", "u32"), ("long long", "i64"), ("unsigned long long", "u64"), ("float", "f32"), + ("double", "f64"), ("void*", "*mut u8"), +]; +const WIDE: &[(&str, &str)] = &[("__int128", "i128"), ("unsigned __int128", "u128")]; + +#[derive(Clone)] +enum Field { + Scalar(String, String), + Array(String, String, usize), + Named(String), +} + +struct Aggregate { + name: String, + union: bool, + packed: bool, + align: Option, + fields: Vec, +} + +struct Gen { + rng: Rng, + wide: bool, + aggregates: Vec, + names: usize, +} + +impl Gen { + fn scalar(&mut self) -> (String, String) { + let pool: Vec<(&str, &str)> = SCALARS.iter().chain(if self.wide { WIDE } else { &[] }).copied().collect(); + let (c, r) = *self.rng.pick(&pool); + (c.into(), r.into()) + } + fn field(&mut self, depth: u32) -> Field { + let r = self.rng.float(); + if depth < 2 && r < 0.15 { + return Field::Named(self.aggregate(depth + 1)); + } + let (c, rs) = self.scalar(); + if r < 0.25 { + let n = *self.rng.pick(&[1, 2, 3, 4, 8]); + return Field::Array(c, rs, n); + } + Field::Scalar(c, rs) + } + fn aggregate(&mut self, depth: u32) -> String { + let name = format!("S{}", self.names); + self.names += 1; + let union = self.rng.float() < 0.12; + let packed = !union && self.rng.float() < 0.08; + // Rust rejects packed with align, and a packed type holding an over-aligned one. + let align = if packed { None } else { *self.rng.pick(&[None, None, None, None, None, None, None, None, None, Some(16), Some(32)]) }; + let n = 1 + self.rng.below(5); + let fields = (0..n).map(|_| self.field(if packed { 2 } else { depth })).collect(); + self.aggregates.push(Aggregate { name: name.clone(), union, packed, align, fields }); + name + } + /// (C type, Rust type) + fn ty(&mut self) -> (String, String) { + if self.rng.float() < 0.4 { + let n = self.aggregate(0); + (n.clone(), n) + } else { + self.scalar() + } + } +} + +fn program(wide: bool, seed: u64, count: usize) -> (String, String, Vec) { + let mut g = Gen { rng: Rng(seed), wide, aggregates: Vec::new(), names: 0 }; + let mut fns = Vec::new(); + for k in 0..count { + let n = g.rng.below(9); + let params: Vec<(String, String)> = (0..n).map(|_| g.ty()).collect(); + let ret = if g.rng.float() < 0.15 { None } else { Some(g.ty()) }; + fns.push((format!("f{k}"), params, ret)); + } + let mut c = vec!["#include ".to_owned()]; + let mut rs: Vec = [ + "#![feature(no_core)]", "#![no_core]", "#![crate_type = \"lib\"]", + "#![allow(improper_ctypes_definitions, unused, non_snake_case)]", "extern crate minicore;", "use minicore::*;", + ] + .iter() + .map(|s| s.to_string()) + .collect(); + for a in &g.aggregates { + let kw = if a.union { "union" } else { "struct" }; + let cf: Vec = a.fields.iter().enumerate().map(|(i, f)| match f { + Field::Scalar(ct, _) => format!("{ct} f{i};"), + Field::Array(ct, _, n) => format!("{ct} f{i}[{n}];"), + Field::Named(n) => format!("{n} f{i};"), + }).collect(); + let rf: Vec = a.fields.iter().enumerate().map(|(i, f)| match f { + Field::Scalar(_, rt) => format!("pub f{i}: {rt},"), + Field::Array(_, rt, n) => format!("pub f{i}: [{rt}; {n}],"), + Field::Named(n) => format!("pub f{i}: {n},"), + }).collect(); + let cattr = format!("{}{}", if a.packed { " __attribute__((packed))" } else { "" }, a.align.map(|x| format!(" __attribute__((aligned({x})))")).unwrap_or_default()); + c.push(format!("typedef {kw} {} {{ {} }}{cattr} {};", a.name, cf.join(" "), a.name)); + let repr = format!("C{}{}", if a.packed { ", packed" } else { "" }, a.align.map(|x| format!(", align({x})")).unwrap_or_default()); + rs.push(format!("#[repr({repr})] pub {kw} {} {{ {} }}", a.name, rf.join(" "))); + // Union fields must be Copy; minicore has no derive. + rs.push(format!("impl Copy for {} {{}}", a.name)); + } + let mut names = Vec::new(); + for (name, params, ret) in &fns { + let cp = if params.is_empty() { "void".into() } else { params.iter().enumerate().map(|(i, t)| format!("{} a{i}", t.0)).collect::>().join(", ") }; + c.push(format!("{} {name}({cp}) {{ for (;;); }}", ret.as_ref().map_or("void", |r| r.0.as_str()))); + let rp = params.iter().enumerate().map(|(i, t)| format!("a{i}: {}", t.1)).collect::>().join(", "); + let rr = ret.as_ref().map(|r| format!(" -> {}", r.1)).unwrap_or_default(); + rs.push(format!("#[no_mangle] pub extern \"C\" fn {name}({rp}){rr} {{ loop {{}} }}")); + names.push(name.clone()); + } + (c.join("\n") + "\n", rs.join("\n") + "\n", names) +} + +// ---- reading LLVM IR signatures ---- + +#[derive(Clone, Debug, PartialEq)] +struct Param { + ty: String, + attrs: Vec, + noundef: bool, +} + +#[derive(Clone, Debug)] +struct Sig { + cc: Vec, + ret: Vec, + ret_attrs: Vec, + params: Vec, +} + +static DEFINE: LazyLock = LazyLock::new(|| Regex::new(r"^define\s+(.*?)@(\w+)\((.*)\)(.*)\{\s*$").unwrap()); +static DROP: LazyLock = LazyLock::new(|| { + Regex::new(r"\b(noundef|nonnull|noalias|nocapture|readonly|readnone|writeonly|writable|dead_on_unwind|captures\([^)]*\)|dereferenceable(_or_null)?\(\d+\)|initializes\([^)]*\)|range\([^)]*\)|nofpclass\([^)]*\)|immarg|returned|local_unnamed_addr|unnamed_addr|dso_local|dso_preemptable|hidden|protected|internal|private|nounwind|noinline|optnone|!\w+ !\d+|#\d+)\b").unwrap() +}); +static NAMED: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^(%[\w.]+) = type (.*)$").unwrap()); +static CC: LazyLock = LazyLock::new(|| Regex::new(r"\b(\w+cc|cc \d+)\b").unwrap()); +static RET_TY: LazyLock = LazyLock::new(|| Regex::new(r"(\{[^{}]*\}|<\{[^{}]*\}>|\[[^\]]*\]|<[^>]*>|[\w.%*]+)\s*$").unwrap()); +static PARAM_NAME: LazyLock = LazyLock::new(|| Regex::new(r"\s+%[\w.]+$").unwrap()); +static PARAM_TY: LazyLock = LazyLock::new(|| Regex::new(r"^(\{[^}]*\}|\[[^\]]*\]|<[^>]*>|[\w.%*]+)(.*)$").unwrap()); +static ABI_ATTR: LazyLock = LazyLock::new(|| { + Regex::new(r"(zeroext|signext|inreg|byval\([^)]*\)|sret\([^)]*\)|byref\([^)]*\)|align \d+|inalloca\([^)]*\))").unwrap() +}); +static ONE_ELEM: LazyLock = LazyLock::new(|| Regex::new(r"^\[1 x (.*)\]$").unwrap()); +static ARRAY: LazyLock = LazyLock::new(|| Regex::new(r"^\[(\d+) x (i\d+|float|double)\]$").unwrap()); +static INT: LazyLock = LazyLock::new(|| Regex::new(r"^i(\d+)$").unwrap()); +static INT_ARRAY: LazyLock = LazyLock::new(|| Regex::new(r"^\[\d+ x i\d+\]$").unwrap()); + +fn split_top(s: &str) -> Vec { + let (mut out, mut depth, mut cur) = (Vec::new(), 0i32, String::new()); + for ch in s.chars() { + match ch { + '(' | '{' | '[' | '<' => depth += 1, + ')' | '}' | ']' | '>' => depth -= 1, + _ => {} + } + if ch == ',' && depth == 0 { + out.push(cur.trim().to_owned()); + cur.clear(); + } else { + cur.push(ch); + } + } + if !cur.trim().is_empty() { + out.push(cur.trim().to_owned()); + } + out +} + +fn param(p: &str) -> (String, Vec, bool) { + let p = PARAM_NAME.replace(p.trim(), "").into_owned(); + let noundef = p.contains("noundef"); + let (ty, attrs) = match PARAM_TY.captures(&p) { + Some(c) => (c[1].to_owned(), c[2].to_owned()), + None => (p.clone(), String::new()), + }; + let attrs = DROP.replace_all(&attrs, ""); + let mut kept: Vec = ABI_ATTR + .find_iter(&attrs) + .map(|m| { + let a = m.as_str(); + if ["byval", "sret", "byref", "inalloca"].iter().any(|k| a.starts_with(k)) { a.split('(').next().unwrap().to_owned() } else { a.to_owned() } + }) + .collect(); + // An `align` on a plain pointer is a hint, not ABI; on byval and sret it is ABI. + if !kept.iter().any(|k| k == "byval" || k == "sret" || k == "byref") { + kept.retain(|k| !k.starts_with("align")); + } + kept.sort(); + (ty, kept, noundef) +} + +fn flatten(ty: &str, named: &BTreeMap) -> Vec { + let mut ty = ty.trim().to_owned(); + if let Some(body) = named.get(&ty) { + ty = body.clone(); + } + if ty.starts_with("<{") && ty.ends_with("}>") { + ty = ty[1..ty.len() - 1].to_owned(); + } + if ty.starts_with('{') && ty.ends_with('}') { + return split_top(&ty[1..ty.len() - 1]).iter().flat_map(|p| flatten(p, named)).collect(); + } + if let Some(c) = ONE_ELEM.captures(&ty) { + return flatten(&c[1], named); + } + vec![ty] +} + +fn signatures(ll: &str) -> BTreeMap { + let named: BTreeMap = NAMED.captures_iter(ll).map(|c| (c[1].to_owned(), c[2].trim().to_owned())).collect(); + let mut out = BTreeMap::new(); + for line in ll.lines().filter(|l| l.starts_with("define")) { + let Some(c) = DEFINE.captures(line) else { continue }; + let cc: Vec = CC.find_iter(&c[1]).map(|m| m.as_str().to_owned()).collect(); + let head = DROP.replace_all(&c[1], "").trim().to_owned(); + let (ret_ty, prefix) = match RET_TY.find(&head) { + Some(m) => (m.as_str().trim().to_owned(), head[..m.start()].to_owned()), + None => ("void".to_owned(), String::new()), + }; + let mut ret_attrs: Vec = prefix.split_whitespace().filter(|a| ["zeroext", "signext", "inreg"].contains(a)).map(str::to_owned).collect(); + ret_attrs.sort(); + let params = split_top(&c[3]) + .iter() + .flat_map(|p| { + let (ty, attrs, noundef) = param(p); + flatten(&ty, &named).into_iter().map(move |t| Param { ty: t, attrs: attrs.clone(), noundef }) + }) + .collect(); + out.insert(c[2].to_owned(), Sig { cc, ret: flatten(&ret_ty, &named), ret_attrs, params }); + } + out +} + +// ---- comparing ---- + +fn klass(ty: &str) -> String { + match ty { + "float" | "double" | "half" | "bfloat" | "fp128" | "x86_fp80" | "ppc_fp128" => "fp".into(), + "void" => "void".into(), + _ if ty.starts_with('<') => "vec".into(), + _ if ty.starts_with('[') => { + let inner = ty.trim_start_matches('[').trim_end_matches(']'); + match inner.split_once(" x ") { + Some((n, e)) => format!("[{n} x {}]", klass(e)), + None => ty.into(), + } + } + _ => "int".into(), + } +} + +/// Register classes, one per register-sized unit. +fn units(types: &[String], arch: &str, ret: bool, width: usize) -> Vec { + let mut out = Vec::new(); + for t in types { + if let Some(c) = INT.captures(t) + && c[1].parse::().unwrap_or(0) > width + { + out.extend(std::iter::repeat_n("int".to_owned(), c[1].parse::().unwrap() / width)); + } else if let Some(c) = ARRAY.captures(t) + && (ret || (arch == "arm" && (&c[2] == "float" || &c[2] == "double"))) + { + // A float array is a homogeneous aggregate: in a return, or an ARM VFP argument, it + // takes consecutive registers like a struct of its elements. + let elems: Vec = std::iter::repeat_n(c[2].to_owned(), c[1].parse().unwrap_or(0)).collect(); + out.extend(units(&elems, arch, ret, width)); + } else if t == "agg" { + out.push("agg".into()); + } else { + let mut k = klass(t); + // x86's SSE registers hold floats, doubles and vectors alike. + if (arch == "x86_64" || arch == "x86") && (k == "fp" || k == "vec") { + k = "sse".into(); + } + out.push(k); + } + } + out +} + +#[derive(Debug, Clone, Serialize)] +#[serde(tag = "kind", rename_all = "kebab-case")] +enum Diff { + CallingConvention { clang: Vec, rustc: Vec }, + ReturnAttributes { clang: Vec, rustc: Vec, ret: Vec }, + Return { clang: Vec, rustc: Vec }, + Parameters { clang: Vec, rustc: Vec }, + ParameterAttributes { index: usize, clang: Vec, rustc: Vec, ty: String }, +} + +impl Diff { + fn label(&self) -> &'static str { + match self { + Diff::CallingConvention { .. } => "calling convention", + Diff::ReturnAttributes { .. } => "return attributes", + Diff::Return { .. } => "return", + Diff::Parameters { .. } => "parameters", + Diff::ParameterAttributes { .. } => "parameter attributes", + } + } +} + +fn compare(c: &Sig, r: &Sig, arch: &str, slot: usize) -> (Vec, Vec) { + let (mut findings, mut notes) = (Vec::new(), Vec::new()); + let width = slot * 8; + if c.cc != r.cc { + findings.push(Diff::CallingConvention { clang: c.cc.clone(), rustc: r.cc.clone() }); + } + if c.ret_attrs != r.ret_attrs { + // x86's psABI leaves the bits above a small integer return undefined (#142389); only + // bool's bits 1-7 must be zero (#163911). + if (arch == "x86_64" || arch == "x86") && c.ret != ["i1"] { + notes.push(format!("return attributes {:?} vs {:?} (x86 psABI: upper bits undefined)", c.ret_attrs, r.ret_attrs)); + } else { + findings.push(Diff::ReturnAttributes { clang: c.ret_attrs.clone(), rustc: r.ret_attrs.clone(), ret: c.ret.clone() }); + } + } + if units(&c.ret, arch, true, width) != units(&r.ret, arch, true, width) { + findings.push(Diff::Return { clang: c.ret.clone(), rustc: r.ret.clone() }); + } else if c.ret != r.ret { + notes.push(format!("return types {:?} vs {:?}", c.ret, r.ret)); + } + // ARM and 64-bit PowerPC split a byval aggregate between registers and stack as they do an + // array argument of the same size: both are "an aggregate". + let agg = |p: &Param| -> Param { + if (arch == "arm" || arch == "powerpc64") && (p.attrs.iter().any(|a| a == "byval") || INT_ARRAY.is_match(&p.ty)) { + Param { ty: "agg".into(), attrs: vec![], noundef: false } + } else { + p.clone() + } + }; + let cp: Vec = c.params.iter().map(agg).collect(); + let rp: Vec = r.params.iter().map(agg).collect(); + let ct: Vec = cp.iter().map(|p| p.ty.clone()).collect(); + let rt: Vec = rp.iter().map(|p| p.ty.clone()).collect(); + if units(&ct, arch, false, width) != units(&rt, arch, false, width) { + // i386 passes every argument on the stack: an expanded struct and a byval copy of it are + // the same bytes. + if arch == "x86" && cp.iter().chain(&rp).any(|p| p.attrs.iter().any(|a| a == "byval")) { + notes.push(format!("parameters {ct:?} vs {rt:?} (x86: same stack bytes)")); + } else { + findings.push(Diff::Parameters { clang: ct, rustc: rt }); + } + } else if ct != rt { + notes.push(format!("parameter types {ct:?} vs {rt:?}")); + } else { + for (i, (a, b)) in cp.iter().zip(&rp).enumerate() { + if a.attrs != b.attrs { + let aligns: Vec = a.attrs.iter().chain(&b.attrs).filter_map(|x| x.strip_prefix("align ")).filter_map(|x| x.parse().ok()).collect(); + let rest_c: Vec<&String> = a.attrs.iter().filter(|x| !x.starts_with("align ")).collect(); + let rest_r: Vec<&String> = b.attrs.iter().filter(|x| !x.starts_with("align ")).collect(); + let only_byval = rest_c.iter().chain(&rest_r).all(|x| x.as_str() == "byval"); + let ext_only_rust = rest_c.is_empty() && rest_r.iter().all(|x| x.as_str() == "zeroext" || x.as_str() == "signext"); + if rest_c == rest_r && !aligns.is_empty() && aligns.iter().max().copied().unwrap_or(0) as usize <= slot { + notes.push(format!("parameter {i} alignment within a stack slot")); + } else if (arch == "wasm32" || arch == "wasm64") && only_byval { + notes.push(format!("parameter {i}: wasm byval is a pointer to a copy")); + } else if arch == "x86" && only_byval { + notes.push(format!("parameter {i}: x86 same stack bytes")); + } else if arch == "x86_64" && ext_only_rust { + // Win64: rustc extends, clang does not; the callee re-extends either way. + notes.push(format!("parameter {i}: win64 extension not relied on")); + } else { + findings.push(Diff::ParameterAttributes { index: i, clang: a.attrs.clone(), rustc: b.attrs.clone(), ty: a.ty.clone() }); + } + } + if a.noundef != b.noundef { + notes.push(format!("parameter {i} noundef")); + } + } + } + (findings, notes) +} + +/// A difference that is a bug already recorded, or one this host cannot decide: its label. +fn label(arch: &str, target: &str, d: &Diff) -> Option<&'static str> { + let fp = |v: &[String]| v.iter().filter(|t| *t == "float" || *t == "double").count(); + match d { + Diff::ReturnAttributes { clang, .. } if arch == "x86_64" && clang == &["zeroext"] => Some("rust-lang/rust#163911"), + Diff::ParameterAttributes { clang, rustc, .. } + if matches!(arch, "riscv64" | "riscv32" | "loongarch64") && rustc.is_empty() && (clang == &["signext"] || clang == &["zeroext"]) => + { + Some("finding 19 (docs/hunt.md)") + } + Diff::Parameters { clang, rustc } | Diff::Return { clang, rustc } if matches!(arch, "riscv64" | "riscv32" | "loongarch64") && fp(rustc) > fp(clang) => { + Some("finding 20 (docs/hunt.md)") + } + // clang returns an 8-byte struct with a 3-byte array field indirectly; rustc and MSVC's + // documentation in edx:eax. Needs MSVC. + Diff::Return { rustc, .. } if arch == "x86" && target.ends_with("windows-msvc") && rustc != &["void"] => { + Some("i686 msvc small-struct return (needs MSVC)") + } + Diff::Parameters { clang, rustc } if arch == "x86" && target.ends_with("windows-msvc") && clang.first().map(String::as_str) == Some("ptr") && clang[1..] == rustc[..] => { + Some("i686 msvc small-struct return (needs MSVC)") + } + // clang marks a float from a single-member aggregate inreg; needs a run on the target. + Diff::ParameterAttributes { clang, rustc, ty, .. } if arch == "powerpc64" && clang == &["inreg"] && rustc.is_empty() && (ty == "float" || ty == "double") => { + Some("ppc64 inreg float (needs a run)") + } + _ => None, + } +} + +// ---- running ---- + +#[derive(Deserialize, Default)] +#[serde(rename_all = "kebab-case")] +struct Spec { + #[serde(default)] + llvm_target: String, + #[serde(default)] + arch: String, + #[serde(default)] + target_pointer_width: serde_json::Value, + #[serde(default)] + c_int_width: serde_json::Value, + #[serde(default)] + cpu: String, + #[serde(default)] + features: String, + #[serde(default)] + llvm_abiname: String, + #[serde(default)] + llvm_floatabi: String, +} + +fn num(v: &serde_json::Value, default: usize) -> usize { + v.as_u64().map(|x| x as usize).or_else(|| v.as_str().and_then(|s| s.parse().ok())).unwrap_or(default) +} + +#[derive(Serialize, Default)] +struct TargetResult { + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + compared: usize, + findings: BTreeMap>, + labelled: BTreeMap, + notes: usize, +} + +fn run_cmd(cmd: &mut Command) -> Result<(), String> { + match cmd.env("RUSTC_BOOTSTRAP", "1").output() { + Ok(o) if o.status.success() => Ok(()), + Ok(o) => Err(String::from_utf8_lossy(&o.stderr).chars().rev().take(1000).collect::().chars().rev().collect()), + Err(e) => Err(e.to_string()), + } +} + +fn one_target(args: &Args, target: &str) -> TargetResult { + let skip = |why: String| TargetResult { skip: Some(why), ..Default::default() }; + let spec_out = Command::new(&args.rustc).args(["--print", "target-spec-json", "-Zunstable-options", "--target", target]).env("RUSTC_BOOTSTRAP", "1").output(); + let Ok(spec_out) = spec_out else { return skip("no spec".into()) }; + let Ok(spec) = serde_json::from_slice::(&spec_out.stdout) else { return skip("no spec".into()) }; + let width = num(&spec.target_pointer_width, 64); + if width < 32 || num(&spec.c_int_width, 32) != 32 { + return skip("16-bit int".into()); + } + let d = args.work.join(target); + let _ = std::fs::create_dir_all(&d); + let (csrc, rsrc, fns) = program(width == 64 && spec.arch != "sparc64", args.seed, args.count); + let _ = std::fs::write(d.join("a.c"), &csrc); + let _ = std::fs::write(d.join("a.rs"), &rsrc); + let base = |c: &mut Command| { + c.args(["--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", "--out-dir"]).arg(&d); + }; + let mut mc = Command::new(&args.rustc); + base(&mut mc); + mc.args(["--crate-type", "rlib", "--crate-name", "minicore", "-Awarnings"]).arg(args.rust.join("tests/auxiliary/minicore.rs")); + if let Err(e) = run_cmd(&mut mc) { + return skip(format!("minicore does not build: {e}")); + } + let mut rc = Command::new(&args.rustc); + base(&mut rc); + rc.args(["--emit=llvm-ir", "-Copt-level=0", "--extern"]).arg(format!("minicore={}", d.join("libminicore.rlib").display())).arg("-o").arg(d.join("r.ll")).arg(d.join("a.rs")); + if let Err(e) = run_cmd(&mut rc) { + return skip(format!("rust side does not build: {e}")); + } + let mut cc = Command::new(&args.clang); + cc.arg(format!("--target={}", spec.llvm_target)).args(["-ffreestanding", "-S", "-emit-llvm", "-O0", "-Wno-everything"]); + // The target's CPU and features decide parts of the ABI in clang too (soft-float, SSE). + if !spec.cpu.is_empty() && spec.cpu != "generic" { + cc.args(["-Xclang", "-target-cpu", "-Xclang", &spec.cpu]); + } + for f in spec.features.split(',').filter(|f| !f.is_empty()) { + cc.args(["-Xclang", "-target-feature", "-Xclang", f]); + } + if !spec.llvm_abiname.is_empty() { + cc.arg(format!("-mabi={}", spec.llvm_abiname)); + } + if spec.llvm_floatabi == "hard" { + cc.arg("-mfloat-abi=hard"); + } + cc.arg("-o").arg(d.join("c.ll")).arg(d.join("a.c")); + if let Err(e) = run_cmd(&mut cc) { + return skip(format!("clang does not build: {e}")); + } + let cs = signatures(&std::fs::read_to_string(d.join("c.ll")).unwrap_or_default()); + let rsig = signatures(&std::fs::read_to_string(d.join("r.ll")).unwrap_or_default()); + let mut res = TargetResult { compared: fns.len(), ..Default::default() }; + for name in &fns { + let (Some(c), Some(r)) = (cs.get(name), rsig.get(name)) else { continue }; + let (f, n) = compare(c, r, &spec.arch, width / 8); + res.notes += n.len(); + let mut unlabelled = Vec::new(); + for diff in f { + match label(&spec.arch, target, &diff) { + Some(l) => *res.labelled.entry(l.to_owned()).or_default() += 1, + None => unlabelled.push(diff), + } + } + if !unlabelled.is_empty() { + res.findings.insert(name.clone(), unlabelled); + } + } + res +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let targets: Vec = if let Some(t) = &args.targets { + t.split(',').map(str::to_owned).collect() + } else if args.all { + let out = Command::new(&args.rustc).args(["--print", "target-list"]).output()?; + String::from_utf8_lossy(&out.stdout).split_whitespace().map(str::to_owned).collect() + } else { + MAIN.iter().map(|s| s.to_string()).collect() + }; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: BTreeMap = pool.install(|| targets.par_iter().map(|t| (t.clone(), one_target(&args, t))).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut labelled: BTreeMap<&str, usize> = BTreeMap::new(); + let mut kinds: BTreeMap<&str, BTreeMap<&str, usize>> = BTreeMap::new(); + let (mut compared, mut skipped) = (0, BTreeMap::::new()); + for (t, r) in &results { + if let Some(s) = &r.skip { + *skipped.entry(s.split(':').next().unwrap_or("").to_owned()).or_default() += 1; + continue; + } + compared += 1; + for (l, n) in &r.labelled { + *labelled.entry(l.as_str()).or_default() += n; + } + for diffs in r.findings.values() { + for d in diffs { + *kinds.entry(d.label()).or_default().entry(t.as_str()).or_default() += 1; + } + } + } + println!("{compared} targets compared; skipped: {skipped:?}"); + if !labelled.is_empty() { + println!("known: {labelled:?}"); + } + for (kind, per) in &kinds { + let total: usize = per.values().sum(); + let mut top: Vec<(&&str, &usize)> = per.iter().collect(); + top.sort_by(|a, b| b.1.cmp(a.1)); + println!("{kind}: {total} in {} targets: {}", per.len(), top.iter().take(8).map(|(t, n)| format!("{t} {n}")).collect::>().join(", ")); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/audit_options.rs b/crates/mirth-lab/src/tools/audit_options.rs new file mode 100644 index 0000000..4973b4e --- /dev/null +++ b/crates/mirth-lab/src/tools/audit_options.rs @@ -0,0 +1,160 @@ +//! Audit rustc's [UNTRACKED] options for stale incremental reuse. +//! +//! An option rustc marks [UNTRACKED] is left out of the dependency-tracking hash, so changing +//! it between incremental sessions reuses the previous session's results. That is only correct +//! if the option cannot change them. For each untracked option that takes no value or a +//! boolean, this builds a crate incrementally without it, then again with it, and compares the +//! result with a clean build that has it: the .rmeta, each .rlib member (object code, with the +//! incremental session suffix removed from names), the diagnostics, and the files written. A +//! difference means the option changes output that incremental compilation reuses: it should +//! be tracked, or the reuse checked. +//! +//! With ARGs, audits those arguments instead of the options found in the checkout's +//! compiler/rustc_session/src/options.rs. The crate should have code of its own for codegen +//! options to change (non-generic functions) and a warning or two for diagnostic options to +//! change. Exits 1 when any option differs. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use regex::Regex; + +use mirth_lab::artifacts::normalized_rlib; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rust checkout, to read the untracked options from. + #[arg(long)] + source: Option, + /// The crate root to build, as a library. + #[arg(long = "crate")] + krate: PathBuf, + #[arg(long, default_value = "audited")] + crate_name: String, + #[arg(long, default_value = "2021")] + edition: String, + #[arg(trailing_var_arg = true, allow_hyphen_values = true)] + args: Vec, +} + +// Options that stop compilation or change how arguments are read. +const SKIP: &[&str] = &["-Chelp", "-Zhelp", "-Zno-analysis", "-Zparse-crate-root-only=yes", "-Zshell-argfiles=yes"]; + +static UNTRACKED: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s{4}(\w+):\s*([^=\n]+?)\s*=\s*\(([^,]*),\s*(parse_\w+),\s*\[UNTRACKED\]").unwrap()); +static CODEGEN: LazyLock = LazyLock::new(|| Regex::new(r"options! \{\s*CodegenOptions,").unwrap()); +static UNSTABLE: LazyLock = LazyLock::new(|| Regex::new(r"options! \{\s*UnstableOptions,").unwrap()); + +fn untracked_boolean_options(source: &Path) -> anyhow::Result> { + let text = std::fs::read_to_string(source.join("compiler/rustc_session/src/options.rs"))?; + let codegen = CODEGEN.find(&text).map_or(0, |m| m.start()); + let unstable = UNSTABLE.find(&text).map_or(0, |m| m.start()); + let mut found = Vec::new(); + for c in UNTRACKED.captures_iter(&text) { + let start = c.get(0).unwrap().start(); + let (name, default, parser) = (&c[1], c[3].trim(), &c[4]); + if !["parse_bool", "parse_no_value", "parse_opt_bool"].contains(&parser) || start < codegen { + continue; + } + let group = if start > unstable { "Z" } else { "C" }; + let mut flag = format!("-{group}{}", name.replace('_', "-")); + if parser != "parse_no_value" { + flag += if default == "true" || default == "Some(true)" { "=no" } else { "=yes" }; + } + if !SKIP.contains(&flag.as_str()) { + found.push(flag); + } + } + Ok(found) +} + +/// One build, always from the same working directory, which rustc records. +fn build(args: &Args, krate: &Path, work: &Path, incremental: &str, out: &str, extra: &[&str]) -> (i32, Vec) { + let _ = std::fs::create_dir_all(work.join(out)); + let r = Command::new(&args.rustc) + .args(["--edition", &args.edition, "--crate-type", "lib", "--crate-name", &args.crate_name, "--emit=metadata,link"]) + .arg(format!("-Cincremental={}", work.join(incremental).display())) + .arg("--out-dir") + .arg(work.join(out)) + .arg(krate) + .args(extra) + .current_dir(work) + .output(); + match r { + Ok(o) => { + let err = String::from_utf8_lossy(&o.stderr); + let mut diagnostics: Vec = + err.lines().filter(|l| l.starts_with("warning") || l.starts_with("error") || l.contains("-->")).map(str::to_owned).collect(); + diagnostics.sort(); + (o.status.code().unwrap_or(-1), diagnostics) + } + Err(_) => (-1, Vec::new()), + } +} + +fn listing(dir: &Path) -> BTreeSet { + std::fs::read_dir(dir).map(|r| r.flatten().map(|e| e.file_name().to_string_lossy().into_owned()).collect()).unwrap_or_default() +} + +fn audit(args: &Args, krate: &Path, flag: &str) -> Vec { + let work = tempfile::Builder::new().prefix("audit-").tempdir().expect("scratch"); + let w = work.path(); + let extra: Vec<&str> = if flag.is_empty() { vec![] } else { vec![flag] }; + let (rc0, _) = build(args, krate, w, "i", "o1", &[]); + let (rc1, diag_inc) = build(args, krate, w, "i", "o1", &extra); + let (rc2, diag_clean) = build(args, krate, w, "j", "o2", &extra); + if rc0 != 0 || rc2 != 0 { + return vec![format!("the crate does not build (exit {rc0} without the option, {rc2} with it)")]; + } + let name = &args.crate_name; + let mut problems = Vec::new(); + if rc1 != rc2 { + problems.push(format!("the incremental rebuild exits {rc1}, a clean build {rc2}")); + } + if std::fs::read(w.join(format!("o1/lib{name}.rmeta"))).ok() != std::fs::read(w.join(format!("o2/lib{name}.rmeta"))).ok() { + problems.push("metadata".into()); + } + let a = normalized_rlib(&w.join(format!("o1/lib{name}.rlib"))); + let b = normalized_rlib(&w.join(format!("o2/lib{name}.rlib"))); + let members: BTreeSet<&String> = a.keys().chain(b.keys()).filter(|m| a.get(*m) != b.get(*m)).collect(); + if !members.is_empty() { + problems.push(format!("object code ({} rlib members differ or exist on one side)", members.len())); + } + if diag_inc != diag_clean { + problems.push(format!("diagnostics ({} lines incrementally, {} clean)", diag_inc.len(), diag_clean.len())); + } + let o1 = listing(&w.join("o1")); + let missing = listing(&w.join("o2")).difference(&o1).count(); + if missing > 0 { + problems.push(format!("{missing} files only a clean build writes")); + } + problems +} + +pub fn run(args: Args) -> anyhow::Result { + let krate = std::fs::canonicalize(&args.krate)?; + let flags = if !args.args.is_empty() { + args.args.clone() + } else { + let source = args.source.as_ref().ok_or_else(|| anyhow::anyhow!("--source or ARGs needed"))?; + untracked_boolean_options(source)? + }; + let show = |p: &[String]| if p.is_empty() { "same".to_owned() } else { p.join("; ") }; + let control = audit(&args, &krate, ""); + println!("{:40} {}", "(control: no option)", show(&control)); + if !control.is_empty() { + eprintln!("the control differs: incremental and clean builds disagree without any option"); + return Ok(ExitCode::from(1)); + } + let mut any = false; + for flag in &flags { + let r = audit(&args, &krate, flag); + println!("{flag:40} {}", show(&r)); + any |= !r.is_empty(); + } + Ok(if any { ExitCode::from(1) } else { ExitCode::SUCCESS }) +} diff --git a/crates/mirth-lab/src/tools/callgraph.rs b/crates/mirth-lab/src/tools/callgraph.rs new file mode 100644 index 0000000..e93724a --- /dev/null +++ b/crates/mirth-lab/src/tools/callgraph.rs @@ -0,0 +1,611 @@ +//! Which of the compiler's functions can run at all: reachability over the call graph a compiler +//! built with rustc/callgraph.toml writes (`.graph`), against the functions (and blocks) +//! a compiler built with rustc/coverage.toml instruments. The functions that cannot be reached +//! leave coverage's denominator. +//! +//! The graph over-approximates what can run, so what it leaves out cannot run (as far as the +//! edges it knows go): direct calls, functions and closures used as values, callees MIR inlining +//! merged in, trait calls resolved in the caller; a call to a trait item reaches every body +//! implementing it, gated by rapid type analysis (a method of an impl for one of the compiler's +//! types only once reachable code builds the type, a trait impl's function only once reachable +//! code demands the trait for the impl's type). Roots: the compiler's and rustdoc's `main`s, +//! foreign-ABI functions, constant and static initializers, impls of traits from outside the +//! compiler, and what programs linking the compiler ran (`--external`). +//! +//! With hits, reports coverage of the reachable functions (and blocks), and checks the analysis: +//! a function that ran must be reachable. + +use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; +use std::fmt::Write as _; +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use regex::Regex; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// The call-graph build's site directory (`.graph` files). + #[arg(long)] + graph: PathBuf, + /// The coverage build's site directory (`.sites` files). + #[arg(long)] + sites: PathBuf, + /// union.txt files of runs of the compiler. + #[arg(long)] + hit: Vec, + /// Directories of raw MIRTH_OUT logs. + #[arg(long)] + logs: Vec, + /// union.txt files of programs outside the compiler that link it: what they ran is a root. + #[arg(long)] + external: Vec, + #[arg(long)] + json: Option, + /// List the unreachable functions of these crates. + #[arg(long)] + unreachable: Vec, + /// Print how the graph reaches these functions. + #[arg(long)] + why: Vec, + /// Write the reachable functions that never ran, by crate and file. + #[arg(long)] + gaps: Option, + /// Write the blocks that never ran in functions that did. + #[arg(long)] + block_gaps: Option, +} + +const ROOTS: &[&str] = &["rustc_main::main", "rustc_driver_impl::main", "rustdoc::main"]; +/// Crates that do not run when the compiler does: proc macros, and a build-script helper. +const NOT_AT_RUN_TIME: &[&str] = + &["rustc_macros", "rustc_type_ir_macros", "rustc_index_macros", "rustc_windows_rc", "rustc_hir_macros", "rustc_fluent_macro"]; +const SCOPE: &[&str] = &["rustc_", "rustdoc"]; +/// Called by the language on any value, not through a bound: drop glue. +const UNGATED_TRAITS: &[&str] = &["core::ops::drop::Drop"]; + +static PARENT: LazyLock = LazyLock::new(|| Regex::new(r"^(.*)::\{[^}]*\}$").unwrap()); +static CRATE_ID: LazyLock = LazyLock::new(|| Regex::new(r"-[0-9a-f]{16}$").unwrap()); + +type Node = String; +type Cond = (u8, String, String); // (0: live type, 1: demand (trait, type)), as strings + +#[derive(Default)] +struct Graph { + path_of: HashMap, + hash_of: HashMap, + implements: HashMap, + external_impl: HashSet, + const_bodies: HashSet, + self_type: HashMap, + constructs: HashMap>, + spec_bounds: HashSet, + ending: HashMap, + diverges_into: HashMap>, + impl_key: HashMap, + demands: HashMap>, + edges: HashMap>, +} + +fn load_graph(dir: &PathBuf) -> Graph { + let mut g = Graph::default(); + let Ok(entries) = std::fs::read_dir(dir) else { return g }; + let mut files: Vec = entries.flatten().map(|e| e.path()).filter(|p| p.extension().is_some_and(|x| x == "graph")).collect(); + files.sort(); + for f in files { + let text = String::from_utf8_lossy(&std::fs::read(&f).unwrap_or_default()).into_owned(); + for line in text.lines() { + let p: Vec<&str> = line.split('\t').collect(); + match p[0] { + "body" if p.len() > 5 => { + let node = p[1].to_owned(); + if p.len() > 7 && SCOPE.iter().any(|s| p[7].starts_with(s)) { + g.self_type.insert(node.clone(), p[6].to_owned()); + } + // A trait with specializing impls: which impl a call reaches is decided by + // more than the bounds say, so such impls are not gated on them. + let specialized = p.len() > 11 && p[11] == "specialized"; + if p.len() > 10 && p[8] != "-" && p[10] != "-" && !UNGATED_TRAITS.contains(&p[9]) && !specialized { + g.impl_key.insert(node.clone(), (p[8].to_owned(), p[10].to_owned())); + } + g.path_of.insert(node.clone(), p[2].to_owned()); + g.hash_of.insert(p[2].to_owned(), node.clone()); + if p.len() > 12 && (p[12] == "ice-only" || p[12] == "diverges") { + g.ending.insert(node.clone(), p[12].to_owned()); + } + if p[5] == "const" || p[5] == "extern" { + g.const_bodies.insert(node.clone()); + } + if p[3] != "-" { + g.implements.insert(node.clone(), p[3].to_owned()); + if !p[4].starts_with("rustc_") { + g.external_impl.insert(node); + } + } + } + "diverges" if p.len() > 2 => { + g.diverges_into.entry(p[1].to_owned()).or_default().insert(p[2].to_owned()); + } + "specbound" if p.len() > 1 => { + g.spec_bounds.insert(p[1].to_owned()); + } + "demand" if p.len() > 3 => { + g.demands.entry(p[1].to_owned()).or_default().insert((p[2].to_owned(), p[3].to_owned())); + } + "edge" if p.len() > 3 => { + let map = if p[3] == "construct" { &mut g.constructs } else { &mut g.edges }; + map.entry(p[1].to_owned()).or_default().insert(p[2].to_owned()); + } + _ => {} + } + } + } + g +} + +fn parent(path: &str) -> Option { + PARENT.captures(path).map(|c| c[1].to_owned()) +} + +struct Site { + krate: String, + path: String, +} + +struct Block { + krate: String, + path: String, + name: String, + span: String, + snippet: String, + panics: bool, + logging: bool, +} + +fn load_sites(dir: &PathBuf) -> (BTreeMap, HashMap, BTreeMap) { + let (mut functions, mut span_of, mut blocks) = (BTreeMap::new(), HashMap::new(), BTreeMap::new()); + let Ok(entries) = std::fs::read_dir(dir) else { return (functions, span_of, blocks) }; + for e in entries.flatten() { + if e.path().extension().is_none_or(|x| x != "sites") { + continue; + } + let text = String::from_utf8_lossy(&std::fs::read(e.path()).unwrap_or_default()).into_owned(); + for line in text.lines() { + let f: Vec<&str> = line.split('\t').collect(); + if f.len() < 7 { + continue; + } + // Newer tables name the crate with its stable id (`rustc_hash-<16 hex digits>`). + let krate = CRATE_ID.replace(f[3], "").into_owned(); + match f[1] { + "cover" => { + functions.insert(f[0].to_owned(), Site { krate, path: f[4].to_owned() }); + span_of.insert(f[4].to_owned(), f[6].to_owned()); + } + "block" => { + let mut parts = f[5].split(' '); + let name = parts.next().unwrap_or("").to_owned(); + let tags: Vec<&str> = parts.collect(); + blocks.insert( + f[0].to_owned(), + Block { + krate, + path: f[4].to_owned(), + name, + span: f[6].to_owned(), + snippet: f.get(7).unwrap_or(&"").to_string(), + panics: tags.contains(&"panics"), + logging: tags.contains(&"log"), + }, + ); + } + _ => {} + } + } + } + (functions, span_of, blocks) +} + +fn words(path: &PathBuf) -> HashSet { + std::fs::read_to_string(path).unwrap_or_default().split_whitespace().map(str::to_owned).collect() +} + +pub fn run(args: Args) -> anyhow::Result { + let g = load_graph(&args.graph); + let bodies: HashSet<&Node> = g.path_of.keys().collect(); + // A body runs only on a compiler bug when every path ends in a panic, directly or through + // bodies that do. + let mut ice_nodes: HashSet = g.ending.iter().filter(|(_, e)| *e == "ice-only").map(|(n, _)| n.clone()).collect(); + loop { + let mut changed = false; + for n in g.ending.keys() { + if !ice_nodes.contains(n) + && let Some(into) = g.diverges_into.get(n) + && !into.is_empty() + && into.iter().all(|x| ice_nodes.contains(x)) + { + ice_nodes.insert(n.clone()); + changed = true; + } + } + if !changed { + break; + } + } + let ice_only: HashSet = ice_nodes.iter().filter_map(|n| g.path_of.get(n).cloned()).collect(); + let mut impl_key: HashMap = g.impl_key.iter().filter(|(_, k)| !g.spec_bounds.contains(&k.0)).map(|(n, k)| (n.clone(), k.clone())).collect(); + let mut implementors: HashMap<&Node, Vec<&Node>> = HashMap::new(); + for (body, item) in &g.implements { + implementors.entry(item).or_default().push(body); + } + let mut roots: HashSet = ROOTS.iter().filter_map(|r| g.hash_of.get(*r).cloned()).collect(); + roots.extend(g.external_impl.iter().cloned()); + roots.extend(g.const_bodies.iter().cloned()); + let mut children: HashMap> = HashMap::new(); + for (node, path) in &g.path_of { + let Some(mut up) = parent(path) else { continue }; + // Nested in a body, or in a static's or constant's initializer, which is not one here. + while !g.hash_of.contains_key(&up) { + match parent(&up) { + Some(p) => up = p, + None => break, + } + } + match g.hash_of.get(&up) { + Some(h) => { + children.entry(h.clone()).or_default().insert(node.clone()); + } + None => { + roots.insert(node.clone()); + } + } + } + let (functions, span_of, blocks) = load_sites(&args.sites); + let mut external: HashSet = HashSet::new(); + for u in &args.external { + external.extend(words(u)); + } + // Crates from crates.io that the scope takes in by name: other dependencies can name their + // types and need their impls without any bound in the compiler saying so. + let from_registry: HashSet<&str> = functions + .values() + .filter(|s| !span_of.get(&s.path).is_none_or(|sp| sp.starts_with("compiler/") || sp.starts_with("src/") || sp.starts_with("library/"))) + .map(|s| s.krate.as_str()) + .collect(); + impl_key.retain(|n, _| !g.path_of.get(n).is_some_and(|p| from_registry.contains(p.split("::").next().unwrap_or("")))); + let external_roots: HashSet = external.iter().filter_map(|x| functions.get(x)).filter_map(|s| g.hash_of.get(&s.path).cloned()).collect(); + roots.extend(external_roots.iter().cloned()); + + // Rapid type analysis. + let mut reachable: HashSet = HashSet::new(); + let mut live: HashSet = HashSet::new(); + let mut demanded: HashSet<(Node, Node)> = HashSet::new(); + let mut came_from: HashMap, &'static str)> = HashMap::new(); + let mut condition_from: HashMap = HashMap::new(); + let mut waiting: HashMap> = HashMap::new(); + let mut queue: VecDeque = VecDeque::new(); + let missing = |node: &Node, live: &HashSet, demanded: &HashSet<(Node, Node)>| -> Option { + if let Some(t) = g.self_type.get(node) + && !live.contains(t) + { + return Some((0, t.clone(), String::new())); + } + if let Some(k) = impl_key.get(node) + && !demanded.contains(k) + { + return Some((1, k.0.clone(), k.1.clone())); + } + None + }; + macro_rules! offer { + ($node:expr, $gated:expr, $source:expr, $how:expr) => {{ + let node: Node = $node; + if !reachable.contains(&node) { + came_from.entry(node.clone()).or_insert(($source, $how)); + match if $gated { missing(&node, &live, &demanded) } else { None } { + Some(c) => { + waiting.entry(c).or_default().insert(node); + } + None => queue.push_back(node), + } + } + }}; + } + let mut sorted_roots: Vec<&Node> = roots.iter().collect(); + sorted_roots.sort(); + for r in sorted_roots { + let gated = g.external_impl.contains(r) && !g.const_bodies.contains(r) && !external_roots.contains(r); + offer!(r.clone(), gated, None, "root"); + } + while let Some(node) = queue.pop_front() { + if !reachable.insert(node.clone()) { + continue; + } + let mut satisfied: Vec = Vec::new(); + if let Some(ts) = g.constructs.get(&node) { + for t in ts { + if live.insert(t.clone()) { + let c = (0, t.clone(), String::new()); + condition_from.insert(c.clone(), node.clone()); + satisfied.push(c); + } + } + } + if let Some(ks) = g.demands.get(&node) { + for k in ks { + if demanded.insert(k.clone()) { + let c = (1, k.0.clone(), k.1.clone()); + condition_from.insert(c.clone(), node.clone()); + satisfied.push(c); + } + } + } + for c in satisfied { + if let Some(nodes) = waiting.remove(&c) { + for n in nodes { + offer!(n, true, None, "root"); + } + } + } + let nexts: HashSet<&Node> = g.edges.get(&node).into_iter().flatten().chain(children.get(&node).into_iter().flatten()).collect(); + for nxt in nexts { + offer!(nxt.clone(), false, Some(node.clone()), "edge"); + // A call to a trait item reaches the bodies implementing it. + for imp in implementors.get(nxt).into_iter().flatten() { + offer!((*imp).clone(), true, Some(node.clone()), "dispatch"); + } + } + } + let reachable_paths: HashSet<&String> = reachable.iter().filter_map(|n| g.path_of.get(n)).collect(); + let mut out = String::new(); + for target in &args.why { + let mut node = g.hash_of.get(target).cloned(); + let ok = node.as_ref().is_some_and(|n| reachable.contains(n)); + let _ = writeln!(out, "\nwhy {target}:{}", if ok { "" } else { " not reachable" }); + let mut seen = HashSet::new(); + while let Some(n) = node.clone() { + if !reachable.contains(&n) || !seen.insert(n.clone()) { + break; + } + let (source, how) = came_from.get(&n).cloned().unwrap_or((None, "?")); + let mut extra = String::new(); + let conds = [g.self_type.get(&n).map(|t| (0u8, t.clone(), String::new())), impl_key.get(&n).map(|k| (1u8, k.0.clone(), k.1.clone()))]; + for c in conds.into_iter().flatten() { + if let Some(from) = condition_from.get(&c) { + let _ = write!(extra, " [{} from {}]", if c.0 == 0 { "live" } else { "demand" }, g.path_of.get(from).unwrap_or(from)); + } + } + let _ = writeln!(out, " {} <- {how}{extra}", g.path_of.get(&n).unwrap_or(&n)); + node = source; + } + } + let functions: BTreeMap<&String, &Site> = functions.iter().filter(|(_, s)| !NOT_AT_RUN_TIME.contains(&s.krate.as_str())).collect(); + let paths: HashSet<&String> = functions.values().map(|s| &s.path).collect(); + let known: HashSet<&String> = paths.iter().filter(|p| g.hash_of.contains_key(**p)).copied().collect(); + let unreach: HashSet<&String> = known.iter().filter(|p| !reachable_paths.contains(**p)).copied().collect(); + let _ = writeln!(out, "{} bodies in the graph, {} roots, {} reachable", bodies.len(), roots.len(), reachable.iter().filter(|n| bodies.contains(n)).count()); + let _ = writeln!( + out, + "{} instrumented functions; {} in the graph; {} unreachable ({:.1}%)", + functions.len(), + known.len(), + unreach.len(), + 100.0 * unreach.len() as f64 / known.len().max(1) as f64 + ); + let mut hit: HashSet = external.clone(); + for u in &args.hit { + hit.extend(words(u)); + } + for d in &args.logs { + for e in WalkDir::new(d).into_iter().filter_map(Result::ok) { + if e.path().extension().is_some_and(|x| x == "log") { + for line in String::from_utf8_lossy(&std::fs::read(e.path()).unwrap_or_default()).lines() { + if let Some(s) = line.strip_prefix("V\t") { + hit.insert(s.to_owned()); + } + } + } + } + } + let hit_paths: HashSet<&String> = hit.iter().filter_map(|s| functions.get(s)).map(|s| &s.path).collect(); + let mut by_crate: BTreeMap<&str, (usize, usize)> = BTreeMap::new(); + for s in functions.values() { + let row = by_crate.entry(s.krate.as_str()).or_default(); + if unreach.contains(&s.path) { + continue; + } + row.0 += 1; + row.1 += hit_paths.contains(&s.path) as usize; + } + if !hit.is_empty() { + let total: usize = by_crate.values().map(|r| r.0).sum(); + let ran: usize = by_crate.values().map(|r| r.1).sum(); + let _ = writeln!(out, "coverage: {} functions ran; of the {total} reachable ones, {ran} ({:.1}%)", hit_paths.len(), 100.0 * ran as f64 / total.max(1) as f64); + let mut wrong: Vec<&&String> = hit_paths.iter().filter(|p| unreach.contains(**p)).collect(); + wrong.sort(); + let ice: HashSet<&String> = functions.values().map(|s| &s.path).filter(|p| ice_only.contains(*p) && !unreach.contains(p)).collect(); + let ice_ran = ice.iter().filter(|p| hit_paths.contains(**p)).count(); + let _ = writeln!( + out, + "of the reachable ones, {} only panic (they run on a compiler bug; {ice_ran} ran): without them, {} of {} ({:.1}%)", + ice.len(), + ran - ice_ran, + total - ice.len(), + 100.0 * (ran - ice_ran) as f64 / (total - ice.len()).max(1) as f64 + ); + let _ = writeln!(out, "ran although unreachable (edges the analysis misses): {}", wrong.len()); + for w in wrong.iter().take(30) { + let _ = writeln!(out, " {w}"); + } + if !blocks.is_empty() { + // Blocks: each function's entry (its own site) and its other blocks, in reachable + // functions; a block only panics when every path from it does, or its function does. + let mut rows: BTreeMap<&str, [usize; 4]> = BTreeMap::new(); + for s in functions.values() { + if !unreach.contains(&s.path) { + let r = rows.entry(s.krate.as_str()).or_default(); + let (h, i) = (hit_paths.contains(&s.path), ice_only.contains(&s.path)); + r[0] += 1; + r[1] += h as usize; + r[2] += i as usize; + r[3] += (i && h) as usize; + } + } + let (mut logged, mut logged_ran) = (0, 0); + for (site, b) in &blocks { + if NOT_AT_RUN_TIME.contains(&b.krate.as_str()) || unreach.contains(&b.path) || !paths.contains(&b.path) { + continue; + } + if b.logging { + logged += 1; + logged_ran += hit.contains(site) as usize; + continue; + } + let r = rows.entry(b.krate.as_str()).or_default(); + let (h, p) = (hit.contains(site), b.panics || ice_only.contains(&b.path)); + r[0] += 1; + r[1] += h as usize; + r[2] += p as usize; + r[3] += (p && h) as usize; + } + let s: [usize; 4] = rows.values().fold([0; 4], |a, r| [a[0] + r[0], a[1] + r[1], a[2] + r[2], a[3] + r[3]]); + let _ = writeln!( + out, + "blocks: of the {} in reachable functions, {} ran ({:.1}%); {} only panic ({} ran): without them, {} of {} ({:.1}%); not counted: {logged} blocks of logging macros, which run only with RUSTC_LOG ({logged_ran} ran)", + s[0], s[1], 100.0 * s[1] as f64 / s[0].max(1) as f64, s[2], s[3], s[1] - s[3], s[0] - s[2], + 100.0 * (s[1] - s[3]) as f64 / (s[0] - s[2]).max(1) as f64 + ); + let _ = writeln!(out, "{:40} {:>7} {:>9} {:>6} (panic-only blocks aside)", "crate", "ran", "blocks", "%"); + let mut sorted: Vec<(&&str, &[usize; 4])> = rows.iter().collect(); + sorted.sort_by(|a, b| { + let f = |r: &[usize; 4]| (r[1] - r[3]) as f64 / (r[0] - r[2]).max(1) as f64; + f(a.1).partial_cmp(&f(b.1)).unwrap() + }); + for (k, r) in sorted { + if r[0] > r[2] { + let _ = writeln!(out, "{k:40} {:7} {:9} {:6.1}", r[1] - r[3], r[0] - r[2], 100.0 * (r[1] - r[3]) as f64 / (r[0] - r[2]) as f64); + } + } + } + let _ = writeln!(out, "{:40} {:>7} {:>9} {:>6}", "crate", "ran", "reachable", "%"); + let mut sorted: Vec<(&&str, &(usize, usize))> = by_crate.iter().collect(); + sorted.sort_by(|a, b| (a.1.1 as f64 / a.1.0.max(1) as f64).partial_cmp(&(b.1.1 as f64 / b.1.0.max(1) as f64)).unwrap()); + for (k, (n, r)) in sorted { + if *n > 0 { + let _ = writeln!(out, "{k:40} {r:7} {n:9} {:6.1}", 100.0 * *r as f64 / *n as f64); + } + } + } + for krate in &args.unreachable { + let _ = writeln!(out, "\nunreachable in {krate}:"); + let mut list: Vec<&&String> = unreach.iter().filter(|p| p.starts_with(&format!("{krate}::"))).collect(); + list.sort(); + for p in list { + let _ = writeln!(out, " {p}"); + } + } + if let Some(j) = &args.json { + let sorted = |set: Vec<&String>| -> Vec { + let mut v: Vec = set.into_iter().cloned().collect(); + v.sort(); + v + }; + let json = serde_json::json!({ + "unreachable": sorted(unreach.iter().copied().collect()), + "ice_only": sorted(ice_only.iter().filter(|p| paths.contains(p)).collect()), + "ran_unreachable": sorted(hit_paths.iter().filter(|p| unreach.contains(**p)).copied().collect()), + "reachable_not_hit": sorted(paths.iter().filter(|p| !unreach.contains(**p) && !hit_paths.contains(**p)).copied().collect()), + }); + std::fs::write(j, serde_json::to_string(&json)?)?; + } + if let Some(gp) = &args.gaps { + let mut files: BTreeMap<(&str, String), Vec<(String, &String)>> = BTreeMap::new(); + for s in functions.values() { + if !unreach.contains(&s.path) && !hit_paths.contains(&s.path) { + let span = span_of.get(&s.path).cloned().unwrap_or_else(|| "?".into()); + let file = span.rsplitn(3, ':').last().unwrap_or("").to_owned(); + files.entry((s.krate.as_str(), file)).or_default().push((span, &s.path)); + } + } + std::fs::write(gp, gaps_text(&files, |span, path| { + let tag = if ice_only.contains(*path) { " (only panics)" } else { "" }; + let line = span.rsplitn(3, ':').nth(1).unwrap_or(""); + format!("- `{path}` {line}{tag}") + }, "Reachable functions that never ran"))?; + let _ = writeln!(out, "gaps written to {}", gp.display()); + } + if let Some(bp) = &args.block_gaps + && !blocks.is_empty() + { + let mut files: BTreeMap<(&str, String), Vec<(String, String)>> = BTreeMap::new(); + for (site, b) in &blocks { + if hit_paths.contains(&b.path) && !hit.contains(site) && !NOT_AT_RUN_TIME.contains(&b.krate.as_str()) && !b.logging { + let file = b.span.rsplitn(3, ':').last().unwrap_or("").to_owned(); + let panics = b.panics || ice_only.contains(&b.path); + let snippet: String = b.snippet.chars().take(100).collect(); + files.entry((b.krate.as_str(), file)).or_default().push(( + b.span.clone(), + format!("`{}` {}{}: `{snippet}`", b.path, b.name, if panics { " (only panics)" } else { "" }), + )); + } + } + // The lines with the most first within a file. + let line_no = |span: &str| -> (u64, u64) { + let p: Vec<&str> = span.rsplitn(3, ':').collect(); + (p.get(1).and_then(|x| x.parse().ok()).unwrap_or(0), p.first().and_then(|x| x.parse().ok()).unwrap_or(0)) + }; + let mut text = format!("# Blocks that never ran in functions that did: {}\n\n", files.values().map(Vec::len).sum::()); + let mut by_crate: BTreeMap<&str, usize> = BTreeMap::new(); + for ((k, _), v) in &files { + *by_crate.entry(k).or_default() += v.len(); + } + let mut crates: Vec<(&&str, &usize)> = by_crate.iter().collect(); + crates.sort_by(|a, b| b.1.cmp(a.1)); + for (krate, n) in crates { + let _ = writeln!(text, "## {krate} ({n})\n"); + let mut fs: Vec<(&(&str, String), &Vec<(String, String)>)> = files.iter().filter(|((k, _), _)| k == krate).collect(); + fs.sort_by(|a, b| b.1.len().cmp(&a.1.len())); + for ((_, file), entries) in fs { + let _ = writeln!(text, "### {file} ({})\n", entries.len()); + let mut es = entries.clone(); + es.sort_by_key(|(span, _)| line_no(span)); + for (span, desc) in es { + let _ = writeln!(text, "- {} {desc}", line_no(&span).0); + } + text.push('\n'); + } + } + std::fs::write(bp, text)?; + let _ = writeln!(out, "block gaps written to {}", bp.display()); + } + print!("{out}"); + Ok(ExitCode::SUCCESS) +} + +/// A gap list: by crate (largest first), then file (largest first), entries sorted by span. +fn gaps_text

(files: &BTreeMap<(&str, String), Vec<(String, P)>>, entry: impl Fn(&str, &P) -> String, title: &str) -> String { + let total: usize = files.values().map(Vec::len).sum(); + let mut text = format!("# {title}: {total}\n\n"); + let mut by_crate: BTreeMap<&str, usize> = BTreeMap::new(); + for ((k, _), v) in files { + *by_crate.entry(k).or_default() += v.len(); + } + let mut crates: Vec<(&&str, &usize)> = by_crate.iter().collect(); + crates.sort_by(|a, b| b.1.cmp(a.1)); + for (krate, n) in crates { + let _ = writeln!(text, "## {krate} ({n})\n"); + let mut fs: Vec<(&(&str, String), &Vec<(String, P)>)> = files.iter().filter(|((k, _), _)| k == krate).collect(); + fs.sort_by(|a, b| b.1.len().cmp(&a.1.len())); + for ((_, file), entries) in fs { + let _ = writeln!(text, "### {file} ({})\n", entries.len()); + let mut es: Vec<&(String, P)> = entries.iter().collect(); + es.sort_by(|a, b| a.0.cmp(&b.0)); + for (span, p) in es { + let _ = writeln!(text, "{}", entry(span, p)); + } + text.push('\n'); + } + } + text +} diff --git a/crates/mirth-lab/src/tools/coverage.rs b/crates/mirth-lab/src/tools/coverage.rs new file mode 100644 index 0000000..793bb0f --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage.rs @@ -0,0 +1,111 @@ +//! Which of the compiler's functions ran, from a compiler built with rustc/coverage.toml. +//! +//! The site tables list each instrumented function (`cover` sites: id, crate, path, span); every +//! rustc process run with MIRTH_OUT set writes, at exit, a `V ` line for each function it +//! entered. This reads both and prints, per crate, how many functions ran; with --files, per +//! source file; with --unhit, the functions that never ran, in the crates or files matching. +//! Several --logs directories (several runs: fixtures, flags) are combined. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::PathBuf; +use std::process::ExitCode; + +use mirth_lab::coverage::{self, to_json_indent}; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + sites: PathBuf, + #[arg(long, required = true)] + logs: Vec, + #[arg(long)] + json: Option, + /// Also per source file. + #[arg(long)] + files: bool, + /// List the functions that never ran in crates or files matching. + #[arg(long)] + unhit: Vec, +} + +#[derive(Serialize, Default, Clone, Copy)] +struct Count { + functions: usize, + ran: usize, +} + +#[derive(Serialize)] +struct Report { + processes: usize, + functions: usize, + ran: usize, + crates: BTreeMap, + files: BTreeMap, + unhit: Vec, +} + +/// A span's file: all but its last two `:` fields. +pub fn span_file(span: &str) -> &str { + span.rsplitn(3, ':').last().unwrap_or(span) +} + +pub fn run(args: Args) -> anyhow::Result { + // A site listed twice counts once (the last listing). + let functions: Vec = + coverage::functions(&args.sites).into_iter().map(|f| (f.site.clone(), f)).collect::>().into_values().collect(); + let mut hit = BTreeSet::new(); + let mut processes = 0; + for d in &args.logs { + let (h, n) = coverage::hits(d); + hit.extend(h); + processes += n; + } + let sites: BTreeSet<&str> = functions.iter().map(|f| f.site.as_str()).collect(); + let unknown = hit.iter().filter(|s| !sites.contains(s.as_str())).count(); + let (mut by_crate, mut by_file) = (BTreeMap::::new(), BTreeMap::::new()); + for f in &functions { + let ran = hit.contains(&f.site) as usize; + for c in [by_crate.entry(f.krate.clone()).or_default(), by_file.entry(span_file(&f.span).to_owned()).or_default()] { + c.functions += 1; + c.ran += ran; + } + } + let total = sites.len(); + let ran = hit.iter().filter(|s| sites.contains(s.as_str())).count(); + print!("{processes} processes; {ran} of {total} functions ran ({:.1}%)", 100.0 * ran as f64 / total.max(1) as f64); + println!("{}", if unknown > 0 { format!("; {unknown} sites not in the tables") } else { String::new() }); + println!("{:40} {:>7} {:>7} {:>6}", "crate", "ran", "of", "%"); + let ratio = |c: &Count| c.ran as f64 / c.functions as f64; + let mut crates: Vec<(&String, &Count)> = by_crate.iter().collect(); + crates.sort_by(|a, b| ratio(a.1).total_cmp(&ratio(b.1))); + for (k, c) in crates { + println!("{k:40} {:7} {:7} {:6.1}", c.ran, c.functions, 100.0 * ratio(c)); + } + if args.files { + println!(); + println!("{:80} {:>6} {:>6}", "file", "ran", "of"); + let mut files: Vec<(&String, &Count)> = by_file.iter().collect(); + files.sort_by(|a, b| ratio(a.1).total_cmp(&ratio(b.1)).then(b.1.functions.cmp(&a.1.functions))); + for (k, c) in files { + println!("{k:80} {:6} {:6}", c.ran, c.functions); + } + } + let mut by_span: Vec<&coverage::Function> = functions.iter().collect(); + by_span.sort_by(|a, b| a.span.cmp(&b.span)); + for pattern in &args.unhit { + println!("\nnever ran, matching '{pattern}':"); + for f in &by_span { + if !hit.contains(&f.site) && (f.krate.contains(pattern.as_str()) || f.span.contains(pattern.as_str())) { + println!(" {} {}", f.path, f.span); + } + } + } + if let Some(j) = &args.json { + let mut unhit: Vec = functions.iter().filter(|f| !hit.contains(&f.site)).map(|f| format!("{}\t{}", f.path, f.span)).collect(); + unhit.sort(); + let report = Report { processes, functions: total, ran, crates: by_crate, files: by_file, unhit }; + std::fs::write(j, to_json_indent(&report, 1))?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/coverage_compact.rs b/crates/mirth-lab/src/tools/coverage_compact.rs new file mode 100644 index 0000000..8461eb2 --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage_compact.rs @@ -0,0 +1,100 @@ +//! Fold coverage logs (MIRTH_OUT, from a compiler built with rustc/coverage.toml) into a running +//! union as they are finished, and delete them: a test suite run starts tens of thousands of rustc +//! processes, whose logs together would not fit on the disk. +//! +//! A log is finished when its last line is the `X` line written at exit, or when it has not +//! changed for ten minutes (a process that crashed). For each, /added.jsonl gets the +//! process's source file argument and the sites it reached that no earlier process did; +//! /union.txt holds every site reached so far, rewritten every pass. Runs until +//! exists, then does a last pass. + +use std::collections::BTreeSet; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; +use std::time::{Duration, SystemTime}; + +use mirth_lab::coverage::{files_with, log_hits, to_json_line}; +use regex::Regex; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + logs: PathBuf, + #[arg(long)] + out: PathBuf, + #[arg(long, default_value = "")] + until: String, +} + +static CRATE_NAME: LazyLock = LazyLock::new(|| Regex::new(r"--crate-name\t(\S+)").unwrap()); + +/// The process's source file argument, from the log's header line, or its crate name. +fn source_of(header: &str) -> String { + if let Some(f) = header.split('\t').skip(3).find(|f| f.ends_with(".rs")) { + return f.to_owned(); + } + CRATE_NAME.captures(header).map(|c| c[1].to_owned()).unwrap_or_default() +} + +#[derive(serde::Serialize)] +struct Added<'a> { + source: String, + new: &'a [&'a str], +} + +fn one_pass(logs: &Path, out: &Path, union: &mut BTreeSet, last: bool) -> anyhow::Result { + let mut done = 0; + let mut added = std::fs::OpenOptions::new().create(true).append(true).open(out.join("added.jsonl"))?; + for log in files_with(logs, "log") { + let (Ok(bytes), Ok(meta)) = (std::fs::read(&log), std::fs::metadata(&log)) else { continue }; + let text = String::from_utf8_lossy(&bytes); + let age = meta.modified().ok().and_then(|m| SystemTime::now().duration_since(m).ok()).unwrap_or_default().as_secs_f64(); + let Some(last_line) = text.lines().last() else { continue }; + if !(last_line.starts_with("X\t") || age > 600.0 || (last && age > 5.0)) { + continue; + } + let sites: BTreeSet<&str> = log_hits(&text).collect(); + let new: Vec<&str> = sites.into_iter().filter(|s| !union.contains(*s)).collect(); + if !new.is_empty() { + let line = to_json_line(&Added { source: source_of(text.lines().next().unwrap_or("")), new: &new }); + writeln!(added, "{line}")?; + union.extend(new.iter().map(|s| s.to_string())); + } + let _ = std::fs::remove_file(&log); + done += 1; + } + let mut text = union.iter().map(String::as_str).collect::>().join("\n"); + text.push('\n'); + std::fs::write(out.join("union.txt"), text)?; + Ok(done) +} + +/// The local time as HH:MM:SS. +fn clock() -> String { + // SAFETY: time and localtime_r write only to the locals passed. + unsafe { + let t = libc::time(std::ptr::null_mut()); + let mut tm: libc::tm = std::mem::zeroed(); + libc::localtime_r(&t, &mut tm); + format!("{:02}:{:02}:{:02}", tm.tm_hour, tm.tm_min, tm.tm_sec) + } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.out)?; + let union_path = args.out.join("union.txt"); + let mut union: BTreeSet = std::fs::read_to_string(&union_path).unwrap_or_default().split_whitespace().map(str::to_owned).collect(); + loop { + let finishing = !args.until.is_empty() && Path::new(&args.until).exists(); + let n = one_pass(&args.logs, &args.out, &mut union, finishing)?; + println!("{} {n} logs folded, {} sites", clock(), union.len()); + if finishing { + one_pass(&args.logs, &args.out, &mut union, true)?; + break; + } + std::thread::sleep(Duration::from_secs(30)); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/coverage_flags.rs b/crates/mirth-lab/src/tools/coverage_flags.rs new file mode 100644 index 0000000..1feb707 --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage_flags.rs @@ -0,0 +1,122 @@ +//! Coverage of the compiler across option configurations: build a fixture with a +//! coverage-instrumented rustc (rustc/coverage.toml) once per row of a PICT transitions table, +//! clean with the A options, then rebuilt after one random edit with the B options, each row's +//! rustc processes logging to /row/. Read the result with `mirth-lab coverage`. + +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +use rand::rngs::StdRng; +use rand::{Rng as _, SeedableRng}; + +use super::flag_model::{self, Opt}; +use super::flag_walk::{copy_fixture, slice}; +use mirth_lab::mutations; +use mirth_lab::rustc::run_command; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// A coverage-instrumented rustc. + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + /// flag-universe's work directory. + #[arg(long)] + flags: PathBuf, + #[arg(long)] + table: PathBuf, + #[arg(long)] + out: PathBuf, + #[arg(long, default_value = "")] + rows: String, + #[arg(long, default_value_t = 6)] + workers: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, +} + +/// One random edit somewhere in the fixture, build scripts aside. +fn edit(src: &Path, rng: &mut StdRng) { + let mut paths: Vec = walkdir::WalkDir::new(src) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && !p.strip_prefix(src).unwrap().components().any(|c| c.as_os_str() == "target")) + .collect(); + paths.sort(); + if paths.is_empty() { + return; + } + for _ in 0..20 { + let path = &paths[rng.random_range(..paths.len())]; + let (_, f) = mutations::pick(rng); + if path.file_name().is_some_and(|n| n == "build.rs") { + continue; + } + let Ok(old) = std::fs::read_to_string(path) else { continue }; + if let Some(new) = f(&old, rng, 0) { + let _ = std::fs::write(path, new); + break; + } + } +} + +fn row(args: &Args, opts: &std::collections::BTreeMap, out: &Path, i: usize, row: &[(String, String)]) -> String { + let logs = out.join(format!("row{i}")); + if logs.exists() { + return "done before".into(); + } + let work = out.join(format!("work{i}")); + let _ = std::fs::remove_dir_all(&work); + let src = work.join("s"); + if let Err(e) = copy_fixture(&args.fixture, &src) { + return format!("cannot copy the fixture: {e}"); + } + let mut results = Vec::new(); + for side in ["A", "B"] { + if side == "B" { + edit(&src, &mut StdRng::seed_from_u64(i as u64)); + } + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target", "x86_64-unknown-linux-gnu", "--target-dir"]) + .arg(work.join("t")) + .current_dir(&src) + .env("RUSTC", &args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("RUSTFLAGS", flag_model::row_flags(row, Some(side), opts).join(" ")) + .env("MIRTH_OUT", work.join("logs")); + let ok = run_command(cmd, Duration::from_secs(1800)).is_ok_and(|f| f.success()); + results.push(if ok { "ok" } else { "failed" }); + } + let _ = std::fs::create_dir_all(work.join("logs")); + let _ = std::fs::rename(work.join("logs"), &logs); + let _ = std::fs::remove_dir_all(&work); + results.join(" ") +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.out)?; + let out = std::fs::canonicalize(&args.out)?; + let opts = flag_model::options(&args.flags)?; + let rows = flag_model::table(&args.table)?; + let idx = slice(&args.rows, rows.len())?; + // Rows in order, a worker taking the next one when it is free. + let next = AtomicUsize::new(0); + std::thread::scope(|s| { + for _ in 0..args.workers.max(1) { + s.spawn(|| { + loop { + let k = next.fetch_add(1, Ordering::SeqCst); + let Some(&i) = idx.get(k) else { break }; + println!("row {i}: {}", row(&args, &opts, &out, i, &rows[i])); + } + }); + } + }); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/coverage_generators.rs b/crates/mirth-lab/src/tools/coverage_generators.rs new file mode 100644 index 0000000..c9dd276 --- /dev/null +++ b/crates/mirth-lab/src/tools/coverage_generators.rs @@ -0,0 +1,309 @@ +//! Compiler runs that rustc's test suites hardly make, for coverage (`mirth-lab callgraph +//! --gaps` lists what is left): run them with the coverage-instrumented compiler and MIRTH_OUT +//! set, and fold the logs with `mirth-lab coverage-compact`. +//! +//! - prints: every `--print` request, on the host and on every target; +//! - targets: `tests/auxiliary/minicore.rs` and a file of functions with every kind of argument +//! and return value, compiled to an object for every target (each target's ABI, layout and +//! codegen code); +//! - links: for every target, a `no_main` binary, a cdylib, a staticlib and a dylib on minicore, +//! linked with `-Clinker=true` (the linker command each target's linker flavor builds, without +//! the linker), with linker options; +//! - dumps: each test of the list (`mirth-lab ui-coverage pick`) with each debugging and +//! printing option (`-Zunpretty=`, `-Zdump-mir`, `-Zprint-type-sizes`, statistics, profiling, ...). + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::time::Duration; + +use mirth_lab::rustc::run_command; +use mirth_lab::uitest; +use rayon::prelude::*; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (tests/auxiliary/minicore.rs, tests/ui). + #[arg(long)] + rust: PathBuf, + /// The picked tests (picked.json of `mirth-lab ui-coverage pick`). + #[arg(long)] + list: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long, default_value = "prints,targets,dumps,links")] + only: String, +} + +const PRINTS: &[&str] = &[ + "all-target-specs-json", "backend-has-mnemonic", "backend-has-zstd", "calling-conventions", "cfg", "check-cfg", + "code-models", "crate-name", "crate-root-lint-levels", "deployment-target", "file-names", "host-tuple", "link-args", + "native-static-libs", "relocation-models", "split-debuginfo", "stack-protector-strategies", "supported-crate-types", + "sysroot", "target-cpus", "target-features", "target-libdir", "target-list", "target-spec-json", + "target-spec-json-schema", "tls-models", "wasm-proc-macro-tuple", +]; +const PER_TARGET: &[&str] = &[ + "cfg", "target-spec-json", "target-cpus", "target-features", "calling-conventions", "code-models", + "relocation-models", "tls-models", "stack-protector-strategies", "split-debuginfo", "supported-crate-types", + "deployment-target", "check-cfg", +]; + +/// Every kind of argument and return value, for each target's calling convention. +const ABI: &str = r#" +#![feature(no_core, lang_items, rustc_attrs, c_variadic, f16, f128)] +#![no_core] +#![crate_type = "lib"] +#![allow(improper_ctypes_definitions, unused)] +extern crate minicore; +use minicore::*; + +#[repr(C)] pub struct Small { a: u8, b: u16 } +#[repr(C)] pub struct Pair { a: u64, b: u64 } +#[repr(C)] pub struct Big { a: [u64; 8] } +#[repr(C)] pub struct Floats { a: f32, b: f64 } +#[repr(C)] pub struct Mixed { a: f32, b: u32 } +#[repr(C)] pub union U { a: u32, b: f32 } +#[repr(C)] pub struct Hfa { a: f32, b: f32, c: f32, d: f32 } +#[repr(C)] pub struct Empty {} +#[repr(transparent)] pub struct T(u64); +#[repr(C, packed)] pub struct Packed { a: u8, b: u32 } +#[repr(C, align(16))] pub struct Aligned { a: u8 } + +#[no_mangle] pub extern "C" fn c_small(x: Small) -> Small { x } +#[no_mangle] pub extern "C" fn c_pair(x: Pair) -> Pair { x } +#[no_mangle] pub extern "C" fn c_big(x: Big) -> Big { x } +#[no_mangle] pub extern "C" fn c_floats(x: Floats, y: f32, z: f64) -> Floats { x } +#[no_mangle] pub extern "C" fn c_mixed(x: Mixed) -> Mixed { x } +#[no_mangle] pub extern "C" fn c_union(x: U) -> U { x } +#[no_mangle] pub extern "C" fn c_hfa(x: Hfa) -> Hfa { x } +#[no_mangle] pub extern "C" fn c_empty(x: Empty) -> Empty { x } +#[no_mangle] pub extern "C" fn c_transparent(x: T) -> T { x } +#[no_mangle] pub extern "C" fn c_packed(x: Packed) -> Packed { x } +#[no_mangle] pub extern "C" fn c_aligned(x: Aligned) -> Aligned { x } +#[no_mangle] pub extern "C" fn c_ints(a: i8, b: u16, c: i32, d: u64, e: i128, f: u128, g: bool, h: char) -> i128 { e } +#[no_mangle] pub extern "C" fn c_ptrs(a: *const u8, b: &u32, c: &mut [u8; 3], f: extern "C" fn()) -> *const u8 { a } +#[no_mangle] pub extern "C" fn c_many(a: u64, b: u64, c: u64, d: u64, e: u64, f: u64, g: u64, h: u64, i: u64, j: Pair, k: f64, l: f64, m: f64, n: f64, o: f64, p: f64, q: f64, r: f64, s: f64) -> u64 { a } +#[no_mangle] pub unsafe extern "C" fn c_variadic(a: u32, mut args: ...) -> u32 { a } +pub fn rust_all(a: Small, b: Pair, c: Big, d: Floats, e: (u8, u64), f: [u32; 5], g: &[u8], h: &str, i: u128) -> Big { c } +pub fn rust_f16(a: f16, b: f128) -> f128 { b } +#[no_mangle] pub extern "C" fn c_f16(a: f16, b: f128) -> f128 { b } +#[no_mangle] pub extern "system" fn system(a: Pair) -> Pair { a } +#[no_mangle] pub extern "C-unwind" fn c_unwind(a: Pair) -> Pair { a } +pub static TABLE: [extern "C" fn(Pair) -> Pair; 2] = [c_pair, c_unwind_shim]; +extern "C" fn c_unwind_shim(a: Pair) -> Pair { a } +extern "C" { fn imported(a: Big, b: Floats) -> Hfa; } +pub unsafe fn call_imported(a: Big, b: Floats) -> Hfa { imported(a, b) } +"#; + +const LINKED: &str = r#" +#![feature(no_core, lang_items)] +#![no_core] +#![no_main] +extern crate minicore; +#[no_mangle] pub extern "C" fn exported(a: u32) -> u32 { a } +#[no_mangle] pub static DATA: u32 = 7; +#[link(name = "c")] extern "C" { fn puts(p: *const u8) -> i32; } +#[link(name = "m", kind = "static")] extern "C" {} +#[link(name = "framework_like", kind = "dylib", modifiers = "+verbatim")] extern "C" {} +"#; + +const LINK_OPTIONS: &[&[&str]] = &[ + &[], &["-Cprefer-dynamic", "-Crelocation-model=pic"], &["-Cstrip=symbols", "-Clink-dead-code"], + &["-Clink-self-contained=yes"], &["-Cdebuginfo=2", "-Csplit-debuginfo=packed"], + &["-Clink-arg=-Wl,--foo", "-Clink-args=-x -y", "-Zpre-link-args=-z"], + &["-Cdefault-linker-libraries", "-Zlink-native-libraries=no"], &["-Ccontrol-flow-guard"], + &["-Zstaticlib-allow-rdylib-deps"], &["-Copt-level=s", "-Clto=fat"], &["-Ccode-model=large"], +]; + +const UNPRETTY: &[&str] = &[ + "normal", "expanded", "expanded,identified", "expanded,hygiene", "ast-tree", "ast-tree,expanded", "hir", + "hir,identified", "hir,typed", "hir-tree", "thir-tree", "thir-flat", "mir", "stable-mir", "mir-cfg", +]; +const DUMPS: &[&[&str]] = &[ + &["-Zdump-mir=all", "-Zdump-mir-dataflow", "-Zdump-mir-graphviz", "-Zmir-include-spans=on"], + &["-Zprint-type-sizes"], &["-Zprint-mono-items=yes", "--emit=link"], &["-Zmeta-stats"], &["-Zhir-stats"], + &["-Zinput-stats"], &["-Zself-profile", "-Zself-profile-events=all"], &["-Ztime-passes"], + &["-Zquery-dep-graph", "-Zdump-dep-graph", "-Cincremental=inc"], &["-Zincremental-info", "-Cincremental=inc"], + &["-Zdump-mono-stats", "-Zdump-mono-stats-format=json", "--emit=link"], &["-Zprint-codegen-stats", "--emit=link"], + &["-Zvalidate-mir", "-Zlint-mir", "-Zmir-opt-level=4"], &["-Zverbose-internals", "-Zidentify-regions"], + &["-Ztrack-diagnostics", "-Zteach"], &["-Zthreads=4"], &["-Zpolonius=next"], &["-Zinline-mir", "-Zmir-opt-level=3"], + &["-Zrandomize-layout"], &["-Zwrite-long-types-to-disk=no", "-Zverbose-internals"], + &["-Zunleash-the-miri-inside-of-you"], &["-Zno-analysis"], &["-Zprofile-closures"], &["-Zui-testing"], + &["-Cinstrument-coverage", "--emit=link"], &["-Zemit-stack-sizes", "--emit=link"], + &["--error-format=json", "--json=diagnostic-rendered-ansi,artifacts,future-incompat,unused-externs"], + &["--error-format=human-annotate-rs"], &["--error-format=short"], &["-Zterminal-urls=yes", "--color=always"], + &["-Wunused", "-Wrust-2018-idioms", "-Wrust-2021-compatibility", "-Wrust-2024-compatibility", "-Wclippy::all"], + &["-Fwarnings", "--cap-lints=warn"], &["-Zcodegen-source-order", "--emit=link"], +]; + +/// The debugging and printing options: each `-Zunpretty=` mode, then the rest. +fn dumps_options() -> Vec> { + let unpretty = UNPRETTY.iter().map(|m| vec![format!("-Zunpretty={m}")]); + unpretty.chain(DUMPS.iter().map(|o| o.iter().map(|s| s.to_string()).collect())).collect() +} + +/// Run with RUSTC_BOOTSTRAP; true when it exits 0. +fn ok>(program: &Path, argv: &[S], cwd: &Path, secs: u64) -> bool { + let mut cmd = Command::new(program); + cmd.args(argv).current_dir(cwd).env("RUSTC_BOOTSTRAP", "1"); + run_command(cmd, Duration::from_secs(secs)).is_ok_and(|f| f.success()) +} + +fn targets(args: &Args) -> Vec { + let out = Command::new(&args.rustc).args(["--print", "target-list"]).env("RUSTC_BOOTSTRAP", "1").output(); + out.map(|o| String::from_utf8_lossy(&o.stdout).split_whitespace().map(str::to_owned).collect()).unwrap_or_default() +} + +fn scratch(args: &Args) -> tempfile::TempDir { + tempfile::tempdir_in(&args.work).expect("scratch directory") +} + +fn prints(args: &Args) { + let d = scratch(args); + let empty = d.path().join("lib.rs"); + let _ = std::fs::write(&empty, ""); + let empty = empty.to_string_lossy().into_owned(); + let mut jobs: Vec> = PRINTS.iter().map(|k| vec!["--print".into(), k.to_string(), "-Zunstable-options".into(), empty.clone()]).collect(); + for target in targets(args) { + for k in PER_TARGET { + jobs.push(vec!["--print".into(), k.to_string(), "--target".into(), target.clone(), "-Zunstable-options".into(), empty.clone()]); + } + } + let succeeded = jobs.par_iter().filter(|j| ok(&args.rustc, j, d.path(), 300)).count(); + println!("prints: {} runs, {succeeded} succeeded", jobs.len()); +} + +/// The common arguments for a `no_core` build for `target` into `d`. +fn target_base(target: &str, d: &Path) -> Vec { + ["--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", "--out-dir"] + .iter() + .map(|s| s.to_string()) + .chain([d.to_string_lossy().into_owned()]) + .collect() +} + +fn one_target(args: &Args, target: &str) -> Vec { + let d = scratch(args); + let p = d.path(); + let _ = std::fs::write(p.join("abi.rs"), ABI); + let minicore = args.rust.join("tests/auxiliary/minicore.rs").to_string_lossy().into_owned(); + let base = target_base(target, p); + let code = |extra: &[String]| { + let mut cmd = Command::new(&args.rustc); + cmd.args(&base).args(extra).current_dir(p).env("RUSTC_BOOTSTRAP", "1"); + match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => match f.exit { + mirth_lab::rustc::Exit::Code(c) => c, + _ => -1, + }, + Err(_) => -1, + } + }; + let s = |v: &[&str]| v.iter().map(|x| x.to_string()).collect::>(); + let mut results = vec![code(&[s(&["--crate-type", "rlib", "--crate-name", "minicore", "-Copt-level=1", "--emit=link,obj"]), vec![minicore]].concat())]; + if results[0] == 0 { + let rlib = format!("minicore={}/libminicore.rlib", p.display()); + for opt in ["0", "3"] { + results.push(code(&[s(&["--emit=obj,asm,llvm-ir"]), vec![format!("-Copt-level={opt}")], s(&["-Cdebuginfo=2", "--extern"]), vec![rlib.clone()], s(&["abi.rs"])].concat())); + } + } + results +} + +fn cross(args: &Args) -> anyhow::Result<()> { + let done: Vec<(String, Vec)> = targets(args).into_par_iter().map(|t| { let r = one_target(args, &t); (t, r) }).collect(); + let built = done.iter().filter(|(_, r)| r.len() == 3 && r.iter().all(|c| *c == 0)).count(); + println!("targets: {} targets, {built} built minicore and the ABI file", done.len()); + let map: BTreeMap> = done.into_iter().collect(); + std::fs::write(args.work.join("targets.json"), mirth_lab::coverage::to_json_indent(&map, 0))?; + Ok(()) +} + +fn one_link(args: &Args, target: &str) -> usize { + let d = scratch(args); + let p = d.path(); + let _ = std::fs::write(p.join("linked.rs"), LINKED); + let minicore = args.rust.join("tests/auxiliary/minicore.rs").to_string_lossy().into_owned(); + let mut base = target_base(target, p); + base.push("-Clinker=true".into()); + let with = |extra: &[&str]| -> Vec { base.iter().cloned().chain(extra.iter().map(|s| s.to_string())).collect() }; + let mut minicore_argv = with(&["--crate-type", "rlib", "--crate-name", "minicore", "--emit=link"]); + minicore_argv.push(minicore); + if !ok(&args.rustc, &minicore_argv, p, 300) { + return 0; + } + let rlib = format!("minicore={}/libminicore.rlib", p.display()); + let mut n = 0; + for options in LINK_OPTIONS { + for kind in ["bin", "cdylib", "staticlib", "dylib"] { + let mut argv = with(&["--crate-type", kind, "--extern", &rlib, "-Csave-temps"]); + argv.extend(options.iter().map(|s| s.to_string())); + argv.push("linked.rs".into()); + n += ok(&args.rustc, &argv, p, 300) as usize; + } + } + n +} + +fn links(args: &Args) { + let done: Vec = targets(args).par_iter().map(|t| one_link(args, t)).collect(); + println!("links: {} targets, {} links succeeded", done.len(), done.iter().sum::()); +} + +fn dump(args: &Args, options: &[Vec], test: &str) -> usize { + let path = args.rust.join("tests/ui").join(test); + let text = String::from_utf8_lossy(&std::fs::read(&path).unwrap_or_default()).into_owned(); + let (flags, edition, _, _) = uitest::headers(&text); + let mut n = 0; + for extra in options { + let d = scratch(args); + let mut argv: Vec = vec![path.to_string_lossy().into_owned(), "--edition".into(), edition.clone().unwrap_or_else(|| "2015".into())]; + if !extra.iter().any(|e| e.starts_with("--emit")) { + argv.push("--emit=metadata".into()); + } + argv.extend(["--out-dir".into(), d.path().to_string_lossy().into_owned()]); + argv.extend(["-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"].map(String::from)); + argv.extend(flags.iter().cloned()); + argv.extend(extra.iter().cloned()); + n += ok(&args.rustc, &argv, d.path(), 120) as usize; + } + n +} + +fn dumps(args: &Args) -> anyhow::Result<()> { + let picked: Vec = serde_json::from_str(&std::fs::read_to_string(&args.list)?)?; + let tests: Vec = picked + .iter() + .filter_map(|t| t.get("test").unwrap_or(t).as_str().map(str::to_owned)) + .collect(); + let options = dumps_options(); + let succeeded: usize = tests.par_iter().map(|t| dump(args, &options, t)).sum(); + println!("dumps: {} tests x {} options, {succeeded} runs succeeded", tests.len(), options.len()); + Ok(()) +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let only: Vec<&str> = args.only.split(',').collect(); + pool.install(|| -> anyhow::Result<()> { + if only.contains(&"prints") { + prints(&args); + } + if only.contains(&"targets") { + cross(&args)?; + } + if only.contains(&"dumps") { + dumps(&args)?; + } + if only.contains(&"links") { + links(&args); + } + Ok(()) + })?; + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/crash_diff.rs b/crates/mirth-lab/src/tools/crash_diff.rs new file mode 100644 index 0000000..e679d80 --- /dev/null +++ b/crates/mirth-lab/src/tools/crash_diff.rs @@ -0,0 +1,100 @@ +//! Internal checks on: what rustc's own invariants say about every UI test. +//! +//! Compiles each standalone UI test with the compiler under test and with a second one built +//! from the same tree with debug assertions (and overflow checks), adding `-Zvalidate-mir`. +//! A crash, failed assertion or MIR validation error under the second that the first does not +//! have is a finding. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// The release compiler under test. + #[arg(long)] + rustc: PathBuf, + /// The same compiler built with debug assertions. + #[arg(long)] + checked: PathBuf, + #[arg(long, default_value = "-Zvalidate-mir")] + extra: String, + #[command(flatten)] + sweep: Sweep, +} + +static MESSAGE: LazyLock> = LazyLock::new(|| { + [r"panicked at [^\n]*\n[^\n]*", r"internal compiler error: [^\n]*", r"broken MIR[^\n]*"] + .iter() + .map(|p| Regex::new(p).unwrap()) + .collect() +}); +static COMPILER_PATH: LazyLock = LazyLock::new(|| Regex::new(r"/\S+/compiler/").unwrap()); + +/// The first line saying what went wrong inside the compiler. +fn message(stderr: &str) -> String { + MESSAGE + .iter() + .find_map(|re| re.find(stderr)) + .map(|m| COMPILER_PATH.replace_all(m.as_str(), "compiler/").chars().take(400).collect()) + .unwrap_or_default() +} + +#[derive(Serialize)] +struct Rec { + test: String, + release: Status, + checked: Status, + #[serde(skip_serializing_if = "Option::is_none")] + note: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn check(args: &Args, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let emit = if Kind::is_check(test.kind) { "metadata" } else { "link" }; + let a = Compile::new(&args.rustc, &test.path, &dir.path().join("release"), &test.flags, test.edition()) + .emit(emit) + .run(); + let b = Compile::new(&args.checked, &test.path, &dir.path().join("checked"), &test.flags, test.edition()) + .extra(args.extra.split_whitespace()) + .emit(emit) + .timeout(600) + .run(); + let mut rec = Rec { test: test.rel.clone(), release: a.status, checked: b.status, note: None, found: Vec::new() }; + if b.status == Status::Ice && a.status != Status::Ice { + let m = message(&b.stderr); + rec.found.push(format!("only with internal checks: {}", m.chars().take(160).collect::())); + let stderr: String = b.stderr.chars().rev().take(4000).collect::().chars().rev().collect(); + driver::write_finding( + &args.sweep.work, + test, + &[], + &serde_json::json!({ "extra": args.extra, "found": [{ "what": "only with internal checks", "message": m, "stderr": stderr }] }), + ); + } else if b.status == Status::Ice && a.status == Status::Ice && message(&a.stderr) != message(&b.stderr) { + rec.note = Some("both crash, differently".into()); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, uitest::ALL, |_| false)); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} diff --git a/crates/mirth-lab/src/tools/diag_check.rs b/crates/mirth-lab/src/tools/diag_check.rs new file mode 100644 index 0000000..cafbbc7 --- /dev/null +++ b/crates/mirth-lab/src/tools/diag_check.rs @@ -0,0 +1,144 @@ +//! Diagnostic invariants: what every diagnostic rustc prints must satisfy, whatever the program. +//! +//! Compiles each standalone UI test with `--error-format=json` and checks every diagnostic, +//! children and suggestions included: +//! +//! - internal: user-facing text (message, labels, suggested code) contains compiler-internal +//! debug output (`DefId(`, region and type-variable debug names, `Opaque(DefId`, `{closure#0}` +//! in suggested code) +//! - span: a span outside its file (offsets past the end, lines past the last, start after end) +//! +//! Errors without any span and exact duplicates are notes. Tests that ask for compiler internals +//! on purpose (verbose printing, dump attributes) are left out. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[command(flatten)] + sweep: Sweep, +} + +static INTERNAL: LazyLock = LazyLock::new(|| { + Regex::new(r"DefId\(|\bRe(LateParam|Bound|Var|Early|Static)\b|'\{erased\}|\?\d+[tif]\b|'\^\d+(_\d+)?\b|Opaque\(DefId|\bAlias\((Projection|Opaque|Inherent|Free)|\bBoundRegionKind|\bDefPath\b|\bLocalDefId\b|\bTyKind::").unwrap() +}); +static INTERNAL_CODE: LazyLock = + LazyLock::new(|| Regex::new(r"\{closure#\d+\}|\{opaque#\d+\}|\{async block@|\{impl#\d+\}|\{constant#\d+\}").unwrap()); +static DEBUG_TEST: LazyLock = LazyLock::new(|| { + Regex::new(r"#!?\[rustc_(dump|effective_visibility|regions|variance|outlives|layout|abi|def_path|symbol_name|object_lifetime_default|evaluate_where_clauses|then_this_would_need|if_this_changed|clean|partition)").unwrap() +}); +static DEBUG_FLAG: LazyLock = LazyLock::new(|| Regex::new(r"verbose|-Zdump|unpretty|print-").unwrap()); +static SUMMARY: LazyLock = + LazyLock::new(|| Regex::new(r"^(aborting due to|could not compile|\d+ (previous )?errors?)").unwrap()); + +#[derive(Serialize, Clone, PartialEq, Eq, PartialOrd, Ord)] +struct Finding { + what: String, + detail: String, + code: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn file_info(cache: &mut BTreeMap>, test_dir: &Path, name: &str) -> Option<(usize, usize)> { + *cache.entry(name.to_owned()).or_insert_with(|| { + let p = Path::new(name); + let data = std::fs::read(if p.is_absolute() { p.to_path_buf() } else { test_dir.join(p) }).ok()?; + Some((data.len(), data.iter().filter(|&&b| b == b'\n').count() + 1)) + }) +} + +fn check(args: &Args, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let c = Compile::new(&args.rustc, &test.path, dir.path(), &test.flags, test.edition()) + .emit("metadata") + .json() + .timeout(120) + .run(); + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new(), notes: Vec::new() }; + if matches!(c.status, Status::Ice | Status::Timeout) { + rec.skip = Some(format!("{:?}", c.status).to_lowercase()); + return rec; + } + let diags = rustc::diagnostics(&c.stderr); + let test_dir = test.path.parent().unwrap_or(Path::new(".")); + let mut cache = BTreeMap::new(); + let mut found: BTreeSet = BTreeSet::new(); + let mut seen: BTreeMap<(String, String, String, Option<(String, usize, usize)>), usize> = BTreeMap::new(); + for diag in &diags { + let prim = diag.primary().map(|s| (s.file_name.clone(), s.byte_start, s.byte_end)); + *seen.entry((diag.level.clone(), diag.code().to_owned(), diag.message.clone(), prim)).or_default() += 1; + if diag.level == "error" && diag.spans.is_empty() && diag.children.is_empty() && !SUMMARY.is_match(&diag.message) { + rec.notes.push(format!("nowhere: {}", diag.message.chars().take(100).collect::())); + } + for node in diag.walk() { + let texts = std::iter::once(node.message.as_str()).chain(node.spans.iter().filter_map(|s| s.label.as_deref())); + for t in texts { + if let Some(m) = INTERNAL.find(t) { + found.insert(Finding { what: "internal".into(), detail: format!("{} :: {}", m.as_str(), t.chars().take(300).collect::()), code: diag.code().into() }); + } + } + for s in &node.spans { + if let Some(repl) = &s.suggested_replacement + && let Some(m) = INTERNAL.find(repl).or_else(|| INTERNAL_CODE.find(repl)) + { + found.insert(Finding { what: "internal".into(), detail: format!("{} :: suggests {:?}", m.as_str(), repl.chars().take(200).collect::()), code: diag.code().into() }); + } + if s.file_name.starts_with('<') { + continue; + } + if let Some((size, lines)) = file_info(&mut cache, test_dir, &s.file_name) + && (s.byte_start > s.byte_end || s.byte_end > size || s.line_start > lines || s.line_end > lines || s.line_start > s.line_end) + { + found.insert(Finding { what: "span".into(), detail: format!("{}:{}..{} lines {}..{} (file: {size} bytes, {lines} lines): {}", s.file_name, s.byte_start, s.byte_end, s.line_start, s.line_end, node.message.chars().take(150).collect::()), code: diag.code().into() }); + } + } + } + } + for ((level, _, message, _), n) in &seen { + if *n > 1 && (level == "error" || level == "warning") { + rec.notes.push(format!("duplicate x{n}: {}", message.chars().take(100).collect::())); + } + } + rec.found = found.iter().map(|f| format!("{}: {}", f.what, f.detail.chars().take(120).collect::())).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found, "notes": rec.notes })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| { + uitest::flag_matches(t, &DEBUG_FLAG) || DEBUG_TEST.is_match(&t.text) || t.text.contains("assumptions_on_binders") + }); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} diff --git a/crates/mirth-lab/src/tools/flag_fuzz.rs b/crates/mirth-lab/src/tools/flag_fuzz.rs new file mode 100644 index 0000000..01f17c1 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_fuzz.rs @@ -0,0 +1,92 @@ +//! Run the fuzzer under option configurations: for each chosen row of a PICT table (the B side +//! of a `flag-model --transitions --cargo` model), fuzz the fixture with those options in +//! RUSTFLAGS for a number of edits. Stops at the first finding (`mirth-lab fuzz +//! --pause-on-finding`, exit 3); rerunning resumes after the rows already done. +//! +//! Writes

/row/ (the fuzzer's work directory) and /rows.jsonl (one line per +//! finished row: options, the fuzzer's totals). + +use std::collections::HashSet; +use std::io::Write as _; +use std::path::PathBuf; +use std::process::{Command, ExitCode}; + +use serde::{Deserialize, Serialize}; + +use super::flag_model; +use super::flag_walk::slice; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + /// flag-universe's work directory. + #[arg(long)] + flags: PathBuf, + #[arg(long)] + table: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value = "")] + rows: String, + /// Per row, over all workers. + #[arg(long, default_value_t = 200)] + edits: usize, + #[arg(long, default_value_t = 8)] + workers: usize, +} + +#[derive(Serialize, Deserialize)] +struct Done { + row: usize, + flags: Vec, + total: String, +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let opts = flag_model::options(&args.flags)?; + let rows = flag_model::table(&args.table)?; + let log = work.join("rows.jsonl"); + let done: HashSet = std::fs::read_to_string(&log) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()?["row"].as_u64().map(|r| r as usize)) + .collect(); + let me = std::env::current_exe()?; + for i in slice(&args.rows, rows.len())? { + if done.contains(&i) { + continue; + } + let flags = flag_model::row_flags(&rows[i], Some("B"), &opts); + let w = work.join(format!("row{i}")); + let out = Command::new(&me) + .arg("fuzz") + .args(["--rustc", &args.rustc, "--fixture"]) + .arg(&args.fixture) + .arg("--work") + .arg(&w) + .args(["--workers", &args.workers.to_string()]) + .args(["--edits", &(args.edits / args.workers.max(1)).max(1).to_string()]) + .args(["--seed", &i.to_string()]) + .args(["--rustflags", &flags.join(" ")]) + .args(["--target", "x86_64-unknown-linux-gnu", "--pause-on-finding"]) + .output()?; + let stdout = String::from_utf8_lossy(&out.stdout); + let total = stdout.trim().lines().last().unwrap_or("").to_owned(); + let paused = std::fs::read_to_string(w.join("PAUSED")).ok(); + match &paused { + Some(p) => println!("row {i}: {total} PAUSED {p}"), + None => println!("row {i}: {total}"), + } + if paused.is_some() { + return Ok(ExitCode::from(3)); // not recorded as done: rerun after patching to do this row again + } + let mut f = std::fs::OpenOptions::new().create(true).append(true).open(&log)?; + writeln!(f, "{}", serde_json::to_string(&Done { row: i, flags: flags[1..].to_vec(), total })?)?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_min.rs b/crates/mirth-lab/src/tools/flag_min.rs new file mode 100644 index 0000000..33f43d4 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_min.rs @@ -0,0 +1,164 @@ +//! Minimize the failures of a flag-walk run: for each distinct first error among rows whose +//! clean build with A failed, find the smallest set of the row's options that still gives the +//! same error (delta debugging, a clean build of the fixture per test). +//! +//! Prints one line per error: the minimal options. Writes /minimized.json. + +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +use mirth_lab::coverage::to_json_indent; +use super::flag_model::{FLAG_BASE}; +use super::flag_walk::copy_fixture; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + fixture: PathBuf, + /// flag-walk's work directory. + #[arg(long)] + walk: PathBuf, + #[arg(long, default_value_t = 1)] + per_error: usize, + #[arg(long, default_value_t = 4)] + jobs: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "x86_64-unknown-linux-gnu")] + target: String, +} + +static VOLATILE: LazyLock = LazyLock::new(|| Regex::new(r"_R\w+|/\S+|`[^`]*`|\b[0-9a-f]{16}\b").unwrap()); +static DIGITS: LazyLock = LazyLock::new(|| Regex::new(r"\d+").unwrap()); + +/// An error line with paths, symbols, hashes and quoted names removed. +pub fn signature(line: &str) -> String { + let line = VOLATILE.replace_all(line, "…"); + DIGITS.replace_all(&line, "N").chars().take(120).collect() +} + +fn first_error(log: &str) -> &str { + for l in log.lines() { + if (l.starts_with("error") || l.starts_with("rustc-LLVM ERROR") || l.starts_with("LLVM ERROR")) && !l.contains("could not compile") { + return l; + } + if l.contains("panicked at") { + return l; + } + } + "" +} + +#[derive(Deserialize)] +struct Row { + #[serde(rename = "A")] + a: Vec, + #[serde(rename = "A_ok")] + a_ok: bool, + #[serde(default)] + error: Option, + #[serde(default)] + errors: Option>, +} + +#[derive(Serialize)] +struct Minimized { + error: String, + minimal: Option>, +} + +fn minimize(args: &Args, mut flags: Vec, sig: &str) -> Option> { + let work = tempfile::tempdir_in(&args.walk).expect("scratch"); + let (src, target) = (work.path().join("s"), work.path().join("t")); + copy_fixture(&args.fixture, &src).ok()?; + let fails = |fl: &[String]| -> bool { + let _ = std::fs::remove_dir_all(&target); + let rustflags = std::iter::once(FLAG_BASE.to_owned()).chain(fl.iter().cloned()).collect::>().join(" "); + let r = Command::new("cargo") + .arg(format!("+{}", args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target", &args.target, "--target-dir"]) + .arg(&target) + .current_dir(&src) + .env("RUSTC", &args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("CARGO_TERM_COLOR", "never") + .env("RUSTFLAGS", rustflags) + .output(); + match r { + Ok(o) => !o.status.success() && signature(first_error(&String::from_utf8_lossy(&o.stderr))) == sig, + Err(_) => false, + } + }; + if !fails(&flags) { + return None; + } + let mut n = 2; + while flags.len() >= 2 { + let chunk = (flags.len() / n).max(1); + let mut reduced = false; + let mut i = 0; + while i < flags.len() { + let rest: Vec = flags[..i].iter().chain(flags[(i + chunk).min(flags.len())..].iter()).cloned().collect(); + if fails(&rest) { + flags = rest; + n = (n - 1).max(2); + reduced = true; + break; + } + i += chunk; + } + if !reduced { + if chunk == 1 { + break; + } + n = flags.len().min(n * 2); + } + } + Some(flags) +} + +pub fn run(args: Args) -> anyhow::Result { + let text = std::fs::read_to_string(args.walk.join("results.jsonl"))?; + // By signature, in first-seen order. + let mut todo: Vec<(String, Vec>)> = Vec::new(); + for line in text.lines().filter(|l| !l.trim().is_empty()) { + let r: Row = serde_json::from_str(line)?; + if r.a_ok { + continue; + } + let err = r.errors.as_ref().and_then(|e| e.first().cloned()).or(r.error.clone()).unwrap_or_default(); + let sig = signature(&err); + let i = match todo.iter().position(|(s, _)| *s == sig) { + Some(i) => i, + None => { + todo.push((sig, Vec::new())); + todo.len() - 1 + } + }; + if todo[i].1.len() < args.per_error { + todo[i].1.push(r.a); + } + } + let jobs: Vec<(&String, &Vec)> = todo.iter().flat_map(|(s, fls)| fls.iter().map(move |f| (s, f))).collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let out: Vec = pool.install(|| { + jobs.par_iter().map(|(sig, fl)| Minimized { error: (*sig).clone(), minimal: minimize(&args, (*fl).clone(), sig) }).collect() + }); + std::fs::write(args.walk.join("minimized.json"), to_json_indent(&out, 1))?; + for o in &out { + let m = match &o.minimal { + Some(v) => format!("[{}]", v.iter().map(|s| format!("'{s}'")).collect::>().join(", ")), + None => "None".into(), + }; + println!("{m} -> {}", o.error); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_model.rs b/crates/mirth-lab/src/tools/flag_model.rs new file mode 100644 index 0000000..2e1889e --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_model.rs @@ -0,0 +1,320 @@ +//! A PICT model of rustc's option universe, from flag-universe's results. +//! +//! One parameter per option. Its values are absence, the values rustc accepted alone, the +//! values it accepts once `-Cunsafe-allow-abi-mismatch` names every target modifier +//! (FLAG_BASE, passed on every row), and the values that need another option, with IF/THEN +//! constraints for those needs. Options that stop compilation early (help, parse-only, +//! link-only) are left out. A value whose need lies outside the subset is dropped. +//! +//! With --cargo the model is for building a Cargo workspace (flag-walk): it leaves out the +//! values in CARGO_DROP, which fail there for reasons of Cargo or this machine, not the options. +//! +//! Combinations that hit bugs already in docs/hunt.md are excluded, so walks look for new +//! ones; --allow-known keeps those that have a local stopgap (for a compiler with them). +//! +//! With --transitions every parameter appears twice, A_ before and B_ after, for covering the +//! changes between two sessions. +//! +//! Run it with PICT (github.com/microsoft/pict): `pict /o:2` gives a pairwise covering +//! array, `/o:3` three-way. +//! +//! Also the parts the other flag tools share: options.json, table rows, a row's flags. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// flag-universe's work directory. + work: PathBuf, + /// all, untracked or tracked. + subset: String, + out: PathBuf, + #[arg(long)] + transitions: bool, + #[arg(long)] + cargo: bool, + #[arg(long)] + allow_known: bool, +} + +pub const FLAG_BASE: &str = "-Cunsafe-allow-abi-mismatch=sanitizer,sanitizer-cfi-normalize-integers,\ +sanitizer-cfi-minimal-runtime,retpoline,retpoline-external-thunk,\ +indirect-branch-cs-prefix,fixed-x18,reg-struct-return,regparm,branch-protection"; + +const STOP: &[&str] = &["-Chelp", "-Zhelp", "-Zparse-crate-root-only", "-Zno-analysis", "-Zlink-only", "-Zimplicit-sysroot-deps"]; + +/// (option, values to drop; None: the whole option) +const CARGO_DROP: &[(&str, Option<&[&str]>)] = &[ + ("-Zassert-incr-state", None), // fails whenever the cache state differs, by design + ("-Zbuild-sdylib-interface", None), // Cargo's target probe fails + ("-Zchecksum-hash-algorithm", Some(&["md5", "sha1"])), // Cargo cannot parse the dep info + ("-Zdirect-access-external-data", None), // link fails + ("-Zfunction-return", Some(&["thunk-extern"])), // link fails: no thunk + ("-Zlink-native-libraries", None), // link fails + ("-Zlint-llvm-ir", None), // aborts on known LLVM lint findings (rust-lang/rust#59793) + ("-Zno-codegen", None), // later crates need the output + ("-Zno-link", None), + ("-Zpanic-in-drop", None), // std is built with unwind + ("-Zsanitizer", None), // no sanitizer runtimes in this sysroot: link fails + ("-Zretpoline-external-thunk", None), // link fails: no thunk + ("-Ztiny-const-eval-limit", None), // the fixture's const evaluation exceeds it + ("-Cpanic", Some(&["immediate-abort"])), // core is built with unwind + ("-Ccode-model", Some(&["tiny"])), // LLVM ERROR: not supported on x86_64 + ("-Ztls-model", Some(&["local-exec", "emulated"])), // dylib cannot link + // the dylib cannot link (static, pie, ropi); rwpi: finding 12 + ("-Crelocation-model", Some(&["static", "pie", "ropi", "rwpi", "ropi-rwpi"])), + ("-Clto", None), // rejected for rlibs and dylibs; Cargo's profile applies it to final artifacts only + // Makes a crate behave like the standard library, which needs stability attributes on + // `const trait`s (fixtures/sink/nightly has one). + ("-Zforce-unstable-if-unmarked", Some(&["yes"])), + // LLVM's pass listing from codegen threads interleaves with rustc's lines on stderr; a split + // -Ztime-passes-format=json line then reaches Cargo as a bare JSON message. + ("-Zprint-llvm-passes", Some(&["yes"])), + ("-Ztime-passes-format", Some(&["json"])), +]; + +/// (option, value, options needed, PICT condition) +type Need = (&'static str, &'static str, &'static [&'static str], &'static str); + +// Constraints that only a real workspace shows: a binary, a dylib, Cargo's own flags. +const CARGO_NEEDS: &[Need] = &[ + ("-Cprefer-dynamic", "yes", &["-Cpanic"], r#"[Cpanic] <> "abort""#), // libstd.so has panic_unwind + ("-Cprefer-dynamic", "yes", &["-Clto"], r#"[Clto] IN {"absent","no","off"}"#), + // Findings 13 and 14 (in LLVM, not patched): retpolines with the machine outliner, or with + // the large code model. + ("-Zretpoline", "yes", &["-Ccode-model"], r#"[Ccode_model] <> "large""#), + // Finding 16 (no stopgap): cached derive expansions with the HIR crate hash. + ("-Zcache-proc-macros", "yes", &["-Zmetadata-crate-hash"], r#"[Zmetadata_crate_hash] <> "no""#), +]; +// Bugs in docs/hunt.md that have a local stopgap: excluded unless --allow-known (for a +// compiler with the stopgaps). +const KNOWN_NEEDS: &[Need] = &[ + // Finding 9: without the default passes, local ThinLTO leaves undefined hidden symbols. + ("-Cno-prepopulate-passes", "present", &["-Zthinlto", "-Copt-level"], r#"[Zthinlto] <> "yes" AND [Copt_level] IN {"absent","0"}"#), +]; +// Values left out of every model: LLVM's machine outliner crashes in many combinations +// (finding 13), which buries everything else; and see below. +const DROP: &[(&str, &[&str])] = &[ + ("-Cllvm-args", &["-enable-machine-outliner"]), + // Not a bug: the limit counts MIR pass runs across the session, so which bodies stay under + // it depends on how many bodies the session computes (an incremental session computes + // fewer) and, with -Zthreads, on thread timing. + ("-Zmir-opt-bisect-limit", &["1", "16"]), + // Not a bug: a testing option that leaves spans out of the incremental hashes, so a rebuild + // keeps stale spans by design. + ("-Zincremental-ignore-spans", &["yes"]), +]; +// Rejected alone; accepted with FLAG_BASE or with the needs below. +const EXTRA: &[(&str, &[&str])] = &[ + ("-Zindirect-branch-cs-prefix", &["yes"]), + ("-Zretpoline-external-thunk", &["yes"]), + ("-Zretpoline", &["yes"]), + ("-Zsanitizer", &["dataflow", "memory", "safestack", "thread", "cfi", "kcfi"]), + ("-Cforce-frame-pointers", &["non-leaf"]), + ("-Cpanic", &["immediate-abort"]), + ("-Zdump-dep-graph", &["yes"]), + ("-Zsanitizer-cfi-canonical-jump-tables", &["no"]), + ("-Zsanitizer-cfi-diag", &["yes"]), + ("-Zsanitizer-cfi-generalize-pointers", &["yes"]), + ("-Zsanitizer-cfi-minimal-runtime", &["yes"]), + ("-Zsanitizer-cfi-normalize-integers", &["yes"]), + ("-Zsanitizer-cfi-recover", &["yes"]), + ("-Zsanitizer-kcfi-arity", &["yes"]), + ("-Zsplit-lto-unit", &["yes"]), + ("-Zvirtual-function-elimination", &["yes"]), +]; +const NEEDS: &[Need] = &[ + ("-Cembed-bitcode", "no", &["-Clto"], r#"[Clto] IN {"absent","no","off"}"#), + ("-Zsplit-lto-unit", "yes", &["-Clto"], r#"[Clto] IN {"yes","on","thin","fat"}"#), + ("-Zvirtual-function-elimination", "yes", &["-Clto"], r#"[Clto] IN {"yes","on","fat"}"#), + ("-Zsanitizer", "cfi", &["-Clto", "-Ccodegen-units"], r#"[Clto] IN {"yes","on","fat"} AND [Ccodegen_units] = "1""#), + ("-Zsanitizer", "kcfi", &["-Cpanic"], r#"[Cpanic] = "abort""#), + ("-Cforce-frame-pointers", "non-leaf", &["-Zunstable-options"], r#"[Zunstable_options] = "present""#), + ("-Cpanic", "immediate-abort", &["-Zunstable-options"], r#"[Zunstable_options] = "present""#), + ("-Zdump-dep-graph", "yes", &["-Zquery-dep-graph"], r#"[Zquery_dep_graph] = "yes""#), + ("-Zsanitizer-cfi-diag", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-recover", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-minimal-runtime", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-canonical-jump-tables", "no", &["-Zsanitizer"], r#"[Zsanitizer] = "cfi""#), + ("-Zsanitizer-cfi-generalize-pointers", "yes", &["-Zsanitizer"], r#"[Zsanitizer] IN {"cfi","kcfi"}"#), + ("-Zsanitizer-cfi-normalize-integers", "yes", &["-Zsanitizer"], r#"[Zsanitizer] IN {"cfi","kcfi"}"#), + ("-Zsanitizer-kcfi-arity", "yes", &["-Zsanitizer"], r#"[Zsanitizer] = "kcfi""#), + ( + "-Zsanitizer-cfi-minimal-runtime", + "yes", + &["-Zsanitizer-cfi-recover", "-Zsanitizer-cfi-diag"], + r#"([Zsanitizer_cfi_recover] = "yes" OR [Zsanitizer_cfi_diag] = "yes")"#, + ), +]; + +/// One -C or -Z option, as flag-universe writes it to options.json. +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct Opt { + pub flag: String, + pub name: String, + pub parser: String, + pub tracking: String, + /// None: an option without a value. + pub values: Vec>, + pub free: bool, +} + +impl Opt { + pub fn full(&self) -> String { + format!("{}{}", self.flag, self.name) + } +} + +/// One value tried alone, as flag-universe writes it to singles.json. +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct Single { + pub arg: String, + pub option: String, + pub value: Option, + pub ok: bool, + pub warn: bool, + pub msg: String, +} + +static NON_WORD: LazyLock = LazyLock::new(|| Regex::new(r"[^A-Za-z0-9]").unwrap()); +static PARAM: LazyLock = LazyLock::new(|| Regex::new(r"\[(\w+)\]").unwrap()); + +/// An option's PICT parameter name. +pub fn pname(opt: &str) -> String { + NON_WORD.replace_all(opt.trim_start_matches('-'), "_").into_owned() +} + +/// options.json of a flag-universe directory, by PICT parameter name. +pub fn options(dir: &Path) -> anyhow::Result> { + let list: Vec = serde_json::from_str(&std::fs::read_to_string(dir.join("options.json"))?)?; + Ok(list.into_iter().map(|o| (pname(&o.full()), o)).collect()) +} + +/// A PICT table: rows of (parameter, value) in column order. +pub fn table(path: &Path) -> anyhow::Result>> { + let text = std::fs::read_to_string(path)?; + let mut lines = text.lines(); + let header: Vec<&str> = lines.next().unwrap_or("").split('\t').collect(); + Ok(lines + .filter(|l| !l.is_empty()) + .map(|l| header.iter().zip(l.split('\t')).map(|(h, v)| (h.to_string(), v.to_string())).collect()) + .collect()) +} + +/// The value of `column` in a row. +pub fn get<'a>(row: &'a [(String, String)], column: &str) -> Option<&'a str> { + row.iter().find(|(k, _)| k == column).map(|(_, v)| v.as_str()) +} + +/// FLAG_BASE and a row's options; with `side` ("A" or "B"), only that side's columns. +pub fn row_flags(row: &[(String, String)], side: Option<&str>, opts: &BTreeMap) -> Vec { + let mut out = vec![FLAG_BASE.to_owned()]; + for (k, v) in row { + if v == "absent" { + continue; + } + let key = match side { + Some(s) => match k.strip_prefix(s).and_then(|r| r.strip_prefix('_')) { + Some(rest) => rest, + None => continue, + }, + None => k.as_str(), + }; + let Some(o) = opts.get(key) else { continue }; + let f = o.full(); + out.push(if v == "present" { f } else { format!("{f}={}", v.replace(';', ",")) }); + } + out +} + +/// A JSON object whose keys keep their order (serde_json's own map sorts them). +pub struct OrderedMap<'a, V>(pub &'a [(String, V)]); + +impl Serialize for OrderedMap<'_, V> { + fn serialize(&self, s: S) -> Result { + use serde::ser::SerializeMap; + let mut m = s.serialize_map(Some(self.0.len()))?; + for (k, v) in self.0 { + m.serialize_entry(k, v)?; + } + m.end() + } +} + +pub fn run(args: Args) -> anyhow::Result { + let known = !args.allow_known; + let opts: BTreeMap = { + let list: Vec = serde_json::from_str(&std::fs::read_to_string(args.work.join("options.json"))?)?; + list.into_iter().map(|o| (o.full(), o)).collect() + }; + let singles: Vec = serde_json::from_str(&std::fs::read_to_string(args.work.join("singles.json"))?)?; + let mut domains: BTreeMap> = BTreeMap::new(); + for s in singles.iter().filter(|s| s.ok && !STOP.contains(&s.option.as_str())) { + domains.entry(s.option.clone()).or_default().push(s.value.clone().unwrap_or_else(|| "present".into())); + } + for (k, vs) in EXTRA { + let d = domains.entry(k.to_string()).or_default(); + for v in *vs { + if !d.iter().any(|x| x == v) { + d.push(v.to_string()); + } + } + } + for (k, vs) in DROP { + domains.entry(k.to_string()).or_default().retain(|v| !vs.contains(&v.as_str())); + } + if args.cargo { + for (k, vs) in CARGO_DROP { + let d = domains.entry(k.to_string()).or_default(); + match vs { + None => d.clear(), + Some(vs) => d.retain(|v| !vs.contains(&v.as_str())), + } + } + domains.retain(|_, v| !v.is_empty()); + } + let untracked = |k: &str| opts.get(k).is_some_and(|o| o.tracking == "UNTRACKED"); + let keep: Vec = + domains.keys().filter(|k| args.subset == "all" || (args.subset == "untracked") == untracked(k)).cloned().collect(); + let mut cons = Vec::new(); + let needs = NEEDS.iter().chain(if args.cargo { CARGO_NEEDS } else { &[] }).chain(if known { KNOWN_NEEDS } else { &[] }); + for (k, v, need, cond) in needs { + if !keep.iter().any(|x| x == k) { + continue; + } + if need.iter().all(|n| keep.iter().any(|x| x == n)) { + cons.push(format!("IF [{}] = \"{v}\" THEN {cond};", pname(k))); + } else if let Some(d) = domains.get_mut(*k) { + if let Some(i) = d.iter().position(|x| x == v) { + d.remove(i); + } + } + } + let mut params: Vec = keep + .iter() + .map(|k| { + let vals: Vec = std::iter::once("absent".to_owned()).chain(domains[k].iter().map(|v| v.replace(',', ";"))).collect(); + format!("{}: {}", pname(k), vals.join(", ")) + }) + .collect(); + if args.transitions { + params = ["A", "B"].iter().flat_map(|t| params.iter().map(move |p| format!("{t}_{p}"))).collect(); + cons = ["A", "B"] + .iter() + .flat_map(|t| cons.iter().map(move |c| PARAM.replace_all(c, |m: ®ex::Captures| format!("[{t}_{}]", &m[1])).into_owned())) + .collect(); + if known && keep.iter().any(|k| k == "-Zprint-type-sizes") { + // Finding 11 in docs/hunt.md: the rebuild ICEs once -Zprint-type-sizes is dropped. + cons.push(r#"IF [A_Zprint_type_sizes] = "yes" THEN [B_Zprint_type_sizes] = "yes";"#.to_owned()); + } + } + std::fs::write(&args.out, format!("{}\n\n{}\n", params.join("\n"), cons.join("\n")))?; + println!("{} parameters, {} constraints", params.len(), cons.len()); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_rows.rs b/crates/mirth-lab/src/tools/flag_rows.rs new file mode 100644 index 0000000..5d2bccc --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_rows.rs @@ -0,0 +1,71 @@ +//! Compile a trivial crate once per row of a PICT table made from flag-model's model, and count +//! the rows rustc rejects, grouped by first error. +//! +//! Tables from a --transitions model are not supported. + +use std::path::PathBuf; +use std::process::{Command, ExitCode}; +use std::time::Duration; + +use rayon::prelude::*; + +use super::flag_model; +use mirth_lab::rustc::run_command; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// flag-universe's work directory (scratch directories go here too). + work: PathBuf, + table: PathBuf, + rustc: PathBuf, + /// Extra rustc arguments, e.g. --emit=metadata. + #[arg(trailing_var_arg = true, allow_hyphen_values = true)] + extra: Vec, +} + +pub fn run(args: Args) -> anyhow::Result { + let opts = flag_model::options(&args.work)?; + let rows = flag_model::table(&args.table)?; + let pool = rayon::ThreadPoolBuilder::new().num_threads(10).build()?; + let res: Vec<(bool, String, usize)> = pool.install(|| { + rows.par_iter() + .map(|row| { + let d = tempfile::tempdir_in(&args.work).expect("scratch"); + let _ = std::fs::write(d.path().join("lib.rs"), "pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n"); + let a = flag_model::row_flags(row, None, &opts); + let mut cmd = Command::new(&args.rustc); + cmd.args(["--edition", "2021", "--crate-type", "lib"]) + .args(&args.extra) + .arg("-o") + .arg(d.path().join("out")) + .args(&a) + .arg(d.path().join("lib.rs")) + .current_dir(d.path()); + match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => { + let err = f.stderr_text().lines().find(|l| l.starts_with("error")).unwrap_or("").to_owned(); + (f.success(), err, a.len() - 1) + } + Err(e) => (false, e.to_string(), a.len() - 1), + } + }) + .collect() + }); + let bad: Vec<&String> = res.iter().filter(|r| !r.0).map(|r| &r.1).collect(); + let avg = res.iter().map(|r| r.2).sum::() as f64 / res.len().max(1) as f64; + println!("{} rows, {} rejected, {avg:.0} options per row on average", rows.len(), bad.len()); + // Most common first; ties in first-seen order, as Counter.most_common. + let mut counts: Vec<(String, usize)> = Vec::new(); + for e in bad { + let k: String = e.chars().take(110).collect(); + match counts.iter_mut().find(|(x, _)| *x == k) { + Some(c) => c.1 += 1, + None => counts.push((k, 1)), + } + } + counts.sort_by(|a, b| b.1.cmp(&a.1)); + for (e, c) in counts { + println!("{c} {e}"); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/flag_universe.rs b/crates/mirth-lab/src/tools/flag_universe.rs new file mode 100644 index 0000000..b471dc5 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_universe.rs @@ -0,0 +1,370 @@ +//! Enumerate rustc's -C and -Z options, find which values and pairs of values it accepts, and +//! size covering arrays over them. +//! +//! Each option's domain is its absence plus the values worth trying: `yes`/`no` for a boolean, +//! present for an option without a value, the values its parser's description lists for an +//! enumerated one, two samples for a number. Options taking free-form strings, paths or lists +//! are counted but left out. Every value is tried alone on a trivial crate (`--emit=metadata`, +//! so only option checking and a tiny compilation run), then every pair of accepted values of +//! different options. A pair is "rejected" when rustc fails with the pair but accepts each +//! value alone. +//! +//! Writes /options.json (domains), /singles.json, /pairs.json, and prints the +//! sizes of pairwise covering arrays built greedily over the accepted domains. + +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use rand::{Rng as _, SeedableRng}; +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +use mirth_lab::coverage::to_json_indent; +use super::flag_model::{Opt, OrderedMap, Single}; +use mirth_lab::rustc::{run_command, Exit}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rust checkout, for compiler/rustc_session/src/options.rs. + #[arg(long)] + source: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + #[arg(long)] + skip_pairs: bool, +} + +const BOOL: &[&str] = &["parse_bool", "parse_opt_bool"]; +const NO_VALUE: &[&str] = &["parse_no_value"]; +const NUMBER: &[&str] = &["parse_number", "parse_opt_number"]; +const FREE: &[&str] = &[ + "parse_string", "parse_opt_string", "parse_string_push", "parse_opt_pathbuf", "parse_list", "parse_comma_list", + "parse_opt_comma_list", "parse_ignore", "parse_target_feature", "parse_list_with_polarity", "parse_llvm_module_flag", + "parse_patchable_function_entry", "parse_autodiff", "parse_offload", "parse_allow_partial_mitigations", + "parse_deny_partial_mitigations", "parse_rust_version", "parse_unpretty", "parse_passes", "parse_branch_protection", + "parse_instrument_xray", "parse_linker_features", "parse_link_self_contained", "parse_align", "parse_location_detail", + "parse_coverage_options", "parse_codegen_retag_options", +]; + +// Values for options whose parser takes a string or whose description lists no values, picked +// by hand (`rustc --print code-models` etc. for the enumerations). +const SAMPLES: &[(&str, &[&str])] = &[ + ("-Copt-level", &["0", "1", "2", "3", "s", "z"]), + ("-Ccode-model", &["tiny", "small", "kernel", "medium", "large"]), + ("-Crelocation-model", &["static", "pic", "pie", "dynamic-no-pic", "ropi", "rwpi", "ropi-rwpi", "default"]), + ("-Ztls-model", &["global-dynamic", "local-dynamic", "initial-exec", "local-exec", "emulated"]), + ("-Ctarget-cpu", &["generic", "native", "x86-64-v2", "x86-64-v3", "x86-64-v4"]), + ("-Ctarget-feature", &["+avx2", "+avx512f", "-sse4.2", "+crt-static"]), + ("-Ztune-cpu", &["generic", "znver4"]), + ("-Zthreads", &["1", "4"]), + ("-Zlocation-detail", &["none", "file", "line,column"]), + ("-Zmin-function-alignment", &["16", "64"]), + ("-Zpatchable-function-entry", &["4", "4,2"]), + ("-Zmir-enable-passes", &["+Inline", "-GVN", "+DeadStoreElimination-final"]), + ("-Zremap-cwd-prefix", &["/remapped"]), + ("-Zsimulate-remapped-rust-src-base", &["/rustc/simulated"]), + ("-Zhint-msrv", &["1.60.0"]), + ("-Cmetadata", &["mirth"]), + ("-Zinstrument-xray", &["always", "never"]), + ("-Cllvm-args", &["-unroll-threshold=0", "-enable-machine-outliner"]), + ("-Zcrate-attr", &["allow(unused)"]), +]; + +static DESC: LazyLock = + LazyLock::new(|| Regex::new(r#"pub\(crate\) const (parse_\w+): &str =\s*((?:"(?:[^"\\]|\\.)*"\s*)+|parse_\w+|[^;]+);"#).unwrap()); +static PARSER_NAME: LazyLock = LazyLock::new(|| Regex::new(r"^parse_\w+$").unwrap()); +static BACKTICKED: LazyLock = LazyLock::new(|| Regex::new(r"`([^`]+)`").unwrap()); +static OPTION_LINE: LazyLock = LazyLock::new(|| Regex::new(r"^\s*(\w+): .*?, (parse_\w+), \[(\w+)").unwrap()); + +fn parse_options(src: &str) -> anyhow::Result> { + // The parsers' descriptions, joined across lines. + let mut descs: HashMap = DESC.captures_iter(src).map(|c| (c[1].to_owned(), c[2].to_owned())).collect(); + let aliases: Vec<(String, String)> = + descs.iter().filter(|(_, v)| PARSER_NAME.is_match(v.trim())).map(|(k, v)| (k.clone(), v.trim().to_owned())).collect(); + for (k, v) in aliases { + let target = descs.get(&v).cloned().unwrap_or_default(); + descs.insert(k, target); + } + let enum_values = |parser: &str| -> Vec { + let mut out: Vec = Vec::new(); + for c in BACKTICKED.captures_iter(descs.get(parser).map_or("", String::as_str)) { + let v = &c[1]; + if out.iter().any(|x| x == v) || v.contains(' ') || v.contains('<') || v.contains('=') { + continue; + } + out.push(v.to_owned()); + } + out + }; + let mut options = Vec::new(); + for (flag, grp) in [("-C", "CodegenOptions"), ("-Z", "UnstableOptions")] { + let i = src.find(&format!("options! {{\n {grp}")).ok_or_else(|| anyhow::anyhow!("no {grp} in options.rs"))?; + let j = i + src[i..].find("\n}").ok_or_else(|| anyhow::anyhow!("unterminated {grp}"))?; + for line in src[i..j].lines() { + let Some(m) = OPTION_LINE.captures(line) else { continue }; + let (name, parser, tracking) = (m[1].replace('_', "-"), m[2].to_owned(), m[3].to_owned()); + let mut values: Vec> = if BOOL.contains(&parser.as_str()) { + vec![Some("yes".into()), Some("no".into())] + } else if NO_VALUE.contains(&parser.as_str()) { + vec![None] + } else if NUMBER.contains(&parser.as_str()) { + vec![Some("1".into()), Some("16".into())] + } else if FREE.contains(&parser.as_str()) { + vec![] + } else { + enum_values(&parser).into_iter().map(Some).collect() + }; + let full = format!("{flag}{name}"); + if let Some((_, s)) = SAMPLES.iter().find(|(k, _)| *k == full) { + values = s.iter().map(|v| Some(v.to_string())).collect(); + } + let free = values.is_empty(); + options.push(Opt { flag: flag.into(), name, parser, tracking, values, free }); + } + } + Ok(options) +} + +fn arg(o: &Opt, v: &Option) -> String { + match v { + None => o.full(), + Some(v) => format!("{}={v}", o.full()), + } +} + +struct Tried { + ok: bool, + warn: bool, + msg: String, +} + +fn try_args(rustc: &Path, work: &Path, argv: &[&str]) -> Tried { + let d = tempfile::tempdir_in(work).expect("scratch"); + let lib = d.path().join("lib.rs"); + let _ = std::fs::write(&lib, "pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n"); + let mut cmd = Command::new(rustc); + cmd.args(["--edition", "2021", "--crate-type", "lib", "--emit=metadata", "-o"]) + .arg(d.path().join("out.rmeta")) + .args(argv) + .arg(&lib) + .current_dir(d.path()); + match run_command(cmd, Duration::from_secs(60)) { + Ok(f) if f.exit == Exit::Timeout => Tried { ok: false, warn: false, msg: "timeout".into() }, + Ok(f) => { + let err = f.stderr_text(); + let first = err.lines().find(|l| l.starts_with("error") || l.starts_with("warning")).unwrap_or(""); + Tried { ok: f.success(), warn: err.contains("warning"), msg: first.chars().take(200).collect() } + } + Err(e) => Tried { ok: false, warn: false, msg: e.to_string() }, + } +} + +#[derive(Serialize, Deserialize)] +struct Pair { + a: String, + b: String, + ok: bool, + warn: bool, + msg: String, +} + +/// A pairwise covering array, built greedily: each row is chosen among random candidates to +/// cover the most uncovered pairs. Values are indices into each domain; index 0 is the +/// option's absence. +fn covering_array(domains: &[Vec>], forbidden: &HashSet<(Option, Option)>, seed: u64) -> Vec> { + let mut rng = rand::rngs::StdRng::seed_from_u64(seed); + let n = domains.len(); + let bad = |i: usize, a: usize, j: usize, b: usize| forbidden.contains(&(domains[i][a].clone(), domains[j][b].clone())); + let mut uncovered: HashSet<(usize, usize, usize, usize)> = HashSet::new(); + for i in 0..n { + for j in i + 1..n { + for a in 0..domains[i].len() { + for b in 0..domains[j].len() { + if !bad(i, a, j, b) { + uncovered.insert((i, a, j, b)); + } + } + } + } + } + let mut rows = Vec::new(); + while let Some(&target) = uncovered.iter().min() { + let (mut best, mut best_gain) = (Vec::new(), -1i64); + for _ in 0..30 { + let mut row: Vec = domains.iter().map(|d| rng.random_range(..d.len())).collect(); + row[target.0] = target.1; + row[target.2] = target.3; + // repair forbidden pairs by falling back to absence + for i in 0..n { + for j in i + 1..n { + if bad(i, row[i], j, row[j]) { + if j != target.0 && j != target.2 { + row[j] = 0; + } else if i != target.0 && i != target.2 { + row[i] = 0; + } + } + } + } + let mut gain = 0i64; + for i in 0..n { + for j in i + 1..n { + gain += uncovered.contains(&(i, row[i], j, row[j])) as i64; + } + } + if gain > best_gain { + best = row; + best_gain = gain; + } + } + for i in 0..n { + for j in i + 1..n { + uncovered.remove(&(i, best[i], j, best[j])); + } + } + rows.push(best); + } + rows +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let src = std::fs::read_to_string(args.source.join("compiler/rustc_session/src/options.rs"))?; + let options = parse_options(&src)?; + std::fs::write(args.work.join("options.json"), to_json_indent(&options, 1))?; + let walkable: Vec<&Opt> = options.iter().filter(|o| !o.free).collect(); + println!("{} options: {} with enumerable values, {} free-form", options.len(), walkable.len(), options.len() - walkable.len()); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + + let singles_path = args.work.join("singles.json"); + let mut singles: Vec = + if singles_path.exists() { serde_json::from_str(&std::fs::read_to_string(&singles_path)?)? } else { Vec::new() }; + // Try the values not tried yet (all of them on a first run). + let done: HashSet = singles.iter().map(|s| s.arg.clone()).collect(); + let jobs: Vec<(&Opt, &Option)> = + walkable.iter().flat_map(|o| o.values.iter().map(move |v| (*o, v))).filter(|(o, v)| !done.contains(&arg(o, v))).collect(); + if !jobs.is_empty() { + let results: Vec = pool.install(|| jobs.par_iter().map(|(o, v)| try_args(&args.rustc, &args.work, &[&arg(o, v)])).collect()); + for ((o, v), r) in jobs.iter().zip(results) { + singles.push(Single { arg: arg(o, v), option: o.full(), value: (*v).clone(), ok: r.ok, warn: r.warn, msg: r.msg }); + } + std::fs::write(&singles_path, to_json_indent(&singles, 1))?; + } + let mut accepted: BTreeMap> = BTreeMap::new(); + for s in singles.iter().filter(|s| s.ok) { + accepted.entry(s.option.clone()).or_default().push(s.arg.clone()); + } + println!( + "{} single values tried, {} accepted ({} options with at least one)", + singles.len(), + accepted.values().map(Vec::len).sum::(), + accepted.len() + ); + + let mut rejected_pairs: HashSet<(String, String)> = HashSet::new(); + if !args.skip_pairs { + let pairs_path = args.work.join("pairs.json"); + let pairs: Vec = if pairs_path.exists() { + serde_json::from_str(&std::fs::read_to_string(&pairs_path)?)? + } else { + let opts: Vec<&String> = accepted.keys().collect(); + let mut jobs: Vec<(String, String)> = Vec::new(); + for (i, x) in opts.iter().enumerate() { + for y in &opts[i + 1..] { + for a in &accepted[*x] { + for b in &accepted[*y] { + jobs.push((a.clone(), b.clone())); + } + } + } + } + // A value rejected alone may need another option: try it with every accepted + // value of every other option. + for r in singles.iter().filter(|s| !s.ok) { + for y in opts.iter().filter(|y| ***y != r.option) { + for b in &accepted[*y] { + jobs.push((r.arg.clone(), b.clone())); + } + } + } + println!("{} pairs to try", jobs.len()); + let results: Vec = pool.install(|| jobs.par_iter().map(|(a, b)| try_args(&args.rustc, &args.work, &[a, b])).collect()); + let pairs: Vec = + jobs.into_iter().zip(results).map(|((a, b), r)| Pair { a, b, ok: r.ok, warn: r.warn, msg: r.msg }).collect(); + std::fs::write(&pairs_path, to_json_indent(&pairs, 1))?; + pairs + }; + let alone: HashSet<&String> = singles.iter().filter(|s| s.ok).map(|s| &s.arg).collect(); + rejected_pairs = + pairs.iter().filter(|x| !x.ok && alone.contains(&x.a) && alone.contains(&x.b)).map(|x| (x.a.clone(), x.b.clone())).collect(); + // Insertion-ordered, as the Python dict was. + let mut requires: Vec<(String, Vec)> = Vec::new(); + for x in pairs.iter().filter(|x| x.ok && !alone.contains(&x.a)) { + match requires.iter_mut().find(|(k, _)| *k == x.a) { + Some((_, v)) => v.push(x.b.clone()), + None => requires.push((x.a.clone(), vec![x.b.clone()])), + } + } + println!( + "{} pairs tried; {} pairs of values accepted alone are rejected together; {} values rejected alone are accepted with another option", + pairs.len(), + rejected_pairs.len(), + requires.len() + ); + std::fs::write(args.work.join("requires.json"), to_json_indent(&OrderedMap(&requires), 1))?; + } + + let forbidden: HashSet<(Option, Option)> = rejected_pairs + .iter() + .flat_map(|(a, b)| [(Some(a.clone()), Some(b.clone())), (Some(b.clone()), Some(a.clone()))]) + .collect(); + let byopt: HashMap = options.iter().map(|o| (o.full(), o)).collect(); + let mut results: Vec<(String, usize)> = Vec::new(); + for (label, pred) in [ + ("untracked", (|o: &Opt| o.tracking == "UNTRACKED") as fn(&Opt) -> bool), + ("tracked", |o: &Opt| o.tracking != "UNTRACKED"), + ("all", |_: &Opt| true), + ] { + let opts: Vec<&String> = accepted.keys().filter(|k| byopt.get(*k).is_some_and(|o| pred(o))).collect(); + if opts.len() <= 1 { + continue; + } + let domains: Vec>> = + opts.iter().map(|o| std::iter::once(None).chain(accepted[*o].iter().cloned().map(Some)).collect()).collect(); + let mut sizes: Vec = domains.iter().map(Vec::len).collect(); + sizes.sort_by(|a, b| b.cmp(a)); + let total: f64 = sizes.iter().map(|&s| s as f64).product(); + let lower = if sizes.len() > 1 { sizes[0] * sizes[1] } else { sizes[0] }; + let rows = covering_array(&domains, &forbidden, 0); + println!( + "{label}: {} options, all combinations {}, pairwise covering array {} rows (lower bound {lower})", + opts.len(), + sci(total), + rows.len() + ); + results.push((label.to_owned(), rows.len())); + } + let covering: Vec = results.iter().map(|(k, v)| format!("\"{k}\": {v}")).collect(); + std::fs::write(args.work.join("covering.json"), format!("{{{}}}", covering.join(", ")))?; + Ok(ExitCode::SUCCESS) +} + +/// Python's `{:.3e}`. +fn sci(x: f64) -> String { + let s = format!("{x:.3e}"); + match s.split_once('e') { + Some((m, e)) => { + let e: i32 = e.parse().unwrap_or(0); + format!("{m}e{}{:02}", if e < 0 { '-' } else { '+' }, e.abs()) + } + None => s, + } +} diff --git a/crates/mirth-lab/src/tools/flag_walk.rs b/crates/mirth-lab/src/tools/flag_walk.rs new file mode 100644 index 0000000..b47a7e3 --- /dev/null +++ b/crates/mirth-lab/src/tools/flag_walk.rs @@ -0,0 +1,541 @@ +//! Walk option transitions on a fixture: for each row of a PICT table made from a +//! `flag-model --transitions` model, build the fixture clean with the A_ options, rebuild it +//! incrementally with the B_ options, build it clean with the B_ options, and compare the +//! rebuild with the clean build (metadata, object code, binary, diagnostics, the binary's +//! output), as the fuzzer does after an edit. +//! +//! Rows change options only, unless --edits asks for random source edits between A and B. +//! +//! The options go in RUSTFLAGS with `--target` set, so they apply to the fixture's crates but +//! not to its build scripts and proc macros. RUSTC_VERIFY_REUSE and RUSTC_REPORT_UNTRACKED are +//! set, for a compiler with mirth's local patches. +//! +//! Writes /results.jsonl (one line per row) and /findings// for each row whose +//! rebuild differs from the clean build, or which crashed. +//! +//! To stay at the frontier: with --pause-on-finding the walk stops taking rows at the first +//! finding not marked known and writes /PAUSED. Patch the compiler, then run the same +//! command with --rustc --recheck: the rows with findings run again first, then the +//! rows not yet walked. A rerun never repeats rows that are done. + +use std::collections::BTreeMap; +use std::io::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Mutex; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; +use rand::rngs::StdRng; +use rand::{Rng as _, SeedableRng}; +use serde_json::Value; + +use mirth_lab::coverage::to_json_indent; +use super::flag_model::{self, Opt}; +use mirth_lab::artifacts::{self, Collected}; +use mirth_lab::mutations; +use mirth_lab::rustc::{run_command, Exit}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + /// flag-universe's work directory. + #[arg(long)] + flags: PathBuf, + #[arg(long)] + table: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + workers: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "x86_64-unknown-linux-gnu")] + target: String, + #[arg(long, default_value_t = 600)] + timeout: u64, + /// a:b, a slice of the table. + #[arg(long, default_value = "")] + rows: String, + /// Random source edits between A and B (the fuzzer's). + #[arg(long, default_value_t = 0)] + edits: usize, + #[arg(long, default_value_t = 0)] + seed: u64, + /// Stop taking rows at the first finding not marked known; rerun to resume. + #[arg(long)] + pause_on_finding: bool, + /// On resume, run the rows that had findings again first (after patching rustc). + #[arg(long)] + recheck: bool, + /// Clean rebuilds before a difference counts as reuse. + #[arg(long, default_value_t = 12)] + p5_builds: usize, +} + +/// Copy a fixture, leaving out build output and edit scratch at any depth. +pub fn copy_fixture(from: &Path, to: &Path) -> std::io::Result<()> { + let skip = |n: &std::ffi::OsStr| n == "target" || n == "edits" || n == "edit"; + for e in walkdir::WalkDir::new(from).follow_links(true).into_iter().filter_entry(|e| e.depth() == 0 || !skip(e.file_name())) { + let e = e?; + let dest = to.join(e.path().strip_prefix(from).expect("under the fixture")); + if e.file_type().is_dir() { + std::fs::create_dir_all(&dest)?; + } else { + std::fs::copy(e.path(), &dest)?; + } + } + Ok(()) +} + +/// Python's slice `a:b` of 0..n. +/// Rows `lo:hi` (Python slice bounds, either may be empty or negative); empty means all. +pub fn slice(spec: &str, n: usize) -> anyhow::Result> { + if spec.is_empty() { + return Ok((0..n).collect()); + } + let Some((lo, hi)) = spec.split_once(':') else { anyhow::bail!("--rows takes lo:hi, not {spec:?}") }; + let at = |s: &str, default: usize| -> anyhow::Result { + if s.is_empty() { + return Ok(default); + } + let v: i64 = s.parse().map_err(|_| anyhow::anyhow!("--rows takes lo:hi, not {spec:?}"))?; + Ok(if v < 0 { (n as i64 + v).max(0) as usize } else { (v as usize).min(n) }) + }; + Ok((at(lo, 0)?..at(hi, n)?).collect()) +} + +/// Python's repr of a string, as the findings have always shown texts. +pub fn py_repr(s: &str) -> String { + let q = if s.contains('\'') && !s.contains('"') { '"' } else { '\'' }; + let mut out = String::from(q); + for c in s.chars() { + match c { + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\r' => out.push_str("\\r"), + '\t' => out.push_str("\\t"), + c if c == q => { + out.push('\\'); + out.push(c); + } + c if (c as u32) < 0x20 || c as u32 == 0x7f => out.push_str(&format!("\\x{:02x}", c as u32)), + c => out.push(c), + } + } + out.push(q); + out +} + +/// Python's repr of a JSON value (lists of strings and numbers, None). +pub fn py_value(v: &Value) -> String { + match v { + Value::Null => "None".into(), + Value::Bool(b) => if *b { "True" } else { "False" }.into(), + Value::String(s) => py_repr(s), + Value::Array(a) => format!("[{}]", a.iter().map(py_value).collect::>().join(", ")), + other => other.to_string(), + } +} + +fn py_list(v: &[String]) -> String { + format!("[{}]", v.iter().map(|s| py_repr(s)).collect::>().join(", ")) +} + +#[derive(Serialize, Deserialize, Clone, Debug)] +pub struct RowResult { + pub row: usize, + #[serde(rename = "A")] + pub a: Vec, + #[serde(rename = "B")] + pub b: Vec, + #[serde(rename = "A_ok")] + pub a_ok: bool, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub diff: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub inc_ok: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub clean_ok: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub reuse: Option>, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub untracked: Option>, + #[serde(default)] + pub error: String, + #[serde(default)] + pub errors: Vec, + #[serde(default)] + pub findings: Vec, + #[serde(default)] + pub rustc: String, +} + +struct Build { + ok: bool, + log: String, + ice: bool, + reuse: Vec, + untracked: Vec, + error: String, + errors: Vec, + art: Option, + exe: Option, +} + +struct Walk<'a> { + args: &'a Args, + opts: BTreeMap, + fixture: PathBuf, + work: PathBuf, + results: Mutex<()>, +} + +fn tail(s: &str, n: usize) -> &str { + let mut i = s.len().saturating_sub(n); + while !s.is_char_boundary(i) { + i += 1; + } + &s[i..] +} + +impl Walk<'_> { + /// RUSTFLAGS for one side of a row. Cargo passes `-Cembed-bitcode=no` unless its profile + /// asks for LTO, and the profile's LTO reaches only the final artifacts, so `-Clto` goes in + /// RUSTFLAGS with `-Cembed-bitcode=yes` after Cargo's flag. + fn flags(&self, row: &[(String, String)], side: &str) -> Vec { + let mut out = flag_model::row_flags(row, Some(side), &self.opts); + let lto = flag_model::get(row, &format!("{side}_Clto")); + if matches!(lto, Some("yes" | "on" | "thin" | "fat")) && flag_model::get(row, &format!("{side}_Cembed_bitcode")) == Some("absent") { + out.push("-Cembed-bitcode=yes".into()); + } + out + } + + fn build(&self, src: &Path, target: &Path, rustflags: &[String]) -> Build { + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", self.args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target", &self.args.target, "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &self.args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("CARGO_TERM_COLOR", "never") + .env("RUSTFLAGS", rustflags.join(" ")) + .env("RUSTC_VERIFY_REUSE", "1") + .env("RUSTC_REPORT_UNTRACKED", "1"); + let (ok, out, log) = match run_command(cmd, Duration::from_secs(self.args.timeout)) { + Ok(f) => { + let mut log = f.stderr_text(); + if f.exit == Exit::Timeout { + log.push_str(&format!("\nkilled after {}s", self.args.timeout)); + } + (f.success(), f.stdout_text(), log) + } + Err(e) => (false, String::new(), e.to_string()), + }; + let name = self.fixture.file_name().map(|n| n.to_string_lossy().into_owned()).unwrap_or_default(); + let mut exe = None; + for line in out.lines() { + let Ok(msg) = serde_json::from_str::(line) else { continue }; + if msg["reason"] == "compiler-artifact" && msg["target"]["name"] == name.as_str() { + if let Some(e) = msg["executable"].as_str().filter(|e| !e.is_empty()) { + exe = Some(e.to_owned()); + } + } + } + let mut reuse: Vec = log + .lines() + .filter(|l| l.starts_with("rustc-verify-reuse:")) + .map(|l| l.split_once(':').unwrap().1.trim().chars().take(200).collect()) + .collect(); + reuse.sort(); + reuse.dedup(); + let mut untracked: Vec = log.lines().filter(|l| l.starts_with("rustc-untracked-read:")).map(|l| l.trim().to_owned()).collect(); + untracked.sort(); + untracked.dedup(); + Build { + ok, + ice: log.contains("internal compiler error") || log.contains("the compiler unexpectedly panicked") || log.contains("rustc interrupted by SIG"), + reuse, + untracked, + error: log.lines().find(|l| l.starts_with("error")).unwrap_or("").to_owned(), + errors: log + .lines() + .filter(|l| (l.starts_with("error") || l.starts_with("rustc-LLVM ERROR") || l.starts_with("LLVM ERROR")) && !l.contains("could not compile")) + .take(6) + .map(|l| l.chars().take(300).collect()) + .collect(), + art: ok.then(|| artifacts::collect(&out, target)), + exe, + log, + } + } + + /// Apply `n` random edits to the fixture's sources; returns the unified diff. + fn edit(&self, src: &Path, rng: &mut StdRng, n: usize) -> String { + let mut diff = String::new(); + for k in 0..n { + let mut paths: Vec = walkdir::WalkDir::new(src) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && !p.strip_prefix(src).unwrap().components().any(|c| c.as_os_str() == "target")) + .collect(); + paths.sort(); + if paths.is_empty() { + break; + } + for _ in 0..20 { + let path = &paths[rng.random_range(..paths.len())]; + let (name, f) = mutations::pick(rng); + if (name == "int_literal" || name == "str_literal") && path.file_name().is_some_and(|n| n == "build.rs") { + continue; // stale OUT_DIR files, or a build script that loops + } + let Ok(old) = std::fs::read_to_string(path) else { continue }; + match f(&old, rng, k) { + Some(new) if new != old => { + let _ = std::fs::write(path, &new); + let rel = path.strip_prefix(src).unwrap().display().to_string(); + let d = similar::TextDiff::from_lines(&old, &new); + diff += &d.unified_diff().header(&format!("a/{rel}"), &format!("b/{rel}")).to_string(); + break; + } + _ => {} + } + } + } + diff + } + + fn run_exe(exe: Option<&str>) -> Value { + let Some(exe) = exe else { return Value::Null }; + match run_command(Command::new(exe), Duration::from_secs(30)) { + Ok(f) if f.exit == Exit::Timeout => serde_json::json!(["timeout", ""]), + Ok(f) => { + let code = match f.exit { + Exit::Code(c) => c, + Exit::Signal(s) => -s, + Exit::Timeout => unreachable!(), + }; + serde_json::json!([code, tail(&f.stdout_text(), 2000)]) + } + Err(_) => Value::Null, + } + } + + fn write_finding(&self, i: usize, res: &RowResult, findings: &[String], logs: &[(&str, &str)]) { + let d = self.work.join("findings").join(format!("r{i}")); + let _ = std::fs::create_dir_all(&d); + let mut r = res.clone(); + r.findings = findings.to_vec(); + let _ = std::fs::write(d.join("row.json"), to_json_indent(&r, 1)); + if let Some(diff) = res.diff.as_ref().filter(|d| !d.is_empty()) { + let _ = std::fs::write(d.join("edit.diff"), diff); + } + for (name, log) in logs { + let _ = std::fs::write(d.join(name), tail(log, 20000)); + } + } + + fn walk(&self, i: usize, row: &[(String, String)]) { + if self.work.join("PAUSED").exists() || self.work.join("STOP").exists() { + return; + } + let home = self.work.join(format!("r{i}")); + let _ = std::fs::remove_dir_all(&home); + let (src, target, inc_target) = (home.join("src"), home.join("target"), home.join("target-inc")); + if let Err(e) = copy_fixture(&self.fixture, &src) { + eprintln!("row {i}: cannot copy the fixture: {e}"); + return; + } + let (a, b) = (self.flags(row, "A"), self.flags(row, "B")); + let mut res = RowResult { + row: i, + a: a[1..].to_vec(), + b: b[1..].to_vec(), + a_ok: false, + diff: None, + inc_ok: None, + clean_ok: None, + reuse: None, + untracked: None, + error: String::new(), + errors: vec![], + findings: vec![], + rustc: String::new(), + }; + let first = self.build(&src, &target, &a); + res.a_ok = first.ok; + let mut findings: Vec = Vec::new(); + if first.ice { + findings.push("ICE in clean A".into()); + } + if first.ok { + let mut rng = StdRng::seed_from_u64(self.args.seed.wrapping_mul(1_000_003).wrapping_add(i as u64)); + res.diff = Some(self.edit(&src, &mut rng, self.args.edits)); + let inc = self.build(&src, &target, &b); + let _ = std::fs::rename(&target, &inc_target); + let clean = self.build(&src, &target, &b); + res.inc_ok = Some(inc.ok); + res.clean_ok = Some(clean.ok); + res.reuse = Some(inc.reuse.clone()); + res.untracked = Some(inc.untracked.clone()); + if inc.ice { + findings.push("ICE in rebuild B".into()); + } + if clean.ice { + findings.push("ICE in clean B".into()); + } + if inc.ok != clean.ok { + findings.push(format!( + "split: rebuild {}, clean {}", + if inc.ok { "ok" } else { "failed" }, + if clean.ok { "ok" } else { "failed" } + )); + } + if inc.ok && clean.ok { + let (ia, ca) = (inc.art.as_ref().unwrap(), clean.art.as_ref().unwrap()); + let mut diff = artifacts::compare(ia, ca); + if matches!(flag_model::get(row, "B_Csplit_debuginfo"), Some("packed" | "unpacked")) { + // Objects and binary name .dwo files by session (DW_AT_GNU_dwo_name, and + // the dwo_id hashed from it), so two clean builds differ too. + diff.remove("rlib"); + diff.remove("exe"); + } + let inc_exe = inc.exe.as_ref().map(|e| e.replace(&*target.to_string_lossy(), &inc_target.to_string_lossy())); + let ra = Self::run_exe(inc_exe.as_deref()); + let rb = Self::run_exe(clean.exe.as_deref()); + if ra != rb { + findings.push(format!("run: {} vs {}", py_value(&ra), py_value(&rb))); + } + if !diff.is_empty() { + // Clean builds may differ among themselves (P5), sometimes only one time in + // five: build clean again up to --p5-builds times before calling it reuse. + let mut p5 = BTreeMap::new(); + for _ in 0..self.args.p5_builds { + let _ = std::fs::remove_dir_all(&target); + let again = self.build(&src, &target, &b); + if let Some(art) = &again.art { + p5 = artifacts::compare(ca, art); + } + if !p5.is_empty() || !again.ok { + break; + } + } + let kind = if p5.is_empty() { "" } else { "P5 " }; + for (k, v) in &diff { + findings.push(format!("{kind}{k}: {}", py_list(&v[..v.len().min(5)]))); + } + } + } + res.error = if inc.error.is_empty() { clean.error.clone() } else { inc.error.clone() }; + res.errors = if inc.errors.is_empty() { clean.errors.clone() } else { inc.errors.clone() }; + if !findings.is_empty() { + self.write_finding(i, &res, &findings, &[("inc.log", &inc.log), ("clean.log", &clean.log)]); + } + } else { + res.error = first.error.clone(); + res.errors = first.errors.clone(); + if !findings.is_empty() { + self.write_finding(i, &res, &findings, &[("a.log", &first.log)]); + } + } + res.findings = findings.clone(); + res.rustc = self.args.rustc.clone(); + let new: Vec<&String> = findings.iter().filter(|f| !f.starts_with("known")).collect(); + if !new.is_empty() && self.args.pause_on_finding { + #[derive(Serialize)] + struct Paused<'a> { + row: usize, + findings: Vec<&'a String>, + } + let _ = std::fs::write(self.work.join("PAUSED"), to_json_indent(&Paused { row: i, findings: new }, 1)); + } + let _ = std::fs::remove_dir_all(&home); + { + let _lock = self.results.lock().unwrap(); + if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open(self.work.join("results.jsonl")) { + let _ = writeln!(f, "{}", serde_json::to_string(&res).unwrap()); + } + } + let mut line = format!("row {i}: A {}", if res.a_ok { "ok" } else { "failed" }); + if res.a_ok { + line += &format!(", rebuild {}", if res.inc_ok == Some(true) { "ok" } else { "failed" }); + } + if !findings.is_empty() { + line += &format!("; {}", py_list(&findings)); + } + if !res.error.is_empty() { + line += &format!("; {}", res.error.chars().take(100).collect::()); + } + println!("{line}"); + } +} + +/// The last result of each row, from /results.jsonl. +fn latest(work: &Path) -> BTreeMap { + let mut out = BTreeMap::new(); + for line in std::fs::read_to_string(work.join("results.jsonl")).unwrap_or_default().lines() { + if let Ok(r) = serde_json::from_str::(line) { + out.insert(r.row, r); + } + } + out +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let fixture = std::fs::canonicalize(&args.fixture)?; + let _ = std::fs::remove_file(work.join("PAUSED")); + let rows = flag_model::table(&args.table)?; + let idx = slice(&args.rows, rows.len())?; + // Resume: rows with a result are done, except, with --recheck, those with findings, which + // run first. + let done = latest(&work); + let again: Vec = + if args.recheck { idx.iter().copied().filter(|i| done.get(i).is_some_and(|r| !r.findings.is_empty())).collect() } else { vec![] }; + let mut todo = again.clone(); + todo.extend(idx.iter().copied().filter(|i| !done.contains_key(i))); + if !again.is_empty() { + println!("rechecking rows {:?}", again); + } + let walk = Walk { opts: flag_model::options(&args.flags)?, fixture, work: work.clone(), results: Mutex::new(()), args: &args }; + // Rows in table order, a worker taking the next one when it is free. + let next = AtomicUsize::new(0); + std::thread::scope(|s| { + for _ in 0..args.workers.max(1) { + s.spawn(|| { + loop { + let k = next.fetch_add(1, Ordering::SeqCst); + let Some(&i) = todo.get(k) else { break }; + walk.walk(i, &rows[i]); + } + }); + } + }); + let results: Vec = latest(&work).into_values().collect(); + let mut summary = format!( + "{{\"rows\": {}, \"done\": {}, \"A ok\": {}, \"compared\": {}, \"findings\": {}", + rows.len(), + results.len(), + results.iter().filter(|r| r.a_ok).count(), + results.iter().filter(|r| r.inc_ok == Some(true) && r.clean_ok == Some(true)).count(), + results.iter().filter(|r| !r.findings.is_empty()).count() + ); + if let Ok(p) = std::fs::read_to_string(work.join("PAUSED")) { + if let Ok(v) = serde_json::from_str::(&p) { + let f: Vec = v["findings"].as_array().into_iter().flatten().map(|x| x.to_string()).collect(); + summary += &format!(", \"paused\": {{\"row\": {}, \"findings\": [{}]}}", v["row"], f.join(", ")); + } + } + println!("{summary}}}"); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/fuzz.rs b/crates/mirth-lab/src/tools/fuzz.rs new file mode 100644 index 0000000..d8fa91a --- /dev/null +++ b/crates/mirth-lab/src/tools/fuzz.rs @@ -0,0 +1,534 @@ +//! Make random edits to a fixture and check each incremental rebuild. +//! +//! Each worker keeps one copy of the fixture and repeats: +//! +//! 1. apply a random mechanical edit (`mutations`: a comment, a moved item, a changed +//! literal, a new function, ...); +//! 2. rebuild incrementally; if the edit does not compile, revert it (the next build then also +//! exercises recovery from a failed session); +//! 3. build the same source from scratch, at the same path; +//! 4. compare: +//! P6 every .rmeta Cargo reports for a workspace member +//! rlib every rlib's members, object code included +//! exe the binary's bytes +//! diag the diagnostics each crate printed +//! run the binaries' output and exit status +//! reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, +//! docs/hunt/verify-reuse.patch) found nothing stale +//! P5 when anything differs, clean builds are made again; if two clean builds +//! differ, that is reported instead (nondeterminism) +//! ICE neither build crashed the compiler +//! split both builds succeed or both fail +//! +//! Every --reset kept edits the worker starts again from the pristine fixture. A finding keeps +//! the edits since the last reset (history.json, which `fuzz-replay` replays), and both builds' +//! differing files and logs. +//! +//! Stop it early by creating /STOP. Progress is in /stats-w.json. With +//! --pause-on-finding all workers stop at the first finding (/PAUSED says which); patch +//! rustc and run again with the patched compiler. +//! +//! Seeds name the same edit sequence only within this implementation: the Python fuzzer drew +//! from Python's generator. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use rand::SeedableRng; +use rand::rngs::StdRng; +use rand::seq::IndexedRandom; +use regex::Regex; +use serde_json::{Value, json}; + +use mirth_lab::artifacts::{self, Collected}; +use mirth_lab::cargo::{self, copy_tree, messages, relative, tail}; +use mirth_lab::mutations; + +#[derive(clap::Args, Debug, Clone)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + workers: u64, + /// Per worker. + #[arg(long, default_value_t = 1_000_000_000)] + edits: u64, + #[arg(long, default_value_t = 40)] + reset: usize, + /// Findings kept per kind. + #[arg(long, default_value_t = 5)] + keep: usize, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "-Zincremental-verify-ich", allow_hyphen_values = true)] + rustflags: String, + #[arg(long, default_value_t = 0)] + seed: u64, + /// Seconds before a build counts as hung. + #[arg(long, default_value_t = 180)] + timeout: u64, + /// Clean builds made again when anything differs, to tell nondeterminism from P6. + #[arg(long, default_value_t = 12)] + p5_builds: usize, + /// Stop when the disk has less free. + #[arg(long, default_value_t = 20)] + min_free_gb: u64, + /// cargo check instead of cargo build: metadata only, no code or binaries. + #[arg(long)] + check: bool, + /// Do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch). + #[arg(long)] + no_verify_reuse: bool, + /// Pass --target to cargo, so RUSTFLAGS skip build scripts and proc macros. + #[arg(long)] + target: Option, + /// Stop all workers at the first finding (writes /PAUSED); patch rustc and run again. + #[arg(long)] + pause_on_finding: bool, +} + +struct Ctx { + args: Args, + work: PathBuf, + fixture: PathBuf, + bin: String, + /// Kinds not compared: metadata is compared separately; with split debuginfo, objects and + /// binaries name .dwo files by session, so two clean builds differ too. + skip: Vec<&'static str>, +} + +struct Build { + ok: bool, + ice: bool, + hang: bool, + reuse: Vec, + log: String, + rmetas: BTreeMap>, + exe: Option, + art: Collected, +} + +impl Ctx { + fn build(&self, src: &Path, target: &Path) -> Build { + let a = &self.args; + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", a.toolchain)) + .args([if a.check { "check" } else { "build" }, "--workspace", "--offline", "-j", "4", "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &a.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("RUSTFLAGS", &a.rustflags) + .env("CARGO_TERM_COLOR", "never"); + if let Some(t) = &a.target { + cmd.args(["--target", t]); + } + if !a.no_verify_reuse { + cmd.env("RUSTC_VERIFY_REUSE", "1").env("RUSTC_REPORT_UNTRACKED", "1"); + } + let r = cargo::run_group(cmd, Some(Duration::from_secs(a.timeout))).unwrap_or_else(|e| cargo::Run { + ok: false, + stdout: String::new(), + stderr: format!("cargo did not start: {e}"), + hang: false, + }); + let (mut rmetas, mut exe) = (BTreeMap::new(), None); + for msg in messages(&r.stdout) { + if msg.reason != "compiler-artifact" { + continue; + } + for f in &msg.filenames { + if f.ends_with(".rmeta") { + rmetas.insert(relative(f, target), std::fs::read(f).unwrap_or_default()); + } else if f.ends_with(".so") && msg.target.as_ref().is_some_and(|t| t.kind.iter().any(|k| k == "proc-macro")) { + // A proc macro's metadata is in the .rustc section of its shared library. + let section = Command::new("objcopy") + .args(["--dump-section", ".rustc=/dev/stdout", f, "/dev/null"]) + .output() + .map(|o| o.stdout) + .unwrap_or_default(); + rmetas.insert(relative(f, target) + ":.rustc", section); + } + } + if let Some(e) = &msg.executable + && msg.target_name() == self.bin + { + exe = Some(e.clone()); + } + } + let log = r.stderr; + Build { + ok: r.ok && !r.hang, + ice: cargo::is_ice(&log), + hang: r.hang, + reuse: cargo::reuse_checks(&log), + art: artifacts::collect(&r.stdout, target), + rmetas, + exe, + log, + } + } +} + +/// Reports seen since the reuse check exists and judged benign (docs/shadow-mode.md): reused +/// codegen units differing only in debuginfo at the end of the file, and constant allocations +/// shared differently between evaluations in different typing modes. +fn known(kind: &str, detail: &[String]) -> bool { + kind == "verify-reuse" + && detail.iter().all(|d| d == "codegen unit" || d.starts_with("allocation sharing eval_to_const_value_raw")) +} + +fn run_exe(exe: Option<&str>) -> Value { + let Some(exe) = exe else { return Value::Null }; + match mirth_lab::rustc::run_command(Command::new(exe), Duration::from_secs(30)) { + Ok(f) => match f.exit { + mirth_lab::rustc::Exit::Timeout => json!(["timeout", ""]), + mirth_lab::rustc::Exit::Code(c) => json!([c, tail(&f.stdout_text(), 2000)]), + // Python's returncode for a signal is its negation. + mirth_lab::rustc::Exit::Signal(s) => json!([-s, tail(&f.stdout_text(), 2000)]), + }, + Err(_) => Value::Null, + } +} + +fn unified_diff(old: &str, new: &str, rel: &str) -> String { + similar::TextDiff::from_lines(old, new) + .unified_diff() + .header(&format!("a/{rel}"), &format!("b/{rel}")) + .to_string() +} + +fn rs_files(src: &Path) -> Vec { + let mut v: Vec = walkdir::WalkDir::new(src) + .into_iter() + .filter_entry(|e| e.depth() == 0 || e.file_name() != "target") + .filter_map(Result::ok) + .filter(|e| e.file_type().is_file() && e.path().extension().is_some_and(|x| x == "rs")) + .map(|e| e.into_path()) + .collect(); + v.sort(); + v +} + +fn tar_gz(src: &Path, out: &Path) -> std::io::Result<()> { + let f = std::fs::File::create(out)?; + let mut t = tar::Builder::new(flate2::write::GzEncoder::new(f, flate2::Compression::default())); + t.follow_symlinks(false); + t.append_dir_all(".", src)?; + t.into_inner()?.finish()?; + Ok(()) +} + +fn rmtree(p: &Path) { + let _ = std::fs::remove_dir_all(p); +} + +struct Worker<'a> { + ctx: &'a Ctx, + k: u64, + home: PathBuf, + src: PathBuf, + target: PathBuf, + inc_target: PathBuf, + findings: PathBuf, + stats: Stats, + kept: BTreeMap, + history: Vec, + previous: BTreeMap>, +} + +#[derive(serde::Serialize, Default)] +struct Stats { + edits: u64, + built: u64, + failed: u64, + compared: u64, + findings: BTreeMap, + secs: f64, + by_edit: BTreeMap, +} + +impl Worker<'_> { + fn reset(&mut self) -> bool { + rmtree(&self.home); + let _ = std::fs::create_dir_all(&self.home); + if let Err(e) = copy_tree(&self.ctx.fixture, &self.src, &["target", "edits", "edit"], false) { + let _ = std::fs::write(self.ctx.work.join(format!("error-w{}.log", self.k)), e.to_string()); + return false; + } + self.history.clear(); + let b = self.ctx.build(&self.src, &self.target); + self.previous = b.rmetas; + if !b.ok { + let _ = std::fs::write(self.ctx.work.join(format!("error-w{}.log", self.k)), tail(&b.log, 6000)); + return false; + } + true + } + + fn report(&mut self, kind: &str, detail: &[String], inc: &Build, clean: Option<&Build>, extra: Value) { + *self.stats.findings.entry(kind.to_owned()).or_default() += 1; + let work = &self.ctx.work; + if self.ctx.args.pause_on_finding && !known(kind, detail) { + let paused = json!({"worker": self.k, "edit": self.stats.edits, "kind": kind, "detail": detail}); + let _ = std::fs::write(work.join("PAUSED"), serde_json::to_string_pretty(&paused).unwrap()); + let _ = std::fs::write(work.join("STOP"), ""); + } + let mut sorted = detail.to_vec(); + sorted.sort(); + let key = format!("{kind}:{}", sorted.join(",")); + let n = self.kept.entry(key).or_default(); + if *n >= self.ctx.args.keep { + return; + } + *n += 1; + let d = self.findings.join(format!("w{}-{:07}-{kind}", self.k, self.stats.edits)); + let _ = std::fs::create_dir_all(&d); + let _ = std::fs::write(d.join("history.json"), serde_json::to_string_pretty(&self.history).unwrap()); + let finding = json!({"kind": kind, "detail": detail, "extra": extra}); + let _ = std::fs::write(d.join("finding.json"), serde_json::to_string_pretty(&finding).unwrap()); + let _ = std::fs::write(d.join("inc.log"), tail(&inc.log, 8000)); + for (name, b) in [("inc", Some(inc)), ("clean", clean)] { + let lines: Vec<&str> = b.map(|b| b.log.lines().collect()).unwrap_or_default(); + let hit = lines.iter().position(|l| { + l.contains("panicked at") + || l.contains("internal compiler error") + || l.contains("unexpectedly panicked") + || l.contains("interrupted by SIG") + }); + if let Some(h) = hit { + let excerpt = lines[h.saturating_sub(5)..(h + 60).min(lines.len())].join("\n"); + let _ = std::fs::write(d.join(format!("{name}-ice.txt")), excerpt); + } + } + let _ = std::fs::write(d.join("clean.log"), clean.map_or("", |c| tail(&c.log, 8000))); + for rel in detail { + let name = Path::new(rel).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + if let Some(v) = inc.rmetas.get(rel) { + let _ = std::fs::write(d.join(format!("{name}.inc")), v); + } + if let Some(v) = clean.and_then(|c| c.rmetas.get(rel)) { + let _ = std::fs::write(d.join(format!("{name}.clean")), v); + } + } + let _ = tar_gz(&self.src, &d.join("src.tar.gz")); + } + + fn write_stats(&self) { + let _ = std::fs::write( + self.ctx.work.join(format!("stats-w{}.json", self.k)), + serde_json::to_string(&self.stats).unwrap(), + ); + } + + fn run(&mut self) { + let _ = std::fs::create_dir_all(&self.findings); + if !self.reset() { + return; + } + let a = &self.ctx.args; + let mut rng = StdRng::seed_from_u64(a.seed * 1000 + self.k); + let started = Instant::now(); + let stop = self.ctx.work.join("STOP"); + while self.stats.edits < a.edits && !stop.exists() { + if cargo::free_bytes(&self.ctx.work) < a.min_free_gb << 30 { + println!("worker {}: less than {} GB free, stopping", self.k, a.min_free_gb); + break; + } + if self.history.iter().filter(|h| h["kept"] == true).count() >= a.reset && !self.reset() { + break; + } + let n = self.stats.edits; + self.stats.edits += 1; + let paths = rs_files(&self.src); + let Some(path) = paths.choose(&mut rng).cloned() else { break }; + let old = std::fs::read_to_string(&path).unwrap_or_default(); + let (edit, apply) = mutations::pick(&mut rng); + if matches!(edit, "int_literal" | "str_literal") && path.file_name().is_some_and(|f| f == "build.rs") { + // Cargo keeps stale OUT_DIR files, so renaming a generated file splits the + // builds, and a changed number can make the build script loop forever. + continue; + } + let new = match apply(&old, &mut rng, n as usize) { + Some(new) if new != old => new, + _ => continue, + }; + let _ = std::fs::write(&path, &new); + let rel = path.strip_prefix(&self.src).unwrap().to_string_lossy().into_owned(); + let diff = unified_diff(&old, &new, &rel); + let inc = self.ctx.build(&self.src, &self.target); + cargo::note_untracked(&self.ctx.work.join("untracked.txt"), &cargo::untracked_reads(&inc.log)); + self.history.push(json!({"edit": edit, "file": rel, "diff": diff, "before": old, "after": new, "kept": inc.ok})); + self.stats.by_edit.entry(edit.to_owned()).or_default()[0] += 1; + if inc.ice { + self.report("ICE", &[], &inc, None, Value::Null); + } + if inc.hang { + self.report("hang", &[], &inc, None, Value::Null); + } + if !inc.reuse.is_empty() { + let lines: Vec = inc + .log + .lines() + .filter(|l| l.starts_with("rustc-verify-reuse")) + .take(20) + .map(|l| l.chars().take(3000).collect()) + .collect(); + let reuse = inc.reuse.clone(); + self.report("verify-reuse", &reuse, &inc, None, json!(lines)); + } + if !inc.ok { + self.stats.failed += 1; + let _ = std::fs::write(&path, &old); + self.history.push(json!({"edit": "revert", "file": rel, "diff": "", "kept": false})); + continue; + } + self.stats.by_edit.get_mut(edit).unwrap()[1] += 1; + self.stats.built += 1; + self.compare(inc); + if self.stats.edits % 20 == 0 { + self.stats.secs = started.elapsed().as_secs_f64(); + self.write_stats(); + } + } + self.stats.secs = started.elapsed().as_secs_f64(); + self.write_stats(); + } + + /// The clean build at the same path, and the comparisons. + fn compare(&mut self, inc: Build) { + let (target, inc_target) = (self.target.clone(), self.inc_target.clone()); + rmtree(&inc_target); + let _ = std::fs::rename(&target, &inc_target); + let clean = self.ctx.build(&self.src, &target); + if clean.ice { + self.report("ICE", &["clean".into()], &inc, Some(&clean), Value::Null); + } + if clean.hang { + self.report("hang", &["clean".into()], &inc, Some(&clean), Value::Null); + } + if !clean.ok { + self.report("split", &[], &inc, Some(&clean), Value::Null); + } else { + self.stats.compared += 1; + let skip = &self.ctx.skip; + let rmeta_differ = |a: &BTreeMap>, b: &BTreeMap>| -> Vec { + let keys: std::collections::BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + keys.into_iter().filter(|r| a.get(*r) != b.get(*r)).cloned().collect() + }; + let mut differ = rmeta_differ(&inc.rmetas, &clean.rmetas); + let mut others: BTreeMap<&str, Vec> = + artifacts::compare(&inc.art, &clean.art).into_iter().filter(|(k, _)| !skip.contains(k)).collect(); + if !differ.is_empty() || !others.is_empty() { + // Build clean again: if two clean builds differ, the difference is + // nondeterminism (P5), not incremental reuse. + let mut p5: Vec = Vec::new(); + let mut again_ok = false; + let mut again = None; + for _ in 0..self.ctx.args.p5_builds { + rmtree(&target); + let b = self.ctx.build(&self.src, &target); + p5 = rmeta_differ(&clean.rmetas, &b.rmetas); + p5.extend( + artifacts::compare(&clean.art, &b.art) + .into_iter() + .filter(|(k, _)| !skip.contains(k)) + .map(|(k, v)| format!("{k}: {}", v[0])), + ); + again_ok = b.ok; + again = Some(b); + if !p5.is_empty() || !again_ok { + break; + } + } + if again_ok && !p5.is_empty() { + p5.truncate(10); + self.report("P5", &p5, &clean, again.as_ref(), Value::Null); + differ.clear(); + others.clear(); + } + } + // Known: metadata reused unchanged from the previous session although a source file + // changed (its hash and length in the source map are stale). + let stale = differ.iter().any(|r| self.previous.get(r).is_some_and(|p| Some(p) == inc.rmetas.get(r))); + if !differ.is_empty() && stale { + *self.stats.findings.entry("P6-stale-reuse".into()).or_default() += 1; + } else if !differ.is_empty() { + self.report("P6", &differ, &inc, Some(&clean), Value::Null); + } + // More oracles: object code in the rlibs, the binary, and the diagnostics. + for (kind, detail) in &others { + self.report(kind, &detail[..detail.len().min(10)], &inc, Some(&clean), Value::Null); + } + let inc_exe = inc.exe.as_ref().map(|e| e.replace(&*target.to_string_lossy(), &inc_target.to_string_lossy())); + let ra = run_exe(inc_exe.as_deref()); + let rb = run_exe(clean.exe.as_deref()); + if ra != rb { + self.report("run", &[], &inc, Some(&clean), json!({"inc": ra, "clean": rb})); + } + } + rmtree(&target); + let _ = std::fs::rename(&inc_target, &target); + self.previous = inc.rmetas; + } +} + +static SPLIT_DEBUGINFO: LazyLock = LazyLock::new(|| Regex::new(r"-Csplit-debuginfo=(packed|unpacked)").unwrap()); + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let fixture = std::fs::canonicalize(&args.fixture)?; + for f in ["STOP", "PAUSED"] { + let _ = std::fs::remove_file(work.join(f)); + } + let mut skip = vec!["rmeta"]; + if SPLIT_DEBUGINFO.is_match(&args.rustflags) { + skip.extend(["rlib", "exe"]); + } + let ctx = Ctx { + bin: fixture.file_name().unwrap().to_string_lossy().into_owned(), + args: args.clone(), + work: work.clone(), + fixture, + skip, + }; + let totals: Vec<(u64, u64, u64)> = std::thread::scope(|s| { + let handles: Vec<_> = (0..args.workers) + .map(|k| { + let ctx = &ctx; + s.spawn(move || { + let home = ctx.work.join(format!("w{k}")); + let mut w = Worker { + ctx, + k, + src: home.join("src"), + target: home.join("target"), + inc_target: home.join("target-inc"), + findings: ctx.work.join("findings"), + home, + stats: Stats::default(), + kept: BTreeMap::new(), + history: Vec::new(), + previous: BTreeMap::new(), + }; + w.run(); + (w.stats.edits, w.stats.built, w.stats.compared) + }) + }) + .collect(); + handles.into_iter().map(|h| h.join().unwrap_or_default()).collect() + }); + let sum = |f: fn(&(u64, u64, u64)) -> u64| totals.iter().map(f).sum::(); + println!("{{\"edits\": {}, \"built\": {}, \"compared\": {}}}", sum(|t| t.0), sum(|t| t.1), sum(|t| t.2)); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/fuzz_replay.rs b/crates/mirth-lab/src/tools/fuzz_replay.rs new file mode 100644 index 0000000..6b372b3 --- /dev/null +++ b/crates/mirth-lab/src/tools/fuzz_replay.rs @@ -0,0 +1,118 @@ +//! Replay a fuzz finding: apply its edits to the pristine fixture in order, with an incremental +//! build after each, then compare with a clean build. +//! +//! Prints which .rmeta files differ, and keeps the compiler's output of the last incremental +//! build and the clean build as inc.log and clean.log. --upto replays only the first N edits +//! that were kept, to find where the difference appears. + +use std::collections::{BTreeMap, BTreeSet}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; + +use serde::Deserialize; + +use mirth_lab::cargo::{self, copy_tree, messages, relative}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + fixture: PathBuf, + #[arg(long)] + finding: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long)] + upto: Option, + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "-Zincremental-verify-ich", allow_hyphen_values = true)] + rustflags: String, + #[arg(long)] + quiet: bool, +} + +#[derive(Deserialize)] +struct Step { + edit: String, + file: String, + #[serde(default)] + before: String, + #[serde(default)] + after: String, + kept: bool, +} + +/// Whether it built, the .rmeta files by path, and the compiler's output. +fn build(args: &Args, src: &Path, target: &Path) -> (bool, BTreeMap>, String) { + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", args.toolchain)) + .args(["build", "--workspace", "--offline", "-j", "4", "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics") + .current_dir(src) + .env("RUSTC", &args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + .env("RUSTFLAGS", &args.rustflags); + let Ok(r) = cargo::run_group(cmd, None) else { return (false, BTreeMap::new(), "cargo did not start".into()) }; + let rmetas = messages(&r.stdout) + .filter(|m| m.reason == "compiler-artifact") + .flat_map(|m| m.filenames) + .filter(|f| f.ends_with(".rmeta")) + .map(|f| (relative(&f, target), std::fs::read(&f).unwrap_or_default())) + .collect(); + (r.ok, rmetas, r.stderr) +} + +pub fn run(args: Args) -> anyhow::Result { + let _ = std::fs::remove_dir_all(&args.work); + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let (src, target, inc_target) = (work.join("src"), work.join("target"), work.join("target-inc")); + copy_tree(&std::fs::canonicalize(&args.fixture)?, &src, &["target", "edits", "edit"], false)?; + build(&args, &src, &target); + + let history: Vec = serde_json::from_str(&std::fs::read_to_string(args.finding.join("history.json"))?)?; + let mut kept = 0; + let mut last: Option<&Step> = None; + for step in &history { + if step.edit == "revert" { + if let Some(l) = last { + std::fs::write(src.join(&l.file), &l.before)?; + } + } else { + if args.upto.is_some_and(|u| step.kept && kept >= u) { + break; + } + std::fs::write(src.join(&step.file), &step.after)?; + last = Some(step); + kept += step.kept as usize; + } + let (ok, _, _) = build(&args, &src, &target); + if !args.quiet { + println!("{:18} {:24} {}", step.edit, step.file, if ok { "built" } else { "failed" }); + } + } + + let (ok, inc, log) = build(&args, &src, &target); + std::fs::write(work.join("inc.log"), log)?; + std::fs::rename(&target, &inc_target)?; + let (ok2, clean, log) = build(&args, &src, &target); + std::fs::write(work.join("clean.log"), log)?; + let keys: BTreeSet<&String> = inc.keys().chain(clean.keys()).collect(); + let differ: Vec<&String> = keys.into_iter().filter(|r| inc.get(*r) != clean.get(*r)).collect(); + let name = |d: &str| Path::new(d).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + let names: Vec = differ.iter().map(|d| name(d)).collect(); + println!( + "{{\"kept_edits\": {kept}, \"inc_ok\": {ok}, \"clean_ok\": {ok2}, \"differ\": {}}}", + serde_json::to_string(&names)?.replace("\",\"", "\", \"") + ); + for d in differ { + let n = name(d); + std::fs::write(work.join(format!("{n}.inc")), inc.get(d).map_or(&[][..], |v| v))?; + std::fs::write(work.join(format!("{n}.clean")), clean.get(d).map_or(&[][..], |v| v))?; + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/gate_check.rs b/crates/mirth-lab/src/tools/gate_check.rs new file mode 100644 index 0000000..38d3a83 --- /dev/null +++ b/crates/mirth-lab/src/tools/gate_check.rs @@ -0,0 +1,241 @@ +//! Feature gates: nothing unstable may be usable from stable code, whatever the spelling. +//! +//! Compiled without `#![feature]` and without `RUSTC_BOOTSTRAP`, as a stable user would: +//! +//! - attributes: every attribute in the "Unstable attributes" part of +//! compiler/rustc_feature/src/builtin_attrs.rs, on each kind of item and position +//! - library: every top-level `pub` item of core, alloc and std marked +//! `#[unstable(feature)]`, reached by `use` (direct, renamed, glob), implemented (traits), +//! taken as a value (functions) or named as a type +//! +//! A program must report the gate. One that compiles is a finding; one that fails without +//! mentioning the gate is noted. Library items whose path does not resolve even with the +//! feature are skipped. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::rustc::{Compile, Status}; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The rust checkout (for the attribute table and the library sources). + #[arg(long)] + rust: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 8)] + jobs: usize, + /// `attributes` or `library`. + #[arg(long)] + only: Option, +} + +static GATE: LazyLock = LazyLock::new(|| { + Regex::new(r"E0658|is experimental|is unstable|unstable feature|use of unstable|internal implementation detail|unstable library feature|requires a nightly|used internally by the standard library|may not be used|are considered unstable|is an internal|cannot be used on stable").unwrap() +}); +/// Arguments for attributes that need them; the rest are written bare. +const ATTR_ARGS: &[(&str, &str)] = &[ + ("optimize", "(speed)"), ("patchable_function_entry", "(prefix_nops = 1, entry_nops = 1)"), + ("instrument_fn", " = \"on\""), ("cfi_encoding", " = \"u1x\""), ("register_tool", "(mytool)"), + ("register_attribute_tool", "(mytool)"), ("register_lint_tool", "(mytool)"), ("linkage", " = \"weak\""), + ("lang", " = \"mirth_nonexistent\""), ("rustc_on_unimplemented", "(message = \"x\")"), + ("rustc_diagnostic_item", " = \"mirth_x\""), ("test_runner", "(crate::r)"), ("pattern_complexity_limit", " = 10"), + ("rustc_legacy_const_generics", "(0)"), ("rustc_abi", "(debug)"), ("rustc_macro_transparency", " = \"semitransparent\""), + ("unstable", "(feature = \"x\", issue = \"none\")"), ("stable", "(feature = \"x\", since = \"1.0.0\")"), + ("rustc_const_unstable", "(feature = \"x\", issue = \"none\")"), ("rustc_const_stable", "(feature = \"x\", since = \"1.0.0\")"), + ("feature", "(mirth_nonexistent)"), ("rustc_objc_class", " = \"X\""), ("rustc_objc_selector", " = \"x\""), + ("rustc_confusables", "(\"x\")"), ("rustc_must_implement_one_of", "(a, b)"), ("allow_internal_unstable", "(core_intrinsics)"), + ("rustc_allow_const_fn_unstable", "(x)"), ("rustc_default_body_unstable", "(feature = \"x\", issue = \"none\")"), + ("rustc_simd_monomorphize_lane_limit", " = \"8\""), ("rustc_scalable_vector", "(4)"), +]; +/// Each position: a name, and a program with `{A}` where the outer attribute goes (`{AI}`: +/// the inner one). +const POSITIONS: &[(&str, &str)] = &[ + ("fn", "{A}\npub fn f() {}\nfn main() {}"), + ("fn-no-body", "pub trait T { {A} fn m(&self); }\nfn main() {}"), + ("foreign-fn", "unsafe extern \"C\" { {A} fn ext(); }\nfn main() {}"), + ("param", "pub fn f({A} x: u32) -> u32 { x }\nfn main() {}"), + ("param-no-body", "pub trait T { fn m(&self, {A} x: u32); }\nfn main() {}"), + ("fn-ptr-param", "pub type F = fn({A} u32);\nfn main() {}"), + ("struct", "{A}\npub struct S;\nfn main() {}"), + ("field", "pub struct S { {A} pub x: u32 }\nfn main() {}"), + ("impl", "pub struct S;\n{A}\nimpl S {}\nfn main() {}"), + ("trait", "{A}\npub trait T {}\nfn main() {}"), + ("mod", "{A}\npub mod m {}\nfn main() {}"), + ("closure", "fn main() { let _c = {A} || (); }"), + ("statement", "fn main() { {A} let _x = 1; }"), + ("crate", "#![{AI}]\nfn main() {}"), +]; + +#[derive(Serialize)] +struct Res { + kind: &'static str, + item: String, + position: String, + result: String, + first: String, +} + +fn compile_text(args: &Args, text: &str, feature: Option<&str>) -> (Status, String) { + let dir = tempfile::tempdir_in(&args.work).expect("scratch"); + let f = dir.path().join("t.rs"); + let body = match feature { + Some(feat) => format!("#![feature({feat})]\n{text}"), + None => text.to_owned(), + }; + let _ = std::fs::write(&f, body); + let mut c = Compile::new(&args.rustc, &f, dir.path(), &[], "2021").emit("metadata").timeout(60); + if feature.is_none() { + c = c.stable(); + } + let c = c.run(); + (c.status, c.stderr) +} + +fn classify(status: Status, stderr: &str, item: &str) -> String { + // `#[feature]` outside the crate root does nothing; rustc warns that it belongs at the root. + if status == Status::Ok && item == "feature" && stderr.contains("crate-level attribute") { + return "gated".into(); + } + match status { + Status::Ok => "accepted".into(), + Status::Ice => "ice".into(), + _ if GATE.is_match(stderr) => "gated".into(), + _ => "no-gate-message".into(), + } +} + +fn first_error(stderr: &str) -> String { + stderr.lines().find(|l| l.starts_with("error")).unwrap_or("").chars().take(200).collect() +} + +fn attributes(args: &Args) -> Vec { + let table = std::fs::read_to_string(args.rust.join("compiler/rustc_feature/src/builtin_attrs.rs")).unwrap_or_default(); + let start = table.find("Unstable attributes:").unwrap_or(0); + let re = Regex::new(r"sym::([a-z_0-9]+)").unwrap(); + let mut names: Vec = re.captures_iter(&table[start..]).map(|c| c[1].to_owned()).collect(); + names.sort(); + names.dedup(); + let args_of: BTreeMap<&str, &str> = ATTR_ARGS.iter().copied().collect(); + let jobs: Vec<(String, &str, String)> = names + .iter() + .flat_map(|name| { + let a = args_of.get(name.as_str()).copied().unwrap_or(""); + POSITIONS.iter().map(move |(pos, t)| { + (name.clone(), *pos, t.replace("{AI}", &format!("{name}{a}")).replace("{A}", &format!("#[{name}{a}]"))) + }) + }) + .collect(); + jobs.par_iter() + .map(|(name, pos, prog)| { + let (status, err) = compile_text(args, prog, None); + Res { kind: "attribute", item: name.clone(), position: pos.to_string(), result: classify(status, &err, name), first: first_error(&err) } + }) + .collect() +} + +struct Item { + path: String, + kind: String, + feature: String, + generic: bool, +} + +fn library_items(rust: &Path) -> Vec { + let unstable = Regex::new(r#"^#\[unstable\(feature = "([a-z_0-9]+)""#).unwrap(); + let decl = Regex::new(r#"^pub (?:const |unsafe |auto |extern "C" )*(struct|enum|trait|union|type|fn|const|static|macro) ([A-Za-z_][A-Za-z_0-9]*)(<)?"#).unwrap(); + let mut items = Vec::new(); + for krate in ["core", "alloc", "std"] { + let root = rust.join("library").join(krate).join("src"); + for entry in WalkDir::new(&root).into_iter().filter_map(Result::ok) { + let p = entry.path(); + if p.extension().is_none_or(|x| x != "rs") { + continue; + } + let rel = p.strip_prefix(&root).unwrap().with_extension(""); + let mut module = vec![krate.to_owned()]; + module.extend(rel.iter().map(|c| c.to_string_lossy().into_owned()).filter(|c| c != "lib" && c != "mod")); + let Ok(text) = std::fs::read_to_string(p) else { continue }; + let lines: Vec<&str> = text.lines().collect(); + for (i, line) in lines.iter().enumerate() { + let Some(m) = unstable.captures(line) else { continue }; + for next in lines.iter().skip(i + 1).take(5) { + if next.starts_with('#') { + continue; + } + if let Some(d) = decl.captures(next) { + items.push(Item { path: format!("{}::{}", module.join("::"), &d[2]), kind: d[1].to_owned(), feature: m[1].to_owned(), generic: d.get(3).is_some() }); + } + break; + } + } + } + } + items +} + +fn library(args: &Args) -> Vec { + library_items(&args.rust) + .par_iter() + .flat_map(|item| { + let (status, _) = compile_text(args, &format!("#[allow(unused_imports)] use {};\nfn main() {{}}", item.path), Some(&item.feature)); + if status != Status::Ok { + return vec![Res { kind: "library", item: item.path.clone(), position: "path".into(), result: "skipped: path".into(), first: String::new() }]; + } + let (parent, name) = item.path.rsplit_once("::").unwrap(); + let p = &item.path; + let mut progs = vec![ + ("use", format!("#[allow(unused_imports)] use {p};\nfn main() {{}}")), + ("use-as", format!("#[allow(unused_imports)] use {p} as Renamed;\nfn main() {{}}")), + ("glob", format!("#[allow(unused_imports)] use {parent}::*;\n#[allow(unused_imports)] use self::{name} as _;\nfn main() {{}}")), + ]; + if item.kind == "trait" && !item.generic { + progs.push(("impl", format!("struct L;\nimpl {p} for L {{}}\nfn main() {{}}"))); + } + if item.kind == "fn" && !item.generic { + progs.push(("value", format!("fn main() {{ let _f = {p}; }}"))); + } + if ["struct", "enum", "union", "type"].contains(&item.kind.as_str()) && !item.generic { + progs.push(("type", format!("pub fn g(_: Option<&{p}>) {{}}\nfn main() {{}}"))); + } + progs + .into_iter() + .map(|(pos, prog)| { + let (status, err) = compile_text(args, &prog, None); + Res { kind: "library", item: item.path.clone(), position: pos.into(), result: classify(status, &err, ""), first: first_error(&err) } + }) + .collect() + }) + .collect() +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build_global().ok(); + let mut results = Vec::new(); + if args.only.as_deref().is_none_or(|o| o == "attributes") { + results.extend(attributes(&args)); + } + if args.only.as_deref().is_none_or(|o| o == "library") { + results.extend(library(&args)); + } + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut counts: BTreeMap<(&str, &str), usize> = BTreeMap::new(); + for r in &results { + *counts.entry((r.kind, r.result.as_str())).or_default() += 1; + } + println!("{counts:?}"); + for r in results.iter().filter(|r| r.result == "accepted" || r.result == "ice") { + println!("{:9} {:9} {:45} {}", r.result.to_uppercase(), r.kind, r.item, r.position); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/grammar_coverage.rs b/crates/mirth-lab/src/tools/grammar_coverage.rs new file mode 100644 index 0000000..2a079ee --- /dev/null +++ b/crates/mirth-lab/src/tools/grammar_coverage.rs @@ -0,0 +1,268 @@ +//! Which parts of the Rust grammar a fixture's sources use, by Ur's grammar. +//! +//! Ur's Rust grammar (Urscal modules, `syntax Sort = Label: ... | Label: ... | Other ;`) names +//! each alternative of each syntax sort. `ur parse --tree` prints a file's tree with every node as +//! `(Sort::Label ...)` and every literal token in quotes. Two measures: +//! +//! alternatives each labeled alternative, by construct: the label within its module, with +//! Conditions.rsc (expressions in condition position) counted as Expressions +//! literals each keyword and operator the syntax rules mention, used or not +//! +//! Prints what is missing, by module, and a summary. Lexical rules, layout and keyword lists are +//! not syntax alternatives and are left out. + +use std::collections::{BTreeMap, BTreeSet, HashMap}; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; + +use mirth_lab::coverage::to_json_indent; +use regex::Regex; +use serde::Serialize; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + ur: PathBuf, + /// Ur's Rust grammar: /ecosystems/rust/language. + #[arg(long)] + grammar: PathBuf, + fixture: PathBuf, + #[arg(long)] + json: Option, +} + +const STRING: &str = r#""(?:[^"\\]|\\.)*""#; +static STRING_RE: LazyLock = LazyLock::new(|| Regex::new(STRING).unwrap()); +static STRING_AT: LazyLock = LazyLock::new(|| Regex::new(&format!("^{STRING}")).unwrap()); +static COMMENT: LazyLock = LazyLock::new(|| Regex::new(r"//[^\n]*").unwrap()); +static SYNTAX: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^syntax\s+(\w+)[^=]*=").unwrap()); +static ATTR_STRING: LazyLock = LazyLock::new(|| Regex::new(&format!(r"@\w+={STRING}")).unwrap()); +static ATTR_EMPTY: LazyLock = LazyLock::new(|| Regex::new(r#"@\w+="[^"]*""#).unwrap()); +static NODE: LazyLock = LazyLock::new(|| Regex::new(r"\((\w+)::(\w+)").unwrap()); + +/// Conditions.rsc repeats the expression sorts for condition position (no struct literals): the +/// same constructs, so counted with the expressions. +fn family(stem: &str) -> &str { + if stem == "Conditions" { "Expressions" } else { stem } +} + +fn is_word(c: char) -> bool { + c.is_alphanumeric() || c == '_' +} + +/// Whether the character before byte `i` is neither a word character nor `:`. +fn free_before(text: &str, i: usize) -> bool { + text[..i].chars().next_back().is_none_or(|c| !is_word(c) && c != ':') +} + +/// (sort, body) for each `syntax` rule of a module: from `syntax Name ... =` to the `;` that +/// ends it, outside quotes and brackets. +fn syntax_rules(text: &str) -> Vec<(String, String)> { + let bytes = text.as_bytes(); + SYNTAX + .captures_iter(text) + .map(|m| { + let start = m.get(0).unwrap().end(); + let (mut i, mut depth) = (start, 0i32); + while i < bytes.len() { + match bytes[i] { + b'"' => { + i += STRING_AT.find(&text[i..]).map_or(1, |q| q.end()); + continue; + } + b'(' | b'[' | b'{' => depth += 1, + b')' | b']' | b'}' => depth -= 1, + b';' if depth == 0 => break, + _ => {} + } + i += 1; + } + (m[1].to_owned(), text[start..i.min(bytes.len())].to_owned()) + }) + .collect() +} + +/// The labels of a rule body: `Label:` (not `Label::`) not preceded by a word character or `:`. +fn labels(body: &str) -> Vec<&str> { + let mut out = Vec::new(); + let mut i = 0; + while i < body.len() { + let c = body[i..].chars().next().unwrap(); + if c.is_ascii_uppercase() && free_before(body, i) { + let end = body[i..].find(|c: char| !is_word(c)).map_or(body.len(), |e| i + e); + let colon = body[end..].find(|c: char| !c.is_whitespace()).map_or(body.len(), |e| end + e); + if body[colon..].starts_with(':') && !body[colon + 1..].starts_with(':') { + out.push(&body[i..end]); + i = colon + 1; + continue; + } + } + i += c.len_utf8(); + } + out +} + +/// The quoted tokens of a tree: a string not preceded by a word character or `:`. +fn quoted(tree: &str) -> Vec<&str> { + let mut out = Vec::new(); + let mut i = 0; + while let Some(q) = tree[i..].find('"').map(|q| i + q) { + if free_before(tree, q) + && let Some(m) = STRING_AT.find(&tree[q..]) + { + out.push(&tree[q + 1..q + m.end() - 1]); + i = q + m.end(); + } else { + i = q + 1; + } + } + out +} + +/// Python's repr of a string. +fn py_repr(s: &str) -> String { + let quote = if s.contains('\'') && !s.contains('"') { '"' } else { '\'' }; + let mut out = String::from(quote); + for c in s.chars() { + match c { + '\\' => out.push_str("\\\\"), + '\n' => out.push_str("\\n"), + '\t' => out.push_str("\\t"), + '\r' => out.push_str("\\r"), + c if c == quote => { + out.push('\\'); + out.push(c); + } + c if (c as u32) < 0x20 || c as u32 == 0x7f => out.push_str(&format!("\\x{:02x}", c as u32)), + c => out.push(c), + } + } + out.push(quote); + out +} + +fn modules(grammar: &Path) -> Vec { + let mut files: Vec = mirth_lab::coverage::files_with(grammar, "rsc"); + files.sort(); + files +} + +#[derive(Serialize)] +struct Report { + alternatives: BTreeMap, + literals: BTreeMap, + failed: Vec, +} + +pub fn run(args: Args) -> anyhow::Result { + let mut grammar: BTreeMap<(String, String), Vec> = BTreeMap::new(); + let mut literals: HashMap = HashMap::new(); + let mut sort_family: HashMap = HashMap::new(); + for f in modules(&args.grammar) { + let stem = f.file_stem().and_then(|s| s.to_str()).unwrap_or("").to_owned(); + let fam = family(&stem).to_owned(); + let text = COMMENT.replace_all(&std::fs::read_to_string(&f)?, "").into_owned(); + let rules = syntax_rules(&text); + for (sort, _) in &rules { + sort_family.entry(sort.clone()).or_insert_with(|| fam.clone()); + } + if ["Testing", "Semantics", "Language", "Rust"].contains(&stem.as_str()) { + continue; + } + for (sort, body) in rules { + let body = ATTR_STRING.replace_all(&body, ""); + for lit in STRING_RE.find_iter(&body) { + let lit = &lit.as_str()[1..lit.as_str().len() - 1]; + let lit = lit.replace("\\<", "<").replace("\\>", ">").replace("\\\"", "\"").replace("\\\\", "\\"); + literals.entry(lit).or_insert_with(|| fam.clone()); + } + let bare = STRING_RE.replace_all(&body, "\"\""); + let bare = ATTR_EMPTY.replace_all(&bare, ""); + for label in labels(&bare) { + grammar.entry((fam.clone(), label.to_owned())).or_default().push(sort.clone()); + } + } + } + let mut files: Vec = WalkDir::new(&args.fixture) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && !p.components().any(|c| c.as_os_str() == "target")) + .collect(); + files.sort(); + let mut used_alts: HashMap<(String, String), usize> = HashMap::new(); + let mut used_lits: HashMap = HashMap::new(); + let mut failed = Vec::new(); + for f in &files { + let out = Command::new(&args.ur).args(["parse", "--tree"]).arg(f).output()?; + let tree = String::from_utf8_lossy(&out.stdout); + if !out.status.success() || !tree.contains("(File::") { + failed.push(f.display().to_string()); + continue; + } + for c in NODE.captures_iter(&tree) { + let fam = sort_family.get(&c[1]).cloned().unwrap_or_else(|| c[1].to_owned()); + *used_alts.entry((fam, c[2].to_owned())).or_default() += 1; + } + for lit in quoted(&tree) { + *used_lits.entry(lit.replace("\\\"", "\"").replace("\\\\", "\\")).or_default() += 1; + } + } + let mut by_module: BTreeMap<&str, Vec> = BTreeMap::new(); + let mut missing_alts = 0; + for ((fam, label), sorts) in &grammar { + if !used_alts.contains_key(&(fam.clone(), label.clone())) { + let sorts: BTreeSet<&String> = sorts.iter().collect(); + by_module.entry(fam).or_default().push(format!("{label} ({})", sorts.into_iter().cloned().collect::>().join("/"))); + missing_alts += 1; + } + } + let mut missing_lits: Vec<(&String, &String)> = literals.iter().filter(|(l, _)| !used_lits.contains_key(*l)).map(|(l, m)| (m, l)).collect(); + missing_lits.sort(); + let mut lit_by_module: BTreeMap<&str, Vec> = BTreeMap::new(); + for (m, lit) in &missing_lits { + lit_by_module.entry(m).or_default().push(py_repr(lit)); + } + let names: BTreeSet<&str> = by_module.keys().chain(lit_by_module.keys()).copied().collect(); + for m in names { + println!("== {m}"); + if let Some(a) = by_module.get(m) { + println!(" alternatives: {}", a.join(", ")); + } + if let Some(l) = lit_by_module.get(m) { + println!(" literals: {}", l.join(" ")); + } + } + let failed_list = if failed.is_empty() { + String::new() + } else { + format!(": [{}]", failed.iter().map(|f| py_repr(f)).collect::>().join(", ")) + }; + println!("{} files, {} failed to parse{failed_list}", files.len(), failed.len()); + println!("alternatives: {} of {} used", grammar.len() - missing_alts, grammar.len()); + println!("literals: {} of {} used", literals.len() - missing_lits.len(), literals.len()); + if let Some(j) = &args.json { + let report = Report { + alternatives: grammar.keys().map(|(f, l)| (format!("{f}::{l}"), used_alts.get(&(f.clone(), l.clone())).copied().unwrap_or(0))).collect(), + literals: literals.keys().map(|l| (l.clone(), used_lits.get(l).copied().unwrap_or(0))).collect(), + failed, + }; + std::fs::write(j, to_json_indent(&report, 1))?; + } + Ok(ExitCode::SUCCESS) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn labels_and_quotes() { + assert_eq!(labels("A: x | Foo::Bar | bC: y | Baz : z | (Q:: r) | Last:"), ["A", "Baz", "Last"]); + assert_eq!(quoted(r#"(X "a" b"c" "d\"e" :"f")"#), ["a", r#"d\"e"#]); + assert_eq!(py_repr("it's"), "\"it's\""); + assert_eq!(py_repr("a\\b"), "'a\\\\b'"); + } +} diff --git a/crates/mirth-lab/src/tools/instr_check.rs b/crates/mirth-lab/src/tools/instr_check.rs new file mode 100644 index 0000000..7645d74 --- /dev/null +++ b/crates/mirth-lab/src/tools/instr_check.rs @@ -0,0 +1,204 @@ +//! Instrumentation round trip: instrumented programs must behave as uninstrumented ones and +//! write profiles that LLVM's tools accept. +//! +//! For each runnable UI test, with a toolchain that ships the profiler runtime and llvm-tools: +//! PGO (`-Cprofile-generate`, run, `llvm-profdata merge`, `-Cprofile-use`, run) and coverage +//! (`-Cinstrument-coverage`, run, merge, `llvm-cov export`). Findings: behavior differing from +//! the plain build, no profile written, the LLVM tools failing or warning about corrupt data, +//! the compiler failing or crashing on `-Cprofile-use`. Threaded tests are left out (scheduling); +//! a test relying on the linker discarding an unused symbol (`linking/executable-no-mangle-strip`) +//! fails to link instrumented, as expected. + +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Exit, Status, run_command}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// A toolchain with the profiler runtime and llvm-tools. + #[arg(long)] + toolchain: String, + #[command(flatten)] + sweep: Sweep, +} + +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"profile|instrument-coverage|coverage-options|^-O$|opt-level|panic=|prefer-dynamic|codegen-backend|-Clto|lto=|no-prepopulate").unwrap() +}); +static THREADS: LazyLock = LazyLock::new(|| Regex::new(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(").unwrap()); +static BAD: LazyLock = LazyLock::new(|| Regex::new(r"(?i)corrupt|malformed|invalid|truncated|failed to|error").unwrap()); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +struct Toolchain { + rustc: PathBuf, + tools: PathBuf, +} + +fn tool(tc: &Toolchain, name: &str, args: &[String]) -> (Exit, String) { + let mut cmd = Command::new(tc.tools.join(name)); + cmd.args(args); + match run_command(cmd, Duration::from_secs(120)) { + Ok(d) => { + let out = d.stdout_text(); + let text = d.stderr_text() + &out[out.len().saturating_sub(200)..]; + (d.exit, text) + } + Err(e) => (Exit::Code(-1), e.to_string()), + } +} + +fn observed(binary: &Path, env: &[(&str, &str)]) -> (Exit, String) { + // Every build runs from the same path (some tests print argv[0]); the test harness's + // timings and result order are normalized. + let fixed = binary.parent().and_then(Path::parent).unwrap_or(Path::new(".")).join("run").join("prog"); + let _ = std::fs::create_dir_all(fixed.parent().unwrap()); + let _ = std::fs::copy(binary, &fixed); + let o = rustc::observe(&fixed, 30, env); + (o.exit, normalize::stdout(&o.stdout)) +} + +fn profraws(dir: &Path) -> Vec { + std::fs::read_dir(dir) + .map(|d| d.flatten().filter(|e| e.path().extension().is_some_and(|x| x == "profraw")).map(|e| e.path().to_string_lossy().into_owned()).collect()) + .unwrap_or_default() +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, tc: &Toolchain, test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new() }; + let dir = driver::scratch_dir(&args.sweep); + let d = dir.path(); + let compile = |name: &str, extra: Vec| Compile::new(&tc.rustc, &test.path, &d.join(name), &test.flags, test.edition()).extra(extra).run(); + let plain = compile("plain", vec!["-Copt-level=1".into()]); + let Some(plain_bin) = plain.binary else { + rec.skip = Some(format!("plain build {:?}", plain.status).to_lowercase()); + return rec; + }; + let base = observed(&plain_bin, &[]); + if (0..2).any(|_| observed(&plain_bin, &[]) != base) { + rec.skip = Some("nondeterministic".into()); + return rec; + } + let mut found: Vec = Vec::new(); + // PGO: generate, merge, use. + let raw = d.join("pgo-raw"); + let generated = compile("gen", vec!["-Copt-level=1".into(), format!("-Cprofile-generate={}", raw.display())]); + match generated.binary { + None => found.push(serde_json::json!({ "what": format!("profile-generate build {:?}", generated.status).to_lowercase(), "stderr": tail(&generated.stderr, 1500) })), + Some(bin) => { + let got = observed(&bin, &[]); + if got != base { + found.push(serde_json::json!({ "what": "profile-generate changes behavior", "base": base, "got": got })); + } + let raws = profraws(&raw); + if raws.is_empty() && got.0 == Exit::Code(0) { + found.push(serde_json::json!({ "what": "no raw profile written" })); + } else if !raws.is_empty() { + let merged = d.join("merged.profdata"); + let mut a = vec!["merge".to_string(), "-o".into(), merged.to_string_lossy().into_owned()]; + a.extend(raws); + let (exit, out) = tool(tc, "llvm-profdata", &a); + if exit != Exit::Code(0) || BAD.is_match(&out) { + found.push(serde_json::json!({ "what": "llvm-profdata merge", "exit": exit, "out": tail(&out, 1500) })); + } else { + let used = compile("use", vec!["-Copt-level=2".into(), format!("-Cprofile-use={}", merged.display())]); + match (used.status, used.binary) { + (Status::Ice, _) => found.push(serde_json::json!({ "what": "profile-use ICE", "stderr": tail(&used.stderr, 2000) })), + (s, None) => found.push(serde_json::json!({ "what": format!("profile-use build {s:?}").to_lowercase(), "stderr": tail(&used.stderr, 1500) })), + (_, Some(ub)) => { + let got = observed(&ub, &[]); + if got != base { + found.push(serde_json::json!({ "what": "profile-use changes behavior", "base": base, "got": got })); + } + } + } + } + } + } + } + // Coverage: instrument, merge, export. + let cov = compile("cov", vec!["-Cinstrument-coverage".into()]); + match cov.binary { + None => found.push(serde_json::json!({ "what": format!("instrument-coverage build {:?}", cov.status).to_lowercase(), "stderr": tail(&cov.stderr, 1500) })), + Some(bin) => { + let craw = d.join("cov-raw"); + let _ = std::fs::create_dir_all(&craw); + let pattern = craw.join("c-%p.profraw").to_string_lossy().into_owned(); + let got = observed(&bin, &[("LLVM_PROFILE_FILE", &pattern)]); + if got != base { + found.push(serde_json::json!({ "what": "instrument-coverage changes behavior", "base": base, "got": got })); + } + let raws = profraws(&craw); + if !raws.is_empty() { + let merged = d.join("cov.profdata"); + let mut a = vec!["merge".to_string(), "-sparse".into(), "-o".into(), merged.to_string_lossy().into_owned()]; + a.extend(raws); + let (exit, out) = tool(tc, "llvm-profdata", &a); + if exit != Exit::Code(0) || BAD.is_match(&out) { + found.push(serde_json::json!({ "what": "llvm-profdata merge (coverage)", "exit": exit, "out": tail(&out, 1500) })); + } else { + let a = vec!["export".to_string(), "-summary-only".into(), format!("-instr-profile={}", merged.display()), bin.to_string_lossy().into_owned()]; + let (exit, out) = tool(tc, "llvm-cov", &a); + // A binary with nothing to instrument (a test harness without tests) has an + // empty coverage map, which llvm-cov refuses. + if exit != Exit::Code(0) && !out.contains("no coverage data found") { + found.push(serde_json::json!({ "what": "llvm-cov export", "exit": exit, "out": tail(&out, 1500) })); + } + } + } else if got.0 == Exit::Code(0) { + found.push(serde_json::json!({ "what": "no coverage profile written" })); + } + } + } + rec.found = found.iter().map(|f| f["what"].as_str().unwrap_or("").to_owned()).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let home = PathBuf::from(std::env::var("HOME")?); + let root = home.join(format!(".rustup/toolchains/{}-x86_64-unknown-linux-gnu", args.toolchain)); + let tc = Toolchain { rustc: root.join("bin/rustc"), tools: root.join("lib/rustlib/x86_64-unknown-linux-gnu/bin") }; + // Threaded tests' output order depends on scheduling, which instrumentation changes. + // Output that depends on a random hash seed. + // hashmap-debug-format: hash order. executable-no-mangle-strip: expects an unreferenced + // `#[no_mangle]` function (naming an undefined symbol) to be collected; IR PGO's profile data + // keeps every instrumented function, as clang -fprofile-generate does with --gc-sections. + const NOISE: &[&str] = &["collections/hashmap/hashmap-debug-format.rs", "linking/executable-no-mangle-strip.rs"]; + let tests = uitest::tests(&args.sweep.tests, &[Some(Kind::RunPass)], |t| { + uitest::flag_matches(t, &OWN) || THREADS.is_match(&t.text) || NOISE.contains(&t.rel.as_str()) + }); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &tc, t))) +} diff --git a/crates/mirth-lab/src/tools/miri_diff.rs b/crates/mirth-lab/src/tools/miri_diff.rs new file mode 100644 index 0000000..896e4f2 --- /dev/null +++ b/crates/mirth-lab/src/tools/miri_diff.rs @@ -0,0 +1,172 @@ +//! Miri differential: an accepted safe program must be free of undefined behavior, MIR +//! optimizations must not introduce any, and the compiled program must do what Miri says. +//! +//! For each runnable UI test, interprets it with Miri at `-Zmir-opt-level` 0, 2 and 4 and builds +//! and runs it natively. Findings: UB at level 0 in a test without `unsafe` code; UB only after +//! MIR optimization; Miri's exit status or stdout changing with the MIR level; the native +//! program's exit status or stdout differing from Miri's. Tests Miri cannot run are skipped, as +//! are comparisons for threaded tests (scheduling differs) and tests asserting what Rust leaves +//! unspecified (function pointer equality, zero-sized addresses), which Miri varies on purpose. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::miri::{Miri, MiriRun, MiriStatus}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Exit}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long, default_value = "nightly-2026-10-06")] + miri_toolchain: String, + /// Seconds per Miri run. + #[arg(long, default_value_t = 120)] + timeout: u64, + #[command(flatten)] + sweep: Sweep, +} + +const FIXED: &[&str] = &["-Coverflow-checks=on", "-Cdebug-assertions=on"]; +/// Miri without preemption: threads switch only where they block, the same at every MIR level. +const MIRI_FLAGS: &[&str] = &["-Zmiri-preemption-rate=0"]; +const LEVELS: &[(&str, &[&str])] = &[("miri0", &[]), ("miri2", &["-Zmir-opt-level=2"]), ("miri4", &["-Zmir-opt-level=4"])]; +/// Tests asserting what Rust leaves unspecified, which Miri varies on purpose: function pointer +/// equality, the addresses of zero-sized values, stack addresses, function alignment; and one +/// Miri limitation (`.init_array` functions called without glibc's arguments). +const UNSPECIFIED: &[&str] = &[ + "consts/const-extern-function.rs", + "consts/zst_no_llvm_alloc.rs", + "layout/null-pointer-optimization.rs", + "mir/mir_misc_casts.rs", + "mir/mir_coercions.rs", + "extern/extern-compare-with-return-type.rs", + "fn/fn-ptr-trait-run.rs", + "mir/mir_raw_fat_ptr.rs", + "codegen/StackColoring-not-blowup-stack-issue-40883.rs", + "attributes/fn-align-dyn.rs", + "runtime/stdout-before-main.rs", +]; +static THREADS: LazyLock = LazyLock::new(|| Regex::new(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(").unwrap()); +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"^-O$|opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|panic=|-Cpanic|prefer-dynamic|-Zbuild-std|-Clink|-Ctarget").unwrap() +}); +/// Code Miri cannot interpret, and compile-time output that would land in Miri's stdout. +static NOT_FOR_MIRI: LazyLock = LazyLock::new(|| { + Regex::new(r#"\basm!|global_asm!|naked_asm!|extern\s+"C"\s*\{|#\[link\(|std::process::Command|\bfork\b|libc::|dlopen|std::os::unix::process|trace_macros|log_syntax"#).unwrap() +}); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + #[serde(skip_serializing_if = "Option::is_none")] + note: Option, + miri: Vec<(String, MiriStatus)>, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn clean(text: &str, test: &Test) -> String { + // argv[0]: the source file under Miri, the binary natively. + let t = text.replace(&test.path.to_string_lossy().into_owned(), ""); + let t = regex_argv0().replace_all(&t, "").into_owned(); + normalize::stdout(&normalize::stderr(&t)) +} + +fn regex_argv0() -> &'static Regex { + static RE: LazyLock = LazyLock::new(|| Regex::new(r"\S*/native/prog\b").unwrap()); + &RE +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, miri: &Miri, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let mut rec = Rec { test: test.rel.clone(), skip: None, note: None, miri: Vec::new(), found: Vec::new() }; + let native = Compile::new(&args.rustc, &test.path, &dir.path().join("native"), &test.flags, test.edition()) + .extra(FIXED.iter().copied().chain(["-Copt-level=0"])) + .run(); + let Some(binary) = native.binary else { + rec.skip = Some(format!("native build {:?}", native.status).to_lowercase()); + return rec; + }; + let mut outs: Vec<(Exit, String)> = + (0..3).map(|_| rustc::observe(&binary, 20, &[])).map(|o| (o.exit, clean(&o.stdout, test))).collect(); + outs.dedup(); + if outs.len() > 1 { + rec.skip = Some("native run is nondeterministic".into()); + return rec; + } + let (native_exit, native_out) = outs.pop().unwrap(); + let mut runs: Vec<(&str, MiriRun)> = Vec::new(); + for &(name, extra) in LEVELS { + let flags: Vec = FIXED.iter().chain(MIRI_FLAGS).chain(extra).map(|s| s.to_string()).collect(); + let mut m = miri.run(&test.path, &test.flags, test.edition(), &flags, args.timeout, dir.path()); + m.stdout = clean(&m.stdout, test); + if name == "miri0" && matches!(m.status, MiriStatus::Unsupported | MiriStatus::Error | MiriStatus::Timeout) { + rec.skip = Some(format!("miri {:?}", m.status).to_lowercase()); + return rec; + } + runs.push((name, m)); + } + rec.miri = runs.iter().map(|(n, m)| (n.to_string(), m.status)).collect(); + let unspecified = UNSPECIFIED.contains(&test.rel.as_str()); + let threaded = THREADS.is_match(&test.text); + let safe = !test.text.contains("unsafe"); + let m0 = &runs[0].1; + let mut found: Vec = Vec::new(); + // UB in a program without `unsafe` code can only be the compiler's. + if m0.status == MiriStatus::Ub && safe && !unspecified { + found.push(serde_json::json!({ "what": "ub (safe code)", "stderr": tail(&m0.stderr, 3000) })); + } else if m0.status == MiriStatus::Ub { + rec.note = Some("ub in a test with unsafe code".into()); + } + for (name, m) in &runs[1..] { + if m.status == MiriStatus::Ub && m0.status != MiriStatus::Ub { + found.push(serde_json::json!({ "what": format!("ub-opt ({name})"), "stderr": tail(&m.stderr, 3000) })); + } else if m.status == MiriStatus::Ice { + found.push(serde_json::json!({ "what": format!("ice ({name})"), "stderr": tail(&m.stderr, 3000) })); + } else if m.status == MiriStatus::Ok && m0.status == MiriStatus::Ok && !threaded && (&m.exit, &m.stdout) != (&m0.exit, &m0.stdout) { + found.push(serde_json::json!({ "what": format!("opt-differs ({name})"), "miri0": tail(&m0.stdout, 1500), "got": tail(&m.stdout, 1500) })); + } + } + if m0.status == MiriStatus::Ok && !threaded && !unspecified { + // Miri exits 1 on a panic that reaches main, native code 101. + let exit_m = if m0.exit == Exit::Code(1) && m0.stderr.contains("panicked") { Exit::Code(101) } else { m0.exit.clone() }; + if (exit_m, &m0.stdout) != (native_exit.clone(), &native_out) { + found.push(serde_json::json!({ "what": "native", "miri": [m0.exit, tail(&m0.stdout, 1500)], "native": [native_exit, tail(&native_out, 1500)] })); + } + } + rec.found = found.iter().map(|f| f["what"].as_str().unwrap_or("").to_owned()).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "fixed": FIXED, "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let miri = Miri::pinned(&args.miri_toolchain); + let tests = uitest::tests(&args.sweep.tests, uitest::RUNNABLE, |t| uitest::flag_matches(t, &OWN) || NOT_FOR_MIRI.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &miri, t))) +} diff --git a/crates/mirth-lab/src/tools/opt_diff.rs b/crates/mirth-lab/src/tools/opt_diff.rs new file mode 100644 index 0000000..a94d8c7 --- /dev/null +++ b/crates/mirth-lab/src/tools/opt_diff.rs @@ -0,0 +1,185 @@ +//! Optimization differential: a program's behavior must not depend on how it was optimized. +//! +//! Builds each runnable UI test under a set of configurations (optimization levels, MIR +//! optimization levels, LTO, target CPU, the Cranelift backend) and runs it, with overflow checks +//! and debug assertions fixed, and compares exit status, stdout and stderr with the unoptimized +//! baseline (`-Copt-level=0 -Zmir-opt-level=0`). A baseline whose output varies between two runs +//! is skipped; a difference counts only if the configuration's binary repeats it. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::normalize; +use mirth_lab::rustc::{self, Compile, Observed, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// A rustc with the Cranelift backend (the pinned nightly's). + #[arg(long)] + cranelift: Option, + /// Comma-separated subset of configurations (`base` is always built). + #[arg(long)] + configs: Option, + #[command(flatten)] + sweep: Sweep, +} + +const FIXED: &[&str] = &["-Coverflow-checks=on", "-Cdebug-assertions=on", "-Cpanic=unwind", "-Cdebuginfo=0"]; +const CONFIGS: &[(&str, &[&str])] = &[ + ("base", &["-Copt-level=0", "-Zmir-opt-level=0"]), + ("O0", &["-Copt-level=0"]), + ("O0-mir4", &["-Copt-level=0", "-Zmir-opt-level=4"]), + ("O1", &["-Copt-level=1"]), + ("O2", &["-Copt-level=2"]), + ("O3", &["-Copt-level=3"]), + ("Os", &["-Copt-level=s"]), + ("Oz", &["-Copt-level=z"]), + ("O3-mir4", &["-Copt-level=3", "-Zmir-opt-level=4"]), + ("O3-lto", &["-Copt-level=3", "-Clto=fat", "-Ccodegen-units=1"]), + ("O2-cgu16", &["-Copt-level=2", "-Ccodegen-units=16"]), + ("O3-native", &["-Copt-level=3", "-Ctarget-cpu=native"]), + ("cranelift", &["-Copt-level=0", "-Zcodegen-backend=cranelift"]), +]; +/// Tests that choose these themselves are left out: the configuration would contradict them. +static OWN: LazyLock = LazyLock::new(|| { + Regex::new(r"^-O$|opt-level|mir-opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|^-Clto|lto=|target-cpu|panic=|-Cpanic|prefer-dynamic|-Zbuild-std").unwrap() +}); +/// Tests whose outcome legitimately depends on optimization: unspecified behavior (whether two +/// equal promoted constants share an address), stack usage, a backend's documented gaps. +const NOISE: &[(&str, &[&str])] = &[ + ("mir/mir_raw_fat_ptr.rs", &["cranelift"]), + ("codegen/StackColoring-not-blowup-stack-issue-40883.rs", &["O0-mir4", "O0"]), + ("attributes/fn-align-dyn.rs", &["cranelift"]), + ("backtrace/backtrace.rs", &["cranelift"]), +]; + +#[derive(Serialize)] +struct Finding { + config: String, + what: String, + #[serde(skip_serializing_if = "Option::is_none")] + base: Option, + #[serde(skip_serializing_if = "Option::is_none")] + got: Option, + #[serde(skip_serializing_if = "String::is_empty")] + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + build: BTreeMap, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn observe(binary: &Path) -> Observed { + // Every configuration's program runs from the same path: some tests print argv[0]. + let fixed = binary.parent().and_then(Path::parent).unwrap_or(Path::new(".")).join("run").join("prog"); + let _ = std::fs::create_dir_all(fixed.parent().unwrap()); + let _ = std::fs::copy(binary, &fixed); + let o = rustc::observe(&fixed, 20, &[]); + Observed { exit: o.exit, stdout: normalize::stdout(&o.stdout), stderr: normalize::stderr(&o.stderr) } +} + +fn check(args: &Args, configs: &[(&str, &[&str])], test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + let mut rec = Rec { test: test.rel.clone(), build: BTreeMap::new(), skip: None, found: Vec::new() }; + let mut runs: BTreeMap<&str, Observed> = BTreeMap::new(); + let mut stderrs: BTreeMap<&str, String> = BTreeMap::new(); + let unwinding = test.text.contains("catch_unwind") || test.text.contains("needs-unwind"); + for &(name, cfg) in configs { + // Cranelift does not unwind on this target yet (catch_unwind catches nothing). + if name == "cranelift" && unwinding { + continue; + } + let rustc = if name == "cranelift" { args.cranelift.as_deref().unwrap() } else { args.rustc.as_path() }; + let out = dir.path().join(name); + let c = Compile::new(rustc, &test.path, &out, &test.flags, test.edition()) + .extra(FIXED.iter().chain(cfg.iter()).copied()) + .run(); + rec.build.insert(name.to_owned(), c.status); + stderrs.insert(name, c.stderr.chars().rev().take(2000).collect::().chars().rev().collect()); + if let Some(b) = c.binary { + runs.insert(name, observe(&b)); + } + } + if rec.build.get("base") != Some(&Status::Ok) || !runs.contains_key("base") { + rec.skip = Some("baseline does not build".into()); + return rec; + } + let base = runs["base"].clone(); + let again = observe(&dir.path().join("base").join("prog")); + if again != base { + rec.skip = Some(format!("baseline is nondeterministic: {:?} vs {:?}", base, again).chars().take(600).collect()); + return rec; + } + let noise: &[&str] = NOISE.iter().find(|(t, _)| *t == test.rel).map_or(&[], |(_, c)| c); + let mut found = Vec::new(); + for &(name, _) in configs { + if name == "base" || !rec.build.contains_key(name) || noise.contains(&name) { + continue; + } + let status = rec.build[name]; + if status != Status::Ok { + // Cranelift's documented gaps (tail calls, some linkages and SIMD intrinsics) show as + // errors or as panics inside the backend. + if name == "cranelift" && (status == Status::Error || stderrs[name].contains("rustc_codegen_cranelift")) { + continue; + } + found.push(Finding { config: name.into(), what: format!("build {status:?}").to_lowercase(), base: None, got: None, stderr: stderrs[name].clone() }); + continue; + } + let Some(got) = runs.get(name) else { continue }; + if name == "cranelift" && got.stderr.contains("failed to initiate panic") { + continue; + } + if *got != base { + let retry = observe(&dir.path().join(name).join("prog")); + if retry != base && retry == *got { + let diff: Vec<&str> = [("exit", got.exit != base.exit), ("stdout", got.stdout != base.stdout), ("stderr", got.stderr != base.stderr)] + .into_iter() + .filter_map(|(k, d)| d.then_some(k)) + .collect(); + found.push(Finding { config: name.into(), what: format!("run differs: {}", diff.join(",")), base: Some(base.clone()), got: Some(got.clone()), stderr: String::new() }); + } + } + } + rec.found = found.iter().map(|f| format!("{}: {}", f.config, f.what)).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "fixed": FIXED, "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let wanted: Option> = args.configs.as_deref().map(|c| c.split(',').collect()); + let configs: Vec<(&str, &[&str])> = CONFIGS + .iter() + .copied() + .filter(|(n, _)| *n == "base" || wanted.as_ref().is_none_or(|w| w.contains(n))) + .filter(|(n, _)| *n != "cranelift" || args.cranelift.is_some()) + .collect(); + let tests = uitest::tests(&args.sweep.tests, uitest::RUNNABLE, |t| uitest::flag_matches(t, &OWN)); + let tests = args.sweep.select(tests); + println!("{} tests, configurations: {}", tests.len(), configs.iter().map(|c| c.0).collect::>().join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &configs, t))) +} diff --git a/crates/mirth-lab/src/tools/release_diff.rs b/crates/mirth-lab/src/tools/release_diff.rs new file mode 100644 index 0000000..b56ae9e --- /dev/null +++ b/crates/mirth-lab/src/tools/release_diff.rs @@ -0,0 +1,158 @@ +//! Release-to-release: code that one toolchain accepts the next must accept too, in comparable +//! time. +//! +//! `cargo check --locked --offline --workspace` of each repository in a corpus of real crates +//! with an older and a newer toolchain (dependencies fetched first), each in its own target +//! directory, removed afterwards. Findings: a regression (old passes, new fails), an ICE, or the +//! new toolchain taking more than --slower times as long. A regression whose failing crate +//! enables unstable features (`#![feature]`, often only when it detects a nightly) is noted, not +//! reported. + +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use mirth_lab::rustc::{Exit, error_codes, is_ice, run_command}; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + /// A directory of repositories (each with a Cargo.toml). + #[arg(long)] + corpus: PathBuf, + #[arg(long)] + old: String, + #[arg(long)] + new: String, + #[arg(long)] + work: PathBuf, + #[arg(long)] + only: Option, + #[arg(long, default_value_t = 2)] + jobs: usize, + #[arg(long, default_value_t = 1.5)] + slower: f64, + #[arg(long, default_value_t = 1800)] + timeout: u64, +} + +static ERROR_LINE: LazyLock = LazyLock::new(|| Regex::new(r"\berror(\[E\d+\])?:").unwrap()); +static CRATE_ROOT: LazyLock = LazyLock::new(|| Regex::new(r"(/\S*?/registry/src/[^/]+/[^/]+|/\S+?)/src/").unwrap()); +static FEATURE: LazyLock = LazyLock::new(|| Regex::new(r"#!\[(cfg_attr\([^\]]*)?feature\(").unwrap()); + +#[derive(Serialize)] +struct Check { + exit: Exit, + seconds: f64, + ice: bool, + codes: Vec, + first: String, + tail: String, +} + +#[derive(Serialize)] +struct Rec { + repo: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + #[serde(skip_serializing_if = "Option::is_none")] + old: Option, + #[serde(skip_serializing_if = "Option::is_none")] + new: Option, + found: Vec, + notes: Vec, +} + +fn check(args: &Args, repo: &Path, toolchain: &str) -> Check { + let name = repo.file_name().unwrap().to_string_lossy(); + let target = args.work.join("target").join(format!("{name}-{toolchain}")); + let _ = std::fs::remove_dir_all(&target); + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{toolchain}")) + .args(["check", "--locked", "--offline", "--workspace", "--message-format=short"]) + .current_dir(repo) + .env("CARGO_TARGET_DIR", &target) + .env("CARGO_TERM_COLOR", "never") + .env("CARGO_INCREMENTAL", "0") + .env("RUSTFLAGS", "--cap-lints=warn") + .env_remove("RUSTC_WRAPPER"); + let start = Instant::now(); + let done = run_command(cmd, Duration::from_secs(args.timeout)); + let seconds = (start.elapsed().as_secs_f64() * 10.0).round() / 10.0; + let _ = std::fs::remove_dir_all(&target); + let (exit, err) = match done { + Ok(d) => (d.exit.clone(), d.stderr_text()), + Err(e) => (Exit::Code(-1), e.to_string()), + }; + let first = err.lines().find(|l| ERROR_LINE.is_match(l)).unwrap_or("").chars().take(300).collect(); + let tail = if exit == Exit::Code(0) { String::new() } else { err.chars().rev().take(3000).collect::().chars().rev().collect() }; + Check { exit, seconds, ice: is_ice(&err), codes: error_codes(&err), first, tail } +} + +/// The crate a first error points into, if its source enables `#![feature(...)]`. +fn uses_unstable(first_error: &str) -> Option { + let root = PathBuf::from(&CRATE_ROOT.captures(first_error)?[1]); + ["lib.rs", "main.rs"].iter().find_map(|f| { + let text = std::fs::read_to_string(root.join("src").join(f)).ok()?; + FEATURE.is_match(&text).then(|| root.file_name().unwrap().to_string_lossy().into_owned()) + }) +} + +fn one(args: &Args, repo: &Path) -> Rec { + let name = repo.file_name().unwrap().to_string_lossy().into_owned(); + let mut fetch = Command::new("cargo"); + fetch.arg(format!("+{}", args.new)).args(["fetch", "--locked"]).current_dir(repo); + match run_command(fetch, Duration::from_secs(1800)) { + Ok(d) if d.success() => {} + other => { + let why = other.map(|d| d.stderr_text()).unwrap_or_else(|e| e.to_string()); + return Rec { repo: name, skip: Some(format!("fetch failed: {}", why.chars().rev().take(300).collect::().chars().rev().collect::())), old: None, new: None, found: vec![], notes: vec![] }; + } + } + let old = check(args, repo, &args.old); + let new = check(args, repo, &args.new); + let (mut found, mut notes) = (Vec::new(), Vec::new()); + let ok = |c: &Check| c.exit == Exit::Code(0); + if ok(&old) && !ok(&new) { + match uses_unstable(&new.first) { + Some(krate) if !new.ice => notes.push(format!("regression in a crate using unstable features ({krate})")), + _ => found.push(if new.ice { "ice".into() } else { "regression".into() }), + } + } else if !ok(&old) && ok(&new) { + notes.push("fixed".into()); + } else if ok(&old) && ok(&new) && old.seconds > 5.0 && new.seconds > args.slower * old.seconds { + found.push(format!("slower: {}s -> {}s", old.seconds, new.seconds)); + } + if new.ice && !found.iter().any(|f| f == "ice") { + found.push("ice".into()); + } + Rec { repo: name, skip: None, old: Some(old), new: Some(new), found, notes } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let mut repos: Vec = std::fs::read_dir(&args.corpus)? + .flatten() + .map(|e| e.path()) + .filter(|p| p.join("Cargo.toml").exists() && args.only.as_ref().is_none_or(|o| p.to_string_lossy().contains(o.as_str()))) + .collect(); + repos.sort(); + println!("{} repositories, {} -> {}", repos.len(), args.old, args.new); + let out = std::sync::Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(args.work.join("results.jsonl"))?); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + pool.install(|| { + repos.par_iter().for_each(|repo| { + let rec = one(&args, repo); + if let Ok(mut f) = out.lock() { + use std::io::Write; + let _ = writeln!(f, "{}", serde_json::to_string(&rec).unwrap_or_default()); + } + let tag = rec.skip.clone().unwrap_or_else(|| if rec.found.is_empty() { rec.notes.join(", ") } else { rec.found.join(", ") }); + println!("{:45} {}", rec.repo, if tag.is_empty() { "same".into() } else { tag }); + }) + }); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/replay.rs b/crates/mirth-lab/src/tools/replay.rs new file mode 100644 index 0000000..ea5d0dc --- /dev/null +++ b/crates/mirth-lab/src/tools/replay.rs @@ -0,0 +1,315 @@ +//! Replay a crate's git history through incremental compilation. +//! +//! For each first-parent commit, oldest first, the workspace is checked out and built +//! incrementally on top of the previous commit's build, then built again from scratch, and the +//! two are compared: +//! +//! P6 every .rmeta Cargo reports for a workspace member is identical +//! rlib every rlib's members are identical, object code included +//! diag both builds printed the same diagnostics +//! reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, +//! docs/hunt/verify-reuse.patch) found nothing stale +//! ICE neither build crashed the compiler +//! split both builds succeed or both fail +//! +//! Registry dependencies are not compiled incrementally by Cargo, so the clean build starts +//! from a copy of the incremental target directory with the workspace members and the +//! incremental cache removed, and only the members are built again. Both builds use the same +//! target directory path, since Cargo derives a crate's identity from paths. +//! +//! Writes /results.jsonl, one line per commit, and keeps the logs of every problem and +//! both .rmeta files of the first few differences per crate in /findings. + +use std::collections::{BTreeMap, BTreeSet}; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::time::Instant; + +use serde_json::json; + +use mirth_lab::artifacts::{self, Collected}; +use mirth_lab::cargo::{self, copy_tree, messages, relative, tail}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: String, + #[arg(long)] + repo: String, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 2000)] + commits: usize, + /// For cargo. + #[arg(long, default_value = "nightly-2026-10-06")] + toolchain: String, + #[arg(long, default_value = "4")] + jobs: String, + /// Differences kept per crate. + #[arg(long, default_value_t = 3)] + keep: usize, + /// First commit index to replay. + #[arg(long = "from", default_value_t = 0)] + start: usize, + /// Last commit index to replay. + #[arg(long = "to")] + end: Option, + /// More flags for every build, such as -Copt-level=2. + #[arg(long, default_value = "", allow_hyphen_values = true)] + rustflags: String, + /// Do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch). + #[arg(long)] + no_verify_reuse: bool, +} + +struct Ctx<'a> { + args: &'a Args, + src: PathBuf, +} + +struct Build { + ok: bool, + ice: bool, + secs: f64, + log: String, + reuse: Vec, + untracked: BTreeSet, + rmetas: BTreeMap>, + fresh: Vec, + art: Collected, +} + +impl Ctx<'_> { + fn cmd(&self, program: &str) -> Command { + let mut c = Command::new(program); + c.current_dir(&self.src) + .env("RUSTC", &self.args.rustc) + .env("RUSTC_WRAPPER", "") + .env("CARGO_INCREMENTAL", "1") + // A commit that denies warnings would otherwise stop building with a newer compiler. + .env("RUSTFLAGS", format!("--cap-lints=warn {}", self.args.rustflags).trim()) + .env("CARGO_TERM_COLOR", "never"); + if !self.args.no_verify_reuse { + c.env("RUSTC_VERIFY_REUSE", "1").env("RUSTC_REPORT_UNTRACKED", "1"); + } + if program == "cargo" { + c.arg(format!("+{}", self.args.toolchain)); + } + c + } + + fn run(&self, program: &str, args: &[&str]) -> (bool, String) { + match self.cmd(program).args(args).output() { + Ok(o) => (o.status.success(), String::from_utf8_lossy(&o.stdout).into_owned()), + Err(_) => (false, String::new()), + } + } + + /// Every package built from a path: workspace members and path dependencies. + fn members(&self) -> Vec { + let (mut ok, mut out) = self.run("cargo", &["metadata", "--format-version", "1"]); + if !ok { + (ok, out) = self.run("cargo", &["metadata", "--no-deps", "--format-version", "1"]); + if !ok { + return Vec::new(); + } + } + let meta: serde_json::Value = serde_json::from_str(&out).unwrap_or_default(); + let names: BTreeSet = meta["packages"] + .as_array() + .into_iter() + .flatten() + .filter(|p| p.get("source").is_none_or(|s| s.is_null())) + .filter_map(|p| p["name"].as_str().map(str::to_owned)) + .collect(); + names.into_iter().collect() + } + + /// Build, and collect the .rmeta files Cargo reports for workspace members. + fn build(&self, target: &Path) -> Build { + let t = Instant::now(); + let mut c = self.cmd("cargo"); + c.args(["build", "--lib", "-j", &self.args.jobs, "--target-dir"]) + .arg(target) + .arg("--message-format=json-render-diagnostics"); + let r = cargo::run_group(c, None).unwrap_or_else(|e| cargo::Run { + ok: false, + stdout: String::new(), + stderr: format!("cargo did not start: {e}"), + hang: false, + }); + let (mut rmetas, mut fresh) = (BTreeMap::new(), Vec::new()); + for msg in messages(&r.stdout) { + if msg.reason != "compiler-artifact" || !msg.from_path() { + continue; + } + for f in &msg.filenames { + if f.ends_with(".rmeta") { + rmetas.insert(relative(f, target), std::fs::read(f).unwrap_or_default()); + if msg.fresh { + fresh.push(msg.target_name().to_owned()); + } + } + } + } + let log = &r.stderr; + Build { + ok: r.ok, + ice: cargo::is_ice(log), + secs: (t.elapsed().as_secs_f64() * 10.0).round() / 10.0, + log: tail(log, 6000).to_owned(), + reuse: log.lines().filter(|l| l.starts_with("rustc-verify-reuse:")).map(|l| l.chars().take(3000).collect()).collect(), + untracked: cargo::untracked_reads(log), + rmetas, + fresh, + art: artifacts::collect(&r.stdout, target), + } + } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let (src, target, inc_target) = (work.join("src"), work.join("target"), work.join("target-inc")); + let findings = work.join("findings"); + std::fs::create_dir_all(&findings)?; + if !src.exists() { + let ok = Command::new("git").args(["clone", "-q", &args.repo]).arg(&src).status()?.success(); + anyhow::ensure!(ok, "git clone {} failed", args.repo); + } + let ctx = Ctx { args: &args, src: src.clone() }; + let max = format!("--max-count={}", args.commits); + let mut commits: Vec = ctx.run("git", &["rev-list", "--first-parent", "--reverse", &max, "origin/HEAD"]).1.split_whitespace().map(str::to_owned).collect(); + if commits.is_empty() { + commits = ctx.run("git", &["rev-list", "--first-parent", "--reverse", &max, "HEAD"]).1.split_whitespace().map(str::to_owned).collect(); + } + let results = work.join("results.jsonl"); + let done: BTreeSet = std::fs::read_to_string(&results) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()) + .filter_map(|v| v["commit"].as_str().map(str::to_owned)) + .collect(); + let mut kept: BTreeMap = BTreeMap::new(); + let keep_dir = |i: usize, commit: &str| -> PathBuf { + let d = findings.join(format!("{i:05}-{}", &commit[..10.min(commit.len())])); + let _ = std::fs::create_dir_all(&d); + d + }; + + for (i, commit) in commits.iter().enumerate() { + if done.contains(commit) || i < args.start || args.end.is_some_and(|e| i > e) { + continue; + } + ctx.run("git", &["checkout", "-q", "--force", commit]); + ctx.run("git", &["clean", "-fdxq"]); + let date = ctx.run("git", &["log", "-1", "--format=%cs", commit]).1.trim().to_owned(); + let names = ctx.members(); + let inc = ctx.build(&target); + // Reads of untracked state, each listed once in /untracked.txt. + cargo::note_untracked(&work.join("untracked.txt"), &inc.untracked); + + // The clean build: the same target directory path, starting from a copy with the + // workspace members and the incremental cache removed. + let _ = std::fs::remove_dir_all(&inc_target); + if target.exists() { + std::fs::rename(&target, &inc_target)?; + copy_tree(&inc_target, &target, &[], true)?; + } else { + std::fs::create_dir_all(&inc_target)?; + } + // Cargo refuses to clean a directory it did not create; this one may have been created + // here when the first build failed early. + std::fs::create_dir_all(&target)?; + let tag = target.join("CACHEDIR.TAG"); + if !tag.exists() { + std::fs::write(&tag, "Signature: 8a477f597d28d172789f06886806bc55\n")?; + } + let target_str = target.to_string_lossy().into_owned(); + for name in &names { + ctx.run("cargo", &["clean", "-p", name, "--target-dir", &target_str]); + } + for e in std::fs::read_dir(&target)?.flatten() { + let inc_dir = e.path().join("incremental"); + if inc_dir.is_dir() { + std::fs::remove_dir_all(inc_dir)?; + } + } + let clean = ctx.build(&target); + + let (mut problems, mut differ): (Vec, Vec) = (Vec::new(), Vec::new()); + if inc.ice || clean.ice { + problems.push("ICE".into()); + } + if inc.ok != clean.ok { + problems.push("split".into()); + } + if !inc.reuse.is_empty() { + // The compiler's own check found something it reused stale. + problems.push("reuse".into()); + std::fs::write(keep_dir(i, commit).join("reuse.txt"), inc.reuse.join("\n"))?; + } + if !clean.fresh.is_empty() { + // A member the clean build did not compile again would be compared with itself. + problems.push("stale".into()); + } + if inc.ok && clean.ok && clean.fresh.is_empty() { + let (a, b) = (&inc.rmetas, &clean.rmetas); + let keys: BTreeSet<&String> = a.keys().chain(b.keys()).collect(); + differ = keys.into_iter().filter(|r| a.get(*r) != b.get(*r)).cloned().collect(); + // More oracles: object code in the rlibs and the diagnostics. + for (kind, detail) in artifacts::compare(&inc.art, &clean.art) { + if kind == "rlib" || kind == "diag" { + std::fs::write(keep_dir(i, commit).join(format!("{kind}.txt")), detail.join("\n"))?; + problems.push(kind.to_owned()); + } + } + if !differ.is_empty() { + problems.push("P6".into()); + for rel in &differ { + let file = Path::new(rel).file_name().map(|s| s.to_string_lossy().into_owned()).unwrap_or_default(); + let krate = file.split('-').next().unwrap_or("").to_owned(); + let n = kept.entry(krate).or_default(); + if *n < args.keep { + *n += 1; + let d = keep_dir(i, commit); + std::fs::write(d.join(format!("{file}.inc")), a.get(rel).map_or(&[][..], |v| v))?; + std::fs::write(d.join(format!("{file}.clean")), b.get(rel).map_or(&[][..], |v| v))?; + } + } + } + } + if !problems.is_empty() { + let d = keep_dir(i, commit); + std::fs::write(d.join("inc.log"), &inc.log)?; + std::fs::write(d.join("clean.log"), &clean.log)?; + } + + // Continue incrementally from the incremental build. + let _ = std::fs::remove_dir_all(&target); + std::fs::rename(&inc_target, &target)?; + let record = json!({ + "i": i, "commit": commit, "date": date, "members": names, + "inc": {"ok": inc.ok, "ice": inc.ice, "secs": inc.secs}, + "clean": {"ok": clean.ok, "ice": clean.ice, "secs": clean.secs}, + "compared": inc.rmetas.len(), "differ": differ, "problems": problems, + }); + let mut f = std::fs::OpenOptions::new().create(true).append(true).open(&results)?; + writeln!(f, "{record}")?; + let status = if inc.ok && clean.ok { + "both ok".to_owned() + } else { + format!("inc {}, clean {}", if inc.ok { "ok" } else { "failed" }, if clean.ok { "ok" } else { "failed" }) + }; + println!( + "{i:5} {date} {} {status}, {} compared, {:.1}s/{:.1}s {}", + &commit[..10.min(commit.len())], + inc.rmetas.len(), + inc.secs, + clean.secs, + problems.join(" ") + ); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/repro_diff.rs b/crates/mirth-lab/src/tools/repro_diff.rs new file mode 100644 index 0000000..0d7ca21 --- /dev/null +++ b/crates/mirth-lab/src/tools/repro_diff.rs @@ -0,0 +1,137 @@ +//! Determinism: what rustc writes must depend only on its inputs and options. +//! +//! Builds each standalone UI test that compiles several times and compares the outputs +//! (`.rmeta`, `.rlib` normalized member by member, executables) with the first build: +//! +//! - repeat: the same build again, in the same directory +//! - path: the same build in another directory, both with `--remap-path-prefix` to one name +//! - threads: `-Zthreads=8` (tests marked `ignore-parallel-frontend` skip it); known: async fns +//! (rust-lang/rust#162202), RPIT and impl Trait in traits (#163878) +//! - decoy: a `-L` directory holding unrelated libraries whose names start with the crate's +//! name (#159677's shape) + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::artifacts; +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Comma-separated subset of repeat, path, threads, decoy. + #[arg(long)] + variants: Option, + #[command(flatten)] + sweep: Sweep, +} + +const VARIANTS: &[&str] = &["repeat", "path", "threads", "decoy"]; +static OWN: LazyLock = + LazyLock::new(|| Regex::new(r"threads|remap-path|-o\b|--out-dir|emit|crate-name|extern|-L\b|-Cincremental").unwrap()); +static PARALLEL_IGNORED: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*ignore-parallel-frontend").unwrap()); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +/// Build the test copied into `src_dir`; its outputs as {file: digest}, or None if it fails. +fn build(args: &Args, test: &Test, src_dir: &Path, extra: &[String]) -> Option> { + let out = src_dir.join("out"); + let _ = std::fs::remove_dir_all(&out); + let _ = std::fs::create_dir_all(&out); + let emit = if test.kind == Some(Kind::CheckPass) { "metadata" } else { "link,metadata" }; + let source = src_dir.join(test.file_name()); + let mut extra_all = vec![ + "--out-dir".to_string(), + out.to_string_lossy().into_owned(), + "--crate-name".into(), + "t".into(), + format!("--remap-path-prefix={}=/src", src_dir.display()), + ]; + extra_all.extend_from_slice(extra); + let c = Compile::new(&args.rustc, &source, &out, &test.flags, test.edition()).extra(extra_all).emit(emit).unnamed_output().run(); + (c.status == Status::Ok).then(|| artifacts::digest_dir(&out)) +} + +fn check(args: &Args, variants: &[&str], test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, found: Vec::new() }; + let dir = driver::scratch_dir(&args.sweep); + let (a, b) = (dir.path().join("a"), dir.path().join("elsewhere-b")); + for d in [&a, &b] { + let _ = std::fs::create_dir_all(d); + let _ = std::fs::copy(&test.path, d.join(test.file_name())); + } + let Some(base) = build(args, test, &a, &[]) else { + rec.skip = Some("does not build".into()); + return rec; + }; + let mut found: Vec<(String, String)> = Vec::new(); + for &v in variants { + if v == "threads" && PARALLEL_IGNORED.is_match(&test.text) { + continue; + } + let got = match v { + "repeat" => build(args, test, &a, &[]), + "path" => build(args, test, &b, &[]), + "threads" => build(args, test, &a, &["-Zthreads=8".into()]), + "decoy" => { + let decoy = dir.path().join("decoy"); + let _ = std::fs::create_dir_all(&decoy); + let _ = std::fs::write(decoy.join("libtother.rlib"), b"!\n"); + let _ = std::fs::write(decoy.join("libt-0123456789abcdef.rmeta"), b"rust\0\0\0\0"); + build(args, test, &a, &["-L".into(), decoy.to_string_lossy().into_owned()]) + } + _ => continue, + }; + match got { + None => found.push((v.into(), "does not build".into())), + Some(g) if g != base => { + let differ: Vec<&str> = g.keys().chain(base.keys()).filter(|k| g.get(*k) != base.get(*k)).map(String::as_str).collect::>().into_iter().collect(); + found.push((v.into(), format!("outputs differ: {}", differ.join(", ")))); + } + _ => {} + } + } + // A difference also in `repeat` is the build's own nondeterminism: report only that. + if found.iter().any(|(v, _)| v == "repeat") { + found.retain(|(v, _)| v == "repeat"); + } + rec.found = found.iter().map(|(v, w)| format!("{v}: {w}")).collect(); + if !found.is_empty() { + let detail: Vec<_> = found.iter().map(|(v, w)| serde_json::json!({ "variant": v, "what": w })).collect(); + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": detail })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let variants: Vec<&str> = match &args.variants { + Some(v) => v.split(',').collect(), + None => VARIANTS.to_vec(), + }; + let kinds = [Some(Kind::BuildPass), Some(Kind::RunPass), Some(Kind::CheckPass)]; + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, &kinds, |t| uitest::flag_matches(t, &OWN))); + println!("{} tests, variants: {}", tests.len(), variants.join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &variants, t))) +} diff --git a/crates/mirth-lab/src/tools/rewrite_diff.rs b/crates/mirth-lab/src/tools/rewrite_diff.rs new file mode 100644 index 0000000..84e63b1 --- /dev/null +++ b/crates/mirth-lab/src/tools/rewrite_diff.rs @@ -0,0 +1,146 @@ +//! Equivalent rewrites: rewriting a program into an equivalent one must not change its verdict. +//! +//! Each standalone UI test is printed back unchanged (the `identity` rewrite: the baseline, +//! since printing drops comments and moves lines) and rewritten by each of mirth-rewrite's +//! rewrites, in process. Each version is compiled with lints capped (a full build when there is +//! a `fn main`), and a changed verdict is a finding; both rejecting with different error codes +//! is a note. Tests whose baseline differs from the original are left out (the printer cannot +//! represent them), as are tests that name files by relative path or have no `core`. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use mirth_rewrite::{Outcome, REWRITES, rewrite}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Comma-separated subset of rewrites. + #[arg(long)] + rewrites: Option, + #[command(flatten)] + sweep: Sweep, +} + +static NOT_MOVABLE: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path|#!\[no_core\]").unwrap()); +/// Item order matters to textual macro scoping: no reordering where macros are defined. +static ORDER_MATTERS: LazyLock = LazyLock::new(|| Regex::new(r"macro_rules!|#\[macro_use\]|macro\s+\w+").unwrap()); +/// Differences that are resource limits, or findings already recorded. +const NOISE: &[(&str, &str)] = &[ + ("consts/chained-constants-stackoverflow.rs", "reorder"), // 10,000 chained consts: query depth + ("consts/interior-mut-const-via-union.rs", "generic-wrap"), // finding 25 + // recursion_limit = "6": evaluation order nests the query stack one level deeper + ("traits/next-solver/overflow/dont-lower-depth-for-witness-and-rigid-opaque.rs", "reorder"), + ("imports/ambiguous-9.rs", "reorder"), // finding 28 + ("imports/ambiguous-14.rs", "reorder"), + ("imports/overwrite-different-ambig-2.rs", "reorder"), +]; + +#[derive(Serialize, Clone)] +struct Verdict { + status: Status, + codes: Vec, + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + applied: Vec, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn verdict(args: &Args, test: &Test, source: &std::path::Path, out: &std::path::Path) -> Verdict { + // A full build whenever there is a program: generic-wrap moves errors to monomorphization. + let emit = if test.text.contains("fn main") || !Kind::is_check(test.kind) { "link" } else { "metadata" }; + let c = Compile::new(&args.rustc, source, out, &test.flags, test.edition()) + .extra(["--cap-lints=warn"]) + .emit(emit) + .timeout(120) + .run(); + let stderr: String = c.stderr.chars().rev().take(2500).collect::().chars().rev().collect(); + Verdict { status: c.status, codes: rustc::error_codes(&c.stderr), stderr } +} + +fn check(args: &Args, names: &[&str], test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, applied: Vec::new(), found: Vec::new(), notes: Vec::new() }; + if NOT_MOVABLE.is_match(&test.text) { + rec.skip = Some("uses files by path or has no core".into()); + return rec; + } + let Outcome::Rewritten(identity) = rewrite("identity", &test.text) else { + rec.skip = Some("does not parse".into()); + return rec; + }; + let dir = driver::scratch_dir(&args.sweep); + let base_src = dir.path().join("identity.rs"); + let _ = std::fs::write(&base_src, &identity); + let original = verdict(args, test, &test.path, &dir.path().join("original")); + let base = verdict(args, test, &base_src, &dir.path().join("identity")); + if (original.status, &original.codes) != (base.status, &base.codes) { + rec.skip = Some("printing changes the verdict".into()); + return rec; + } + let mut kept: Vec<(String, Vec)> = vec![("identity.rs".into(), identity.into_bytes())]; + let (mut found, mut notes) = (Vec::new(), Vec::new()); + for &name in names { + if name == "identity" + || (name == "reorder" && ORDER_MATTERS.is_match(&test.text)) + || NOISE.contains(&(test.rel.as_str(), name)) + // generic_const_exprs requires bounds in generic contexts that a concrete one does not. + || (name == "generic-wrap" && test.text.contains("generic_const_exprs")) + { + continue; + } + let Outcome::Rewritten(text) = rewrite(name, &test.text) else { continue }; + rec.applied.push(name.into()); + let src = dir.path().join(format!("{name}.rs")); + let _ = std::fs::write(&src, &text); + let v = verdict(args, test, &src, &dir.path().join(name)); + let entry = |what: String| serde_json::json!({ "rewrite": name, "what": what, "codes": v.codes, "base_codes": base.codes, "stderr": v.stderr, "base_stderr": base.stderr }); + if v.status != base.status && v.status != Status::Timeout && base.status != Status::Timeout { + found.push(entry(format!("verdict: {:?} -> {:?}", base.status, v.status).to_lowercase())); + kept.push((format!("{name}.rs"), text.into_bytes())); + } else if v.status == Status::Error && base.status == Status::Error && v.codes != base.codes { + // Which error suppresses which may depend on order: a note. + notes.push(entry(format!("codes: {:?} -> {:?}", base.codes, v.codes))); + } + } + let label = |e: &serde_json::Value| format!("{}: {}", e["rewrite"].as_str().unwrap_or(""), e["what"].as_str().unwrap_or("")); + rec.found = found.iter().map(label).collect(); + rec.notes = notes.iter().map(label).collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &kept, &serde_json::json!({ "kind": test.kind, "found": found, "notes": notes })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let names: Vec<&str> = match &args.rewrites { + Some(r) => r.split(',').collect(), + None => REWRITES.to_vec(), + }; + let tests = args.sweep.select(uitest::tests(&args.sweep.tests, uitest::ALL, |_| false)); + println!("{} tests, rewrites: {}", tests.len(), names.iter().filter(|n| **n != "identity").copied().collect::>().join(", ")); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &names, t))) +} diff --git a/crates/mirth-lab/src/tools/scale_check.rs b/crates/mirth-lab/src/tools/scale_check.rs new file mode 100644 index 0000000..50f840b --- /dev/null +++ b/crates/mirth-lab/src/tools/scale_check.rs @@ -0,0 +1,282 @@ +//! Scaling and budgets: compile time, memory, future sizes and stack frames must grow at most +//! about linearly with the size of a program of a fixed shape. +//! +//! Each generator writes a program of size N for N in a doubling series. For each N: the +//! compiler's user CPU time and peak memory (wait4's rusage), the largest `sub $X, %rsp` in the +//! assembly, and for some shapes a size the program reports (`size_of_val` of a future). The +//! growth exponent k (value ~ N^k) is the log-log slope over the three largest sizes. Findings: +//! k above --max-exponent for time or memory, above 1.3 for a frame or a size, or a timeout. + +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode, Stdio}; +use std::sync::LazyLock; +use std::time::{Duration, Instant}; + +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + work: PathBuf, + /// Comma-separated shapes (default: all). + #[arg(long)] + only: Option, + #[arg(long, default_value = "0,2")] + opt: String, + #[arg(long, default_value = "50,100,200,400,800")] + sizes: String, + #[arg(long, default_value_t = 300)] + timeout: u64, + #[arg(long, default_value_t = 1.6)] + max_exponent: f64, + #[arg(long, default_value_t = 4)] + jobs: usize, +} + +fn g_fields(n: usize) -> String { + let mut f = String::new(); + for i in 0..n { + let _ = writeln!(f, " pub f{i}: u{},", 8 << (i % 4)); + } + format!("#[derive(Debug, Clone, PartialEq, Eq, Hash, Default, PartialOrd, Ord)]\npub struct S {{\n{f}}}\nfn main() {{ let s = S::default(); println!(\"{{}}\", format!(\"{{:?}}\", s.clone()).len()); }}\n") +} + +fn g_enum(n: usize) -> String { + let variants: String = (0..n).map(|i| format!(" V{i}(u32),\n")).collect(); + let arms: String = (0..n).map(|i| format!(" E::V{i}(x) => x + {i},\n")).collect(); + format!("#[derive(Debug, Clone, PartialEq)]\npub enum E {{\n{variants}}}\npub fn f(e: &E) -> u32 {{\n match *e {{\n{arms} }}\n}}\nfn main() {{ println!(\"{{}}\", f(&E::V0(1))); }}\n") +} + +fn g_nested_generic(n: usize) -> String { + let ty = (0..n).fold("u8".to_owned(), |t, _| format!("W<{t}>")); + format!("#[derive(Clone, Debug, Default)] pub struct W(T);\npub trait T {{ fn t(&self) -> usize; }}\nimpl T for u8 {{ fn t(&self) -> usize {{ 1 }} }}\nimpl T for W {{ fn t(&self) -> usize {{ self.0.t() + 1 }} }}\nfn main() {{ let v: {ty} = Default::default(); println!(\"{{}}\", v.t()); }}\n") +} + +fn g_iter_chain(n: usize) -> String { + let chain: String = (0..n).map(|i| format!(".map(|x| x.wrapping_add({i}))")).collect(); + format!("fn main() {{ let s: u64 = (0u64..10){chain}.sum(); println!(\"{{}}\", s); }}\n") +} + +fn g_async_forward(n: usize) -> String { + let mut fns = vec!["async fn f0(x: [u8; 64]) -> u8 { x[0] }".to_owned()]; + fns.extend((1..n).map(|i| format!("async fn f{i}(x: [u8; 64]) -> u8 {{ f{}(x).await }}", i - 1))); + format!("{}\nfn main() {{ let fut = f{}([1; 64]); println!(\"SIZE {{}}\", std::mem::size_of_val(&fut)); }}\n", fns.join("\n"), n - 1) +} + +fn g_seq_calls(n: usize) -> String { + let calls: String = (0..n).map(|i| format!(" let a{i} = big({i}); acc ^= a{i}[{}];\n", i % 512)).collect(); + format!("#[inline(never)] fn big(x: u64) -> [u64; 512] {{ [x; 512] }}\n#[inline(never)] pub fn many() -> u64 {{\n let mut acc = 0u64;\n{calls} acc\n}}\nfn main() {{ println!(\"{{}}\", many()); }}\n") +} + +fn g_trait_impls(n: usize) -> String { + let impls: String = (0..n).map(|i| format!("pub struct S{i}; impl Tr for S{i} {{ fn v(&self) -> u32 {{ {i} }} }}\n")).collect(); + let uses = (0..n).map(|i| format!("S{i}.v()")).collect::>().join(" + "); + format!("pub trait Tr {{ fn v(&self) -> u32; }}\n{impls}fn main() {{ println!(\"{{}}\", {uses}); }}\n") +} + +fn g_nested_expr(n: usize) -> String { + let expr = (0..n).fold("1u64".to_owned(), |e, i| format!("({e} + {})", i % 7)); + format!("fn main() {{ let x = std::hint::black_box({expr}); println!(\"{{}}\", x); }}\n") +} + +#[derive(Clone, Copy, PartialEq)] +enum Extra { + None, + RunSize, + Frame, +} + +/// (name, generator, what else to measure, largest meaningful N) +const SHAPES: &[(&str, fn(usize) -> String, Extra, usize)] = &[ + ("fields", g_fields, Extra::None, usize::MAX), + ("enum", g_enum, Extra::None, usize::MAX), + ("nested-generic", g_nested_generic, Extra::None, 120), + ("iter-chain", g_iter_chain, Extra::None, usize::MAX), + ("async-forward", g_async_forward, Extra::RunSize, 200), + ("seq-calls", g_seq_calls, Extra::Frame, usize::MAX), + ("trait-impls", g_trait_impls, Extra::None, usize::MAX), + ("nested-expr", g_nested_expr, Extra::None, 400), +]; + +#[derive(Serialize, Default, Clone)] +struct Row { + n: usize, + #[serde(skip_serializing_if = "Option::is_none")] + user: Option, + #[serde(skip_serializing_if = "Option::is_none")] + rss_kb: Option, + max_frame: u64, + #[serde(skip_serializing_if = "Option::is_none")] + run_size: Option, + timeout: bool, + #[serde(skip_serializing_if = "String::is_empty")] + error: String, +} + +/// Compile in `dir` with rusage from wait4; None user time on timeout. +fn compile_measured(args: &Args, dir: &Path, opt: u32) -> Row { + let mut child = match Command::new(&args.rustc) + .args(["m.rs", "--edition", "2021", &format!("-Copt-level={opt}"), "-o", "m", "--emit=link,asm"]) + .current_dir(dir) + .stdout(Stdio::null()) + .stderr(Stdio::piped()) + .spawn() + { + Ok(c) => c, + Err(e) => return Row { error: e.to_string(), ..Default::default() }, + }; + let pid = child.id() as libc::pid_t; + let deadline = Instant::now() + Duration::from_secs(args.timeout); + let mut status: libc::c_int = 0; + // SAFETY: rusage is plain data; wait4 fills it for the child we spawned. + let mut usage: libc::rusage = unsafe { std::mem::zeroed() }; + loop { + // SAFETY: waiting on our own child; status and usage outlive the call. + let r = unsafe { libc::wait4(pid, &mut status, libc::WNOHANG, &mut usage) }; + if r == pid { + break; + } + if Instant::now() > deadline { + let _ = child.kill(); + // SAFETY: reap the killed child. + unsafe { libc::wait4(pid, &mut status, 0, &mut usage) }; + return Row { timeout: true, ..Default::default() }; + } + std::thread::sleep(Duration::from_millis(50)); + } + let mut err = String::new(); + if let Some(mut e) = child.stderr.take() { + use std::io::Read; + let _ = e.read_to_string(&mut err); + } + if !(libc::WIFEXITED(status) && libc::WEXITSTATUS(status) == 0) { + return Row { error: err.chars().rev().take(500).collect::().chars().rev().collect(), ..Default::default() }; + } + let user = usage.ru_utime.tv_sec as f64 + usage.ru_utime.tv_usec as f64 / 1e6; + Row { user: Some(user), rss_kb: Some(usage.ru_maxrss), ..Default::default() } +} + +static SUB_RSP: LazyLock = LazyLock::new(|| Regex::new(r"subq?\s+\$(0x[0-9a-f]+|\d+),\s*%rsp").unwrap()); +static SIZE: LazyLock = LazyLock::new(|| Regex::new(r"SIZE (\d+)").unwrap()); + +fn measure(args: &Args, source: &str, opt: u32) -> Row { + let dir = tempfile::tempdir_in(&args.work).expect("scratch"); + let _ = std::fs::write(dir.path().join("m.rs"), source); + let mut row = compile_measured(args, dir.path(), opt); + if row.user.is_none() { + return row; + } + let asm = std::fs::read_to_string(dir.path().join("m.s")).unwrap_or_default(); + row.max_frame = SUB_RSP + .captures_iter(&asm) + .filter_map(|c| { + let v = &c[1]; + if let Some(h) = v.strip_prefix("0x") { u64::from_str_radix(h, 16).ok() } else { v.parse().ok() } + }) + .max() + .unwrap_or(0); + if let Ok(out) = Command::new(dir.path().join("m")).output() { + row.run_size = SIZE.captures(&String::from_utf8_lossy(&out.stdout)).and_then(|c| c[1].parse().ok()); + } + row +} + +/// The log-log slope over the three largest positive points. +fn exponent(points: &[(usize, f64)]) -> Option { + let pts: Vec<(f64, f64)> = points.iter().filter(|(_, v)| *v > 0.0).map(|&(n, v)| ((n as f64).ln(), v.ln())).collect(); + if pts.len() < 3 { + return None; + } + let pts = &pts[pts.len() - 3..]; + let mx = pts.iter().map(|p| p.0).sum::() / 3.0; + let my = pts.iter().map(|p| p.1).sum::() / 3.0; + let den: f64 = pts.iter().map(|p| (p.0 - mx).powi(2)).sum(); + (den > 0.0).then(|| pts.iter().map(|p| (p.0 - mx) * (p.1 - my)).sum::() / den) +} + +#[derive(Serialize)] +struct Res { + shape: String, + opt: u32, + rows: Vec, + k_time: Option, + k_rss: Option, + k_frame: Option, + k_size: Option, + found: Vec, +} + +fn one(args: &Args, sizes: &[usize], shape: &(&str, fn(usize) -> String, Extra, usize), opt: u32) -> Res { + let (name, gen_fn, extra, cap) = *shape; + let mut rows = Vec::new(); + for &n in sizes.iter().filter(|&&n| n <= cap) { + let mut r = measure(args, &gen_fn(n), opt); + r.n = n; + let stop = r.timeout || !r.error.is_empty(); + rows.push(r); + if stop { + break; + } + } + let ok: Vec<&Row> = rows.iter().filter(|r| r.user.is_some()).collect(); + let first_user = ok.first().and_then(|r| r.user).unwrap_or(0.0); + let k_time = exponent(&ok.iter().map(|r| (r.n, (r.user.unwrap() - first_user * 0.5).max(1e-3))).collect::>()); + let k_rss = exponent(&ok.iter().map(|r| (r.n, r.rss_kb.unwrap_or(0) as f64)).collect::>()); + let k_frame = (extra == Extra::Frame).then(|| exponent(&ok.iter().map(|r| (r.n, r.max_frame as f64)).collect::>())).flatten(); + let k_size = (extra == Extra::RunSize).then(|| exponent(&ok.iter().map(|r| (r.n, r.run_size.unwrap_or(0) as f64)).collect::>())).flatten(); + let mut found = Vec::new(); + if let Some(last) = rows.last() { + if last.timeout { + found.push(format!("timeout at N={}", last.n)); + } + if !last.error.is_empty() { + found.push(format!("error at N={}: {}", last.n, last.error.chars().rev().take(200).collect::().chars().rev().collect::())); + } + } + for (label, k, limit) in [("k_time", k_time, args.max_exponent), ("k_rss", k_rss, args.max_exponent), ("k_frame", k_frame, 1.3), ("k_size", k_size, 1.3)] { + if let Some(k) = k + && k > limit + { + found.push(format!("{label} = {k:.2}")); + } + } + Res { shape: name.into(), opt, rows, k_time, k_rss, k_frame, k_size, found } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let sizes: Vec = args.sizes.split(',').filter_map(|s| s.parse().ok()).collect(); + let opts: Vec = args.opt.split(',').filter_map(|s| s.parse().ok()).collect(); + let wanted: Option> = args.only.as_deref().map(|o| o.split(',').collect()); + let jobs: Vec<(&(&str, fn(usize) -> String, Extra, usize), u32)> = SHAPES + .iter() + .filter(|s| wanted.as_ref().is_none_or(|w| w.contains(&s.0))) + .flat_map(|s| opts.iter().map(move |&o| (s, o))) + .collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: Vec = pool.install(|| jobs.par_iter().map(|(s, o)| one(&args, &sizes, s, *o)).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + for r in &results { + let last = r.rows.iter().rev().find(|x| x.user.is_some()); + let ks: Vec = [("k_time", r.k_time), ("k_rss", r.k_rss), ("k_frame", r.k_frame), ("k_size", r.k_size)] + .iter() + .filter_map(|(l, k)| k.map(|k| format!("{l}={k:.2}"))) + .collect(); + println!( + "{:15} O{} N<={} user {:.1}s rss {}MB {} {}", + r.shape, + r.opt, + last.map_or(0, |x| x.n), + last.and_then(|x| x.user).unwrap_or(0.0), + last.and_then(|x| x.rss_kb).unwrap_or(0) / 1024, + ks.join(" "), + r.found.join("; ") + ); + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/solver_diff.rs b/crates/mirth-lab/src/tools/solver_diff.rs new file mode 100644 index 0000000..6d8c5ed --- /dev/null +++ b/crates/mirth-lab/src/tools/solver_diff.rs @@ -0,0 +1,138 @@ +//! Solver differential: the old and new trait solvers, and NLL and Polonius, must agree. +//! +//! Compiles each standalone UI test four ways (old solver = `-Znext-solver=coherence`, the +//! default new solver, each with `-Zpolonius=next`) and compares with the old solver and NLL: a +//! crash, a timeout or a different verdict is a finding; both rejecting with different error +//! codes is a note. A program accepted only by a non-reference configuration is interpreted with +//! Miri under that configuration: undefined behavior means the other one accepted something +//! unsound. Tests that name a solver or Polonius in their headers are left out. + +use std::path::PathBuf; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::miri::{Miri, MiriStatus}; +use mirth_lab::rustc::{self, Compile, Status}; +use mirth_lab::uitest::{self, Kind, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// The toolchain whose Miri interprets one-sided acceptances. + #[arg(long, default_value = "nightly-2026-10-06")] + miri_toolchain: String, + #[command(flatten)] + sweep: Sweep, +} + +const CONFIGS: &[(&str, &[&str])] = &[ + ("old", &["-Znext-solver=coherence"]), + ("next", &[]), + ("old-polonius", &["-Znext-solver=coherence", "-Zpolonius=next"]), + ("next-polonius", &["-Zpolonius=next"]), +]; +static OWN: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)next-solver|polonius|^//@\s*revisions:.*\bnext\b").unwrap()); + +#[derive(Serialize)] +struct Result1 { + status: Status, + codes: Vec, + stderr: String, +} + +#[derive(Serialize)] +struct Finding { + config: String, + what: String, + #[serde(skip_serializing_if = "Option::is_none")] + miri: Option, + stderr: String, +} + +#[derive(Serialize)] +struct Rec { + test: String, + status: Vec<(String, Status)>, + found: Vec, + notes: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn tail(s: &str, n: usize) -> String { + let v: Vec = s.chars().collect(); + v[v.len().saturating_sub(n)..].iter().collect() +} + +fn check(args: &Args, miri: &Miri, test: &Test) -> Rec { + let dir = driver::scratch_dir(&args.sweep); + // Metadata for check tests; build and run tests get a full build (monomorphization errors). + let emit = if Kind::is_check(test.kind) { "metadata" } else { "link" }; + let results: Vec<(&str, Result1)> = CONFIGS + .iter() + .map(|&(name, cfg)| { + let c = Compile::new(&args.rustc, &test.path, &dir.path().join(name), &test.flags, test.edition()) + .extra(cfg.iter().copied()) + .emit(emit) + .timeout(120) + .run(); + (name, Result1 { status: c.status, codes: rustc::error_codes(&c.stderr), stderr: tail(&c.stderr, 2500) }) + }) + .collect(); + let reference = &results[0].1; + let (mut found, mut notes) = (Vec::new(), Vec::new()); + for (name, r) in &results[1..] { + if r.status == Status::Ice && reference.status != Status::Ice { + found.push(Finding { config: name.to_string(), what: "ice".into(), miri: None, stderr: r.stderr.clone() }); + } else if r.status == Status::Timeout && reference.status != Status::Timeout { + found.push(Finding { config: name.to_string(), what: "timeout".into(), miri: None, stderr: String::new() }); + } else if matches!((r.status, reference.status), (Status::Ok, Status::Error) | (Status::Error, Status::Ok)) { + let accepted_by = if r.status == Status::Ok { name } else { "old" }; + let mut what = format!("verdict: old {:?}, {name} {:?}", reference.status, r.status).to_lowercase(); + let mut miri_status = None; + if test.text.contains("fn main") { + let cfg = CONFIGS.iter().find(|c| c.0 == accepted_by).map_or(&[][..], |c| c.1); + let extra: Vec = cfg.iter().map(|s| s.to_string()).collect(); + let m = miri.run(&test.path, &test.flags, test.edition(), &extra, 120, dir.path()); + if m.status == MiriStatus::Ub { + what.push_str(&format!("; Miri: UB under {accepted_by}")); + } + miri_status = Some(format!("{:?}", m.status)); + } + let stderr = if r.status == Status::Error { r.stderr.clone() } else { reference.stderr.clone() }; + found.push(Finding { config: name.to_string(), what, miri: miri_status, stderr }); + } else if r.status == Status::Error && reference.status == Status::Error && r.codes != reference.codes { + notes.push(format!("{name} codes {:?} vs {:?}", r.codes, reference.codes)); + } + } + let rec = Rec { + test: test.rel.clone(), + status: results.iter().map(|(n, r)| (n.to_string(), r.status)).collect(), + found: found.iter().map(|f| format!("{}: {}", f.config, f.what)).collect(), + notes, + }; + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &[], &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let miri = Miri::pinned(&args.miri_toolchain); + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| OWN.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests, configurations: old, next, old-polonius, next-polonius", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, &miri, t))) +} diff --git a/crates/mirth-lab/src/tools/suggest_diff.rs b/crates/mirth-lab/src/tools/suggest_diff.rs new file mode 100644 index 0000000..ec38573 --- /dev/null +++ b/crates/mirth-lab/src/tools/suggest_diff.rs @@ -0,0 +1,193 @@ +//! Suggestions apply: a machine-applicable suggestion must produce code that compiles the way +//! the suggestion promises. +//! +//! For each standalone UI test without `//@ run-rustfix` (compiletest checks those), applies +//! each `MachineApplicable` suggestion inside the test file alone and compiles again: +//! +//! - lint-breaks: a warning's (a lint's) fix introduces an error; lints never stop a build, and +//! `cargo fix` applies their fixes without asking +//! - parse: the fixed file no longer parses +//! - not-fixed: not one fewer of the same diagnostic +//! - ice: the fixed file crashes the compiler +//! +//! Errors appearing after an error's suggestion is applied are expected and not reported. + +use std::collections::BTreeSet; +use std::path::{Path, PathBuf}; +use std::process::ExitCode; +use std::sync::LazyLock; + +use mirth_lab::driver::{self, Record, Sweep}; +use mirth_lab::rustc::{self, Compile, Diagnostic, Status}; +use mirth_lab::uitest::{self, Test}; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + /// Include tests with `//@ run-rustfix`. + #[arg(long)] + with_rustfix: bool, + /// Suggestions tried per test. + #[arg(long, default_value_t = 8)] + max: usize, + #[command(flatten)] + sweep: Sweep, +} + +static BY_PATH: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path").unwrap()); +static RUSTFIX: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*run-rustfix").unwrap()); + +type Part = (usize, usize, String); + +#[derive(Serialize)] +struct Rec { + test: String, + #[serde(skip_serializing_if = "Option::is_none")] + skip: Option, + tried: usize, + found: Vec, +} + +impl Record for Rec { + fn findings(&self) -> Vec { + self.found.clone() + } + fn test(&self) -> &str { + &self.test + } +} + +fn diagnose(args: &Args, test: &Test, source: &Path, out: &Path) -> (Status, Vec) { + let c = Compile::new(&args.rustc, source, out, &test.flags, test.edition()) + .emit("metadata") + .json() + .timeout(120) + .run(); + (c.status, rustc::diagnostics(&c.stderr)) +} + +/// The machine-applicable suggestions of a diagnostic (and its direct children) in `file`. +fn suggestions(diag: &Diagnostic, file: &str) -> Vec> { + std::iter::once(diag) + .chain(diag.children.iter()) + .filter_map(|node| { + let mut parts: Vec = node + .spans + .iter() + .filter(|s| { + s.suggestion_applicability.as_deref() == Some("MachineApplicable") + && Path::new(&s.file_name).file_name().and_then(|n| n.to_str()) == Some(file) + }) + .filter_map(|s| s.suggested_replacement.clone().map(|r| (s.byte_start, s.byte_end, r))) + .collect(); + parts.sort(); + (!parts.is_empty()).then_some(parts) + }) + .collect() +} + +fn key(d: &Diagnostic) -> (String, String) { + (d.code().to_owned(), d.message.clone()) +} + +/// The errors, by code (or lint name) when they have one: a renamed identifier changes the +/// message of the same lint. +fn errors(diags: &[Diagnostic]) -> BTreeSet<(String, String)> { + diags + .iter() + .filter(|d| d.level == "error" && !d.message.starts_with("aborting")) + .map(|d| if d.code().is_empty() { (String::new(), d.message.clone()) } else { (d.code().to_owned(), String::new()) }) + .collect() +} + +fn check(args: &Args, test: &Test) -> Rec { + let mut rec = Rec { test: test.rel.clone(), skip: None, tried: 0, found: Vec::new() }; + if BY_PATH.is_match(&test.text) { + rec.skip = Some("uses files by path".into()); + return rec; + } + let dir = driver::scratch_dir(&args.sweep); + let src = dir.path().join(test.file_name()); + let _ = std::fs::copy(&test.path, &src); + let (status, diags) = diagnose(args, test, &src, &dir.path().join("orig")); + if matches!(status, Status::Ice | Status::Timeout) { + rec.skip = Some(format!("original {status:?}").to_lowercase()); + return rec; + } + let base_errors = errors(&diags); + let text = std::fs::read(&src).unwrap_or_default(); + let mut found: Vec = Vec::new(); + let mut kept: Vec<(String, Vec)> = Vec::new(); + 'outer: for diag in &diags { + for parts in suggestions(diag, test.file_name()) { + if rec.tried >= args.max { + break 'outer; + } + rec.tried += 1; + // Apply from the end so earlier offsets stay valid; overlapping parts are skipped. + let mut fixed = text.clone(); + let mut last: Option = None; + let mut ok = true; + for (start, end, repl) in parts.iter().rev() { + if last.is_some_and(|l| *end > l) || *end > fixed.len() || start > end { + ok = false; + break; + } + fixed.splice(*start..*end, repl.bytes()); + last = Some(*start); + } + if !ok { + continue; + } + let fdir = dir.path().join(format!("fix{}", rec.tried)); + let _ = std::fs::create_dir_all(&fdir); + let fsrc = fdir.join(test.file_name()); + let _ = std::fs::write(&fsrc, &fixed); + let (fstatus, fdiags) = diagnose(args, test, &fsrc, &fdir); + let new: BTreeSet<_> = errors(&fdiags).difference(&base_errors).cloned().collect(); + let what = if fstatus == Status::Ice { + Some("ice") + } else if new.iter().any(|(_, m)| m.contains("expected") || m.contains("unexpected") || m.contains("unknown start of token")) { + Some("parse") + } else if diag.level == "warning" && !new.is_empty() { + Some("lint-breaks") + } else if fdiags.iter().filter(|d| key(d) == key(diag)).count() >= diags.iter().filter(|d| key(d) == key(diag)).count() { + // Nested braces legitimately report the next level, but then there is one fewer. + Some("not-fixed") + } else { + None + }; + if let Some(what) = what { + let name = format!("fix{}.rs", rec.tried); + found.push(serde_json::json!({ + "what": what, "diagnostic": diag.message, "code": diag.code(), "level": diag.level, + "parts": parts, "fixed_name": name, + "new_errors": new.iter().map(|(c, m)| if c.is_empty() { m.clone() } else { c.clone() }).take(5).collect::>(), + })); + kept.push((name, fixed)); + } + } + } + rec.found = found + .iter() + .map(|f| { + let who = if f["code"].as_str().unwrap_or("").is_empty() { f["level"].as_str().unwrap_or("") } else { f["code"].as_str().unwrap_or("") }; + format!("{}: {} {}", f["what"].as_str().unwrap_or(""), who, f["diagnostic"].as_str().unwrap_or("").chars().take(80).collect::()) + }) + .collect(); + if !found.is_empty() { + driver::write_finding(&args.sweep.work, test, &kept, &serde_json::json!({ "found": found })); + } + rec +} + +pub fn run(args: Args) -> anyhow::Result { + let tests = uitest::tests(&args.sweep.tests, uitest::ALL, |t| !args.with_rustfix && RUSTFIX.is_match(&t.text)); + let tests = args.sweep.select(tests); + println!("{} tests", tests.len()); + Ok(driver::drive(&tests, &args.sweep, |t| check(&args, t))) +} diff --git a/crates/mirth-lab/src/tools/ui_coverage.rs b/crates/mirth-lab/src/tools/ui_coverage.rs new file mode 100644 index 0000000..5d9a61f --- /dev/null +++ b/crates/mirth-lab/src/tools/ui_coverage.rs @@ -0,0 +1,270 @@ +//! Which compiler functions each of rustc's UI tests reaches that a baseline (the fixture's +//! builds) does not, with a coverage-instrumented rustc (rustc/coverage.toml); then a small set +//! of tests that reaches the most of them. +//! +//! `run` compiles each test file the way its `//@` headers say, as far as one rustc call can: +//! `compile-flags`, `edition`, the first of `revisions` (as `--cfg` with its own flags), metadata +//! only for tests that do not build (check-pass, and tests expected to fail before codegen), a +//! full build otherwise. Tests that need auxiliary crates, proc macros, another target or +//! `minicore` are skipped. Writes /tests.jsonl: per test, whether it compiled and the +//! indices (into /functions.json) of the functions it reached beyond the baseline. +//! +//! `pick` chooses tests greedily, each adding the most functions not yet reached, and writes +//! /picked.json. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::io::Write; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{LazyLock, Mutex}; +use std::time::Duration; + +use mirth_lab::coverage::{self, to_json_indent, to_json_line}; +use mirth_lab::rustc::{is_ice, run_command, Exit}; +use mirth_lab::uitest::{self, Kind}; +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; +use walkdir::WalkDir; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[command(subcommand)] + cmd: Cmd, +} + +#[derive(clap::Subcommand, Debug)] +enum Cmd { + /// Compile each UI test; record the functions it reaches beyond the baseline. + Run(RunArgs), + /// Choose the tests that together reach the most. + Pick(PickArgs), +} + +#[derive(clap::Args, Debug)] +struct RunArgs { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + tests: PathBuf, + #[arg(long)] + sites: PathBuf, + /// Log directories (or directories of them) of the baseline. + #[arg(long)] + baseline: Vec, + #[arg(long)] + out: PathBuf, + #[arg(long, default_value_t = 6)] + jobs: usize, + #[arg(long, default_value_t = 0)] + limit: usize, +} + +#[derive(clap::Args, Debug)] +struct PickArgs { + #[arg(long)] + out: PathBuf, + #[arg(long, default_value_t = 200)] + count: usize, +} + +static SKIP: LazyLock = LazyLock::new(|| { + Regex::new(r"(?m)^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|only-|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|needs-asm-support|known-bug)").unwrap() +}); +/// `only-` directives this host satisfies. +const HOST_ONLY: &[&str] = &["x86_64", "linux", "unix", "64bit"]; + +fn skipped(text: &str) -> bool { + SKIP.captures_iter(text).any(|c| { + let m = c.get(1).unwrap(); + m.as_str() != "only-" || !HOST_ONLY.iter().any(|h| text[m.end()..].starts_with(h)) + }) +} + +#[derive(Serialize)] +#[serde(untagged)] +enum Outcome { + Ran { test: String, status: String, kind: Option, error: String, new: Vec }, + Skipped { test: String, status: String }, +} + +#[derive(Serialize)] +struct Functions<'a> { + functions: Vec<[&'a str; 2]>, + baseline: Vec, +} + +fn run_one(a: &RunArgs, path: &Path, index: &BTreeMap, baseline: &HashSet) -> Outcome { + let text = String::from_utf8_lossy(&std::fs::read(path).unwrap_or_default()).into_owned(); + let rel = path.strip_prefix(&a.tests).unwrap_or(path).to_string_lossy().into_owned(); + if skipped(&text) { + return Outcome::Skipped { test: rel, status: "skipped".into() }; + } + let (flags, edition, kind, _) = uitest::headers(&text); + let d = tempfile::tempdir_in(a.out.join("scratch")).expect("scratch directory"); + let emit = if Kind::is_check(kind) { "--emit=metadata" } else { "--emit=link" }; + let logs = d.path().join("logs"); + let mut cmd = Command::new(&a.rustc); + cmd.arg(path) + .args(["--edition", edition.as_deref().unwrap_or("2015"), emit, "--out-dir"]) + .arg(d.path()) + .args(["-Zunstable-options", "-Ainternal_features"]) + .args(&flags) + .env("MIRTH_OUT", &logs) + .env("RUSTC_BOOTSTRAP", "1") + .current_dir(d.path()); + let (status, error) = match run_command(cmd, Duration::from_secs(120)) { + Ok(f) if f.exit == Exit::Timeout => ("timeout".to_owned(), String::new()), + Ok(f) => { + let stderr = f.stderr_text(); + let status = if f.success() { "ok" } else if is_ice(&stderr) { "ice" } else { "error" }; + let first: String = stderr.lines().find(|l| l.starts_with("error")).unwrap_or("").chars().take(160).collect(); + (status.to_owned(), first) + } + Err(e) => ("error".to_owned(), e.to_string()), + }; + let (hit, _) = coverage::hits(&logs); + let mut new: Vec = hit.iter().filter(|s| !baseline.contains(*s)).filter_map(|s| index.get(s).copied()).collect(); + new.sort(); + Outcome::Ran { test: rel, status, kind, error, new } +} + +fn run_tests(a: RunArgs) -> anyhow::Result<()> { + let mut functions: Vec<(String, String, String)> = coverage::functions(&a.sites).into_iter().map(|f| (f.site, f.path, f.span)).collect(); + functions.sort(); + let index: BTreeMap = functions.iter().enumerate().map(|(i, f)| (f.0.clone(), i)).collect(); + let mut baseline: HashSet = HashSet::new(); + for b in &a.baseline { + let mut dirs = vec![b.clone()]; + if let Ok(entries) = std::fs::read_dir(b) { + dirs.extend(entries.flatten().map(|e| e.path()).filter(|p| !p.file_name().is_some_and(|n| n.to_string_lossy().starts_with('.')))); + } + for d in dirs.iter().filter(|d| d.is_dir()) { + baseline.extend(coverage::hits(d).0); + } + } + std::fs::create_dir_all(a.out.join("scratch"))?; + let mut base_idx: Vec = baseline.iter().filter_map(|s| index.get(s).copied()).collect(); + base_idx.sort(); + let info = Functions { functions: functions.iter().map(|f| [f.1.as_str(), f.2.as_str()]).collect(), baseline: base_idx.clone() }; + std::fs::write(a.out.join("functions.json"), to_json_line(&info))?; + let results = a.out.join("tests.jsonl"); + let done: HashSet = std::fs::read_to_string(&results) + .unwrap_or_default() + .lines() + .filter_map(|l| serde_json::from_str::(l).ok()) + .filter_map(|v| v["test"].as_str().map(str::to_owned)) + .collect(); + let mut tests: Vec = WalkDir::new(&a.tests) + .into_iter() + .filter_map(Result::ok) + .map(|e| e.into_path()) + .filter(|p| p.extension().is_some_and(|x| x == "rs") && p.is_file()) + .filter(|p| { + let rel = p.strip_prefix(&a.tests).unwrap_or(p); + !rel.components().any(|c| c.as_os_str() == "auxiliary") && !done.contains(&*rel.to_string_lossy()) + }) + .collect(); + tests.sort(); + if a.limit > 0 { + tests.truncate(a.limit); + } + println!("{} functions, {} in the baseline; {} tests to run", functions.len(), base_idx.len(), tests.len()); + let out = Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(&results)?); + let n = AtomicUsize::new(0); + let pool = rayon::ThreadPoolBuilder::new().num_threads(a.jobs).build()?; + pool.install(|| { + tests.par_iter().for_each(|t| { + let res = run_one(&a, t, &index, &baseline); + let mut f = out.lock().unwrap(); + let _ = writeln!(f, "{}", to_json_line(&res)); + let k = n.fetch_add(1, Ordering::Relaxed); + if k % 500 == 0 { + let _ = f.flush(); + println!("{k} tests"); + } + }) + }); + let _ = std::fs::remove_dir_all(a.out.join("scratch")); + Ok(()) +} + +/// What `pick` reads of a line of tests.jsonl. +#[derive(Deserialize)] +struct Line { + test: String, + status: String, + #[serde(default)] + new: Vec, +} + +#[derive(Deserialize)] +struct Info { + functions: Vec, + baseline: Vec, +} + +#[derive(Serialize)] +struct Picked { + test: String, + status: String, + adds: usize, + total: usize, +} + +fn pick(a: PickArgs) -> anyhow::Result<()> { + let info: Info = serde_json::from_str(&std::fs::read_to_string(a.out.join("functions.json"))?)?; + // In file order: a tie goes to the test listed first, as Python's max does. + let mut sets: Vec<(String, BTreeSet)> = Vec::new(); + let mut status: BTreeMap = BTreeMap::new(); + let mut reached = BTreeSet::new(); + for line in std::fs::read_to_string(a.out.join("tests.jsonl"))?.lines() { + let Ok(Line { test, status: s, new }) = serde_json::from_str(line) else { continue }; + if new.is_empty() { + continue; + } + reached.extend(new.iter().copied()); + let set: BTreeSet = new.into_iter().collect(); + match sets.iter_mut().find(|(t, _)| *t == test) { + Some(entry) => entry.1 = set, + None => sets.push((test.clone(), set)), + } + status.insert(test, s); + } + let mut covered: BTreeSet = BTreeSet::new(); + let mut picked = Vec::new(); + while picked.len() < a.count && !sets.is_empty() { + let gains: Vec = sets.iter().map(|(_, s)| s.difference(&covered).count()).collect(); + let best = (0..sets.len()).fold(0, |b, i| if gains[i] > gains[b] { i } else { b }); + if gains[best] == 0 { + break; + } + let (test, set) = sets.remove(best); + covered.extend(set); + picked.push(Picked { status: status[&test].clone(), test, adds: gains[best], total: covered.len() }); + } + let total = info.functions.len(); + let base = info.baseline.len(); + println!( + "baseline {base} of {total} functions ({:.1}%); all tests reach {} more; {} picked tests reach {} more ({:.1}% in all)", + 100.0 * base as f64 / total as f64, + reached.len(), + picked.len(), + covered.len(), + 100.0 * (base + covered.len()) as f64 / total as f64 + ); + for t in picked.iter().take(40) { + println!(" +{:5} {:6} {:7} {}", t.adds, t.total, t.status, t.test); + } + std::fs::write(a.out.join("picked.json"), to_json_indent(&picked, 1))?; + Ok(()) +} + +pub fn run(args: Args) -> anyhow::Result { + match args.cmd { + Cmd::Run(a) => run_tests(a)?, + Cmd::Pick(a) => pick(a)?, + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/ui_fuzz.rs b/crates/mirth-lab/src/tools/ui_fuzz.rs new file mode 100644 index 0000000..8afa0ab --- /dev/null +++ b/crates/mirth-lab/src/tools/ui_fuzz.rs @@ -0,0 +1,280 @@ +//! Incremental rebuilds of rustc's UI tests, most of which fail to compile on purpose: the +//! fuzzer's edits applied to each test file, each rebuild compared with a clean build of the same +//! source, so that error reporting and recovery are exercised under incremental compilation. +//! +//! For each test in the list (ui-coverage's pick), compiled the way its `//@` headers say: build +//! it incrementally, then repeatedly apply a random edit (`mutations`) and rebuild it +//! incrementally, build the edited file again with a fresh incremental directory (same file, same +//! working directory), and compare: +//! +//! status both succeed, both fail, or both crash +//! diag the diagnostics, with paths and the incremental directory taken out +//! output the .rmeta and .rlib (normalized as `artifacts` does) when both succeed +//! ice both crash or neither does +//! +//! A clean build is made again before a difference counts (nondeterminism: such findings are +//! marked P5). Findings go to /findings/-/ with the source, both outputs and the +//! edit history. + +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::io::Write as _; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::{LazyLock, Mutex}; +use std::time::Duration; + +use mirth_lab::artifacts::{normalized_rlib, sha256}; +use mirth_lab::rustc::{Exit, is_ice, run_command}; +use mirth_lab::uitest::{self, Kind}; +use mirth_lab::mutations; +use rayon::prelude::*; +use regex::Regex; +use serde::Serialize; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + rustc: PathBuf, + #[arg(long)] + tests: PathBuf, + /// The tests to fuzz: a JSON list of paths, or of objects with a "test" path. + #[arg(long)] + list: PathBuf, + #[arg(long)] + work: PathBuf, + #[arg(long, default_value_t = 20)] + edits: usize, + #[arg(long, default_value_t = 8)] + jobs: usize, + /// Extra rustc options for every build. + #[arg(long, default_value = "", allow_hyphen_values = true)] + flags: String, + #[arg(long)] + pause_on_finding: bool, +} + +/// Tests that hit bugs already in docs/hunt.md every time: finding 17. +const KNOWN: &[&str] = &["unleash-the-miri-inside-of-you"]; + +#[derive(PartialEq)] +enum Digest { + File(String), + Rlib(BTreeMap), +} + +struct Build { + code: i32, + ice: bool, + diag: Vec, + files: BTreeMap, + stderr: String, +} + +static COUNT: LazyLock = LazyLock::new(|| Regex::new(r"\(\d+\)").unwrap()); +static NON_WORD: LazyLock = LazyLock::new(|| Regex::new(r"\W").unwrap()); + +struct Ctx<'a> { + args: &'a Args, + work: PathBuf, +} + +impl Ctx<'_> { + /// Compile `source` (a file in `dir`) with outputs in `dir/out`. + fn build(&self, dir: &Path, source: &Path, flags: &[String], edition: &str, kind: Option, incremental: Option<&Path>) -> Build { + let out = dir.join("out"); + let _ = std::fs::remove_dir_all(&out); + let _ = std::fs::create_dir_all(&out); + let emit = if Kind::is_check(kind) { "--emit=metadata" } else { "--emit=link,metadata" }; + let mut cmd = Command::new(&self.args.rustc); + cmd.arg(source.file_name().unwrap()) + .args(["--edition", edition, emit, "--out-dir", "out", "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features"]) + .arg("--error-format=short"); + // Before the test's own flags, which may end with an option expecting a value. + if let Some(i) = incremental { + cmd.arg(format!("-Cincremental={}", i.display())); + } + cmd.args(flags).args(self.args.flags.split_whitespace()); + cmd.current_dir(dir).env("RUSTC_BOOTSTRAP", "1").env("RUST_BACKTRACE", "0"); + let (code, mut err) = match run_command(cmd, Duration::from_secs(300)) { + Ok(f) => match f.exit { + Exit::Code(c) => (c, f.stderr_text()), + Exit::Signal(s) => (-s, f.stderr_text()), + Exit::Timeout => (-1, "timeout".to_owned()), + }, + Err(e) => (-1, e.to_string()), + }; + let ice = is_ice(&err); + if let Some(i) = incremental { + err = err.replace(&i.display().to_string(), ""); + } + let diag: BTreeSet = err + .lines() + .filter(|l| !l.is_empty() && !["note: ", " ", "query stack", "#"].iter().any(|p| l.starts_with(p))) + .map(|l| COUNT.replace_all(l, "(…)").into_owned()) + .collect(); + let mut files = BTreeMap::new(); + for e in std::fs::read_dir(&out).into_iter().flatten().flatten() { + let p = e.path(); + let name = e.file_name().to_string_lossy().into_owned(); + match p.extension().and_then(|x| x.to_str()) { + Some("rmeta") => { + files.insert(name, Digest::File(sha256(&std::fs::read(&p).unwrap_or_default()))); + } + Some("rlib") => { + files.insert(name, Digest::Rlib(normalized_rlib(&p))); + } + _ => {} + } + } + Build { code, ice, diag: diag.into_iter().collect(), files, stderr: err } + } + + fn fuzz_one(&self, test: &str) -> anyhow::Result { + if self.work.join("PAUSED").exists() { + return Ok("not run".into()); + } + let path = self.args.tests.join(test); + let text = String::from_utf8_lossy(&std::fs::read(&path)?).into_owned(); + let (flags, edition, kind, _) = uitest::headers(&text); + let edition = edition.as_deref().unwrap_or("2015"); + if KNOWN.iter().any(|k| text.contains(k)) { + return Ok("skipped (known)".into()); + } + let name = format!("{}-{}", NON_WORD.replace_all(test, "_"), &sha256(test.as_bytes())[..8]); + let home = self.work.join("w").join(&name); + let _ = std::fs::remove_dir_all(&home); + // The clean build uses the same directory and file, with a fresh incremental directory, + // so that nothing but incremental state tells the two builds apart. + let dir = home.join("src"); + std::fs::create_dir_all(&dir)?; + let file_name = path.file_name().unwrap(); + let src = dir.join(file_name); + std::fs::write(&src, &text)?; + let (incr, incr_clean) = (home.join("incr"), home.join("incr-clean")); + let mut rng = mutations::seeded(test); + let mut history: Vec = Vec::new(); + self.build(&dir, &src, &flags, edition, kind, Some(&incr)); + let mut found_any = 0; + for n in 0..self.args.edits { + let old = std::fs::read_to_string(&src)?; + let (edit, f) = mutations::pick(&mut rng); + let Some(new) = f(&old, &mut rng, n).filter(|new| *new != old) else { continue }; + std::fs::write(&src, &new)?; + let diff = similar::TextDiff::from_lines(&old, &new).unified_diff().header("a", "b").to_string(); + history.push(serde_json::json!({"edit": edit, "diff": diff})); + let inc = self.build(&dir, &src, &flags, edition, kind, Some(&incr)); + let _ = std::fs::remove_dir_all(&incr_clean); + let clean = self.build(&dir, &src, &flags, edition, kind, Some(&incr_clean)); + let mut found = compare(&inc, &clean); + if !found.is_empty() && !found.iter().any(|f| f.starts_with("ICE")) { + let _ = std::fs::remove_dir_all(&incr_clean); + let again = self.build(&dir, &src, &flags, edition, kind, Some(&incr_clean)); + if !compare(&clean, &again).is_empty() { + found = found.into_iter().map(|f| format!("P5 {f}")).collect(); + } + } + if found.is_empty() { + continue; + } + found_any += 1; + let d = self.work.join("findings").join(format!("{name}-{n}")); + std::fs::create_dir_all(&d)?; + std::fs::write(d.join(file_name), &new)?; + let detail = serde_json::json!({"test": test, "found": found, "flags": flags, "edition": edition, + "kind": kind, "history": history}); + std::fs::write(d.join("finding.json"), to_json(&detail))?; + std::fs::write(d.join("inc.stderr"), &inc.stderr)?; + std::fs::write(d.join("clean.stderr"), &clean.stderr)?; + if self.args.pause_on_finding && !found.iter().all(|f| f.starts_with("P5") || f.starts_with("known")) { + std::fs::write(self.work.join("PAUSED"), to_json(&serde_json::json!({"test": test, "edit": n, "found": found})))?; + break; + } + } + let _ = std::fs::remove_dir_all(&home); + Ok(format!("{} edits, {found_any} findings", history.len())) + } +} + +/// JSON indented by one space, as the findings have always been written. +fn to_json(v: &impl Serialize) -> Vec { + let mut out = Vec::new(); + let mut ser = serde_json::Serializer::with_formatter(&mut out, serde_json::ser::PrettyFormatter::with_indent(b" ")); + v.serialize(&mut ser).expect("serializable"); + out +} + +fn compare(inc: &Build, clean: &Build) -> Vec { + let mut found = Vec::new(); + // Some tests crash the compiler on purpose; only a crash on one side counts. + if inc.ice != clean.ice { + found.push(format!("ICE {} only", if inc.ice { "incremental" } else { "clean" })); + } + if (inc.code == 0) != (clean.code == 0) { + found.push(format!("status: incremental {}, clean {}", inc.code, clean.code)); + } + let message = |l: &str| -> String { + let l = l.split_once(": error").map_or(l, |x| x.1); + l.split_once(": warning").map_or(l, |x| x.1).to_owned() + }; + let summary = |l: &str| l.starts_with("error: aborting") || (l.starts_with("warning:") && l.contains("emitted")); + if inc.diag != clean.diag { + let clean_set: HashSet<&String> = clean.diag.iter().collect(); + let inc_messages: HashSet = inc.diag.iter().map(|d| message(d)).collect(); + let fewer = inc.code != 0 + && clean.code != 0 + && inc.diag.iter().filter(|d| !summary(d)).all(|d| clean_set.contains(d)) + && clean.diag.iter().filter(|d| !inc.diag.contains(d)).all(|d| inc_messages.contains(&message(d)) || summary(d)); + if fewer { + found.push("known diag (finding 18): the rebuild stopped at a fatal error sooner".into()); + } else { + let only_inc: Vec<&String> = inc.diag.iter().filter(|d| !clean.diag.contains(d)).take(3).collect(); + let only_clean: Vec<&String> = clean.diag.iter().filter(|d| !inc.diag.contains(d)).take(3).collect(); + found.push(format!("diag: incremental only {only_inc:?}; clean only {only_clean:?}")); + } + } + if inc.code == 0 && clean.code == 0 && inc.files != clean.files { + let names: BTreeSet<&String> = inc.files.keys().chain(clean.files.keys()).collect(); + let differ: Vec<&str> = names.into_iter().filter(|k| inc.files.get(*k) != clean.files.get(*k)).map(String::as_str).collect(); + found.push(format!("output: {}", differ.join(", "))); + } + found +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let work = std::fs::canonicalize(&args.work)?; + let _ = std::fs::remove_file(work.join("PAUSED")); + let ctx = Ctx { args: &args, work: work.clone() }; + // A compiler without its standard library fails every test the same way: stop instead. + let probe_dir = work.join("probe"); + std::fs::create_dir_all(&probe_dir)?; + std::fs::write(probe_dir.join("probe.rs"), "fn main() {}\n")?; + let probe = ctx.build(&probe_dir, &probe_dir.join("probe.rs"), &[], "2021", Some(Kind::BuildPass), None); + if probe.code != 0 { + eprintln!("{} cannot build an empty program:\n{}", args.rustc.display(), probe.stderr); + return Ok(ExitCode::from(1)); + } + let picked: Vec = serde_json::from_slice(&std::fs::read(&args.list)?)?; + let tests: Vec = picked + .iter() + .filter_map(|t| t.get("test").unwrap_or(t).as_str().map(str::to_owned)) + .collect(); + let done_path = work.join("done.txt"); + let done: HashSet = std::fs::read_to_string(&done_path).unwrap_or_default().split_whitespace().map(str::to_owned).collect(); + let log = Mutex::new(std::fs::OpenOptions::new().create(true).append(true).open(&done_path)?); + let todo: Vec<&String> = tests.iter().filter(|t| !done.contains(*t)).collect(); + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + pool.install(|| { + todo.par_iter().for_each(|test| { + // A broken test or harness case must not stop the run. + let result = ctx.fuzz_one(test).unwrap_or_else(|e| format!("harness error: {e:#}").chars().take(300).collect()); + println!("{test}: {result}"); + if !result.starts_with("not run") && !work.join("PAUSED").exists() { + let mut log = log.lock().unwrap(); + let _ = writeln!(log, "{test}"); + let _ = log.flush(); + } + }) + }); + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/tools/xlink.rs b/crates/mirth-lab/src/tools/xlink.rs new file mode 100644 index 0000000..79e4529 --- /dev/null +++ b/crates/mirth-lab/src/tools/xlink.rs @@ -0,0 +1,175 @@ +//! Cross-target build and link: every target must build `core` and `alloc` and link a program +//! with its documented linker, with no undefined symbols. +//! +//! For each target, builds rustc/xlink-probe (a `no_std` program needing compiler-builtins: 128-bit +//! integers, float conversions and math, float formatting, large copies, atomics) with +//! `cargo -Zbuild-std=core,alloc` and links it; targets whose default linker is a C compiler +//! driver link with `rust-lld` in the flavor their spec names. Results: ok, link-undefined (the +//! finding: lld reports undefined symbols), link, env (a library or startup file of the target's +//! C sysroot missing here), build, ice, skipped. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; +use std::process::{Command, ExitCode}; +use std::sync::LazyLock; +use std::time::Duration; + +use mirth_lab::rustc::{Exit, run_command}; +use rayon::prelude::*; +use regex::Regex; +use serde::{Deserialize, Serialize}; + +#[derive(clap::Args, Debug)] +pub struct Args { + #[arg(long)] + toolchain: String, + #[arg(long)] + work: PathBuf, + /// Comma-separated targets (default: every target rustc knows). + #[arg(long)] + targets: Option, + #[arg(long, default_value_t = 6)] + jobs: usize, + /// The probe crate (default: rustc/xlink-probe in the mirth checkout). + #[arg(long)] + probe: Option, +} + +/// Targets that need more than a target name (a CPU, a linker that is not lld). +static SKIP: LazyLock = LazyLock::new(|| Regex::new(r"^(amdgcn|nvptx|bpf|spirv)|avr-none").unwrap()); +static UNDEFINED: LazyLock = LazyLock::new(|| Regex::new(r"undefined symbol: (\S+)").unwrap()); +static ENV_MISSING: LazyLock = LazyLock::new(|| Regex::new(r"unable to find library|cannot open crt|cannot open .*\.o\b|No such file").unwrap()); + +/// The parts of a target spec the link options depend on. +#[derive(Deserialize, Default)] +#[serde(rename_all = "kebab-case")] +struct Spec { + #[serde(default)] + linker_flavor: String, + #[serde(default)] + linker: String, + #[serde(default)] + is_like_wasm: bool, + #[serde(default)] + is_like_msvc: bool, + #[serde(default)] + is_like_darwin: bool, +} + +#[derive(Serialize)] +struct Res { + result: String, + #[serde(skip_serializing_if = "Vec::is_empty")] + flags: Vec, + #[serde(skip_serializing_if = "Vec::is_empty")] + undefined: Vec, + #[serde(skip_serializing_if = "String::is_empty")] + first: String, + #[serde(skip_serializing_if = "String::is_empty")] + tail: String, +} + +fn link_flags(spec: &Spec) -> Vec { + let f = spec.linker_flavor.as_str(); + let own_lld = spec.linker.contains("lld") || f.ends_with("-lld") || f.starts_with("wasm-lld"); + let mut flags: Vec<&str> = Vec::new(); + if spec.is_like_wasm || f.starts_with("wasm") { + flags.extend(["-Clink-arg=--no-entry", "-Clink-arg=--export=probe_entry"]); + } else if spec.is_like_msvc || f.starts_with("msvc") { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=lld-link"]); + } + flags.extend(["-Clink-arg=/ENTRY:probe_entry", "-Clink-arg=/NODEFAULTLIB"]); + } else if spec.is_like_darwin || f.starts_with("darwin") { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=ld64.lld"]); + } + flags.extend(["-Clink-arg=-e", "-Clink-arg=_probe_entry", "-Clink-arg=-undefined", "-Clink-arg=dynamic_lookup"]); + } else { + if !own_lld { + flags.extend(["-Clinker=rust-lld", "-Clinker-flavor=ld.lld"]); + } + flags.push("-Clink-arg=--entry=probe_entry"); + } + flags.into_iter().map(str::to_owned).collect() +} + +fn rustc_out(toolchain: &str, a: &[&str]) -> Option { + let out = Command::new("rustc").arg(format!("+{toolchain}")).args(a).env("RUSTC_BOOTSTRAP", "1").output().ok()?; + out.status.success().then(|| String::from_utf8_lossy(&out.stdout).into_owned()) +} + +fn one(args: &Args, probe: &Path, target: &str) -> Res { + let skipped = |why: &str| Res { result: "skipped".into(), flags: vec![], undefined: vec![], first: why.into(), tail: String::new() }; + if SKIP.is_match(target) { + return skipped(""); + } + let Some(spec_json) = rustc_out(&args.toolchain, &["--print", "target-spec-json", "-Zunstable-options", "--target", target]) else { + return skipped("no spec"); + }; + let spec: Spec = serde_json::from_str(&spec_json).unwrap_or_default(); + let flags = link_flags(&spec); + let tdir = args.work.join("target").join(target); + let mut cmd = Command::new("cargo"); + cmd.arg(format!("+{}", args.toolchain)) + .args(["build", "--release", "-Zbuild-std=core,alloc", "-Zbuild-std-features=compiler-builtins-mem", "--target", target]) + .current_dir(probe) + .env("CARGO_TARGET_DIR", &tdir) + .env("RUSTFLAGS", flags.join(" ")) + .env("CARGO_TERM_COLOR", "never") + .env_remove("RUSTC_WRAPPER"); + let done = run_command(cmd, Duration::from_secs(1500)); + let _ = std::fs::remove_dir_all(&tdir); + let (exit, err) = match done { + Ok(d) => (d.exit.clone(), d.stderr_text()), + Err(e) => (Exit::Code(-1), e.to_string()), + }; + let result = if exit == Exit::Code(0) { + "ok" + } else if exit == Exit::Timeout { + "timeout" + } else if err.contains("internal compiler error") || err.contains("panicked at") { + "ice" + } else if err.contains("linking with") || err.contains("lld: error") { + if err.contains("undefined symbol") || err.contains("undefined reference") { + "link-undefined" + } else if ENV_MISSING.is_match(&err) { + "env" + } else { + "link" + } + } else { + "build" + }; + let mut undefined: Vec = UNDEFINED.captures_iter(&err).map(|c| c[1].to_owned()).collect(); + undefined.sort(); + undefined.dedup(); + undefined.truncate(20); + let first = err.lines().find(|l| l.contains("error[") || l.contains("error:")).unwrap_or("").chars().take(300).collect(); + let tail = if result == "ok" { String::new() } else { err.chars().rev().take(2500).collect::().chars().rev().collect() }; + Res { result: result.into(), flags, undefined, first, tail } +} + +pub fn run(args: Args) -> anyhow::Result { + std::fs::create_dir_all(&args.work)?; + let probe = args.probe.clone().unwrap_or_else(|| PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../../rustc/xlink-probe")); + let targets: Vec = match &args.targets { + Some(t) => t.split(',').map(str::to_owned).collect(), + None => rustc_out(&args.toolchain, &["--print", "target-list"]).unwrap_or_default().split_whitespace().map(str::to_owned).collect(), + }; + let pool = rayon::ThreadPoolBuilder::new().num_threads(args.jobs).build()?; + let results: BTreeMap = pool.install(|| targets.par_iter().map(|t| (t.clone(), one(&args, &probe, t))).collect()); + std::fs::write(args.work.join("results.json"), serde_json::to_string_pretty(&results)?)?; + let mut counts: BTreeMap<&str, usize> = BTreeMap::new(); + for r in results.values() { + *counts.entry(r.result.as_str()).or_default() += 1; + } + println!("{counts:?}"); + for (t, r) in &results { + if r.result != "ok" && r.result != "skipped" && r.result != "env" { + let what = if r.undefined.is_empty() { r.first.chars().take(120).collect() } else { r.undefined.join(" ") }; + println!("{:15} {t:40} {what}", r.result); + } + } + Ok(ExitCode::SUCCESS) +} diff --git a/crates/mirth-lab/src/uitest.rs b/crates/mirth-lab/src/uitest.rs new file mode 100644 index 0000000..60387d4 --- /dev/null +++ b/crates/mirth-lab/src/uitest.rs @@ -0,0 +1,182 @@ +//! rustc's UI tests as the checks use them: their `//@` headers (the first revision of a test +//! with revisions), and which of them can be compiled on their own on this host. + +use std::path::{Path, PathBuf}; +use std::sync::LazyLock; + +use regex::Regex; +use walkdir::WalkDir; + +/// What a test expects of the compiler, from its `//@ ` header. +#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)] +#[serde(rename_all = "kebab-case")] +pub enum Kind { + CheckPass, + BuildPass, + RunPass, + CheckFail, + BuildFail, + RunFail, +} + +impl Kind { + fn parse(key: &str) -> Option { + Some(match key { + "check-pass" => Kind::CheckPass, + "build-pass" => Kind::BuildPass, + "run-pass" => Kind::RunPass, + "check-fail" => Kind::CheckFail, + "build-fail" => Kind::BuildFail, + "run-fail" => Kind::RunFail, + _ => return None, + }) + } + + /// Whether type checking is all the test asks for (metadata is enough to compile it). + pub fn is_check(kind: Option) -> bool { + matches!(kind, None | Some(Kind::CheckPass) | Some(Kind::CheckFail)) + } +} + +/// Every kind, and tests without a kind header. +pub const ALL: &[Option] = &[ + Some(Kind::CheckPass), + Some(Kind::BuildPass), + Some(Kind::RunPass), + Some(Kind::CheckFail), + Some(Kind::BuildFail), + Some(Kind::RunFail), + None, +]; +pub const RUNNABLE: &[Option] = &[Some(Kind::RunPass), Some(Kind::RunFail)]; + +/// A UI test, as one of its revisions compiles it. +#[derive(Clone, Debug)] +pub struct Test { + pub path: PathBuf, + /// The path below the test root, as findings and known lists name it. + pub rel: String, + pub text: String, + pub flags: Vec, + pub edition: Option, + pub kind: Option, + pub revision: Option, +} + +impl Test { + pub fn edition(&self) -> &str { + self.edition.as_deref().unwrap_or("2015") + } + + pub fn file_name(&self) -> &str { + self.path.file_name().and_then(|n| n.to_str()).unwrap_or("test.rs") + } +} + +static REVISIONS: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*revisions:\s*(.*)$").unwrap()); +static DIRECTIVE: LazyLock = + LazyLock::new(|| Regex::new(r"(?m)^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$").unwrap()); +/// Tests that need more than one file, another target, or a tool this host may lack. +static NOT_STANDALONE: LazyLock = LazyLock::new(|| { + Regex::new( + r"(?m)^//@\s*(aux-build|aux-crate|aux-bin|aux-codegen-backend|proc-macro|add-minicore|needs-llvm-components|needs-sanitizer|needs-profiler|needs-rust-lld|needs-enzyme|ignore-x86_64|ignore-linux|ignore-unix|ignore-64bit|known-bug|rustc-env|unset-rustc-env)\b", + ) + .unwrap() +}); +static ONLY: LazyLock = LazyLock::new(|| Regex::new(r"(?m)^//@\s*only-(\S+)").unwrap()); +/// `only-` directives this host satisfies. +const HOST: &[&str] = &["x86_64", "linux", "unix", "64bit", "elf", "gnu"]; + +/// The headers of a test's first revision. +pub fn headers(text: &str) -> (Vec, Option, Option, Option) { + let revision = + REVISIONS.captures(text).and_then(|c| c[1].split_whitespace().next().map(str::to_owned)); + let (mut flags, mut edition, mut kind) = (Vec::new(), None, None); + for c in DIRECTIVE.captures_iter(text) { + if let Some(only) = c.get(1) + && !revision.as_deref().is_some_and(|r| only.as_str().split(',').any(|o| o == r)) + { + continue; + } + let value = c.get(3).map_or("", |v| v.as_str().trim()); + match &c[2] { + "compile-flags" => flags.extend(value.split_whitespace().map(str::to_owned)), + "edition" => edition = value.split_whitespace().next().map(str::to_owned), + key => { + if let Some(k) = Kind::parse(key) { + kind = Some(k); + } + } + } + } + if let Some(r) = &revision { + flags.extend(["--cfg".to_owned(), r.clone()]); + } + (flags, edition, kind, revision) +} + +/// Whether a test can be compiled alone on this host. +pub fn standalone(text: &str) -> bool { + !NOT_STANDALONE.is_match(text) + && ONLY.captures_iter(text).all(|c| HOST.iter().any(|h| c[1].starts_with(h))) +} + +/// The standalone tests under `root` of the given kinds, minus those `skip` rejects. +pub fn tests(root: &Path, kinds: &[Option], skip: impl Fn(&Test) -> bool) -> Vec { + let mut out = Vec::new(); + let mut paths: Vec = WalkDir::new(root) + .into_iter() + .filter_map(Result::ok) + .filter(|e| { + e.file_type().is_file() + && e.path().extension().is_some_and(|x| x == "rs") + && !e.path().components().any(|c| c.as_os_str() == "auxiliary") + }) + .map(|e| e.into_path()) + .collect(); + paths.sort(); + for path in paths { + let Ok(text) = std::fs::read_to_string(&path) else { continue }; + if !standalone(&text) { + continue; + } + let (flags, edition, kind, revision) = headers(&text); + if !kinds.contains(&kind) { + continue; + } + let rel = path.strip_prefix(root).unwrap_or(&path).to_string_lossy().into_owned(); + let test = Test { path, rel, text, flags, edition, kind, revision }; + if !skip(&test) { + out.push(test); + } + } + out +} + +/// Whether the test (or its flags) names one of these, as a substring of a flag. +pub fn flag_matches(test: &Test, re: &Regex) -> bool { + test.flags.iter().any(|f| re.is_match(f)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn first_revision_headers() { + let text = "//@ revisions: a b\n//@[a] compile-flags: -Zfoo\n//@[b] compile-flags: -Zbar\n//@ edition: 2021\n//@[a] check-pass\n"; + let (flags, edition, kind, rev) = headers(text); + assert_eq!(flags, ["-Zfoo", "--cfg", "a"]); + assert_eq!(edition.as_deref(), Some("2021")); + assert_eq!(kind, Some(Kind::CheckPass)); + assert_eq!(rev.as_deref(), Some("a")); + } + + #[test] + fn host_only_directives() { + assert!(standalone("//@ only-x86_64\n")); + assert!(standalone("//@ only-linux\n")); + assert!(!standalone("//@ only-aarch64\n")); + assert!(!standalone("//@ aux-build: x.rs\n")); + } +} diff --git a/crates/mirth-rewrite/src/lib.rs b/crates/mirth-rewrite/src/lib.rs new file mode 100644 index 0000000..a1aa938 --- /dev/null +++ b/crates/mirth-rewrite/src/lib.rs @@ -0,0 +1,319 @@ +//! Meaning-preserving rewrites of a Rust file, for metamorphic testing of rustc: a rewritten +//! program must get the same verdict (and the same error codes) as the original. +//! +//! [`rewrite`] returns the rewritten file, or why there is none. Comments are not kept (the +//! tokens are printed back), so line numbers change: compare verdicts and error codes, not +//! spans. The `mirth-rewrite` binary wraps it for the command line. +//! +//! Rewrites: +//! +//! - `identity`: the file printed back unchanged: the baseline for the others, since printing +//! tokens back drops comments and moves every line. +//! - `generic-wrap`: the body of each free function with plain parameters moves into a generic +//! inner function, called with `()` for its unused type parameter. What the body does is the +//! same; it is now checked in a generic context and instantiated. +//! - `alias`: each struct, enum and union gets a type alias with the same generic parameters, +//! and every type mentioning it by its bare name mentions the alias instead. +//! - `reorder`: the top-level items in reverse order (item order does not matter in Rust; +//! files with `macro_rules!` or item-position macro calls are left out, where it does). +//! - `unused`: an unused function, struct and trait added at the end. + +use std::collections::{BTreeMap, BTreeSet}; + +use proc_macro2::Span; +use quote::{ToTokens, format_ident, quote}; +use syn::visit_mut::VisitMut; +use syn::{FnArg, GenericParam, Ident, Item, ItemFn, Pat, Type, TypePath}; + +/// The rewrites, `identity` first (the baseline for the others). +pub const REWRITES: [&str; 5] = ["identity", "generic-wrap", "alias", "reorder", "unused"]; + +/// What a rewrite of a file came to. +#[derive(Debug, PartialEq, Eq)] +pub enum Outcome { + /// The rewritten file. + Rewritten(String), + /// `syn` does not parse the file. + DoesNotParse, + /// The rewrite has nothing to change in this file. + DoesNotApply, + /// No rewrite has this name. + Unknown, +} + +/// Rewrite `text` (a whole Rust file) by the rewrite called `name`. +pub fn rewrite(name: &str, text: &str) -> Outcome { + let Ok(mut file) = syn::parse_file(text) else { return Outcome::DoesNotParse }; + let applied = match name { + "generic-wrap" => generic_wrap(&mut file), + "alias" => alias(&mut file), + "reorder" => reorder(&mut file), + "unused" => unused(&mut file), + "identity" => true, + _ => return Outcome::Unknown, + }; + if applied { Outcome::Rewritten(file.into_token_stream().to_string()) } else { Outcome::DoesNotApply } +} + +/// Moves the body of `fn f(a: A, b: B) -> R { body }` into +/// `fn __mirth_inner<__MirthT>(a: A, b: B) -> R { body }`, called as `__mirth_inner::<()>(a, b)`. +fn generic_wrap(file: &mut syn::File) -> bool { + let mut applied = false; + for item in &mut file.items { + if let Item::Fn(f) = item + && wrappable(f) + { + wrap(f); + applied = true; + } + } + applied +} + +fn wrappable(f: &ItemFn) -> bool { + let sig = &f.sig; + // Plain functions only: no generics (an inner fn cannot use the outer's parameters), no + // `impl Trait`, no qualifiers that change how the body runs, no `self`, no attributes that + // name the function (`#[test]`, `#[no_mangle]`, `#[track_caller]` would change meaning). + sig.generics.params.is_empty() + && sig.generics.where_clause.is_none() + && sig.constness.is_none() + && sig.asyncness.is_none() + && sig.unsafety.is_none() + && sig.abi.is_none() + && sig.variadic.is_none() + && f.attrs.iter().all(|a| a.path().is_ident("allow") || a.path().is_ident("inline")) + && !sig.to_token_stream().to_string().contains("impl ") + && !sig.to_token_stream().to_string().contains('\'') + // Parameters without attributes: a `#[cfg]`-ed out one cannot be passed on by name. + && sig.inputs.iter().all(|arg| matches!(arg, FnArg::Typed(t) if t.attrs.is_empty() && matches!(&*t.pat, Pat::Ident(p) if p.by_ref.is_none() && p.subpat.is_none()))) +} + +fn wrap(f: &mut ItemFn) { + let sig = &f.sig; + let inputs = &sig.inputs; + let output = &sig.output; + let names: Vec<&Ident> = sig + .inputs + .iter() + .map(|arg| match arg { + FnArg::Typed(t) => match &*t.pat { + Pat::Ident(p) => &p.ident, + _ => unreachable!("checked in wrappable"), + }, + FnArg::Receiver(_) => unreachable!("checked in wrappable"), + }) + .collect(); + let body = &f.block; + // No attributes (a test may `forbid` the lint they would allow); names no lint objects to. + let new: syn::Block = syn::parse_quote!({ + fn _mirth_inner(#inputs) #output #body + _mirth_inner::<()>(#(#names),*) + }); + // Parameters declared `mut` are mutated in the body, now the inner function's. + for arg in f.sig.inputs.iter_mut() { + if let FnArg::Typed(t) = arg + && let Pat::Ident(p) = &mut *t.pat + { + p.mutability = None; + } + } + *f.block = new; +} + +/// `type __MirthAlias_S = S;` for each struct, enum and union `S`, and every +/// type that names `S` by its bare name names the alias instead. +fn alias(file: &mut syn::File) -> bool { + // Names also used for generic parameters somewhere: a bare path may mean the parameter. + struct Params(BTreeSet); + impl VisitMut for Params { + fn visit_generic_param_mut(&mut self, p: &mut GenericParam) { + match p { + GenericParam::Type(t) => self.0.insert(t.ident.to_string()), + GenericParam::Const(c) => self.0.insert(c.ident.to_string()), + GenericParam::Lifetime(_) => false, + }; + syn::visit_mut::visit_generic_param_mut(self, p); + } + } + let mut params_seen = Params(BTreeSet::new()); + params_seen.visit_file_mut(&mut file.clone()); + // Types defined more than once (a nested item shadowing a top-level one): a bare name may + // mean either. + struct Defined(BTreeMap); + impl VisitMut for Defined { + fn visit_item_struct_mut(&mut self, i: &mut syn::ItemStruct) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_struct_mut(self, i); + } + fn visit_item_enum_mut(&mut self, i: &mut syn::ItemEnum) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_enum_mut(self, i); + } + fn visit_item_union_mut(&mut self, i: &mut syn::ItemUnion) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_union_mut(self, i); + } + fn visit_item_type_mut(&mut self, i: &mut syn::ItemType) { + *self.0.entry(i.ident.to_string()).or_default() += 1; + syn::visit_mut::visit_item_type_mut(self, i); + } + } + let mut defined = Defined(BTreeMap::new()); + defined.visit_file_mut(&mut file.clone()); + let mut aliases = BTreeMap::new(); + let mut new_items = Vec::new(); + for item in &file.items { + let (ident, generics) = match item { + Item::Struct(s) => (&s.ident, &s.generics), + Item::Enum(e) => (&e.ident, &e.generics), + Item::Union(u) => (&u.ident, &u.generics), + _ => continue, + }; + if params_seen.0.contains(&ident.to_string()) || defined.0.get(&ident.to_string()) != Some(&1) { + continue; + } + // Lifetime parameters elide differently through an alias; defaults may name other + // parameters: such types are left alone. + if generics.params.iter().any(|p| match p { + GenericParam::Lifetime(_) => true, + GenericParam::Type(t) => t.default.is_some(), + GenericParam::Const(c) => c.default.is_some(), + }) { + continue; + } + let alias = format_ident!("_MirthAlias{}", ident); + // The type's own `cfg`s: an alias of a configured-out type would name nothing. + let cfgs: Vec<&syn::Attribute> = match item { + Item::Struct(x) => &x.attrs, + Item::Enum(x) => &x.attrs, + Item::Union(x) => &x.attrs, + _ => unreachable!(), + } + .iter() + // `cfg` only: a `cfg_attr` may expand to a `derive`, which an alias cannot have. + .filter(|a| a.path().is_ident("cfg")) + .collect(); + // The alias's parameters: the type's, without bounds (aliases ignore them), with defaults. + let mut params = generics.clone(); + params.where_clause = None; + for p in params.params.iter_mut() { + match p { + GenericParam::Type(t) => { + // `?Sized` must stay: without it the alias requires `Sized`. + let maybe: Vec = t + .bounds + .iter() + .filter(|b| matches!(b, syn::TypeParamBound::Trait(tb) if matches!(tb.modifier, syn::TraitBoundModifier::Maybe(_)))) + .cloned() + .collect(); + t.bounds = maybe.into_iter().collect(); + if t.bounds.is_empty() { + t.colon_token = None; + } + } + GenericParam::Lifetime(l) => { + l.bounds.clear(); + l.colon_token = None; + } + GenericParam::Const(_) => {} + } + } + let args: Vec = generics + .params + .iter() + .map(|p| match p { + GenericParam::Type(t) => t.ident.to_token_stream(), + GenericParam::Lifetime(l) => l.lifetime.to_token_stream(), + GenericParam::Const(c) => c.ident.to_token_stream(), + }) + .collect(); + let target = if args.is_empty() { quote!(#ident) } else { quote!(#ident<#(#args),*>) }; + new_items.push(syn::parse_quote!( + #(#cfgs)* + type #alias #params = #target; + )); + aliases.insert(ident.to_string(), alias); + } + if aliases.is_empty() { + return false; + } + struct Rename<'a> { + aliases: &'a BTreeMap, + count: usize, + } + impl VisitMut for Rename<'_> { + fn visit_type_path_mut(&mut self, ty: &mut TypePath) { + if ty.qself.is_none() + && ty.path.leading_colon.is_none() + && ty.path.segments.len() == 1 + && let Some(alias) = self.aliases.get(&ty.path.segments[0].ident.to_string()) + { + ty.path.segments[0].ident = Ident::new(&alias.to_string(), Span::call_site()); + self.count += 1; + } + syn::visit_mut::visit_type_path_mut(self, ty); + } + // Inside a type's own definition, `Self`-like uses must stay (a recursive type through + // its alias would be a cycle in the alias), so definitions are not visited. + fn visit_item_struct_mut(&mut self, _: &mut syn::ItemStruct) {} + fn visit_item_enum_mut(&mut self, _: &mut syn::ItemEnum) {} + fn visit_item_union_mut(&mut self, _: &mut syn::ItemUnion) {} + // Macros' tokens are not types to `syn`; derive input stays as it is. + fn visit_macro_mut(&mut self, _: &mut syn::Macro) {} + // In an inline module the bare name reaches the type through a `use`, the alias would + // need one too: modules are left as they are. + fn visit_item_mod_mut(&mut self, _: &mut syn::ItemMod) {} + // A receiver typed with the impl's own name (`self: &mut Test`) elides lifetimes like + // `&mut self`; through an alias it does not (resolution does not see through aliases). + fn visit_receiver_mut(&mut self, _: &mut syn::Receiver) {} + // The same rule compares a receiver with the impl's self type: that stays as written. + fn visit_item_impl_mut(&mut self, imp: &mut syn::ItemImpl) { + let self_ty = std::mem::replace(&mut *imp.self_ty, Type::Verbatim(Default::default())); + syn::visit_mut::visit_item_impl_mut(self, imp); + *imp.self_ty = self_ty; + } + } + let mut rename = Rename { aliases: &aliases, count: 0 }; + for item in &mut file.items { + rename.visit_item_mut(item); + } + if rename.count == 0 { + return false; + } + file.items.extend(new_items); + true +} + +fn reorder(file: &mut syn::File) -> bool { + let has_macros = file.items.iter().any(|item| matches!(item, Item::Macro(_))); + if has_macros || file.items.len() < 2 { + return false; + } + // `use` and `extern crate` first, as written, so that preludes and `#[macro_use]` stay put. + let (mut head, mut rest): (Vec, Vec) = + file.items.drain(..).partition(|item| matches!(item, Item::Use(_) | Item::ExternCrate(_))); + rest.reverse(); + head.extend(rest); + file.items = head; + true +} + +fn unused(file: &mut syn::File) -> bool { + let names: BTreeSet = file + .items + .iter() + .filter_map(|item| match item { + Item::Fn(f) => Some(f.sig.ident.to_string()), + _ => None, + }) + .collect(); + if names.contains("_mirth_unused") { + return false; + } + // Leading underscores keep `dead_code` quiet without an attribute a test could `forbid`. + file.items.push(syn::parse_quote!(fn _mirth_unused() {})); + file.items.push(syn::parse_quote!(struct _MirthUnused;)); + file.items.push(syn::parse_quote!(trait _MirthUnusedTrait {})); + true +} diff --git a/crates/mirth-rewrite/src/main.rs b/crates/mirth-rewrite/src/main.rs index 93b037e..9a791fd 100644 --- a/crates/mirth-rewrite/src/main.rs +++ b/crates/mirth-rewrite/src/main.rs @@ -1,320 +1,28 @@ -//! Meaning-preserving rewrites of a Rust file, for metamorphic testing of rustc: a rewritten -//! program must get the same verdict (and the same error codes) as the original. -//! -//! mirth-rewrite -//! -//! prints the rewritten file to stdout; exits 2 when the file does not parse (as `syn` sees -//! Rust) and 3 when the rewrite does not apply to it. Comments are not kept (the tokens are -//! printed back), so line numbers change: compare verdicts and error codes, not spans. -//! -//! Rewrites: -//! -//! - `identity`: the file printed back unchanged: the baseline for the others, since printing -//! tokens back drops comments and moves every line. -//! - `generic-wrap`: the body of each free function with plain parameters moves into a generic -//! inner function, called with `()` for its unused type parameter. What the body does is the -//! same; it is now checked in a generic context and instantiated. -//! - `alias`: each struct, enum and union gets a type alias with the same generic parameters, -//! and every type mentioning it by its bare name mentions the alias instead. -//! - `reorder`: the top-level items in reverse order (item order does not matter in Rust; -//! files with `macro_rules!` or item-position macro calls are left out, where it does). -//! - `unused`: an unused function, struct and trait added at the end. +//! `mirth-rewrite `: prints the rewritten file to stdout; exits 2 when the file +//! does not parse (as `syn` sees Rust) and 3 when the rewrite does not apply to it. The rewrites +//! are documented in the library. -use std::collections::{BTreeMap, BTreeSet}; use std::process::exit; -use proc_macro2::Span; -use quote::{ToTokens, format_ident, quote}; -use syn::visit_mut::VisitMut; -use syn::{FnArg, GenericParam, Ident, Item, ItemFn, Pat, Type, TypePath}; +use mirth_rewrite::{Outcome, REWRITES, rewrite}; fn main() { let args: Vec = std::env::args().collect(); if args.len() != 3 { - eprintln!("usage: mirth-rewrite "); + eprintln!("usage: mirth-rewrite <{}> ", REWRITES.join("|")); exit(64); } let text = std::fs::read_to_string(&args[2]).unwrap_or_else(|error| { eprintln!("{}: {error}", args[2]); exit(64) }); - let Ok(mut file) = syn::parse_file(&text) else { exit(2) }; - let applied = match args[1].as_str() { - "generic-wrap" => generic_wrap(&mut file), - "alias" => alias(&mut file), - "reorder" => reorder(&mut file), - "unused" => unused(&mut file), - "identity" => true, - other => { - eprintln!("unknown rewrite {other}"); + match rewrite(&args[1], &text) { + Outcome::Rewritten(out) => println!("{out}"), + Outcome::DoesNotParse => exit(2), + Outcome::DoesNotApply => exit(3), + Outcome::Unknown => { + eprintln!("unknown rewrite {}", args[1]); exit(64) } - }; - if !applied { - exit(3); } - println!("{}", file.into_token_stream()); -} - -/// Moves the body of `fn f(a: A, b: B) -> R { body }` into -/// `fn __mirth_inner<__MirthT>(a: A, b: B) -> R { body }`, called as `__mirth_inner::<()>(a, b)`. -fn generic_wrap(file: &mut syn::File) -> bool { - let mut applied = false; - for item in &mut file.items { - if let Item::Fn(f) = item - && wrappable(f) - { - wrap(f); - applied = true; - } - } - applied -} - -fn wrappable(f: &ItemFn) -> bool { - let sig = &f.sig; - // Plain functions only: no generics (an inner fn cannot use the outer's parameters), no - // `impl Trait`, no qualifiers that change how the body runs, no `self`, no attributes that - // name the function (`#[test]`, `#[no_mangle]`, `#[track_caller]` would change meaning). - sig.generics.params.is_empty() - && sig.generics.where_clause.is_none() - && sig.constness.is_none() - && sig.asyncness.is_none() - && sig.unsafety.is_none() - && sig.abi.is_none() - && sig.variadic.is_none() - && f.attrs.iter().all(|a| a.path().is_ident("allow") || a.path().is_ident("inline")) - && !sig.to_token_stream().to_string().contains("impl ") - && !sig.to_token_stream().to_string().contains('\'') - // Parameters without attributes: a `#[cfg]`-ed out one cannot be passed on by name. - && sig.inputs.iter().all(|arg| matches!(arg, FnArg::Typed(t) if t.attrs.is_empty() && matches!(&*t.pat, Pat::Ident(p) if p.by_ref.is_none() && p.subpat.is_none()))) -} - -fn wrap(f: &mut ItemFn) { - let sig = &f.sig; - let inputs = &sig.inputs; - let output = &sig.output; - let names: Vec<&Ident> = sig - .inputs - .iter() - .map(|arg| match arg { - FnArg::Typed(t) => match &*t.pat { - Pat::Ident(p) => &p.ident, - _ => unreachable!("checked in wrappable"), - }, - FnArg::Receiver(_) => unreachable!("checked in wrappable"), - }) - .collect(); - let body = &f.block; - // No attributes (a test may `forbid` the lint they would allow); names no lint objects to. - let new: syn::Block = syn::parse_quote!({ - fn _mirth_inner(#inputs) #output #body - _mirth_inner::<()>(#(#names),*) - }); - // Parameters declared `mut` are mutated in the body, now the inner function's. - for arg in f.sig.inputs.iter_mut() { - if let FnArg::Typed(t) = arg - && let Pat::Ident(p) = &mut *t.pat - { - p.mutability = None; - } - } - *f.block = new; -} - -/// `type __MirthAlias_S = S;` for each struct, enum and union `S`, and every -/// type that names `S` by its bare name names the alias instead. -fn alias(file: &mut syn::File) -> bool { - // Names also used for generic parameters somewhere: a bare path may mean the parameter. - struct Params(BTreeSet); - impl VisitMut for Params { - fn visit_generic_param_mut(&mut self, p: &mut GenericParam) { - match p { - GenericParam::Type(t) => self.0.insert(t.ident.to_string()), - GenericParam::Const(c) => self.0.insert(c.ident.to_string()), - GenericParam::Lifetime(_) => false, - }; - syn::visit_mut::visit_generic_param_mut(self, p); - } - } - let mut params_seen = Params(BTreeSet::new()); - params_seen.visit_file_mut(&mut file.clone()); - // Types defined more than once (a nested item shadowing a top-level one): a bare name may - // mean either. - struct Defined(BTreeMap); - impl VisitMut for Defined { - fn visit_item_struct_mut(&mut self, i: &mut syn::ItemStruct) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_struct_mut(self, i); - } - fn visit_item_enum_mut(&mut self, i: &mut syn::ItemEnum) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_enum_mut(self, i); - } - fn visit_item_union_mut(&mut self, i: &mut syn::ItemUnion) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_union_mut(self, i); - } - fn visit_item_type_mut(&mut self, i: &mut syn::ItemType) { - *self.0.entry(i.ident.to_string()).or_default() += 1; - syn::visit_mut::visit_item_type_mut(self, i); - } - } - let mut defined = Defined(BTreeMap::new()); - defined.visit_file_mut(&mut file.clone()); - let mut aliases = BTreeMap::new(); - let mut new_items = Vec::new(); - for item in &file.items { - let (ident, generics) = match item { - Item::Struct(s) => (&s.ident, &s.generics), - Item::Enum(e) => (&e.ident, &e.generics), - Item::Union(u) => (&u.ident, &u.generics), - _ => continue, - }; - if params_seen.0.contains(&ident.to_string()) || defined.0.get(&ident.to_string()) != Some(&1) { - continue; - } - // Lifetime parameters elide differently through an alias; defaults may name other - // parameters: such types are left alone. - if generics.params.iter().any(|p| match p { - GenericParam::Lifetime(_) => true, - GenericParam::Type(t) => t.default.is_some(), - GenericParam::Const(c) => c.default.is_some(), - }) { - continue; - } - let alias = format_ident!("_MirthAlias{}", ident); - // The type's own `cfg`s: an alias of a configured-out type would name nothing. - let cfgs: Vec<&syn::Attribute> = match item { - Item::Struct(x) => &x.attrs, - Item::Enum(x) => &x.attrs, - Item::Union(x) => &x.attrs, - _ => unreachable!(), - } - .iter() - // `cfg` only: a `cfg_attr` may expand to a `derive`, which an alias cannot have. - .filter(|a| a.path().is_ident("cfg")) - .collect(); - // The alias's parameters: the type's, without bounds (aliases ignore them), with defaults. - let mut params = generics.clone(); - params.where_clause = None; - for p in params.params.iter_mut() { - match p { - GenericParam::Type(t) => { - // `?Sized` must stay: without it the alias requires `Sized`. - let maybe: Vec = t - .bounds - .iter() - .filter(|b| matches!(b, syn::TypeParamBound::Trait(tb) if matches!(tb.modifier, syn::TraitBoundModifier::Maybe(_)))) - .cloned() - .collect(); - t.bounds = maybe.into_iter().collect(); - if t.bounds.is_empty() { - t.colon_token = None; - } - } - GenericParam::Lifetime(l) => { - l.bounds.clear(); - l.colon_token = None; - } - GenericParam::Const(_) => {} - } - } - let args: Vec = generics - .params - .iter() - .map(|p| match p { - GenericParam::Type(t) => t.ident.to_token_stream(), - GenericParam::Lifetime(l) => l.lifetime.to_token_stream(), - GenericParam::Const(c) => c.ident.to_token_stream(), - }) - .collect(); - let target = if args.is_empty() { quote!(#ident) } else { quote!(#ident<#(#args),*>) }; - new_items.push(syn::parse_quote!( - #(#cfgs)* - type #alias #params = #target; - )); - aliases.insert(ident.to_string(), alias); - } - if aliases.is_empty() { - return false; - } - struct Rename<'a> { - aliases: &'a BTreeMap, - count: usize, - } - impl VisitMut for Rename<'_> { - fn visit_type_path_mut(&mut self, ty: &mut TypePath) { - if ty.qself.is_none() - && ty.path.leading_colon.is_none() - && ty.path.segments.len() == 1 - && let Some(alias) = self.aliases.get(&ty.path.segments[0].ident.to_string()) - { - ty.path.segments[0].ident = Ident::new(&alias.to_string(), Span::call_site()); - self.count += 1; - } - syn::visit_mut::visit_type_path_mut(self, ty); - } - // Inside a type's own definition, `Self`-like uses must stay (a recursive type through - // its alias would be a cycle in the alias), so definitions are not visited. - fn visit_item_struct_mut(&mut self, _: &mut syn::ItemStruct) {} - fn visit_item_enum_mut(&mut self, _: &mut syn::ItemEnum) {} - fn visit_item_union_mut(&mut self, _: &mut syn::ItemUnion) {} - // Macros' tokens are not types to `syn`; derive input stays as it is. - fn visit_macro_mut(&mut self, _: &mut syn::Macro) {} - // In an inline module the bare name reaches the type through a `use`, the alias would - // need one too: modules are left as they are. - fn visit_item_mod_mut(&mut self, _: &mut syn::ItemMod) {} - // A receiver typed with the impl's own name (`self: &mut Test`) elides lifetimes like - // `&mut self`; through an alias it does not (resolution does not see through aliases). - fn visit_receiver_mut(&mut self, _: &mut syn::Receiver) {} - // The same rule compares a receiver with the impl's self type: that stays as written. - fn visit_item_impl_mut(&mut self, imp: &mut syn::ItemImpl) { - let self_ty = std::mem::replace(&mut *imp.self_ty, Type::Verbatim(Default::default())); - syn::visit_mut::visit_item_impl_mut(self, imp); - *imp.self_ty = self_ty; - } - } - let mut rename = Rename { aliases: &aliases, count: 0 }; - for item in &mut file.items { - rename.visit_item_mut(item); - } - if rename.count == 0 { - return false; - } - file.items.extend(new_items); - true -} - -fn reorder(file: &mut syn::File) -> bool { - let has_macros = file.items.iter().any(|item| matches!(item, Item::Macro(_))); - if has_macros || file.items.len() < 2 { - return false; - } - // `use` and `extern crate` first, as written, so that preludes and `#[macro_use]` stay put. - let (mut head, mut rest): (Vec, Vec) = - file.items.drain(..).partition(|item| matches!(item, Item::Use(_) | Item::ExternCrate(_))); - rest.reverse(); - head.extend(rest); - file.items = head; - true -} - -fn unused(file: &mut syn::File) -> bool { - let names: BTreeSet = file - .items - .iter() - .filter_map(|item| match item { - Item::Fn(f) => Some(f.sig.ident.to_string()), - _ => None, - }) - .collect(); - if names.contains("_mirth_unused") { - return false; - } - // Leading underscores keep `dead_code` quiet without an attribute a test could `forbid`. - file.items.push(syn::parse_quote!(fn _mirth_unused() {})); - file.items.push(syn::parse_quote!(struct _MirthUnused;)); - file.items.push(syn::parse_quote!(trait _MirthUnusedTrait {})); - true } diff --git a/crates/mirth-watch/src/config.rs b/crates/mirth-watch/src/config.rs index 4da2e17..940f566 100644 --- a/crates/mirth-watch/src/config.rs +++ b/crates/mirth-watch/src/config.rs @@ -70,7 +70,7 @@ pub struct Diagnostics { pub paths: bool, /// Write each body's outgoing edges to `.graph` beside the site table: direct calls, /// functions and closures used as values, and callees MIR inlining merged in; and, for a - /// method implementing a trait's, which trait item (rustc/callgraph.py reads them). + /// method implementing a trait's, which trait item (`mirth-lab callgraph` reads them). #[serde(default)] pub callgraph: bool, } diff --git a/docs/checks.md b/docs/checks.md index 7027f53..1e149f5 100644 --- a/docs/checks.md +++ b/docs/checks.md @@ -153,7 +153,7 @@ DeadStoreElimination creating overlapping call arguments (#162997), parameter de default impl (#163856), and an optimization introducing a dereference on an unreachable path (#159591). The check: build every runnable program at `-Copt-level` 0, 1, 2, 3, s and z, with `-Zmir-opt-level` 0 to 4, each MIR pass toggled singly, and fat or thin LTO; compare output. -**It is cheap, because mirth already has the option matrix (`flag-model.py`) and edit +**It is cheap, because mirth already has the option matrix (`mirth-lab flag-model`) and edit transitions; what is missing is running the program and comparing its output.** Plain compiler-fuzzer programs (Csmith-style, rustlantis) feed it well. @@ -210,7 +210,7 @@ partly covered by checks 1, 9 and 17. UB (#160670, #160669), and diagnostic differences between the solvers (#162919). The check: compile everything with `-Znext-solver=globally` and with the old solver, and with NLL and `-Zpolonius=next`. Compare the verdicts and error codes, and run one-sided acceptances under -Miri. **The cheapest check here: `ui-solver-diff.py` already does half of it** +Miri. **The cheapest check here: `ui-mirth-lab solver-diff` already does half of it** ([solver.md](solver.md)). ### 10. Scaling and budgets (31) @@ -316,7 +316,7 @@ Ordered by bugs caught per unit of effort, and by what mirth already has: 1. **Optimization and pass differential** (45). It extends the option matrix with a run step. Two afternoons. -2. **Solver differential** (32, plus up to 73 ICEs). It is `ui-solver-diff.py` plus Polonius, +2. **Solver differential** (32, plus up to 73 ICEs). It is `ui-mirth-lab solver-diff` plus Polonius, and the next solver as a configuration column for the whole corpus. 3. **Miri differential** (52). It runs Miri over the run-pass tests, then once per MIR pass. 4. **Equivalent rewrites in ui-fuzz** (63). Four rewrites to start with. @@ -345,18 +345,44 @@ about 600 of the 1,000. ## Built (2026-10-09/10) -The first seven checks of the build order, as scripts in `rustc/` sharing `rustc/uitest.py`, run +The first seven checks of the build order, as `mirth-lab` subcommands sharing its `uitest` module, run over the standalone UI tests at the pin (and real crates for release-to-release). Each has `--recheck`, `--known` and `--pause-on-finding` for the frontier loop. | check | script | swept | result | |---|---|---|---| -| optimization and pass differential | `opt-diff.py` | 3,217 runnable tests × 13 configurations (opt levels, MIR opt levels, LTO, target CPU, Cranelift) | nothing; Cranelift's gaps (tail calls, some linkages and SIMD intrinsics) noted | -| solver differential | `solver-diff.py` | 17,634 tests × old/new solver × NLL/Polonius | the 26 rejections and 3 crashes of [`solver.md`](solver.md); Polonius agrees with NLL everywhere | -| Miri differential | `miri-diff.py` | 3,094 runnable tests at MIR opt levels 0, 2, 4 and natively | nothing; tests asserting unspecified behavior (function pointer equality, ZST addresses) listed | -| equivalent rewrites | `mirth-rewrite` + `rewrite-diff.py` | 18,624 tests × generic-wrap, alias, reorder, unused | findings 25 (generic-wrap) and 28 (reorder) | -| ABI vs clang | `abi-diff.py` | 21 main targets × 10 seeds × 300 random signatures | findings 19 and 20; #163911 reproduced; i686 MSVC small-struct returns and a PowerPC64 `inreg` float undecided | -| internal checks on | `crash-diff.py` + a debug-assertions compiler | 18,624 tests with `-Zvalidate-mir` | findings 21–24 (17 tests) | -| release-to-release | `release-diff.py` | 87 real repositories, nightly-2026-07-18 → 10-06 | findings 26 and 27; `allocative` (unstable features) noted | +| optimization and pass differential | `mirth-lab opt-diff` | 3,217 runnable tests × 13 configurations (opt levels, MIR opt levels, LTO, target CPU, Cranelift) | nothing; Cranelift's gaps (tail calls, some linkages and SIMD intrinsics) noted | +| solver differential | `mirth-lab solver-diff` | 17,634 tests × old/new solver × NLL/Polonius | the 26 rejections and 3 crashes of [`solver.md`](solver.md); Polonius agrees with NLL everywhere | +| Miri differential | `mirth-lab miri-diff` | 3,094 runnable tests at MIR opt levels 0, 2, 4 and natively | nothing; tests asserting unspecified behavior (function pointer equality, ZST addresses) listed | +| equivalent rewrites | `mirth-rewrite` + `mirth-lab rewrite-diff` | 18,624 tests × generic-wrap, alias, reorder, unused | findings 25 (generic-wrap) and 28 (reorder) | +| ABI vs clang | `mirth-lab abi-diff` | 21 main targets × 10 seeds × 300 random signatures | findings 19 and 20; #163911 reproduced; i686 MSVC small-struct returns and a PowerPC64 `inreg` float undecided | +| internal checks on | `mirth-lab crash-diff` + a debug-assertions compiler | 18,624 tests with `-Zvalidate-mir` | findings 21–24 (17 tests) | +| release-to-release | `mirth-lab release-diff` | 87 real repositories, nightly-2026-07-18 → 10-06 | findings 26 and 27; `allocative` (unstable features) noted | Ten new findings (19–28) in [`hunt.md`](hunt.md), none from the checks mirth had before. + +### Second batch (2026-10-10) + +| check | script | swept | result | +|---|---|---|---| +| suggestions apply (18) | `mirth-lab suggest-diff` | 17,945 tests without `run-rustfix`, 7,385 machine-applicable suggestions applied one at a time | finding 29: 111 lint fixes break builds (six shapes reduced); error-recovery suggestions that leave the error or do not parse noted | +| diagnostic invariants (13) | `mirth-lab diag-check` | 18,374 tests | finding 30: debug output in two diagnostics; spans all in bounds | +| determinism (15) | `mirth-lab repro-diff` | 6,886 tests × repeat, other directory with `--remap-path-prefix`, `-Zthreads=8`, decoy `-L` library | nothing new: only `-Zthreads` differences, all in the known async fn (#162202) and RPIT (#163878) families | +| feature gates (17) | `mirth-lab gate-check` | 143 unstable attributes × 14 positions; 156 unstable library items with resolvable paths × use, renamed use, glob, impl, value, type | every library spelling gated; finding 31 (an ICE after the gate error for `#[rustc_main]` on non-functions); `#[feature]` outside the crate root only warns (intended) | + +## Running the checks + +The checks are subcommands of `mirth-lab` (`crates/mirth-lab`; `mirth-lab --help` lists them): + +```sh +cargo build --release -p mirth-lab +R=~/mirth-work/campaign/rustc/bin/rustc T=~/mirth-work/rust/tests/ui +target/release/mirth-lab opt-diff --rustc $R --cranelift "$(rustup +nightly-2026-10-06 which rustc)" --tests $T --work +target/release/mirth-lab solver-diff --rustc $R --tests $T --work +target/release/mirth-lab abi-diff --rustc $R --rust ~/mirth-work/rust --work --seed 3 +target/release/mirth-lab release-diff --corpus ~/proofhouse-repos/rust --old nightly-2026-07-18 --new nightly-2026-10-06 --work +``` + +Sweeps over UI tests share `--tests`, `--work`, `--only`, `--known`, `--jobs`, `--recheck` and +`--pause-on-finding` (exit 3 at the first finding: the frontier loop). Results go to +`/results.jsonl`, findings to `/findings//`. diff --git a/docs/coverage-handoff.md b/docs/coverage-handoff.md index 6fe60af..e4e86cc 100644 --- a/docs/coverage-handoff.md +++ b/docs/coverage-handoff.md @@ -65,15 +65,15 @@ $WORK` runs the ui-fulldeps tests compiletest skips at stage 1. **Any other command** (the generators, the fuzzers): ```sh -rustc/coverage-run.sh ui-fuzz python3 rustc/ui-fuzz.py --rustc $COV_RUSTC \ +rustc/coverage-run.sh ui-fuzz target/release/mirth-lab ui-fuzz --rustc $COV_RUSTC \ --tests $MIRTH_RUST/tests/ui --list $WORK/ui-cov/all-runnable.json --work $WORK/ui-fuzz-cov \ --edits 3 --jobs 6 # about 45 minutes -rustc/coverage-run.sh generators python3 rustc/coverage-generators.py --rustc $COV_RUSTC \ +rustc/coverage-run.sh generators target/release/mirth-lab coverage-generators --rustc $COV_RUSTC \ --rust $MIRTH_RUST --list $WORK/ui-cov/picked.json --work $WORK/gen-work --jobs 6 \ --only prints,targets,dumps,links # prints and targets: minutes; dumps: 30 minutes ``` -A new generator is a function in `coverage-generators.py` named in `--only`. A run's directory +A new generator is a function in `crates/mirth-lab/src/tools/coverage_generators.rs` named in `--only`. A run's directory under `cov-suites` is picked up by the report automatically; name it with `fulldeps` or `compiler-unit` if the programs are outside the compiler (their entry points become roots). @@ -154,11 +154,11 @@ MIRTH_RUNTIME=$WORK/build-cg/mirth-runtime/libmirth_runtime.rlib MIRTH_WATCH=cg. 1. **Saturate with generators** (cheap, broad). Next: the dumps over all 18,553 runnable UI tests instead of 300; ui-fuzz and `tests/incremental` under `RUSTC_VERIFY_REUSE`; every `rustc --explain` example compiled; rustdoc over the UI tests (`rustdoc --document-private-items`, - `--output-format json`, `--test`); target × option combinations (PICT, as `flag-model.py`). + `--output-format json`, `--test`); target × option combinations (PICT, as `mirth-lab flag-model`). Measure each with `coverage-report.sh` and keep the ones that add. 2. **Tighten the denominator** where the graph says reachable but nothing can run it. For each suspicious class, `--why` a sample: if the chain has a spurious step (a demand nothing really - makes), fix the analysis in `sites.rs` (`graph`) or `callgraph.py`, rebuild the graph, and + makes), fix the analysis in `sites.rs` (`graph`) or `mirth-lab callgraph`, rebuild the graph, and check 0 misses. Classes to look at first: the AST's encode/decode, `handle_cycle_error` per query, `format_value` (only on a verification failure or `RUSTC_VERIFY_REUSE`), derived impls of types only built in tests. A function that only runs on a compiler bug is already diff --git a/docs/coverage.md b/docs/coverage.md index 76db752..8605dc3 100644 --- a/docs/coverage.md +++ b/docs/coverage.md @@ -4,12 +4,12 @@ the compiler compiling it reaches. mirth-watch's `[coverage]` mode (`rustc/coverage.toml`) instruments every function and closure of the compiler's own crates (`rustc_*`) with one call at entry, which records the function's first call in each process; at exit each rustc -process writes the functions it entered. `rustc/coverage.py` joins those with the site +process writes the functions it entered. `mirth-lab coverage` joins those with the site tables. MIRTH_RUST= BUILD_DIR= MIRTH_WATCH=rustc/coverage.toml rustc/build.sh MIRTH_OUT= RUSTC=//stage1/bin/rustc cargo build # any number of builds - rustc/coverage.py --sites /mirth-sites --logs [--files] [--unhit rustc_borrowck] + target/release/mirth-lab coverage --sites /mirth-sites --logs [--files] [--unhit rustc_borrowck] The instrumented compiler is the same source as `rustc-verify12` (pinned nightly plus the local patches); 72,774 functions in 80 crates. @@ -54,7 +54,7 @@ programs that fail to compile, built incrementally, will. ## With option configurations -`rustc/coverage-flags.py` builds sink once per row of a pairwise transitions table (clean with +`mirth-lab coverage-flags` builds sink once per row of a pairwise transitions table (clean with the row's A options, then rebuilt after one random edit with its B options). 60 rows, 1,586 rustc processes in all with the run above: **30,657 functions (42.8%)**. The largest gains: MIR optimization passes (+460, from `-Copt-level` and `-Zmir-opt-level`), `rustc_trait_selection` @@ -63,7 +63,7 @@ handling), `rustc_thread_pool` (+226, `-Zthreads`). ## rustc's UI tests -`rustc/ui-coverage.py run` compiles each UI test file (the way its `//@` headers say) with the +`mirth-lab ui-coverage run` compiles each UI test file (the way its `//@` headers say) with the instrumented compiler and keeps the functions it reaches beyond sink's; `pick` chooses tests greedily. Of 20,419 files, 1,839 were skipped (auxiliary crates, other targets); together the rest reach **17,668 functions sink does not (61% of the compiler with sink's)**; **300 picked tests @@ -77,7 +77,7 @@ reach 13,626 of them (55%)**. The first few: | `attributes/malformed-attrs.rs` | 444 | | `abi/stack-protector.rs` | 432 | -`rustc/ui-fuzz.py` runs the picked tests through incremental rebuilds after the fuzzer's edits, +`mirth-lab ui-fuzz` runs the picked tests through incremental rebuilds after the fuzzer's edits, each compared with a clean build (status, diagnostics, outputs): error reporting and recovery under incremental compilation, which sink cannot reach. @@ -86,7 +86,7 @@ solver, while nightly defaults to the new one: [`solver.md`](solver.md). ### Through incremental rebuilds -`rustc/ui-fuzz.py` over every UI test that compiles standalone (18,553 files), up to 8 random +`mirth-lab ui-fuzz` over every UI test that compiles standalone (18,553 files), up to 8 random edits each: **123,063 edit, incremental rebuild and clean rebuild cycles**, each compared on exit status, diagnostics and outputs. Checked first on finding 7's shape (a warning from inline assembly, lost when a codegen unit is reused), which it finds in 2 of 37 edits. @@ -123,7 +123,7 @@ of rustc-hash share a name): `T: From`, a derive's `T: Encodable` for every field) and of the trait's associated types (`type Domain: JoinSemiLattice`), recursively. -`rustc/callgraph.py` computes reachability from the compiler's `main`s, every function with a +`mirth-lab callgraph` computes reachability from the compiler's `main`s, every function with a foreign ABI (callbacks from C, C++ and LLVM), every implementation of a trait from outside the compiler (which `std` may call), and every initializer. A call to a trait item reaches its implementations (class hierarchy analysis), but an implementation for one of the compiler's @@ -156,7 +156,7 @@ a normal error path and is not counted. Coverage is reported with and without th `rustc/coverage-suites.sh` runs a suite through compiletest (`./x.py test`) on the instrumented compiler, with `MIRTH_OUT` set, folding logs as they finish -(`rustc/coverage-compact.py`): compiletest handles what a standalone runner cannot (auxiliary +(`mirth-lab coverage-compact`): compiletest handles what a standalone runner cannot (auxiliary crates, every revision, `minicore` cross-targets, run-make). With `--keep-stage 0 --keep-stage 1`, and a refusal when the log shows the compiler compiling: a changed mirth runtime once made `x.py` rebuild the compiler without the instrumentation, so `build.sh` now keeps a runtime per @@ -201,11 +201,11 @@ logs the same way. | run | functions reached | |---|---:| -| ui tests through incremental rebuilds (`ui-fuzz.py`, 18,553 tests, 3 edits each) | 44,196 | -| `coverage-generators.py`: every `--print` request on the host and on all 334 targets; minicore and an ABI file compiled for every target at `-Copt-level=0` and 3; the 300 picked UI tests under 48 debugging and printing options | 41,768 | -| `coverage-generators.py --only links`: a binary, cdylib, staticlib and dylib on minicore for every target with `-Clinker=true`, under 11 sets of linker options | 18,228 | +| ui tests through incremental rebuilds (`mirth-lab ui-fuzz`, 18,553 tests, 3 edits each) | 44,196 | +| `mirth-lab coverage-generators`: every `--print` request on the host and on all 334 targets; minicore and an ABI file compiled for every target at `-Copt-level=0` and 3; the 300 picked UI tests under 48 debugging and printing options | 41,768 | +| `mirth-lab coverage-generators --only links`: a binary, cdylib, staticlib and dylib on minicore for every target with `-Clinker=true`, under 11 sets of linker options | 18,228 | **All together, with sink and its option configurations: 50,762 of the 62,516 functions that can run (81.2%); without the 911 that only panic, 50,715 of 61,605 (82.3%).** -`rustc/coverage-report.sh` recomputes this; `rustc/callgraph.py --gaps ` lists the rest +`rustc/coverage-report.sh` recomputes this; `mirth-lab callgraph --gaps ` lists the rest by crate and file, largest first. What is left, and the plan for it: [coverage-handoff.md](coverage-handoff.md). diff --git a/docs/flags.md b/docs/flags.md index 0414182..b73b3a5 100644 --- a/docs/flags.md +++ b/docs/flags.md @@ -6,7 +6,7 @@ builds cover every pair or triple of option values? Pinned compiler: nightly-202 ## Enumeration -`rustc/flag-universe.py` reads `compiler/rustc_session/src/options.rs`: +`mirth-lab flag-universe` reads `compiler/rustc_session/src/options.rs`: | | options | enumerable | free-form (left out) | |---|---|---|---| @@ -17,7 +17,7 @@ builds cover every pair or triple of option values? Pinned compiler: nightly-202 An option's domain is absence plus: `yes`/`no` for a boolean, present for an option without a value, the backticked values in its parser's description for an enumerated one, `1` and `16` for a number. Strings, paths, lists, target features, passes and the like are left out, -except 19 options with hand-picked samples (`SAMPLES` in `flag-universe.py`): the first tables +except 19 options with hand-picked samples (`SAMPLES` in `mirth-lab flag-universe`): the first tables below were made before these were added, and left out `-Copt-level` (its parser takes a string), so the walks there ran at opt-level 0. With the samples, 235 options are enumerable, 556 values were tried alone and 508 accepted; the pairs were not tried again. @@ -64,7 +64,7 @@ The 47 values rejected alone fall into four groups: `-Zlink-only`, `-Zimplicit-sysroot-deps=no` (needs `#![no_std]`). So the declared constraints are few: 1 pairwise exclusion and 17 implications. They are in -`rustc/flag-model.py`, which writes a [PICT](https://github.com/microsoft/pict) model. +`mirth-lab flag-model`, which writes a [PICT](https://github.com/microsoft/pict) model. ## Covering array sizes @@ -100,14 +100,14 @@ Each row is a clean build with A, then a rebuild with B, compared with a clean b ## Walking transitions on sink -`rustc/flag-walk.py` takes a table from `flag-model.py --transitions --cargo` and, per row, +`mirth-lab flag-walk` takes a table from `mirth-lab flag-model --transitions --cargo` and, per row, builds `fixtures/sink` clean with the A options, rebuilds with the B options, builds clean with the B options, and compares (metadata, object code, binaries, diagnostics, the program's output). A difference is checked against up to 12 more clean builds first, and reported as P5 (nondeterminism) if clean builds differ among themselves. A real workspace adds constraints the trivial crate does not show (`CARGO_DROP` and -`CARGO_NEEDS` in `flag-model.py`): `-Clto` is rejected for rlibs and dylibs, Cargo's target +`CARGO_NEEDS` in `mirth-lab flag-model`): `-Clto` is rejected for rlibs and dylibs, Cargo's target probe fails on values that need another option, there are no sanitizer runtimes here, `-Cprefer-dynamic` with `-Cpanic=abort` or LTO cannot link, and so on. Single values on sink: 428 of 466 build; no single option, set the same in both sessions, changes a rebuild. @@ -145,9 +145,9 @@ a message (both left out of `--cargo` models). A walk that keeps hitting a known bug finds nothing behind it, and excluding the bug from the model loses the coverage. So a found bug is patched locally and the walk resumes: -1. `flag-walk.py --pause-on-finding` stops taking rows at the first finding and writes +1. `mirth-lab flag-walk --pause-on-finding` stops taking rows at the first finding and writes `PAUSED` (the row and what it found); rows in flight finish. -2. Minimize (`flag-min.py` for failing rows, delta debugging by hand for differences), +2. Minimize (`mirth-lab flag-min` for failing rows, delta debugging by hand for differences), reproduce with plain rustc, write the facts (`docs/hunt/`). 3. Patch `~/mirth-work/rust` (a stopgap in `docs/hunt/*-stopgap.patch`; the reproduction in `docs/hunt/repro.sh` must change), build stage 1, freeze it as a new toolchain. @@ -170,10 +170,10 @@ each is about a day. ## Reproduce - rustc/flag-universe.py --rustc --source --work --jobs 10 - rustc/flag-model.py all model.txt # or untracked / tracked, --transitions + mirth-lab flag-universe --rustc --source --work --jobs 10 + mirth-lab flag-model all model.txt # or untracked / tracked, --transitions pict model.txt /o:2 /r:1 > rows.tsv - rustc/flag-rows.py rows.tsv --emit=metadata - rustc/flag-model.py untracked tr.txt --transitions --cargo + mirth-lab flag-rows rows.tsv --emit=metadata + mirth-lab flag-model untracked tr.txt --transitions --cargo pict tr.txt /o:3 /r:1 > tr.tsv - rustc/flag-walk.py --rustc --fixture fixtures/sink --flags --table tr.tsv --work + mirth-lab flag-walk --rustc --fixture fixtures/sink --flags --table tr.tsv --work diff --git a/docs/grammar.md b/docs/grammar.md index 6596162..6b8411a 100644 --- a/docs/grammar.md +++ b/docs/grammar.md @@ -1,11 +1,11 @@ # Covering the grammar `fixtures/sink` is the code the fuzzer, the flag walks and the history of findings run on, so -what it does not contain is never tested. `rustc/grammar-coverage.py` measures it against +what it does not contain is never tested. `mirth-lab grammar-coverage` measures it against [Ur](https://github.com/PowderworksCode/codebase/tree/main/projects/ur)'s Rust grammar (`ecosystems/rust/language/*.rsc`): - rustc/grammar-coverage.py --ur --grammar /ecosystems/rust/language fixtures/sink + target/release/mirth-lab grammar-coverage --ur --grammar /ecosystems/rust/language fixtures/sink It parses every `.rs` file with `ur parse --tree` and counts two things: diff --git a/docs/hunt.md b/docs/hunt.md index 69e35b7..e52fcc5 100644 --- a/docs/hunt.md +++ b/docs/hunt.md @@ -36,15 +36,15 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 10 | with `-g -Clto=thin` and incremental compilation, clean builds give different object files from run to run, and since 1.90 (rust-lld) different binaries | **looks new**; found by the flag transitions walk ([`flags.md`](flags.md)), first as a rebuild differing from a clean build; objects differ since at least 1.60; cause found (ThinLTO input in codegen completion order); [facts](hunt/thinlto-module-order.md), local stopgap: inputs sorted by name ([patch](hunt/thinlto-order-stopgap.patch)) | | 11 | an incremental rebuild after an edit panics ("`trimmed_def_paths` called, diagnostics were expected but none were emitted") when the previous session had `-Zprint-type-sizes` and the crate has an `async fn` awaiting another | **looks new**; found by the three-way walk over untracked option transitions ([`flags.md`](flags.md)); since 1.79; cause found (an awaited type formatted with trimmed paths inside `layout_of`); [facts](hunt/print-type-sizes-trimmed-paths.md), local stopgap: the field type formatted without trimmed paths ([patch](hunt/print-type-sizes-trimmed-stopgap.patch)) | | 12 | rustc segfaults in LLVM's DWARF emission with `-g -Crelocation-model=rwpi` on x86_64 when the crate has a writable static | **looks new**; stable flags; found by trying `-Crelocation-model` values on sink ([`flags.md`](flags.md)); since 1.60 (LLVM 14); [facts](hunt/rwpi-debuginfo-segfault.md), local stopgap: `rwpi` and `ropi-rwpi` rejected off ARM ([patch](hunt/rwpi-stopgap.patch)) | -| 13 | LLVM's machine outliner (`-Cllvm-args=-enable-machine-outliner`) segfaults with retpolines at `-Copt-level` 1 and up, and fails in other combinations (`-Zcf-protection` with `-Zpatchable-function-entry`, the large code model) | in LLVM, reproduced with `llc` alone; unstable or raw LLVM flags; found by the flag walk with `flag-min.py`; since at least 1.71; [facts](hunt/llvm-retpoline.md); excluded from the models | +| 13 | LLVM's machine outliner (`-Cllvm-args=-enable-machine-outliner`) segfaults with retpolines at `-Copt-level` 1 and up, and fails in other combinations (`-Zcf-protection` with `-Zpatchable-function-entry`, the large code model) | in LLVM, reproduced with `llc` alone; unstable or raw LLVM flags; found by the flag walk with `mirth-lab flag-min`; since at least 1.71; [facts](hunt/llvm-retpoline.md); excluded from the models | | 14 | LLVM 23 lowers a retpoline tail call under the large code model to the APX instruction `jmpabs`: binaries die with SIGILL on CPUs without APX (every program that allocates), dylibs fail to link | in LLVM, a regression in LLVM 23 (fine on LLVM 22.1.8); unstable flags in rustc; reproduced with `llc` on 13 lines of IR; [facts](hunt/llvm-retpoline.md); excluded from the models | | 15 | under the incomplete `guard_patterns` feature, a guard pattern's guard is ignored: `Some(x if x > 3)` matches `Some(2)`, and the binding cannot be used in the arm | incomplete feature; found while covering nightly syntax in sink ([`grammar.md`](grammar.md)); [facts](hunt/guard-patterns-ignored.md), no fix | | 16 | with `-Zcache-proc-macros=yes -Zmetadata-crate-hash=no`, an incremental rebuild of a crate using derives gets a different crate hash (SVH) from a clean build after an edit upstream | unstable options (one "potentially unsound"); found by the fuzzer under walk configurations; not root-caused; [facts](hunt/cached-proc-macros-crate-hash.md); excluded from the models | | 17 | with `-Zunleash-the-miri-inside-of-you`, the "skipping const checks" warning is not shown again on an incremental rebuild | testing-only option; found by the UI-test fuzzer ([`coverage.md`](coverage.md)); since at least 1.60; [facts](hunt/unleash-warning-lost.md); tests using the option skipped | -| 18 | after a fatal error (a missing lang item), an incremental rebuild reports fewer errors than a clean build: the fatal error is reached in a different query order | diagnostics only; found by the UI-test fuzzer; stock nightly; [facts](hunt/fatal-error-order.md); labelled known in `ui-fuzz.py` | +| 18 | after a fatal error (a missing lang item), an incremental rebuild reports fewer errors than a clean build: the fatal error is reached in a different query order | diagnostics only; found by the UI-test fuzzer; stock nightly; [facts](hunt/fatal-error-order.md); labelled known in `mirth-lab ui-fuzz` | | 19 | on riscv64 and loongarch64, an `extern "C"` call passes an `i32` (or narrower integer) that lands on the stack without sign-extending it; a clang-compiled callee reads the slot as already extended | **looks new**; ABI, stable code; found by the ABI differential against clang ([`checks.md`](checks.md)); since at least 1.80; cause found (extension only `if *avail_gprs >= 1` in `callconv/riscv.rs`, same in `loongarch.rs`); [facts](hunt/riscv-stack-arg-extension.md) | | 20 | on RISC-V and LoongArch hard-float targets, a `repr(C)` struct of one float and one pointer is passed in a floating-point and an integer register; clang passes it by the integer convention, so C and Rust disagree on where it is | **looks new**; ABI, stable code; found by the ABI differential; since at least 1.80; cause found (`Primitive::Pointer` counted as an integer in `should_use_fp_conv_helper`, `callconv/riscv.rs` and `loongarch.rs`); [facts](hunt/riscv-float-pointer-struct.md) | -| 21 | `-Zvalidate-mir` rejects MIR the compiler builds from accepted code: projections into `#[repr(simd)]` types (banned by MCP#838) in 9 SIMD tests, and an unsize coercion to `Pin>` in `async-await/issue-86507.rs` | found by the internal-checks sweep (`crash-diff.py`); stock nightly with `-Zvalidate-mir`; compiletest does not validate UI tests; [facts](hunt/internal-checks.md) | +| 21 | `-Zvalidate-mir` rejects MIR the compiler builds from accepted code: projections into `#[repr(simd)]` types (banned by MCP#838) in 9 SIMD tests, and an unsize coercion to `Pin>` in `async-await/issue-86507.rs` | found by the internal-checks sweep (`mirth-lab crash-diff`); stock nightly with `-Zvalidate-mir`; compiletest does not validate UI tests; [facts](hunt/internal-checks.md) | | 22 | the new trait solver trips a debug assertion in region outlives (`regions.rs:37`, `!type_outlives.has_non_rigid_aliases()`) on 5 UI tests | debug-assertion builds with nightly's default solver; hidden in CI by compiletest's solver pin; a sibling of closed #160206; [facts](hunt/internal-checks.md) | | 23 | an `attempt to add with overflow` in `ty/instance.rs:421` compiling `recursion/issue-83150.rs` under the new solver | overflow-checked builds; hidden by the solver pin; [facts](hunt/internal-checks.md) | | 24 | `-Zvalidate-mir` rejects a move of a dereferenced unsized place into a call (`unsized-locals/unsized-exprs2.rs`) | incomplete `unsized_fn_params`; [facts](hunt/internal-checks.md) | @@ -52,6 +52,10 @@ with `-Zthreads=8`. `rustc/check.sh wide` runs the ordinary checks. | 26 | meilisearch (edition 2021) stops compiling on nightly-2026-10-06: `Ok(()) as Result<_>` now infers `!` (never-type fallback in edition 2021), and neither 1.98 nor the July nightly warned, also with the future-compatibility lints on | **looks new** as a lint false negative; found by release-to-release; reduced to 11 lines; [facts](hunt/release-regressions.md) | | 27 | under the new trait solver (nightly's default), a type parameter that appears only in a projection (`T0::Of<'_>`) of a function-pointer coercion is not inferred (E0283); breaks surrealdb through `diskann-wide 0.54.0` | **known, intended**: `diskann-wide` is listed in #160895 ("higher-ranked associated type", the intended breakage of trait-system-refactor-initiative#168; 0.55 not yet patched); surrealdb is an affected project not on that list; found by release-to-release; [facts](hunt/release-regressions.md) | | 28 | glob-import ambiguity depends on item order: with two modules re-exporting each other's globs, one order is E0659 and the other compiles and calls a different function (1.98, nightly); the accepted order has swapped between releases | **looks new**; stable code; found by the equivalent-rewrite differential (`reorder`) on 3 UI tests; [facts](hunt/glob-ambiguity-order.md) | +| 29 | machine-applicable lint fixes (what `cargo fix` applies unasked) break builds: `unused_variables` turns `ref b` into a moving `_b` and renames only the declaration of variables mentioned elsewhere, `unused_mut` changes one or-pattern alternative or a variable a `move` closure assigns, `unused_imports` removes a glob that resolution needs | **looks new**; stable 1.98; found by the suggestions-apply check (111 lint fixes in UI tests); six 3–7 line reductions; [facts](hunt/lint-fixes-break-builds.md) | +| 30 | compiler-internal debug output in user-facing diagnostics: under the default (new) solver, E0308 help suggests `as fn(?0t) -> ?0t`; an E0391 cycle note prints `Binder { value: ConstEvaluatable(AliasConst(… DefId(0:7 ~ …` (blessed in `offset-of/inside-array-length.stderr`) | low, diagnostics; found by the diagnostic-invariants check over 18,374 UI tests (excluding tests that ask for verbose output); the first not in CI because of the solver pin ([`solver-triage.md`](solver-triage.md) item I) | +| 31 | `#[rustc_main]` on a struct, impl, trait or module, on stable: after the expected E0658 and "cannot be used on structs", rustc ICEs ("unexpected sort of node in fn_sig()", `collect.rs`): the item is still taken as the entry point | **looks new**, low (error recovery, internal attribute); regression between 1.91.0 and 1.93.0; found by the feature-gate check; [repro](hunt/tests/rustc-main-on-struct.rs) | +| 32 | new-solver compile-time regression: a chain of N `.map()` calls type-checks in 4.5 s / 520 MB at N=200 on nightly-2026-08-03 and 37–58 s / 2.0–2.9 GB from nightly-2026-08-04, with a new "overflow evaluating the requirement `Map<…>: Iterator`" future-compat warning; nightly's default solver is the new one, so default builds regressed from 2.2 s (old solver, July) to 53 s | **looks new**, medium (compile time, realistic code shape); bisected over nightlies to #160254 (the only solver PR in the range); found by the scaling check; [facts](hunt/iter-chain-solver-regression.md) | Findings 1 and 2 are single-threaded: an ordinary `cargo build`, an edit, another `cargo build`, and the metadata differs from a clean build of the edited source. Both come diff --git a/docs/hunt/cached-proc-macros-crate-hash.md b/docs/hunt/cached-proc-macros-crate-hash.md index a7be33b..eb2f8a5 100644 --- a/docs/hunt/cached-proc-macros-crate-hash.md +++ b/docs/hunt/cached-proc-macros-crate-hash.md @@ -19,5 +19,5 @@ depends on `sink-core`, where the edit is. reverts to computing the crate hash from the HIR. Since the crate hash is what dependents compare, a wrong one can make them reuse or reject the wrong things. -**How mirth found it.** The fuzzer under walk configurations (`rustc/flag-fuzz.py`), first +**How mirth found it.** The fuzzer under walk configurations (`mirth-lab flag-fuzz`), first edit of a row; excluded from the models since (a known bug without a stopgap). diff --git a/docs/hunt/fatal-error-order.md b/docs/hunt/fatal-error-order.md index 554ed20..5e8623b 100644 --- a/docs/hunt/fatal-error-order.md +++ b/docs/hunt/fatal-error-order.md @@ -21,6 +21,6 @@ rebuild, a query reached earlier (while checking what can be reused) hits the mi outside that loop, and the first fatal error ends the session. Any fatal error reached in a different order would do the same. -**How mirth found it.** `rustc/ui-fuzz.py` over rustc's UI tests, after an edit renamed the lang +**How mirth found it.** `mirth-lab ui-fuzz` over rustc's UI tests, after an edit renamed the lang item. Such differences (the rebuild's diagnostics a subset of the clean build's, each missing line repeating a message the rebuild has) are labelled known since. diff --git a/docs/hunt/glob-ambiguity-order.md b/docs/hunt/glob-ambiguity-order.md index 8f2d388..2f1d10d 100644 --- a/docs/hunt/glob-ambiguity-order.md +++ b/docs/hunt/glob-ambiguity-order.md @@ -1,6 +1,6 @@ # Glob-import ambiguity depends on the order of items -Facts for finding 28. Found by the equivalent-rewrite differential (`rustc/rewrite-diff.py`, the +Facts for finding 28. Found by the equivalent-rewrite differential (`mirth-lab rewrite-diff`, the `reorder` rewrite: top-level items in reverse order, `use` items first) on `tests/ui/imports/ambiguous-9.rs`; the same rewrite also flips `imports/ambiguous-14.rs` (error → compiles) and `imports/overwrite-different-ambig-2.rs` (compiles → error). diff --git a/docs/hunt/guard-patterns-ignored.md b/docs/hunt/guard-patterns-ignored.md index 841bd72..490b1aa 100644 --- a/docs/hunt/guard-patterns-ignored.md +++ b/docs/hunt/guard-patterns-ignored.md @@ -29,4 +29,4 @@ the same name (`Some(x if x > 3) | Some(x if x == 0) => x`) gives the same error **Versions.** nightly-2026-07-18 and nightly-2026-10-06 (and the local compiler). **How mirth found it.** Writing nightly syntax into `fixtures/sink` to cover every -alternative of Ur's Rust grammar (`rustc/grammar-coverage.py`); the fixture's runtime check. +alternative of Ur's Rust grammar (`mirth-lab grammar-coverage`); the fixture's runtime check. diff --git a/docs/hunt/internal-checks.md b/docs/hunt/internal-checks.md index 1af986d..8e75653 100644 --- a/docs/hunt/internal-checks.md +++ b/docs/hunt/internal-checks.md @@ -1,6 +1,6 @@ # rustc's internal checks on the UI tests -Facts for findings 21 to 24. Found by `rustc/crash-diff.py` ([`checks.md`](../checks.md), check 19): +Facts for findings 21 to 24. Found by `mirth-lab crash-diff` ([`checks.md`](../checks.md), check 19): every standalone UI test (18,624) compiled with the release compiler under test and again with a compiler built from the same tree with `rust.debug-assertions = true`, `rust.debug-assertions-std = true` and `rust.overflow-checks = true`, plus `-Zvalidate-mir`. @@ -80,4 +80,4 @@ rust.debug-assertions-std=true --set rust.overflow-checks=true` (mirth: `~/mirth ## Local stopgap None: these are checks failing, not wrong output, and they do not affect mirth's incremental -checks. `crash-diff.py` takes them as `--known` (`rustc/crash-known.txt`). +checks. `mirth-lab crash-diff` takes them as `--known` (`rustc/crash-known.txt`). diff --git a/docs/hunt/issue-untracked-options.md b/docs/hunt/issue-untracked-options.md index a21fa44..06d8064 100644 --- a/docs/hunt/issue-untracked-options.md +++ b/docs/hunt/issue-untracked-options.md @@ -83,7 +83,7 @@ warnings and keeps its full build. A tool that does this for every untracked boolean option, comparing metadata, each rlib member (object names have their incremental session suffix removed, since the objects are otherwise identical), diagnostics and files written, is -[`rustc/audit-options.py`](../../rustc/audit-options.py) in mirth: +[`mirth-lab audit-options`](../../crates/mirth-lab/src/tools/audit_options.rs) in mirth: ```text (control: no option) same diff --git a/docs/hunt/iter-chain-solver-regression.md b/docs/hunt/iter-chain-solver-regression.md new file mode 100644 index 0000000..13a04d5 --- /dev/null +++ b/docs/hunt/iter-chain-solver-regression.md @@ -0,0 +1,69 @@ +# New solver: long iterator chains take 10× longer and 4× the memory since nightly-2026-08-04 + +Facts for finding 32. Found by the scaling check (`mirth-lab scale-check`, shape `iter-chain`: +growth exponent of compile time 2.6–3.5 at N ≤ 100). + +## What happens + +[`tests/iter-chain-200.rs`](tests/iter-chain-200.rs) is one statement: +`(0u64..10).map(|x| x.wrapping_add(0)) … .map(|x| x.wrapping_add(199)).sum()`, 200 `map` +calls. User time and peak memory of `rustc iter-chain-200.rs` (-Copt-level=0): + +| toolchain | default | `-Znext-solver=coherence` (old solver) | `-Znext-solver=globally` | +|---|---|---|---| +| 1.80.0 | 5.7 s, 115 MB | | | +| 1.90.0 | 7.0 s, 125 MB | | | +| 1.98.0 | 2.3 s, 118 MB | | | +| nightly-2026-07-18 | 2.2 s, 119 MB | 2.2 s, 118 MB | 4.6 s, 517 MB | +| nightly-2026-08-03 | | | 4.5 s, 519 MB | +| nightly-2026-08-04 | | | 37.3 s, 2.42 GB | +| nightly-2026-10-06 | 53.4 s, 2.04 GB | 6.1 s, 131 MB | 52.8 s, 2.04 GB | + +At the mirth pin (default solver): N=50 0.19 s, N=100 1.2 s, N=200 68.6 s and 2.0 GB. +`-Ztime-passes` puts 65 of 73 s in `type_check_crate`. From nightly-2026-08-04 the build also +warns once, "overflow evaluating the requirement `Map…>>: Iterator`" +("this was previously accepted by the compiler but is being phased out"); earlier nightlies +do not warn. + +Two separate changes show in the table: the new-solver slowdown between 2026-08-03 and +2026-08-04 (this finding), and the default switching to the new solver between July and +October ([`solver.md`](../solver.md)). The old solver also went from 2.2 s to 6.1 s over the +same period; not bisected. + +## Bisection + +Over nightlies, `-Znext-solver=globally`, "bad" above 20 s user time: 2026-07-28 4.5 s, +2026-08-02 4.7 s, 2026-08-03 4.5 s, 2026-08-04 37.3 s, 2026-08-07 38.0 s, 2026-08-27 58.4 s. + +- last good: nightly-2026-08-03, 11177f2235f0c842b00f82c558ad9480c0c3a895 +- first bad: nightly-2026-08-04, 504869653f510b279c542e65ccd1ea9710c119ba + +The range has four merges. One touches the trait solver: rollup #160414, containing #160254 +("fix-fcw-missing", commit 1489e477b62 "rerun even if the goal has ty vars"). Not confirmed by +building a compiler with it reverted. + +## Where + +`compiler/rustc_next_trait_solver/src/solve/eval_ctxt/mod.rs`, +`maybe_evaluate_root_goal_with_higher_recursion_limit` (and the proof-tree variant). #160254 +removed this early return: + +```rust + // Some goals no longer overflow after the stalled infers are resolved. + // Thus we don't have to rerun eagerly here. + let has_stalled_infers = match predicate.kind().skip_binder() { … }; + if has_stalled_infers { + return; + } +``` + +so a root goal that overflows while it still has inference variables is now re-evaluated with +twice the recursion limit (to decide whether to emit the overflow future-compat warning). The +goal in the warning has one (`Map<_, …>`, innermost). Whether the rerun happens once per +fulfillment iteration, which would fit the growth with N, was not checked. + +## Scope + +Any method chain long enough that the solver overflows on the receiver's trait goal while its +innermost type is still being inferred. 200 is long for hand-written code; generated code and +builder- or iterator-heavy macros reach it. diff --git a/docs/hunt/lint-fixes-break-builds.md b/docs/hunt/lint-fixes-break-builds.md new file mode 100644 index 0000000..fd1f67b --- /dev/null +++ b/docs/hunt/lint-fixes-break-builds.md @@ -0,0 +1,50 @@ +# Machine-applicable lint fixes that break builds + +Facts for finding 29. Found by the suggestions-apply check (`mirth-lab suggest-diff`, check 18 in +[`checks.md`](../checks.md)): every `MachineApplicable` suggestion of every UI test without +`//@ run-rustfix` (17,945 tests, 7,385 suggestions), each applied alone and compiled again. A +lint's machine-applicable fix is what `cargo fix` and `cargo clippy --fix` apply without asking; +a lint never stops a build, so its fix must not introduce an error. 111 lint fixes did. + +## Six shapes, reduced, on stable 1.98.0 + +Each file in [`tests/lint-fixes/`](tests/lint-fixes/) compiles with warnings; applying the one +named suggestion (`apply-one-suggestion.py 1.98.0`) gives the error shown. The same on +nightly-2026-10-06. + +| file | lint | code | suggestion | after applying it | +|---|---|---|---|---| +| `ref.rs` | `unused_variables` | `let ref b = u; drop(u);` | `ref b` → `_b` | E0382 use of moved value: the binding now moves | +| `update.rs` | `unused_variables` | `fn test(f: Foo) { Foo { foo: 4, ..f } }` (every field given) | `f` → `_f` | E0425 cannot find value `f`: the base still names it | +| `unreach.rs` | `unused_variables` | `let x = f(); let _ = x;` with `f() -> Never` | `x` → `_x` | E0425: the later (unreachable) use still names `x` | +| `closure.rs` | `unused_variables` | `let mut x = 0; to_fn(move \|\| { x = 42; })` | `mut x` → `_x` | E0425: the closure still names `x` | +| `closure.rs` | `unused_mut` | the same | remove `mut` | E0594 cannot assign to `x`: the closure assigns | +| `ormut.rs` | `unused_mut` | `Ok(mut y) \| &Err(mut y) => drop(y)` | remove one `mut` | E0409 bound inconsistently across alternatives | +| `glob.rs` | `unused_imports` | `mod one_private { use crate::m::*; pub use crate::m::*; } use crate::one_private::S;` | remove `pub use crate::m::*;` | E0603 struct import `S` is private | + +In the UI tests the same shapes appear as: `ref`/`ref mut` bindings, including `ref x @ pat` and +unsized `ref rest @ ..`, where E0277 follows (25 suggestions); variables mentioned again only in +unreachable code, struct-update bases or closures (48); or-patterns where one alternative's +`mut` or name is changed alone (12); glob re-exports that take part in ambiguity or visibility +(5). + +## Expected + +A machine-applicable suggestion keeps the program compiling, with the same meaning: +`ref _b` (or `_`) for a `ref` binding, renaming every mention or not offering the rename where +the variable is mentioned elsewhere, removing `mut` from every alternative, and not calling an +import unused when removing it changes resolution. + +## Also found, lower priority + +- Lint suggestions inside macro input that the macro then fails to match: `unexpected_cfgs` + (`FALSE` → `false` in a macro's meta argument), `missing_abi` next to a literal with a suffix, + `unused_parens` around `let` chains passed to a macro. +- 106 error suggestions that leave the same error, and 63 whose fix no longer parses, mostly in + parser-recovery and `fn_delegation` tests (error-recovery suggestions on deliberately broken + code). +- An ICE after applying an E0308 fix (`consts/const-eval/array-len-mismatch-type.rs`, stable + 1.98): const evaluation runs on a body that failed with E0277 and panics with "expected wide + pointer extra data"; that ICE family has open #154779. + +Not found in the issue tracker (searched for each shape). diff --git a/docs/hunt/llvm-retpoline.md b/docs/hunt/llvm-retpoline.md index 2806fce..17fc504 100644 --- a/docs/hunt/llvm-retpoline.md +++ b/docs/hunt/llvm-retpoline.md @@ -34,14 +34,14 @@ this is LLVM's. The outliner is not on by default for x86-64, which may make it low-priority report there. **The outliner fails in other combinations too** (minimized from walk rows with -`rustc/flag-min.py`, on sink): +`mirth-lab flag-min`, on sink): - `-Cllvm-args=-enable-machine-outliner -Copt-level=3 -Zcf-protection=full -Zpatchable-function-entry=4,2 -Cdebuginfo=none`: SIGSEGV. - `-Cllvm-args=-enable-machine-outliner -Copt-level=2 -Ccode-model=large` (with a few more options): `error: symbol '.L6$pb' can not be undefined in a subtraction expression`. -The models now leave the outliner out entirely (`DROP` in `rustc/flag-model.py`). +The models now leave the outliner out entirely (`DROP` in `mirth-lab flag-model`). ## 14: LLVM 23 emits APX `jmpabs` for retpoline tail calls under the large code model @@ -85,4 +85,4 @@ nightly-2026-10-06 (LLVM 23.1.3). A regression in LLVM 23: the large-code-model the retpoline thunk is lowered to `jmpabs` without checking that APX is available. **How mirth found them.** The pairwise walk over all options with the sample values -(`docs/flags.md`), then `rustc/flag-min.py` on the rows that failed. +(`docs/flags.md`), then `mirth-lab flag-min` on the rows that failed. diff --git a/docs/hunt/no-prepopulate-link.md b/docs/hunt/no-prepopulate-link.md index 5ee6b4c..e475223 100644 --- a/docs/hunt/no-prepopulate-link.md +++ b/docs/hunt/no-prepopulate-link.md @@ -23,7 +23,7 @@ too. With `-Clto=thin` it links. Not specific to incremental compilation. binary and a cdylib fail as well (a dylib links): `rustc --crate-type bin -Ccodegen-units=16 -Clink-dead-code=yes -Cno-prepopulate-passes -Zthinlto=yes` on a `main` calling `core::mem::swap`. Found by minimizing the link failures of the pairwise walk -(`rustc/flag-min.py`). +(`mirth-lab flag-min`). **Versions.** Stable releases with `RUSTC_BOOTSTRAP=1`: links on 1.53.0 through 1.77.0, fails on 1.78.0 through 1.98.1 and nightly-2026-10-06. 1.78 is when these `ub_checks` diff --git a/docs/hunt/promoted-validation-generic.md b/docs/hunt/promoted-validation-generic.md index 9a4dbaf..10aea4f 100644 --- a/docs/hunt/promoted-validation-generic.md +++ b/docs/hunt/promoted-validation-generic.md @@ -1,6 +1,6 @@ # An invalid constant is accepted when its unused reference sits in a generic function -Facts for finding 25. Found by the equivalent-rewrite differential (`rustc/rewrite-diff.py`, the +Facts for finding 25. Found by the equivalent-rewrite differential (`mirth-lab rewrite-diff`, the `generic-wrap` rewrite: a function's body moved into a generic inner function called with `()`) on `tests/ui/consts/interior-mut-const-via-union.rs`. @@ -68,4 +68,4 @@ parameter it gives E0080 at every level. ## Local stopgap -None; not an incremental difference. `rewrite-diff.py` lists the test as known for `generic-wrap`. +None; not an incremental difference. `mirth-lab rewrite-diff` lists the test as known for `generic-wrap`. diff --git a/docs/hunt/release-regressions.md b/docs/hunt/release-regressions.md index b5a44cc..2382e22 100644 --- a/docs/hunt/release-regressions.md +++ b/docs/hunt/release-regressions.md @@ -1,6 +1,6 @@ # Regressions in real crates, nightly-2026-07-18 to nightly-2026-10-06 -Facts for findings 26 and 27. Found by the release-to-release check (`rustc/release-diff.py`, +Facts for findings 26 and 27. Found by the release-to-release check (`mirth-lab release-diff`, [`checks.md`](../checks.md) check 2): `cargo check --locked` of 87 popular repositories (`~/proofhouse-repos/rust`) under both nightlies. 53 behave the same, 18 fail on both, 10 could not fetch their locked dependencies. Of the 5 regressions, two are `allocative 0.3.4`, which enables diff --git a/docs/hunt/riscv-float-pointer-struct.md b/docs/hunt/riscv-float-pointer-struct.md index ff76089..db475ec 100644 --- a/docs/hunt/riscv-float-pointer-struct.md +++ b/docs/hunt/riscv-float-pointer-struct.md @@ -1,7 +1,7 @@ # RISC-V and LoongArch: a struct of a float and a pointer goes in the wrong registers Facts for finding 20. Found by the ABI differential ([`checks.md`](../checks.md), check 14: -`rustc/abi-diff.py`). +`mirth-lab abi-diff`). ## What happens diff --git a/docs/hunt/riscv-stack-arg-extension.md b/docs/hunt/riscv-stack-arg-extension.md index 5bf73ad..303166b 100644 --- a/docs/hunt/riscv-stack-arg-extension.md +++ b/docs/hunt/riscv-stack-arg-extension.md @@ -1,7 +1,7 @@ # riscv64 and loongarch64: integer arguments passed on the stack are not sign-extended Facts for finding 19. Found by the ABI differential ([`checks.md`](../checks.md), check 14: -`rustc/abi-diff.py`, rustc's `extern "C"` lowering against clang's for random C signatures). +`mirth-lab abi-diff`, rustc's `extern "C"` lowering against clang's for random C signatures). ## What happens diff --git a/docs/hunt/tests/iter-chain-200.rs b/docs/hunt/tests/iter-chain-200.rs new file mode 100644 index 0000000..075917d --- /dev/null +++ b/docs/hunt/tests/iter-chain-200.rs @@ -0,0 +1 @@ +fn main() { let s: u64 = (0u64..10).map(|x| x.wrapping_add(0)).map(|x| x.wrapping_add(1)).map(|x| x.wrapping_add(2)).map(|x| x.wrapping_add(3)).map(|x| x.wrapping_add(4)).map(|x| x.wrapping_add(5)).map(|x| x.wrapping_add(6)).map(|x| x.wrapping_add(7)).map(|x| x.wrapping_add(8)).map(|x| x.wrapping_add(9)).map(|x| x.wrapping_add(10)).map(|x| x.wrapping_add(11)).map(|x| x.wrapping_add(12)).map(|x| x.wrapping_add(13)).map(|x| x.wrapping_add(14)).map(|x| x.wrapping_add(15)).map(|x| x.wrapping_add(16)).map(|x| x.wrapping_add(17)).map(|x| x.wrapping_add(18)).map(|x| x.wrapping_add(19)).map(|x| x.wrapping_add(20)).map(|x| x.wrapping_add(21)).map(|x| x.wrapping_add(22)).map(|x| x.wrapping_add(23)).map(|x| x.wrapping_add(24)).map(|x| x.wrapping_add(25)).map(|x| x.wrapping_add(26)).map(|x| x.wrapping_add(27)).map(|x| x.wrapping_add(28)).map(|x| x.wrapping_add(29)).map(|x| x.wrapping_add(30)).map(|x| x.wrapping_add(31)).map(|x| x.wrapping_add(32)).map(|x| x.wrapping_add(33)).map(|x| x.wrapping_add(34)).map(|x| x.wrapping_add(35)).map(|x| x.wrapping_add(36)).map(|x| x.wrapping_add(37)).map(|x| x.wrapping_add(38)).map(|x| x.wrapping_add(39)).map(|x| x.wrapping_add(40)).map(|x| x.wrapping_add(41)).map(|x| x.wrapping_add(42)).map(|x| x.wrapping_add(43)).map(|x| x.wrapping_add(44)).map(|x| x.wrapping_add(45)).map(|x| x.wrapping_add(46)).map(|x| x.wrapping_add(47)).map(|x| x.wrapping_add(48)).map(|x| x.wrapping_add(49)).map(|x| x.wrapping_add(50)).map(|x| x.wrapping_add(51)).map(|x| x.wrapping_add(52)).map(|x| x.wrapping_add(53)).map(|x| x.wrapping_add(54)).map(|x| x.wrapping_add(55)).map(|x| x.wrapping_add(56)).map(|x| x.wrapping_add(57)).map(|x| x.wrapping_add(58)).map(|x| x.wrapping_add(59)).map(|x| x.wrapping_add(60)).map(|x| x.wrapping_add(61)).map(|x| x.wrapping_add(62)).map(|x| x.wrapping_add(63)).map(|x| x.wrapping_add(64)).map(|x| x.wrapping_add(65)).map(|x| x.wrapping_add(66)).map(|x| x.wrapping_add(67)).map(|x| x.wrapping_add(68)).map(|x| x.wrapping_add(69)).map(|x| x.wrapping_add(70)).map(|x| x.wrapping_add(71)).map(|x| x.wrapping_add(72)).map(|x| x.wrapping_add(73)).map(|x| x.wrapping_add(74)).map(|x| x.wrapping_add(75)).map(|x| x.wrapping_add(76)).map(|x| x.wrapping_add(77)).map(|x| x.wrapping_add(78)).map(|x| x.wrapping_add(79)).map(|x| x.wrapping_add(80)).map(|x| x.wrapping_add(81)).map(|x| x.wrapping_add(82)).map(|x| x.wrapping_add(83)).map(|x| x.wrapping_add(84)).map(|x| x.wrapping_add(85)).map(|x| x.wrapping_add(86)).map(|x| x.wrapping_add(87)).map(|x| x.wrapping_add(88)).map(|x| x.wrapping_add(89)).map(|x| x.wrapping_add(90)).map(|x| x.wrapping_add(91)).map(|x| x.wrapping_add(92)).map(|x| x.wrapping_add(93)).map(|x| x.wrapping_add(94)).map(|x| x.wrapping_add(95)).map(|x| x.wrapping_add(96)).map(|x| x.wrapping_add(97)).map(|x| x.wrapping_add(98)).map(|x| x.wrapping_add(99)).map(|x| x.wrapping_add(100)).map(|x| x.wrapping_add(101)).map(|x| x.wrapping_add(102)).map(|x| x.wrapping_add(103)).map(|x| x.wrapping_add(104)).map(|x| x.wrapping_add(105)).map(|x| x.wrapping_add(106)).map(|x| x.wrapping_add(107)).map(|x| x.wrapping_add(108)).map(|x| x.wrapping_add(109)).map(|x| x.wrapping_add(110)).map(|x| x.wrapping_add(111)).map(|x| x.wrapping_add(112)).map(|x| x.wrapping_add(113)).map(|x| x.wrapping_add(114)).map(|x| x.wrapping_add(115)).map(|x| x.wrapping_add(116)).map(|x| x.wrapping_add(117)).map(|x| x.wrapping_add(118)).map(|x| x.wrapping_add(119)).map(|x| x.wrapping_add(120)).map(|x| x.wrapping_add(121)).map(|x| x.wrapping_add(122)).map(|x| x.wrapping_add(123)).map(|x| x.wrapping_add(124)).map(|x| x.wrapping_add(125)).map(|x| x.wrapping_add(126)).map(|x| x.wrapping_add(127)).map(|x| x.wrapping_add(128)).map(|x| x.wrapping_add(129)).map(|x| x.wrapping_add(130)).map(|x| x.wrapping_add(131)).map(|x| x.wrapping_add(132)).map(|x| x.wrapping_add(133)).map(|x| x.wrapping_add(134)).map(|x| x.wrapping_add(135)).map(|x| x.wrapping_add(136)).map(|x| x.wrapping_add(137)).map(|x| x.wrapping_add(138)).map(|x| x.wrapping_add(139)).map(|x| x.wrapping_add(140)).map(|x| x.wrapping_add(141)).map(|x| x.wrapping_add(142)).map(|x| x.wrapping_add(143)).map(|x| x.wrapping_add(144)).map(|x| x.wrapping_add(145)).map(|x| x.wrapping_add(146)).map(|x| x.wrapping_add(147)).map(|x| x.wrapping_add(148)).map(|x| x.wrapping_add(149)).map(|x| x.wrapping_add(150)).map(|x| x.wrapping_add(151)).map(|x| x.wrapping_add(152)).map(|x| x.wrapping_add(153)).map(|x| x.wrapping_add(154)).map(|x| x.wrapping_add(155)).map(|x| x.wrapping_add(156)).map(|x| x.wrapping_add(157)).map(|x| x.wrapping_add(158)).map(|x| x.wrapping_add(159)).map(|x| x.wrapping_add(160)).map(|x| x.wrapping_add(161)).map(|x| x.wrapping_add(162)).map(|x| x.wrapping_add(163)).map(|x| x.wrapping_add(164)).map(|x| x.wrapping_add(165)).map(|x| x.wrapping_add(166)).map(|x| x.wrapping_add(167)).map(|x| x.wrapping_add(168)).map(|x| x.wrapping_add(169)).map(|x| x.wrapping_add(170)).map(|x| x.wrapping_add(171)).map(|x| x.wrapping_add(172)).map(|x| x.wrapping_add(173)).map(|x| x.wrapping_add(174)).map(|x| x.wrapping_add(175)).map(|x| x.wrapping_add(176)).map(|x| x.wrapping_add(177)).map(|x| x.wrapping_add(178)).map(|x| x.wrapping_add(179)).map(|x| x.wrapping_add(180)).map(|x| x.wrapping_add(181)).map(|x| x.wrapping_add(182)).map(|x| x.wrapping_add(183)).map(|x| x.wrapping_add(184)).map(|x| x.wrapping_add(185)).map(|x| x.wrapping_add(186)).map(|x| x.wrapping_add(187)).map(|x| x.wrapping_add(188)).map(|x| x.wrapping_add(189)).map(|x| x.wrapping_add(190)).map(|x| x.wrapping_add(191)).map(|x| x.wrapping_add(192)).map(|x| x.wrapping_add(193)).map(|x| x.wrapping_add(194)).map(|x| x.wrapping_add(195)).map(|x| x.wrapping_add(196)).map(|x| x.wrapping_add(197)).map(|x| x.wrapping_add(198)).map(|x| x.wrapping_add(199)).sum(); println!("{}", s); } diff --git a/docs/hunt/tests/lint-fixes/apply-one-suggestion.py b/docs/hunt/tests/lint-fixes/apply-one-suggestion.py new file mode 100644 index 0000000..1f8fdea --- /dev/null +++ b/docs/hunt/tests/lint-fixes/apply-one-suggestion.py @@ -0,0 +1,16 @@ +import json,subprocess,sys +f=sys.argv[1]; tc=sys.argv[2] +r=subprocess.run(['rustc','+'+tc,'--edition','2021','--emit=metadata','--error-format=json','-o','/dev/null',f],capture_output=True,text=True) +src=open(f,'rb').read(); n=0 +for l in r.stderr.splitlines(): + try: d=json.loads(l) + except: continue + for c in [d]+d.get('children',[]): + parts=[(s['byte_start'],s['byte_end'],s['suggested_replacement']) for s in c.get('spans',[]) if s.get('suggestion_applicability')=='MachineApplicable' and s.get('suggested_replacement') is not None] + if not parts or d['level']!='warning': continue + n+=1; fixed=bytearray(src) + for a,b,t in sorted(parts,reverse=True): fixed[a:b]=t.encode() + g=f.replace('.rs',f'_fix{n}.rs'); open(g,'wb').write(fixed) + rr=subprocess.run(['rustc','+'+tc,'--edition','2021','--emit=metadata','-o','/dev/null',g],capture_output=True,text=True) + errs=[x for x in rr.stderr.splitlines() if x.startswith('error')] + print(f" {d['code']['code'] if d.get('code') else '-'}: {[src[a:b].decode() for a,b,_ in parts]} -> {[t for _,_,t in parts]} => {errs[0] if errs else 'compiles'}") diff --git a/docs/hunt/tests/lint-fixes/closure.rs b/docs/hunt/tests/lint-fixes/closure.rs new file mode 100644 index 0000000..7ad537c --- /dev/null +++ b/docs/hunt/tests/lint-fixes/closure.rs @@ -0,0 +1,2 @@ +fn to_fn(f: F) -> F { f } +fn main() { let mut x = 0; let _f = to_fn(move || { x = 42; }); } diff --git a/docs/hunt/tests/lint-fixes/glob.rs b/docs/hunt/tests/lint-fixes/glob.rs new file mode 100644 index 0000000..435d2c2 --- /dev/null +++ b/docs/hunt/tests/lint-fixes/glob.rs @@ -0,0 +1,7 @@ +mod m { pub struct S {} } +mod one_private { + use crate::m::*; + pub use crate::m::*; +} +use crate::one_private::S; +fn main() { let _ = S {}; } diff --git a/docs/hunt/tests/lint-fixes/ormut.rs b/docs/hunt/tests/lint-fixes/ormut.rs new file mode 100644 index 0000000..b3af771 --- /dev/null +++ b/docs/hunt/tests/lint-fixes/ormut.rs @@ -0,0 +1,6 @@ +fn main() { + let res: &Result = &Ok(1); + match res { Ok(mut x) | &Err(mut x) => { x += 1; drop::(x) } } + let r2: &Result = &Ok(1); + match r2 { Ok(mut y) | &Err(mut y) => drop::(y) } +} diff --git a/docs/hunt/tests/lint-fixes/ref.rs b/docs/hunt/tests/lint-fixes/ref.rs new file mode 100644 index 0000000..8e7780e --- /dev/null +++ b/docs/hunt/tests/lint-fixes/ref.rs @@ -0,0 +1,6 @@ +struct U; +fn main() { + let u = U; + let ref b = u; // unused `b`: suggestion replaces `ref b` with `_b`, which moves `u` + drop(u); +} diff --git a/docs/hunt/tests/lint-fixes/unreach.rs b/docs/hunt/tests/lint-fixes/unreach.rs new file mode 100644 index 0000000..2818afb --- /dev/null +++ b/docs/hunt/tests/lint-fixes/unreach.rs @@ -0,0 +1,3 @@ +enum Never {} +fn f() -> Never { panic!() } +fn main() { let x = f(); let _ = x; } diff --git a/docs/hunt/tests/lint-fixes/update.rs b/docs/hunt/tests/lint-fixes/update.rs new file mode 100644 index 0000000..4c6b87d --- /dev/null +++ b/docs/hunt/tests/lint-fixes/update.rs @@ -0,0 +1,3 @@ +struct Foo { foo: i32 } +fn test(f: Foo) -> i32 { let g = Foo { foo: 4, ..f }; g.foo } // `f` reported unused +fn main() { test(Foo { foo: 1 }); } diff --git a/docs/hunt/tests/rustc-main-on-struct.rs b/docs/hunt/tests/rustc-main-on-struct.rs new file mode 100644 index 0000000..847cc99 --- /dev/null +++ b/docs/hunt/tests/rustc-main-on-struct.rs @@ -0,0 +1,3 @@ +#[rustc_main] +pub struct S; +fn main() {} diff --git a/docs/hunt/unleash-warning-lost.md b/docs/hunt/unleash-warning-lost.md index 48b035a..f52c8ff 100644 --- a/docs/hunt/unleash-warning-lost.md +++ b/docs/hunt/unleash-warning-lost.md @@ -22,5 +22,5 @@ list at the end of the session. A rebuild that takes const checking's results fr incremental cache records nothing, so there is nothing to warn about: a side effect of a query that is not replayed, the same kind as finding 7. The option is for testing the compiler. -**How mirth found it.** `rustc/ui-fuzz.py` over rustc's UI tests ([`coverage.md`](../coverage.md)): +**How mirth found it.** `mirth-lab ui-fuzz` over rustc's UI tests ([`coverage.md`](../coverage.md)): the first edit to this test, its rebuild's diagnostics compared with a clean build's. diff --git a/docs/scale.md b/docs/scale.md index 15ae618..f55255e 100644 --- a/docs/scale.md +++ b/docs/scale.md @@ -2,11 +2,11 @@ Three ways to find more bugs than one fixture and ten hand-written edits can: -1. **Replay real history.** `rustc/replay.py` walks a crate's git history, oldest +1. **Replay real history.** `mirth-lab replay` walks a crate's git history, oldest first, building each commit incrementally on top of the last and again from scratch, and compares them (P6). Ten crates, up to 2,000 commits each. 2. **Fuzz edits on a large fixture.** `fixtures/sink` is a five-crate workspace with as - many stable language features as fit. `rustc/fuzz.py` makes random mechanical edits to it + many stable language features as fit. `mirth-lab fuzz` makes random mechanical edits to it and checks every incremental rebuild against a clean build. 3. **Survey past bugs for properties.** Agents read 1,000 fixed rustc bugs and extracted the invariants they violated; a second pass checked every citation. The result is @@ -49,7 +49,7 @@ codegen-llvm tests pass. ## History replay -`rustc/replay.py --rustc --repo --work [--commits N] [--from i --to j]` +`mirth-lab replay --rustc --repo --work [--commits N] [--from i --to j]` For each first-parent commit, oldest first: check it out, `cargo build --lib` incrementally on the previous commit's target directory, then build the same source from @@ -119,7 +119,7 @@ module, `#[no_mangle]` and `#[used]`, `cfg_attr`, `Any`, a glob re-export, an ex using `$crate`, a proc macro with a `mixed_site` binding, and a generic `#[inline]` function that `sink-mid` inlines from the metadata. -`rustc/fuzz.py --rustc --fixture fixtures/sink --work [--workers N]` +`mirth-lab fuzz --rustc --fixture fixtures/sink --work [--workers N]` Each worker keeps one evolving copy of the fixture. It makes a random edit, chosen from 16 kinds (a comment, a blank line, an indented line, swapped or moved or deleted items, a @@ -131,7 +131,8 @@ the same source is built from scratch at the same path, and every `.rmeta` and e macro's embedded metadata are compared (P6), the two binaries are run and their output compared, and ICEs, hangs and one-sided failures are reported. Every 40 kept edits the worker starts again from the pristine fixture. A finding keeps every edit since the last reset, and -`rustc/fuzz-replay.py` replays it exactly. +`mirth-lab fuzz-replay` replays it exactly. (`--seed` names an edit sequence of the Rust +fuzzer; runs of the earlier Python fuzzer, which drew from Python's generator, do not repeat.) Since [`shadow-mode.md`](shadow-mode.md), the fuzzer and the replay also run every build with the compiler's own check of what it reused (`RUSTC_VERIFY_REUSE`, on a compiler with diff --git a/docs/solver-triage.md b/docs/solver-triage.md index 9e31507..95fc352 100644 --- a/docs/solver-triage.md +++ b/docs/solver-triage.md @@ -6,7 +6,7 @@ solver everywhere, since #160895), checked against the tracking issue affected-crates tables and the "unintended breakage" list), rust-lang/rust issues and rust-lang/trait-system-refactor-initiative (tsri) issues, on 2026-10-10. -Sources: the UI-test solver differential ([`solver.md`](solver.md), `rustc/solver-diff.py`: +Sources: the UI-test solver differential ([`solver.md`](solver.md), `mirth-lab solver-diff`: 26 tests accepted by the old solver and rejected by the new, 3 crashes), the internal-checks sweep (findings 22 and 23), release-to-release (finding 27). @@ -30,6 +30,7 @@ Sources: the UI-test solver differential ([`solver.md`](solver.md), `rustc/solve | F | higher-ranked associated type no longer guides inference (`escaping-bounds`; `diskann-wide`) | 1 + crate | yes | #160895, tsri#168: intended | low | surrealdb as an affected project on #160895 | | G | type alias `impl Trait`: "does not constrain", a cycle | 2 | no | #160895: RPIT/TAIT handling changed | low | no | | H | `fn_delegation` with `impl Trait` returns: E0282 | 1 | no (incomplete) | no | low | optional | +| I | a help suggestion containing inference variables: "consider casting both fn items to fn pointers using `as fn(?0t) -> ?0t`" (`fn/fn_def_opaque_coercion_to_fn_ptr.rs`; the old solver gives no such help) | 1 | yes | no | low | optional (found by `mirth-lab diag-check`) | | 22 | debug assertion `!type_outlives.has_non_rigid_aliases()` in region outlives | 5 | some | sibling of closed #160206 | low | optional: debug builds only, but an invariant broken | | 23 | integer overflow in `ty/instance.rs:421` (`recursion/issue-83150.rs`) | 1 | yes | no | low | optional | diff --git a/docs/solver.md b/docs/solver.md index b28ca82..7bfb49b 100644 --- a/docs/solver.md +++ b/docs/solver.md @@ -5,11 +5,11 @@ everywhere by default (`CFG_DEFAULT_NEXT_SOLVER_GLOBALLY`), while compiletest pa `-Znext-solver=coherence` to every UI test, so the suite checks the old solver only. Code that nightly users compile is therefore not what the suite checks. -`rustc/ui-solver-diff.py` compiles every UI test both ways with stock nightly-2026-10-06, the way +`rustc/ui-mirth-lab solver-diff` compiles every UI test both ways with stock nightly-2026-10-06, the way its `//@` headers say (tests needing auxiliary crates or another target, and tests that set `-Znext-solver` themselves, are left out): - rustc/ui-solver-diff.py --rustc --tests /tests/ui --out + rustc/ui-mirth-lab solver-diff --rustc --tests /tests/ui --out 17,716 tests compiled both ways; **300 differ**: 271 fail either way with a different first error, **26 compile under the pinned solver and fail under nightly's default**, and **3 crash under the diff --git a/docs/untracked-reads.md b/docs/untracked-reads.md index b39f44b..7957c19 100644 --- a/docs/untracked-reads.md +++ b/docs/untracked-reads.md @@ -153,6 +153,6 @@ ten crates. ## In the runs -`rustc/fuzz.py` and `rustc/replay.py` set `RUSTC_REPORT_UNTRACKED` with the reuse check and +`mirth-lab fuzz` and `mirth-lab replay` set `RUSTC_REPORT_UNTRACKED` with the reuse check and keep every distinct line in `/untracked.txt`, so a code path that only a real crate reaches adds a line. diff --git a/docs/ur-queries.md b/docs/ur-queries.md index ee511a5..66f96b1 100644 --- a/docs/ur-queries.md +++ b/docs/ur-queries.md @@ -129,7 +129,7 @@ every site was triaged; what was: **Three untracked options that change reused output.** `optionRead`, joined with the options marked `[UNTRACKED]`, gave 74 such options read outside the session and the driver. Most only affect linking, which runs every session, or debugging output. To test -the rest without judging each by hand, [`rustc/audit-options.py`](../rustc/audit-options.py) +the rest without judging each by hand, [`mirth-lab audit-options`](../crates/mirth-lab/src/tools/audit_options.rs) builds a crate incrementally without an option, then with it, and compares with a clean build that has it, as a comment on rust-lang/rust#84232 ("Audit all UNTRACKED options", open since 2021) suggests. Of 50 boolean options, three change what an incremental session diff --git a/fixtures/audit/lib.rs b/fixtures/audit/lib.rs index 0934e34..9761aa1 100644 --- a/fixtures/audit/lib.rs +++ b/fixtures/audit/lib.rs @@ -1,4 +1,4 @@ -//! A crate for rustc/audit-options.py: generic code, code of its own for codegen +//! A crate for mirth-lab audit-options: generic code, code of its own for codegen //! options to change, a closure, a static, and warnings for diagnostic options. use std::collections::BTreeMap; diff --git a/fixtures/sink/core/src/grammar.rs b/fixtures/sink/core/src/grammar.rs index d15ed14..269a27b 100644 --- a/fixtures/sink/core/src/grammar.rs +++ b/fixtures/sink/core/src/grammar.rs @@ -1,5 +1,5 @@ //! Stable syntax the rest of the sink does not use, found by measuring it against Ur's Rust -//! grammar (`rustc/grammar-coverage.py`): one function per construct or small group. +//! grammar (`mirth-lab grammar-coverage`): one function per construct or small group. use core::cmp::Ordering; use core::fmt::Debug; diff --git a/fixtures/sink/nightly/src/lib.rs b/fixtures/sink/nightly/src/lib.rs index 220145f..da2ff72 100644 --- a/fixtures/sink/nightly/src/lib.rs +++ b/fixtures/sink/nightly/src/lib.rs @@ -1,5 +1,5 @@ //! Nightly-only syntax: the parts of Ur's Rust grammar no stable construct reaches -//! (`rustc/grammar-coverage.py`). One module per feature, each with a function `main` checks. +//! (`mirth-lab grammar-coverage`). One module per feature, each with a function `main` checks. #![feature( auto_traits, builtin_syntax, diff --git a/rustc/abi-diff.py b/rustc/abi-diff.py deleted file mode 100644 index b49508d..0000000 --- a/rustc/abi-diff.py +++ /dev/null @@ -1,468 +0,0 @@ -#!/usr/bin/env python3 -"""ABI differential: rustc's `extern "C"` must lower a signature the way clang lowers the same C -signature, on every target both support. - -Generates random C signatures (bool, integers of each width, float, double, pointers, __int128 -on 64-bit targets, and repr(C) structs, unions and arrays of them, nested, packed or -over-aligned), writes each as a C function (compiled by clang for the target's LLVM triple) and -as a Rust `#[no_mangle] extern "C" fn` (compiled by rustc for the target against minicore, so no -sysroot is needed), and compares the two LLVM IR signatures, parameter by parameter after -first-class aggregates are flattened (LLVM assigns their elements to registers one by one): - - finding a parameter or the return value differs in its register class (integer, floating - point, vector, memory), in an extension attribute (zeroext, signext), in inreg, - byval or sret, in the alignment of a byval or sret pointer, in the number of - parameters, or in the calling convention - note the same register classes with different IR types (`{double, double}` against - `float, double`), or a `noundef` difference - - rustc/abi-diff.py --rustc --rust --work [--targets t1,t2 | --all] - [--count 200] [--seed 1] [--jobs 8] [--clang clang] - -Known bugs are labelled, not reported: rust-lang/rust#163911 (x86_64 bool returns) and findings -19 and 20 in docs/hunt.md (RISC-V and LoongArch); so are two differences this host cannot decide -(i686 MSVC small-struct returns, a PowerPC64 `inreg` float). - -Writes //{a.c,a.rs,c.ll,r.ll} and /results.json (per target: functions -compared, findings, notes), and prints the findings grouped by kind. -""" - -import argparse -import ast -import json -import random -import re -import subprocess -from collections import defaultdict -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path -import os - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--rust", required=True) -p.add_argument("--work", required=True) -p.add_argument("--targets") -p.add_argument("--all", action="store_true", help="every target rustc knows; the default is MAIN, the tier 1 and " - "2 targets whose differences have been triaged (the others still show representation differences)") -p.add_argument("--count", type=int, default=200) -p.add_argument("--seed", type=int, default=1) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--clang", default="clang") -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -ENV = dict(os.environ, RUSTC_BOOTSTRAP="1") - -SCALARS = [ # (C, Rust) - ("_Bool", "bool"), ("signed char", "i8"), ("unsigned char", "u8"), ("short", "i16"), - ("unsigned short", "u16"), ("int", "i32"), ("unsigned int", "u32"), ("long long", "i64"), - ("unsigned long long", "u64"), ("float", "f32"), ("double", "f64"), ("void*", "*mut u8"), -] -WIDE = [("__int128", "i128"), ("unsigned __int128", "u128")] - - -class Gen: - def __init__(self, rng, wide): - self.rng, self.wide, self.structs, self.names = rng, wide, [], 0 - - def scalar(self): - pool = SCALARS + (WIDE if self.wide else []) - return self.rng.choice(pool) - - def field_type(self, depth): - r = self.rng.random() - if depth < 2 and r < 0.15: - return self.aggregate(depth + 1) - if r < 0.25: - c, rs = self.scalar() - n = self.rng.choice([1, 2, 3, 4, 8]) - return ("array", c, rs, n) - return self.scalar() - - def aggregate(self, depth=0): - name = f"S{self.names}" - self.names += 1 - union = self.rng.random() < 0.12 - packed = not union and self.rng.random() < 0.08 - # Rust rejects packed with align, and a packed type containing an over-aligned one: - # packed structs get scalars and arrays only. - align = None if packed else self.rng.choice([None] * 9 + [16, 32]) - fields = [self.field_type(2 if packed else depth) for _ in range(self.rng.randint(1, 5))] - self.structs.append((name, union, packed, align, fields)) - return (name, name) - - def ty(self): - return self.aggregate() if self.rng.random() < 0.4 else self.scalar() - - -def c_field(i, f): - if f[0] == "array": - return f"{f[1]} f{i}[{f[3]}];" - c = f[0] - return f"{c} f{i};" - - -def r_field(i, f): - if f[0] == "array": - return f"pub f{i}: [{f[2]}; {f[3]}]," - return f"pub f{i}: {f[1]}," - - -def c_ty(t): - return t[0] - - -def program(target_wide, seed): - rng = random.Random(seed) - g = Gen(rng, target_wide) - fns = [] - for k in range(args.count): - params = [g.ty() for _ in range(rng.randint(0, 8))] - ret = None if rng.random() < 0.15 else g.ty() - fns.append((f"f{k}", params, ret)) - c = ["#include "] - rs = ["#![feature(no_core)]", "#![no_core]", '#![crate_type = "lib"]', - "#![allow(improper_ctypes_definitions, unused, non_snake_case)]", - "extern crate minicore;", "use minicore::*;"] - for name, union, packed, align, fields in g.structs: - kw = "union" if union else "struct" - attrs = (" __attribute__((packed))" if packed else "") + (f" __attribute__((aligned({align})))" if align else "") - c.append(f"typedef {kw} {name} {{ {' '.join(c_field(i, f) for i, f in enumerate(fields))} }}{attrs} {name};") - repr_ = "C" + (", packed" if packed else "") + (f", align({align})" if align else "") - rs.append(f"#[repr({repr_})] pub {kw} {name} {{ {' '.join(r_field(i, f) for i, f in enumerate(fields))} }}") - # Union fields must be Copy; minicore has no derive. - rs.append(f"impl Copy for {name} {{}}") - for name, params, ret in fns: - cp = ", ".join(f"{c_ty(t)} a{i}" for i, t in enumerate(params)) or "void" - c.append(f"{c_ty(ret) if ret else 'void'} {name}({cp}) {{ for (;;); }}") - rp = ", ".join(f"a{i}: {t[1]}" for i, t in enumerate(params)) - rr = f" -> {ret[1]}" if ret else "" - rs.append(f"#[no_mangle] pub extern \"C\" fn {name}({rp}){rr} {{ loop {{}} }}") - return "\n".join(c) + "\n", "\n".join(rs) + "\n", fns - - -DEFINE = re.compile(r"^define\s+(.*?)@(\w+)\((.*)\)(.*)\{\s*$") -DROP = re.compile(r"\b(noundef|nonnull|noalias|nocapture|readonly|readnone|writeonly|writable|dead_on_unwind|" - r"captures\([^)]*\)|dereferenceable(_or_null)?\(\d+\)|initializes\([^)]*\)|range\([^)]*\)|" - r"nofpclass\([^)]*\)|immarg|returned|local_unnamed_addr|unnamed_addr|dso_local|" - r"dso_preemptable|hidden|protected|internal|private|nounwind|noinline|optnone|" - r"!\w+ !\d+|#\d+)\b") - - -def split_top(s): - out, depth, cur = [], 0, "" - for ch in s: - if ch in "({[<": - depth += 1 - elif ch in ")}]>": - depth -= 1 - if ch == "," and depth == 0: - out.append(cur.strip()) - cur = "" - else: - cur += ch - if cur.strip(): - out.append(cur.strip()) - return out - - -def param_type(p): - """The IR type of a parameter declaration, its attributes, and whether it has noundef.""" - p = re.sub(r"\s+%[\w.]+$", "", p.strip()) - noundef = "noundef" in p - m = re.match(r"^(\{[^}]*\}|\[[^\]]*\]|<[^>]*>|[\w.%*]+)(.*)$", p) - ty, attrs = (m.group(1), m.group(2)) if m else (p, "") - attrs = DROP.sub("", attrs) - kept = [] - for a in re.findall(r"(zeroext|signext|inreg|byval\([^)]*\)|sret\([^)]*\)|byref\([^)]*\)|align \d+|inalloca\([^)]*\))", attrs): - kept.append(re.sub(r"\(.*\)$", "", a) if a.startswith(("byval", "sret", "byref", "inalloca")) else a) - # An `align` on a plain pointer is a hint, not ABI; on byval and sret it is ABI. - if not any(k.startswith(("byval", "sret", "byref")) for k in kept): - kept = [k for k in kept if not k.startswith("align")] - return ty, tuple(sorted(kept)), noundef - - -def flatten(ty, named): - """A type's register-assignable parts; `named` maps the module's %struct names to bodies.""" - ty = ty.strip() - if ty in named: - ty = named[ty] - if ty.startswith("<{") and ty.endswith("}>"): # packed struct - ty = ty[1:-1] - if ty.startswith("{") and ty.endswith("}"): - return [x for part in split_top(ty[1:-1]) for x in flatten(part.strip(), named)] - m = re.fullmatch(r"\[1 x (.*)\]", ty) - if m: # a one-element array goes where its element goes - return flatten(m.group(1), named) - return [ty] - - -def units(types, arch=None, ret=False, width=64): - """Register classes, one per register-sized unit: an integer wider than a register takes - several. For a return value, an array is its elements (returned in consecutive registers).""" - out = [] - for t in types: - m = re.fullmatch(r"\[(\d+) x (i\d+|float|double)\]", t) - mi = re.fullmatch(r"i(\d+)", t) - if mi and int(mi.group(1)) > width: - out += ["int"] * (int(mi.group(1)) // width) - # A float array is a homogeneous aggregate: in a return, or an ARM VFP argument, it - # takes consecutive floating-point registers like a struct of its elements. - elif m and (ret or (arch == "arm" and m.group(2) in ("float", "double"))): - out += units([m.group(2)] * int(m.group(1)), arch, ret, width) - elif t == "agg": - out.append("agg") - else: - k = klass(t) - # x86's SSE registers hold floats, doubles and vectors alike. - if arch in ("x86_64", "x86") and k in ("fp", "vec"): - k = "sse" - out.append(k) - return out - - -def klass(ty): - if ty in ("float", "double", "half", "bfloat", "fp128", "x86_fp80", "ppc_fp128") or ty.startswith("<"): - return "fp" if not ty.startswith("<") else "vec" - if ty.startswith("["): - m = re.match(r"\[(\d+) x (.*)\]", ty) - return f"[{m.group(1)} x {klass(m.group(2))}]" if m else ty - if ty == "void": - return "void" - return "int" - - -def signature(line, named): - m = DEFINE.match(line) - if not m: - return None - head, name, params, tail = m.groups() - cc = re.findall(r"\b(\w+cc|cc \d+)\b", head) - head = DROP.sub("", head).strip() - tm = re.search(r"(\{[^{}]*\}|<\{[^{}]*\}>|\[[^\]]*\]|<[^>]*>|[\w.%*]+)\s*$", head) - ret_ty = tm.group(1) if tm else "void" - ret_attrs = tuple(sorted(a for a in head[:tm.start() if tm else 0].split() if a in ("zeroext", "signext", "inreg"))) - ps = [] - for p in split_top(params): - ty, attrs, noundef = param_type(p) - for t in flatten(ty, named): - ps.append((t, attrs, noundef)) - rets = flatten(ret_ty, named) - return {"name": name, "cc": tuple(cc), "ret": rets, "ret_attrs": ret_attrs, "params": ps, - "ret_noundef": "noundef" in m.group(1)} - - -def signatures(ll): - named = {m.group(1): m.group(2).strip() for m in re.finditer(r"^(%[\w.]+) = type (.*)$", ll, re.M)} - return {s["name"]: s for s in (signature(l, named) for l in ll.splitlines() if l.startswith("define")) if s} - - -def compare(c, r, arch, slot): - """`arch`: the target's arch; `slot`: its stack slot size in bytes (the pointer width).""" - findings, notes = [], [] - if c["cc"] != r["cc"]: - findings.append(f"calling convention: clang {c['cc']} rustc {r['cc']}") - if c["ret_attrs"] != r["ret_attrs"]: - what = f"return attributes: clang {c['ret_attrs']} rustc {r['ret_attrs']}" - # x86's psABI leaves the bits above a small integer return undefined (rustc stopped - # extending them in #142389); only bool's bits 1-7 must be zero (#163911). - if arch in ("x86_64", "x86") and c["ret"] != ["i1"]: - notes.append(what + " (x86 psABI: upper bits undefined)") - else: - findings.append(what) - if units(c["ret"], arch, True, slot * 8) != units(r["ret"], arch, True, slot * 8): - findings.append(f"return: clang {c['ret']} rustc {r['ret']}") - elif c["ret"] != r["ret"]: - notes.append(f"return types: clang {c['ret']} rustc {r['ret']}") - cp, rp = c["params"], r["params"] - # ARM and 64-bit PowerPC split a byval aggregate between registers and the stack as they do - # an array argument of the same size: both are "an aggregate". - if arch in ("arm", "powerpc64"): - agg = lambda t, a: ("agg", (), False) if ("byval" in a or re.match(r"\[\d+ x i\d+\]", t)) else (t, a, None) - cp = [agg(t, a) if agg(t, a)[0] == "agg" else (t, a, n) for t, a, n in cp] - rp = [agg(t, a) if agg(t, a)[0] == "agg" else (t, a, n) for t, a, n in rp] - if units([t for t, _, _ in cp], arch, False, slot * 8) != units([t for t, _, _ in rp], arch, False, slot * 8): - what = f"parameters: clang {[t for t, _, _ in cp]} rustc {[t for t, _, _ in rp]}" - # i386 passes every argument on the stack: a struct expanded into its scalars and a - # byval copy of it are the same bytes. - if arch == "x86" and any("byval" in a for _, a, _ in cp + rp): - notes.append(what + " (x86: same stack bytes)") - else: - findings.append(what) - elif [t for t, _, _ in cp] != [t for t, _, _ in rp]: - notes.append(f"parameter types: clang {[t for t, _, _ in cp]} rustc {[t for t, _, _ in rp]}") - elif True: - for i, ((ct, ca, cn), (rt, ra, rn)) in enumerate(zip(cp, rp)): - if ca != ra: - what = f"parameter {i} attributes: clang {ca} rustc {ra} ({ct})" - aligns = [int(a.split()[1]) for a in ca + ra if a.startswith("align ")] - rest_c = [a for a in ca if not a.startswith("align ")] - rest_r = [a for a in ra if not a.startswith("align ")] - ext_only_rust = not rest_c and set(rest_r) <= {"zeroext", "signext"} - if rest_c == rest_r and aligns and max(aligns) <= slot: - # A byval copy goes in stack slots at least `slot` bytes aligned either way. - notes.append(what + " (both within a stack slot)") - elif arch in ("wasm32", "wasm64") and {*rest_c, *rest_r} <= {"byval"}: - # WebAssembly lowers byval to a pointer to a copy the caller makes. - notes.append(what + " (wasm: byval is a pointer to a copy)") - elif arch == "x86" and {*rest_c, *rest_r} <= {"byval"}: - # i386 passes everything on the stack: a byval copy and a struct expanded - # into its scalars occupy the same bytes. - notes.append(what + " (x86: same stack bytes)") - elif arch == "x86_64" and ext_only_rust: - # Win64: rustc extends small integers, clang does not; LLVM does not rely on - # it in the callee (checked: both re-extend), so nothing observable. - notes.append(what + " (win64: extension not relied on)") - else: - findings.append(what) - if ct != rt: - notes.append(f"parameter {i}: clang {ct} rustc {rt}") - if cn != rn: - notes.append(f"parameter {i} noundef: clang {cn} rustc {rn}") - return findings, notes - - -MAIN = ("x86_64-unknown-linux-gnu x86_64-unknown-linux-musl x86_64-pc-windows-msvc x86_64-pc-windows-gnu " - "x86_64-apple-darwin i686-unknown-linux-gnu i686-pc-windows-msvc aarch64-unknown-linux-gnu " - "aarch64-apple-darwin aarch64-pc-windows-msvc aarch64-unknown-linux-musl armv7-unknown-linux-gnueabihf " - "arm-unknown-linux-gnueabi thumbv7em-none-eabihf riscv64gc-unknown-linux-gnu riscv32imac-unknown-none-elf " - "loongarch64-unknown-linux-gnu powerpc64le-unknown-linux-gnu s390x-unknown-linux-gnu wasm32-unknown-unknown " - "wasm32-wasip1").split() - - -def known(arch, finding): - """A finding that is a bug already recorded: the label, or None.""" - if arch == "x86_64" and finding.startswith("return attributes: clang ('zeroext',)"): - return "rust-lang/rust#163911" - if arch in ("riscv64", "riscv32", "loongarch64"): - if re.match(r"parameter \d+ attributes: clang \('(signext|zeroext)',\) rustc \(\)", finding): - return "finding 19 (docs/hunt.md)" - if finding.startswith(("parameters:", "return:")): - c, _, r = finding.partition(" rustc ") - fp = lambda text: len(re.findall(r"'(float|double)'", text)) - if fp(r) > fp(c): - return "finding 20 (docs/hunt.md)" - return None - - -def unresolved(arch, target, finding): - """A difference whose correct side needs a reference this host lacks: the label, or None.""" - lists = re.fullmatch(r"parameters: clang (\[.*?\]) rustc (\[.*\])", finding) - sret_only = bool(lists) and (lambda c, r: c[:1] == ["ptr"] and c[1:] == r)(*map(ast.literal_eval, lists.groups())) - if target.endswith("windows-msvc") and arch == "x86" and ( - (finding.startswith("return:") and "'void'" in finding) or sret_only): - # clang returns an 8-byte struct with an array field of 3 bytes indirectly (its - # register-size rule recurses into fields); rustc and MSVC's documentation return any - # 8-byte struct in edx:eax. Needs MSVC to decide. - return "i686 msvc small-struct return (needs MSVC)" - if arch == "powerpc64" and re.match(r"parameter \d+ attributes: clang \('inreg',\) rustc \(\) \((float|double)\)", finding): - # clang marks a float from a single-member aggregate inreg; whether the PowerPC backend - # then places it differently needs a run on the target. - return "ppc64 inreg float (needs a run)" - return None - - -def run(argv, cwd): - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=600, cwd=cwd, env=ENV) - return r.returncode, r.stderr - except subprocess.TimeoutExpired: - return -1, "timeout" - - -def spec(target): - out = subprocess.run([args.rustc, "--print", "target-spec-json", "-Zunstable-options", "--target", target], - capture_output=True, text=True, env=ENV) - return json.loads(out.stdout) if out.returncode == 0 else None - - -def one_target(target): - s = spec(target) - if not s: - return target, {"skip": "no target spec"} - width = int(s.get("target-pointer-width", 64)) - if width < 32 or s.get("c-int-width", "32") != "32": - return target, {"skip": "16-bit int"} - d = WORK / target - d.mkdir(parents=True, exist_ok=True) - csrc, rsrc, fns = program(width == 64 and s.get("arch") not in ("sparc64",), args.seed) - (d / "a.c").write_text(csrc) - (d / "a.rs").write_text(rsrc) - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", - "--out-dir", str(d)] - code, err = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "-Awarnings", - str(Path(args.rust) / "tests/auxiliary/minicore.rs")], d) - if code: - return target, {"skip": "minicore does not build", "err": err[-500:]} - code, err = run(base + ["--emit=llvm-ir", "-Copt-level=0", "--extern", f"minicore={d}/libminicore.rlib", - "-o", str(d / "r.ll"), str(d / "a.rs")], d) - if code: - return target, {"skip": "rust side does not build", "err": err[-1500:]} - cflags = [] - # The target's CPU and features decide parts of the ABI in clang too (soft-float, SSE). - if s.get("cpu") and s["cpu"] != "generic": - cflags += ["-Xclang", "-target-cpu", "-Xclang", s["cpu"]] - for feature in filter(None, s.get("features", "").split(",")): - cflags += ["-Xclang", "-target-feature", "-Xclang", feature] - if s.get("llvm-abiname"): - cflags.append(f"-mabi={s['llvm-abiname']}") - if s.get("llvm-floatabi") == "hard": - cflags.append("-mfloat-abi=hard") - code, err = run([args.clang, f"--target={s['llvm-target']}", "-ffreestanding", "-S", "-emit-llvm", "-O0", - "-Wno-everything", *cflags, "-o", str(d / "c.ll"), str(d / "a.c")], d) - if code: - return target, {"skip": "clang does not build", "err": err[-800:]} - cs, rs = signatures((d / "c.ll").read_text()), signatures((d / "r.ll").read_text()) - findings, notes, known_hits = {}, {}, {} - for name, _, _ in fns: - if name not in cs or name not in rs: - continue - f, n = compare(cs[name], rs[name], s.get("arch"), width // 8) - labelled = [(x, known(s.get("arch"), x) or unresolved(s.get("arch"), target, x)) for x in f] - f = [x for x, k in labelled if not k] - for x, k in labelled: - if k: - known_hits.setdefault(k, 0) - known_hits[k] += 1 - if f: - findings[name] = f - if n: - notes[name] = n - return target, {"compared": len(fns), "findings": findings, "notes": notes, "known": known_hits} - - -def main(): - if args.targets: - targets = args.targets.split(",") - elif not args.all: - targets = MAIN - else: - targets = subprocess.run([args.rustc, "--print", "target-list"], capture_output=True, text=True, - env=ENV).stdout.split() - with ThreadPoolExecutor(args.jobs) as ex: - results = dict(ex.map(one_target, targets)) - (WORK / "results.json").write_text(json.dumps(results, indent=1)) - kinds = defaultdict(lambda: defaultdict(int)) - skipped = defaultdict(list) - for t, r in results.items(): - if "skip" in r: - skipped[r["skip"]].append(t) - continue - for fs in r["findings"].values(): - for f in fs: - kinds[re.sub(r"\(.*|:.*", "", f).strip()][t] += 1 - compared = [t for t, r in results.items() if "skip" not in r] - knowns = defaultdict(int) - for r in results.values(): - for k, n in r.get("known", {}).items(): - knowns[k] += n - if knowns: - print("known: " + ", ".join(f"{k} {n}" for k, n in sorted(knowns.items()))) - print(f"{len(compared)} targets compared; skipped: " + ", ".join(f"{k} {len(v)}" for k, v in skipped.items())) - for kind, per in sorted(kinds.items(), key=lambda kv: -sum(kv[1].values())): - print(f"{kind}: {sum(per.values())} in {len(per)} targets: " - + ", ".join(f"{t} {n}" for t, n in sorted(per.items(), key=lambda kv: -kv[1])[:8])) - - -main() diff --git a/rustc/artifacts.py b/rustc/artifacts.py deleted file mode 100644 index aa0ae18..0000000 --- a/rustc/artifacts.py +++ /dev/null @@ -1,108 +0,0 @@ -"""What a Cargo build produced, in a form two builds can be compared in. - -Shared by fuzz.py and replay.py. From the JSON messages of `cargo build ---message-format=json-render-diagnostics` it collects, for packages built -from a path: - - rmeta every .rmeta, by path - rlib every .rlib's members by name, with the incremental session suffix - of codegen-unit object names removed (`...rcgu.o`), - since it differs between sessions while the objects are identical - exe every executable - diag every diagnostic, rendered, counted per crate -""" - -import hashlib -import json -import re -from collections import Counter -from pathlib import Path - -SESSION = re.compile(rb"\.[0-9a-z]{7}(\.rcgu\.(?:o|dwo))") - - -def ar_members(data): - """The members of a Unix ar archive (GNU format), as {name: bytes}.""" - if not data.startswith(b"!\n"): - return {"": data} - members, names, pos = {}, b"", 8 - while pos + 60 <= len(data): - header = data[pos:pos + 60] - name = header[:16].rstrip() - size = int(header[48:58].strip() or 0) - body = data[pos + 60:pos + 60 + size] - pos += 60 + size + (size & 1) - if name == b"//": - names = body - continue - if name in (b"/", b"/SYM64/"): - continue - if name.startswith(b"/") and name[1:].isdigit(): - offset = int(name[1:]) - name = names[offset:names.index(b"/\n", offset)] - name = name.rstrip(b"/") - members[SESSION.sub(rb"\1", name).decode("utf-8", "replace")] = body - return members - - -def normalized_rlib(path): - out = {} - for name, body in ar_members(Path(path).read_bytes()).items(): - # The link metadata and, with split debuginfo, the objects name session-suffixed files. - body = SESSION.sub(rb"\1", body) - out[name] = hashlib.sha256(body).hexdigest() - return out - - -def collect(stdout, target): - """Artifacts from cargo's JSON messages, keyed by path relative to `target`.""" - target = Path(target) - found = {"rmeta": {}, "rlib": {}, "exe": {}, "diag": Counter()} - for line in stdout.splitlines(): - try: - msg = json.loads(line) - except ValueError: - continue - if "path+file" not in msg.get("package_id", ""): - continue - if msg.get("reason") == "compiler-message": - m = msg.get("message", {}) - text = m.get("rendered") or m.get("message") or "" - found["diag"][(msg["target"]["name"], text)] += 1 - continue - if msg.get("reason") != "compiler-artifact": - continue - for f in msg.get("filenames", []): - rel = str(Path(f).relative_to(target)) if f.startswith(str(target)) else f - if f.endswith(".rmeta"): - found["rmeta"][rel] = hashlib.sha256(Path(f).read_bytes()).hexdigest() - elif f.endswith(".rlib"): - found["rlib"][rel] = normalized_rlib(f) - if msg.get("executable"): - f = msg["executable"] - rel = str(Path(f).relative_to(target)) if f.startswith(str(target)) else f - found["exe"][rel] = hashlib.sha256(SESSION.sub(rb"\1", Path(f).read_bytes())).hexdigest() - return found - - -def compare(a, b): - """{kind: [what differs]} for the kinds that differ between two collections.""" - out = {} - for kind in ("rmeta", "exe"): - diff = sorted(k for k in set(a[kind]) | set(b[kind]) if a[kind].get(k) != b[kind].get(k)) - if diff: - out[kind] = diff - diff = [] - for rel in sorted(set(a["rlib"]) | set(b["rlib"])): - ma, mb = a["rlib"].get(rel, {}), b["rlib"].get(rel, {}) - members = sorted(m for m in set(ma) | set(mb) if ma.get(m) != mb.get(m)) - if members: - diff.append(f"{rel}: {', '.join(members[:5])}{' …' if len(members) > 5 else ''}") - if diff: - out["rlib"] = diff - if a["diag"] != b["diag"]: - only_a = a["diag"] - b["diag"] - only_b = b["diag"] - a["diag"] - out["diag"] = [f"{c} only in the first: {t[:200]!r}" for (c, t), n in only_a.items()] + \ - [f"{c} only in the second: {t[:200]!r}" for (c, t), n in only_b.items()] - return out diff --git a/rustc/audit-options.py b/rustc/audit-options.py deleted file mode 100755 index 18d7dd0..0000000 --- a/rustc/audit-options.py +++ /dev/null @@ -1,119 +0,0 @@ -#!/usr/bin/env python3 -"""Audit rustc's [UNTRACKED] options for stale incremental reuse. - -An option rustc marks [UNTRACKED] is left out of the dependency-tracking hash, -so changing it between incremental sessions reuses the previous session's -results. That is only correct if the option cannot change them. For each -untracked option that takes no value or a boolean, this builds a crate -incrementally without it, then again with it, and compares the result with a -clean build that has it: the .rmeta, each .rlib member (object code, with the -incremental session suffix removed from names), the diagnostics, and the files -written. A difference means the option changes output that incremental -compilation reuses: it should be tracked, or the reuse checked. - - rustc/audit-options.py --rustc --source --crate [--edition 2018] [ARG…] - -With ARGs, audits those arguments instead of the options found in the -checkout's compiler/rustc_session/src/options.rs. The crate should have code of -its own for codegen options to change (non-generic functions) and a warning or -two for diagnostic options to change. -""" - -import argparse -import os -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).resolve().parent)) -import artifacts # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--source", help="a rust checkout, to read the untracked options from") -p.add_argument("--crate", required=True, help="the crate root to build, as a library") -p.add_argument("--crate-name", default="audited") -p.add_argument("--edition", default="2021") -p.add_argument("args", nargs="*") -args = p.parse_args() - -# Options that stop compilation or change how arguments are read. -SKIP = {"-Chelp", "-Zhelp", "-Zno-analysis", "-Zparse-crate-root-only=yes", "-Zshell-argfiles=yes"} - - -def untracked_boolean_options(source): - text = (Path(source) / "compiler/rustc_session/src/options.rs").read_text() - codegen = re.search(r"options! \{\s*CodegenOptions,", text).start() - unstable = re.search(r"options! \{\s*UnstableOptions,", text).start() - found = [] - for m in re.finditer(r"^\s{4}(\w+):\s*([^=\n]+?)\s*=\s*\(([^,]*),\s*(parse_\w+),\s*\[UNTRACKED\]", text, re.M): - name, _ty, default, parser = m.groups() - if parser not in ("parse_bool", "parse_no_value", "parse_opt_bool"): - continue - if m.start() < codegen: - continue - group = "Z" if m.start() > unstable else "C" - flag = f"-{group}{name.replace('_', '-')}" - if parser != "parse_no_value": - flag += "=no" if default.strip() in ("true", "Some(true)") else "=yes" - if flag not in SKIP: - found.append(flag) - return found - - -crate = Path(args.crate).resolve() -name = args.crate_name - - -def build(work, incremental, out, extra): - """One build, always from the same working directory, which rustc records.""" - (work / out).mkdir(exist_ok=True) - r = subprocess.run([args.rustc, "--edition", args.edition, "--crate-type", "lib", "--crate-name", name, - "--emit=metadata,link", f"-Cincremental={work / incremental}", "--out-dir", str(work / out), - str(crate)] + extra, capture_output=True, text=True, cwd=work) - diagnostics = sorted(l for l in r.stderr.splitlines() if l.startswith(("warning", "error")) or "-->" in l) - return r.returncode, diagnostics - - -def audit(flag): - work = Path(tempfile.mkdtemp(prefix="audit-")) - try: - extra = [flag] if flag else [] - rc0, _ = build(work, "i", "o1", []) - rc1, diag_inc = build(work, "i", "o1", extra) - rc2, diag_clean = build(work, "j", "o2", extra) - if rc0 or rc2: - return [f"the crate does not build (exit {rc0} without the option, {rc2} with it)"] - problems = [] - if rc1 != rc2: - problems.append(f"the incremental rebuild exits {rc1}, a clean build {rc2}") - if (work / f"o1/lib{name}.rmeta").read_bytes() != (work / f"o2/lib{name}.rmeta").read_bytes(): - problems.append("metadata") - a = artifacts.normalized_rlib(work / f"o1/lib{name}.rlib") - b = artifacts.normalized_rlib(work / f"o2/lib{name}.rlib") - members = [m for m in set(a) | set(b) if a.get(m) != b.get(m)] - if members: - problems.append(f"object code ({len(members)} rlib members differ or exist on one side)") - if diag_inc != diag_clean: - problems.append(f"diagnostics ({len(diag_inc)} lines incrementally, {len(diag_clean)} clean)") - missing = set(os.listdir(work / "o2")) - set(os.listdir(work / "o1")) - if missing: - problems.append(f"{len(missing)} files only a clean build writes") - return problems - finally: - shutil.rmtree(work) - - -flags = args.args or untracked_boolean_options(args.source) -results = {} -control = audit("") -print(f"{'(control: no option)':40} {'; '.join(control) or 'same'}", flush=True) -if control: - sys.exit("the control differs: incremental and clean builds disagree without any option") -for flag in flags: - results[flag] = audit(flag) - print(f"{flag:40} {'; '.join(results[flag]) or 'same'}", flush=True) -sys.exit(1 if any(results.values()) else 0) diff --git a/rustc/callgraph.py b/rustc/callgraph.py deleted file mode 100644 index b658ed6..0000000 --- a/rustc/callgraph.py +++ /dev/null @@ -1,398 +0,0 @@ -#!/usr/bin/env python3 -"""Which of the compiler's functions can run at all: reachability over the call graph that a -compiler built with rustc/callgraph.toml writes (`.graph`), against the functions a -compiler built with rustc/coverage.toml instruments (`cover` sites). The functions that cannot -be reached are taken out of coverage's denominator. - - rustc/callgraph.py --graph /mirth-sites --sites /mirth-sites - [--hit ...] [--external ...] [--logs ...] - [--json out.json] [--unreachable ] [--why ] [--gaps ] - -The graph over-approximates what can run, so what it leaves out cannot run (as far as the -edges it knows go): - -- edges: direct calls, functions and closures used as values, callees MIR inlining merged in, - and trait calls resolved in the caller's context; -- a call to a trait item reaches every body implementing it, and the trait's own default body; - but (rapid type analysis) a method of an impl for one of the compiler's structs or enums only - once reachable code builds that type (an aggregate, a constructor, a constant of it), and a - function of a trait impl only once reachable code demands the trait for the impl's type (a - call whose bounds say so, an impl selected for one, a cast to `dyn Trait`): see - docs/coverage.md for the exceptions; -- a body nested in another (a closure, an inline const) is reached with it, and one nested in - something that is not a body (a static's or a constant's initializer) is a root; -- roots: the compiler's and rustdoc's `main`s, every function with a foreign ABI (callbacks from C, C++ and - LLVM), every body implementing a trait from outside the compiler (for one of the compiler's - types: once the type is built), - which the standard library may call (`Iterator::next`, `Drop::drop`, `Debug::fmt`, ...), and - every constant's and static's initializer (tables of function pointers, callbacks). - -With --hit or --logs, reports coverage of the reachable functions, and checks the analysis: a -function that ran must be reachable; any that are not are listed (an edge kind it misses). -""" - -import argparse -import json -import re -from collections import defaultdict, deque -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--graph", required=True) -p.add_argument("--sites", required=True) -p.add_argument("--hit", action="append", default=[]) -p.add_argument("--logs", action="append", default=[]) -p.add_argument("--external", action="append", default=[], - help="union.txt of programs outside the compiler that link it (ui-fulldeps): what they " - "ran counts, and is a root, since their own mains call it") -p.add_argument("--json") -p.add_argument("--unreachable", action="append", default=[]) -p.add_argument("--why", action="append", default=[], help="print how a function is reached") -p.add_argument("--gaps", help="write the reachable functions that never ran, by crate and file, to this file") -p.add_argument("--block-gaps", help="write the blocks that never ran in functions that did, by crate and file, " - "to this file (a compiler built with `[coverage] blocks`)") -args = p.parse_args() - -ROOTS = {"rustc_main::main", "rustc_driver_impl::main", "rustdoc::main"} -# Crates that do not run when the compiler does: proc macros (run while it is built) and the -# Windows resource helper of its build script. -NOT_AT_RUN_TIME = {"rustc_macros", "rustc_type_ir_macros", "rustc_index_macros", "rustc_windows_rc", - "rustc_hir_macros", "rustc_fluent_macro"} - -# Nodes are DefPathHashes; bodies also have their path (as coverage sites name them). -path_of, hash_of = {}, {} -implements, external_impl, const_bodies = {}, set(), set() -SCOPE = ("rustc_", "rustdoc") -self_type = {} # a method in an impl for one of the compiler's structs or enums -> that type -constructs = defaultdict(set) # body -> types it builds -spec_bounds = set() # traits a specializing impl's bounds name -impl_trait = {} -ending = {} # a body no path of which returns -> "ice-only" (it panics) or "diverges" -diverges_into = defaultdict(set) # such a body -> the other functions ending its paths -impl_key = {} # a function of a trait impl -> (trait, the struct or enum the impl is for) -demands = defaultdict(set) # body -> (trait, type) pairs it needs implemented -# Called by the language on any value, not through a bound: drop glue. -UNGATED_TRAITS = {"core::ops::drop::Drop"} -edges = defaultdict(set) -for f in Path(args.graph).glob("*.graph"): - for line in f.read_text(errors="replace").splitlines(): - parts = line.split("\t") - if parts[0] == "body": - node, path, item, item_path, kind = parts[1], parts[2], parts[3], parts[4], parts[5] - if len(parts) > 7 and parts[7].startswith(SCOPE): - self_type[node] = parts[6] - # A trait with specializing impls: which impl a call reaches is decided by more than - # the bounds say, so such impls are not gated on them. - specialized = len(parts) > 11 and parts[11] == "specialized" - if (len(parts) > 10 and parts[8] != "-" and parts[10] != "-" and parts[9] not in UNGATED_TRAITS - and not specialized): - impl_key[node] = (parts[8], parts[10]) - - path_of[node] = path - hash_of[path] = node - if len(parts) > 12 and parts[12] in ("ice-only", "diverges"): - ending[node] = parts[12] - if kind in ("const", "extern"): - const_bodies.add(node) - if item != "-": - implements[node] = item - if not item_path.startswith("rustc_"): - external_impl.add(node) - elif parts[0] == "diverges": - diverges_into[parts[1]].add(parts[2]) - elif parts[0] == "specbound": - spec_bounds.add(parts[1]) - elif parts[0] == "demand": - demands[parts[1]].add((parts[2], parts[3])) - elif parts[0] == "edge": - if parts[3] == "construct": - constructs[parts[1]].add(parts[2]) - else: - edges[parts[1]].add(parts[2]) -bodies = set(path_of) -# A body runs only on a compiler bug when every path ends in a panic, directly or through bodies -# that do (`default_extern_query` and the closures that call it). -ice_nodes = {n for n, e in ending.items() if e == "ice-only"} -changed = True -while changed: - changed = False - for node, e in ending.items(): - if node not in ice_nodes and diverges_into[node] and diverges_into[node] <= ice_nodes: - ice_nodes.add(node) - changed = True -ice_only = {path_of[n] for n in ice_nodes} -impl_key = {node: key for node, key in impl_key.items() if key[0] not in spec_bounds} - -implementors = defaultdict(set) -for body, item in implements.items(): - implementors[item].add(body) - - -def parent(path): - m = re.match(r"^(.*)::\{[^}]*\}$", path) - return m.group(1) if m else None - - -children = defaultdict(set) -roots = {hash_of[r] for r in ROOTS if r in hash_of} | external_impl | const_bodies -for node, path in path_of.items(): - up = parent(path) - if up is None: - continue - # Nested in a body, or in a static's or a constant's initializer, which is not one here. - while up is not None and up not in hash_of and parent(up) is not None: - up = parent(up) - if up in hash_of: - children[hash_of[up]].add(node) - else: - roots.add(node) - -functions = {} # site -> (crate, path) -span_of = {} -blocks = {} # site -> (crate, function path, block, span, snippet, only panics) -logging = set() # blocks all of whose code is a logging macro's: they run only with RUSTC_LOG -for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) < 7: - continue - # Newer tables name the crate with its stable id (`rustc_hash-<16 hex digits>`). - krate = re.sub(r"-[0-9a-f]{16}$", "", f[3]) - if f[1] == "cover": - functions[f[0]] = (krate, f[4]) - span_of[f[4]] = f[6] - elif f[1] == "block": - name, *tags = f[5].split(" ") - if "log" in tags: - logging.add(f[0]) - blocks[f[0]] = (krate, f[4], name, f[6], f[7] if len(f) > 7 else "", "panics" in tags) -external = set() -for u in args.external: - external |= set(Path(u).read_text().split()) - -# Crates from crates.io that the scope takes in by name (rustc-hash, rustc-stable-hash): other -# dependencies, which the graph does not cover, can name their types and need their impls -# without any bound in the compiler saying so. Their impls are not gated on demand. -from_registry = {krate for _, (krate, path) in functions.items() - if not span_of.get(path, "compiler/").startswith(("compiler/", "src/", "library/"))} -impl_key = {node: key for node, key in impl_key.items() - if path_of[node].split("::", 1)[0] not in from_registry} - -# What programs outside the compiler ran of it: their mains call it, so it is a root. -external_roots = {hash_of[functions[x][1]] for x in external if x in functions and functions[x][1] in hash_of} -roots |= external_roots - -# Rapid type analysis: a method of an impl for one of the compiler's types counts for trait -# dispatch (and as an external trait's root) only once reachable code builds the type; and a -# function of a trait impl only once reachable code needs that type to implement that trait (a -# call whose bounds say so, a cast to `dyn Trait`, the trait's item used with that `Self`). A -# call the caller's types resolve to the implementation reaches it directly. -reachable, live, demanded = set(), set(), set() -came_from = {} # node -> (from node, how) -condition_from = {} # condition -> the body that met it -waiting = defaultdict(set) # condition -> functions waiting for it -queue = deque() - - -def missing(node): - out = [] - t = self_type.get(node) - if t is not None and t not in live: - out.append(("live", t)) - k = impl_key.get(node) - if k is not None and k not in demanded: - out.append(("demand", k)) - return out - - -def offer(node, gated, source=None, how="root"): - if node in reachable: - return - came_from.setdefault(node, (source, how)) - lacking = missing(node) if gated else [] - if lacking: - waiting[lacking[0]].add(node) - else: - queue.append(node) - - -def satisfied(condition): - for node in waiting.pop(condition, ()): - offer(node, True) - - -for r in roots: - offer(r, r in external_impl and r not in const_bodies and r not in external_roots) -while queue: - node = queue.popleft() - if node in reachable: - continue - reachable.add(node) - for t in constructs.get(node, ()): - if t not in live: - live.add(t) - condition_from[("live", t)] = node - satisfied(("live", t)) - for k in demands.get(node, ()): - if k not in demanded: - demanded.add(k) - condition_from[("demand", k)] = node - satisfied(("demand", k)) - for nxt in edges.get(node, set()) | children.get(node, set()): - offer(nxt, False, node, "edge") - # A call to a trait item reaches the bodies implementing it. - for impl in implementors.get(nxt, ()): - offer(impl, True, node, "dispatch") -reachable_paths = {path_of[n] for n in reachable if n in path_of} -for target in args.why: - node = hash_of.get(target) - print(f"\nwhy {target}:" + ("" if node in reachable else " not reachable")) - seen = set() - while node is not None and node in reachable and node not in seen: - seen.add(node) - source, how = came_from.get(node, (None, "?")) - extra = "" - if True: - for c in [("live", self_type.get(node)), ("demand", impl_key.get(node))]: - if c[1] is not None and c in condition_from: - extra += f" [{c[0]} from {path_of.get(condition_from[c], condition_from[c])}]" - print(f" {path_of.get(node, node)} <- {how}{extra}") - node = source - -functions = {s: v for s, v in functions.items() if v[0] not in NOT_AT_RUN_TIME} -paths = {path for _, path in functions.values()} -known = paths & set(hash_of) -unreach = {path for path in known if path not in reachable_paths} -print(f"{len(bodies)} bodies in the graph, {len(roots)} roots, {len(reachable & bodies)} reachable") -print(f"{len(functions)} instrumented functions; {len(known)} in the graph; " - f"{len(unreach)} unreachable ({100 * len(unreach) / max(len(known), 1):.1f}%)") - -hit = set(external) -for u in args.hit: - hit |= set(Path(u).read_text().split()) -for d in args.logs: - for log in Path(d).rglob("*.log"): - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - hit.add(line[2:]) -hit_paths = {functions[s][1] for s in hit if s in functions} -by_crate = defaultdict(lambda: [0, 0, 0]) -for site, (krate, path) in functions.items(): - row = by_crate[krate] - if path in unreach: - continue - row[0] += 1 - row[1] += path in hit_paths -if hit: - total = sum(r[0] for r in by_crate.values()) - ran = sum(r[1] for r in by_crate.values()) - print(f"coverage: {len(hit_paths)} functions ran; of the {total} reachable ones, {ran} " - f"({100 * ran / max(total, 1):.1f}%)") - wrong = sorted(hit_paths & unreach) - ice = {path for _, (k, path) in functions.items() if path in ice_only and path not in unreach} - ice_ran = ice & hit_paths - print(f"of the reachable ones, {len(ice)} only panic (they run on a compiler bug; {len(ice_ran)} ran): " - f"without them, {ran - len(ice_ran)} of {total - len(ice)} " - f"({100 * (ran - len(ice_ran)) / max(total - len(ice), 1):.1f}%)") - print(f"ran although unreachable (edges the analysis misses): {len(wrong)}") - for w in wrong[:30]: - print(" ", w) - if blocks: - # Blocks: each function's entry (its own site) and its other blocks, in the functions - # that can run. A block only panics when every path from it does, or its function does. - rows = defaultdict(lambda: [0, 0, 0, 0]) # crate -> reachable, ran, panic-only, panic-only ran - for site, (krate, path) in functions.items(): - if path not in unreach: - row = rows[krate] - row[0] += 1 - row[1] += path in hit_paths - row[2] += path in ice_only - row[3] += path in ice_only and path in hit_paths - logged = [0, 0] - for site, (krate, path, _, _, _, panics) in blocks.items(): - if krate in NOT_AT_RUN_TIME or path in unreach or path not in paths: - continue - if site in logging: - logged[0] += 1 - logged[1] += site in hit - continue - row = rows[krate] - row[0] += 1 - row[1] += site in hit - panics = panics or path in ice_only - row[2] += panics - row[3] += panics and site in hit - n, r, pn, pr = (sum(row[i] for row in rows.values()) for i in range(4)) - print(f"blocks: of the {n} in reachable functions, {r} ran ({100 * r / max(n, 1):.1f}%); " - f"{pn} only panic ({pr} ran): without them, {r - pr} of {n - pn} " - f"({100 * (r - pr) / max(n - pn, 1):.1f}%); not counted: {logged[0]} blocks of logging " - f"macros, which run only with RUSTC_LOG ({logged[1]} ran)") - print(f"{'crate':40} {'ran':>7} {'blocks':>9} {'%':>6} (panic-only blocks aside)") - for krate, (n, r, pn, pr) in sorted(rows.items(), key=lambda kv: (kv[1][1] - kv[1][3]) / max(kv[1][0] - kv[1][2], 1)): - if n - pn: - print(f"{krate:40} {r - pr:7} {n - pn:9} {100 * (r - pr) / (n - pn):6.1f}") - print(f"{'crate':40} {'ran':>7} {'reachable':>9} {'%':>6}") - for krate, (n, r, _) in sorted(by_crate.items(), key=lambda kv: kv[1][1] / max(kv[1][0], 1)): - if n: - print(f"{krate:40} {r:7} {n:9} {100 * r / n:6.1f}") -for crate in args.unreachable: - print(f"\nunreachable in {crate}:") - for path in sorted(unreach): - if path.startswith(crate + "::"): - print(" ", path) -if args.json: - Path(args.json).write_text(json.dumps({"unreachable": sorted(unreach), - "ice_only": sorted(ice_only & paths), - "ran_unreachable": sorted(hit_paths & unreach), - "reachable_not_hit": sorted((paths - unreach) - hit_paths)}, indent=0)) - -if args.gaps: - files = defaultdict(list) - for site, (krate, path) in functions.items(): - if path not in unreach and path not in hit_paths: - span = span_of.get(path, "?") - files[(krate, span.rsplit(":", 2)[0])].append((span, path)) - with open(args.gaps, "w") as out: - out.write(f"# Reachable functions that never ran: {sum(map(len, files.values()))}\n\n") - by_crate = defaultdict(int) - for (krate, _), fs in files.items(): - by_crate[krate] += len(fs) - for krate in sorted(by_crate, key=lambda k: -by_crate[k]): - out.write(f"## {krate} ({by_crate[krate]})\n\n") - for (k, file), fs in sorted(files.items(), key=lambda kv: -len(kv[1])): - if k != krate: - continue - out.write(f"### {file} ({len(fs)})\n\n") - for span, path in sorted(fs): - tag = " (only panics)" if path in ice_only else "" - out.write(f"- `{path}` {span.rsplit(':', 2)[-2] if ':' in span else ''}{tag}\n") - out.write("\n") - print(f"gaps written to {args.gaps}") - -if args.block_gaps and blocks: - # The blocks that never ran, in functions that did (gaps.md has the functions that did not), - # by crate and file, the lines with the most first. - files = defaultdict(list) - for site, (krate, path, name, span, snippet, panics) in blocks.items(): - if path in hit_paths and site not in hit and krate not in NOT_AT_RUN_TIME and site not in logging: - files[(krate, span.rsplit(":", 2)[0])].append((span, path, name, snippet, panics or path in ice_only)) - with open(args.block_gaps, "w") as out: - out.write(f"# Blocks that never ran in functions that did: {sum(map(len, files.values()))}\n\n") - by_crate = defaultdict(int) - for (krate, _), bs in files.items(): - by_crate[krate] += len(bs) - for krate in sorted(by_crate, key=lambda k: -by_crate[k]): - out.write(f"## {krate} ({by_crate[krate]})\n\n") - for (k, file), bs in sorted(files.items(), key=lambda kv: -len(kv[1])): - if k != krate: - continue - out.write(f"### {file} ({len(bs)})\n\n") - def line_of(b): - parts = b[0].rsplit(":", 2) - return (int(parts[1]), int(parts[2])) if len(parts) == 3 and parts[1].isdigit() else (0, 0) - for b in sorted(bs, key=line_of): - span, path, name, snippet, panics = b - tag = " (only panics)" if panics else "" - out.write(f"- {line_of(b)[0]} `{path}` {name}{tag}: `{snippet[:100]}`\n") - out.write("\n") - print(f"block gaps written to {args.block_gaps}") diff --git a/rustc/callgraph.toml b/rustc/callgraph.toml index ec5dd3d..7b6f2d9 100644 --- a/rustc/callgraph.toml +++ b/rustc/callgraph.toml @@ -1,6 +1,6 @@ # The compiler's call graph: every body in the compiler's own crates (and rustdoc, a second # entry point into them: build with WITH_RUSTDOC=1) writes its calls and -# references to .graph (rustc/callgraph.py reads them, with rustc/coverage.toml's sites). +# references to .graph (`mirth-lab callgraph` reads them, with rustc/coverage.toml's sites). [scope] crates = ["rustc_*", "rustdoc"] diff --git a/rustc/coverage-compact.py b/rustc/coverage-compact.py deleted file mode 100644 index 3d808fb..0000000 --- a/rustc/coverage-compact.py +++ /dev/null @@ -1,71 +0,0 @@ -#!/usr/bin/env python3 -"""Fold coverage logs (MIRTH_OUT, from a compiler built with rustc/coverage.toml) into a running -union as they are finished, and delete them: a test suite run starts tens of thousands of rustc -processes, whose logs together would not fit on the disk. - - rustc/coverage-compact.py --logs --out [--until ] - -A log is finished when its last line is the `X` line written at exit, or when it has not changed -for ten minutes (a process that crashed). For each, /added.jsonl gets the process's source -file argument and the sites it reached that no earlier process did; /union.txt holds every -site reached so far, rewritten every pass. Runs until exists, then does a last pass. -""" - -import argparse -import json -import os -import re -import time -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--logs", required=True) -p.add_argument("--out", required=True) -p.add_argument("--until", default="") -args = p.parse_args() - -LOGS, OUT = Path(args.logs), Path(args.out) -OUT.mkdir(parents=True, exist_ok=True) -union_path = OUT / "union.txt" -union = set(union_path.read_text().split()) if union_path.exists() else set() - - -def source_of(header): - for field in header.split("\t")[3:]: - if field.endswith(".rs"): - return field - m = re.search(r"--crate-name\t(\S+)", header) - return m.group(1) if m else "" - - -def one_pass(final): - done = 0 - with (OUT / "added.jsonl").open("a") as added: - for log in list(LOGS.glob("*.log")): - try: - text = log.read_text(errors="replace") - age = time.time() - log.stat().st_mtime - except FileNotFoundError: - continue - lines = text.splitlines() - if not lines or not (lines[-1].startswith("X\t") or age > 600 or (final and age > 5)): - continue - sites = {l[2:] for l in lines if l.startswith("V\t")} - new = sites - union - union.update(new) - if new: - added.write(json.dumps({"source": source_of(lines[0]), "new": sorted(new)}) + "\n") - log.unlink() - done += 1 - union_path.write_text("\n".join(sorted(union)) + "\n") - return done - - -while True: - finishing = bool(args.until) and os.path.exists(args.until) - n = one_pass(finishing) - print(f"{time.strftime('%H:%M:%S')} {n} logs folded, {len(union)} sites", flush=True) - if finishing: - one_pass(True) - break - time.sleep(30) diff --git a/rustc/coverage-flags.py b/rustc/coverage-flags.py deleted file mode 100644 index 4787555..0000000 --- a/rustc/coverage-flags.py +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env python3 -"""Coverage of the compiler across option configurations: build a fixture with a -coverage-instrumented rustc (rustc/coverage.toml) once per row of a PICT transitions table, -clean with the A options, then rebuilt after one random edit with the B options, each row's -rustc processes logging to /row/. Read the result with rustc/coverage.py. - - rustc/coverage-flags.py --rustc --fixture fixtures/sink - --flags --table rows.tsv --out [--rows 0:40] [--workers 6] -""" - -import argparse -import csv -import importlib.util -import json -import os -import random -import shutil -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -here = Path(__file__).parent -sys.path.insert(0, str(here)) -import mutations # noqa: E402 - -spec = importlib.util.spec_from_file_location("flag_model", here / "flag-model.py") -flag_model = importlib.util.module_from_spec(spec) -spec.loader.exec_module(flag_model) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--flags", required=True) -p.add_argument("--table", required=True) -p.add_argument("--out", required=True) -p.add_argument("--rows", default="") -p.add_argument("--workers", type=int, default=6) -p.add_argument("--toolchain", default="nightly-2026-10-06") -args = p.parse_args() - -OUT = Path(args.out).resolve() -opts = {flag_model.pname(o["flag"] + o["name"]): o for o in json.load(open(Path(args.flags) / "options.json"))} -rows = list(csv.DictReader(open(args.table), delimiter="\t")) -idx = list(range(len(rows))) -if args.rows: - lo, hi = (int(x) if x else None for x in args.rows.split(":")) - idx = idx[lo:hi] - - -def flags(row, side): - out = [flag_model.FLAG_BASE] - for k, v in row.items(): - if k.startswith(side + "_") and v != "absent": - o = opts[k[2:]] - f = o["flag"] + o["name"] - out.append(f if v == "present" else f + "=" + v.replace(";", ",")) - return out - - -def run(i): - logs = OUT / f"row{i}" - if logs.exists(): - return i, "done before" - work = OUT / f"work{i}" - shutil.rmtree(work, ignore_errors=True) - src = work / "s" - shutil.copytree(args.fixture, src, ignore=shutil.ignore_patterns("target")) - results = [] - for side in "AB": - if side == "B": - rng = random.Random(i) - paths = sorted(p for p in src.rglob("*.rs") if "target" not in p.parts) - for _ in range(20): - path = rng.choice(paths) - fn = rng.choices([e for e, _ in mutations.EDITS], weights=[w for _, w in mutations.EDITS])[0] - if path.name == "build.rs": - continue - new = fn(path.read_text(), rng, 0) - if new is not None: - path.write_text(new) - break - e = dict(os.environ, RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", - RUSTFLAGS=" ".join(flags(rows[i], side)), MIRTH_OUT=str(work / "logs")) - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", - "--target", "x86_64-unknown-linux-gnu", "--target-dir", str(work / "t")], - cwd=src, env=e, capture_output=True, text=True, timeout=1800) - results.append("ok" if r.returncode == 0 else "failed") - (work / "logs").mkdir(exist_ok=True) - (work / "logs").rename(logs) - shutil.rmtree(work, ignore_errors=True) - return i, " ".join(results) - - -OUT.mkdir(parents=True, exist_ok=True) -with ThreadPoolExecutor(args.workers) as ex: - for i, result in ex.map(run, idx): - print(f"row {i}: {result}", flush=True) diff --git a/rustc/coverage-fulldeps.sh b/rustc/coverage-fulldeps.sh index 53fd8f1..9bfdb0e 100755 --- a/rustc/coverage-fulldeps.sh +++ b/rustc/coverage-fulldeps.sh @@ -11,7 +11,9 @@ host=x86_64-unknown-linux-gnu out=$1/ui-fulldeps-stage1 mkdir -p "$out/logs" "$out/bin" rm -f "$out/done" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! pass=0; fail=0 for t in $(grep -l '^//@ ignore-stage1' -r "$MIRTH_RUST/tests/ui-fulldeps" --include=*.rs | grep -v /auxiliary/ | sort); do diff --git a/rustc/coverage-generators.py b/rustc/coverage-generators.py deleted file mode 100644 index c389026..0000000 --- a/rustc/coverage-generators.py +++ /dev/null @@ -1,264 +0,0 @@ -#!/usr/bin/env python3 -"""Compiler runs that rustc's test suites hardly make, for coverage (rustc/callgraph.py --gaps -lists what is left): run them with the coverage-instrumented compiler and MIRTH_OUT set, and fold -the logs with rustc/coverage-compact.py. - - MIRTH_OUT= rustc/coverage-generators.py --rustc --rust - --list picked.json --work [--jobs 8] [--only prints,targets,dumps] - -- prints: every `--print` request, on the host and on every target; -- targets: `tests/auxiliary/minicore.rs` and a file of functions with every kind of argument - and return value, compiled to an object for every target (each target's ABI, layout and - codegen code); -- links: for every target, a `no_main` binary, a cdylib, a staticlib and a dylib on minicore, - linked with `-Clinker=true` (the linker command each target's linker flavor builds, without - the linker), with linker options; -- dumps: each test of the list (rustc/ui-coverage.py pick) with each debugging and printing - option (`-Zunpretty=`, `-Zdump-mir`, `-Zprint-type-sizes`, statistics, profiling, ...). -""" - -import argparse -import json -import os -import re -import subprocess -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--rust", required=True) -p.add_argument("--list", required=True) -p.add_argument("--work", required=True) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--only", default="prints,targets,dumps,links") -args = p.parse_args() -WORK = Path(args.work) -WORK.mkdir(parents=True, exist_ok=True) -ENV = dict(os.environ, RUSTC_BOOTSTRAP="1") - -PRINTS = ["all-target-specs-json", "backend-has-mnemonic", "backend-has-zstd", "calling-conventions", - "cfg", "check-cfg", "code-models", "crate-name", "crate-root-lint-levels", "deployment-target", - "file-names", "host-tuple", "link-args", "native-static-libs", "relocation-models", - "split-debuginfo", "stack-protector-strategies", "supported-crate-types", "sysroot", - "target-cpus", "target-features", "target-libdir", "target-list", "target-spec-json", - "target-spec-json-schema", "tls-models", "wasm-proc-macro-tuple"] -PER_TARGET = ["cfg", "target-spec-json", "target-cpus", "target-features", "calling-conventions", - "code-models", "relocation-models", "tls-models", "stack-protector-strategies", - "split-debuginfo", "supported-crate-types", "deployment-target", "check-cfg"] - -# Every kind of argument and return value, for each target's calling convention. -ABI = r""" -#![feature(no_core, lang_items, rustc_attrs, c_variadic, f16, f128)] -#![no_core] -#![crate_type = "lib"] -#![allow(improper_ctypes_definitions, unused)] -extern crate minicore; -use minicore::*; - -#[repr(C)] pub struct Small { a: u8, b: u16 } -#[repr(C)] pub struct Pair { a: u64, b: u64 } -#[repr(C)] pub struct Big { a: [u64; 8] } -#[repr(C)] pub struct Floats { a: f32, b: f64 } -#[repr(C)] pub struct Mixed { a: f32, b: u32 } -#[repr(C)] pub union U { a: u32, b: f32 } -#[repr(C)] pub struct Hfa { a: f32, b: f32, c: f32, d: f32 } -#[repr(C)] pub struct Empty {} -#[repr(transparent)] pub struct T(u64); -#[repr(C, packed)] pub struct Packed { a: u8, b: u32 } -#[repr(C, align(16))] pub struct Aligned { a: u8 } - -#[no_mangle] pub extern "C" fn c_small(x: Small) -> Small { x } -#[no_mangle] pub extern "C" fn c_pair(x: Pair) -> Pair { x } -#[no_mangle] pub extern "C" fn c_big(x: Big) -> Big { x } -#[no_mangle] pub extern "C" fn c_floats(x: Floats, y: f32, z: f64) -> Floats { x } -#[no_mangle] pub extern "C" fn c_mixed(x: Mixed) -> Mixed { x } -#[no_mangle] pub extern "C" fn c_union(x: U) -> U { x } -#[no_mangle] pub extern "C" fn c_hfa(x: Hfa) -> Hfa { x } -#[no_mangle] pub extern "C" fn c_empty(x: Empty) -> Empty { x } -#[no_mangle] pub extern "C" fn c_transparent(x: T) -> T { x } -#[no_mangle] pub extern "C" fn c_packed(x: Packed) -> Packed { x } -#[no_mangle] pub extern "C" fn c_aligned(x: Aligned) -> Aligned { x } -#[no_mangle] pub extern "C" fn c_ints(a: i8, b: u16, c: i32, d: u64, e: i128, f: u128, g: bool, h: char) -> i128 { e } -#[no_mangle] pub extern "C" fn c_ptrs(a: *const u8, b: &u32, c: &mut [u8; 3], f: extern "C" fn()) -> *const u8 { a } -#[no_mangle] pub extern "C" fn c_many(a: u64, b: u64, c: u64, d: u64, e: u64, f: u64, g: u64, h: u64, i: u64, j: Pair, k: f64, l: f64, m: f64, n: f64, o: f64, p: f64, q: f64, r: f64, s: f64) -> u64 { a } -#[no_mangle] pub unsafe extern "C" fn c_variadic(a: u32, mut args: ...) -> u32 { a } -pub fn rust_all(a: Small, b: Pair, c: Big, d: Floats, e: (u8, u64), f: [u32; 5], g: &[u8], h: &str, i: u128) -> Big { c } -pub fn rust_f16(a: f16, b: f128) -> f128 { b } -#[no_mangle] pub extern "C" fn c_f16(a: f16, b: f128) -> f128 { b } -#[no_mangle] pub extern "system" fn system(a: Pair) -> Pair { a } -#[no_mangle] pub extern "C-unwind" fn c_unwind(a: Pair) -> Pair { a } -pub static TABLE: [extern "C" fn(Pair) -> Pair; 2] = [c_pair, c_unwind_shim]; -extern "C" fn c_unwind_shim(a: Pair) -> Pair { a } -extern "C" { fn imported(a: Big, b: Floats) -> Hfa; } -pub unsafe fn call_imported(a: Big, b: Floats) -> Hfa { imported(a, b) } -""" - - -def run(argv, cwd=None, timeout=300): - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=timeout, cwd=cwd, env=ENV) - return r.returncode, r.stderr - except subprocess.TimeoutExpired: - return -1, "timeout" - - -def targets(): - out = subprocess.run([args.rustc, "--print", "target-list"], capture_output=True, text=True, env=ENV) - return out.stdout.split() - - -def prints(): - jobs = [[args.rustc, "--print", kind, "-Zunstable-options", "-"] for kind in PRINTS] - for target in targets(): - for kind in PER_TARGET: - jobs.append([args.rustc, "--print", kind, "--target", target, "-Zunstable-options", "-"]) - with tempfile.TemporaryDirectory(dir=WORK) as d: - empty = Path(d) / "lib.rs" - empty.write_text("") - jobs = [[a if a != "-" else str(empty) for a in j] for j in jobs] - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(lambda j: run(j, cwd=d), jobs)) - print(f"prints: {len(jobs)} runs, {sum(c == 0 for c, _ in done)} succeeded", flush=True) - - -def one_target(target): - with tempfile.TemporaryDirectory(dir=WORK) as d: - minicore = Path(args.rust) / "tests/auxiliary/minicore.rs" - (Path(d) / "abi.rs").write_text(ABI) - results = [] - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", - "-Cpanic=abort", "--out-dir", d] - code, err = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "-Copt-level=1", - "--emit=link,obj", str(minicore)], cwd=d) - results.append(code) - if code == 0: - for opt in ("0", "3"): - code, err = run(base + ["--emit=obj,asm,llvm-ir", f"-Copt-level={opt}", "-Cdebuginfo=2", - "--extern", f"minicore={d}/libminicore.rlib", "abi.rs"], cwd=d) - results.append(code) - return target, results - - -def cross(): - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(one_target, targets())) - built = sum(all(c == 0 for c in r) and len(r) == 3 for _, r in done) - print(f"targets: {len(done)} targets, {built} built minicore and the ABI file", flush=True) - (WORK / "targets.json").write_text(json.dumps(dict(done), indent=0)) - - -LINKED = r""" -#![feature(no_core, lang_items)] -#![no_core] -#![no_main] -extern crate minicore; -#[no_mangle] pub extern "C" fn exported(a: u32) -> u32 { a } -#[no_mangle] pub static DATA: u32 = 7; -#[link(name = "c")] extern "C" { fn puts(p: *const u8) -> i32; } -#[link(name = "m", kind = "static")] extern "C" {} -#[link(name = "framework_like", kind = "dylib", modifiers = "+verbatim")] extern "C" {} -""" -LINK_OPTIONS = [[], ["-Cprefer-dynamic", "-Crelocation-model=pic"], ["-Cstrip=symbols", "-Clink-dead-code"], - ["-Clink-self-contained=yes"], ["-Cdebuginfo=2", "-Csplit-debuginfo=packed"], - ["-Clink-arg=-Wl,--foo", "-Clink-args=-x -y", "-Zpre-link-args=-z"], - ["-Cdefault-linker-libraries", "-Zlink-native-libraries=no"], ["-Ccontrol-flow-guard"], - ["-Zstaticlib-allow-rdylib-deps"], ["-Copt-level=s", "-Clto=fat"], ["-Ccode-model=large"]] - - -def one_link(target): - with tempfile.TemporaryDirectory(dir=WORK) as d: - minicore = Path(args.rust) / "tests/auxiliary/minicore.rs" - (Path(d) / "linked.rs").write_text(LINKED) - base = [args.rustc, "--target", target, "-Zunstable-options", "--edition", "2021", "-Cpanic=abort", - "--out-dir", d, "-Clinker=true"] - code, _ = run(base + ["--crate-type", "rlib", "--crate-name", "minicore", "--emit=link", - str(minicore)], cwd=d) - ok = 0 - if code == 0: - for options in LINK_OPTIONS: - for kind in ("bin", "cdylib", "staticlib", "dylib"): - c, _ = run(base + ["--crate-type", kind, "--extern", f"minicore={d}/libminicore.rlib", - "-Csave-temps", *options, "linked.rs"], cwd=d) - ok += c == 0 - return ok - - -def links(): - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(one_link, targets())) - print(f"links: {len(done)} targets, {sum(done)} links succeeded", flush=True) - - -DUMPS = [[f"-Zunpretty={m}"] for m in ("normal", "expanded", "expanded,identified", "expanded,hygiene", - "ast-tree", "ast-tree,expanded", "hir", "hir,identified", - "hir,typed", "hir-tree", "thir-tree", "thir-flat", "mir", - "stable-mir", "mir-cfg")] + [ - ["-Zdump-mir=all", "-Zdump-mir-dataflow", "-Zdump-mir-graphviz", "-Zmir-include-spans=on"], - ["-Zprint-type-sizes"], ["-Zprint-mono-items=yes", "--emit=link"], ["-Zmeta-stats"], ["-Zhir-stats"], - ["-Zinput-stats"], ["-Zself-profile", "-Zself-profile-events=all"], ["-Ztime-passes"], - ["-Zquery-dep-graph", "-Zdump-dep-graph", "-Cincremental=inc"], ["-Zincremental-info", "-Cincremental=inc"], - ["-Zdump-mono-stats", "-Zdump-mono-stats-format=json", "--emit=link"], ["-Zprint-codegen-stats", "--emit=link"], - ["-Zvalidate-mir", "-Zlint-mir", "-Zmir-opt-level=4"], ["-Zverbose-internals", "-Zidentify-regions"], - ["-Ztrack-diagnostics", "-Zteach"], ["-Zthreads=4"], ["-Zpolonius=next"], ["-Zinline-mir", "-Zmir-opt-level=3"], - ["-Zrandomize-layout"], ["-Zwrite-long-types-to-disk=no", "-Zverbose-internals"], - ["-Zunleash-the-miri-inside-of-you"], ["-Zno-analysis"], ["-Zprofile-closures"], ["-Zui-testing"], - ["-Cinstrument-coverage", "--emit=link"], ["-Zemit-stack-sizes", "--emit=link"], - ["--error-format=json", "--json=diagnostic-rendered-ansi,artifacts,future-incompat,unused-externs"], - ["--error-format=human-annotate-rs"], ["--error-format=short"], ["-Zterminal-urls=yes", "--color=always"], - ["-Wunused", "-Wrust-2018-idioms", "-Wrust-2021-compatibility", "-Wrust-2024-compatibility", "-Wclippy::all"], - ["-Fwarnings", "--cap-lints=warn"], ["-Zcodegen-source-order", "--emit=link"], -] - - -def headers(text): - flags, edition, revision = [], None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - if revision: - flags += ["--cfg", revision] - return flags, edition - - -def dump(test): - path = Path(args.rust) / "tests/ui" / test - text = path.read_text(errors="replace") - flags, edition = headers(text) - ok = 0 - for extra in DUMPS: - with tempfile.TemporaryDirectory(dir=WORK) as d: - emit = [] if any(e.startswith("--emit") for e in extra) else ["--emit=metadata"] - code, _ = run([args.rustc, str(path), "--edition", edition or "2015", *emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", - *flags, *extra], cwd=d, timeout=120) - ok += code == 0 - return ok - - -def dumps(): - picked = json.loads(Path(args.list).read_text()) - tests = [t["test"] if isinstance(t, dict) else t for t in picked] - with ThreadPoolExecutor(args.jobs) as ex: - done = list(ex.map(dump, tests)) - print(f"dumps: {len(tests)} tests x {len(DUMPS)} options, {sum(done)} runs succeeded", flush=True) - - -only = args.only.split(",") -if "prints" in only: - prints() -if "targets" in only: - cross() -if "dumps" in only: - dumps() -if "links" in only: - links() diff --git a/rustc/coverage-report.sh b/rustc/coverage-report.sh index 879950a..8a4ead9 100755 --- a/rustc/coverage-report.sh +++ b/rustc/coverage-report.sh @@ -3,7 +3,7 @@ # and option-configuration logs, against the call graph's denominator. Writes # $WORK/coverage-report.txt, $WORK/gaps.md (what can run and did not, by crate and file) and # $WORK/gaps.json; with a compiler built with `[coverage] blocks`, also $WORK/gaps-blocks.md (the -# blocks that never ran in functions that did). Extra arguments go to callgraph.py (`--why `, `--unreachable `). +# blocks that never ran in functions that did). Extra arguments go to `mirth-lab callgraph` (`--why `, `--unreachable `). # # WORK=~/mirth-work rustc/coverage-report.sh [--why ...] # @@ -34,7 +34,8 @@ dirs=${COV_LOGS-$work/cov-sink $work/cov-flags} for d in $dirs; do [ -d "$d" ] && logs+=(--logs "$d") done -python3 "$here/callgraph.py" --graph "$work/build-cg/mirth-sites" --sites "$build/mirth-sites" \ +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" callgraph --graph "$work/build-cg/mirth-sites" --sites "$build/mirth-sites" \ "${runs[@]}" "${logs[@]}" --gaps "$report/gaps.md" --block-gaps "$report/gaps-blocks.md" --json "$report/gaps.json" "$@" \ > "$report/coverage-report.txt" sed -n '1,5p;/^blocks:/p' "$report/coverage-report.txt" diff --git a/rustc/coverage-run.sh b/rustc/coverage-run.sh index 77a09a5..088128d 100755 --- a/rustc/coverage-run.sh +++ b/rustc/coverage-run.sh @@ -1,12 +1,12 @@ #!/usr/bin/env bash # Run any command with MIRTH_OUT set, folding the coverage logs of the instrumented compiler's -# processes into $WORK/cov-suites//union.txt as they finish (rustc/coverage-compact.py), +# processes into $WORK/cov-suites//union.txt as they finish (`mirth-lab coverage-compact`), # where rustc/coverage-report.sh picks them up (COV_SUITES: another directory instead). # # WORK=~/mirth-work rustc/coverage-run.sh # # For example, the UI tests through incremental rebuilds: -# rustc/coverage-run.sh ui-fuzz python3 rustc/ui-fuzz.py --rustc $COV_RUSTC \ +# rustc/coverage-run.sh ui-fuzz target/release/mirth-lab ui-fuzz --rustc $COV_RUSTC \ # --tests $MIRTH_RUST/tests/ui --list ~/mirth-work/ui-cov/all-runnable.json \ # --work ~/mirth-work/ui-fuzz-cov --edits 3 --jobs 6 set -uo pipefail @@ -16,7 +16,9 @@ name=$1; shift out=${COV_SUITES:-$work/cov-suites}/$name mkdir -p "$out/logs" rm -f "$out/done" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! MIRTH_OUT=$out/logs "$@" > "$out/run.log" 2>&1 status=$? diff --git a/rustc/coverage-suites.sh b/rustc/coverage-suites.sh index 87e32bd..bef8686 100755 --- a/rustc/coverage-suites.sh +++ b/rustc/coverage-suites.sh @@ -2,12 +2,12 @@ # Run rustc's own test suites through compiletest with the coverage-instrumented compiler # (rustc/build.sh with MIRTH_WATCH=rustc/coverage.toml and BUILD_DIR), recording which of the # compiler's functions each rustc process reaches. Logs are folded as they finish -# (rustc/coverage-compact.py), so the disk holds only the union and what each test added. +# (`mirth-lab coverage-compact`), so the disk holds only the union and what each test added. # # MIRTH_RUST= BUILD_DIR= \ # rustc/coverage-suites.sh [-- ] # -# Writes //{union.txt,added.jsonl,x.log}. Read with rustc/coverage.py +# Writes //{union.txt,added.jsonl,x.log}. Read with `mirth-lab coverage` # --union /*/union.txt. set -uo pipefail : "${MIRTH_RUST:?set MIRTH_RUST}" @@ -27,7 +27,9 @@ rm -f "$out/done" export RUSTFLAGS_BOOTSTRAP="-L dependency=$BUILD_DIR/mirth-runtime" export RUSTFLAGS_NOT_BOOTSTRAP="$RUSTFLAGS_BOOTSTRAP" -python3 "$here/coverage-compact.py" --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & +# The compactor is a mirth-lab subcommand: build it first (a no-op when up to date). +(cd "$here/.." && cargo build --release -q --offline -p mirth-lab) || exit 1 +"$here/../target/release/mirth-lab" coverage-compact --logs "$out/logs" --out "$out" --until "$out/done" > "$out/compact.log" 2>&1 & compactor=$! # A test compile running more than five minutes is killed and named. diff --git a/rustc/coverage.py b/rustc/coverage.py deleted file mode 100644 index 161a67c..0000000 --- a/rustc/coverage.py +++ /dev/null @@ -1,75 +0,0 @@ -#!/usr/bin/env python3 -"""Which of the compiler's functions ran, from a compiler built with rustc/coverage.toml. - - rustc/coverage.py --sites /mirth-sites --logs [--logs ...] - [--json out.json] [--files] [--unhit ] - -The site tables list each instrumented function (`cover` sites: id, crate, path, span); every -rustc process run with MIRTH_OUT set writes, at exit, a `V ` line for each function it -entered. This reads both and prints, per crate, how many functions ran; with --files, per -source file; with --unhit, the functions that never ran, in the crates or files matching. - -Several --logs directories (several runs: fixtures, flags) are combined. -""" - -import argparse -import json -from collections import defaultdict -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--sites", required=True) -p.add_argument("--logs", action="append", required=True) -p.add_argument("--json") -p.add_argument("--files", action="store_true") -p.add_argument("--unhit", action="append", default=[]) -args = p.parse_args() - -functions = {} # site -> (crate, path, span) -for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) >= 7 and f[1] == "cover": - functions[f[0]] = (f[3], f[4], f[6]) - -hit = set() -processes = 0 -for directory in args.logs: - for log in Path(directory).glob("*.log"): - processes += 1 - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - hit.add(line[2:]) - -unknown = hit - functions.keys() -by_crate = defaultdict(lambda: [0, 0]) -by_file = defaultdict(lambda: [0, 0]) -for site, (krate, path, span) in functions.items(): - file = span.rsplit(":", 2)[0] - ran = site in hit - for table, key in ((by_crate, krate), (by_file, file)): - table[key][0] += 1 - table[key][1] += ran - -total, ran = len(functions), len(hit & functions.keys()) -print(f"{processes} processes; {ran} of {total} functions ran ({100 * ran / max(total, 1):.1f}%)" - + (f"; {len(unknown)} sites not in the tables" if unknown else "")) -print(f"{'crate':40} {'ran':>7} {'of':>7} {'%':>6}") -for krate, (n, r) in sorted(by_crate.items(), key=lambda kv: kv[1][1] / kv[1][0]): - print(f"{krate:40} {r:7} {n:7} {100 * r / n:6.1f}") -if args.files: - print() - print(f"{'file':80} {'ran':>6} {'of':>6}") - for file, (n, r) in sorted(by_file.items(), key=lambda kv: (kv[1][1] / kv[1][0], -kv[1][0])): - print(f"{file:80} {r:6} {n:6}") -for pattern in args.unhit: - print(f"\nnever ran, matching {pattern!r}:") - for site, (krate, path, span) in sorted(functions.items(), key=lambda kv: kv[1][2]): - if site not in hit and (pattern in krate or pattern in span): - print(f" {path} {span}") -if args.json: - Path(args.json).write_text(json.dumps({ - "processes": processes, "functions": total, "ran": ran, - "crates": {k: {"functions": n, "ran": r} for k, (n, r) in by_crate.items()}, - "files": {k: {"functions": n, "ran": r} for k, (n, r) in by_file.items()}, - "unhit": sorted(f"{functions[s][1]}\t{functions[s][2]}" for s in functions.keys() - hit)}, indent=1)) diff --git a/rustc/coverage.toml b/rustc/coverage.toml index 2540cb6..d747088 100644 --- a/rustc/coverage.toml +++ b/rustc/coverage.toml @@ -1,9 +1,9 @@ # Which functions of the compiler run: every function and closure in the compiler's own -# crates records its first call in each rustc process (rustc/coverage.py reads the result). +# crates records its first call in each rustc process (`mirth-lab coverage` reads the result). [scope] crates = ["rustc_*"] [coverage] functions = true -# Every basic block too (rustc/callgraph.py --block-gaps). +# Every basic block too (`mirth-lab callgraph --block-gaps`). blocks = true diff --git a/rustc/crash-diff.py b/rustc/crash-diff.py deleted file mode 100644 index c11f049..0000000 --- a/rustc/crash-diff.py +++ /dev/null @@ -1,93 +0,0 @@ -#!/usr/bin/env python3 -"""Internal checks on: what the compiler's own invariants say about every UI test. - -Compiles each standalone UI test with the compiler under test and again with a second compiler -built from the same source with debug assertions (`rust.debug-assertions`), with -`-Zvalidate-mir` added. Findings are a crash, a failed assertion or a MIR validation error -under the second that the first does not have: rustc's invariants failing where release -builds go on silently (19 of the last 1,000 ICE reports needed such a build). - - rustc/crash-diff.py --rustc --checked - --tests /tests/ui --work [--extra "-Zvalidate-mir"] [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--checked", required=True) -p.add_argument("--extra", default="-Zvalidate-mir") -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def message(stderr): - """The first line saying what went wrong inside the compiler.""" - for pattern in (r"panicked at [^\n]*\n[^\n]*", r"internal compiler error: [^\n]*", r"broken MIR[^\n]*"): - m = re.search(pattern, stderr) - if m: - return re.sub(r"/\S+/compiler/", "compiler/", m.group(0))[:400] - return "" - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - emit = "metadata" if kind in ("check-pass", "check-fail", None) else "link" - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - a, ea, _ = uitest.compile(args.rustc, path.resolve(), d / "release", flags, edition, timeout=300, emit=emit) - b, eb, _ = uitest.compile(args.checked, path.resolve(), d / "checked", flags, edition, args.extra.split(), - timeout=600, emit=emit) - record = {"test": rel, "release": a, "checked": b} - found = [] - if b == "ice" and a != "ice": - found.append({"what": "only with internal checks", "message": message(eb), "stderr": eb[-4000:]}) - elif b == "ice" and a == "ice" and message(ea) != message(eb): - record["note"] = "both crash, differently" - record["found"] = [f"{f['what']}: {f['message'][:160]}" for f in found] - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({"test": rel, "flags": flags, "edition": edition, - "extra": args.extra, "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/flag-campaign.sh b/rustc/flag-campaign.sh index 617d995..9143bb8 100755 --- a/rustc/flag-campaign.sh +++ b/rustc/flag-campaign.sh @@ -9,7 +9,7 @@ # /rustc at the patched toolchain, and rerun with --recheck: the rows with findings run # again first, then everything not yet walked. # -# Needs PICT (PICT=..., default ~/mirth-work/tools/pict/pict) and flag-universe.py's results +# Needs PICT (PICT=..., default ~/mirth-work/tools/pict/pict) and `mirth-lab flag-universe`'s results # (FLAGS=..., default ~/mirth-work/flags). set -u D=$(cd "$1" && pwd); shift @@ -20,7 +20,7 @@ FIXTURE=${FIXTURE:-$here/../fixtures/sink} cd "$D" model() { # name subset - [ -f "$1.txt" ] || python3 "$here/flag-model.py" "$FLAGS" "$2" "$1.txt" --transitions --cargo --allow-known > /dev/null + [ -f "$1.txt" ] || "$here/../target/release/mirth-lab" flag-model "$FLAGS" "$2" "$1.txt" --transitions --cargo --allow-known > /dev/null } table() { # model strength seed t="$1-t$2-r$3.tsv" @@ -30,7 +30,7 @@ table() { # model strength seed walk() { # table edits seed [flag-walk options] local tab=$1 edits=$2 seed=$3; shift 3 w="walk-${tab%.tsv}-e$edits" - python3 "$here/flag-walk.py" --rustc "$D/rustc/bin/rustc" --fixture "$FIXTURE" --flags "$FLAGS" \ + "$here/../target/release/mirth-lab" flag-walk --rustc "$D/rustc/bin/rustc" --fixture "$FIXTURE" --flags "$FLAGS" \ --table "$tab" --work "$w" --workers 10 --edits "$edits" --seed "$seed" --pause-on-finding "$@" \ > "$w.log" 2>&1 echo "$(date +%T) $w $(tail -1 "$w.log")" | tee -a campaign.log diff --git a/rustc/flag-fuzz.py b/rustc/flag-fuzz.py deleted file mode 100644 index 25d27a3..0000000 --- a/rustc/flag-fuzz.py +++ /dev/null @@ -1,69 +0,0 @@ -#!/usr/bin/env python3 -"""Run fuzz.py under option configurations: for each chosen row of a PICT table (the B side of -a `flag-model.py --transitions --cargo` model), fuzz the fixture with those options in -RUSTFLAGS for a number of edits. Stops at the first finding (fuzz.py --pause-on-finding); -rerunning resumes after the rows already done. - - rustc/flag-fuzz.py --rustc --fixture fixtures/sink --flags - --table rows.tsv --work [--rows 0:20] [--edits 200] [--workers 8] - -Writes /row/ (fuzz.py's work directory) and /rows.jsonl (one line per finished -row: options, fuzz.py's totals, findings). -""" - -import argparse -import csv -import importlib.util -import json -import subprocess -import sys -from pathlib import Path - -spec = importlib.util.spec_from_file_location("flag_model", Path(__file__).with_name("flag-model.py")) -flag_model = importlib.util.module_from_spec(spec) -spec.loader.exec_module(flag_model) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--flags", required=True) -p.add_argument("--table", required=True) -p.add_argument("--work", required=True) -p.add_argument("--rows", default="") -p.add_argument("--edits", type=int, default=200, help="per row, over all workers") -p.add_argument("--workers", type=int, default=8) -args = p.parse_args() - -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -opts = {flag_model.pname(o["flag"] + o["name"]): o for o in json.load(open(Path(args.flags) / "options.json"))} -rows = list(csv.DictReader(open(args.table), delimiter="\t")) -idx = list(range(len(rows))) -if args.rows: - lo, hi = (int(x) if x else None for x in args.rows.split(":")) - idx = idx[lo:hi] -log = WORK / "rows.jsonl" -done = {json.loads(l)["row"] for l in log.read_text().splitlines()} if log.exists() else set() - -for i in idx: - if i in done: - continue - flags = [flag_model.FLAG_BASE] - for k, v in rows[i].items(): - if k.startswith("B_") and v != "absent": - o = opts[k[2:]] - f = o["flag"] + o["name"] - flags.append(f if v == "present" else f + "=" + v.replace(";", ",")) - w = WORK / f"row{i}" - r = subprocess.run([sys.executable, str(Path(__file__).with_name("fuzz.py")), "--rustc", args.rustc, - "--fixture", args.fixture, "--work", str(w), "--workers", str(args.workers), - "--edits", str(max(1, args.edits // args.workers)), "--seed", str(i), - "--rustflags", " ".join(flags), "--target", "x86_64-unknown-linux-gnu", - "--pause-on-finding"], capture_output=True, text=True) - total = r.stdout.strip().splitlines()[-1] if r.stdout.strip() else "" - paused = (w / "PAUSED").read_text() if (w / "PAUSED").exists() else None - print(f"row {i}: {total}" + (f" PAUSED {paused}" if paused else ""), flush=True) - if paused: - sys.exit(3) # not recorded as done: rerun after patching to do this row again - with log.open("a") as f: - f.write(json.dumps({"row": i, "flags": flags[1:], "total": total}) + "\n") diff --git a/rustc/flag-min.py b/rustc/flag-min.py deleted file mode 100644 index f9c37a3..0000000 --- a/rustc/flag-min.py +++ /dev/null @@ -1,104 +0,0 @@ -#!/usr/bin/env python3 -"""Minimize the failures of a flag-walk.py run: for each distinct first error among rows -whose clean build with A failed, find the smallest set of the row's options that still gives -the same error (delta debugging, a clean build of the fixture per test). - - rustc/flag-min.py --rustc --fixture fixtures/sink --walk - [--per-error 1] [--jobs 4] - -Prints one line per error: the minimal options. Writes /minimized.json. -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--walk", required=True) -p.add_argument("--per-error", type=int, default=1) -p.add_argument("--jobs", type=int, default=4) -p.add_argument("--toolchain", default="nightly-2026-10-06") -p.add_argument("--target", default="x86_64-unknown-linux-gnu") -args = p.parse_args() - -FLAG_BASE = ("-Cunsafe-allow-abi-mismatch=sanitizer,sanitizer-cfi-normalize-integers," - "sanitizer-cfi-minimal-runtime,retpoline,retpoline-external-thunk," - "indirect-branch-cs-prefix,fixed-x18,reg-struct-return,regparm,branch-protection") -WALK = Path(args.walk) - - -def signature(line): - """An error line with paths, symbols, hashes and quoted names removed.""" - line = re.sub(r"_R\w+|/\S+|`[^`]*`|\b[0-9a-f]{16}\b", "…", line) - line = re.sub(r"\d+", "N", line) - return line[:120] - - -def first_error(log): - for l in log.splitlines(): - if l.startswith(("error", "rustc-LLVM ERROR", "LLVM ERROR")) and "could not compile" not in l: - return l - if "panicked at" in l: - return l - return "" - - -def minimize(flags, sig): - work = Path(tempfile.mkdtemp(dir=WALK)) - src, target = work / "s", work / "t" - shutil.copytree(Path(args.fixture), src, ignore=shutil.ignore_patterns("target", "edits", "edit")) - e = dict(os.environ, RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", CARGO_TERM_COLOR="never") - - def fails(fl): - shutil.rmtree(target, ignore_errors=True) - e["RUSTFLAGS"] = " ".join([FLAG_BASE] + fl) - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", - "--target", args.target, "--target-dir", str(target)], - cwd=src, env=e, capture_output=True, text=True) - return r.returncode != 0 and signature(first_error(r.stderr)) == sig - - if not fails(flags): - shutil.rmtree(work) - return None - n = 2 - while len(flags) >= 2: - chunk = max(1, len(flags) // n) - for i in range(0, len(flags), chunk): - rest = flags[:i] + flags[i + chunk:] - if fails(rest): - flags, n = rest, max(n - 1, 2) - break - else: - if chunk == 1: - break - n = min(len(flags), n * 2) - shutil.rmtree(work) - return flags - - -rows = [json.loads(l) for l in open(WALK / "results.jsonl")] -todo = {} -for r in rows: - if r["A_ok"]: - continue - err = (r.get("errors") or [r.get("error", "")])[0] - sig = signature(err) - todo.setdefault(sig, []) - if len(todo[sig]) < args.per_error: - todo[sig].append(r["A"]) - -jobs = [(sig, fl) for sig, fls in todo.items() for fl in fls] -with ThreadPoolExecutor(args.jobs) as ex: - found = list(ex.map(lambda j: (j[0], minimize(list(j[1]), j[0])), jobs)) -out = [{"error": sig, "minimal": fl} for sig, fl in found] -(WALK / "minimized.json").write_text(json.dumps(out, indent=1)) -for o in out: - print(o["minimal"], "->", o["error"]) diff --git a/rustc/flag-model.py b/rustc/flag-model.py deleted file mode 100644 index c55c84d..0000000 --- a/rustc/flag-model.py +++ /dev/null @@ -1,170 +0,0 @@ -#!/usr/bin/env python3 -"""Write a PICT model of rustc's option universe from flag-universe.py's results. - - rustc/flag-model.py [--transitions] [--cargo] [--allow-known] - -One parameter per option. Its values are absence, the values rustc accepted alone, the -values it accepts once `-Cunsafe-allow-abi-mismatch` names every target modifier (FLAG_BASE, -passed on every row), and the values that need another option, with IF/THEN constraints for -those needs. Options that stop compilation early (help, parse-only, link-only) are left out. -A value whose need lies outside the subset is dropped. - -With --cargo the model is for building a Cargo workspace (flag-walk.py): it leaves out the -values in CARGO_DROP, which fail there for reasons of Cargo or this machine, not the options. - -Combinations that hit bugs already in docs/hunt.md are excluded, so walks look for new -ones; --allow-known keeps those that have a local stopgap (for a compiler with them). - -With --transitions every parameter appears twice, A_ before and B_ after, for covering the -changes between two sessions. - -Run it with PICT (github.com/microsoft/pict): `pict /o:2` gives a pairwise covering -array, `/o:3` three-way. -""" - -import json -import re -import sys - -FLAG_BASE = ("-Cunsafe-allow-abi-mismatch=sanitizer,sanitizer-cfi-normalize-integers," - "sanitizer-cfi-minimal-runtime,retpoline,retpoline-external-thunk," - "indirect-branch-cs-prefix,fixed-x18,reg-struct-return,regparm,branch-protection") -STOP = {"-Chelp", "-Zhelp", "-Zparse-crate-root-only", "-Zno-analysis", "-Zlink-only", - "-Zimplicit-sysroot-deps"} -CARGO_DROP = { - "-Zassert-incr-state": None, # fails whenever the cache state differs, by design - "-Zbuild-sdylib-interface": None, # Cargo's target probe fails - "-Zchecksum-hash-algorithm": ["md5", "sha1"], # Cargo cannot parse the dep info - "-Zdirect-access-external-data": None, # link fails - "-Zfunction-return": ["thunk-extern"], # link fails: no thunk - "-Zlink-native-libraries": None, # link fails - "-Zlint-llvm-ir": None, # aborts on known LLVM lint findings (rust-lang/rust#59793) - "-Zno-codegen": None, "-Zno-link": None, # later crates need the output - "-Zpanic-in-drop": None, # std is built with unwind - "-Zsanitizer": None, # no sanitizer runtimes in this sysroot: link fails - "-Zretpoline-external-thunk": None, # link fails: no thunk - "-Ztiny-const-eval-limit": None, # the fixture's const evaluation exceeds it - "-Cpanic": ["immediate-abort"], # core is built with unwind - "-Ccode-model": ["tiny"], # LLVM ERROR: not supported on x86_64 - "-Ztls-model": ["local-exec", "emulated"], # dylib cannot link - # the dylib cannot link (static, pie, ropi); rwpi: finding 12 - "-Crelocation-model": ["static", "pie", "ropi", "rwpi", "ropi-rwpi"], - "-Clto": None, # rejected for rlibs and dylibs; Cargo's profile applies it to final artifacts only - # LLVM's pass listing from codegen threads interleaves with rustc's lines on stderr; a split - # -Ztime-passes-format=json line then reaches Cargo as a bare JSON message. - # Makes a crate behave like the standard library, which needs stability attributes on - # `const trait`s (fixtures/sink/nightly has one). - "-Zforce-unstable-if-unmarked": ["yes"], - "-Zprint-llvm-passes": ["yes"], - "-Ztime-passes-format": ["json"], -} -# Constraints that only a real workspace shows: a binary, a dylib, Cargo's own flags. -CARGO_NEEDS = [ - ("-Cprefer-dynamic", "yes", ["-Cpanic"], '[Cpanic] <> "abort"'), # libstd.so has panic_unwind - ("-Cprefer-dynamic", "yes", ["-Clto"], '[Clto] IN {"absent","no","off"}'), - # Findings 13 and 14 (in LLVM, not patched): retpolines with the machine outliner, or with - # the large code model. - ("-Zretpoline", "yes", ["-Ccode-model"], '[Ccode_model] <> "large"'), - # Finding 16 (no stopgap): cached derive expansions with the HIR crate hash. - ("-Zcache-proc-macros", "yes", ["-Zmetadata-crate-hash"], '[Zmetadata_crate_hash] <> "no"'), -] -# Bugs in docs/hunt.md that have a local stopgap: excluded unless --allow-known (for a -# compiler with the stopgaps). -KNOWN_NEEDS = [ - # Finding 9: without the default passes, local ThinLTO leaves undefined hidden symbols. - ("-Cno-prepopulate-passes", "present", ["-Zthinlto", "-Copt-level"], - '[Zthinlto] <> "yes" AND [Copt_level] IN {"absent","0"}'), -] -# Values left out of every model: LLVM's machine outliner crashes in many combinations -# (finding 13), which buries everything else; and see below. -DROP = {"-Cllvm-args": ["-enable-machine-outliner"], - # Not a bug: the limit counts MIR pass runs across the session, so which bodies stay - # under it depends on how many bodies the session computes (an incremental session - # computes fewer) and, with -Zthreads, on thread timing. - "-Zmir-opt-bisect-limit": ["1", "16"], - # Not a bug: a testing option that leaves spans out of the incremental hashes, so a - # rebuild keeps stale spans by design. - "-Zincremental-ignore-spans": ["yes"]} -# Rejected alone; accepted with FLAG_BASE or with the needs below. -EXTRA = {"-Zindirect-branch-cs-prefix": ["yes"], "-Zretpoline-external-thunk": ["yes"], - "-Zretpoline": ["yes"], - "-Zsanitizer": ["dataflow", "memory", "safestack", "thread", "cfi", "kcfi"], - "-Cforce-frame-pointers": ["non-leaf"], "-Cpanic": ["immediate-abort"], - "-Zdump-dep-graph": ["yes"], "-Zsanitizer-cfi-canonical-jump-tables": ["no"], - "-Zsanitizer-cfi-diag": ["yes"], "-Zsanitizer-cfi-generalize-pointers": ["yes"], - "-Zsanitizer-cfi-minimal-runtime": ["yes"], "-Zsanitizer-cfi-normalize-integers": ["yes"], - "-Zsanitizer-cfi-recover": ["yes"], "-Zsanitizer-kcfi-arity": ["yes"], - "-Zsplit-lto-unit": ["yes"], "-Zvirtual-function-elimination": ["yes"]} -LTO_ON = '{"yes","on","thin","fat"}' -LTO_FAT = '{"yes","on","fat"}' -# (option, value, options needed, PICT condition) -NEEDS = [ - ("-Cembed-bitcode", "no", ["-Clto"], '[Clto] IN {"absent","no","off"}'), - ("-Zsplit-lto-unit", "yes", ["-Clto"], "[Clto] IN " + LTO_ON), - ("-Zvirtual-function-elimination", "yes", ["-Clto"], "[Clto] IN " + LTO_FAT), - ("-Zsanitizer", "cfi", ["-Clto", "-Ccodegen-units"], "[Clto] IN " + LTO_FAT + ' AND [Ccodegen_units] = "1"'), - ("-Zsanitizer", "kcfi", ["-Cpanic"], '[Cpanic] = "abort"'), - ("-Cforce-frame-pointers", "non-leaf", ["-Zunstable-options"], '[Zunstable_options] = "present"'), - ("-Cpanic", "immediate-abort", ["-Zunstable-options"], '[Zunstable_options] = "present"'), - ("-Zdump-dep-graph", "yes", ["-Zquery-dep-graph"], '[Zquery_dep_graph] = "yes"'), - ("-Zsanitizer-cfi-diag", "yes", ["-Zsanitizer"], '[Zsanitizer] = "cfi"'), - ("-Zsanitizer-cfi-recover", "yes", ["-Zsanitizer"], '[Zsanitizer] = "cfi"'), - ("-Zsanitizer-cfi-minimal-runtime", "yes", ["-Zsanitizer"], '[Zsanitizer] = "cfi"'), - ("-Zsanitizer-cfi-canonical-jump-tables", "no", ["-Zsanitizer"], '[Zsanitizer] = "cfi"'), - ("-Zsanitizer-cfi-generalize-pointers", "yes", ["-Zsanitizer"], '[Zsanitizer] IN {"cfi","kcfi"}'), - ("-Zsanitizer-cfi-normalize-integers", "yes", ["-Zsanitizer"], '[Zsanitizer] IN {"cfi","kcfi"}'), - ("-Zsanitizer-kcfi-arity", "yes", ["-Zsanitizer"], '[Zsanitizer] = "kcfi"'), - ("-Zsanitizer-cfi-minimal-runtime", "yes", ["-Zsanitizer-cfi-recover", "-Zsanitizer-cfi-diag"], - '([Zsanitizer_cfi_recover] = "yes" OR [Zsanitizer_cfi_diag] = "yes")'), -] - - -def pname(opt): - return re.sub(r"[^A-Za-z0-9]", "_", opt.lstrip("-")) - - -def main(): - work, subset, out = sys.argv[1:4] - transitions = "--transitions" in sys.argv[4:] - cargo = "--cargo" in sys.argv[4:] - known = "--allow-known" not in sys.argv[4:] - opts = {o["flag"] + o["name"]: o for o in json.load(open(work + "/options.json"))} - domains = {} - for s in json.load(open(work + "/singles.json")): - if s["ok"] and s["option"] not in STOP: - domains.setdefault(s["option"], []).append("present" if s["value"] is None else s["value"]) - for k, vs in EXTRA.items(): - for v in vs: - if v not in domains.setdefault(k, []): - domains[k].append(v) - for k, vs in DROP.items(): - domains[k] = [v for v in domains.get(k, []) if v not in vs] - if cargo: - for k, vs in CARGO_DROP.items(): - domains[k] = [] if vs is None else [v for v in domains.get(k, []) if v not in vs] - domains = {k: v for k, v in domains.items() if v} - - keep = [k for k in sorted(domains) - if subset == "all" or (subset == "untracked") == (opts[k]["tracking"] == "UNTRACKED")] - cons = [] - for k, v, needs, cond in NEEDS + (CARGO_NEEDS if cargo else []) + (KNOWN_NEEDS if known else []): - if k not in keep: - continue - if all(n in keep for n in needs): - cons.append(f'IF [{pname(k)}] = "{v}" THEN {cond};') - elif v in domains[k]: - domains[k].remove(v) - - params = [f"{pname(k)}: {', '.join(['absent'] + [v.replace(',', ';') for v in domains[k]])}" for k in keep] - if transitions: - params = [f"{t}_{p}" for t in "AB" for p in params] - cons = [re.sub(r"\[(\w+)\]", lambda m: f"[{t}_{m.group(1)}]", c) for t in "AB" for c in cons] - if known and "-Zprint-type-sizes" in keep: - # Finding 11 in docs/hunt.md: the rebuild ICEs once -Zprint-type-sizes is dropped. - cons.append('IF [A_Zprint_type_sizes] = "yes" THEN [B_Zprint_type_sizes] = "yes";') - open(out, "w").write("\n".join(params) + "\n\n" + "\n".join(cons) + "\n") - print(f"{len(params)} parameters, {len(cons)} constraints") - - -if __name__ == "__main__": - main() diff --git a/rustc/flag-rows.py b/rustc/flag-rows.py deleted file mode 100644 index 196bdcc..0000000 --- a/rustc/flag-rows.py +++ /dev/null @@ -1,53 +0,0 @@ -#!/usr/bin/env python3 -"""Compile a trivial crate once per row of a PICT table made from flag-model.py's model, and -count the rows rustc rejects, grouped by first error. - - rustc/flag-rows.py [extra rustc args, e.g. --emit=metadata] - -Tables from a --transitions model are not supported. -""" - -import csv -import importlib.util -import json -import subprocess -import sys -import tempfile -from collections import Counter -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -spec = importlib.util.spec_from_file_location("flag_model", Path(__file__).with_name("flag-model.py")) -flag_model = importlib.util.module_from_spec(spec) -spec.loader.exec_module(flag_model) - -work, tsv, rustc = sys.argv[1:4] -extra = sys.argv[4:] -opts = {flag_model.pname(o["flag"] + o["name"]): o for o in json.load(open(work + "/options.json"))} -rows = list(csv.DictReader(open(tsv), delimiter="\t")) - - -def argv(row): - out = [flag_model.FLAG_BASE] - for k, v in row.items(): - if v != "absent": - f = opts[k]["flag"] + opts[k]["name"] - out.append(f if v == "present" else f + "=" + v.replace(";", ",")) - return out - - -def run(row): - with tempfile.TemporaryDirectory(dir=work) as d: - Path(d, "lib.rs").write_text("pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n") - a = argv(row) - p = subprocess.run([rustc, "--edition", "2021", "--crate-type", "lib", *extra, "-o", d + "/out", - *a, d + "/lib.rs"], capture_output=True, text=True, cwd=d, timeout=300) - err = next((l for l in p.stderr.splitlines() if l.startswith("error")), "") - return p.returncode == 0, err, len(a) - 1 - - -res = list(ThreadPoolExecutor(10).map(run, rows)) -bad = [e for ok, e, _ in res if not ok] -print(f"{len(rows)} rows, {len(bad)} rejected, {sum(n for *_, n in res) / len(res):.0f} options per row on average") -for e, c in Counter(e[:110] for e in bad).most_common(): - print(c, e) diff --git a/rustc/flag-universe.py b/rustc/flag-universe.py deleted file mode 100644 index 7f9bc19..0000000 --- a/rustc/flag-universe.py +++ /dev/null @@ -1,251 +0,0 @@ -#!/usr/bin/env python3 -"""Enumerate rustc's -C and -Z options, find which values and pairs of values it accepts, -and size covering arrays over them. - - rustc/flag-universe.py --rustc --source --work [--jobs N] - -Each option's domain is its absence plus the values worth trying: `yes`/`no` for a boolean, -present for an option without a value, the values its parser's description lists for an -enumerated one, two samples for a number. Options taking free-form strings, paths or lists -are counted but left out. Every value is tried alone on a trivial crate (`--emit=metadata`, -so only option checking and a tiny compilation run), then every pair of accepted values of -different options. A pair is "rejected" when rustc fails with the pair but accepts each value -alone. - -Writes /options.json (domains), /singles.json, /pairs.json, and prints -the sizes of pairwise covering arrays built greedily over the accepted domains. -""" - -import argparse -import itertools -import json -import os -import random -import re -import subprocess -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--source", required=True, help="a rust checkout, for compiler/rustc_session/src/options.rs") -p.add_argument("--work", required=True) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--skip-pairs", action="store_true") -args = p.parse_args() - -work = Path(args.work) -work.mkdir(parents=True, exist_ok=True) -src = (Path(args.source) / "compiler/rustc_session/src/options.rs").read_text() - -# The parsers' descriptions, joined across lines. -descs = {} -for m in re.finditer(r'pub\(crate\) const (parse_\w+): &str =\s*((?:"(?:[^"\\]|\\.)*"\s*)+|parse_\w+|[^;]+);', src): - descs[m.group(1)] = m.group(2) -for k, v in list(descs.items()): - if re.fullmatch(r"parse_\w+", v.strip()): - descs[k] = descs.get(v.strip(), "") - -BOOL = {"parse_bool", "parse_opt_bool"} -NO_VALUE = {"parse_no_value"} -NUMBER = {"parse_number", "parse_opt_number"} -FREE = {"parse_string", "parse_opt_string", "parse_string_push", "parse_opt_pathbuf", "parse_list", - "parse_comma_list", "parse_opt_comma_list", "parse_ignore", "parse_target_feature", - "parse_list_with_polarity", "parse_llvm_module_flag", "parse_patchable_function_entry", - "parse_autodiff", "parse_offload", "parse_allow_partial_mitigations", - "parse_deny_partial_mitigations", "parse_rust_version", "parse_unpretty", - "parse_passes", "parse_branch_protection", "parse_instrument_xray", - "parse_linker_features", "parse_link_self_contained", "parse_align", - "parse_location_detail", "parse_coverage_options", "parse_codegen_retag_options"} - - -# Values for options whose parser takes a string or whose description lists no values, picked -# by hand (`rustc --print code-models` etc. for the enumerations). -SAMPLES = { - "-Copt-level": ["0", "1", "2", "3", "s", "z"], - "-Ccode-model": ["tiny", "small", "kernel", "medium", "large"], - "-Crelocation-model": ["static", "pic", "pie", "dynamic-no-pic", "ropi", "rwpi", "ropi-rwpi", "default"], - "-Ztls-model": ["global-dynamic", "local-dynamic", "initial-exec", "local-exec", "emulated"], - "-Ctarget-cpu": ["generic", "native", "x86-64-v2", "x86-64-v3", "x86-64-v4"], - "-Ctarget-feature": ["+avx2", "+avx512f", "-sse4.2", "+crt-static"], - "-Ztune-cpu": ["generic", "znver4"], - "-Zthreads": ["1", "4"], - "-Zlocation-detail": ["none", "file", "line,column"], - "-Zmin-function-alignment": ["16", "64"], - "-Zpatchable-function-entry": ["4", "4,2"], - "-Zmir-enable-passes": ["+Inline", "-GVN", "+DeadStoreElimination-final"], - "-Zremap-cwd-prefix": ["/remapped"], - "-Zsimulate-remapped-rust-src-base": ["/rustc/simulated"], - "-Zhint-msrv": ["1.60.0"], - "-Cmetadata": ["mirth"], - "-Zinstrument-xray": ["always", "never"], - "-Cllvm-args": ["-unroll-threshold=0", "-enable-machine-outliner"], - "-Zcrate-attr": ["allow(unused)"], -} - - -def enum_values(parser): - vals = re.findall(r"`([^`]+)`", descs.get(parser, "")) - out = [] - for v in vals: - if v in out or " " in v or "<" in v or "=" in v: - continue - out.append(v) - return out - - -options = [] -for flag, grp in (("-C", "CodegenOptions"), ("-Z", "UnstableOptions")): - i = src.index("options! {\n " + grp) - j = src.index("\n}", i) - for line in src[i:j].splitlines(): - m = re.match(r"^\s*(\w+): .*?, (parse_\w+), \[(\w+)", line) - if not m: - continue - name, parser, tracking = m.group(1).replace("_", "-"), m.group(2), m.group(3) - if parser in BOOL: - values = ["yes", "no"] - elif parser in NO_VALUE: - values = [None] - elif parser in NUMBER: - values = ["1", "16"] - elif parser in FREE: - values = [] - else: - values = enum_values(parser) - values = SAMPLES.get(flag + name.replace("_", "-"), values) - options.append({"flag": flag, "name": name, "parser": parser, "tracking": tracking, - "values": values, "free": not values}) - - -def arg(o, v): - return f"{o['flag']}{o['name']}" if v is None else f"{o['flag']}{o['name']}={v}" - - -def run(argv): - with tempfile.TemporaryDirectory(dir=work) as d: - lib = Path(d) / "lib.rs" - lib.write_text("pub fn f(x: u32) -> u32 { x.wrapping_mul(3) }\n") - try: - r = subprocess.run([args.rustc, "--edition", "2021", "--crate-type", "lib", "--emit=metadata", - "-o", str(Path(d) / "out.rmeta"), *argv, str(lib)], - capture_output=True, text=True, timeout=60, cwd=d) - err = r.stderr - first = next((l for l in err.splitlines() if l.startswith(("error", "warning"))), "") - return {"ok": r.returncode == 0, "warn": "warning" in err, "msg": first[:200]} - except subprocess.TimeoutExpired: - return {"ok": False, "warn": False, "msg": "timeout"} - - -(work / "options.json").write_text(json.dumps(options, indent=1)) -walkable = [o for o in options if not o["free"]] -print(f"{len(options)} options: {len(walkable)} with enumerable values, " - f"{len(options) - len(walkable)} free-form") - -singles_path = work / "singles.json" -singles = json.loads(singles_path.read_text()) if singles_path.exists() else [] -# Try the values not tried yet (all of them on a first run). -done = {s["arg"] for s in singles} -jobs = [(o, v) for o in walkable for v in o["values"] if arg(o, v) not in done] -if jobs: - with ThreadPoolExecutor(args.jobs) as ex: - results = list(ex.map(lambda ov: run([arg(*ov)]), jobs)) - singles += [{"arg": arg(o, v), "option": o["flag"] + o["name"], "value": v, **r} - for (o, v), r in zip(jobs, results)] - singles_path.write_text(json.dumps(singles, indent=1)) -accepted = {} -for s in singles: - if s["ok"]: - accepted.setdefault(s["option"], []).append(s["arg"]) -print(f"{len(singles)} single values tried, {sum(map(len, accepted.values()))} accepted " - f"({len(accepted)} options with at least one)") - -rejected_pairs = set() -if not args.skip_pairs: - pairs_path = work / "pairs.json" - if pairs_path.exists(): - pairs = json.loads(pairs_path.read_text()) - else: - opts = sorted(accepted) - jobs = [(a, b) for x, y in itertools.combinations(opts, 2) for a in accepted[x] for b in accepted[y]] - # A value rejected alone may need another option: try it with every accepted value - # of every other option. - for r in (x for x in singles if not x["ok"]): - jobs += [(r["arg"], b) for y in opts if y != r["option"] for b in accepted[y]] - print(f"{len(jobs)} pairs to try", flush=True) - with ThreadPoolExecutor(args.jobs) as ex: - results = list(ex.map(lambda ab: run(list(ab)), jobs)) - pairs = [{"a": a, "b": b, **r} for (a, b), r in zip(jobs, results)] - pairs_path.write_text(json.dumps(pairs, indent=1)) - alone = {x["arg"] for x in singles if x["ok"]} - rejected_pairs = {(x["a"], x["b"]) for x in pairs if not x["ok"] and x["a"] in alone and x["b"] in alone} - requires = {} - for x in pairs: - if x["ok"] and x["a"] not in alone: - requires.setdefault(x["a"], []).append(x["b"]) - print(f"{len(pairs)} pairs tried; {len(rejected_pairs)} pairs of values accepted alone " - f"are rejected together; {len(requires)} values rejected alone are accepted with " - f"another option") - (work / "requires.json").write_text(json.dumps(requires, indent=1)) - - -def covering_array(domains, forbidden, seed=0): - """A pairwise covering array, built greedily: each row is chosen among random - candidates to cover the most uncovered pairs. Values are indices into each domain; index - 0 is the option's absence.""" - rng = random.Random(seed) - names = list(domains) - n = len(names) - uncovered = set() - for i, j in itertools.combinations(range(n), 2): - for a in range(len(domains[names[i]])): - for b in range(len(domains[names[j]])): - if (domains[names[i]][a], domains[names[j]][b]) not in forbidden: - uncovered.add((i, a, j, b)) - rows = [] - while uncovered: - best, best_gain = None, -1 - target = next(iter(uncovered)) - for _ in range(30): - row = [rng.randrange(len(domains[k])) for k in names] - row[target[0]], row[target[2]] = target[1], target[3] - # repair forbidden pairs by falling back to absence - for i, j in itertools.combinations(range(n), 2): - if (domains[names[i]][row[i]], domains[names[j]][row[j]]) in forbidden: - if j not in (target[0], target[2]): - row[j] = 0 - elif i not in (target[0], target[2]): - row[i] = 0 - gain = sum(1 for i, j in itertools.combinations(range(n), 2) if (i, row[i], j, row[j]) in uncovered) - if gain > best_gain: - best, best_gain = row, gain - rows.append(best) - for i, j in itertools.combinations(range(n), 2): - uncovered.discard((i, best[i], j, best[j])) - return rows - - -def summarize(label, opts): - domains = {o: [None] + accepted[o] for o in opts} - sizes = sorted((len(v) for v in domains.values()), reverse=True) - total = 1 - for s in sizes: - total *= s - lower = sizes[0] * sizes[1] if len(sizes) > 1 else sizes[0] - forbidden = {(a, b) for a, b in rejected_pairs} | {(b, a) for a, b in rejected_pairs} - rows = covering_array(domains, forbidden) - print(f"{label}: {len(opts)} options, all combinations {total:.3e}, " - f"pairwise covering array {len(rows)} rows (lower bound {lower})") - return rows - - -byopt = {o["flag"] + o["name"]: o for o in options} -results = {} -for label, pred in (("untracked", lambda o: o["tracking"] == "UNTRACKED"), - ("tracked", lambda o: o["tracking"] != "UNTRACKED"), - ("all", lambda o: True)): - opts = sorted(k for k in accepted if pred(byopt[k])) - if len(opts) > 1: - results[label] = summarize(label, opts) -(work / "covering.json").write_text(json.dumps({k: len(v) for k, v in results.items()})) diff --git a/rustc/flag-walk.py b/rustc/flag-walk.py deleted file mode 100644 index 25df97b..0000000 --- a/rustc/flag-walk.py +++ /dev/null @@ -1,272 +0,0 @@ -#!/usr/bin/env python3 -"""Walk option transitions on a fixture: for each row of a PICT table made from a -`flag-model.py --transitions` model, build the fixture clean with the A_ options, rebuild it -incrementally with the B_ options, build it clean with the B_ options, and compare the -rebuild with the clean build (metadata, object code, binary, diagnostics, the binary's -output), as fuzz.py does after an edit. - - rustc/flag-walk.py --rustc --fixture fixtures/sink --flags - --table rows.tsv --work [--workers 8] [--rows a:b] - -Rows change options only; the source is not edited. - -The options go in RUSTFLAGS with `--target` set, so they apply to the fixture's crates but -not to its build scripts and proc macros. RUSTC_VERIFY_REUSE and RUSTC_REPORT_UNTRACKED are -set, for a compiler with mirth's local patches. - -Writes /results.jsonl (one line per row) and /findings// for each row whose -rebuild differs from the clean build, or which crashed. - -To stay at the frontier: with --pause-on-finding the walk stops taking rows at the first -finding not marked known and writes /PAUSED. Patch the compiler, then run the same -command with --rustc --recheck: the rows with findings run again first, then the -rows not yet walked. A rerun never repeats rows that are done. -""" - -import argparse -import csv -import difflib -import importlib.util -import json -import os -import random -import shutil -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import artifacts # noqa: E402 -import mutations # noqa: E402 - -spec = importlib.util.spec_from_file_location("flag_model", Path(__file__).with_name("flag-model.py")) -flag_model = importlib.util.module_from_spec(spec) -spec.loader.exec_module(flag_model) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--flags", required=True, help="flag-universe.py's work directory") -p.add_argument("--table", required=True) -p.add_argument("--work", required=True) -p.add_argument("--workers", type=int, default=8) -p.add_argument("--toolchain", default="nightly-2026-10-06") -p.add_argument("--target", default="x86_64-unknown-linux-gnu") -p.add_argument("--timeout", type=int, default=600) -p.add_argument("--rows", default="", help="a:b, a slice of the table") -p.add_argument("--edits", type=int, default=0, help="random source edits between A and B (fuzz.py's)") -p.add_argument("--seed", type=int, default=0) -p.add_argument("--pause-on-finding", action="store_true", - help="stop taking rows at the first finding not marked known; rerun to resume") -p.add_argument("--recheck", action="store_true", - help="on resume, run the rows that had findings again first (after patching rustc)") -p.add_argument("--p5-builds", type=int, default=12, help="clean rebuilds before a difference counts as reuse") -args = p.parse_args() - -FIXTURE = Path(args.fixture).resolve() -WORK = Path(args.work).resolve() -opts = {flag_model.pname(o["flag"] + o["name"]): o for o in json.load(open(Path(args.flags) / "options.json"))} - - -def flags(row, side): - """RUSTFLAGS for one side of a row. Cargo passes `-Cembed-bitcode=no` unless its profile - asks for LTO, and the profile's LTO reaches only the final artifacts, so `-Clto` goes in - RUSTFLAGS with `-Cembed-bitcode=yes` after Cargo's flag.""" - out = [flag_model.FLAG_BASE] - for k, v in row.items(): - if k.startswith(side + "_") and v != "absent": - o = opts[k[2:]] - f = o["flag"] + o["name"] - out.append(f if v == "present" else f + "=" + v.replace(";", ",")) - if row.get(side + "_Clto") in ("yes", "on", "thin", "fat") and row.get(side + "_Cembed_bitcode") == "absent": - out.append("-Cembed-bitcode=yes") - return out - - -def build(src, target, rustflags): - e = dict(os.environ) - e.update(RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", CARGO_TERM_COLOR="never", - RUSTFLAGS=" ".join(rustflags), RUSTC_VERIFY_REUSE="1", RUSTC_REPORT_UNTRACKED="1") - try: - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", - "--target", args.target, "--target-dir", str(target), - "--message-format=json-render-diagnostics"], - cwd=src, env=e, capture_output=True, text=True, timeout=args.timeout) - rc, out, log = r.returncode, r.stdout, r.stderr - except subprocess.TimeoutExpired as t: - rc, out, log = -1, t.stdout or "", (t.stderr or "") + f"\nkilled after {args.timeout}s" - out = out.decode() if isinstance(out, bytes) else out - log = log.decode() if isinstance(log, bytes) else log - exe = None - for line in out.splitlines(): - try: - msg = json.loads(line) - except ValueError: - continue - if msg.get("reason") == "compiler-artifact" and msg.get("executable") and msg["target"]["name"] == FIXTURE.name: - exe = msg["executable"] - return {"ok": rc == 0, "log": log, - "ice": "internal compiler error" in log or "the compiler unexpectedly panicked" in log - or "rustc interrupted by SIG" in log, - "reuse": sorted({l.split(":", 1)[1].strip()[:200] for l in log.splitlines() - if l.startswith("rustc-verify-reuse:")}), - "untracked": sorted({l.strip() for l in log.splitlines() if l.startswith("rustc-untracked-read:")}), - "error": next((l for l in log.splitlines() if l.startswith("error")), ""), - "errors": [l[:300] for l in log.splitlines() if l.startswith(("error", "rustc-LLVM ERROR", "LLVM ERROR")) - and "could not compile" not in l][:6], - "art": artifacts.collect(out, target) if rc == 0 else None, "exe": exe} - - -def run_exe(exe): - if not exe: - return None - try: - r = subprocess.run([exe], capture_output=True, text=True, timeout=30) - return [r.returncode, r.stdout[-2000:]] - except subprocess.TimeoutExpired: - return ["timeout", ""] - - -def edit(src, rng, n): - """Apply n random edits to the fixture's sources; returns the unified diff.""" - diff = "" - for k in range(n): - paths = sorted(p for p in src.rglob("*.rs") if "target" not in p.parts) - for _ in range(20): - path = rng.choice(paths) - fn = rng.choices([e for e, _ in mutations.EDITS], weights=[w for _, w in mutations.EDITS])[0] - if fn in (mutations.str_literal, mutations.int_literal) and path.name == "build.rs": - continue # as in fuzz.py: stale OUT_DIR files, or a build script that loops - old = path.read_text() - new = fn(old, rng, k) - if new is not None and new != old: - path.write_text(new) - rel = str(path.relative_to(src)) - diff += "".join(difflib.unified_diff(old.splitlines(True), new.splitlines(True), - "a/" + rel, "b/" + rel)) - break - return diff - - -def walk(i_row): - i, row = i_row - if (WORK / "PAUSED").exists() or (WORK / "STOP").exists(): - return None - home = WORK / f"r{i}" - if home.exists(): - shutil.rmtree(home) - src, target, inc_target = home / "src", home / "target", home / "target-inc" - shutil.copytree(FIXTURE, src, ignore=shutil.ignore_patterns("target", "edits", "edit")) - a, b = flags(row, "A"), flags(row, "B") - res = {"row": i, "A": a[1:], "B": b[1:]} - first = build(src, target, a) - res["A_ok"] = first["ok"] - findings = [] - if first["ice"]: - findings.append("ICE in clean A") - if first["ok"]: - res["diff"] = edit(src, random.Random(args.seed * 1_000_003 + i), args.edits) - inc = build(src, target, b) - target.rename(inc_target) - clean = build(src, target, b) - res.update(inc_ok=inc["ok"], clean_ok=clean["ok"], reuse=inc["reuse"], untracked=inc["untracked"]) - if inc["ice"]: - findings.append("ICE in rebuild B") - if clean["ice"]: - findings.append("ICE in clean B") - if inc["ok"] != clean["ok"]: - findings.append("split: rebuild " + ("ok" if inc["ok"] else "failed") + - ", clean " + ("ok" if clean["ok"] else "failed")) - if inc["ok"] and clean["ok"]: - diff = artifacts.compare(inc["art"], clean["art"]) - if row.get("B_Csplit_debuginfo") in ("packed", "unpacked"): - # Objects and binary name .dwo files by session (DW_AT_GNU_dwo_name, and the - # dwo_id hashed from it), so two clean builds differ too. - diff.pop("rlib", None) - diff.pop("exe", None) - ra = run_exe(inc["exe"].replace(str(target), str(inc_target)) if inc["exe"] else None) - rb = run_exe(clean["exe"]) - if ra != rb: - findings.append(f"run: {ra} vs {rb}") - if diff: - # Clean builds may differ among themselves (P5), sometimes only one time in - # five: build clean again up to --p5-builds times before calling it reuse. - p5 = {} - for _ in range(args.p5_builds): - shutil.rmtree(target) - again = build(src, target, b) - if again["ok"]: - p5 = artifacts.compare(clean["art"], again["art"]) - if p5 or not again["ok"]: - break - kind = "P5 " if p5 else "" - findings += [f"{kind}{k}: {v[:5]}" for k, v in diff.items()] - res["error"] = inc["error"] or clean["error"] - res["errors"] = inc["errors"] or clean["errors"] - if findings: - d = WORK / "findings" / f"r{i}" - d.mkdir(parents=True, exist_ok=True) - (d / "row.json").write_text(json.dumps({**res, "findings": findings}, indent=1)) - if res.get("diff"): - (d / "edit.diff").write_text(res["diff"]) - (d / "inc.log").write_text(inc["log"][-20000:]) - (d / "clean.log").write_text(clean["log"][-20000:]) - else: - res["error"] = first["error"] - res["errors"] = first["errors"] - if findings: - d = WORK / "findings" / f"r{i}" - d.mkdir(parents=True, exist_ok=True) - (d / "row.json").write_text(json.dumps({**res, "findings": findings}, indent=1)) - (d / "a.log").write_text(first["log"][-20000:]) - res["findings"] = findings - res["rustc"] = args.rustc - new = [f for f in findings if not f.startswith("known")] - if new and args.pause_on_finding: - (WORK / "PAUSED").write_text(json.dumps({"row": i, "findings": new}, indent=1)) - shutil.rmtree(home) - with open(WORK / "results.jsonl", "a") as f: - f.write(json.dumps(res) + "\n") - print(f"row {i}: A {'ok' if res['A_ok'] else 'failed'}" - + (f", rebuild {'ok' if res.get('inc_ok') else 'failed'}" if res["A_ok"] else "") - + (f"; {findings}" if findings else "") + (f"; {res['error'][:100]}" if res.get("error") else ""), - flush=True) - return res - - -def latest(): - """The last result of each row, from /results.jsonl.""" - out = {} - path = WORK / "results.jsonl" - if path.exists(): - for line in path.read_text().splitlines(): - r = json.loads(line) - out[r["row"]] = r - return out - - -if __name__ == "__main__": - WORK.mkdir(parents=True, exist_ok=True) - (WORK / "PAUSED").unlink(missing_ok=True) - rows = list(csv.DictReader(open(args.table), delimiter="\t")) - idx = list(range(len(rows))) - if args.rows: - lo, hi = (int(x) if x else None for x in args.rows.split(":")) - idx = idx[lo:hi] - # Resume: rows with a result are done, except, with --recheck, those with findings, which - # run first. - done = latest() - again = [i for i in idx if i in done and done[i]["findings"]] if args.recheck else [] - idx = again + [i for i in idx if i not in done] - if again: - print(f"rechecking rows {again}", flush=True) - with ThreadPoolExecutor(args.workers) as ex: - list(ex.map(walk, [(i, rows[i]) for i in idx])) - results = list(latest().values()) - summary = {"rows": len(rows), "done": len(results), "A ok": sum(r["A_ok"] for r in results), - "compared": sum(bool(r.get("inc_ok") and r.get("clean_ok")) for r in results), - "findings": sum(bool(r["findings"]) for r in results)} - if (WORK / "PAUSED").exists(): - summary["paused"] = json.loads((WORK / "PAUSED").read_text()) - print(json.dumps(summary)) diff --git a/rustc/fuzz-replay.py b/rustc/fuzz-replay.py deleted file mode 100755 index f85a58f..0000000 --- a/rustc/fuzz-replay.py +++ /dev/null @@ -1,85 +0,0 @@ -#!/usr/bin/env python3 -"""Replay a fuzz finding: apply its edits to the pristine fixture in order, -with an incremental build after each, then compare with a clean build. - - rustc/fuzz-replay.py --rustc --fixture fixtures/sink --finding --work [--upto N] - -Prints which .rmeta files differ, and keeps the compiler's output of the last -incremental build and the clean build as inc.log and clean.log. --upto replays only the first N edits that -were kept, to find where the difference appears. -""" - -import argparse -import json -import os -import shutil -import subprocess -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--finding", required=True) -p.add_argument("--work", required=True) -p.add_argument("--upto", type=int, default=None) -p.add_argument("--toolchain", default="nightly-2026-10-06") -p.add_argument("--rustflags", default="-Zincremental-verify-ich") -p.add_argument("--quiet", action="store_true") -args = p.parse_args() - -work = Path(args.work).resolve() -src, target, inc_target = work / "src", work / "target", work / "target-inc" -env = dict(os.environ, RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", RUSTFLAGS=args.rustflags) - - -def build(t): - r = subprocess.run(["cargo", f"+{args.toolchain}", "build", "--workspace", "--offline", "-j", "4", - "--target-dir", str(t), "--message-format=json-render-diagnostics"], - cwd=src, env=env, capture_output=True, text=True) - rmetas = {} - for line in r.stdout.splitlines(): - try: - msg = json.loads(line) - except ValueError: - continue - if msg.get("reason") == "compiler-artifact": - for f in msg["filenames"]: - if f.endswith(".rmeta"): - rmetas[str(Path(f).relative_to(t))] = Path(f).read_bytes() - build.log = r.stderr - return r.returncode == 0, rmetas - - -if work.exists(): - shutil.rmtree(work) -work.mkdir(parents=True) -shutil.copytree(Path(args.fixture).resolve(), src, ignore=shutil.ignore_patterns("target", "edits", "edit")) -build(target) - -history = json.loads((Path(args.finding) / "history.json").read_text()) -kept = 0 -last = None -for step in history: - if step["edit"] == "revert": - (src / last["file"]).write_text(last["before"]) - else: - if args.upto is not None and step["kept"] and kept >= args.upto: - break - (src / step["file"]).write_text(step["after"]) - last = step - kept += step["kept"] - ok, _ = build(target) - if not args.quiet: - print(f"{step['edit']:18} {step['file']:24} {'built' if ok else 'failed'}") - -ok, inc = build(target) -(work / "inc.log").write_text(build.log) -target.rename(inc_target) -ok2, clean = build(target) -(work / "clean.log").write_text(build.log) -differ = sorted(r for r in set(inc) | set(clean) if inc.get(r) != clean.get(r)) -print(json.dumps({"kept_edits": kept, "inc_ok": ok, "clean_ok": ok2, "differ": [Path(d).name for d in differ]})) -for d in differ: - name = Path(d).name - (work / (name + ".inc")).write_bytes(inc.get(d, b"")) - (work / (name + ".clean")).write_bytes(clean.get(d, b"")) diff --git a/rustc/fuzz.py b/rustc/fuzz.py deleted file mode 100755 index d7a7909..0000000 --- a/rustc/fuzz.py +++ /dev/null @@ -1,383 +0,0 @@ -#!/usr/bin/env python3 -"""Make random edits to a fixture and check each incremental rebuild. - -Each worker keeps one copy of the fixture and repeats: - - 1. apply a random mechanical edit (a comment, a moved item, a changed - literal, a new function, ...); - 2. rebuild incrementally; if the edit does not compile, revert it (the next - build then also exercises recovery from a failed session); - 3. build the same source from scratch, at the same path; - 4. compare: - P6 every .rmeta Cargo reports for a workspace member - rlib every rlib's members, object code included (artifacts.py) - exe the binary's bytes - diag the diagnostics each crate printed - run the binaries' output and exit status - reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, - docs/hunt/verify-reuse.patch) found nothing stale - P5 when anything differs, a second clean build is made; if the two - clean builds differ, that is reported instead (nondeterminism) - ICE neither build crashed the compiler - split both builds succeed or both fail - -Every --reset edits the worker starts again from the pristine fixture. A -finding keeps the diffs since the last reset, which replay it, and both -builds' differing files and logs. - - rustc/fuzz.py --rustc --fixture fixtures/sink --work [--workers 8] [--edits N] - -Stop it early by creating /STOP. Progress is in /stats.json. With ---pause-on-finding all workers stop at the first finding (/PAUSED says which); patch -rustc and run again with the patched compiler. -""" - -import argparse -import difflib -import json -import multiprocessing -import os -import random -import re -import shutil -import subprocess -import sys -import time -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).resolve().parent)) -import artifacts # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--fixture", required=True) -p.add_argument("--work", required=True) -p.add_argument("--workers", type=int, default=8) -p.add_argument("--edits", type=int, default=10**9, help="per worker") -p.add_argument("--reset", type=int, default=40) -p.add_argument("--keep", type=int, default=5, help="findings kept per kind") -p.add_argument("--toolchain", default="nightly-2026-10-06") -p.add_argument("--rustflags", default="-Zincremental-verify-ich") -p.add_argument("--seed", type=int, default=0) -p.add_argument("--timeout", type=int, default=180, help="seconds before a build counts as hung") -p.add_argument("--p5-builds", type=int, default=12, - help="clean builds made again when anything differs, to tell nondeterminism from P6") -p.add_argument("--min-free-gb", type=int, default=20, help="stop when the disk has less free") -p.add_argument("--check", action="store_true", - help="cargo check instead of cargo build: metadata only, no code or binaries") -p.add_argument("--no-verify-reuse", action="store_true", - help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") -p.add_argument("--target", default=None, - help="pass --target to cargo, so RUSTFLAGS skip build scripts and proc macros") -p.add_argument("--pause-on-finding", action="store_true", - help="stop all workers at the first finding (writes /PAUSED); patch rustc and " - "run again to resume") -args = p.parse_args() - -WORK = Path(args.work).resolve() -FIXTURE = Path(args.fixture).resolve() -BIN = FIXTURE.name -# Kinds not compared here: metadata is compared separately; with split debuginfo, objects and -# binaries name .dwo files by session, so two clean builds differ too. -SKIP = {"rmeta"} | ({"rlib", "exe"} if re.search(r"-Csplit-debuginfo=(packed|unpacked)", args.rustflags) else set()) - -from mutations import EDITS, int_literal, str_literal # noqa: E402 - -# ---------------------------------------------------------------- building - - -def env(): - e = dict(os.environ) - e.update(RUSTC=args.rustc, RUSTC_WRAPPER="", CARGO_INCREMENTAL="1", - RUSTFLAGS=args.rustflags, CARGO_TERM_COLOR="never") - if not args.no_verify_reuse: - e["RUSTC_VERIFY_REUSE"] = "1" - e["RUSTC_REPORT_UNTRACKED"] = "1" - return e - - -class Result: - def __init__(self, returncode, stdout, stderr): - self.returncode, self.stdout, self.stderr = returncode, stdout, stderr - - -def build(src, target): - """Build; a build running longer than --timeout is killed and reported as a hang - if a rustc was still running (a looping build script is the fixture's problem).""" - t = time.time() - proc = subprocess.Popen( - ["cargo", f"+{args.toolchain}", "check" if args.check else "build", "--workspace", "--offline", "-j", "4", - "--target-dir", str(target), "--message-format=json-render-diagnostics", - *(["--target", args.target] if args.target else [])], - cwd=src, env=env(), stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, - start_new_session=True) - hang = False - try: - out, err = proc.communicate(timeout=args.timeout) - except subprocess.TimeoutExpired: - group = subprocess.run(["ps", "-o", "args=", "-g", str(proc.pid)], capture_output=True, text=True).stdout - hang = any(line.split()[0].endswith("/rustc") for line in group.splitlines() if line.strip()) - os.killpg(proc.pid, 9) - out, err = proc.communicate() - err += f"\nkilled after {args.timeout}s; still running:\n{group}" - r = Result(proc.returncode if not hang else 1, out, err) - rmetas, exe = {}, None - for line in r.stdout.splitlines(): - try: - msg = json.loads(line) - except ValueError: - continue - if msg.get("reason") != "compiler-artifact": - continue - for f in msg.get("filenames", []): - if f.endswith(".rmeta"): - rmetas[str(Path(f).relative_to(target))] = Path(f).read_bytes() - elif f.endswith(".so") and "proc-macro" in msg["target"]["kind"]: - # A proc macro's metadata is in the .rustc section of its shared library. - section = subprocess.run(["objcopy", "--dump-section", ".rustc=/dev/stdout", f, "/dev/null"], - capture_output=True).stdout - rmetas[str(Path(f).relative_to(target)) + ":.rustc"] = section - if msg.get("executable") and msg["target"]["name"] == BIN: - exe = msg["executable"] - log = r.stderr - return { - "ok": r.returncode == 0, - "ice": "internal compiler error" in log or "the compiler unexpectedly panicked" in log, - "hang": hang, - "reuse": reuse_checks(log), - "untracked": untracked_reads(log), - "secs": time.time() - t, "log": log, "rmetas": rmetas, "exe": exe, - "art": artifacts.collect(r.stdout, target), - } - - -def untracked_reads(log): - """Reads of untracked state the compiler reported (docs/hunt/report-untracked.patch).""" - return {line.strip() for line in log.splitlines() if line.startswith("rustc-untracked-read:")} - - -def note_untracked(lines): - """Adds new reports of untracked reads to /untracked.txt, which lists each once.""" - path = WORK / "untracked.txt" - known = set(path.read_text().splitlines()) if path.exists() else set() - new = sorted(set(lines) - known) - if new: - with path.open("a") as f: - f.write("".join(line + "\n" for line in new)) - - -def known(kind, detail): - """Reports seen since the reuse check exists and judged benign (docs/shadow-mode.md): reused - codegen units differing only in debuginfo at the end of the file, and constant allocations - shared differently between evaluations in different typing modes.""" - return kind == "verify-reuse" and all( - d == "codegen unit" or d.startswith("allocation sharing eval_to_const_value_raw") for d in detail) - - -def reuse_checks(log): - """What the compiler's own check of reused results (docs/hunt/verify-reuse.patch) found - stale: `query `, `metadata`, `codegen unit` or `allocation sharing `, - once each.""" - found = set() - for line in log.splitlines(): - if line.startswith("rustc-verify-reuse: query `"): - found.add("query " + line.split("`")[1]) - elif line.startswith("rustc-verify-reuse: metadata"): - found.add("metadata") - elif line.startswith("rustc-verify-reuse: codegen unit"): - found.add("codegen unit") - elif line.startswith("rustc-verify-reuse: allocation shared differently"): - # Named by the two queries and typing modes, so that a new pattern is kept apart - # from a known one. - queries = re.findall(r"query `(\w+)`", line) - modes = re.findall(r"TypingModeEqWrapper\((\w+)\)", line) - found.add("allocation sharing " + " / ".join(queries + sorted(set(modes)))) - return sorted(found) - - -def run_exe(exe): - if not exe: - return None - try: - r = subprocess.run([exe], capture_output=True, text=True, timeout=30) - return (r.returncode, r.stdout[-2000:]) - except subprocess.TimeoutExpired: - return ("timeout", "") - - -def files(root): - return {str(p.relative_to(root)): p.read_text() for p in root.rglob("*.rs") if "target" not in p.parts} - - -# ---------------------------------------------------------------- a worker - - -def worker(k): - rng = random.Random(args.seed * 1000 + k) - home = WORK / f"w{k}" - src, target, inc_target = home / "src", home / "target", home / "target-inc" - findings = WORK / "findings" - findings.mkdir(parents=True, exist_ok=True) - stats = {"edits": 0, "built": 0, "failed": 0, "compared": 0, "findings": {}, "secs": 0.0, - "by_edit": {}} - kept = {} - history = [] - previous = {} - - def reset(): - if home.exists(): - shutil.rmtree(home) - home.mkdir(parents=True) - shutil.copytree(FIXTURE, src, ignore=shutil.ignore_patterns("target", "edits", "edit")) - history.clear() - previous.clear() - b = build(src, target) - previous.update(b["rmetas"]) - if not b["ok"]: - # Raising here would hang the pool; record it and stop this worker. - (WORK / f"error-w{k}.log").write_text(b["log"][-6000:]) - return False - return True - - def report(kind, detail, inc, clean, extra=None): - stats["findings"][kind] = stats["findings"].get(kind, 0) + 1 - if args.pause_on_finding and not known(kind, detail): - (WORK / "PAUSED").write_text(json.dumps({"worker": k, "edit": stats["edits"], "kind": kind, - "detail": detail}, indent=1)) - (WORK / "STOP").touch() - key = kind + ":" + ",".join(sorted(detail)) - if kept.get(key, 0) >= args.keep: - return - kept[key] = kept.get(key, 0) + 1 - d = findings / f"w{k}-{stats['edits']:07}-{kind}" - d.mkdir(parents=True, exist_ok=True) - (d / "history.json").write_text(json.dumps(history, indent=1)) - (d / "finding.json").write_text(json.dumps({"kind": kind, "detail": detail, "extra": extra}, indent=1)) - (d / "inc.log").write_text(inc["log"][-8000:]) - for name, b in (("inc", inc), ("clean", clean)): - lines = b["log"].splitlines() if b else [] - hits = [i for i, l in enumerate(lines) if "panicked at" in l or "internal compiler error" in l - or "unexpectedly panicked" in l or "interrupted by SIG" in l] - if hits: - (d / f"{name}-ice.txt").write_text("\n".join(lines[max(0, hits[0] - 5):hits[0] + 60])) - (d / "clean.log").write_text(clean["log"][-8000:] if clean else "") - for rel in detail: - if rel in inc["rmetas"]: - (d / (Path(rel).name + ".inc")).write_bytes(inc["rmetas"][rel]) - if clean and rel in clean["rmetas"]: - (d / (Path(rel).name + ".clean")).write_bytes(clean["rmetas"][rel]) - shutil.make_archive(str(d / "src"), "gztar", src) - - if not reset(): - return stats - started = time.time() - while stats["edits"] < args.edits and not (WORK / "STOP").exists(): - if shutil.disk_usage(WORK).free < args.min_free_gb * 2**30: - print(f"worker {k}: less than {args.min_free_gb} GB free, stopping", flush=True) - break - if len([h for h in history if h["kept"]]) >= args.reset and not reset(): - break - n = stats["edits"] - stats["edits"] += 1 - paths = sorted(p for p in src.rglob("*.rs") if "target" not in p.parts) - path = rng.choice(paths) - old = path.read_text() - fn = rng.choices([e for e, _ in EDITS], weights=[w for _, w in EDITS])[0] - if fn in (str_literal, int_literal) and path.name == "build.rs": - # Cargo keeps stale OUT_DIR files, so renaming a generated file splits the builds, - # and a changed number can make the build script loop forever. - continue - new = fn(old, rng, n) - if new is None or new == old: - continue - path.write_text(new) - rel = str(path.relative_to(src)) - diff = "".join(difflib.unified_diff(old.splitlines(True), new.splitlines(True), "a/" + rel, "b/" + rel)) - inc = build(src, target) - note_untracked(inc["untracked"]) - history.append({"edit": fn.__name__, "file": rel, "diff": diff, "before": old, "after": new, "kept": inc["ok"]}) - by = stats["by_edit"].setdefault(fn.__name__, [0, 0]) - by[0] += 1 - if inc["ice"]: - report("ICE", [], inc, None) - if inc["hang"]: - report("hang", [], inc, None) - if inc["reuse"]: - lines = [l[:3000] for l in inc["log"].splitlines() if l.startswith("rustc-verify-reuse")] - report("verify-reuse", inc["reuse"], inc, None, lines[:20]) - if not inc["ok"]: - stats["failed"] += 1 - path.write_text(old) - history.append({"edit": "revert", "file": rel, "diff": "", "kept": False}) - continue - by[1] += 1 - stats["built"] += 1 - - # The clean build, at the same path. - if inc_target.exists(): - shutil.rmtree(inc_target) - target.rename(inc_target) - clean = build(src, target) - if clean["ice"]: - report("ICE", ["clean"], inc, clean) - if clean["hang"]: - report("hang", ["clean"], inc, clean) - if not clean["ok"]: - report("split", [], inc, clean) - else: - stats["compared"] += 1 - a, b = inc["rmetas"], clean["rmetas"] - differ = sorted(r for r in set(a) | set(b) if a.get(r) != b.get(r)) - others = {k: v for k, v in artifacts.compare(inc["art"], clean["art"]).items() if k not in SKIP} - if differ or others: - # Build clean once more: if two clean builds differ, the difference is - # nondeterminism (P5), not incremental reuse. - p5 = [] - for _ in range(args.p5_builds): - shutil.rmtree(target) - again = build(src, target) - c = again["rmetas"] - p5 = sorted(r for r in set(b) | set(c) if b.get(r) != c.get(r)) - p5 += [f"{k}: {v[0]}" for k, v in artifacts.compare(clean["art"], again["art"]).items() - if k not in SKIP] - if p5 or not again["ok"]: - break - if again["ok"] and p5: - report("P5", p5[:10], clean, again) - differ, others = [], {} - # Known: metadata reused unchanged from the previous session although a source - # file changed (its hash and length in the source map are stale). - stale = [r for r in differ if r in previous and previous[r] == a.get(r)] - if differ and stale: - stats["findings"]["P6-stale-reuse"] = stats["findings"].get("P6-stale-reuse", 0) + 1 - elif differ: - report("P6", differ, inc, clean) - # More oracles: object code in the rlibs, the binary, and the diagnostics. - for kind, detail in others.items(): - report(kind, detail[:10], inc, clean) - ra = run_exe(str(inc["exe"]).replace(str(target), str(inc_target)) if inc["exe"] else None) - rb = run_exe(clean["exe"]) - if ra != rb: - report("run", [], inc, clean, {"inc": ra, "clean": rb}) - shutil.rmtree(target) - inc_target.rename(target) - previous.clear() - previous.update(inc["rmetas"]) - - if stats["edits"] % 20 == 0: - stats["secs"] = time.time() - started - (WORK / f"stats-w{k}.json").write_text(json.dumps(stats)) - stats["secs"] = time.time() - started - (WORK / f"stats-w{k}.json").write_text(json.dumps(stats)) - return stats - - -if __name__ == "__main__": - WORK.mkdir(parents=True, exist_ok=True) - for f in ("STOP", "PAUSED"): - (WORK / f).unlink(missing_ok=True) - with multiprocessing.Pool(args.workers) as pool: - results = pool.map(worker, range(args.workers)) - total = {"edits": sum(r["edits"] for r in results), "built": sum(r["built"] for r in results), - "compared": sum(r["compared"] for r in results)} - print(json.dumps(total)) diff --git a/rustc/grammar-coverage.py b/rustc/grammar-coverage.py deleted file mode 100644 index 3c85ebb..0000000 --- a/rustc/grammar-coverage.py +++ /dev/null @@ -1,123 +0,0 @@ -#!/usr/bin/env python3 -"""Which parts of the Rust grammar a fixture's sources use, by Ur's grammar. - - rustc/grammar-coverage.py --ur --grammar /ecosystems/rust/language \\ - [--json out.json] [--edition 2024] - -Ur's Rust grammar (Urscal modules, `syntax Sort = Label: ... | Label: ... | Other ;`) names -each alternative of each syntax sort. `ur parse --tree` prints a file's tree with every node as -`(Sort::Label ...)` and every literal token in quotes. Two measures: - - alternatives each labeled alternative, by construct: the label within its module, with - Conditions.rsc (expressions in condition position) counted as Expressions - literals each keyword and operator the syntax rules mention, used or not - -Prints what is missing, by module, and a summary. Lexical rules, layout and keyword lists are -not syntax alternatives and are left out. -""" - -import argparse -import json -import re -import subprocess -from collections import defaultdict -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--ur", required=True) -p.add_argument("--grammar", required=True) -p.add_argument("fixture") -p.add_argument("--json") -args = p.parse_args() - - -def strip_comments(text): - return re.sub(r"//[^\n]*", "", text) - - -STRING = r'"(?:[^"\\]|\\.)*"' - - -def syntax_rules(text): - """(sort, body) for each `syntax` rule of a module: from `syntax Name ... =` to the `;` - that ends it, outside quotes and brackets.""" - for m in re.finditer(r"(?m)^syntax\s+(\w+)[^=]*=", text): - i, depth = m.end(), 0 - while i < len(text): - c = text[i] - if c == '"': - q = re.match(STRING, text[i:]) - i += q.end() if q else 1 - continue - if c in "([{": - depth += 1 - elif c in ")]}": - depth -= 1 - elif c == ";" and depth == 0: - break - i += 1 - yield m.group(1), text[m.end():i] - - -# Conditions.rsc repeats the expression sorts for condition position (no struct literals): the -# same constructs, so counted with the expressions. -FAMILY = {"Conditions": "Expressions"} - -grammar = {} # (family, label) -> [sorts] -literals = {} # literal -> family -for f in sorted(Path(args.grammar).glob("*.rsc")): - if f.stem in ("Testing", "Semantics", "Language", "Rust"): - continue - family = FAMILY.get(f.stem, f.stem) - text = strip_comments(f.read_text()) - for sort, body in syntax_rules(text): - body = re.sub(r'@\w+=' + STRING, "", body) - for lit in re.findall(STRING, body): - lit = lit[1:-1].replace("\\<", "<").replace("\\>", ">").replace('\\"', '"').replace("\\\\", "\\") - literals.setdefault(lit, family) - bare = re.sub(STRING, '""', body) - bare = re.sub(r'@\w+="[^"]*"', "", bare) - for label in re.findall(r"(? --tests /tests/ui --work [--only ] - [--known ] [--jobs 8] [--timeout 120] [--pause-on-finding] [--recheck] - -Writes /results.jsonl and /findings//. -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -FIXED = ["-Coverflow-checks=on", "-Cdebug-assertions=on"] -# Miri without preemption: threads switch only where they block, the same at every MIR level. -MIRI_FLAGS = ["-Zmiri-preemption-rate=0"] -# Tests asserting what Rust leaves unspecified, which Miri varies on purpose: function pointer -# equality, the addresses of zero-sized values, stack addresses, function alignment. -UNSPECIFIED = {"consts/const-extern-function.rs", "consts/zst_no_llvm_alloc.rs", - "layout/null-pointer-optimization.rs", "mir/mir_misc_casts.rs", "mir/mir_coercions.rs", - "extern/extern-compare-with-return-type.rs", "fn/fn-ptr-trait-run.rs", "mir/mir_raw_fat_ptr.rs", - "codegen/StackColoring-not-blowup-stack-issue-40883.rs", "attributes/fn-align-dyn.rs", - # Miri calls .init_array functions without glibc's (argc, argv, envp). - "runtime/stdout-before-main.rs"} -THREADS = re.compile(r"thread::(spawn|scope)|std::sync::mpsc|\bspawn\(") -LEVELS = {"miri0": [], "miri2": ["-Zmir-opt-level=2"], "miri4": ["-Zmir-opt-level=4"]} -OWN = re.compile(r"^-O$|opt-level|overflow-checks|debug-assertions|codegen-backend|mir-enable-passes|" - r"panic=|-Cpanic|prefer-dynamic|-Zbuild-std|-Clink|-Ctarget") -# Code Miri cannot interpret: skip without trying. -NOT_FOR_MIRI = re.compile(r"\basm!|global_asm!|naked_asm!|extern\s+\"C\"\s*\{|#\[link\(|std::process::Command|" - r"\bfork\b|libc::|dlopen|std::os::unix::process") - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--timeout", type=int, default=120) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def clean(text, path): - text = re.sub(r"(thread '[^']*') \(\d+\)", r"\1", text) - text = re.sub(r"\S*/lib/rustlib/src/rust/library/", "library/", text) - # argv[0]: the source file under Miri, the binary natively. - text = text.replace(str(path.resolve()), "") - text = re.sub(r"\S*/native/prog\b", "", text) - # The test harness: timings, and result lines in completion order. - text = re.sub(r"finished in \d+\.\d+s", "finished in …s", text) - lines = text.split("\n") - results = iter(sorted(l for l in lines if l.startswith("test ") and " ... " in l)) - return "\n".join(next(results) if (l.startswith("test ") and " ... " in l) else l for l in lines) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - status, _, binary = uitest.compile(args.rustc, path.resolve(), d / "native", flags, edition, - FIXED + ["-Copt-level=0"]) - if binary is None: - record["skip"] = f"native build {status}" - return record, [] - runs = [uitest.run(binary) for _ in range(3)] - outs = {(c, clean(o.decode(errors="replace"), path)) for c, o, _ in runs} - if len(outs) > 1: - record["skip"] = "native run is nondeterministic" - return record, [] - code, out = outs.pop() - native = {"exit": code, "stdout": out} - miri = {} - for name, extra in LEVELS.items(): - m = uitest.miri(path.resolve(), flags, edition, FIXED + MIRI_FLAGS + extra, timeout=args.timeout, cwd=d) - m["stdout"] = clean(m["stdout"], path) - miri[name] = m - if name == "miri0" and m["status"] in ("unsupported", "error", "timeout"): - record["skip"] = f"miri {m['status']}" - record["why"] = m["stderr"][-300:] - return record, [] - record["miri"] = {k: v["status"] for k, v in miri.items()} - found = [] - # UB in a program without `unsafe` code can only be the compiler's; in a test with its - # own unsafe code it is most likely the test's (noted, not a finding). - safe = "unsafe" not in path.read_text(errors="replace") - if miri["miri0"]["status"] == "ub" and safe and rel not in UNSPECIFIED: - found.append({"what": "ub (safe code)", "stderr": miri["miri0"]["stderr"][-3000:]}) - elif miri["miri0"]["status"] == "ub": - record["note"] = "ub in a test with unsafe code" - threaded = THREADS.search(path.read_text(errors="replace")) - for name in ("miri2", "miri4"): - m = miri[name] - if m["status"] == "ub" and miri["miri0"]["status"] != "ub": - found.append({"what": f"ub-opt ({name})", "stderr": m["stderr"][-3000:]}) - elif m["status"] == "ice": - found.append({"what": f"ice ({name})", "stderr": m["stderr"][-3000:]}) - elif (m["status"] == "ok" and miri["miri0"]["status"] == "ok" and not threaded - and (m["exit"], m["stdout"]) != (miri["miri0"]["exit"], miri["miri0"]["stdout"])): - found.append({"what": f"opt-differs ({name})", "miri0": miri["miri0"]["stdout"][-1500:], - "got": m["stdout"][-1500:], "exits": [miri["miri0"]["exit"], m["exit"]]}) - m0 = miri["miri0"] - # Native threads race; Miri's do not without preemption: no comparison for threaded tests. - if m0["status"] == "ok" and not threaded and rel not in UNSPECIFIED: - # Miri exits 1 on a panic that reaches main, native code 101. - exit_m = 101 if m0["exit"] == 1 and "panicked" in m0["stderr"] else m0["exit"] - if (exit_m, m0["stdout"]) != (native["exit"], native["stdout"]): - found.append({"what": "native", "miri": [m0["exit"], m0["stdout"][-1500:]], - "native": [native["exit"], native["stdout"][-1500:]]}) - if found: - outdir = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "fixed": FIXED, "found": found}, indent=1)) - record["found"] = [f["what"] for f in found] - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests( - args.tests, ("run-pass", "run-fail"), - lambda text, flags: any(OWN.search(f) for f in flags) or NOT_FOR_MIRI.search(text) - # Compile-time output (trace_macros, log_syntax) would land in Miri's stdout. - or "trace_macros" in text or "log_syntax" in text): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/mutations.py b/rustc/mutations.py deleted file mode 100644 index df6a2a6..0000000 --- a/rustc/mutations.py +++ /dev/null @@ -1,201 +0,0 @@ -"""Random mechanical edits to Rust source, shared by fuzz.py and flag-walk.py. - -Each edit takes the file's text, a random.Random and a counter, and returns the new text, or -None when it does not apply. EDITS lists them with weights. -""" - -import re - - - -def blocks(text): - """Top-level items: blank-line separated, continuation blocks merged.""" - out = [] - for b in text.split("\n\n"): - if out and (b[:1].isspace() or b.startswith("}") or b.startswith("where")): - out[-1] += "\n\n" + b - else: - out.append(b) - return out - - -def edit_lines(text, rng, f): - lines = text.split("\n") - r = f(lines, rng) - return None if r is None else "\n".join(r) - - -def comment_line(text, rng, n): - def f(lines, rng): - i = rng.randrange(len(lines) + 1) - indent = re.match(r"\s*", lines[i] if i < len(lines) else "").group(0) - return lines[:i] + [f"{indent}// fuzz {n}"] + lines[i:] - return edit_lines(text, rng, f) - - -def blank_line(text, rng, n): - return edit_lines(text, rng, lambda lines, rng: (lambda i: lines[:i] + [""] + lines[i:])(rng.randrange(len(lines) + 1))) - - -def remove_comment(text, rng, n): - def f(lines, rng): - idx = [i for i, l in enumerate(lines) if l.strip().startswith("//") and not l.strip().startswith("//!")] - if not idx: - return None - i = rng.choice(idx) - return lines[:i] + lines[i + 1:] - return edit_lines(text, rng, f) - - -def indent_line(text, rng, n): - def f(lines, rng): - idx = [i for i, l in enumerate(lines) if l.strip()] - i = rng.choice(idx) - lines[i] = " " + lines[i] - return lines - return edit_lines(text, rng, f) - - -def swap_items(text, rng, n): - b = blocks(text) - if len(b) < 3: - return None - i = rng.randrange(1, len(b) - 1) - b[i], b[i + 1] = b[i + 1], b[i] - return "\n\n".join(b) - - -def move_item_to_end(text, rng, n): - b = blocks(text) - if len(b) < 3: - return None - i = rng.randrange(1, len(b)) - item = b.pop(i) - return "\n\n".join(b + [item.rstrip("\n")]) + "\n" - - -def delete_item(text, rng, n): - b = blocks(text) - if len(b) < 3: - return None - b.pop(rng.randrange(1, len(b))) - return "\n\n".join(b) - - -def duplicate_fn(text, rng, n): - b = blocks(text) - fns = [i for i, x in enumerate(b) if re.search(r"^(pub(\([^)]*\))? )?(const )?(async )?fn \w+", x, re.M)] - if not fns: - return None - i = rng.choice(fns) - copy = re.sub(r"\bfn (\w+)", lambda m: f"fn {m.group(1)}_fuzz{n}", b[i], count=1) - b.insert(i + 1, copy) - return "\n\n".join(b) - - -ADDITIONS = [ - "fn fuzz_private_{n}() -> u32 {{ {n} }}", - "pub fn fuzz_public_{n}(x: u32) -> u32 {{ x.wrapping_mul({n}) }}", - "#[inline]\npub fn fuzz_inline_{n}(x: &T) -> (T, u32) {{ (x.clone(), {n}) }}", - "pub const FUZZ_{n}: &str = \"fuzz {n}\";", - "pub static FUZZ_STATIC_{n}: [u8; 3] = [{n} as u8, 1, 2];", - "#[derive(Debug, Clone, PartialEq)]\npub struct Fuzz{n} {{ pub items: [T; N], pub tag: &'static str }}", - "pub enum FuzzEnum{n} {{ A(u32), B {{ x: i64 }}, C }}", - "pub trait FuzzTrait{n} {{ fn go(&self) -> impl Sized; const K: u32 = {n}; }}", - "pub async fn fuzz_async_{n}() -> u32 {{ {n} }}", - "pub type FuzzAlias{n} = Vec<(T, u32)>;", - "macro_rules! fuzz_macro_{n} {{ ($e:expr) => {{ $e + {n} }}; }}", - "pub mod fuzz_mod_{n} {{ pub fn inner() -> &'static str {{ \"{n}\" }} }}", -] - - -def add_item(text, rng, n): - b = blocks(text) - i = rng.randrange(1, len(b) + 1) - b.insert(i, rng.choice(ADDITIONS).format(n=n)) - return "\n\n".join(b) - - -def int_literal(text, rng, n): - ms = [m for m in re.finditer(r"(? --tests /tests/ui --work - [--cranelift ] [--configs O3,O3-lto] [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] - -Writes /results.jsonl, and /findings// with the source, the configurations' -argv and outputs. With --pause-on-finding, stops starting new tests at the first finding and -exits 3 (the frontier loop: docs/hunt.md). --recheck runs only the tests with findings. -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -FIXED = ["-Coverflow-checks=on", "-Cdebug-assertions=on", "-Cpanic=unwind", "-Cdebuginfo=0"] -CONFIGS = { - "base": ["-Copt-level=0", "-Zmir-opt-level=0"], - "O0": ["-Copt-level=0"], - "O0-mir4": ["-Copt-level=0", "-Zmir-opt-level=4"], - "O1": ["-Copt-level=1"], - "O2": ["-Copt-level=2"], - "O3": ["-Copt-level=3"], - "Os": ["-Copt-level=s"], - "Oz": ["-Copt-level=z"], - "O3-mir4": ["-Copt-level=3", "-Zmir-opt-level=4"], - "O3-lto": ["-Copt-level=3", "-Clto=fat", "-Ccodegen-units=1"], - "O2-cgu16": ["-Copt-level=2", "-Ccodegen-units=16"], - "O3-native": ["-Copt-level=3", "-Ctarget-cpu=native"], - "cranelift": ["-Copt-level=0", "-Zcodegen-backend=cranelift"], -} -# Tests whose outcome legitimately depends on optimization: unspecified behavior (whether two -# equal promoted constants share an address), stack usage, or a backend's documented gaps. -NOISE = { - "mir/mir_raw_fat_ptr.rs": {"cranelift"}, # compares the addresses of two `&0u8` - "codegen/StackColoring-not-blowup-stack-issue-40883.rs": {"O0-mir4", "O0"}, # stack usage - "attributes/fn-align-dyn.rs": {"cranelift"}, # Cranelift ignores #[align] on functions - "backtrace/backtrace.rs": {"cranelift"}, # Cranelift backtraces lack frames -} -# Tests that choose these themselves are left out: the configuration would contradict them. -OWN = re.compile(r"^-O$|opt-level|mir-opt-level|overflow-checks|debug-assertions|codegen-backend|" - r"mir-enable-passes|^-Clto|lto=|target-cpu|panic=|-Cpanic|prefer-dynamic|-Zbuild-std") - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--cranelift", help="a rustc with the cranelift backend (the pinned nightly)") -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--configs", help="comma-separated subset (base is always built)") -p.add_argument("--only", help="tests whose path contains this") -p.add_argument("--known", help="file of test paths to leave out (known findings)") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) -(WORK / "scratch").mkdir(exist_ok=True) -configs = {k: v for k, v in CONFIGS.items() - if k == "base" or not args.configs or k in args.configs.split(",")} -if not args.cranelift: - configs.pop("cranelift", None) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def normalized(stderr): - lines = [l for l in stderr.decode(errors="replace").splitlines() - if not l.startswith(("note: run with `RUST_BACKTRACE", "note: Some details are omitted"))] - text = "\n".join(lines) - # Panic messages name the thread with its OS id: `thread 'main' (909942) panicked`. - text = re.sub(r"(thread '[^']*') \(\d+\)", r"\1", text) - # Toolchains print std's and dependencies' paths differently: in full (rust-src installed), - # remapped to /rustc//, or relative. - text = re.sub(r"\S*/lib/rustlib/src/rust/library/|/rustc/[0-9a-f]+/library/", "library/", text) - return re.sub(r"\S*/registry/(src/)?[^/\s]+/([^/\s]+-\d[^/\s]*)/", r"/\2/", text) - - -def observe(binary): - # Every configuration's program runs from the same path: some tests print argv[0]. - fixed = binary.parent.parent / "run" / "prog" - fixed.parent.mkdir(exist_ok=True) - shutil.copy2(binary, fixed) - code, out, err = uitest.run(fixed) - stdout = out.decode(errors="replace") - # The test harness prints how long tests took, and runs tests on several threads: their - # result lines come in any order. - stdout = re.sub(r"finished in \d+\.\d+s", "finished in …s", stdout) - lines = stdout.split("\n") - results = sorted(l for l in lines if l.startswith("test ") and " ... " in l) - it = iter(results) - stdout = "\n".join(next(it) if (l.startswith("test ") and " ... " in l) else l for l in lines) - return {"exit": code, "stdout": stdout, "stderr": normalized(err)} - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - built, runs = {}, {} - text = path.read_text(errors="replace") - for name, cfg in configs.items(): - # Cranelift does not unwind on this target yet (catch_unwind catches nothing). - if name == "cranelift" and ("catch_unwind" in text or "needs-unwind" in text): - continue - rustc = args.cranelift if name == "cranelift" else args.rustc - status, stderr, binary = uitest.compile(rustc, path.resolve(), d / name, flags, edition, FIXED + cfg) - built[name] = {"status": status, "stderr": stderr[-2000:]} - if binary: - runs[name] = observe(binary) - record = {"test": rel, "kind": kind, "build": {k: v["status"] for k, v in built.items()}} - if built["base"]["status"] != "ok": - record["skip"] = "baseline does not build" - return record, [] - again = observe(d / "base" / "prog") - if again != runs["base"]: - record["skip"] = "baseline is nondeterministic" - return record, [] - found = [] - for name in configs: - if name == "base" or name not in built or name in NOISE.get(rel, ()): - continue - b = built[name]["status"] - if b != "ok": - # The Cranelift backend has documented gaps (tail calls, some linkages and SIMD - # intrinsics), which it reports as errors or as panics inside itself. - if name == "cranelift" and (b == "error" or "rustc_codegen_cranelift" in built[name]["stderr"]): - continue - found.append({"config": name, "what": f"build {b}", "stderr": built[name]["stderr"]}) - continue - # Cranelift cannot unwind on this target yet: a panic aborts. - if name == "cranelift" and "failed to initiate panic" in runs[name]["stderr"]: - continue - if runs[name] != runs["base"]: - retry = observe(d / name / "prog") - if retry != runs["base"] and retry == runs[name]: - diff = [k for k in ("exit", "stdout", "stderr") if runs[name][k] != runs["base"][k]] - found.append({"config": name, "what": "run differs: " + ",".join(diff), - "base": runs["base"], "got": runs[name]}) - if found: - out = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - shutil.copy(path, out / path.name) - (out / "finding.json").write_text(json.dumps({ - "test": rel, "flags": flags, "edition": edition, "fixed": FIXED, - "configs": {f["config"]: configs[f["config"]] for f in found}, "found": found, - "base": runs["base"]}, indent=1)) - record["found"] = [f"{f['config']}: {f['what']}" for f in found] - return record, found - - -def main(): - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, ("run-pass", "run-fail"), - lambda text, flags: any(OWN.search(f) for f in flags)): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (args.recheck and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, configurations: {', '.join(configs)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/release-diff.py b/rustc/release-diff.py deleted file mode 100644 index 87d2ebd..0000000 --- a/rustc/release-diff.py +++ /dev/null @@ -1,124 +0,0 @@ -#!/usr/bin/env python3 -"""Release-to-release: code that one toolchain accepts the next must accept too, in comparable -time. - -Runs `cargo check --locked` on each repository of a corpus of real crates with two toolchains -(an older and a newer rustup toolchain, or a local rustc via a rustup-linked name), each in its -own target directory, deleted afterwards. Compared: - - regression the older toolchain checks the repository, the newer one does not (the new - error codes and the first error are recorded); noted instead when the failing crate - enables unstable features (`#![feature]`, often only when it detects a nightly) - fixed the other way round (reported, not a finding) - slower the newer toolchain takes more than --slower times as long (both succeed) - ice the newer toolchain crashes - -Dependencies are fetched first (`cargo fetch --locked`), so the timed runs are offline. - - rustc/release-diff.py --corpus --old --new - --work [--only ] [--jobs 2] [--slower 1.5] [--timeout 1800] - -Writes /results.jsonl and prints regressions, crashes and slowdowns. -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import time -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--corpus", required=True) -p.add_argument("--old", required=True) -p.add_argument("--new", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--jobs", type=int, default=2) -p.add_argument("--slower", type=float, default=1.5) -p.add_argument("--timeout", type=int, default=1800) -args = p.parse_args() -WORK = Path(args.work).resolve() -WORK.mkdir(parents=True, exist_ok=True) - - -def check(repo, toolchain): - target = WORK / "target" / f"{repo.name}-{toolchain}" - shutil.rmtree(target, ignore_errors=True) - env = dict(os.environ, CARGO_TARGET_DIR=str(target), CARGO_TERM_COLOR="never", - CARGO_INCREMENTAL="0", RUSTFLAGS="--cap-lints=warn") - env.pop("RUSTC_WRAPPER", None) - start = time.time() - try: - r = subprocess.run(["cargo", f"+{toolchain}", "check", "--locked", "--offline", "--workspace", - "--message-format=short"], cwd=repo, env=env, capture_output=True, text=True, - timeout=args.timeout) - code, err = r.returncode, r.stderr - except subprocess.TimeoutExpired: - code, err = "timeout", "" - seconds = round(time.time() - start, 1) - shutil.rmtree(target, ignore_errors=True) - ice = "internal compiler error" in err or "the compiler unexpectedly panicked" in err - codes = sorted(set(re.findall(r"error\[(E\d{4})\]", err))) - first = next((l for l in err.splitlines() if re.search(r"\berror(\[E\d+\])?:", l)), "") - return {"code": code, "seconds": seconds, "ice": ice, "codes": codes, "first": first[:300], - "tail": err[-3000:] if code != 0 else ""} - - -def uses_unstable(first_error): - """The crate a first error points into, if its source enables `#![feature(...)]`.""" - m = re.search(r"(/\S*?/registry/src/[^/]+/[^/]+|/\S+?)/src/", first_error) - if not m: - return None - root = Path(m.group(1)) - for f in list((root / "src").glob("lib.rs")) + list((root / "src").glob("main.rs")): - text = f.read_text(errors="replace") - if re.search(r"#!\[(cfg_attr\([^]]*)?feature\(", text): - return root.name - return None - - -def one(repo): - fetch = subprocess.run(["cargo", f"+{args.new}", "fetch", "--locked"], cwd=repo, capture_output=True, - text=True, timeout=1800) - if fetch.returncode != 0: - return {"repo": repo.name, "skip": "fetch failed", "why": fetch.stderr[-400:]} - old = check(repo, args.old) - new = check(repo, args.new) - rec = {"repo": repo.name, "old": old, "new": new, "found": []} - if old["code"] == 0 and new["code"] != 0: - unstable = uses_unstable(new["first"]) - if unstable and not new["ice"]: - # A crate that turns on unstable features when it detects a nightly compiler breaks - # when they change: expected between nightlies, noted rather than reported. - rec["notes"] = [f"regression in a crate using unstable features ({unstable})"] - else: - rec["found"].append("ice" if new["ice"] else "regression") - elif old["code"] != 0 and new["code"] == 0: - rec["notes"] = ["fixed"] - elif old["code"] == 0 and new["code"] == 0 and old["seconds"] > 5 and new["seconds"] > args.slower * old["seconds"]: - rec["found"].append(f"slower: {old['seconds']}s -> {new['seconds']}s") - if new["ice"] and "ice" not in rec["found"]: - rec["found"].append("ice") - return rec - - -def main(): - repos = sorted(p for p in Path(args.corpus).iterdir() if (p / "Cargo.toml").exists() - and (not args.only or args.only in p.name)) - print(f"{len(repos)} repositories, {args.old} -> {args.new}", flush=True) - with ThreadPoolExecutor(args.jobs) as ex, (WORK / "results.jsonl").open("a") as out: - for rec in ex.map(one, repos): - out.write(json.dumps(rec) + "\n") - out.flush() - tag = rec.get("skip") or ", ".join(rec.get("found", [])) or "same" - old, new = rec.get("old", {}), rec.get("new", {}) - print(f"{rec['repo']:45} {tag:30} old {old.get('code')} {old.get('seconds')}s " - f"new {new.get('code')} {new.get('seconds')}s {new.get('first', '')[:80]}", flush=True) - - -main() diff --git a/rustc/replay.py b/rustc/replay.py deleted file mode 100755 index 2188b9a..0000000 --- a/rustc/replay.py +++ /dev/null @@ -1,219 +0,0 @@ -#!/usr/bin/env python3 -"""Replay a crate's git history through incremental compilation. - -For each first-parent commit, oldest first, the workspace is checked out and -built incrementally on top of the previous commit's build, then built again -from scratch, and the two are compared: - - P6 every .rmeta Cargo reports for a workspace member is identical - rlib every rlib's members are identical, object code included - diag both builds printed the same diagnostics - reuse the compiler's own check of what it reused (RUSTC_VERIFY_REUSE, - docs/hunt/verify-reuse.patch) found nothing stale - ICE neither build crashed the compiler - split both builds succeed or both fail - -Registry dependencies are not compiled incrementally by Cargo, so the clean -build starts from a copy of the incremental target directory with the -workspace members and the incremental cache removed, and only the members are -built again. Both builds use the same target directory path, since Cargo -derives a crate's identity from paths. - - rustc/replay.py --rustc --repo --work [--commits 2000] - -Writes /results.jsonl, one line per commit, and keeps the logs of every -problem and both .rmeta files of the first few differences per crate in -/findings. -""" - -import argparse -import json -import os -import shutil -import subprocess -import sys -import time -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).resolve().parent)) -import artifacts # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--repo", required=True) -p.add_argument("--work", required=True) -p.add_argument("--commits", type=int, default=2000) -p.add_argument("--toolchain", default="nightly-2026-10-06", help="for cargo") -p.add_argument("--jobs", default="4") -p.add_argument("--keep", type=int, default=3, help="differences kept per crate") -p.add_argument("--from", dest="start", type=int, default=0, help="first commit index to replay") -p.add_argument("--to", dest="end", type=int, default=None, help="last commit index to replay") -p.add_argument("--rustflags", default="", help="more flags for every build, such as -Copt-level=2") -p.add_argument("--no-verify-reuse", action="store_true", - help="do not set RUSTC_VERIFY_REUSE (needs a compiler with docs/hunt/verify-reuse.patch)") -args = p.parse_args() - -work = Path(args.work).resolve() -src, target, inc_target = work / "src", work / "target", work / "target-inc" -findings = work / "findings" -work.mkdir(parents=True, exist_ok=True) -findings.mkdir(exist_ok=True) -if not src.exists(): - subprocess.run(["git", "clone", "-q", args.repo, str(src)], check=True) - -env = dict(os.environ) -env.update( - RUSTC=args.rustc, - RUSTC_WRAPPER="", - CARGO_INCREMENTAL="1", - # A commit that denies warnings would otherwise stop building with a newer compiler. - RUSTFLAGS=f"--cap-lints=warn {args.rustflags}".strip(), - CARGO_TERM_COLOR="never", -) -if not args.no_verify_reuse: - env["RUSTC_VERIFY_REUSE"] = "1" - env["RUSTC_REPORT_UNTRACKED"] = "1" -cargo = ["cargo", f"+{args.toolchain}"] - - -def run(cmd, **kw): - return subprocess.run(cmd, cwd=src, env=env, capture_output=True, text=True, **kw) - - -def members(): - """Every package built from a path: workspace members and path dependencies.""" - r = run(cargo + ["metadata", "--format-version", "1"]) - if r.returncode != 0: - r = run(cargo + ["metadata", "--no-deps", "--format-version", "1"]) - if r.returncode != 0: - return [] - return sorted({pkg["name"] for pkg in json.loads(r.stdout)["packages"] if pkg.get("source") is None}) - - -def build(target_dir): - """Build, and collect the .rmeta files Cargo reports for workspace members.""" - t = time.time() - r = run(cargo + ["build", "--lib", "-j", args.jobs, "--target-dir", str(target_dir), - "--message-format=json-render-diagnostics"]) - rmetas, fresh = {}, [] - for line in r.stdout.splitlines(): - try: - msg = json.loads(line) - except ValueError: - continue - if msg.get("reason") != "compiler-artifact" or "path+file" not in msg.get("package_id", ""): - continue - for f in msg.get("filenames", []): - if f.endswith(".rmeta"): - rmetas[str(Path(f).relative_to(target_dir))] = Path(f).read_bytes() - if msg.get("fresh"): - fresh.append(msg["target"]["name"]) - log = r.stderr - return { - "ok": r.returncode == 0, - "ice": "internal compiler error" in log or "the compiler unexpectedly panicked" in log, - "secs": round(time.time() - t, 1), - "log": log[-6000:], - "reuse": [line[:3000] for line in log.splitlines() if line.startswith("rustc-verify-reuse:")], - "untracked": {line for line in log.splitlines() if line.startswith("rustc-untracked-read:")}, - "rmetas": rmetas, - "fresh": fresh, - "art": artifacts.collect(r.stdout, target_dir), - } - - -def keep_dir(i, commit): - d = findings / f"{i:05}-{commit[:10]}" - d.mkdir(exist_ok=True) - return d - - -commits = run(["git", "rev-list", "--first-parent", "--reverse", f"--max-count={args.commits}", "origin/HEAD"]).stdout.split() -if not commits: - commits = run(["git", "rev-list", "--first-parent", "--reverse", f"--max-count={args.commits}", "HEAD"]).stdout.split() -results = work / "results.jsonl" -done = {json.loads(line)["commit"] for line in results.open()} if results.exists() else set() -kept = {} - -for i, commit in enumerate(commits): - if commit in done or i < args.start or (args.end is not None and i > args.end): - continue - run(["git", "checkout", "-q", "--force", commit]) - run(["git", "clean", "-fdxq"]) - date = run(["git", "log", "-1", "--format=%cs", commit]).stdout.strip() - names = members() - inc = build(target) - # Reads of untracked state, each listed once in /untracked.txt. - known = set((work / "untracked.txt").read_text().splitlines()) if (work / "untracked.txt").exists() else set() - with (work / "untracked.txt").open("a") as f: - f.write("".join(line + "\n" for line in sorted(inc["untracked"] - known))) - - # The clean build: the same target directory path, starting from a copy with the - # workspace members and the incremental cache removed. - if inc_target.exists(): - shutil.rmtree(inc_target) - if target.exists(): - target.rename(inc_target) - shutil.copytree(inc_target, target, symlinks=True) - else: - inc_target.mkdir() - # Cargo refuses to clean a directory it did not create; this one may have been - # created here when the first build failed early. - target.mkdir(exist_ok=True) - tag = target / "CACHEDIR.TAG" - if not tag.exists(): - tag.write_text("Signature: 8a477f597d28d172789f06886806bc55\n") - for name in names: - run(cargo + ["clean", "-p", name, "--target-dir", str(target)]) - for d in target.glob("*/incremental"): - shutil.rmtree(d) - clean = build(target) - - problems, differ = [], [] - if inc["ice"] or clean["ice"]: - problems.append("ICE") - if inc["ok"] != clean["ok"]: - problems.append("split") - if inc["reuse"]: - # The compiler's own check found something it reused stale. - problems.append("reuse") - (keep_dir(i, commit) / "reuse.txt").write_text("\n".join(inc["reuse"])) - if clean["fresh"]: - # A member the clean build did not compile again would be compared with itself. - problems.append("stale") - if inc["ok"] and clean["ok"] and not clean["fresh"]: - a, b = inc["rmetas"], clean["rmetas"] - differ = sorted(rel for rel in set(a) | set(b) if a.get(rel) != b.get(rel)) - # More oracles: object code in the rlibs and the diagnostics. - for kind, detail in artifacts.compare(inc["art"], clean["art"]).items(): - if kind in ("rlib", "diag"): - problems.append(kind) - (keep_dir(i, commit) / f"{kind}.txt").write_text("\n".join(detail)) - if differ: - problems.append("P6") - for rel in differ: - crate = Path(rel).name.split("-")[0] - if kept.get(crate, 0) < args.keep: - kept[crate] = kept.get(crate, 0) + 1 - d = keep_dir(i, commit) - (d / f"{Path(rel).name}.inc").write_bytes(a.get(rel, b"")) - (d / f"{Path(rel).name}.clean").write_bytes(b.get(rel, b"")) - if problems: - d = keep_dir(i, commit) - (d / "inc.log").write_text(inc["log"]) - (d / "clean.log").write_text(clean["log"]) - - # Continue incrementally from the incremental build. - shutil.rmtree(target, ignore_errors=True) - inc_target.rename(target) - record = { - "i": i, "commit": commit, "date": date, "members": names, - "inc": {k: inc[k] for k in ("ok", "ice", "secs")}, - "clean": {k: clean[k] for k in ("ok", "ice", "secs")}, - "compared": len(inc["rmetas"]), "differ": differ, "problems": problems, - } - with results.open("a") as f: - f.write(json.dumps(record) + "\n") - status = "both ok" if inc["ok"] and clean["ok"] else f"inc {'ok' if inc['ok'] else 'failed'}, clean {'ok' if clean['ok'] else 'failed'}" - print(f"{i:5} {date} {commit[:10]} {status}, {len(inc['rmetas'])} compared, " - f"{inc['secs']}s/{clean['secs']}s {' '.join(problems)}", flush=True) diff --git a/rustc/rewrite-diff.py b/rustc/rewrite-diff.py deleted file mode 100644 index f87ad5a..0000000 --- a/rustc/rewrite-diff.py +++ /dev/null @@ -1,176 +0,0 @@ -#!/usr/bin/env python3 -"""Equivalent rewrites: rewriting a program into an equivalent one must not change its verdict. - -Each standalone UI test is printed back unchanged (`mirth-rewrite identity`: the baseline, since -printing drops comments and moves lines) and rewritten by each of mirth-rewrite's rewrites -(generic-wrap, alias, reorder, unused). Each version is compiled the way the test's headers say -(metadata for check tests, a full build for build and run tests), and compared with the -baseline: - - verdict accepted against rejected, or a crash on one side only (a finding) - codes both rejected with different sets of error codes (a finding for reorder and unused, - which change nothing a diagnostic could depend on; noted for the others) - -A test whose baseline differs from the original file's verdict is left out (the printer cannot -represent it faithfully). - - rustc/rewrite-diff.py --rustc --tests /tests/ui --work - [--rewrites generic-wrap,alias] [--only ] [--known ] [--jobs 8] - [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import subprocess -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -REWRITER = Path(__file__).resolve().parent.parent / "target/release/mirth-rewrite" -REWRITES = ["generic-wrap", "alias", "reorder", "unused"] -# Error codes may differ legitimately for every rewrite (which error suppresses which depends on -# order): only a changed verdict is a finding. -STRICT = set() -# Differences that are resource limits or legitimate requirements of a generic context. -NOISE = { - ("consts/chained-constants-stackoverflow.rs", "reorder"), # 10,000 chained consts: query depth - ("consts/interior-mut-const-via-union.rs", "generic-wrap"), # finding 25 (docs/hunt.md) - # recursion_limit = "6": evaluation order nests the query stack one level deeper - ("traits/next-solver/overflow/dont-lower-depth-for-witness-and-rigid-opaque.rs", "reorder"), - ("imports/ambiguous-9.rs", "reorder"), # finding 28 - ("imports/ambiguous-14.rs", "reorder"), # finding 28 - ("imports/overwrite-different-ambig-2.rs", "reorder"), # finding 28 -} -NOT_MOVABLE = re.compile(r"^\s*(pub(\([^)]*\))?\s+)?mod\s+\w+\s*;|include(_str|_bytes)?!|#\[path|#!\[no_core\]", re.M) -# Item order matters to textual macro scoping: no reordering where macros are defined. -ORDER_MATTERS = re.compile(r"macro_rules!|#\[macro_use\]|macro\s+\w+") -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--rewrites", help="comma-separated subset") -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -rewrites = args.rewrites.split(",") if args.rewrites else REWRITES -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def codes(stderr): - return sorted(set(re.findall(r"error\[(E\d{4})\]", stderr))) - - -def verdict(source, flags, edition, kind, out, has_main): - # A full build whenever there is a program: generic-wrap moves errors to monomorphization. - emit = "link" if has_main or kind not in ("check-pass", "check-fail", None) else "metadata" - # Lints capped: a rewrite may add or move a warning, and lint levels are not the subject. - status, stderr, _ = uitest.compile(args.rustc, source, out, flags, edition, ["--cap-lints=warn"], - timeout=120, emit=emit) - return {"status": status, "codes": codes(stderr), "stderr": stderr[-2500:]} - - -def rewrite(name, source, target): - r = subprocess.run([str(REWRITER), name, str(source)], capture_output=True, text=True, timeout=60) - if r.returncode != 0: - return False - target.write_text(r.stdout) - return True - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - text = path.read_text(errors="replace") - # The rewritten file is compiled elsewhere: files it names by relative path are not there. - # Without `core`, a new trait or generic parameter does not compile. - if NOT_MOVABLE.search(text): - record["skip"] = "uses files by path or has no core" - return record, [] - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - base_src = d / "identity.rs" - if not rewrite("identity", path, base_src): - record["skip"] = "does not parse" - return record, [] - has_main = "fn main" in text - original = verdict(path.resolve(), flags, edition, kind, d / "original", has_main) - base = verdict(base_src, flags, edition, kind, d / "identity", has_main) - if (original["status"], original["codes"]) != (base["status"], base["codes"]): - record["skip"] = "printing changes the verdict" - return record, [] - record["base"] = base["status"] - found, notes, applied = [], [], [] - for name in rewrites: - if name == "reorder" and ORDER_MATTERS.search(text): - continue - if (rel, name) in NOISE: - continue - # generic_const_exprs requires `where` bounds in generic contexts that a concrete one - # does not. - if name == "generic-wrap" and "generic_const_exprs" in text: - continue - src = d / f"{name}.rs" - if not rewrite(name, path, src): - continue - applied.append(name) - v = verdict(src, flags, edition, kind, d / name, has_main) - if v["status"] != base["status"] and "timeout" not in (v["status"], base["status"]): - found.append({"rewrite": name, "what": f"verdict: {base['status']} -> {v['status']}", - "base_codes": base["codes"], "codes": v["codes"], "stderr": v["stderr"], - "base_stderr": base["stderr"]}) - elif v["status"] == base["status"] == "error" and v["codes"] != base["codes"]: - entry = {"rewrite": name, "what": f"codes: {base['codes']} -> {v['codes']}", - "stderr": v["stderr"], "base_stderr": base["stderr"]} - (found if name in STRICT else notes).append(entry) - if any(f["rewrite"] == name for f in found + notes): - shutil.copy(src, d / f"keep-{name}.rs") - record["applied"] = applied - record["found"] = [f"{f['rewrite']}: {f['what']}" for f in found] - record["notes"] = [f"{f['rewrite']}: {f['what']}" for f in notes] - if found or notes: - outdir = WORK / ("findings" if found else "notes") / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - shutil.copy(base_src, outdir / "identity.rs") - for f in found + notes: - shutil.copy(d / f"keep-{f['rewrite']}.rs", outdir / f"{f['rewrite']}.rs") - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "kind": kind, "found": found, "notes": notes}, - indent=1)) - return record, found - - -def main(): - global REWRITER - if not REWRITER.exists(): - sys.exit("build mirth-rewrite first: cargo build --release -p mirth-rewrite") - # A private copy: rebuilding mirth-rewrite must not change a sweep halfway. - shutil.copy2(REWRITER, WORK / "mirth-rewrite") - REWRITER = WORK / "mirth-rewrite" - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, rewrites: {', '.join(rewrites)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/solver-diff.py b/rustc/solver-diff.py deleted file mode 100644 index 0fec9da..0000000 --- a/rustc/solver-diff.py +++ /dev/null @@ -1,127 +0,0 @@ -#!/usr/bin/env python3 -"""Solver differential: the trait solvers and the borrow checkers must agree. - -Compiles each standalone UI test four ways: the old trait solver (`-Znext-solver=coherence`, -what compiletest pins), nightly's default (the new solver everywhere), the old solver with -Polonius (`-Zpolonius=next`), and the new solver with Polonius. Compared with the old solver -and NLL: - - ice a configuration crashes where the reference does not - verdict accepted by one and rejected by the other - codes both reject, with different sets of error codes (reported, not a finding: the - solvers word errors differently) - -A program accepted only by a non-reference configuration and runnable (it has `fn main`) is -interpreted with Miri under that configuration's flags: undefined behavior there means the -other configuration accepted something unsound (the Polonius soundness bugs had that shape). - -Tests that name a solver or Polonius in their headers (they test the difference on purpose) are -left out. - - rustc/solver-diff.py --rustc --tests /tests/ui --work [--only ] - [--known ] [--jobs 8] [--pause-on-finding] [--recheck] -""" - -import argparse -import json -import re -import shutil -import sys -import tempfile -from pathlib import Path - -sys.path.insert(0, str(Path(__file__).parent)) -import uitest # noqa: E402 - -CONFIGS = { - "old": ["-Znext-solver=coherence"], - "next": [], - "old-polonius": ["-Znext-solver=coherence", "-Zpolonius=next"], - "next-polonius": ["-Zpolonius=next"], -} -REFERENCE = "old" -KINDS = ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail", None) - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--work", required=True) -p.add_argument("--only") -p.add_argument("--known") -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--pause-on-finding", action="store_true") -p.add_argument("--recheck", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -(WORK / "scratch").mkdir(parents=True, exist_ok=True) -known = set(Path(args.known).read_text().split()) if args.known else set() - - -def codes(stderr): - return sorted(set(re.findall(r"error\[(E\d{4})\]", stderr))) - - -def one(path, flags, edition, kind): - rel = str(path.relative_to(args.tests)) - record = {"test": rel, "kind": kind} - results = {} - with tempfile.TemporaryDirectory(dir=WORK / "scratch") as d: - d = Path(d) - for name, cfg in CONFIGS.items(): - # Metadata is enough for check tests (type checking and borrow checking run for it); - # build and run tests get a full build, which reaches monomorphization-time errors. - emit = "metadata" if kind in ("check-pass", "check-fail", None) else "link" - status, stderr, _ = uitest.compile(args.rustc, path.resolve(), d / name, flags, edition, cfg, - timeout=120, emit=emit) - results[name] = {"status": status, "codes": codes(stderr), "stderr": stderr[-2500:]} - record["status"] = {k: v["status"] for k, v in results.items()} - ref = results[REFERENCE] - found, notes = [], [] - for name, r in results.items(): - if name == REFERENCE: - continue - if r["status"] == "ice" and ref["status"] != "ice": - found.append({"config": name, "what": "ice", "stderr": r["stderr"]}) - elif r["status"] == "timeout" and ref["status"] != "timeout": - found.append({"config": name, "what": "timeout"}) - elif {r["status"], ref["status"]} == {"ok", "error"}: - f = {"config": name, "what": f"verdict: {REFERENCE} {ref['status']}, {name} {r['status']}", - "ref_codes": ref["codes"], "codes": r["codes"], - "stderr": (r if r["status"] == "error" else ref)["stderr"]} - accepted_by = name if r["status"] == "ok" else REFERENCE - if "fn main" in path.read_text(errors="replace"): - m = uitest.miri(path.resolve(), flags, edition, CONFIGS[accepted_by], timeout=120, cwd=WORK / "scratch") - f["miri"] = {"config": accepted_by, "status": m["status"], "stderr": m["stderr"][-1500:]} - if m["status"] == "ub": - f["what"] += f"; Miri: UB under {accepted_by}" - found.append(f) - elif r["status"] == ref["status"] == "error" and r["codes"] != ref["codes"]: - notes.append(f"{name} codes {r['codes']} vs {ref['codes']}") - record["found"] = [f"{f['config']}: {f['what']}" for f in found] - record["notes"] = notes - if found: - outdir = WORK / "findings" / rel.replace("/", "__") - shutil.rmtree(outdir, ignore_errors=True) - outdir.mkdir(parents=True) - shutil.copy(path, outdir / path.name) - (outdir / "finding.json").write_text(json.dumps( - {"test": rel, "flags": flags, "edition": edition, "configs": CONFIGS, "found": found}, indent=1)) - return record, found - - -def main(): - wanted = None - if args.recheck: - wanted = {json.loads((f / "finding.json").read_text())["test"] for f in (WORK / "findings").glob("*")} - skip = lambda text, flags: re.search(r"next-solver|polonius|^//@\s*revisions:.*\bnext\b", text, re.M) - todo = [] - for path, flags, edition, kind in uitest.tests(args.tests, KINDS, skip): - rel = str(path.relative_to(args.tests)) - if rel in known or (args.only and args.only not in rel) or (wanted is not None and rel not in wanted): - continue - todo.append((path, flags, edition, kind)) - print(f"{len(todo)} tests, configurations: {', '.join(CONFIGS)}", flush=True) - sys.exit(uitest.drive(todo, one, WORK / "results.jsonl", args.jobs, args.pause_on_finding)) - - -main() diff --git a/rustc/ui-coverage.py b/rustc/ui-coverage.py deleted file mode 100644 index 92b4a61..0000000 --- a/rustc/ui-coverage.py +++ /dev/null @@ -1,168 +0,0 @@ -#!/usr/bin/env python3 -"""Which compiler functions each of rustc's UI tests reaches that a baseline (the fixture's -builds) does not, with a coverage-instrumented rustc (rustc/coverage.toml); then a small set -of tests that reaches the most of them. - - rustc/ui-coverage.py run --rustc --tests /tests/ui - --sites /mirth-sites --baseline [--baseline ...] --out [--jobs 6] - rustc/ui-coverage.py pick --out [--count 200] - -`run` compiles each test file the way its `//@` headers say, as far as one rustc call can: -`compile-flags`, `edition`, the first of `revisions` (as `--cfg` with its own flags), metadata -only for tests that do not build (check-pass, and tests expected to fail before codegen), a -full build otherwise. Tests that need auxiliary crates, proc macros, another target or -`minicore` are skipped. Writes /tests.jsonl: per test, whether it compiled and the -indices (into /functions.json) of the functions it reached beyond the baseline. - -`pick` chooses tests greedily, each adding the most functions not yet reached, and writes -/picked.json. -""" - -import argparse -import json -import os -import re -import shutil -import subprocess -import sys -import tempfile -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -sub = p.add_subparsers(dest="cmd", required=True) -r = sub.add_parser("run") -r.add_argument("--rustc", required=True) -r.add_argument("--tests", required=True) -r.add_argument("--sites", required=True) -r.add_argument("--baseline", action="append", default=[]) -r.add_argument("--out", required=True) -r.add_argument("--jobs", type=int, default=6) -r.add_argument("--limit", type=int, default=0) -k = sub.add_parser("pick") -k.add_argument("--out", required=True) -k.add_argument("--count", type=int, default=200) -args = p.parse_args() -OUT = Path(args.out) - -SKIP = re.compile(r"^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|" - r"only-(?!x86_64|linux|unix|64bit)|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|" - r"needs-asm-support|known-bug)", re.M) -NO_BUILD = ("check-pass", "check-fail") - - -def hits(directory): - out = set() - for log in Path(directory).glob("*.log"): - for line in log.read_text(errors="replace").splitlines(): - if line.startswith("V\t"): - out.add(line[2:]) - return out - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def run_one(path): - text = path.read_text(errors="replace") - rel = str(path.relative_to(args.tests)) - if SKIP.search(text): - return {"test": rel, "status": "skipped"} - flags, edition, kind = headers(text) - with tempfile.TemporaryDirectory(dir=OUT / "scratch") as d: - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link" - argv = [args.rustc, str(path), "--edition", edition or "2015", emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", *flags] - env = dict(os.environ, MIRTH_OUT=d + "/logs", RUSTC_BOOTSTRAP="1") - try: - done = subprocess.run(argv, capture_output=True, text=True, timeout=120, env=env, cwd=d) - ice = ("internal compiler error" in done.stderr or "the compiler unexpectedly panicked" in done.stderr - or "rustc interrupted by SIG" in done.stderr) - status = "ok" if done.returncode == 0 else ("ice" if ice else "error") - first = next((l for l in done.stderr.splitlines() if l.startswith("error")), "")[:160] - except subprocess.TimeoutExpired: - status, first = "timeout", "" - new = sorted(index[s] for s in hits(d + "/logs") - baseline if s in index) - return {"test": rel, "status": status, "kind": kind, "error": first, "new": new} - - -if args.cmd == "run": - functions = [] - for table in Path(args.sites).glob("*.sites"): - for line in table.read_text(errors="replace").splitlines(): - f = line.split("\t") - if len(f) >= 7 and f[1] == "cover": - functions.append((f[0], f[4], f[6])) - functions.sort() - index = {site: i for i, (site, _, _) in enumerate(functions)} - baseline = set() - for b in args.baseline: - for sub_dir in [Path(b), *Path(b).glob("*")]: - if sub_dir.is_dir(): - baseline |= hits(sub_dir) - OUT.mkdir(parents=True, exist_ok=True) - (OUT / "scratch").mkdir(exist_ok=True) - (OUT / "functions.json").write_text(json.dumps( - {"functions": [[path, span] for _, path, span in functions], - "baseline": sorted(index[s] for s in baseline if s in index)})) - done = set() - results = OUT / "tests.jsonl" - if results.exists(): - done = {json.loads(l)["test"] for l in results.read_text().splitlines()} - tests = sorted(p for p in Path(args.tests).rglob("*.rs") - if "auxiliary" not in p.parts and str(p.relative_to(args.tests)) not in done) - if args.limit: - tests = tests[:args.limit] - print(f"{len(functions)} functions, {len(baseline & index.keys())} in the baseline; " - f"{len(tests)} tests to run", flush=True) - with ThreadPoolExecutor(args.jobs) as ex, results.open("a") as out: - for n, res in enumerate(ex.map(run_one, tests)): - out.write(json.dumps(res) + "\n") - if n % 500 == 0: - out.flush() - print(f"{n} tests", flush=True) - shutil.rmtree(OUT / "scratch", ignore_errors=True) - -if args.cmd == "pick": - info = json.loads((OUT / "functions.json").read_text()) - tests = [json.loads(l) for l in (OUT / "tests.jsonl").read_text().splitlines()] - tests = [t for t in tests if t.get("new")] - reached = set() - for t in tests: - reached |= set(t["new"]) - covered, picked = set(), [] - sets = {t["test"]: set(t["new"]) for t in tests} - status = {t["test"]: t["status"] for t in tests} - while len(picked) < args.count and sets: - best = max(sets, key=lambda name: len(sets[name] - covered)) - gain = sets[best] - covered - if not gain: - break - covered |= gain - picked.append({"test": best, "status": status[best], "adds": len(gain), "total": len(covered)}) - del sets[best] - total = len(info["functions"]) - base = len(info["baseline"]) - print(f"baseline {base} of {total} functions ({100 * base / total:.1f}%); all tests reach " - f"{len(reached)} more; {len(picked)} picked tests reach {len(covered)} more " - f"({100 * (base + len(covered)) / total:.1f}% in all)") - for t in picked[:40]: - print(f" +{t['adds']:5} {t['total']:6} {t['status']:7} {t['test']}") - (OUT / "picked.json").write_text(json.dumps(picked, indent=1)) diff --git a/rustc/ui-fuzz.py b/rustc/ui-fuzz.py deleted file mode 100644 index 66ac3de..0000000 --- a/rustc/ui-fuzz.py +++ /dev/null @@ -1,215 +0,0 @@ -#!/usr/bin/env python3 -"""Incremental rebuilds of rustc's UI tests, most of which fail to compile on purpose: the -fuzzer's edits applied to each test file, each rebuild compared with a clean build of the same -source, so that error reporting and recovery are exercised under incremental compilation. - - rustc/ui-fuzz.py --rustc --tests /tests/ui --list picked.json --work - [--edits 20] [--jobs 8] [--pause-on-finding] - -For each test in the list (rustc/ui-coverage.py pick), compiled the way its `//@` headers say: -build it incrementally, then repeatedly apply a random edit (mutations.py) and rebuild it -incrementally, build the edited file again with a fresh incremental directory (same file, same -working directory), and compare: - - status both succeed, both fail, or both crash - diag the diagnostics, with paths and the incremental directory taken out - output the .rmeta and .rlib (normalized as artifacts.py does) when both succeed - ice both crash or neither does - -A clean build is made again before a difference counts (nondeterminism). Findings go to -/findings/-/ with the source, both outputs and the edit history. -""" - -import argparse -import difflib -import hashlib -import json -import os -import random -import re -import shutil -import subprocess -import sys -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -here = Path(__file__).parent -sys.path.insert(0, str(here)) -import artifacts # noqa: E402 -import mutations # noqa: E402 - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--list", required=True) -p.add_argument("--work", required=True) -p.add_argument("--edits", type=int, default=20) -p.add_argument("--jobs", type=int, default=8) -p.add_argument("--flags", default="", help="extra rustc options for every build") -p.add_argument("--pause-on-finding", action="store_true") -args = p.parse_args() -WORK = Path(args.work).resolve() -NO_BUILD = ("check-pass", "check-fail") -# Tests that hit bugs already in docs/hunt.md every time: finding 17. -KNOWN = ("unleash-the-miri-inside-of-you",) - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def build(directory, source, flags, edition, kind, incremental): - """Compile `source` (a file in `directory`) with outputs in `directory/out`.""" - out = directory / "out" - shutil.rmtree(out, ignore_errors=True) - out.mkdir(parents=True) - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link,metadata" - argv = [args.rustc, source.name, "--edition", edition or "2015", emit, "--out-dir", "out", - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=short"] - # Before the test's own flags, which may end with an option expecting a value. - if incremental: - argv.append(f"-Cincremental={incremental}") - argv += [*flags, *args.flags.split()] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=300, cwd=directory, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - code, err = r.returncode, r.stderr - except subprocess.TimeoutExpired: - code, err = -1, "timeout" - ice = ("internal compiler error" in err or "the compiler unexpectedly panicked" in err - or "rustc interrupted by SIG" in err) - if incremental: - err = err.replace(incremental, "") - diag = sorted(set(re.sub(r"\(\d+\)", "(…)", l) for l in err.splitlines() - if l and not l.startswith(("note: ", " ", "query stack", "#")))) - files = {} - for f in sorted(out.iterdir()): - if f.suffix == ".rmeta": - files[f.name] = hashlib.sha256(f.read_bytes()).hexdigest() - elif f.suffix == ".rlib": - files[f.name] = artifacts.normalized_rlib(f) - return {"code": code, "ice": ice, "diag": diag, "files": files, "stderr": err} - - -def compare(inc, clean): - found = [] - # Some tests crash the compiler on purpose; only a crash on one side counts. - if inc["ice"] != clean["ice"]: - found.append("ICE " + ("incremental" if inc["ice"] else "clean") + " only") - if (inc["code"] == 0) != (clean["code"] == 0): - found.append(f"status: incremental {inc['code']}, clean {clean['code']}") - message = lambda line: line.split(": error", 1)[-1].split(": warning", 1)[-1] - summary = lambda line: line.startswith("error: aborting") or line.startswith("warning:") and "emitted" in line - fewer = (inc["code"] != 0 and clean["code"] != 0 - and {d for d in inc["diag"] if not summary(d)} <= set(clean["diag"]) - and all(message(d) in {message(e) for e in inc["diag"]} or summary(d) - for d in clean["diag"] if d not in inc["diag"])) - if inc["diag"] != clean["diag"] and fewer: - found.append("known diag (finding 18): the rebuild stopped at a fatal error sooner") - elif inc["diag"] != clean["diag"]: - only_inc = [d for d in inc["diag"] if d not in clean["diag"]][:3] - only_clean = [d for d in clean["diag"] if d not in inc["diag"]][:3] - found.append(f"diag: incremental only {only_inc}; clean only {only_clean}") - if inc["code"] == 0 and clean["code"] == 0 and inc["files"] != clean["files"]: - found.append("output: " + ", ".join(k for k in set(inc["files"]) | set(clean["files"]) - if inc["files"].get(k) != clean["files"].get(k))) - return found - - -def fuzz(test): - try: - return fuzz_one(test) - except Exception as error: # a broken test or harness case must not stop the run - return test, f"harness error: {error!r}"[:300] - - -def fuzz_one(test): - if (WORK / "PAUSED").exists(): - return test, "not run" - path = Path(args.tests) / test - text = path.read_text(errors="replace") - flags, edition, kind = headers(text) - if any(k in text for k in KNOWN): - return test, "skipped (known)" - name = re.sub(r"\W", "_", test) + "-" + hashlib.sha256(test.encode()).hexdigest()[:8] - home = WORK / "w" / name - shutil.rmtree(home, ignore_errors=True) - # The clean build uses the same directory and file, with a fresh incremental directory, so - # that nothing but incremental state tells the two builds apart. - inc_dir = clean_dir = home / "src" - inc_dir.mkdir(parents=True) - src_inc = src_clean = inc_dir / path.name - src_inc.write_text(text) - rng = random.Random(test) - history = [] - build(inc_dir, src_inc, flags, edition, kind, str(home / "incr")) - found_any = 0 - for n in range(args.edits): - old = src_inc.read_text() - fn = rng.choices([e for e, _ in mutations.EDITS], weights=[w for _, w in mutations.EDITS])[0] - new = fn(old, rng, n) - if new is None or new == old: - continue - src_inc.write_text(new) - history.append({"edit": fn.__name__, "diff": "".join(difflib.unified_diff( - old.splitlines(True), new.splitlines(True), "a", "b"))}) - inc = build(inc_dir, src_inc, flags, edition, kind, str(home / "incr")) - shutil.rmtree(home / "incr-clean", ignore_errors=True) - clean = build(clean_dir, src_clean, flags, edition, kind, str(home / "incr-clean")) - found = compare(inc, clean) - if found and not any(f.startswith("ICE") for f in found): - shutil.rmtree(home / "incr-clean", ignore_errors=True) - again = build(clean_dir, src_clean, flags, edition, kind, str(home / "incr-clean")) - if compare(clean, again): - found = ["P5 " + f for f in found] - if found: - found_any += 1 - d = WORK / "findings" / f"{name}-{n}" - d.mkdir(parents=True, exist_ok=True) - (d / path.name).write_text(new) - (d / "finding.json").write_text(json.dumps({"test": test, "found": found, "flags": flags, - "edition": edition, "kind": kind, - "history": history}, indent=1)) - (d / "inc.stderr").write_text(inc["stderr"]) - (d / "clean.stderr").write_text(clean["stderr"]) - if args.pause_on_finding and not all(f.startswith(("P5", "known")) for f in found): - (WORK / "PAUSED").write_text(json.dumps({"test": test, "edit": n, "found": found}, indent=1)) - break - shutil.rmtree(home, ignore_errors=True) - return test, f"{len(history)} edits, {found_any} findings" - - -WORK.mkdir(parents=True, exist_ok=True) -(WORK / "PAUSED").unlink(missing_ok=True) -# A compiler without its standard library fails every test the same way: stop instead. -(WORK / "probe").mkdir(exist_ok=True) -(WORK / "probe" / "probe.rs").write_text("fn main() {}\n") -probe = build(WORK / "probe", WORK / "probe" / "probe.rs", [], "2021", "build-pass", None) -if probe["code"] != 0: - sys.exit(f"{args.rustc} cannot build an empty program:\n{probe['stderr']}") -picked = json.loads(Path(args.list).read_text()) -tests = [t["test"] if isinstance(t, dict) else t for t in picked] -done_path = WORK / "done.txt" -done = set(done_path.read_text().split()) if done_path.exists() else set() -with ThreadPoolExecutor(args.jobs) as ex, done_path.open("a") as log: - for test, result in ex.map(fuzz, [t for t in tests if t not in done]): - print(f"{test}: {result}", flush=True) - if not result.startswith("not run") and not (WORK / "PAUSED").exists(): - log.write(test + "\n") - log.flush() diff --git a/rustc/ui-solver-diff.py b/rustc/ui-solver-diff.py deleted file mode 100644 index d4ee965..0000000 --- a/rustc/ui-solver-diff.py +++ /dev/null @@ -1,102 +0,0 @@ -#!/usr/bin/env python3 -"""rustc's UI tests under nightly's default trait solver and under the one the test suite pins. - -Nightly builds use the new trait solver everywhere by default (compiler-team MCP #1014, -rust-lang/rust#160895), but compiletest passes `-Znext-solver=coherence` to every UI test, so -the suite checks the old behaviour only. This compiles each test both ways, the way its `//@` -headers say (as rustc/ui-coverage.py does), and lists the tests whose outcome differs: an ICE, -success against failure, or a different first error. - - rustc/ui-solver-diff.py --rustc --tests /tests/ui --out [--jobs 8] - -Writes /results.jsonl and prints the differences, ICEs first. -""" - -import argparse -import json -import os -import re -import subprocess -import tempfile -from collections import Counter -from concurrent.futures import ThreadPoolExecutor -from pathlib import Path - -p = argparse.ArgumentParser() -p.add_argument("--rustc", required=True) -p.add_argument("--tests", required=True) -p.add_argument("--out", required=True) -p.add_argument("--jobs", type=int, default=8) -args = p.parse_args() -OUT = Path(args.out) - -SKIP = re.compile(r"^//@\s*(aux-build|aux-crate|aux-bin|proc-macro|add-minicore|needs-llvm-components|" - r"only-(?!x86_64|linux|unix|64bit)|ignore-x86_64|ignore-linux|needs-sanitizer|needs-profiler|" - r"known-bug)|-Znext-solver", re.M) -NO_BUILD = ("check-pass", "check-fail") - - -def headers(text): - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind - - -def compile(path, flags, edition, kind, solver): - with tempfile.TemporaryDirectory(dir=OUT / "scratch") as d: - emit = "--emit=metadata" if kind in NO_BUILD or kind is None else "--emit=link" - argv = [args.rustc, str(path), "--edition", edition or "2015", emit, "--out-dir", d, - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", *solver, *flags] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=120, cwd=d, - env=dict(os.environ, RUSTC_BOOTSTRAP="1")) - except subprocess.TimeoutExpired: - return {"status": "timeout", "error": ""} - ice = ("internal compiler error" in r.stderr or "the compiler unexpectedly panicked" in r.stderr - or "rustc interrupted by SIG" in r.stderr) - first = next((l for l in r.stderr.splitlines() if l.startswith("error")), "") - first = re.sub(r"/\S+|`[^`]*`", "…", first)[:160] - return {"status": "ok" if r.returncode == 0 else ("ice" if ice else "error"), "error": first} - - -def one(path): - text = path.read_text(errors="replace") - rel = str(path.relative_to(args.tests)) - if SKIP.search(text): - return {"test": rel, "skipped": True} - flags, edition, kind = headers(text) - default = compile(path, flags, edition, kind, []) - pinned = compile(path, flags, edition, kind, ["-Znext-solver=coherence"]) - return {"test": rel, "kind": kind, "default": default, "pinned": pinned} - - -OUT.mkdir(parents=True, exist_ok=True) -(OUT / "scratch").mkdir(exist_ok=True) -tests = sorted(p for p in Path(args.tests).rglob("*.rs") if "auxiliary" not in p.parts) -results = [] -with ThreadPoolExecutor(args.jobs) as ex, (OUT / "results.jsonl").open("w") as out: - for res in ex.map(one, tests): - out.write(json.dumps(res) + "\n") - results.append(res) - -ran = [r for r in results if not r.get("skipped")] -differ = [r for r in ran if r["default"]["status"] != r["pinned"]["status"] - or r["default"]["error"] != r["pinned"]["error"]] -print(f"{len(ran)} tests compiled both ways; {len(differ)} differ") -print(Counter((r["pinned"]["status"], r["default"]["status"]) for r in differ).most_common()) -for r in sorted(differ, key=lambda r: (r["default"]["status"] != "ice", r["test"])): - print(f"{r['pinned']['status']:>7} -> {r['default']['status']:<7} {r['test']} {r['default']['error'][:100]}") diff --git a/rustc/uitest.py b/rustc/uitest.py deleted file mode 100644 index 8ac1156..0000000 --- a/rustc/uitest.py +++ /dev/null @@ -1,161 +0,0 @@ -"""What the oracle scripts need from rustc's UI tests: their `//@` headers, which of them can be -compiled on their own on this host, and a way to build and run one. - -The header reading is the same as in ui-fuzz.py, ui-coverage.py and ui-solver-diff.py (the first -revision of a test with revisions). -""" - -import os -import re -import subprocess -from pathlib import Path - -# Tests that need more than one file, another target, or a tool this host may lack. -NOT_STANDALONE = re.compile( - r"^//@\s*(aux-build|aux-crate|aux-bin|aux-codegen-backend|proc-macro|add-minicore|" - r"needs-llvm-components|needs-sanitizer|needs-profiler|needs-rust-lld|needs-enzyme|" - r"only-(?!x86_64|linux|unix|64bit|elf|gnu)|ignore-x86_64|ignore-linux|ignore-unix|ignore-64bit|" - r"known-bug|rustc-env|unset-rustc-env)", re.M) - - -def headers(text): - """(flags, edition, kind, revision) of a test, for its first revision.""" - flags, edition, revision, kind = [], None, None, None - revs = re.search(r"^//@\s*revisions:\s*(.*)$", text, re.M) - if revs: - revision = revs.group(1).split()[0] - for m in re.finditer(r"^//@(?:\[([\w,-]+)\])?\s*([a-z-]+)(?::\s*(.*))?$", text, re.M): - only, key, value = m.group(1), m.group(2), (m.group(3) or "").strip() - if only and (revision is None or revision not in only.split(",")): - continue - if key == "compile-flags": - flags += value.split() - elif key == "edition": - edition = value.split()[0] - elif key in ("check-pass", "build-pass", "run-pass", "check-fail", "build-fail", "run-fail"): - kind = key - if revision: - flags += ["--cfg", revision] - return flags, edition, kind, revision - - -def tests(root, kinds, extra_skip=None): - """The standalone tests under `root` of the given kinds, as (path, flags, edition, kind).""" - root = Path(root) - for path in sorted(root.rglob("*.rs")): - if "auxiliary" in path.parts: - continue - text = path.read_text(errors="replace") - if NOT_STANDALONE.search(text): - continue - flags, edition, kind, _ = headers(text) - if kind not in kinds: - continue - if extra_skip and extra_skip(text, flags): - continue - yield path, flags, edition, kind - - -def is_ice(stderr): - return ("internal compiler error" in stderr or "the compiler unexpectedly panicked" in stderr - or "rustc interrupted by SIG" in stderr) - - -def compile(rustc, source, out, flags, edition, extra=(), timeout=300, emit="link"): - """Build `source` into the directory `out`; returns (status, stderr, binary). - status: ok, error, ice, timeout.""" - out.mkdir(parents=True, exist_ok=True) - binary = out / "prog" - argv = [str(rustc), str(source), "--edition", edition or "2015", f"--emit={emit}", "-o", str(binary), - "-Zunstable-options", "-Ainternal_features", "-Aincomplete_features", "--error-format=short", - *flags, *extra] - try: - r = subprocess.run(argv, capture_output=True, text=True, timeout=timeout, cwd=out, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return "timeout", "", None - if is_ice(r.stderr): - return "ice", r.stderr, None - return ("ok" if r.returncode == 0 else "error"), r.stderr, binary if r.returncode == 0 else None - - -def run(binary, timeout=20): - """Run a built program; (exit, stdout, stderr), exit a number, or 'signal N' or 'timeout'.""" - try: - r = subprocess.run([str(binary)], capture_output=True, timeout=timeout, cwd=binary.parent, - stdin=subprocess.DEVNULL, env=dict(os.environ, RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return "timeout", b"", b"" - code = r.returncode if r.returncode >= 0 else f"signal {-r.returncode}" - return code, r.stdout, r.stderr - - -MIRI_TOOLCHAIN = Path.home() / ".rustup/toolchains/nightly-2026-10-06-x86_64-unknown-linux-gnu" -MIRI_SYSROOT = Path.home() / ".cache/miri" - - -def miri(source, flags, edition, extra=(), timeout=120, cwd=None): - """Interpret `source` with Miri (the pinned nightly's, sysroot from `cargo miri setup`). - Returns {status, exit, stdout, stderr}; status: ok (ran to the end, any exit code), ub, - unsupported, error (did not compile), timeout.""" - argv = [str(MIRI_TOOLCHAIN / "bin/miri"), "--sysroot", str(MIRI_SYSROOT), str(source), - "--edition", edition or "2015", "-Zunstable-options", "-Ainternal_features", - "-Aincomplete_features", "-Zmiri-disable-isolation", "-Zmiri-deterministic-floats", *flags, *extra] - try: - r = subprocess.run(argv, capture_output=True, timeout=timeout, cwd=cwd, stdin=subprocess.DEVNULL, - env=dict(os.environ, RUSTC_BOOTSTRAP="1", RUST_BACKTRACE="0")) - except subprocess.TimeoutExpired: - return {"status": "timeout", "exit": None, "stdout": "", "stderr": ""} - err = r.stderr.decode(errors="replace") - if "Undefined Behavior:" in err: - status = "ub" - elif "unsupported operation" in err or "can't call foreign function" in err: - status = "unsupported" - elif is_ice(err): - status = "ice" - elif re.search(r"^error(\[E\d+\])?: ", err, re.M) and r.returncode == 1 and "panicked" not in err: - status = "error" - else: - status = "ok" - return {"status": status, "exit": r.returncode, "stdout": r.stdout.decode(errors="replace"), "stderr": err} - - -def drive(todo, one, results, jobs, pause_on_finding): - """Run `one(*item)` for each item on `jobs` threads; each returns (record, findings). Records - go to the `results` file as JSON lines. With `pause_on_finding`, stops starting new items at - the first finding (the frontier loop) and returns 3; otherwise 0.""" - import json - import sys - from concurrent.futures import ThreadPoolExecutor, FIRST_COMPLETED, wait - it = iter(todo) - findings = done = 0 - stop = False - with ThreadPoolExecutor(jobs) as ex, open(results, "a") as out: - pending = set() - - def submit(): - nxt = None if stop else next(it, None) - if nxt is not None: - pending.add(ex.submit(one, *nxt)) - for _ in range(jobs * 2): - submit() - while pending: - finished, _ = wait(pending, return_when=FIRST_COMPLETED) - for fut in finished: - pending.discard(fut) - try: - record, found = fut.result() - except Exception as error: # a harness bug must not stop the run - record, found = {"test": "?", "harness": repr(error)[:300]}, [] - out.write(json.dumps(record) + "\n") - out.flush() - done += 1 - if found: - findings += 1 - print(f"FINDING {record.get('test')}: {record.get('found')}", flush=True) - stop = stop or pause_on_finding - if done % 100 == 0: - print(f"{done}/{len(todo)} done, {findings} with findings", flush=True) - submit() - print(f"{done} tests, {findings} with findings", flush=True) - return 3 if findings and pause_on_finding else 0 diff --git a/rustc/xlink-probe/Cargo.lock b/rustc/xlink-probe/Cargo.lock new file mode 100644 index 0000000..30e7b5f --- /dev/null +++ b/rustc/xlink-probe/Cargo.lock @@ -0,0 +1,7 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "probe" +version = "0.1.0" diff --git a/rustc/xlink-probe/Cargo.toml b/rustc/xlink-probe/Cargo.toml new file mode 100644 index 0000000..61d6a70 --- /dev/null +++ b/rustc/xlink-probe/Cargo.toml @@ -0,0 +1,10 @@ +[package] +name = "probe" +version = "0.1.0" +edition = "2021" +[profile.dev] +panic = "abort" +[profile.release] +panic = "abort" + +[workspace] diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt new file mode 100644 index 0000000..a72e1e2 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_52_34-1073611.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f74e2e5eb95 - ::create + 1: 0x7f74e2e5eae5 - ::force_capture + 2: 0x7f74e1b98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f74e2e74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f74e2e526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f74e2e4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f74e2e5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f74df8aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f74dfe60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f74e4d15f4e - ::set_output_kind + 10: 0x7f74e4d32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f74e4a3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f74e4a3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f74e4732f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f74e4731eb2 - ::link + 15: 0x7f74e48ebec6 - ::link + 16: 0x7f74e48e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f74e491cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f74e491caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f74e4917d9b - ::new::thread_start + 20: 0x7f74dd889b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f74dd916ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt new file mode 100644 index 0000000..2305d9b --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_54_55-1317662.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fd12bc5eb95 - ::create + 1: 0x7fd12bc5eae5 - ::force_capture + 2: 0x7fd12a998fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fd12bc74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fd12bc526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fd12bc4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fd12bc5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fd1286aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fd128c60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fd12db15f4e - ::set_output_kind + 10: 0x7fd12db32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fd12d83f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fd12d83240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fd12d532f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fd12d531eb2 - ::link + 15: 0x7fd12d6ebec6 - ::link + 16: 0x7fd12d6e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fd12d71cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fd12d71caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fd12d717d9b - ::new::thread_start + 20: 0x7fd126689b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fd126716ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt new file mode 100644 index 0000000..f8961a3 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_57_29-1334708.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fdf2945eb95 - ::create + 1: 0x7fdf2945eae5 - ::force_capture + 2: 0x7fdf28198fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fdf29474fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fdf294526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fdf2944ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fdf2945424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fdf25eaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fdf26460be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fdf2b315f4e - ::set_output_kind + 10: 0x7fdf2b332a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fdf2b03f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fdf2b03240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fdf2ad32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fdf2ad31eb2 - ::link + 15: 0x7fdf2aeebec6 - ::link + 16: 0x7fdf2aee5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fdf2af1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fdf2af1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fdf2af17d9b - ::new::thread_start + 20: 0x7fdf23e89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fdf23f16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt new file mode 100644 index 0000000..fe99e3f --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_46-1343102.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f0a5e05eb95 - ::create + 1: 0x7f0a5e05eae5 - ::force_capture + 2: 0x7f0a5cd98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f0a5e074fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f0a5e0526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f0a5e04ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f0a5e05424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f0a5aaaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f0a5b060be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f0a5ff15f4e - ::set_output_kind + 10: 0x7f0a5ff32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f0a5fc3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f0a5fc3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f0a5f932f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f0a5f931eb2 - ::link + 15: 0x7f0a5faebec6 - ::link + 16: 0x7f0a5fae5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f0a5fb1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f0a5fb1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f0a5fb17d9b - ::new::thread_start + 20: 0x7f0a58a89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f0a58b16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt new file mode 100644 index 0000000..3cef0de --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T06_59_59-1343572.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f3c0305eb95 - ::create + 1: 0x7f3c0305eae5 - ::force_capture + 2: 0x7f3c01d98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f3c03074fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f3c030526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f3c0304ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f3c0305424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f3bffaaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f3c00060be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f3c04f15f4e - ::set_output_kind + 10: 0x7f3c04f32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f3c04c3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f3c04c3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f3c04932f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f3c04931eb2 - ::link + 15: 0x7f3c04aebec6 - ::link + 16: 0x7f3c04ae5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f3c04b1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f3c04b1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f3c04b17d9b - ::new::thread_start + 20: 0x7f3bfda89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f3bfdb16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt new file mode 100644 index 0000000..ccbb9f7 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_16-1344795.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fb2b765eb95 - ::create + 1: 0x7fb2b765eae5 - ::force_capture + 2: 0x7fb2b6398fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fb2b7674fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fb2b76526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fb2b764ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fb2b765424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fb2b40aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fb2b4660be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fb2b9515f4e - ::set_output_kind + 10: 0x7fb2b9532a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fb2b923f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fb2b923240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fb2b8f32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fb2b8f31eb2 - ::link + 15: 0x7fb2b90ebec6 - ::link + 16: 0x7fb2b90e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fb2b911cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fb2b911caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fb2b9117d9b - ::new::thread_start + 20: 0x7fb2b2089b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fb2b2116ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt new file mode 100644 index 0000000..eb8596b --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_00_31-1345593.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7fb84445eb95 - ::create + 1: 0x7fb84445eae5 - ::force_capture + 2: 0x7fb843198fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7fb844474fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7fb8444526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7fb84444ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7fb84445424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7fb840eaa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7fb841460be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7fb846315f4e - ::set_output_kind + 10: 0x7fb846332a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7fb84603f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7fb84603240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7fb845d32f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7fb845d31eb2 - ::link + 15: 0x7fb845eebec6 - ::link + 16: 0x7fb845ee5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7fb845f1cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7fb845f1caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7fb845f17d9b - ::new::thread_start + 20: 0x7fb83ee89b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7fb83ef16ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt new file mode 100644 index 0000000..d206804 --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_02_04-1382261.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f6800c5eb95 - ::create + 1: 0x7f6800c5eae5 - ::force_capture + 2: 0x7f67ff998fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f6800c74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f6800c526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f6800c4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f6800c5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f67fd6aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f67fdc60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f6802b15f4e - ::set_output_kind + 10: 0x7f6802b32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f680283f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f680283240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f6802532f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f6802531eb2 - ::link + 15: 0x7f68026ebec6 - ::link + 16: 0x7f68026e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f680271cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f680271caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f6802717d9b - ::new::thread_start + 20: 0x7f67fb689b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f67fb716ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt b/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt new file mode 100644 index 0000000..bc9fd8d --- /dev/null +++ b/rustc/xlink-probe/rustc-ice-2026-10-10T07_10_33-1720154.txt @@ -0,0 +1,32 @@ +thread 'rustc' panicked at /rustc-dev/ea137335b78829b4514bf1b4c16302f74fab8581/compiler/rustc_codegen_ssa/src/back/linker.rs:225:5: +assertion failed: l.is_cc() +stack backtrace: + 0: 0x7f60ede5eb95 - ::create + 1: 0x7f60ede5eae5 - ::force_capture + 2: 0x7f60ecb98fb2 - std[f097bc1422754316]::panicking::update_hook::>::{closure#0} + 3: 0x7f60ede74fb2 - std[f097bc1422754316]::panicking::panic_with_hook + 4: 0x7f60ede526a4 - std[f097bc1422754316]::panicking::panic_handler::{closure#0} + 5: 0x7f60ede4ae59 - std[f097bc1422754316]::sys::backtrace::__rust_end_short_backtrace:: + 6: 0x7f60ede5424d - __rustc[e66452c6972d9e87]::rust_begin_unwind + 7: 0x7f60ea8aa8ec - core[fd4243e10de856d2]::panicking::panic_fmt + 8: 0x7f60eae60be2 - core[fd4243e10de856d2]::panicking::panic + 9: 0x7f60efd15f4e - ::set_output_kind + 10: 0x7f60efd32a23 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::add_order_independent_options + 11: 0x7f60efa3f039 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::linker_with_args + 12: 0x7f60efa3240f - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_natively + 13: 0x7f60ef732f26 - rustc_codegen_ssa[8f359b88cd05c65d]::back::link::link_binary + 14: 0x7f60ef731eb2 - ::link + 15: 0x7f60ef8ebec6 - ::link + 16: 0x7f60ef8e5751 - rustc_interface[1a497e1e07f914a9]::interface::run_compiler::<(), rustc_driver_impl[538812ad21a3ed]::compiler_entrypoint::{closure#0}>::{closure#2} + 17: 0x7f60ef91cd42 - std[f097bc1422754316]::sys::backtrace::__rust_begin_short_backtrace::::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()> + 18: 0x7f60ef91caed - ::{closure#2}, ()>::{closure#0}, ()>::{closure#0}::{closure#0}, ()>::{closure#1} as core[fd4243e10de856d2]::ops::function::FnOnce<()>>::call_once::{shim:vtable#0} + 19: 0x7f60ef917d9b - ::new::thread_start + 20: 0x7f60e8889b84 - start_thread + at ./nptl/pthread_create.c:447:8 + 21: 0x7f60e8916ecc - clone3 + at ./misc/../sysdeps/unix/sysv/linux/x86_64/clone3.S:78:0 + 22: 0x0 - + + +rustc version: 1.101.0-nightly (ea137335b 2026-10-05) +platform: x86_64-unknown-linux-gnu \ No newline at end of file diff --git a/rustc/xlink-probe/src/main.rs b/rustc/xlink-probe/src/main.rs new file mode 100644 index 0000000..298d0d4 --- /dev/null +++ b/rustc/xlink-probe/src/main.rs @@ -0,0 +1,39 @@ +#![no_std] +#![no_main] +#![feature(core_float_math)] +use core::fmt::Write; +use core::hint::black_box; + +struct Buf { b: [u8; 256], n: usize } +impl Write for Buf { + fn write_str(&mut self, s: &str) -> core::fmt::Result { + for &c in s.as_bytes() { if self.n < 256 { self.b[self.n] = c; self.n += 1; } } + Ok(()) + } +} + +#[unsafe(no_mangle)] +pub extern "C" fn probe_entry() -> u32 { + let a: u128 = black_box(0x1234_5678_9abc_def0_1122_3344_5566_7788); + let b: u128 = black_box(12345); + let i: i128 = black_box(-987654321987654321i128); + let f: f64 = black_box(1.5e10); + let g: f32 = black_box(2.5); + let mut buf = Buf { b: [0; 256], n: 0 }; + let _ = write!(buf, "{} {} {} {:.3} {:e} {}", a / b, a % b, i / 7, f, g as f64, (f as i128) as f32); + let big = black_box([7u8; 4096]); + let mut dst = [0u8; 4096]; + dst.copy_from_slice(&big); + let m = core::f64::math::mul_add(f, black_box(1.25), black_box(3.0)) + core::f64::math::sqrt(f) + core::f64::math::floor(f) + core::f32::math::mul_add(g, g, g) as f64; + let x = (m as u64) + black_box(f) as u64 + black_box(g) as u64 + (i as f64) as u64; + #[cfg(target_has_atomic = "32")] + { + use core::sync::atomic::{AtomicU32, Ordering}; + static C: AtomicU32 = AtomicU32::new(0); + C.fetch_add(1, Ordering::SeqCst); + } + (buf.n as u32) ^ (dst[100] as u32) ^ (x as u32) +} + +#[panic_handler] +fn panic(_: &core::panic::PanicInfo) -> ! { loop {} }