Skip to main content

rustc_codegen_llvm/back/
write.rs

1use std::ffi::{CStr, CString};
2use std::io::{self, Write};
3use std::path::{Path, PathBuf};
4use std::sync::Arc;
5use std::{fs, slice, str};
6
7use libc::{c_char, c_int, c_void, size_t};
8use rustc_codegen_ssa::back::link::ensure_removed;
9use rustc_codegen_ssa::back::versioned_llvm_target;
10use rustc_codegen_ssa::back::write::{
11    BitcodeSection, CodegenContext, EmitObj, InlineAsmError, ModuleConfig, SharedEmitter,
12    TargetMachineFactoryConfig, TargetMachineFactoryFn,
13};
14use rustc_codegen_ssa::base::wants_wasm_eh;
15use rustc_codegen_ssa::common::TypeKind;
16use rustc_codegen_ssa::traits::*;
17use rustc_codegen_ssa::{CompiledModule, ModuleCodegen, ModuleKind};
18use rustc_data_structures::profiling::SelfProfilerRef;
19use rustc_data_structures::small_c_str::SmallCStr;
20use rustc_errors::{DiagCtxt, DiagCtxtHandle, Level};
21use rustc_fs_util::{link_or_copy, path_to_c_string};
22use rustc_middle::ty::TyCtxt;
23use rustc_session::Session;
24use rustc_session::config::{self, Lto, OutputType, Passes, SplitDwarfKind, SwitchWithOptPath};
25use rustc_span::{BytePos, InnerSpan, Pos, RemapPathScopeComponents, SpanData, SyntaxContext};
26use rustc_target::spec::{CodeModel, FloatAbi, RelocModel, SanitizerSet, SplitDebuginfo, TlsModel};
27use tracing::{debug, trace};
28
29use crate::back::lto::{Buffer, ModuleBuffer};
30use crate::back::owned_target_machine::OwnedTargetMachine;
31use crate::back::profiling::{
32    LlvmSelfProfiler, selfprofile_after_pass_callback, selfprofile_before_pass_callback,
33};
34use crate::builder::SBuilder;
35use crate::builder::gpu_offload::scalar_width;
36use crate::common::AsCCharPtr;
37use crate::diagnostics::{
38    CopyBitcode, FromLlvmDiag, FromLlvmOptimizationDiag, LlvmError, ParseTargetMachineConfig,
39    UnsupportedCompression, WithLlvmError, WriteBytecode,
40};
41use crate::llvm::diagnostic::OptimizationDiagnosticKind::*;
42use crate::llvm::{self, DiagnosticInfo};
43use crate::type_::llvm_type_ptr;
44use crate::{LlvmCodegenBackend, ModuleLlvm, SimpleCx, attributes, base, common, llvm_util};
45
46pub(crate) fn llvm_err<'a>(dcx: DiagCtxtHandle<'_>, err: LlvmError<'a>) -> ! {
47    match llvm::last_error() {
48        Some(llvm_err) => dcx.emit_fatal(WithLlvmError(err, llvm_err)),
49        None => dcx.emit_fatal(err),
50    }
51}
52
53fn write_output_file<'ll>(
54    dcx: DiagCtxtHandle<'_>,
55    target: &'ll llvm::TargetMachine,
56    no_builtins: bool,
57    m: &'ll llvm::Module,
58    output: &Path,
59    dwo_output: Option<&Path>,
60    file_type: llvm::FileType,
61    self_profiler_ref: &SelfProfilerRef,
62    verify_llvm_ir: bool,
63) {
64    {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event compiler/rustc_codegen_llvm/src/back/write.rs:64",
                        "rustc_codegen_llvm::back::write", ::tracing::Level::DEBUG,
                        ::tracing_core::__macro_support::Option::Some("compiler/rustc_codegen_llvm/src/back/write.rs"),
                        ::tracing_core::__macro_support::Option::Some(64u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_codegen_llvm::back::write"),
                        ::tracing_core::field::FieldSet::new(&["message"],
                            ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::DEBUG <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::DEBUG <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&format_args!("write_output_file output={0:?} dwo_output={1:?}",
                                                    output, dwo_output) as &dyn ::tracing::field::Value))])
            });
    } else { ; }
};debug!("write_output_file output={:?} dwo_output={:?}", output, dwo_output);
65    let output_c = path_to_c_string(output);
66    let dwo_output_c;
67    let dwo_output_ptr = if let Some(dwo_output) = dwo_output {
68        dwo_output_c = path_to_c_string(dwo_output);
69        dwo_output_c.as_ptr()
70    } else {
71        std::ptr::null()
72    };
73    let result = unsafe {
74        let pm = llvm::LLVMCreatePassManager();
75        llvm::LLVMAddAnalysisPasses(target, pm);
76        llvm::LLVMRustAddLibraryInfo(target, pm, m, no_builtins);
77        llvm::LLVMRustWriteOutputFile(
78            target,
79            pm,
80            m,
81            output_c.as_ptr(),
82            dwo_output_ptr,
83            file_type,
84            verify_llvm_ir,
85        )
86    };
87
88    // Record artifact sizes for self-profiling
89    if result == llvm::LLVMRustResult::Success {
90        let artifact_kind = match file_type {
91            llvm::FileType::ObjectFile => "object_file",
92            llvm::FileType::AssemblyFile => "assembly_file",
93        };
94        record_artifact_size(self_profiler_ref, artifact_kind, output);
95        if let Some(dwo_file) = dwo_output {
96            record_artifact_size(self_profiler_ref, "dwo_file", dwo_file);
97        }
98    }
99
100    result.into_result().unwrap_or_else(|()| llvm_err(dcx, LlvmError::WriteOutput { path: output }))
101}
102
103/// If `for_cfg` is `true` then we are creating this machine for the purpose of populating
104/// [`rustc_codegen_ssa::TargetConfig`] based on what LLVM actually enables in this configuration.
105/// `-Ctarget-feature` should be ignored in that case since it is already processed separately.
106pub(crate) fn create_informational_target_machine(
107    sess: &Session,
108    for_cfg: bool,
109) -> OwnedTargetMachine {
110    let config = TargetMachineFactoryConfig { split_dwarf_file: None, output_obj_file: None };
111    // Can't use query system here quite yet because this function is invoked before the query
112    // system/tcx is set up.
113    let features = llvm_util::global_llvm_features(sess, for_cfg);
114    target_machine_factory(sess, config::OptLevel::No, &features)(sess.dcx(), config)
115}
116
117pub(crate) fn create_target_machine(tcx: TyCtxt<'_>, mod_name: &str) -> OwnedTargetMachine {
118    let split_dwarf_file = if tcx.sess.target_can_use_split_dwarf() {
119        tcx.output_filenames(()).split_dwarf_path(
120            tcx.sess.split_debuginfo(),
121            tcx.sess.opts.unstable_opts.split_dwarf_kind,
122            mod_name,
123        )
124    } else {
125        None
126    };
127
128    let output_obj_file =
129        Some(tcx.output_filenames(()).temp_path_for_cgu(OutputType::Object, mod_name));
130    let config = TargetMachineFactoryConfig { split_dwarf_file, output_obj_file };
131
132    target_machine_factory(
133        tcx.sess,
134        tcx.backend_optimization_level(()),
135        tcx.global_backend_features(()),
136    )(tcx.dcx(), config)
137}
138
139fn to_llvm_opt_settings(cfg: config::OptLevel) -> (llvm::CodeGenOptLevel, llvm::CodeGenOptSize) {
140    use self::config::OptLevel::*;
141    match cfg {
142        No => (llvm::CodeGenOptLevel::None, llvm::CodeGenOptSizeNone),
143        Less => (llvm::CodeGenOptLevel::Less, llvm::CodeGenOptSizeNone),
144        More => (llvm::CodeGenOptLevel::Default, llvm::CodeGenOptSizeNone),
145        Aggressive => (llvm::CodeGenOptLevel::Aggressive, llvm::CodeGenOptSizeNone),
146        Size => (llvm::CodeGenOptLevel::Default, llvm::CodeGenOptSizeDefault),
147        SizeMin => (llvm::CodeGenOptLevel::Default, llvm::CodeGenOptSizeAggressive),
148    }
149}
150
151fn to_pass_builder_opt_level(cfg: config::OptLevel) -> llvm::PassBuilderOptLevel {
152    use config::OptLevel::*;
153    match cfg {
154        No => llvm::PassBuilderOptLevel::O0,
155        Less => llvm::PassBuilderOptLevel::O1,
156        More => llvm::PassBuilderOptLevel::O2,
157        Aggressive => llvm::PassBuilderOptLevel::O3,
158        Size => llvm::PassBuilderOptLevel::Os,
159        SizeMin => llvm::PassBuilderOptLevel::Oz,
160    }
161}
162
163fn to_llvm_relocation_model(relocation_model: RelocModel) -> llvm::RelocModel {
164    match relocation_model {
165        RelocModel::Static => llvm::RelocModel::Static,
166        // LLVM doesn't have a PIE relocation model, it represents PIE as PIC with an extra
167        // attribute.
168        RelocModel::Pic | RelocModel::Pie => llvm::RelocModel::PIC,
169        RelocModel::DynamicNoPic => llvm::RelocModel::DynamicNoPic,
170        RelocModel::Ropi => llvm::RelocModel::ROPI,
171        RelocModel::Rwpi => llvm::RelocModel::RWPI,
172        RelocModel::RopiRwpi => llvm::RelocModel::ROPI_RWPI,
173    }
174}
175
176pub(crate) fn to_llvm_code_model(code_model: Option<CodeModel>) -> llvm::CodeModel {
177    match code_model {
178        Some(CodeModel::Tiny) => llvm::CodeModel::Tiny,
179        Some(CodeModel::Small) => llvm::CodeModel::Small,
180        Some(CodeModel::Kernel) => llvm::CodeModel::Kernel,
181        Some(CodeModel::Medium) => llvm::CodeModel::Medium,
182        Some(CodeModel::Large) => llvm::CodeModel::Large,
183        None => llvm::CodeModel::None,
184    }
185}
186
187fn to_llvm_float_abi(float_abi: Option<FloatAbi>) -> llvm::FloatAbi {
188    match float_abi {
189        None => llvm::FloatAbi::Default,
190        Some(FloatAbi::Soft) => llvm::FloatAbi::Soft,
191        Some(FloatAbi::Hard) => llvm::FloatAbi::Hard,
192    }
193}
194
195pub(crate) fn target_machine_factory(
196    sess: &Session,
197    optlvl: config::OptLevel,
198    target_features: &[String],
199) -> TargetMachineFactoryFn<LlvmCodegenBackend> {
200    // Self-profile timer for creating a _factory_.
201    let _prof_timer = sess.prof.generic_activity("target_machine_factory");
202
203    let reloc_model = to_llvm_relocation_model(sess.relocation_model());
204
205    let (opt_level, _) = to_llvm_opt_settings(optlvl);
206    let float_abi = to_llvm_float_abi(sess.target.llvm_floatabi);
207
208    let ffunction_sections =
209        sess.opts.unstable_opts.function_sections.unwrap_or(sess.target.function_sections);
210    let fdata_sections = ffunction_sections;
211    let funique_section_names = !sess.opts.unstable_opts.no_unique_section_names;
212
213    let code_model = to_llvm_code_model(sess.code_model());
214
215    // This is used to set cfg_has_threads, so all logic must be in this method.
216    let singlethread = sess.target.singlethread(&sess.internal_target_features);
217
218    let triple = SmallCStr::new(&versioned_llvm_target(sess));
219    let cpu = SmallCStr::new(llvm_util::target_cpu(sess));
220    let features = CString::new(target_features.join(",")).unwrap();
221    let abi = SmallCStr::new(sess.target.llvm_abiname.desc());
222    let trap_unreachable =
223        sess.opts.unstable_opts.trap_unreachable.unwrap_or(sess.target.trap_unreachable);
224    let emit_stack_size_section = sess.opts.unstable_opts.emit_stack_sizes;
225
226    let verbose_asm = sess.opts.unstable_opts.verbose_asm;
227    let relax_elf_relocations =
228        sess.opts.unstable_opts.relax_elf_relocations.unwrap_or(sess.target.relax_elf_relocations);
229
230    let use_init_array =
231        !sess.opts.unstable_opts.use_ctors_section.unwrap_or(sess.target.use_ctors_section);
232
233    let path_mapping = sess.source_map().path_mapping().clone();
234    let working_dir = sess.source_map().working_dir().clone();
235
236    let use_emulated_tls = #[allow(non_exhaustive_omitted_patterns)] match sess.tls_model() {
    TlsModel::Emulated => true,
    _ => false,
}matches!(sess.tls_model(), TlsModel::Emulated);
237
238    let debuginfo_compression = match sess.opts.unstable_opts.debuginfo_compression {
239        config::DebugInfoCompression::None => llvm::CompressionKind::None,
240        config::DebugInfoCompression::Zlib => {
241            if llvm::LLVMRustLLVMHasZlibCompression() {
242                llvm::CompressionKind::Zlib
243            } else {
244                sess.dcx().emit_warn(UnsupportedCompression { algorithm: "zlib" });
245                llvm::CompressionKind::None
246            }
247        }
248        config::DebugInfoCompression::Zstd => {
249            if llvm::LLVMRustLLVMHasZstdCompression() {
250                llvm::CompressionKind::Zstd
251            } else {
252                sess.dcx().emit_warn(UnsupportedCompression { algorithm: "zstd" });
253                llvm::CompressionKind::None
254            }
255        }
256    };
257
258    let use_wasm_eh = wants_wasm_eh(sess);
259
260    let large_data_threshold = sess.opts.unstable_opts.large_data_threshold.unwrap_or(0);
261
262    let prof = SelfProfilerRef::clone(&sess.prof);
263    Arc::new(move |dcx: DiagCtxtHandle<'_>, config: TargetMachineFactoryConfig| {
264        // Self-profile timer for invoking a factory to create a target machine.
265        let _prof_timer = prof.generic_activity("target_machine_factory_inner");
266
267        let path_to_cstring_helper = |path: Option<PathBuf>| -> CString {
268            let path = path.unwrap_or_default();
269            let path = path_mapping
270                .to_real_filename(&working_dir, path)
271                .path(RemapPathScopeComponents::DEBUGINFO)
272                .to_string_lossy()
273                .into_owned();
274            CString::new(path).unwrap()
275        };
276
277        let split_dwarf_file = path_to_cstring_helper(config.split_dwarf_file);
278        let output_obj_file = path_to_cstring_helper(config.output_obj_file);
279
280        OwnedTargetMachine::new(
281            &triple,
282            &cpu,
283            &features,
284            &abi,
285            code_model,
286            reloc_model,
287            opt_level,
288            float_abi,
289            ffunction_sections,
290            fdata_sections,
291            funique_section_names,
292            trap_unreachable,
293            singlethread,
294            verbose_asm,
295            emit_stack_size_section,
296            relax_elf_relocations,
297            use_init_array,
298            &split_dwarf_file,
299            &output_obj_file,
300            debuginfo_compression,
301            use_emulated_tls,
302            use_wasm_eh,
303            large_data_threshold,
304        )
305        .unwrap_or_else(|err| dcx.emit_fatal(ParseTargetMachineConfig(err)))
306    })
307}
308
309pub(crate) fn save_temp_bitcode(
310    cgcx: &CodegenContext,
311    module: &ModuleCodegen<ModuleLlvm>,
312    name: &str,
313) {
314    if !cgcx.save_temps {
315        return;
316    }
317    let ext = ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("{0}.bc", name))
    })format!("{name}.bc");
318    let path = cgcx.output_filenames.temp_path_ext_for_cgu(&ext, &module.name);
319    write_bitcode_to_file(&module.module_llvm, &path)
320}
321
322fn write_bitcode_to_file(module: &ModuleLlvm, path: &Path) {
323    unsafe {
324        let path = path_to_c_string(&path);
325        let llmod = module.llmod();
326        llvm::LLVMWriteBitcodeToFile(llmod, path.as_ptr());
327    }
328}
329
330/// In what context is a diagnostic handler being attached to a codegen unit?
331pub(crate) enum CodegenDiagnosticsStage {
332    /// Prelink optimization stage.
333    Opt,
334    /// LTO/ThinLTO postlink optimization stage.
335    LTO,
336    /// Code generation.
337    Codegen,
338}
339
340pub(crate) struct DiagnosticHandlers<'a> {
341    data: *mut (&'a CodegenContext, &'a SharedEmitter),
342    llcx: &'a llvm::Context,
343    old_handler: Option<&'a llvm::DiagnosticHandler>,
344}
345
346impl<'a> DiagnosticHandlers<'a> {
347    pub(crate) fn new(
348        cgcx: &'a CodegenContext,
349        shared_emitter: &'a SharedEmitter,
350        llcx: &'a llvm::Context,
351        module: &ModuleCodegen<ModuleLlvm>,
352        stage: CodegenDiagnosticsStage,
353    ) -> Self {
354        let remark_passes_all: bool;
355        let remark_passes: Vec<CString>;
356        match &cgcx.remark {
357            Passes::All => {
358                remark_passes_all = true;
359                remark_passes = Vec::new();
360            }
361            Passes::Some(passes) => {
362                remark_passes_all = false;
363                remark_passes =
364                    passes.iter().map(|name| CString::new(name.as_str()).unwrap()).collect();
365            }
366        };
367        let remark_passes: Vec<*const c_char> =
368            remark_passes.iter().map(|name: &CString| name.as_ptr()).collect();
369        let remark_file = cgcx
370            .remark_dir
371            .as_ref()
372            // Use the .opt.yaml file suffix, which is supported by LLVM's opt-viewer.
373            .map(|dir| {
374                let stage_suffix = match stage {
375                    CodegenDiagnosticsStage::Codegen => "codegen",
376                    CodegenDiagnosticsStage::Opt => "opt",
377                    CodegenDiagnosticsStage::LTO => "lto",
378                };
379                dir.join(::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("{0}.{1}.opt.yaml", module.name,
                stage_suffix))
    })format!("{}.{stage_suffix}.opt.yaml", module.name))
380            })
381            .and_then(|dir| dir.to_str().and_then(|p| CString::new(p).ok()));
382
383        let pgo_available = cgcx.module_config.pgo_use.is_some();
384        let data = Box::into_raw(Box::new((cgcx, shared_emitter)));
385        unsafe {
386            let old_handler = llvm::LLVMRustContextGetDiagnosticHandler(llcx);
387            llvm::LLVMRustContextConfigureDiagnosticHandler(
388                llcx,
389                diagnostic_handler,
390                data.cast(),
391                remark_passes_all,
392                remark_passes.as_ptr(),
393                remark_passes.len(),
394                // The `as_ref()` is important here, otherwise the `CString` will be dropped
395                // too soon!
396                remark_file.as_ref().map(|dir| dir.as_ptr()).unwrap_or(std::ptr::null()),
397                pgo_available,
398            );
399            DiagnosticHandlers { data, llcx, old_handler }
400        }
401    }
402}
403
404impl<'a> Drop for DiagnosticHandlers<'a> {
405    fn drop(&mut self) {
406        unsafe {
407            llvm::LLVMRustContextSetDiagnosticHandler(self.llcx, self.old_handler);
408            drop(Box::from_raw(self.data));
409        }
410    }
411}
412
413fn report_inline_asm(
414    cgcx: &CodegenContext,
415    msg: String,
416    level: llvm::DiagnosticLevel,
417    cookie: u64,
418    source: Option<(String, Vec<InnerSpan>)>,
419) -> InlineAsmError {
420    // In LTO build we may get srcloc values from other crates which are invalid
421    // since they use a different source map. To be safe we just suppress these
422    // in LTO builds.
423    let span = if cookie == 0 || #[allow(non_exhaustive_omitted_patterns)] match cgcx.lto {
    Lto::Fat | Lto::Thin => true,
    _ => false,
}matches!(cgcx.lto, Lto::Fat | Lto::Thin) {
424        SpanData::default()
425    } else {
426        SpanData {
427            lo: BytePos::from_u32(cookie as u32),
428            hi: BytePos::from_u32((cookie >> 32) as u32),
429            ctxt: SyntaxContext::root(),
430            parent: None,
431        }
432    };
433    let level = match level {
434        llvm::DiagnosticLevel::Error => Level::Error,
435        llvm::DiagnosticLevel::Warning => Level::Warning,
436        llvm::DiagnosticLevel::Note | llvm::DiagnosticLevel::Remark => Level::Note,
437    };
438    let msg = msg.trim_prefix("error: ").to_string();
439    InlineAsmError { span, msg, level, source }
440}
441
442unsafe extern "C" fn diagnostic_handler(info: &DiagnosticInfo, user: *mut c_void) {
443    if user.is_null() {
444        return;
445    }
446    let (cgcx, shared_emitter) = unsafe { *(user as *const (&CodegenContext, &SharedEmitter)) };
447
448    let dcx = DiagCtxt::new(Box::new(shared_emitter.clone()));
449    let dcx = dcx.handle();
450
451    match unsafe { llvm::diagnostic::Diagnostic::unpack(info) } {
452        llvm::diagnostic::InlineAsm(inline) => {
453            // FIXME use dcx
454            shared_emitter.inline_asm_error(report_inline_asm(
455                cgcx,
456                inline.message,
457                inline.level,
458                inline.cookie,
459                inline.source,
460            ));
461        }
462
463        llvm::diagnostic::Optimization(opt) => {
464            dcx.emit_note(FromLlvmOptimizationDiag {
465                filename: &opt.filename,
466                line: opt.line,
467                column: opt.column,
468                pass_name: &opt.pass_name,
469                kind: match opt.kind {
470                    OptimizationRemark => "success",
471                    OptimizationMissed | OptimizationFailure => "missed",
472                    OptimizationAnalysis
473                    | OptimizationAnalysisFPCommute
474                    | OptimizationAnalysisAliasing => "analysis",
475                    OptimizationRemarkOther => "other",
476                },
477                message: &opt.message,
478            });
479        }
480        llvm::diagnostic::PGO(diagnostic_ref) | llvm::diagnostic::Linker(diagnostic_ref) => {
481            let message = llvm::build_string(|s| unsafe {
482                llvm::LLVMRustWriteDiagnosticInfoToString(diagnostic_ref, s)
483            })
484            .expect("non-UTF8 diagnostic");
485            dcx.emit_warn(FromLlvmDiag { message });
486        }
487        llvm::diagnostic::Unsupported(diagnostic_ref) => {
488            let message = llvm::build_string(|s| unsafe {
489                llvm::LLVMRustWriteDiagnosticInfoToString(diagnostic_ref, s)
490            })
491            .expect("non-UTF8 diagnostic");
492            dcx.emit_err(FromLlvmDiag { message });
493        }
494        llvm::diagnostic::UnknownDiagnostic(..) => {}
495    }
496}
497
498fn get_pgo_gen_path(config: &ModuleConfig) -> Option<CString> {
499    match config.pgo_gen {
500        SwitchWithOptPath::Enabled(ref opt_dir_path) => {
501            let path = if let Some(dir_path) = opt_dir_path {
502                dir_path.join("default_%m.profraw")
503            } else {
504                PathBuf::from("default_%m.profraw")
505            };
506
507            Some(CString::new(::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("{0}", path.display()))
    })format!("{}", path.display())).unwrap())
508        }
509        SwitchWithOptPath::Disabled => None,
510    }
511}
512
513fn get_pgo_use_path(config: &ModuleConfig) -> Option<CString> {
514    config
515        .pgo_use
516        .as_ref()
517        .map(|path_buf| CString::new(path_buf.to_string_lossy().as_bytes()).unwrap())
518}
519
520fn get_pgo_sample_use_path(config: &ModuleConfig) -> Option<CString> {
521    config
522        .pgo_sample_use
523        .as_ref()
524        .map(|path_buf| CString::new(path_buf.to_string_lossy().as_bytes()).unwrap())
525}
526
527fn get_instr_profile_output_path(config: &ModuleConfig) -> Option<CString> {
528    config.instrument_coverage.then(|| c"default_%m_%p.profraw".to_owned())
529}
530
531// PreAD will run llvm opts but disable size increasing opts (vectorization, loop unrolling)
532// DuringAD is the same as above, but also runs the enzyme opt and autodiff passes.
533// PostAD will run all opts, including size increasing opts.
534#[derive(#[automatically_derived]
impl ::core::fmt::Debug for AutodiffStage {
    #[inline]
    fn fmt(&self, f: &mut ::core::fmt::Formatter) -> ::core::fmt::Result {
        ::core::fmt::Formatter::write_str(f,
            match self {
                AutodiffStage::PreAD => "PreAD",
                AutodiffStage::DuringAD => "DuringAD",
                AutodiffStage::PostAD => "PostAD",
            })
    }
}Debug, #[automatically_derived]
impl ::core::cmp::Eq for AutodiffStage {
    #[inline]
    #[doc(hidden)]
    #[coverage(off)]
    fn assert_fields_are_eq(&self) {}
}Eq, #[automatically_derived]
impl ::core::cmp::PartialEq for AutodiffStage {
    #[inline]
    fn eq(&self, other: &AutodiffStage) -> bool {
        let __self_discr = ::core::intrinsics::discriminant_value(self);
        let __arg1_discr = ::core::intrinsics::discriminant_value(other);
        __self_discr == __arg1_discr
    }
}PartialEq)]
535pub(crate) enum AutodiffStage {
536    PreAD,
537    DuringAD,
538    PostAD,
539}
540
541pub(crate) unsafe fn llvm_optimize(
542    cgcx: &CodegenContext,
543    prof: &SelfProfilerRef,
544    dcx: DiagCtxtHandle<'_>,
545    module: &ModuleCodegen<ModuleLlvm>,
546    thin_lto_buffer: Option<&mut Option<Buffer>>,
547    thin_lto_summary_buffer: Option<&mut Option<Buffer>>,
548    config: &ModuleConfig,
549    opt_level: config::OptLevel,
550    opt_stage: llvm::OptStage,
551    autodiff_stage: AutodiffStage,
552) {
553    // Enzyme:
554    // The whole point of compiler based AD is to differentiate optimized IR instead of unoptimized
555    // source code. However, benchmarks show that optimizations increasing the code size
556    // tend to reduce AD performance. Therefore deactivate them before AD, then differentiate the code
557    // and finally re-optimize the module, now with all optimizations available.
558    // FIXME(ZuseZ4): In a future update we could figure out how to only optimize individual functions getting
559    // differentiated.
560
561    let consider_ad = config.autodiff.contains(&config::AutoDiff::Enable);
562    let run_enzyme = autodiff_stage == AutodiffStage::DuringAD;
563    let print_before_enzyme = config.autodiff.contains(&config::AutoDiff::PrintModBefore);
564    let print_after_enzyme = config.autodiff.contains(&config::AutoDiff::PrintModAfter);
565    let print_passes = config.autodiff.contains(&config::AutoDiff::PrintPasses);
566    let passes_after_enzyme = if autodiff_stage == AutodiffStage::PostAD {
567        config.autodiff_post_passes.as_deref()
568    } else {
569        None
570    };
571    let passes_after_enzyme_ptr =
572        passes_after_enzyme.map_or(std::ptr::null(), |s| s.as_c_char_ptr());
573    let passes_after_enzyme_len = passes_after_enzyme.map_or(0, |s| s.len());
574    let merge_functions;
575    let unroll_loops;
576    let vectorize_slp;
577    let vectorize_loop;
578
579    // When we build rustc with enzyme/autodiff support, we want to postpone size-increasing
580    // optimizations until after differentiation. Our pipeline is thus: (opt + enzyme), (full opt).
581    // We therefore have two calls to llvm_optimize, if autodiff is used.
582    //
583    // We also must disable merge_functions, since autodiff placeholder/dummy bodies tend to be
584    // identical. We run opts before AD, so there is a chance that LLVM will merge our dummies.
585    // In that case, we lack some dummy bodies and can't replace them with the real AD code anymore.
586    // We then would need to abort compilation. This was especially common in test cases.
587    if consider_ad && autodiff_stage != AutodiffStage::PostAD {
588        merge_functions = false;
589        unroll_loops = false;
590        vectorize_slp = false;
591        vectorize_loop = false;
592    } else {
593        unroll_loops =
594            opt_level != config::OptLevel::Size && opt_level != config::OptLevel::SizeMin;
595        merge_functions = config.merge_functions;
596        vectorize_slp = config.vectorize_slp;
597        vectorize_loop = config.vectorize_loop;
598    }
599    {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event compiler/rustc_codegen_llvm/src/back/write.rs:599",
                        "rustc_codegen_llvm::back::write", ::tracing::Level::TRACE,
                        ::tracing_core::__macro_support::Option::Some("compiler/rustc_codegen_llvm/src/back/write.rs"),
                        ::tracing_core::__macro_support::Option::Some(599u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_codegen_llvm::back::write"),
                        ::tracing_core::field::FieldSet::new(&[{
                                            const NAME:
                                                ::tracing::__macro_support::FieldName<{
                                                    ::tracing::__macro_support::FieldName::len("unroll_loops")
                                                }> =
                                                ::tracing::__macro_support::FieldName::new("unroll_loops");
                                            NAME.as_str()
                                        },
                                        {
                                            const NAME:
                                                ::tracing::__macro_support::FieldName<{
                                                    ::tracing::__macro_support::FieldName::len("vectorize_slp")
                                                }> =
                                                ::tracing::__macro_support::FieldName::new("vectorize_slp");
                                            NAME.as_str()
                                        },
                                        {
                                            const NAME:
                                                ::tracing::__macro_support::FieldName<{
                                                    ::tracing::__macro_support::FieldName::len("vectorize_loop")
                                                }> =
                                                ::tracing::__macro_support::FieldName::new("vectorize_loop");
                                            NAME.as_str()
                                        },
                                        {
                                            const NAME:
                                                ::tracing::__macro_support::FieldName<{
                                                    ::tracing::__macro_support::FieldName::len("run_enzyme")
                                                }> =
                                                ::tracing::__macro_support::FieldName::new("run_enzyme");
                                            NAME.as_str()
                                        }], ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::TRACE <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::TRACE <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&::tracing::field::debug(&unroll_loops)
                                            as &dyn ::tracing::field::Value)),
                                (::tracing::__macro_support::Option::Some(&::tracing::field::debug(&vectorize_slp)
                                            as &dyn ::tracing::field::Value)),
                                (::tracing::__macro_support::Option::Some(&::tracing::field::debug(&vectorize_loop)
                                            as &dyn ::tracing::field::Value)),
                                (::tracing::__macro_support::Option::Some(&::tracing::field::debug(&run_enzyme)
                                            as &dyn ::tracing::field::Value))])
            });
    } else { ; }
};trace!(?unroll_loops, ?vectorize_slp, ?vectorize_loop, ?run_enzyme);
600    if thin_lto_buffer.is_some() {
601        if !#[allow(non_exhaustive_omitted_patterns)] match opt_stage {
            llvm::OptStage::PreLinkNoLTO | llvm::OptStage::PreLinkFatLTO |
                llvm::OptStage::PreLinkThinLTO => true,
            _ => false,
        } {
    {
        ::core::panicking::panic_fmt(format_args!("the bitcode for LTO can only be obtained at the pre-link stage"));
    }
};assert!(
602            matches!(
603                opt_stage,
604                llvm::OptStage::PreLinkNoLTO
605                    | llvm::OptStage::PreLinkFatLTO
606                    | llvm::OptStage::PreLinkThinLTO
607            ),
608            "the bitcode for LTO can only be obtained at the pre-link stage"
609        );
610    }
611    let pgo_gen_path = get_pgo_gen_path(config);
612    let pgo_use_path = get_pgo_use_path(config);
613    let pgo_sample_use_path = get_pgo_sample_use_path(config);
614    let is_lto = opt_stage == llvm::OptStage::ThinLTO || opt_stage == llvm::OptStage::FatLTO;
615    let instr_profile_output_path = get_instr_profile_output_path(config);
616    let sanitize_dataflow_abilist: Vec<_> = config
617        .sanitizer_dataflow_abilist
618        .iter()
619        .map(|file| CString::new(file.as_str()).unwrap())
620        .collect();
621    let sanitize_dataflow_abilist_ptrs: Vec<_> =
622        sanitize_dataflow_abilist.iter().map(|file| file.as_ptr()).collect();
623    // Sanitizer instrumentation is only inserted during the pre-link optimization stage.
624    let sanitizer_options = if !is_lto {
625        Some(llvm::SanitizerOptions {
626            sanitize_address: config.sanitizer.contains(SanitizerSet::ADDRESS),
627            sanitize_address_recover: config.sanitizer_recover.contains(SanitizerSet::ADDRESS),
628            sanitize_cfi: config.sanitizer.contains(SanitizerSet::CFI),
629            sanitize_dataflow: config.sanitizer.contains(SanitizerSet::DATAFLOW),
630            sanitize_dataflow_abilist: sanitize_dataflow_abilist_ptrs.as_ptr(),
631            sanitize_dataflow_abilist_len: sanitize_dataflow_abilist_ptrs.len(),
632            sanitize_kcfi: config.sanitizer.contains(SanitizerSet::KCFI),
633            sanitize_memory: config.sanitizer.contains(SanitizerSet::MEMORY),
634            sanitize_memory_recover: config.sanitizer_recover.contains(SanitizerSet::MEMORY),
635            sanitize_memory_track_origins: config.sanitizer_memory_track_origins as c_int,
636            sanitize_realtime: config.sanitizer.contains(SanitizerSet::REALTIME),
637            sanitize_thread: config.sanitizer.contains(SanitizerSet::THREAD),
638            sanitize_hwaddress: config.sanitizer.contains(SanitizerSet::HWADDRESS),
639            sanitize_hwaddress_recover: config.sanitizer_recover.contains(SanitizerSet::HWADDRESS),
640            sanitize_kernel_address: config.sanitizer.contains(SanitizerSet::KERNELADDRESS),
641            sanitize_kernel_address_recover: config
642                .sanitizer_recover
643                .contains(SanitizerSet::KERNELADDRESS),
644            sanitize_kernel_hwaddress: config.sanitizer.contains(SanitizerSet::KERNELHWADDRESS),
645            sanitize_kernel_hwaddress_recover: config
646                .sanitizer_recover
647                .contains(SanitizerSet::KERNELHWADDRESS),
648        })
649    } else {
650        None
651    };
652
653    fn handle_offload<'ll>(cx: &'ll SimpleCx<'_>, old_fn: &llvm::Value) {
654        let old_fn_ty = cx.get_type_of_global(old_fn);
655        let old_param_types = cx.func_params_types(old_fn_ty);
656        let old_param_count = old_param_types.len();
657        if old_param_count == 0 {
658            return;
659        }
660
661        let first_param = llvm::get_param(old_fn, 0);
662        let c_name = llvm::get_value_name(first_param);
663        let first_arg_name = str::from_utf8(&c_name).unwrap();
664        // We might call llvm_optimize (and thus this code) multiple times on the same IR,
665        // but we shouldn't add this helper ptr multiple times.
666        // FIXME(offload): This could break if the user calls his first argument `dyn_ptr`.
667        if first_arg_name == "dyn_ptr" {
668            return;
669        }
670
671        // Create the new parameter list, with ptr as the first argument
672        let mut new_param_types = Vec::with_capacity(old_param_count as usize + 1);
673        new_param_types.push(cx.type_ptr());
674
675        // This relies on undocumented LLVM knowledge that scalars must be passed as i64
676        for &old_ty in &old_param_types {
677            let new_ty = match cx.type_kind(old_ty) {
678                TypeKind::Half | TypeKind::Float | TypeKind::Double | TypeKind::Integer => {
679                    cx.type_i64()
680                }
681                _ => old_ty,
682            };
683            new_param_types.push(new_ty);
684        }
685
686        // Create the new function type
687        let ret_ty = unsafe { llvm::LLVMGetReturnType(old_fn_ty) };
688        let new_fn_ty = cx.type_func(&new_param_types, ret_ty);
689
690        // Create the new function, with a temporary .offload name to avoid a name collision.
691        let old_fn_name = String::from_utf8(llvm::get_value_name(old_fn)).unwrap();
692        let new_fn_name = ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!("{0}.offload", &old_fn_name))
    })format!("{}.offload", &old_fn_name);
693        let new_fn = cx.add_func(&new_fn_name, new_fn_ty);
694        let a0 = llvm::get_param(new_fn, 0);
695        llvm::set_value_name(a0, CString::new("dyn_ptr").unwrap().as_bytes());
696
697        let bb = SBuilder::append_block(cx, new_fn, "entry");
698        let mut builder = SBuilder::build(cx, bb);
699
700        let mut old_args_rebuilt = Vec::with_capacity(old_param_types.len());
701
702        for (i, &old_ty) in old_param_types.iter().enumerate() {
703            let new_arg = llvm::get_param(new_fn, (i + 1) as u32);
704
705            let rebuilt = match cx.type_kind(old_ty) {
706                TypeKind::Half | TypeKind::Float | TypeKind::Double | TypeKind::Integer => {
707                    let num_bits = scalar_width(cx, old_ty);
708
709                    let trunc = builder.trunc(new_arg, cx.type_ix(num_bits));
710                    builder.bitcast(trunc, old_ty)
711                }
712                _ => new_arg,
713            };
714
715            old_args_rebuilt.push(rebuilt);
716        }
717
718        builder.ret_void();
719
720        // Here we map the old arguments to the new arguments, with an offset of 1 to make sure
721        // that we don't use the newly added `%dyn_ptr`.
722        unsafe {
723            llvm::RustOffloadWrapper::get_instance().llvm_rust_offload_wrapper(
724                old_fn,
725                new_fn,
726                old_args_rebuilt.as_slice(),
727            );
728        }
729
730        llvm::set_linkage(new_fn, llvm::get_linkage(old_fn));
731        llvm::set_visibility(new_fn, llvm::get_visibility(old_fn));
732
733        // Replace all uses of old_fn with new_fn (RAUW)
734        unsafe {
735            llvm::LLVMReplaceAllUsesWith(old_fn, new_fn);
736        }
737        let name = llvm::get_value_name(old_fn);
738        unsafe {
739            llvm::LLVMDeleteFunction(old_fn);
740        }
741        // Now we can re-use the old name, without name collision.
742        llvm::set_value_name(new_fn, &name);
743    }
744
745    if cgcx.target_is_like_gpu && config.offload.contains(&config::Offload::Device) {
746        let cx =
747            SimpleCx::new(module.module_llvm.llmod(), module.module_llvm.llcx, cgcx.pointer_size);
748        for func in cx.get_functions() {
749            let offload_kernel = "offload-kernel";
750            if attributes::has_string_attr(func, offload_kernel) {
751                handle_offload(&cx, func);
752            }
753            attributes::remove_string_attr_from_llfn(func, offload_kernel);
754        }
755    }
756
757    let mut llvm_profiler = prof
758        .llvm_recording_enabled()
759        .then(|| LlvmSelfProfiler::new(prof.get_self_profiler().unwrap()));
760
761    let llvm_selfprofiler =
762        llvm_profiler.as_mut().map(|s| s as *mut _ as *mut c_void).unwrap_or(std::ptr::null_mut());
763
764    let extra_passes = if !is_lto { config.passes.join(",") } else { "".to_string() };
765
766    let llvm_plugins = config.llvm_plugins.join(",");
767
768    let enzyme_fn = if consider_ad {
769        let wrapper = llvm::EnzymeWrapper::get_instance();
770        wrapper.registerEnzymeAndPassPipeline
771    } else {
772        std::ptr::null()
773    };
774
775    let result = unsafe {
776        llvm::LLVMRustOptimize(
777            module.module_llvm.llmod(),
778            &*module.module_llvm.tm.raw(),
779            to_pass_builder_opt_level(opt_level),
780            opt_stage,
781            cgcx.use_linker_plugin_lto,
782            config.no_prepopulate_passes,
783            config.verify_llvm_ir,
784            config.lint_llvm_ir,
785            thin_lto_buffer,
786            thin_lto_summary_buffer,
787            merge_functions,
788            unroll_loops,
789            vectorize_slp,
790            vectorize_loop,
791            config.no_builtins,
792            config.emit_lifetime_markers,
793            enzyme_fn,
794            print_before_enzyme,
795            print_after_enzyme,
796            print_passes,
797            sanitizer_options.as_ref(),
798            pgo_gen_path.as_ref().map_or(std::ptr::null(), |s| s.as_ptr()),
799            pgo_use_path.as_ref().map_or(std::ptr::null(), |s| s.as_ptr()),
800            config.instrument_coverage,
801            instr_profile_output_path.as_ref().map_or(std::ptr::null(), |s| s.as_ptr()),
802            pgo_sample_use_path.as_ref().map_or(std::ptr::null(), |s| s.as_ptr()),
803            config.debug_info_for_profiling,
804            llvm_selfprofiler,
805            selfprofile_before_pass_callback,
806            selfprofile_after_pass_callback,
807            passes_after_enzyme_ptr,
808            passes_after_enzyme_len,
809            extra_passes.as_c_char_ptr(),
810            extra_passes.len(),
811            llvm_plugins.as_c_char_ptr(),
812            llvm_plugins.len(),
813        )
814    };
815
816    if cgcx.target_is_like_gpu && config.offload.contains(&config::Offload::Device) {
817        let device_path = cgcx.output_filenames.path(OutputType::Object);
818        let device_dir = device_path.parent().unwrap();
819        let device_out = device_dir.join("device.bin");
820        let device_out_c = path_to_c_string(device_out.as_path());
821        // 1) Bundle device module into offload image device.bin (device TM)
822        let ok = unsafe {
823            llvm::RustOffloadWrapper::get_instance().llvm_rust_bundle_images(
824                module.module_llvm.llmod(),
825                module.module_llvm.tm.raw(),
826                device_out_c.as_c_str(),
827            )
828        };
829        if !ok || !device_out.exists() {
830            dcx.emit_err(crate::diagnostics::OffloadBundleImagesFailed);
831        }
832    }
833
834    // This assumes that we previously compiled our kernels for a gpu target, which created a
835    // `device.bin` artifact. The user is supposed to provide us with a path to this artifact, we
836    // don't need any other artifacts from the previous run. We will embed this artifact into our
837    // LLVM-IR host module, to create a `host.o` ObjectFile, which we will write to disk.
838    // The last, not yet automated steps uses the `clang-linker-wrapper` to process `host.o`.
839    if !cgcx.target_is_like_gpu {
840        if let Some(device_path) = config
841            .offload
842            .iter()
843            .find_map(|o| if let config::Offload::Host(path) = o { Some(path) } else { None })
844        {
845            let device_pathbuf = PathBuf::from(device_path);
846            if device_pathbuf.is_relative() {
847                dcx.emit_err(crate::diagnostics::OffloadWithoutAbsPath);
848            } else if device_pathbuf
849                .file_name()
850                .and_then(|n| n.to_str())
851                .is_some_and(|n| n != "device.bin")
852            {
853                dcx.emit_err(crate::diagnostics::OffloadWrongFileName);
854            } else if !device_pathbuf.exists() {
855                dcx.emit_err(crate::diagnostics::OffloadNonexistingPath);
856            }
857            let host_path = cgcx.output_filenames.path(OutputType::Object);
858            let host_dir = host_path.parent().unwrap();
859            let out_obj = host_dir.join("host.o");
860            let device_bin_c = path_to_c_string(device_pathbuf.as_path());
861
862            // 2) Finalize host: lib.bc + device.bin -> host.o (host TM)
863            // We create a full clone of our LLVM host module, since we will embed the device IR
864            // into it, and this might break caching or incremental compilation otherwise.
865            let llmod2 = llvm::LLVMCloneModule(module.module_llvm.llmod());
866            let ok = unsafe {
867                llvm::RustOffloadWrapper::get_instance()
868                    .llvm_rust_offload_embed_buffer_in_module(llmod2, device_bin_c.as_c_str())
869            };
870            if !ok {
871                dcx.emit_err(crate::diagnostics::OffloadEmbedFailed);
872            }
873            write_output_file(
874                dcx,
875                module.module_llvm.tm.raw(),
876                config.no_builtins,
877                llmod2,
878                &out_obj,
879                None,
880                llvm::FileType::ObjectFile,
881                prof,
882                true,
883            );
884            // We ignore cgcx.save_temps here and unconditionally always keep our `device.bin` artifact.
885            // Otherwise, recompiling the host code would fail since we deleted that device artifact
886            // in the previous host compilation, which would be confusing at best.
887        }
888    }
889    result.into_result().unwrap_or_else(|()| llvm_err(dcx, LlvmError::RunLlvmPasses))
890}
891
892// Unsafe due to LLVM calls.
893pub(crate) fn optimize(
894    cgcx: &CodegenContext,
895    prof: &SelfProfilerRef,
896    shared_emitter: &SharedEmitter,
897    module: &mut ModuleCodegen<ModuleLlvm>,
898    config: &ModuleConfig,
899) {
900    let _timer = prof.generic_activity_with_arg("LLVM_module_optimize", &*module.name);
901
902    let dcx = DiagCtxt::new(Box::new(shared_emitter.clone()));
903    let dcx = dcx.handle();
904
905    let llcx = &*module.module_llvm.llcx;
906    let _handlers =
907        DiagnosticHandlers::new(cgcx, shared_emitter, llcx, module, CodegenDiagnosticsStage::Opt);
908
909    if module.kind == ModuleKind::Regular {
910        save_temp_bitcode(cgcx, module, "no-opt");
911    }
912
913    // FIXME(ZuseZ4): support SanitizeHWAddress and prevent illegal/unsupported opts
914
915    if let Some(opt_level) = config.opt_level {
916        let opt_stage = match cgcx.lto {
917            Lto::Fat => llvm::OptStage::PreLinkFatLTO,
918            Lto::Thin | Lto::ThinLocal => llvm::OptStage::PreLinkThinLTO,
919            _ if cgcx.use_linker_plugin_lto => llvm::OptStage::PreLinkThinLTO,
920            _ => llvm::OptStage::PreLinkNoLTO,
921        };
922
923        // If we know that we will later run AD, then we disable vectorization and loop unrolling.
924        // Otherwise we pretend AD is already done and run the normal opt pipeline (=PostAD).
925        let consider_ad = config.autodiff.contains(&config::AutoDiff::Enable);
926        let autodiff_stage = if consider_ad { AutodiffStage::PreAD } else { AutodiffStage::PostAD };
927        // The embedded bitcode is used to run LTO/ThinLTO.
928        // The bitcode obtained during the `codegen` phase is no longer suitable for performing LTO.
929        // It may have undergone LTO due to ThinLocal, so we need to obtain the embedded bitcode at
930        // this point.
931        let (mut thin_lto_buffer, mut thin_lto_summary_buffer) = if (module.kind
932            == ModuleKind::Regular
933            && config.emit_obj == EmitObj::ObjectCode(BitcodeSection::Full))
934            || config.emit_thin_lto_summary
935        {
936            (Some(None), config.emit_thin_lto_summary.then_some(None))
937        } else {
938            (None, None)
939        };
940        unsafe {
941            llvm_optimize(
942                cgcx,
943                prof,
944                dcx,
945                module,
946                thin_lto_buffer.as_mut(),
947                thin_lto_summary_buffer.as_mut(),
948                config,
949                opt_level,
950                opt_stage,
951                autodiff_stage,
952            )
953        };
954        if let Some(thin_lto_buffer) = thin_lto_buffer {
955            let thin_lto_buffer = thin_lto_buffer.unwrap();
956            module.thin_lto_buffer = Some(thin_lto_buffer.data().to_vec());
957            let bc_summary_out =
958                cgcx.output_filenames.temp_path_for_cgu(OutputType::ThinLinkBitcode, &module.name);
959            if let Some(thin_lto_summary_buffer) = thin_lto_summary_buffer
960                && let Some(thin_link_bitcode_filename) = bc_summary_out.file_name()
961            {
962                let thin_lto_summary_buffer = thin_lto_summary_buffer.unwrap();
963                let summary_data = thin_lto_summary_buffer.data();
964                prof.artifact_size(
965                    "llvm_bitcode_summary",
966                    thin_link_bitcode_filename.to_string_lossy(),
967                    summary_data.len() as u64,
968                );
969                let _timer = prof.generic_activity_with_arg(
970                    "LLVM_module_codegen_emit_bitcode_summary",
971                    &*module.name,
972                );
973                if let Err(err) = fs::write(&bc_summary_out, summary_data) {
974                    dcx.emit_err(WriteBytecode { path: &bc_summary_out, err });
975                }
976            }
977        }
978    }
979}
980
981pub(crate) fn codegen(
982    cgcx: &CodegenContext,
983    prof: &SelfProfilerRef,
984    shared_emitter: &SharedEmitter,
985    module: ModuleCodegen<ModuleLlvm>,
986    config: &ModuleConfig,
987) -> CompiledModule {
988    let _timer = prof.generic_activity_with_arg("LLVM_module_codegen", &*module.name);
989
990    let dcx = DiagCtxt::new(Box::new(shared_emitter.clone()));
991    let dcx = dcx.handle();
992
993    {
994        let llmod = module.module_llvm.llmod();
995        let llcx = &*module.module_llvm.llcx;
996        let tm = &*module.module_llvm.tm;
997        let _handlers = DiagnosticHandlers::new(
998            cgcx,
999            shared_emitter,
1000            llcx,
1001            &module,
1002            CodegenDiagnosticsStage::Codegen,
1003        );
1004
1005        if cgcx.msvc_imps_needed {
1006            create_msvc_imps(cgcx, llcx, llmod);
1007        }
1008
1009        // Note that if object files are just LLVM bitcode we write bitcode,
1010        // copy it to the .o file, and delete the bitcode if it wasn't
1011        // otherwise requested.
1012
1013        let bc_out = cgcx.output_filenames.temp_path_for_cgu(OutputType::Bitcode, &module.name);
1014        let obj_out = cgcx.output_filenames.temp_path_for_cgu(OutputType::Object, &module.name);
1015
1016        if config.bitcode_needed() {
1017            if config.emit_bc || config.emit_obj == EmitObj::Bitcode {
1018                let thin = {
1019                    let _timer = prof.generic_activity_with_arg(
1020                        "LLVM_module_codegen_make_bitcode",
1021                        &*module.name,
1022                    );
1023                    ModuleBuffer::new(llmod, cgcx.lto != Lto::Fat)
1024                };
1025                let data = thin.data();
1026                let _timer = prof
1027                    .generic_activity_with_arg("LLVM_module_codegen_emit_bitcode", &*module.name);
1028                if let Some(bitcode_filename) = bc_out.file_name() {
1029                    prof.artifact_size(
1030                        "llvm_bitcode",
1031                        bitcode_filename.to_string_lossy(),
1032                        data.len() as u64,
1033                    );
1034                }
1035                if let Err(err) = fs::write(&bc_out, data) {
1036                    dcx.emit_err(WriteBytecode { path: &bc_out, err });
1037                }
1038            }
1039
1040            if config.embed_bitcode() && module.kind == ModuleKind::Regular {
1041                let _timer = prof
1042                    .generic_activity_with_arg("LLVM_module_codegen_embed_bitcode", &*module.name);
1043                let thin_bc =
1044                    module.thin_lto_buffer.as_deref().expect("cannot find embedded bitcode");
1045                embed_bitcode(cgcx, llcx, llmod, &thin_bc);
1046            }
1047        }
1048
1049        if config.emit_ir {
1050            let _timer =
1051                prof.generic_activity_with_arg("LLVM_module_codegen_emit_ir", &*module.name);
1052            let out =
1053                cgcx.output_filenames.temp_path_for_cgu(OutputType::LlvmAssembly, &module.name);
1054            let out_c = path_to_c_string(&out);
1055
1056            extern "C" fn demangle_callback(
1057                input_ptr: *const c_char,
1058                input_len: size_t,
1059                output_ptr: *mut c_char,
1060                output_len: size_t,
1061            ) -> size_t {
1062                let input =
1063                    unsafe { slice::from_raw_parts(input_ptr as *const u8, input_len as usize) };
1064
1065                let Ok(input) = str::from_utf8(input) else { return 0 };
1066
1067                let output = unsafe {
1068                    slice::from_raw_parts_mut(output_ptr as *mut u8, output_len as usize)
1069                };
1070                let mut cursor = io::Cursor::new(output);
1071
1072                let Ok(demangled) = rustc_demangle::try_demangle(input) else { return 0 };
1073
1074                if cursor.write_fmt(format_args!("{0:#}", demangled))write!(cursor, "{demangled:#}").is_err() {
1075                    // Possible only if provided buffer is not big enough
1076                    return 0;
1077                }
1078
1079                cursor.position() as size_t
1080            }
1081
1082            let result =
1083                unsafe { llvm::LLVMRustPrintModule(llmod, out_c.as_ptr(), demangle_callback) };
1084
1085            if result == llvm::LLVMRustResult::Success {
1086                record_artifact_size(prof, "llvm_ir", &out);
1087            }
1088
1089            result
1090                .into_result()
1091                .unwrap_or_else(|()| llvm_err(dcx, LlvmError::WriteIr { path: &out }));
1092        }
1093
1094        if config.emit_asm {
1095            let _timer =
1096                prof.generic_activity_with_arg("LLVM_module_codegen_emit_asm", &*module.name);
1097            let path = cgcx.output_filenames.temp_path_for_cgu(OutputType::Assembly, &module.name);
1098
1099            // We can't use the same module for asm and object code output,
1100            // because that triggers various errors like invalid IR or broken
1101            // binaries. So we must clone the module to produce the asm output
1102            // if we are also producing object code.
1103            let llmod = if let EmitObj::ObjectCode(_) = config.emit_obj {
1104                llvm::LLVMCloneModule(llmod)
1105            } else {
1106                llmod
1107            };
1108            write_output_file(
1109                dcx,
1110                tm.raw(),
1111                config.no_builtins,
1112                llmod,
1113                &path,
1114                None,
1115                llvm::FileType::AssemblyFile,
1116                prof,
1117                config.verify_llvm_ir,
1118            );
1119        }
1120
1121        match config.emit_obj {
1122            EmitObj::ObjectCode(_) => {
1123                let _timer =
1124                    prof.generic_activity_with_arg("LLVM_module_codegen_emit_obj", &*module.name);
1125
1126                let dwo_out = cgcx.output_filenames.temp_path_dwo_for_cgu(&module.name);
1127                let dwo_out = match (cgcx.split_debuginfo, cgcx.split_dwarf_kind) {
1128                    // Don't change how DWARF is emitted when disabled.
1129                    (SplitDebuginfo::Off, _) => None,
1130                    // Don't provide a DWARF object path if split debuginfo is enabled but this is
1131                    // a platform that doesn't support Split DWARF.
1132                    _ if !cgcx.target_can_use_split_dwarf => None,
1133                    // Don't provide a DWARF object path in single mode, sections will be written
1134                    // into the object as normal but ignored by linker.
1135                    (_, SplitDwarfKind::Single) => None,
1136                    // Emit (a subset of the) DWARF into a separate dwarf object file in split
1137                    // mode.
1138                    (_, SplitDwarfKind::Split) => Some(dwo_out.as_path()),
1139                };
1140
1141                write_output_file(
1142                    dcx,
1143                    tm.raw(),
1144                    config.no_builtins,
1145                    llmod,
1146                    &obj_out,
1147                    dwo_out,
1148                    llvm::FileType::ObjectFile,
1149                    prof,
1150                    config.verify_llvm_ir,
1151                );
1152            }
1153
1154            EmitObj::Bitcode => {
1155                {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event compiler/rustc_codegen_llvm/src/back/write.rs:1155",
                        "rustc_codegen_llvm::back::write", ::tracing::Level::DEBUG,
                        ::tracing_core::__macro_support::Option::Some("compiler/rustc_codegen_llvm/src/back/write.rs"),
                        ::tracing_core::__macro_support::Option::Some(1155u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_codegen_llvm::back::write"),
                        ::tracing_core::field::FieldSet::new(&["message"],
                            ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::DEBUG <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::DEBUG <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&format_args!("copying bitcode {0:?} to obj {1:?}",
                                                    bc_out, obj_out) as &dyn ::tracing::field::Value))])
            });
    } else { ; }
};debug!("copying bitcode {:?} to obj {:?}", bc_out, obj_out);
1156                if let Err(err) = link_or_copy(&bc_out, &obj_out) {
1157                    dcx.emit_err(CopyBitcode { err });
1158                }
1159
1160                if !config.emit_bc {
1161                    {
    use ::tracing::__macro_support::Callsite as _;
    static __CALLSITE: ::tracing::callsite::DefaultCallsite =
        {
            static META: ::tracing::Metadata<'static> =
                {
                    ::tracing_core::metadata::Metadata::new("event compiler/rustc_codegen_llvm/src/back/write.rs:1161",
                        "rustc_codegen_llvm::back::write", ::tracing::Level::DEBUG,
                        ::tracing_core::__macro_support::Option::Some("compiler/rustc_codegen_llvm/src/back/write.rs"),
                        ::tracing_core::__macro_support::Option::Some(1161u32),
                        ::tracing_core::__macro_support::Option::Some("rustc_codegen_llvm::back::write"),
                        ::tracing_core::field::FieldSet::new(&["message"],
                            ::tracing_core::callsite::Identifier(&__CALLSITE)),
                        ::tracing::metadata::Kind::EVENT)
                };
            ::tracing::callsite::DefaultCallsite::new(&META)
        };
    let enabled =
        ::tracing::Level::DEBUG <= ::tracing::level_filters::STATIC_MAX_LEVEL
                &&
                ::tracing::Level::DEBUG <=
                    ::tracing::level_filters::LevelFilter::current() &&
            {
                let interest = __CALLSITE.interest();
                !interest.is_never() &&
                    ::tracing::__macro_support::__is_enabled(__CALLSITE.metadata(),
                        interest)
            };
    if enabled {
        (|value_set: ::tracing::field::ValueSet|
                    {
                        let meta = __CALLSITE.metadata();
                        ::tracing::Event::dispatch(meta, &value_set);
                        ;
                    })({
                #[allow(unused_imports)]
                use ::tracing::field::{debug, display, Value};
                __CALLSITE.metadata().fields().value_set_all(&[(::tracing::__macro_support::Option::Some(&format_args!("removing_bitcode {0:?}",
                                                    bc_out) as &dyn ::tracing::field::Value))])
            });
    } else { ; }
};debug!("removing_bitcode {:?}", bc_out);
1162                    ensure_removed(dcx, &bc_out);
1163                }
1164            }
1165
1166            EmitObj::None => {}
1167        }
1168
1169        record_llvm_cgu_instructions_stats(prof, &module.name, llmod);
1170    }
1171
1172    // `.dwo` files are only emitted if:
1173    //
1174    // - Object files are being emitted (i.e. bitcode only or metadata only compilations will not
1175    //   produce dwarf objects, even if otherwise enabled)
1176    // - Target supports Split DWARF
1177    // - Split debuginfo is enabled
1178    // - Split DWARF kind is `split` (i.e. debuginfo is split into `.dwo` files, not different
1179    //   sections in the `.o` files).
1180    let dwarf_object_emitted = #[allow(non_exhaustive_omitted_patterns)] match config.emit_obj {
    EmitObj::ObjectCode(_) => true,
    _ => false,
}matches!(config.emit_obj, EmitObj::ObjectCode(_))
1181        && cgcx.target_can_use_split_dwarf
1182        && cgcx.split_debuginfo != SplitDebuginfo::Off
1183        && cgcx.split_dwarf_kind == SplitDwarfKind::Split;
1184    module.into_compiled_module(
1185        config.emit_obj != EmitObj::None,
1186        dwarf_object_emitted,
1187        config.emit_bc,
1188        config.emit_asm,
1189        config.emit_ir,
1190        &cgcx.output_filenames,
1191    )
1192}
1193
1194fn create_section_with_flags_asm(section_name: &str, section_flags: &str, data: &[u8]) -> Vec<u8> {
1195    let mut asm = ::alloc::__export::must_use({
        ::alloc::fmt::format(format_args!(".section {0},\"{1}\"\n",
                section_name, section_flags))
    })format!(".section {section_name},\"{section_flags}\"\n").into_bytes();
1196    asm.extend_from_slice(b".ascii \"");
1197    asm.reserve(data.len());
1198    for &byte in data {
1199        if byte == b'\\' || byte == b'"' {
1200            asm.push(b'\\');
1201            asm.push(byte);
1202        } else if byte < 0x20 || byte >= 0x80 {
1203            // Avoid non UTF-8 inline assembly. Use octal escape sequence, because it is fixed
1204            // width, while hex escapes will consume following characters.
1205            asm.push(b'\\');
1206            asm.push(b'0' + ((byte >> 6) & 0x7));
1207            asm.push(b'0' + ((byte >> 3) & 0x7));
1208            asm.push(b'0' + ((byte >> 0) & 0x7));
1209        } else {
1210            asm.push(byte);
1211        }
1212    }
1213    asm.extend_from_slice(b"\"\n");
1214    asm
1215}
1216
1217pub(crate) fn bitcode_section_name(cgcx: &CodegenContext) -> &'static CStr {
1218    if cgcx.target_is_like_darwin {
1219        c"__LLVM,__bitcode"
1220    } else if cgcx.target_is_like_aix {
1221        c".ipa"
1222    } else {
1223        c".llvmbc"
1224    }
1225}
1226
1227/// Embed the bitcode of an LLVM module for LTO in the LLVM module itself.
1228fn embed_bitcode(
1229    cgcx: &CodegenContext,
1230    llcx: &llvm::Context,
1231    llmod: &llvm::Module,
1232    bitcode: &[u8],
1233) {
1234    // We're adding custom sections to the output object file, but we definitely
1235    // do not want these custom sections to make their way into the final linked
1236    // executable. The purpose of these custom sections is for tooling
1237    // surrounding object files to work with the LLVM IR, if necessary. For
1238    // example rustc's own LTO will look for LLVM IR inside of the object file
1239    // in these sections by default.
1240    //
1241    // To handle this is a bit different depending on the object file format
1242    // used by the backend, broken down into a few different categories:
1243    //
1244    // * Mach-O - this is for macOS. Inspecting the source code for the native
1245    //   linker here shows that the `.llvmbc` and `.llvmcmd` sections are
1246    //   automatically skipped by the linker. In that case there's nothing extra
1247    //   that we need to do here. We do need to make sure that the
1248    //   `__LLVM,__cmdline` section exists even though it is empty as otherwise
1249    //   ld64 rejects the object file.
1250    //
1251    // * Wasm - the native LLD linker is hard-coded to skip `.llvmbc` and
1252    //   `.llvmcmd` sections, so there's nothing extra we need to do.
1253    //
1254    // * COFF - if we don't do anything the linker will by default copy all
1255    //   these sections to the output artifact, not what we want! To subvert
1256    //   this we want to flag the sections we inserted here as
1257    //   `IMAGE_SCN_LNK_REMOVE`.
1258    //
1259    // * ELF - this is very similar to COFF above. One difference is that these
1260    //   sections are removed from the output linked artifact when
1261    //   `--gc-sections` is passed, which we pass by default. If that flag isn't
1262    //   passed though then these sections will show up in the final output.
1263    //   Additionally the flag that we need to set here is `SHF_EXCLUDE`.
1264    //
1265    // * XCOFF - AIX linker ignores content in .ipa and .info if no auxiliary
1266    //   symbol associated with these sections.
1267    //
1268    // Unfortunately, LLVM provides no way to set custom section flags. For ELF
1269    // and COFF we emit the sections using module level inline assembly for that
1270    // reason (see issue #90326 for historical background).
1271
1272    if cgcx.target_is_like_darwin
1273        || cgcx.target_is_like_aix
1274        || cgcx.target_arch == "wasm32"
1275        || cgcx.target_arch == "wasm64"
1276    {
1277        // We don't need custom section flags, create LLVM globals.
1278        let llconst = common::bytes_in_context(llcx, bitcode);
1279        let llglobal = llvm::add_global(llmod, common::val_ty(llconst), c"rustc.embedded.module");
1280        llvm::set_initializer(llglobal, llconst);
1281
1282        llvm::set_section(llglobal, bitcode_section_name(cgcx));
1283        llvm::set_linkage(llglobal, llvm::Linkage::PrivateLinkage);
1284        llvm::LLVMSetGlobalConstant(llglobal, llvm::TRUE);
1285
1286        let llconst = common::bytes_in_context(llcx, &[]);
1287        let llglobal = llvm::add_global(llmod, common::val_ty(llconst), c"rustc.embedded.cmdline");
1288        llvm::set_initializer(llglobal, llconst);
1289        let section = if cgcx.target_is_like_darwin {
1290            c"__LLVM,__cmdline"
1291        } else if cgcx.target_is_like_aix {
1292            c".info"
1293        } else {
1294            c".llvmcmd"
1295        };
1296        llvm::set_section(llglobal, section);
1297        llvm::set_linkage(llglobal, llvm::Linkage::PrivateLinkage);
1298    } else {
1299        // We need custom section flags, so emit module-level inline assembly.
1300        let section_flags = if cgcx.is_pe_coff { "n" } else { "e" };
1301        let asm = create_section_with_flags_asm(".llvmbc", section_flags, bitcode);
1302        llvm::append_module_inline_asm(llmod, &asm);
1303        let asm = create_section_with_flags_asm(".llvmcmd", section_flags, &[]);
1304        llvm::append_module_inline_asm(llmod, &asm);
1305    }
1306}
1307
1308// Create a `__imp_<symbol> = &symbol` global for each externally visible
1309// static data symbol, including aliases to static data.
1310// This is required to satisfy `dllimport` references to static data in .rlibs
1311// when using MSVC linker. We do this only for data, as linker can fix up
1312// code references on its own.
1313// See #26591, #27438
1314fn create_msvc_imps(cgcx: &CodegenContext, llcx: &llvm::Context, llmod: &llvm::Module) {
1315    if !cgcx.msvc_imps_needed {
1316        return;
1317    }
1318    // The x86 ABI seems to require that leading underscores are added to symbol
1319    // names, so we need an extra underscore on x86. There's also a leading
1320    // '\x01' here which disables LLVM's symbol mangling (e.g., no extra
1321    // underscores added in front).
1322    let prefix: &[u8] = if cgcx.target_arch == "x86" { b"\x01__imp__" } else { b"\x01__imp_" };
1323
1324    let ptr_ty = llvm_type_ptr(llcx);
1325    let symbols = std::iter::chain(
1326        base::iter_globals(llmod),
1327        base::iter_global_aliases(llmod).filter(|&val| {
1328            llvm::LLVMGetTypeKind(unsafe { llvm::LLVMGlobalGetValueType(val) }).to_rust()
1329                != llvm::TypeKind::Function
1330        }),
1331    )
1332    .map(|val| (val, llvm::get_linkage(val)))
1333    .filter(|&(val, linkage)| {
1334        #[allow(non_exhaustive_omitted_patterns)] match linkage {
    llvm::Linkage::ExternalLinkage | llvm::Linkage::WeakAnyLinkage => true,
    _ => false,
}matches!(linkage, llvm::Linkage::ExternalLinkage | llvm::Linkage::WeakAnyLinkage)
1335            && !llvm::is_declaration(val)
1336    })
1337    .collect::<Vec<_>>();
1338
1339    for (val, linkage) in symbols {
1340        let name = llvm::get_value_name(val);
1341        // Exclude some symbols that we know are not Rust symbols.
1342        if ignored(&name) {
1343            continue;
1344        }
1345
1346        let mut imp_name = prefix.to_vec();
1347        imp_name.extend(name);
1348        let imp_name = CString::new(imp_name).unwrap();
1349
1350        let imp = llvm::add_global(llmod, ptr_ty, &imp_name);
1351
1352        llvm::set_initializer(imp, val);
1353        llvm::set_linkage(imp, linkage);
1354    }
1355
1356    // Use this function to exclude certain symbols from `__imp` generation.
1357    fn ignored(symbol_name: &[u8]) -> bool {
1358        // These are symbols generated by LLVM's profiling instrumentation
1359        symbol_name.starts_with(b"__llvm_profile_")
1360    }
1361}
1362
1363fn record_artifact_size(
1364    self_profiler_ref: &SelfProfilerRef,
1365    artifact_kind: &'static str,
1366    path: &Path,
1367) {
1368    // Don't stat the file if we are not going to record its size.
1369    if !self_profiler_ref.enabled() {
1370        return;
1371    }
1372
1373    if let Some(artifact_name) = path.file_name() {
1374        let file_size = std::fs::metadata(path).map(|m| m.len()).unwrap_or(0);
1375        self_profiler_ref.artifact_size(artifact_kind, artifact_name.to_string_lossy(), file_size);
1376    }
1377}
1378
1379fn record_llvm_cgu_instructions_stats(prof: &SelfProfilerRef, name: &str, llmod: &llvm::Module) {
1380    if !prof.enabled() {
1381        return;
1382    }
1383
1384    let total = unsafe { llvm::LLVMRustModuleInstructionStats(llmod) };
1385    prof.artifact_size("cgu_instructions", name, total);
1386}