/
usr
/
local
/
lib64
/
python3.6
/
site-packages
/
torch
/
include
/
torch
/
csrc
/
jit
/
tensorexpr
/
/usr/local/lib64/python3.6/site-packages/torch/include/torch/csrc/jit/tensorexpr
mkdir
upload
Name
Size
Mode
Actions
operators/
-
0755
rm
analysis.h
5888
0644
edit
dl
rm
block_codegen.h
4211
0644
edit
dl
rm
bounds_inference.h
2230
0644
edit
dl
rm
bounds_overlap.h
3329
0644
edit
dl
rm
codegen.h
6402
0644
edit
dl
rm
cpp_codegen.h
2278
0644
edit
dl
rm
cpp_intrinsics.h
719
0644
edit
dl
rm
cuda_codegen.h
7782
0644
edit
dl
rm
cuda_random.h
2642
0644
edit
dl
rm
dim_arg.h
884
0644
edit
dl
rm
eval.h
9639
0644
edit
dl
rm
exceptions.h
3253
0644
edit
dl
rm
expr.h
11588
0644
edit
dl
rm
external_functions.h
1274
0644
edit
dl
rm
external_functions_registry.h
2343
0644
edit
dl
rm
fwd_decls.h
2806
0644
edit
dl
rm
graph_opt.h
2553
0644
edit
dl
rm
half_support.h
5038
0644
edit
dl
rm
hash_provider.h
7930
0644
edit
dl
rm
intrinsic_symbols.h
420
0644
edit
dl
rm
ir.h
22622
0644
edit
dl
rm
ir_cloner.h
2069
0644
edit
dl
rm
ir_mutator.h
2010
0644
edit
dl
rm
ir_printer.h
3693
0644
edit
dl
rm
ir_simplifier.h
15090
0644
edit
dl
rm
ir_verifier.h
1240
0644
edit
dl
rm
ir_visitor.h
1825
0644
edit
dl
rm
kernel.h
9210
0644
edit
dl
rm
llvm_codegen.h
3180
0644
edit
dl
rm
llvm_jit.h
1965
0644
edit
dl
rm
loopnest.h
21599
0644
edit
dl
rm
mem_dependency_checker.h
13003
0644
edit
dl
rm
reduction.h
6742
0644
edit
dl
rm
registerizer.h
12498
0644
edit
dl
rm
stmt.h
21138
0644
edit
dl
rm
tensor.h
7640
0644
edit
dl
rm
tensorexpr_init.h
268
0644
edit
dl
rm
types.h
3880
0644
edit
dl
rm
unique_name_manager.h
940
0644
edit
dl
rm
var_substitutor.h
1753
0644
edit
dl
rm
Edit:
/usr/local/lib64/python3.6/site-packages/torch/include/torch/csrc/jit/tensorexpr/graph_opt.h
(2553B)
#pragma once #include <torch/csrc/jit/ir/ir.h> namespace torch { namespace jit { namespace tensorexpr { // Optimize aten::cat ops in the given subgraph. // // Moving users of cat to its inputs. // Cat ops get lowered into multiple loops, one per input. When the result // of cat is used by some other op, it results in a situation where inlining // of cat does not happen. This in turn results in intermediate buffers // being created for the result of cat, since it is not inlined. // // For example, consider the following graph: // graph(%x : Float(10, strides=[1], device=cpu), // %y : Float(20, strides=[1], device=cpu)): // %dim : int = prim::Constant[value=0]() // %xy_list : Tensor[] = prim::ListConstruct(%x, %y) // %cat : Float(60, strides=[1], device=cpu) = aten::cat(%xy_list, %dim) // %5 : Float(60, strides=[1], device=cpu) = aten::log(%cat) // return (%5))IR"; // // This will get lowered into: // Allocate(aten_cat); // for (...) // aten_cat[...] = x[...] // for (...) // aten_cat[...] = y[...] // for (...) // aten_log[...] = log(aten_cat[...]) // Free(aten_cat); // Note that aten_cat is not inlined into aten_log and it results in // an intermediate buffer allocation as well. // // Optimization: // We move the ops that use the result of `cat` into its inputs whenever // possible. // // The graph above will be transformed to: // graph(%x : Float(10, strides=[1], device=cpu), // %y : Float(20, strides=[1], device=cpu)): // %3 : int = prim::Constant[value=0]() // %7 : Float(10, strides=[1], device=cpu) = aten::log(%x) // %8 : Float(20, strides=[1], device=cpu) = aten::log(%y) // %9 : Tensor[] = prim::ListConstruct(%7, %8) // %10 : Float(60, strides=[1], device=cpu) = aten::cat(%9, %3) // return (%10) // // This will get lowered into: // for (...) // aten_cat[...] = log(x[...]) // for (...) // aten_cat[...] = log(y[...]) // aten_cat is the output buffer here. bool OptimizeCat(const std::shared_ptr<Graph>& graph); TORCH_API void annotateInputShapes( const std::shared_ptr<Graph>& graph, const std::vector<c10::optional<at::Tensor>>& example_inputs); TORCH_API std::shared_ptr<Graph> removeUnusedSelfArgument( const std::shared_ptr<Graph>& graph); } // namespace tensorexpr } // namespace jit } // namespace torch
Save
cmd:
run