Repository navigation
Expand file tree
/
Copy pathCMakeLists.txt
More file actions
256 lines (242 loc) · 11.4 KB
/
Copy pathCMakeLists.txt
File metadata and controls
256 lines (242 loc) · 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
# Copyright (c) Meta Platforms, Inc. and affiliates.
# All rights reserved.
#
# This source code is licensed under the BSD-style license found in the
# LICENSE file in the root directory of this source tree.
# Kernel library for quantized operators. Please keep this file formatted by
# running:
# ~~~
# cmake-format -i CMakeLists.txt
# ~~~
cmake_minimum_required(VERSION 3.19)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
if(NOT CMAKE_CXX_STANDARD)
set(CMAKE_CXX_STANDARD 17)
endif()
# Source root directory for executorch.
if(NOT EXECUTORCH_ROOT)
set(EXECUTORCH_ROOT ${CMAKE_CURRENT_SOURCE_DIR}/../..)
endif()
set(_common_compile_options
$<$<CXX_COMPILER_ID:MSVC>:/wd4996>
$<$<NOT:$<CXX_COMPILER_ID:MSVC>>:-Wno-deprecated-declarations>
)
include(${EXECUTORCH_ROOT}/tools/cmake/Utils.cmake)
include(${EXECUTORCH_ROOT}/tools/cmake/Codegen.cmake)
# Quantized ops kernel sources TODO(larryliu0820): use buck2 to gather the
# sources
list(TRANSFORM _quantized_kernels__srcs PREPEND "${EXECUTORCH_ROOT}/")
# Generate C++ bindings to register kernels into both PyTorch (for AOT) and
# Executorch (for runtime). Here select all ops in quantized.yaml
set(_yaml_file ${CMAKE_CURRENT_LIST_DIR}/quantized.yaml)
gen_selected_ops(LIB_NAME "quantized_ops_lib" OPS_SCHEMA_YAML "${_yaml_file}")
# Expect gen_selected_ops output file to be selected_operators.yaml
generate_bindings_for_kernels(
LIB_NAME "quantized_ops_lib" CUSTOM_OPS_YAML "${_yaml_file}"
)
message("Generated files ${gen_command_sources}")
# Not targeting Xcode, because the custom command generating
# kernels/quantized/selected_operators.yaml is attached to multiple targets:
# quantized_ops_aot_lib quantized_ops_lib but none of these is a common
# dependency of the other(s). This is not allowed by the Xcode "new build
# system".
if(NOT CMAKE_GENERATOR STREQUAL "Xcode"
AND EXECUTORCH_BUILD_KERNELS_QUANTIZED_AOT
)
# Not targeting ARM_BAREMETAL as aot_lib depends on incompatible libraries
if(NOT EXECUTORCH_BUILD_ARM_BAREMETAL)
set(_quantized_aot_ops
"quantized_decomposed::add.out"
"quantized_decomposed::choose_qparams.Tensor_out"
"quantized_decomposed::choose_qparams_per_token_asymmetric.out"
"quantized_decomposed::dequantize_per_channel.out"
"quantized_decomposed::dequantize_per_tensor.out"
"quantized_decomposed::dequantize_per_tensor.Tensor_out"
"quantized_decomposed::dequantize_per_token.out"
"quantized_decomposed::mixed_linear.out"
"quantized_decomposed::mixed_mm.out"
"quantized_decomposed::quantize_per_channel.out"
"quantized_decomposed::quantize_per_tensor.out"
"quantized_decomposed::quantize_per_tensor.Tensor_out"
"quantized_decomposed::quantize_per_token.out"
)
gen_selected_ops(
LIB_NAME "quantized_ops_aot_lib" ROOT_OPS ${_quantized_aot_ops}
)
# Expect gen_selected_ops output file to be
# quantized_ops_aot_lib/selected_operators.yaml
generate_bindings_for_kernels(
LIB_NAME "quantized_ops_aot_lib" CUSTOM_OPS_YAML "${_yaml_file}"
)
# Build a AOT library to register quantized ops into PyTorch. This is a
# hack.
set(_quantized_sources
${_quantized_kernels__srcs}
${EXECUTORCH_ROOT}/kernels/portable/cpu/util/reduce_util.cpp
${EXECUTORCH_ROOT}/runtime/core/exec_aten/util/tensor_util_aten.cpp
)
gen_custom_ops_aot_lib(
LIB_NAME "quantized_ops_aot_lib" KERNEL_SOURCES "${_quantized_sources}"
)
# The route to the runtime, through the same helper every other target uses.
# Written by hand here before, in two separate blocks that between them
# recorded only the wheel layout and only when the wheel flag was set, so a
# plain -DEXECUTORCH_BUILD_SHARED=ON build produced a library with no route
# to the runtime it links.
executorch_target_shared_runtime_path(
quantized_ops_aot_lib "kernels/quantized" "executorch/kernels/quantized"
)
# Register quantized ops to portable_lib, so that they're available via
# pybindings.
#
# Only when this build has no shared quantized library to link instead. That
# library and this block compile the same sources against the same yaml, so
# a build producing both put two identical registrars in one process: the
# operator sets are identical by construction, and measured on the shipped
# wheel neither library carried one the other lacked. Registration is a
# static initializer, so both fire on load and the second aborts the
# process. Where the shared library exists the AOT plugin links it below
# rather than carrying its own copy.
if(TARGET portable_lib AND NOT EXECUTORCH_BUILD_SHARED)
add_library(quantized_pybind_kernels_lib ${_quantized_kernels__srcs})
target_link_libraries(
quantized_pybind_kernels_lib PRIVATE portable_lib executorch_core
kernels_util_all_deps
)
target_compile_options(
quantized_pybind_kernels_lib PUBLIC ${_common_compile_options}
)
target_include_directories(
quantized_pybind_kernels_lib PUBLIC "${_common_include_directories}"
)
gen_selected_ops(
LIB_NAME "quantized_ops_pybind_lib" OPS_SCHEMA_YAML "${_yaml_file}"
)
generate_bindings_for_kernels(
LIB_NAME "quantized_ops_pybind_lib" CUSTOM_OPS_YAML "${_yaml_file}"
)
# Build a library for pybind usage. quantized_ops_pybind_lib: Register
# quantized ops kernels into Executorch runtime for pybind.
gen_operators_lib(
LIB_NAME "quantized_ops_pybind_lib" KERNEL_LIBS
quantized_pybind_kernels_lib DEPS portable_lib
)
target_link_libraries(
quantized_ops_aot_lib PUBLIC quantized_ops_pybind_lib
)
endif()
# Outside the gate above. That gate only decides whether this build compiles
# its own copy of the kernels; the runtime search path below is needed
# either way, because the plugin links torch and the Python extension
# directly through Codegen.cmake regardless. Removing it left an Apple build
# with no route to torch/lib, where the loader cannot fall back on a
# non-absolute name the way Linux can.
if(TARGET portable_lib)
# pip wheels will need to be able to find the dependent libraries. On
# Linux, the .so has non-absolute dependencies on libs like "_C.so"
# without paths; as long as we `import torch` first, those dependencies
# will work. But Apple dylibs do not support non-absolute dependencies, so
# we need to tell the loader where to look for its libraries. The
# LC_LOAD_DYLIB entries for the extension will look like
# "@rpath/_C.cpython-310-darwin.so", so we can add an LC_RPATH entry to
# look in a directory relative to where that extension is installed. To
# see these LC_* values, run `otool -l libquantized_ops_lib.dylib`.
# "extension", not "extensions": the plural directory does not exist, so
# the parent's path reached nothing and this library could not find the
# extension it needs. torch is three directories up from here, and this
# library links it directly, so the hop has to be recorded or the only
# route is the absolute path from the build machine.
if(APPLE)
set(RPATH
"@loader_path/../../extension/pybindings;@loader_path/../../../torch/lib"
)
else()
set(RPATH
"$ORIGIN/../../extension/pybindings:$ORIGIN/../../../torch/lib"
)
endif()
# Appended rather than assigned. The helper above already recorded the
# route to the runtime, and overwriting the property here would drop it.
get_target_property(_existing quantized_ops_aot_lib INSTALL_RPATH)
if(_existing)
# Mach-O keeps one entry per load command, so the two are a CMake list
# there. Joining them with a colon produced a single unusable path
# containing both, and the library then found neither the runtime nor
# the extension.
if(APPLE)
set(RPATH "${_existing};${RPATH}")
else()
set(RPATH "${_existing}:${RPATH}")
endif()
endif()
# Quoted, because on Apple this is a list: unquoted it expands into
# separate arguments and the trailing entries are read as further property
# keywords, leaving the search path empty.
set_target_properties(
quantized_ops_aot_lib PROPERTIES BUILD_RPATH "${RPATH}" INSTALL_RPATH
"${RPATH}"
)
endif()
endif()
endif()
add_library(quantized_kernels ${_quantized_kernels__srcs})
target_link_libraries(
quantized_kernels PRIVATE executorch_core kernels_util_all_deps
)
# The thread pool carries the define that switches parallel_for from a serial
# fallback to the real threaded implementation. Without it choose_qparams runs
# on one core, as does the ARM path in quantize. Guarded because a bare metal
# target builds these kernels without a thread pool at all, where the serial
# fallback is the only correct choice.
if(TARGET extension_threadpool)
target_link_libraries(quantized_kernels PRIVATE extension_threadpool)
endif()
target_compile_options(quantized_kernels PUBLIC ${_common_compile_options})
# Build a library for _quantized_kernels_srcs
#
# quantized_ops_lib: Register quantized ops kernels into Executorch runtime
gen_operators_lib(
LIB_NAME "quantized_ops_lib" KERNEL_LIBS quantized_kernels DEPS
executorch_core
)
install(
TARGETS quantized_kernels quantized_ops_lib
EXPORT ExecuTorchTargets
DESTINATION ${CMAKE_INSTALL_LIBDIR}
PUBLIC_HEADER
DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/executorch/kernels/quantized/
)
if(EXECUTORCH_BUILD_SHARED)
executorch_add_shared_library(
executorch_quantized_ops quantized_ops_lib quantized_kernels
executorch_shared
)
# This library holds no symbol a consumer names, only registration
# constructors, so a link with --as-needed would drop it and register nothing.
executorch_target_link_options_shared_lib(executorch_quantized_ops)
# The export plugin gets its kernels from this one library rather than
# compiling its own copy, so the process holds a single registrar. Attached
# here rather than beside the plugin's other properties because that block
# runs before this target exists. The plugin already routes kernels/quantized/
# to lib/ for its runtime search path, which is where this library ships, and
# the --as-needed retention above is an interface property so it reaches the
# plugin too.
if(TARGET quantized_ops_aot_lib)
target_link_libraries(quantized_ops_aot_lib PUBLIC executorch_quantized_ops)
endif()
# Named after what the library provides rather than after the target that
# produces it, matching the optimized kernels next to it, so the shipped file
# reads as libexecutorch_kernels_quantized.so. The target name stays as it is
# because a source build already refers to it. Only for the wheel, for the
# same reason as the optimized kernels: a source install has been shipping the
# old file name with a versioned soname.
if(EXECUTORCH_BUILD_WHEEL_DO_NOT_USE)
set_target_properties(
executorch_quantized_ops PROPERTIES OUTPUT_NAME
executorch_kernels_quantized
)
endif()
# Ships beside libexecutorch.so in the wheel's lib/ directory, so it resolves
# the runtime from there rather than from wherever it was built.
executorch_target_shipped_runtime_path(executorch_quantized_ops)
endif()