# Copyright 2026 Gentoo Authors
# Distributed under the terms of the GNU General Public License v2

# k-bit quantization kernels for PyTorch: QLoRA 4/8-bit loading in
# transformers (the localai-backend/transformers venv) and soup-cli
# training. Upstream ships prebuilt CUDA-only wheels; this builds the
# native library from source through the CMake COMPUTE_BACKEND matrix.
# The python package always carries the CPU library and USE=rocm/cuda
# adds the GPU one beside it — the exact layout upstream wheels use;
# the loader picks at runtime by the torch build (the HIP library name
# embeds hipconfig's version, e.g. libbitsandbytes_rocm72.so).

EAPI=8

DISTUTILS_USE_PEP517=standalone
DISTUTILS_SINGLE_IMPL=1
PYTHON_COMPAT=( python3_{12..14} )

inherit distutils-r1 local-ai-rocm multiprocessing

DESCRIPTION="k-bit optimizers and quantization for PyTorch"
HOMEPAGE="https://github.com/bitsandbytes-foundation/bitsandbytes"
SRC_URI="https://github.com/bitsandbytes-foundation/bitsandbytes/archive/${PV}.tar.gz -> ${P}.gh.tar.gz"

LICENSE="MIT"
SLOT="0"
KEYWORDS="~amd64"
IUSE="cuda rocm"
REQUIRED_USE="?? ( cuda rocm ) rocm? ( ${ROCM_REQUIRED_USE} )"

RDEPEND="
	$(python_gen_cond_dep '
		sci-ml/pytorch[${PYTHON_SINGLE_USEDEP}]
		dev-python/numpy[${PYTHON_USEDEP}]
		dev-python/packaging[${PYTHON_USEDEP}]
	')
	cuda? ( dev-util/nvidia-cuda-toolkit:= )
	rocm? (
		>=dev-util/hip-${ROCM_VERSION}:=
		>=sci-libs/hipBLAS-${ROCM_VERSION}:=
		>=sci-libs/hipRAND-${ROCM_VERSION}:=
		>=sci-libs/hipBLASLt-${ROCM_VERSION}:=
	)
"
DEPEND="${RDEPEND}"
BDEPEND="
	dev-build/cmake
	$(python_gen_cond_dep '
		dev-python/scikit-build-core[${PYTHON_USEDEP}]
		dev-python/setuptools[${PYTHON_USEDEP}]
		dev-python/trove-classifiers[${PYTHON_USEDEP}]
	')
"

src_configure() {
	# scikit-build-core reads CMAKE_ARGS: the pep517 wheel build runs ONE
	# CMake pass, so the wheel carries the primary (GPU when enabled)
	# library; the always-needed CPU library is a second plain CMake
	# build appended in python_install.
	local args=( -DCOMPUTE_BACKEND=cpu )
	if use rocm; then
		local amdgpu_flags=$(get_amdgpu_flags)
		args=(
			-DCOMPUTE_BACKEND=hip
			-DCMAKE_HIP_ARCHITECTURES="${amdgpu_flags%;}"
		)
	elif use cuda; then
		args=( -DCOMPUTE_BACKEND=cuda )
	fi
	export CMAKE_ARGS="${args[*]}"
	distutils-r1_src_configure
}

src_compile() {
	distutils-r1_src_compile

	if use rocm || use cuda; then
		# The loader falls back to libbitsandbytes_cpu.so whenever torch
		# reports no GPU; upstream wheels ship it unconditionally.
		cmake -S "${S}" -B "${WORKDIR}/cpu-build" -DCOMPUTE_BACKEND=cpu || die
		cmake --build "${WORKDIR}/cpu-build" -j "$(makeopts_jobs)" || die
	fi
}

python_install() {
	distutils-r1_python_install

	if use rocm || use cuda; then
		python_moduleinto bitsandbytes
		# Upstream CMake forces every output into the SOURCE package dir
		# (so the wheel's package-data picks libraries up); -B only moves
		# the object files. The CPU fallback therefore lands in ${S}.
		python_domodule "${S}"/bitsandbytes/libbitsandbytes_cpu.so
	fi
}