Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 5 additions & 1 deletion .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -84,8 +84,12 @@ android/gradle/
!.yarn/sdks
!.yarn/versions

# Generated by the build pipelines from generate_tokenizers_header_file.js
android/c_sources
# Android does not generate the sources on its own, leave this in for CI
# c_sources/ is also generated (copied out of the app's own c_sources by the
# podspec/Package.swift) but stays tracked: SwiftPM evaluates Package.swift in a
# sandbox that can deny that copy, so an SPM build off a fresh checkout needs the
# files to already be there. Android no longer depends on it.
# c_sources/

scripts/sqlite-vec-*
Expand Down
65 changes: 41 additions & 24 deletions Package.swift
Original file line number Diff line number Diff line change
Expand Up @@ -103,34 +103,51 @@ if !tokenizers.isEmpty && useTurso {
print("[OP-SQLITE] SPM configuration found at \(appPackageJSONPath)")

// MARK: - Tokenizer header generation
// Direct translation of generate_tokenizers_header_file.rb.
// The header contents come from generate_tokenizers_header_file.js, the one
// implementation shared by every build pipeline (CocoaPods and Gradle call the
// same script). This only has to find node and run it.

func generateTokenizersHeaderFile(names: [String], filePath: String) {
let fileURL = URL(fileURLWithPath: filePath)
try? FileManager.default.createDirectory(
at: fileURL.deletingLastPathComponent(), withIntermediateDirectories: true)

let tokenizerList = names.map { "opsqlite_\($0)_init(db,&errMsg,nullptr);" }.joined()

var content = ""
content += "#ifndef TOKENIZERS_H\n"
content += "#define TOKENIZERS_H\n"
content += "\n"
content += "#define TOKENIZER_LIST \(tokenizerList)\n"
content += "\n"
content += "#include <sqlite3.h>\n"
content += "\n"
content += "namespace opsqlite {\n"
content += "\n"
for name in names {
content += "int opsqlite_\(name)_init(sqlite3 *db, char **error, sqlite3_api_routines const *api);\n"
let scriptPath = packageRoot.appendingPathComponent("generate_tokenizers_header_file.js").path

// NODE_BINARY is React Native's own convention (it writes one into
// ios/.xcode.env). Xcode does not necessarily inherit the user's shell PATH,
// so fall back to a PATH lookup and then to the usual install locations.
var candidates: [String] = []
if let fromEnv = ProcessInfo.processInfo.environment["NODE_BINARY"], !fromEnv.isEmpty {
candidates.append(fromEnv)
}
candidates += ["node", "/opt/homebrew/bin/node", "/usr/local/bin/node", "/usr/bin/node"]

for node in candidates {
let process = Process()
// Going through `env` is what resolves a bare "node" against PATH.
process.executableURL = URL(fileURLWithPath: "/usr/bin/env")
process.arguments = [node, scriptPath, filePath] + names
process.standardOutput = FileHandle.standardError

do {
try process.run()
process.waitUntilExit()
if process.terminationStatus == 0 {
return
}
} catch {
continue
}
}
content += "\n"
content += "} // namespace opsqlite\n"
content += "\n"
content += "#endif // TOKENIZERS_H\n"

try? content.write(to: fileURL, atomically: true, encoding: .utf8)
// Not fatal, on purpose. Manifest evaluation happens in contexts we do not
// control (Xcode's package resolution, `swift package dump-package`), and
// SwiftPM sandboxes writes there, so the generation can legitimately fail
// while an already up-to-date header sits on disk -- the script exits 0
// without writing in that case, so reaching this point means it really could
// not produce the file. Killing the manifest would be worse than letting the
// compiler report the missing header.
FileHandle.standardError.write(
Data(
("[OP-SQLITE] warning: could not generate \(filePath) with \(scriptPath). Make sure node is "
+ "on your PATH or set NODE_BINARY to its location.\n").utf8))
}

// Mirrors FileUtils.cp_r: copies the contents of `source` into `destination`,
Expand Down
7 changes: 7 additions & 0 deletions android/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -93,6 +93,13 @@ find_library(LOG_LIB log)
if (USER_DEFINED_SOURCE_FILES)
target_sources(${PACKAGE_NAME} PRIVATE ${USER_DEFINED_SOURCE_FILES})

# tokenizers.h is generated next to those sources, outside ../cpp, so it is
# found through the include path instead of a path relative to whichever file
# does the #include.
if (USER_DEFINED_TOKENIZERS_HEADER_DIR)
include_directories(${USER_DEFINED_TOKENIZERS_HEADER_DIR})
endif()

add_definitions("-DTOKENIZERS_HEADER_PATH=\"${USER_DEFINED_TOKENIZERS_HEADER_PATH}\"")
endif()

Expand Down
47 changes: 46 additions & 1 deletion android/build.gradle
Original file line number Diff line number Diff line change
Expand Up @@ -143,6 +143,38 @@ if(!tokenizers.isEmpty()) {
println "[OP-SQLITE] Tokenizers enabled. Detected tokenizers: " + tokenizers
}

// Generates the tokenizers.h that cpp/OPBridge.cpp includes (through
// TOKENIZERS_HEADER_PATH) and that the copied tokenizers.cpp includes by name.
//
// The generation itself lives in generate_tokenizers_header_file.js, shared with
// the CocoaPods (generate_tokenizers_header_file.rb) and SwiftPM (Package.swift)
// pipelines, so all three emit the same header. Before this, Android had no
// generator of its own and could only build if an iOS pod install (or a header
// committed to the repo) had produced the file first.
//
// Uses a plain ProcessBuilder rather than project.exec: this runs at
// configuration time, where that API is deprecated/removed in recent Gradle.
def generateTokenizersHeaderFile(List<String> names, File headerFile) {
def packageDir = buildscript.sourceFile.parentFile.parentFile
def script = new File(packageDir, "generate_tokenizers_header_file.js")
// NODE_BINARY is React Native's own convention for pointing at a specific node.
def node = System.getenv("NODE_BINARY")
if (!node) {
node = "node"
}

def command = [node, script.absolutePath, headerFile.absolutePath] + names
def process = new ProcessBuilder(command).redirectErrorStream(true).start()
def output = process.inputStream.text
def status = process.waitFor()

if (status != 0) {
throw new GradleException(
"[OP-SQLITE] Could not generate ${headerFile}. Command: ${command.join(' ')}\n" +
"Make sure node is on your PATH or set NODE_BINARY to its location.\n${output}")
}
}

// Build the list of library-default SQLite compile flags driven by package.json
// toggles. These are emitted via CMake add_definitions() BEFORE the user-supplied
// sqliteFlags, so that user flags take precedence when both define the same macro
Expand Down Expand Up @@ -194,17 +226,29 @@ android {
// This are zeroes because they will be passed as C flags, so they become falsy
def sourceFiles = 0
// def tokenizerInitStrings = 0
def tokenizersHeaderDir = 0
def tokenizersHeaderPath = 0
if (!tokenizers.isEmpty()) {
def sourceDir = isUserApp ? file("$rootDir/../../../c_sources") : file("$rootDir/../c_sources")
def destDir = file("$buildscript.sourceFile.parentFile/c_sources")
// Generated into the app's own c_sources, exactly where the podspec and
// Package.swift put it, so the header the user's tokenizers.cpp includes
// is there no matter which platform was built last. The copy below then
// carries it over with the sources.
generateTokenizersHeaderFile(tokenizers, new File(sourceDir, "tokenizers.h"))
copy {
from sourceDir
into destDir
include "**/*.cpp", "**/*.h"
}
sourceFiles = fileTree(dir: destDir, include: ["**/*.cpp", "**/*.h"]).files.join(";")
tokenizersHeaderPath = "../c_sources/tokenizers.h"
// The header sits next to the copied tokenizer sources, not in cpp/, so
// it is reached through the include path. Passing a path relative to the
// file doing the #include would hardcode where in cpp/ that include is,
// and an absolute path in a C string literal would break on Windows
// separators.
tokenizersHeaderDir = destDir.absolutePath.replace('\\', '/')
tokenizersHeaderPath = "tokenizers.h"
}

cppFlags "-O3 -frtti -fexceptions -Wall -fstack-protector-all"
Expand All @@ -217,6 +261,7 @@ android {
"-DUSE_TURSO=${useTurso ? 1 : 0}",
"-DUSE_SQLITE_VEC=${useSqliteVec ? 1 : 0}",
"-DUSER_DEFINED_SOURCE_FILES=${sourceFiles}",
"-DUSER_DEFINED_TOKENIZERS_HEADER_DIR='${tokenizersHeaderDir}'",
"-DUSER_DEFINED_TOKENIZERS_HEADER_PATH='${tokenizersHeaderPath}'",
"-DANDROID_SUPPORT_FLEXIBLE_PAGE_SIZES=ON"
}
Expand Down
2 changes: 1 addition & 1 deletion c_sources/tokenizers.h
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
#ifndef TOKENIZERS_H
#define TOKENIZERS_H

#define TOKENIZER_LIST opsqlite_wordtokenizer_init(db,&errMsg,nullptr);opsqlite_porter_init(db,&errMsg,nullptr);
#define TOKENIZER_LIST opsqlite_wordtokenizer_init(db,&err_msg,nullptr);opsqlite_porter_init(db,&err_msg,nullptr);

#include <sqlite3.h>

Expand Down
8 changes: 4 additions & 4 deletions cpp/OPBridge.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -140,7 +140,7 @@ sqlite3 *opsqlite_open(std::string const &name, std::string const &path,
// Written to only on failure by both sqlite3_load_extension below and the
// tokenizer init calls TOKENIZER_LIST expands into, so it starts out null.
// Unused when neither of those is configured into the build.
[[maybe_unused]] char *errMsg = nullptr;
[[maybe_unused]] char *err_msg = nullptr;
sqlite3 *db;

int flags = SQLITE_OPEN_FULLMUTEX;
Expand Down Expand Up @@ -187,11 +187,11 @@ sqlite3 *opsqlite_open(std::string const &name, std::string const &path,
const char *vec_entry_point = "sqlite3_vec_init";

int vec_status = sqlite3_load_extension(db, _sqlite_vec_path.c_str(),
vec_entry_point, &errMsg);
vec_entry_point, &err_msg);

if (vec_status != SQLITE_OK) {
std::string message = errMsg != nullptr ? errMsg : "unknown error";
sqlite3_free(errMsg);
std::string message = err_msg != nullptr ? err_msg : "unknown error";
sqlite3_free(err_msg);

throw build_error("could not load sqlite-vec", message, vec_status & 0xff,
vec_status);
Expand Down
4 changes: 2 additions & 2 deletions docs/docs/tokenizers.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@ op-sqlite has a novel way for you to create your tokenizers.
}
```

2. Run `pod install`. The podspec now contains a code generation step. It will create a `c_sources` folder at the root of your project. It will create a `tokenizers.h` file. DON’T TOUCH THIS FILE. It will be overwritten every time. You need to create a `c_sources/tokenizers.cpp` file. Here you need to provide your tokenizer implementation. The `tokenizer.h` file contains the function declaration that will be executed when registering your tokenizer. In this case here is a sample `tokenizers.cpp` implementation
2. Run `pod install` (iOS) or a Gradle build (Android). Both contain a code generation step. It will create a `c_sources` folder at the root of your project. It will create a `tokenizers.h` file. DON’T TOUCH THIS FILE. It will be overwritten every time. You need to create a `c_sources/tokenizers.cpp` file. Here you need to provide your tokenizer implementation. The `tokenizer.h` file contains the function declaration that will be executed when registering your tokenizer. In this case here is a sample `tokenizers.cpp` implementation

```cpp
#include "tokenizers.h"
Expand Down Expand Up @@ -102,7 +102,7 @@ op-sqlite has a novel way for you to create your tokenizers.
You need to keep the namespace and the function signature intact. For now the `sqlite3_api_routines` parameter will always be a null pointer.

3. Once you are done. You need to run `pod install` again. It will then copy the files you created to the pod sources in order to compile `op-sqlite` together with your new C++ code in one go.
4. The code generation step is only implemented in Cocoapods. Every time you create/change a file inside of `c_sources` you will need to do a `pod install` to re-add the newly created files into the compilation process. This also applies for Android, at least the header file generation step. On your CI, you will also need to do a pod install even if your pods are cached, in order to copy the sources.
4. The `tokenizers.h` generation runs in every pipeline: CocoaPods, SwiftPM and Gradle all call the same `generate_tokenizers_header_file.js` from the package, so an Android build no longer needs an iOS `pod install` to have happened first. Adding or removing a **file** inside `c_sources` still needs a `pod install` on iOS, since the pod's file list is fixed at install time; Android picks up new files on the next Gradle build. On your CI, you will also need to do a pod install even if your pods are cached, in order to copy the sources.
5. You can then create a FTS5 virtual table with your tokenizer:

```tsx
Expand Down
2 changes: 1 addition & 1 deletion example/c_sources/tokenizers.h
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
#ifndef TOKENIZERS_H
#define TOKENIZERS_H

#define TOKENIZER_LIST opsqlite_wordtokenizer_init(db,&errMsg,nullptr);opsqlite_porter_init(db,&errMsg,nullptr);
#define TOKENIZER_LIST opsqlite_wordtokenizer_init(db,&err_msg,nullptr);opsqlite_porter_init(db,&err_msg,nullptr);

#include <sqlite3.h>

Expand Down
80 changes: 80 additions & 0 deletions generate_tokenizers_header_file.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,80 @@
#!/usr/bin/env node

// Single source of truth for the generated `tokenizers.h`.
//
// Every build pipeline needs this file: CocoaPods (via
// generate_tokenizers_header_file.rb), SwiftPM (Package.swift) and Gradle
// (android/build.gradle). They all shell out to this script instead of
// reimplementing the generation, so the header cannot drift between platforms
// and TOKENIZER_LIST only has to be kept in sync with cpp/OPBridge.cpp here.
//
// Usage: node generate_tokenizers_header_file.js <output-file> <name>...

const fs = require("node:fs");
const path = require("node:path");

/**
* The expansion of TOKENIZER_LIST is pasted into opsqlite_open(), so `db` and
* `err_msg` refer to that function's locals.
*/
function generateTokenizersHeader(names) {
const tokenizerList = names.map((name) => `opsqlite_${name}_init(db,&err_msg,nullptr);`).join("");

const declarations = names.map(
(name) =>
`int opsqlite_${name}_init(sqlite3 *db, char **error, sqlite3_api_routines const *api);`,
);

return [
"#ifndef TOKENIZERS_H",
"#define TOKENIZERS_H",
"",
`#define TOKENIZER_LIST ${tokenizerList}`,
"",
"#include <sqlite3.h>",
"",
"namespace opsqlite {",
"",
...declarations,
"",
"} // namespace opsqlite",
"",
"#endif // TOKENIZERS_H",
"",
].join("\n");
}

/**
* Returns true when the file was written, false when it was already up to date.
*
* Rewriting an identical file would bump its mtime and force a recompile of
* everything that includes it. Gradle in particular runs this on every
* configuration.
*/
function writeTokenizersHeader(names, filePath) {
const contents = generateTokenizersHeader(names);

if (fs.existsSync(filePath) && fs.readFileSync(filePath, "utf8") === contents) {
return false;
}

fs.mkdirSync(path.dirname(filePath), { recursive: true });
fs.writeFileSync(filePath, contents);

return true;
}

module.exports = { generateTokenizersHeader, writeTokenizersHeader };

if (require.main === module) {
const [filePath, ...names] = process.argv.slice(2);

if (!filePath) {
console.error(
"[OP-SQLITE] usage: node generate_tokenizers_header_file.js <output-file> <tokenizer-name>...",
);
process.exit(1);
}

writeTokenizersHeader(names, filePath);
}
38 changes: 14 additions & 24 deletions generate_tokenizers_header_file.rb
Original file line number Diff line number Diff line change
@@ -1,29 +1,19 @@
require 'fileutils'
require 'shellwords'

# Thin wrapper around generate_tokenizers_header_file.js, which holds the actual
# generation logic so CocoaPods, SwiftPM and Gradle all emit the exact same
# header. Node is already a hard requirement for building a React Native app.
def generate_tokenizers_header_file(names, file_path)
# Ensure the directory exists
dir_path = File.dirname(file_path)
FileUtils.mkdir_p(dir_path) unless Dir.exist?(dir_path)
tokenizer_list = names.map { |name| "opsqlite_#{name}_init(db,&errMsg,nullptr);" }.join
script_path = File.join(__dir__, "generate_tokenizers_header_file.js")
# NODE_BINARY is React Native's own convention for pointing at a specific
# node (see the .xcode.env files it generates).
node = ENV["NODE_BINARY"]
node = "node" if node.nil? || node.empty?

File.open(file_path, 'w') do |file|
file.puts "#ifndef TOKENIZERS_H"
file.puts "#define TOKENIZERS_H"
file.puts
file.puts "#define TOKENIZER_LIST #{tokenizer_list}"
file.puts
file.puts "#include <sqlite3.h>"
file.puts
file.puts "namespace opsqlite {"
file.puts
command = [node, script_path, file_path, *names]

names.each do |name|
file.puts "int opsqlite_#{name}_init(sqlite3 *db, char **error, sqlite3_api_routines const *api);"
end

file.puts
file.puts "} // namespace opsqlite"
file.puts
file.puts "#endif // TOKENIZERS_H"
unless system(*command)
raise "[OP-SQLITE] Could not generate the tokenizers header. Command: #{command.shelljoin}. " \
"Make sure node is on your PATH or set NODE_BINARY to its location."
end
end
end
1 change: 1 addition & 0 deletions package.json
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@
"node/package.json",
"*.podspec",
"*.rb",
"generate_tokenizers_header_file.js",
"react-native.config.js",
"ios/**.xcframework",
"Package.swift",
Expand Down
Loading