From 8fa313587267fcf095fbfbc1bd2cb556843d9b5f Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 07:27:27 +0800 Subject: [PATCH 01/21] Expose in-process dictionary generation for native preparation callers --- Core/Portable/dictionary/Cargo.toml | 3 + Core/Portable/dictionary/README.md | 10 + .../dictionary/include/inkflow_dictionary.h | 42 +++ .../dictionary/include/module.modulemap | 4 + Core/Portable/dictionary/src/ffi.rs | 319 ++++++++++++++++++ Core/Portable/dictionary/src/lib.rs | 1 + Core/Portable/dictionary/test.sh | 17 +- .../dictionary/tests/NativeGenerator.swift | 95 ++++++ Core/Portable/dictionary/tests/native.c | 45 +++ Core/Portable/dictionary/verify.py | 14 + 10 files changed, 549 insertions(+), 1 deletion(-) create mode 100644 Core/Portable/dictionary/include/inkflow_dictionary.h create mode 100644 Core/Portable/dictionary/include/module.modulemap create mode 100644 Core/Portable/dictionary/src/ffi.rs create mode 100644 Core/Portable/dictionary/tests/NativeGenerator.swift create mode 100644 Core/Portable/dictionary/tests/native.c diff --git a/Core/Portable/dictionary/Cargo.toml b/Core/Portable/dictionary/Cargo.toml index 577a6e7..0353c1b 100644 --- a/Core/Portable/dictionary/Cargo.toml +++ b/Core/Portable/dictionary/Cargo.toml @@ -4,6 +4,9 @@ version = "0.1.0" edition = "2024" publish = false +[lib] +crate-type = ["rlib", "staticlib"] + [dependencies] serde = { version = "1.0", features = ["derive"] } serde_json = "1.0" diff --git a/Core/Portable/dictionary/README.md b/Core/Portable/dictionary/README.md index 5391989..147706a 100644 --- a/Core/Portable/dictionary/README.md +++ b/Core/Portable/dictionary/README.md @@ -30,6 +30,16 @@ build/dictionary-parity/cargo/release/inkflow-dictionary spelling \ It refuses an existing output directory. Validation completes before creating output; a write failure removes only the new directory it created. +## In-process preparation boundary + +`include/inkflow_dictionary.h` and its Clang module map expose an experimental C ABI from `libinkflow_dictionary.a`. It calls the same Rust generation, receipt validation, and spelling functions as the CLI. It does not launch a process, touch the filesystem, or initialize Rime. Call it from preparation/update workers, never from key handling. Shipping callers have not switched to this ABI yet. + +Inputs are borrowed pointer/length buffers. Catalog and receipt buffers contain UTF-8 JSON; dictionary and correction inputs remain raw bytes so validation can reject malformed text. The boundary accepts at most 64 inputs, 1 MiB of catalog/corrections, 16 KiB per receipt, and the existing 128 MiB per source. Malformed transport inputs return `bridge-input` or `bridge-json`; domain failures retain their code, source, and line. Recoverable Rust panics return `bridge-panic`. Invalid foreign pointers, allocator aborts, and process crashes are outside that guarantee. + +Every operation returns an owned opaque result. Successful generation supplies the dictionary and manifest; spelling supplies 32 named schema files; receipt validation supplies no files. Failure supplies error JSON and no files. Output pointers and names are length-delimited, not NUL-terminated, and remain valid until the caller frees the result. Result accessors require a live non-null handle. Independent calls share no mutable generator state. + +The test script compiles and links a C consumer on both desktops. On macOS it also builds a standalone Swift consumer, releases input storage before reading results, and compares the complete corpus and all spelling bytes through the ABI against the Rust CLI and original Swift implementation. `swift-ffi-corpus.json` retains that consumer's actual summary. This proves the preparation boundary, not shipping worker sandbox, packaging, or resource activation behavior. + ## Reference and compatibility contract `fixtures/cases.json` supplies 43 inputs to both implementations. The Swift test executable exports `catalog.json` and `reference.json`; Rust does not maintain another hand-written source catalog. The authoritative production catalog remains `IFDictionaryCatalog` in Swift. The copied catalog is a pinned fixture; the Mac check rejects drift from the live catalog. diff --git a/Core/Portable/dictionary/include/inkflow_dictionary.h b/Core/Portable/dictionary/include/inkflow_dictionary.h new file mode 100644 index 0000000..53c4f19 --- /dev/null +++ b/Core/Portable/dictionary/include/inkflow_dictionary.h @@ -0,0 +1,42 @@ +#ifndef INKFLOW_DICTIONARY_H +#define INKFLOW_DICTIONARY_H +#include +#include +#ifdef __cplusplus +extern "C" { +#endif + +/* Experimental in-process preparation ABI; never call on the key-event path. + * Inputs are borrowed for the call. Nonempty buffers/arrays must point to valid, + * aligned readable storage; NULL is accepted only for an empty buffer/array. + * JSON uses the existing catalog and receipt formats. Dictionary/correction + * buffers are raw bytes, including invalid UTF-8, for normal validation. + * Each operation returns an owned result, including on validation failure. + * Output buffers are borrowed until result_free; they are NOT NUL-terminated. + * Accessors require a live, non-NULL result. Callers must not mutate buffers + * or free a result while it is being read. + * Independent calls/results may be used on different threads. No global state, + * filesystem, network, Rime, or process launching is involved. + * Rust panics become bridge-panic errors; invalid pointers, allocator aborts, + * and process crashes are not recoverable. Free every result exactly once. + */ +typedef struct { const uint8_t* data; size_t len; } IFDBytes; +typedef struct { IFDBytes receipt; IFDBytes data; } IFDInput; +typedef struct IFDResult IFDResult; + +IFDResult* ifd_generate(IFDBytes catalog, const IFDInput* inputs, size_t count, + IFDBytes corrections); +IFDResult* ifd_spelling(IFDBytes dictionary); +IFDResult* ifd_validate(IFDInput input); +/* Empty on success; otherwise JSON {code, source, line}. Failed results have + * no files. bridge-input/bridge-json report malformed ABI requests. */ +IFDBytes ifd_result_error(const IFDResult* result); +size_t ifd_result_count(const IFDResult* result); +/* Out-of-range indices return empty buffers. Names are UTF-8 basenames. */ +IFDBytes ifd_result_name(const IFDResult* result, size_t index); +IFDBytes ifd_result_data(const IFDResult* result, size_t index); +void ifd_result_free(IFDResult* result); /* NULL is a no-op. */ +#ifdef __cplusplus +} +#endif +#endif diff --git a/Core/Portable/dictionary/include/module.modulemap b/Core/Portable/dictionary/include/module.modulemap new file mode 100644 index 0000000..f60399c --- /dev/null +++ b/Core/Portable/dictionary/include/module.modulemap @@ -0,0 +1,4 @@ +module InkFlowDictionary { + header "inkflow_dictionary.h" + export * +} diff --git a/Core/Portable/dictionary/src/ffi.rs b/Core/Portable/dictionary/src/ffi.rs new file mode 100644 index 0000000..3801cb4 --- /dev/null +++ b/Core/Portable/dictionary/src/ffi.rs @@ -0,0 +1,319 @@ +use crate::{Error, Input, MAX_SOURCE_BYTES, Result, SourceSpec}; +use std::{collections::BTreeMap, panic::catch_unwind, ptr, slice}; + +#[repr(C)] +#[derive(Clone, Copy)] +pub struct Bytes { + data: *const u8, + len: usize, +} +impl Bytes { + fn borrowed(data: &[u8]) -> Self { + Self { + data: if data.is_empty() { + ptr::null() + } else { + data.as_ptr() + }, + len: data.len(), + } + } + unsafe fn read<'a>(self, limit: usize) -> Result<&'a [u8]> { + if self.len > limit || (self.len != 0 && self.data.is_null()) { + return Err(Error::new("bridge-input")); + } + if self.len == 0 { + return Ok(&[]); + } + // The C caller guarantees the lifetime and readable extent of this buffer. + Ok(unsafe { slice::from_raw_parts(self.data, self.len) }) + } +} + +#[repr(C)] +#[derive(Clone, Copy)] +pub struct RawInput { + receipt: Bytes, + data: Bytes, +} +impl RawInput { + unsafe fn owned(self) -> Result { + let receipt = serde_json::from_slice(unsafe { self.receipt.read(16_384)? }) + .map_err(|_| Error::new("bridge-json"))?; + Ok(Input { + receipt, + data: unsafe { self.data.read(MAX_SOURCE_BYTES)? }.to_vec(), + }) + } +} + +pub struct Output { + files: Vec<(String, Vec)>, + error: Vec, +} +fn boundary(work: impl FnOnce() -> Result>>) -> *mut Output { + let result = catch_unwind(std::panic::AssertUnwindSafe(work)) + .unwrap_or_else(|_| Err(Error::new("bridge-panic"))); + let output = match result { + Ok(files) => Output { + files: files.into_iter().collect(), + error: Vec::new(), + }, + Err(error) => Output { + files: Vec::new(), + error: + serde_json::json!({"code": error.code, "source": error.source, "line": error.line}) + .to_string() + .into_bytes(), + }, + }; + Box::into_raw(Box::new(output)) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_generate( + catalog: Bytes, + inputs: *const RawInput, + count: usize, + corrections: Bytes, +) -> *mut Output { + boundary(|| { + if count > 64 || (count != 0 && inputs.is_null()) { + return Err(Error::new("bridge-input")); + } + let catalog: Vec = serde_json::from_slice(unsafe { catalog.read(1_048_576)? }) + .map_err(|_| Error::new("bridge-json"))?; + let inputs = if count == 0 { + &[] + } else { + unsafe { slice::from_raw_parts(inputs, count) } + }; + let inputs = inputs + .iter() + .map(|input| unsafe { input.owned() }) + .collect::>>()?; + let generated = + crate::generate(&inputs, unsafe { corrections.read(1_048_576)? }, &catalog)?; + let mut manifest = serde_json::to_vec_pretty(&generated.manifest) + .map_err(|_| Error::new("bridge-json"))?; + manifest.push(b'\n'); + Ok(BTreeMap::from([ + ("pinyin_simp.dict.yaml".to_owned(), generated.dictionary), + ("dictionary-manifest.json".to_owned(), manifest), + ])) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_spelling(dictionary: Bytes) -> *mut Output { + boundary(|| crate::spelling(unsafe { dictionary.read(MAX_SOURCE_BYTES)? })) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_validate(input: RawInput) -> *mut Output { + boundary(|| { + crate::validate(&unsafe { input.owned()? })?; + Ok(BTreeMap::new()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_result_error(result: *const Output) -> Bytes { + Bytes::borrowed(&unsafe { &*result }.error) +} +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_result_count(result: *const Output) -> usize { + unsafe { &*result }.files.len() +} +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_result_name(result: *const Output, index: usize) -> Bytes { + Bytes::borrowed( + unsafe { &*result } + .files + .get(index) + .map_or(&[], |(name, _)| name.as_bytes()), + ) +} +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_result_data(result: *const Output, index: usize) -> Bytes { + Bytes::borrowed( + unsafe { &*result } + .files + .get(index) + .map_or(&[], |(_, data)| data), + ) +} +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifd_result_free(result: *mut Output) { + if !result.is_null() { + drop(unsafe { Box::from_raw(result) }); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + unsafe fn error(result: *mut Output) -> String { + let error = unsafe { ifd_result_error(result).read(16_384) }.unwrap(); + let value: serde_json::Value = serde_json::from_slice(error).unwrap(); + assert_eq!(unsafe { ifd_result_count(result) }, 0); + let code = value["code"].as_str().unwrap().to_owned(); + unsafe { ifd_result_free(result) }; + code + } + + #[test] + fn invalid_requests_and_panic_containment() { + unsafe { + let empty = Bytes::borrowed(&[]); + assert_eq!(error(ifd_spelling(empty)), "source-format"); + assert_eq!( + error(ifd_spelling(Bytes { + data: ptr::null(), + len: 1 + })), + "bridge-input" + ); + assert_eq!( + error(ifd_spelling(Bytes { + data: ptr::null(), + len: usize::MAX + })), + "bridge-input" + ); + assert_eq!( + error(ifd_generate(empty, ptr::null(), 1, empty)), + "bridge-input" + ); + assert_eq!( + error(ifd_generate(empty, ptr::null(), 65, empty)), + "bridge-input" + ); + assert_eq!( + error(ifd_generate(empty, ptr::null(), 0, empty)), + "bridge-json" + ); + assert_eq!( + error(ifd_generate(Bytes::borrowed(b"[]"), ptr::null(), 0, empty)), + "source-set" + ); + assert_eq!( + error(boundary(|| panic!("test panic boundary"))), + "bridge-panic" + ); + ifd_result_free(ptr::null_mut()); + } + } + + #[test] + fn generation_matches_direct_calls_for_all_contract_cases() { + let cases: Vec = + serde_json::from_str(include_str!("../fixtures/cases.json")).unwrap(); + for case in cases { + let mut catalog: Vec = + serde_json::from_str(include_str!("../fixtures/catalog.json")).unwrap(); + let header = case["header"] + .as_str() + .unwrap_or("---\nimport_tables: [ignored]\n...\n"); + let inputs: Vec<_> = catalog + .iter_mut() + .map(|spec| { + let body = case["bodies"][&spec.id] + .as_str() + .unwrap_or("甲\tjia\t100\n"); + let data = format!("{header}{body}").into_bytes(); + spec.pinned_byte_count = data.len(); + spec.pinned_blob_sha = crate::git_blob(&data); + spec.pinned_sha256 = crate::sha256(&data); + Input { + receipt: spec.pinned_receipt(), + data, + } + }) + .collect(); + let corrections = case["corrections"].as_str().unwrap_or("").as_bytes(); + let expected = crate::generate(&inputs, corrections, &catalog); + let catalog = serde_json::to_vec(&catalog).unwrap(); + let receipts: Vec<_> = inputs + .iter() + .map(|i| serde_json::to_vec(&i.receipt).unwrap()) + .collect(); + let raw: Vec<_> = inputs + .iter() + .zip(&receipts) + .map(|(input, receipt)| RawInput { + receipt: Bytes::borrowed(receipt), + data: Bytes::borrowed(&input.data), + }) + .collect(); + let output = unsafe { + ifd_generate( + Bytes::borrowed(&catalog), + raw.as_ptr(), + raw.len(), + Bytes::borrowed(corrections), + ) + }; + drop(raw); + drop(receipts); + drop(inputs); + drop(catalog); + let output = unsafe { Box::from_raw(output) }; + match expected { + Ok(expected) => { + assert!(output.error.is_empty(), "{}", case["name"]); + assert_eq!(output.files.len(), 2); + assert_eq!(output.files[1].0, "pinyin_simp.dict.yaml"); + assert_eq!(output.files[1].1, expected.dictionary); + let mut manifest = serde_json::to_vec_pretty(&expected.manifest).unwrap(); + manifest.push(b'\n'); + assert_eq!(output.files[0].0, "dictionary-manifest.json"); + assert_eq!(output.files[0].1, manifest); + } + Err(expected) => { + assert!(output.files.is_empty()); + let error: serde_json::Value = serde_json::from_slice(&output.error).unwrap(); + assert_eq!( + error, + serde_json::json!({"code": expected.code, "source": expected.source, "line": expected.line}) + ); + } + } + } + } + + #[test] + fn owned_spelling_results_and_independent_calls() { + let threads: Vec<_> = (0..4) + .map(|_| { + std::thread::spawn(|| unsafe { + let source = b"---\n...\n\xe4\xbd\xa0\tni\t100\n".to_vec(); + let expected = crate::spelling(&source).unwrap(); + let result = ifd_spelling(Bytes::borrowed(&source)); + drop(source); + assert_eq!(ifd_result_error(result).len, 0); + assert_eq!(ifd_result_count(result), 32); + for (index, (name, data)) in expected.iter().enumerate() { + assert_eq!( + ifd_result_name(result, index).read(1024).unwrap(), + name.as_bytes() + ); + assert_eq!( + ifd_result_data(result, index) + .read(MAX_SOURCE_BYTES) + .unwrap(), + data + ); + } + assert_eq!(ifd_result_data(result, usize::MAX).len, 0); + assert_eq!(ifd_result_name(result, usize::MAX).len, 0); + ifd_result_free(result); + }) + }) + .collect(); + for thread in threads { + thread.join().unwrap(); + } + } +} diff --git a/Core/Portable/dictionary/src/lib.rs b/Core/Portable/dictionary/src/lib.rs index 7b4f5f3..da8641b 100644 --- a/Core/Portable/dictionary/src/lib.rs +++ b/Core/Portable/dictionary/src/lib.rs @@ -1,4 +1,5 @@ //! Offline dictionary generation. Inputs and provenance are supplied by the caller. +mod ffi; mod model; mod spelling; pub use model::*; diff --git a/Core/Portable/dictionary/test.sh b/Core/Portable/dictionary/test.sh index 6d2f613..9568284 100644 --- a/Core/Portable/dictionary/test.sh +++ b/Core/Portable/dictionary/test.sh @@ -9,6 +9,7 @@ export CARGO_TARGET_DIR="$root/build/dictionary-parity/cargo" fixtures="$root/Core/Portable/dictionary/fixtures" reference="$fixtures" comparison=() +bridge=() if [[ $(uname -s) == Darwin ]]; then bash macOS/scripts/test-dictionary-generator.sh build/dictionary-generator-tests --export-reference "$fixtures/cases.json" "$report" @@ -18,5 +19,19 @@ fi export INKFLOW_DICTIONARY_REFERENCE="$reference" cargo test --locked --release --manifest-path Core/Portable/dictionary/Cargo.toml -- --nocapture cargo build --locked --release --manifest-path Core/Portable/dictionary/Cargo.toml +include="$root/Core/Portable/dictionary/include" +archive="$CARGO_TARGET_DIR/release/libinkflow_dictionary.a" +libs=(-ldl -lpthread -lm) +if [[ $(uname -s) == Darwin ]]; then + libs=(-lresolv -liconv) + swiftc -O -warnings-as-errors -I "$include" \ + Core/Portable/dictionary/tests/NativeGenerator.swift "$archive" "${libs[@]}" \ + -o "$CARGO_TARGET_DIR/release/swift-native-generator" + bridge=(--bridge "$CARGO_TARGET_DIR/release/swift-native-generator") +fi +cc -std=c11 -Wall -Wextra -Werror -I "$include" \ + Core/Portable/dictionary/tests/native.c "$archive" "${libs[@]}" \ + -o "$CARGO_TARGET_DIR/release/dictionary-native-test" +"$CARGO_TARGET_DIR/release/dictionary-native-test" python3 Core/Portable/dictionary/verify.py --binary "$CARGO_TARGET_DIR/release/inkflow-dictionary" \ - --catalog "$reference/catalog.json" --report "$report" "${comparison[@]}" + --catalog "$reference/catalog.json" --report "$report" "${comparison[@]}" "${bridge[@]}" diff --git a/Core/Portable/dictionary/tests/NativeGenerator.swift b/Core/Portable/dictionary/tests/NativeGenerator.swift new file mode 100644 index 0000000..894bde0 --- /dev/null +++ b/Core/Portable/dictionary/tests/NativeGenerator.swift @@ -0,0 +1,95 @@ +import Foundation +import InkFlowDictionary + +// Standalone comparison host. Shipping preparation still uses its existing worker. +private final class Buffer { + private let pointer: UnsafeMutablePointer + private let count: Int + init(_ data: Data) { + count = data.count + pointer = .allocate(capacity: max(1, count)) + data.copyBytes(to: pointer, count: count) + } + var borrowed: IFDBytes { .init(data: UnsafePointer(pointer), len: count) } + deinit { pointer.deallocate() } +} +private struct Failure: Error { let detail: String } +private func copied(_ bytes: IFDBytes) -> Data { + bytes.len == 0 ? Data() : Data(bytes: bytes.data!, count: bytes.len) +} +private func check(_ result: OpaquePointer) throws { + let error = copied(ifd_result_error(result)) + if !error.isEmpty { throw Failure(detail: String(decoding: error, as: UTF8.self)) } +} +private func generated(_ args: [String]) throws -> OpaquePointer { + let catalogData = try Data(contentsOf: URL(fileURLWithPath: args[1])) + let catalog = Buffer(catalogData) + let specs = try JSONSerialization.jsonObject(with: catalogData) as! [[String: Any]] + let corrections = Buffer(try Data(contentsOf: URL(fileURLWithPath: args[4]))) + var buffers: [Buffer] = [] + var inputs: [IFDInput] = [] + for spec in specs { + let id = spec["id"] as! String + let source = spec["group"] as! String == "legacy" + ? URL(fileURLWithPath: args[3]) + : URL(fileURLWithPath: args[2]).appendingPathComponent(id + ".yaml") + let receipt: [String: Any] = [ + "id": id, "name": spec["name"]!, "repository": spec["repository"]!, + "path": spec["path"]!, "commit": spec["pinnedCommit"]!, + "blobSHA": spec["pinnedBlobSHA"]!, "sha256": spec["pinnedSHA256"]!, + "byteCount": spec["pinnedByteCount"]!, "recordCount": 0, + ] + let receiptBuffer = Buffer(try JSONSerialization.data(withJSONObject: receipt)) + let dataBuffer = Buffer(try Data(contentsOf: source)) + buffers += [receiptBuffer, dataBuffer] + let input = IFDInput(receipt: receiptBuffer.borrowed, data: dataBuffer.borrowed) + let validation = ifd_validate(input)! + defer { ifd_result_free(validation) } + try check(validation) + inputs.append(input) + } + return withExtendedLifetime((buffers, catalog, corrections)) { + inputs.withUnsafeBufferPointer { + ifd_generate(catalog.borrowed, $0.baseAddress, $0.count, corrections.borrowed)! + } + } +} +private func spell(_ path: String) throws -> OpaquePointer { + let buffer = Buffer(try Data(contentsOf: URL(fileURLWithPath: path))) + return withExtendedLifetime(buffer) { ifd_spelling(buffer.borrowed)! } +} +private func run() throws { + let args = Array(CommandLine.arguments.dropFirst()) + let result: OpaquePointer + let destination: URL + if args.count == 6, args[0] == "generate" { + result = try generated(args) + destination = URL(fileURLWithPath: args[5]) + } else if args.count == 3, args[0] == "spelling" { + result = try spell(args[1]) + destination = URL(fileURLWithPath: args[2]) + } else { + throw Failure(detail: "Expected generate CATALOG SOURCES LEGACY CORRECTIONS NEW_OUTPUT or spelling DICTIONARY NEW_OUTPUT") + } + // All input buffers have been released before results are inspected or written. + defer { ifd_result_free(result) } + try check(result) + guard !FileManager.default.fileExists(atPath: destination.path) else { + throw Failure(detail: "Output already exists") + } + try FileManager.default.createDirectory(at: destination, withIntermediateDirectories: false) + do { + for index in 0.. +#include +#include + +static IFDBytes bytes(const char* text) { + IFDBytes result = {(const uint8_t*)text, strlen(text)}; + return result; +} + +int main(void) { + char dictionary[] = "---\n...\n你\tni\t100\n"; + IFDResult* spelling = ifd_spelling(bytes(dictionary)); + memset(dictionary, 0, sizeof(dictionary)); + assert(ifd_result_error(spelling).len == 0); + assert(ifd_result_count(spelling) == 32); + for (size_t i = 0; i < 32; ++i) { + IFDBytes name = ifd_result_name(spelling, i); + IFDBytes data = ifd_result_data(spelling, i); + assert(name.len > 0 && data.len > 0); + assert(memchr(name.data, '/', name.len) == NULL); + } + assert(ifd_result_name(spelling, 32).len == 0); + assert(ifd_result_data(spelling, SIZE_MAX).len == 0); + + IFDResult* malformed = ifd_generate(bytes("[]"), NULL, 1, bytes("")); + assert(ifd_result_error(malformed).len > 0); + assert(ifd_result_count(malformed) == 0); + ifd_result_free(malformed); + assert(ifd_result_count(spelling) == 32); + ifd_result_free(spelling); + + const uint8_t invalid_utf8[] = {0xff, 0x00}; + IFDBytes invalid = {invalid_utf8, sizeof(invalid_utf8)}; + malformed = ifd_spelling(invalid); + assert(ifd_result_error(malformed).len > 0); + ifd_result_free(malformed); + IFDInput input = {bytes("{}"), bytes("")}; + malformed = ifd_validate(input); + assert(ifd_result_error(malformed).len > 0); + ifd_result_free(malformed); + ifd_result_free(NULL); + puts("PASS C ABI layout, linking, ownership, errors, and spelling outputs"); + return 0; +} diff --git a/Core/Portable/dictionary/verify.py b/Core/Portable/dictionary/verify.py index 1f4779d..c580b10 100644 --- a/Core/Portable/dictionary/verify.py +++ b/Core/Portable/dictionary/verify.py @@ -65,6 +65,7 @@ def main(): parser.add_argument('--binary', type=Path, required=True) parser.add_argument('--catalog', type=Path, required=True) parser.add_argument('--swift', type=Path) + parser.add_argument('--bridge', type=Path) parser.add_argument('--report', type=Path, required=True) args = parser.parse_args() catalog = json.loads(args.catalog.read_text()) @@ -88,6 +89,19 @@ def main(): (args.report / 'corpus.json').write_text(json.dumps(expected, ensure_ascii=False, indent=2) + '\n') (args.report / 'rust-corpus.json').write_text(json.dumps(actual, ensure_ascii=False, indent=2) + '\n') print(f"PASS pinned corpus: {actual['manifest']['entryCount']} rows, dictionary and all 32 profiles match Swift") + if args.bridge: + native_dictionary, native_schemas = root / 'native-dictionary', root / 'native-spelling' + subprocess.run([str(args.bridge), 'generate', str(args.catalog), str(source), str(source / 'legacy.yaml'), + str(ROOT / 'Core/config/chinese-overrides.tsv'), str(native_dictionary)], check=True) + subprocess.run([str(args.bridge), 'spelling', str(native_dictionary / 'pinyin_simp.dict.yaml'), + str(native_schemas)], check=True) + native = snapshot(native_dictionary, native_schemas) + assert native == actual, 'Swift FFI and Rust CLI summaries differ' + assert (native_dictionary / 'pinyin_simp.dict.yaml').read_bytes() == (generated / 'pinyin_simp.dict.yaml').read_bytes() + for file in schemas.glob('*.schema.yaml'): + assert (native_schemas / file.name).read_bytes() == file.read_bytes(), file.name + (args.report / 'swift-ffi-corpus.json').write_text(json.dumps(native, ensure_ascii=False, indent=2) + '\n') + print('PASS Swift in-process Rust ABI: complete dictionary, manifest, and 32 spelling profiles') if args.swift: for name in ('catalog.json', 'reference.json', 'corpus.json'): equivalent(json.loads((args.report / name).read_text()), json.loads((FIXTURES / name).read_text()), name) From 0d0969c20a03bddeaeb163e897b124e8c0252554 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 07:30:54 +0800 Subject: [PATCH 02/21] Record Swift consumer parity through the Rust dictionary ABI --- docs/remote-mac-baseline.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/remote-mac-baseline.md b/docs/remote-mac-baseline.md index 51b509f..4b3c8f9 100644 --- a/docs/remote-mac-baseline.md +++ b/docs/remote-mac-baseline.md @@ -42,7 +42,7 @@ The remote report contains `00-portable.log`, `native-build.json`, and the norma ## Dictionary generator comparison -`dictionary` runs `Core/Portable/dictionary/test.sh`. It exercises the existing Swift generator, exports its contract, and compares the Rust generator against fresh Swift outputs and recorded fixtures. It returns `catalog.json`, `reference.json`, the Swift/Rust summaries `corpus.json` and `rust-corpus.json`, plus `00-dictionary.log` and the normal receipt. See [the comparison contract](../Core/Portable/dictionary/README.md). +`dictionary` runs `Core/Portable/dictionary/test.sh`. It exercises the existing Swift generator, exports its contract, and compares the Rust generator against fresh Swift outputs and recorded fixtures. It also compiles C and Swift consumers of the in-process Rust dictionary ABI. It returns `catalog.json`, `reference.json`, the Swift/Rust summaries `corpus.json` and `rust-corpus.json`, the Swift ABI consumer's `swift-ffi-corpus.json`, plus `00-dictionary.log` and the normal receipt. See [the comparison contract](../Core/Portable/dictionary/README.md). ## Baseline recipe From 385b5e8fe257a9d38d991dc71db0ec30dcd718e0 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 07:42:47 +0800 Subject: [PATCH 03/21] Use one Rust dictionary generator for preparation and the Swift worker --- Core/Package.swift | 8 +- .../dictionary/include/inkflow_dictionary.h | 2 + Core/Portable/dictionary/src/ffi.rs | 5 + Core/Portable/dictionary/src/lib.rs | 5 + Core/Portable/dictionary/src/main.rs | 74 +++- Core/Portable/dictionary/test.sh | 4 +- Core/Portable/dictionary/tests/cli.rs | 33 ++ .../InkFlowDictionary/build_identity.c | 2 + .../include/InkFlowDictionary.h | 1 + .../InkFlowDomain/DictionaryGenerator.swift | 326 ++++-------------- .../InkFlowDomain/DictionaryModels.swift | 12 +- .../DictionaryGenerationReference.swift | 4 +- .../DictionaryGeneratorTests.swift | 19 +- .../SwiftReferenceGenerator.swift | 286 +++++++++++++++ Core/config/chinese-sources.json | 110 ++++++ Core/scripts/build-dictionary-generator.sh | 15 +- Core/scripts/prepare-rime.sh | 2 +- Core/scripts/prepare-spelling.sh | 5 +- Core/scripts/swift-package.sh | 1 + Core/scripts/test-prepare-rime.sh | 8 +- Package.swift | 8 +- macOS/Tools/QualityBuildMetadata.swift | 4 +- macOS/scripts/swift-package.sh | 1 + macOS/scripts/test-quality-metadata.sh | 3 +- scripts/mac-remote.py | 1 + 25 files changed, 632 insertions(+), 307 deletions(-) create mode 100644 Core/Sources/InkFlowDictionary/build_identity.c create mode 100644 Core/Sources/InkFlowDictionary/include/InkFlowDictionary.h create mode 100644 Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift create mode 100644 Core/config/chinese-sources.json diff --git a/Core/Package.swift b/Core/Package.swift index e96c67c..458ca4d 100644 --- a/Core/Package.swift +++ b/Core/Package.swift @@ -60,7 +60,13 @@ let package = Package( .executableTarget(name: "AIAdoptionLearningTests", dependencies: ["InkFlowDomain", "InkFlowRime", "InkFlowCoreTestSupport"], path: "Tests/AIAdoptionLearningTests", swiftSettings: strictSwiftSettings, linkerSettings: buildRimeRuntime), .executableTarget(name: "RankingTests", dependencies: ["InkFlowRankingTestSupport"], path: "Tests/RankingTests", swiftSettings: strictSwiftSettings, linkerSettings: buildRimeRuntime), - .target(name: "InkFlowDomain", swiftSettings: strictSwiftSettings), + .target(name: "InkFlowDictionary", path: "Sources/InkFlowDictionary", + publicHeadersPath: "include", + cSettings: [.unsafeFlags(["-I\(repository.path)/build/dictionary/include"])], + linkerSettings: [.unsafeFlags(["-L\(repository.path)/build/dictionary/cargo/release"]), + .linkedLibrary("inkflow_dictionary"), .linkedLibrary("resolv"), .linkedLibrary("iconv")]), + .target(name: "InkFlowDomain", dependencies: ["InkFlowDictionary"], path: "Sources/InkFlowDomain", + swiftSettings: strictSwiftSettings), .target(name: "CRime", dependencies: ["InkFlowRimeNative"], publicHeadersPath: "include", cSettings: strictCSettings + [.unsafeFlags(["-I\(dependencyRoot)/include"])], linkerSettings: rimeLinkerSettings), diff --git a/Core/Portable/dictionary/include/inkflow_dictionary.h b/Core/Portable/dictionary/include/inkflow_dictionary.h index 53c4f19..2aa349e 100644 --- a/Core/Portable/dictionary/include/inkflow_dictionary.h +++ b/Core/Portable/dictionary/include/inkflow_dictionary.h @@ -24,6 +24,8 @@ typedef struct { const uint8_t* data; size_t len; } IFDBytes; typedef struct { IFDBytes receipt; IFDBytes data; } IFDInput; typedef struct IFDResult IFDResult; +/* Immutable compiled catalog, valid for the process lifetime; do not free. */ +IFDBytes ifd_catalog(void); IFDResult* ifd_generate(IFDBytes catalog, const IFDInput* inputs, size_t count, IFDBytes corrections); IFDResult* ifd_spelling(IFDBytes dictionary); diff --git a/Core/Portable/dictionary/src/ffi.rs b/Core/Portable/dictionary/src/ffi.rs index 3801cb4..fc98e46 100644 --- a/Core/Portable/dictionary/src/ffi.rs +++ b/Core/Portable/dictionary/src/ffi.rs @@ -70,6 +70,11 @@ fn boundary(work: impl FnOnce() -> Result>>) -> *mut Ou Box::into_raw(Box::new(output)) } +#[unsafe(no_mangle)] +pub extern "C" fn ifd_catalog() -> Bytes { + Bytes::borrowed(crate::CATALOG_JSON) +} + #[unsafe(no_mangle)] pub unsafe extern "C" fn ifd_generate( catalog: Bytes, diff --git a/Core/Portable/dictionary/src/lib.rs b/Core/Portable/dictionary/src/lib.rs index da8641b..ea776f4 100644 --- a/Core/Portable/dictionary/src/lib.rs +++ b/Core/Portable/dictionary/src/lib.rs @@ -5,6 +5,11 @@ mod spelling; pub use model::*; pub use spelling::spelling; +pub const CATALOG_JSON: &[u8] = include_bytes!("../../../config/chinese-sources.json"); +pub fn catalog() -> Result> { + serde_json::from_slice(CATALOG_JSON).map_err(|_| Error::new("catalog-format")) +} + use std::collections::{HashMap, HashSet}; use std::fmt::Write; use unicode_general_category::{GeneralCategory, get_general_category}; diff --git a/Core/Portable/dictionary/src/main.rs b/Core/Portable/dictionary/src/main.rs index 58e3bea..1d3e74b 100644 --- a/Core/Portable/dictionary/src/main.rs +++ b/Core/Portable/dictionary/src/main.rs @@ -15,7 +15,7 @@ fn read(path: &Path, limit: usize) -> Result> { } fn write_new(directory: &Path, files: BTreeMap>) -> Result<()> { - // Refuse replacement. Preparation callers publish their own completed staging tree. + // Preparation callers publish their own completed staging tree. fs::create_dir(directory)?; let result = (|| { for (name, bytes) in files { @@ -29,33 +29,69 @@ fn write_new(directory: &Path, files: BTreeMap>) -> Result<()> { result } +fn dictionary( + catalog: &[SourceSpec], + sources: &str, + legacy: &str, + corrections: &str, + output: &str, +) -> Result<()> { + let mut inputs = Vec::new(); + for spec in catalog { + if spec.id.is_empty() + || !spec + .id + .bytes() + .all(|b| b.is_ascii_alphanumeric() || b == b'-') + { + return Err("Invalid source identifier".into()); + } + let path = if spec.group == "legacy" { + Path::new(legacy).to_owned() + } else { + Path::new(sources).join(format!("{}.yaml", spec.id)) + }; + inputs.push(Input { + receipt: spec.pinned_receipt(), + data: read(&path, MAX_SOURCE_BYTES)?, + }); + } + let corrections = read(Path::new(corrections), 1_048_576)?; + let result = generate(&inputs, &corrections, catalog)?; + let mut manifest = serde_json::to_vec_pretty(&result.manifest)?; + manifest.push(b'\n'); + write_new( + Path::new(output), + BTreeMap::from([ + ("pinyin_simp.dict.yaml".to_owned(), result.dictionary), + ("dictionary-manifest.json".to_owned(), manifest), + ]), + ) +} + fn run(args: &[String]) -> Result<()> { match args { + [command] if command == "sources" => { + for spec in inkflow_dictionary::catalog()?.iter().filter(|s| s.group != "legacy") { + println!("{}\t{}\t{}\thttps://raw.githubusercontent.com/{}/{}/{}", + spec.id, spec.pinned_sha256, spec.pinned_byte_count, spec.repository, spec.pinned_commit, spec.path); + } + } + [command] if command == "catalog" => { + print!("{}", std::str::from_utf8(inkflow_dictionary::CATALOG_JSON)?); + } + [command, sources, legacy, corrections, output] if command == "generate" => { + dictionary(&inkflow_dictionary::catalog()?, sources, legacy, corrections, output)?; + } [command, catalog, sources, legacy, corrections, output] if command == "generate" => { let catalog: Vec = serde_json::from_slice(&read(Path::new(catalog), 1_048_576)?)?; - let mut inputs = Vec::new(); - for spec in &catalog { - if spec.id.is_empty() || !spec.id.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'-') { - return Err("Invalid source identifier".into()); - } - let path = if spec.group == "legacy" { Path::new(legacy).to_owned() } - else { Path::new(sources).join(format!("{}.yaml", spec.id)) }; - inputs.push(Input { receipt: spec.pinned_receipt(), data: read(&path, MAX_SOURCE_BYTES)? }); - } - let corrections = read(Path::new(corrections), 1_048_576)?; - let result = generate(&inputs, &corrections, &catalog)?; - let mut manifest = serde_json::to_vec_pretty(&result.manifest)?; - manifest.push(b'\n'); - write_new(Path::new(output), BTreeMap::from([ - ("pinyin_simp.dict.yaml".to_owned(), result.dictionary), - ("dictionary-manifest.json".to_owned(), manifest), - ]))?; + dictionary(&catalog, sources, legacy, corrections, output)?; } [command, dictionary, output] if command == "spelling" => { let dictionary = read(Path::new(dictionary), MAX_SOURCE_BYTES)?; write_new(Path::new(output), spelling(&dictionary)?)?; } - _ => return Err("Usage: inkflow-dictionary generate CATALOG SOURCES LEGACY CORRECTIONS NEW_OUTPUT | spelling DICTIONARY NEW_OUTPUT".into()), + _ => return Err("Usage: dictionary-generator sources | catalog | generate [CATALOG] SOURCES LEGACY CORRECTIONS NEW_OUTPUT | spelling DICTIONARY NEW_OUTPUT".into()), } Ok(()) } diff --git a/Core/Portable/dictionary/test.sh b/Core/Portable/dictionary/test.sh index 9568284..38811dd 100644 --- a/Core/Portable/dictionary/test.sh +++ b/Core/Portable/dictionary/test.sh @@ -5,7 +5,7 @@ root=$PWD report=${1:-"$root/build/dictionary-parity/report"} mkdir -p "$report" report=$(cd "$report" && pwd) -export CARGO_TARGET_DIR="$root/build/dictionary-parity/cargo" +export CARGO_TARGET_DIR="$root/build/dictionary/cargo" fixtures="$root/Core/Portable/dictionary/fixtures" reference="$fixtures" comparison=() @@ -14,7 +14,7 @@ if [[ $(uname -s) == Darwin ]]; then bash macOS/scripts/test-dictionary-generator.sh build/dictionary-generator-tests --export-reference "$fixtures/cases.json" "$report" reference="$report" - comparison=(--swift "$root/build/test-chinese") + comparison=(--swift "$root/build/test-chinese-reference") fi export INKFLOW_DICTIONARY_REFERENCE="$reference" cargo test --locked --release --manifest-path Core/Portable/dictionary/Cargo.toml -- --nocapture diff --git a/Core/Portable/dictionary/tests/cli.rs b/Core/Portable/dictionary/tests/cli.rs index 35ca6f7..2c5ee16 100644 --- a/Core/Portable/dictionary/tests/cli.rs +++ b/Core/Portable/dictionary/tests/cli.rs @@ -13,6 +13,39 @@ impl Drop for Scratch { } } +#[test] +fn compiled_catalog_and_source_download_list_match() { + let output = Command::new(env!("CARGO_BIN_EXE_inkflow-dictionary")) + .arg("catalog") + .output() + .unwrap(); + assert!(output.status.success()); + assert_eq!(output.stdout, inkflow_dictionary::CATALOG_JSON); + let catalog = inkflow_dictionary::catalog().unwrap(); + let reference: serde_json::Value = + serde_json::from_str(include_str!("../fixtures/catalog.json")).unwrap(); + assert_eq!( + serde_json::from_slice::(&output.stdout).unwrap(), + reference + ); + let output = Command::new(env!("CARGO_BIN_EXE_inkflow-dictionary")) + .arg("sources") + .output() + .unwrap(); + assert!(output.status.success()); + let expected: String = catalog + .iter() + .filter(|s| s.group != "legacy") + .map(|s| { + format!( + "{}\t{}\t{}\thttps://raw.githubusercontent.com/{}/{}/{}\n", + s.id, s.pinned_sha256, s.pinned_byte_count, s.repository, s.pinned_commit, s.path + ) + }) + .collect(); + assert_eq!(output.stdout, expected.as_bytes()); +} + #[test] fn cli_checks_inputs_before_publishing_and_refuses_replacement() { let root = std::env::temp_dir().join(format!( diff --git a/Core/Sources/InkFlowDictionary/build_identity.c b/Core/Sources/InkFlowDictionary/build_identity.c new file mode 100644 index 0000000..c947258 --- /dev/null +++ b/Core/Sources/InkFlowDictionary/build_identity.c @@ -0,0 +1,2 @@ +#include "build_identity.h" +const char inkflow_dictionary_build_identity[] = INKFLOW_DICTIONARY_BUILD_ID; diff --git a/Core/Sources/InkFlowDictionary/include/InkFlowDictionary.h b/Core/Sources/InkFlowDictionary/include/InkFlowDictionary.h new file mode 100644 index 0000000..ef7196e --- /dev/null +++ b/Core/Sources/InkFlowDictionary/include/InkFlowDictionary.h @@ -0,0 +1 @@ +#include "../../../Portable/dictionary/include/inkflow_dictionary.h" diff --git a/Core/Sources/InkFlowDomain/DictionaryGenerator.swift b/Core/Sources/InkFlowDomain/DictionaryGenerator.swift index c7a2969..7f4142d 100644 --- a/Core/Sources/InkFlowDomain/DictionaryGenerator.swift +++ b/Core/Sources/InkFlowDomain/DictionaryGenerator.swift @@ -1,285 +1,99 @@ import Foundation +import InkFlowDictionary -/// Reads only the tab-separated body. Remote headers, imports and executable configuration are never interpreted. -package enum IFDictionaryGenerator { - package static let maximumWeight = Int(Int32.max) - private struct Key: Hashable { let text: String; let reading: String } - private struct Row { let key: Key; let weight: Int } - - package static func generate(inputs: [IFDictionaryInput], corrections: Data = Data(), - catalog: [IFDictionarySourceSpec] = IFDictionaryCatalog.sources) throws -> IFDictionaryGeneration { - let expected = catalog.map(\.id) - guard inputs.map(\.receipt.id) == expected else { - throw IFDictionaryError("source-set", "Expected every catalog source exactly once, in precedence order") - } - var groups = [String: [Key: Int]]() - var receipts = [IFDictionarySourceReceipt]() - for (spec, input) in zip(catalog, inputs) { - try validate(input) - guard input.receipt.repository == spec.repository, input.receipt.path == spec.path else { - throw IFDictionaryError("source-location", source: spec.id, "Source repository or path differs from the catalog") - } - if !spec.isUpdatable { - guard input.receipt.commit == spec.pinnedCommit, input.receipt.blobSHA == spec.pinnedBlobSHA, - input.receipt.sha256 == spec.pinnedSHA256 else { - throw IFDictionaryError("legacy-changed", source: spec.id, "Compatibility source must remain pinned") - } - } - let rows = try parse(input.data, source: spec.id) - var receipt = input.receipt - receipt.recordCount = rows.count - receipts.append(receipt) - var group = groups.removeValue(forKey: spec.group) ?? [:] - for row in rows where group[row.key] == nil { group[row.key] = row.weight } - groups[spec.group] = group - } - guard let frost = groups["frost"] else { throw IFDictionaryError("source-set", "Missing Frost baseline") } - var union = frost - var calibrations = [IFDictionaryCalibration]() - for groupName in ["ice", "legacy"] { - guard let group = groups[groupName] else { - throw IFDictionaryError("source-set", source: groupName, "Missing source group") - } - let calibration = try calibrate(source: group, baseline: frost, name: groupName) - calibrations.append(calibration) - for (key, weight) in group where union[key] == nil { - let factor = calibration.buckets[bucket(key.reading) - 1].multiplier - let mapped = Double(weight) * factor - // Saturate only mapped source weights at Rime's signed integer boundary. - union[key] = weight == 0 ? 0 : max(1, Int(min(Double(maximumWeight), mapped.rounded()))) - } - } - // Specialty dictionaries only fill gaps after the established calibrated union. - // They cannot change existing weights or the overlap used for calibration. - for (key, weight) in groups["specialty"] ?? [:] where union[key] == nil { union[key] = weight } - for row in try parseCorrections(corrections) { union[row.key] = row.weight } - let keys = union.keys.sorted { - $0.text == $1.text ? $0.reading.utf8.lexicographicallyPrecedes($1.reading.utf8) - : $0.text.utf8.lexicographicallyPrecedes($1.text.utf8) - } - var body = String() - body.reserveCapacity(union.count * 36) - for key in keys { body += "\(key.text)\t\(key.reading)\t\(union[key]!)\n" } - let contentHash = IFDictionaryHash.sha256(Data(body.utf8)) - let versionHash = IFDictionaryHash.sha256(Data("recipe:\(IFDictionaryCatalog.recipeVersion)\n\(body)".utf8)) - let version = "r\(IFDictionaryCatalog.recipeVersion)-\(versionHash)" - let header = """ - # Generated by InkFlow from Rime Frost, Rime Ice and pinned pinyin_simp. - # Baseline weights are retained; specialty data fills gaps. See bundled Licenses. - --- - name: pinyin_simp - version: '\(version)' - sort: by_weight - use_preset_vocabulary: false - ... - - """ - let dictionary = Data((header + body).utf8) - let manifest = IFDictionaryManifest(formatVersion: 1, recipeVersion: IFDictionaryCatalog.recipeVersion, - contentVersion: version, entryCount: union.count, contentSHA256: contentHash, - dictionarySHA256: IFDictionaryHash.sha256(dictionary), correctionsSHA256: IFDictionaryHash.sha256(corrections), - sources: receipts, calibrations: calibrations) - return IFDictionaryGeneration(dictionary: dictionary, manifest: manifest) +// Foreign input buffers stay alive through each synchronous preparation call. +private final class DictionaryBuffer { + private let pointer: UnsafeMutablePointer + private let count: Int + init(_ data: Data) { + count = data.count + pointer = .allocate(capacity: max(1, count)) + data.copyBytes(to: pointer, count: count) } + var borrowed: IFDBytes { .init(data: UnsafePointer(pointer), len: count) } + deinit { pointer.deallocate() } +} - package static func validate(_ input: IFDictionaryInput) throws { - let receipt = input.receipt - guard IFDictionaryHash.isHex(receipt.commit, length: 40), IFDictionaryHash.isHex(receipt.blobSHA, length: 40), - IFDictionaryHash.isHex(receipt.sha256, length: 64) else { - throw IFDictionaryError("invalid-receipt", source: receipt.id, "Invalid commit or content identifier") - } - guard !input.data.isEmpty, input.data.count <= IFDictionaryCatalog.maximumSourceBytes, - input.data.count == receipt.byteCount else { - throw IFDictionaryError("source-size", source: receipt.id, "Received \(input.data.count) bytes; expected \(receipt.byteCount)") - } - guard IFDictionaryHash.gitBlob(input.data) == receipt.blobSHA, - IFDictionaryHash.sha256(input.data) == receipt.sha256 else { - throw IFDictionaryError("source-checksum", source: receipt.id, "Git blob or SHA-256 verification failed") - } +private enum DictionaryNative { + private struct Failure: Decodable { + let code: String + let source: String? + let line: Int? } - - package static func normalizedReading(_ value: String) throws -> String { - let code = value.precomposedStringWithCanonicalMapping.lowercased().replacingOccurrences(of: "ü", with: "v") - let syllables = code.split(whereSeparator: \.isWhitespace) - guard !syllables.isEmpty, syllables.count <= 128, - syllables.allSatisfy({ !$0.isEmpty && $0.utf8.count <= 8 && $0.utf8.allSatisfy { (97...122).contains($0) } }) else { - throw IFDictionaryError("invalid-reading", "Expected explicit, toneless Pinyin syllables") - } - return syllables.joined(separator: " ") + static func copy(_ bytes: IFDBytes) -> Data { + bytes.len == 0 ? Data() : Data(bytes: bytes.data!, count: bytes.len) } - - private static func parse(_ data: Data, source: String) throws -> [Row] { - var rows = [Row]() - try readRows(data, source: source) { rows.append($0) } - return rows + static func consume(_ result: OpaquePointer) throws -> [String: Data] { + defer { ifd_result_free(result) } + let error = copy(ifd_result_error(result)) + if !error.isEmpty { + let failure = try JSONDecoder().decode(Failure.self, from: error) + throw IFDictionaryError(failure.code, source: failure.source, line: failure.line, + "Dictionary input validation or generation failed") + } + var files = [String: Data]() + for index in 0.. Set { - var syllables = Set() - try readRows(dictionary, source: "generated-spelling") { row in - syllables.formUnion(row.key.reading.split(separator: " ").map(String.init)) - } - return syllables +/// Calls the shared Rust generator in-process, outside interactive input handling. +package enum IFDictionaryGenerator { + package static func catalog() -> [IFDictionarySourceSpec] { + // This immutable JSON is compiled into the same verified Rust library. + try! JSONDecoder().decode([IFDictionarySourceSpec].self, from: DictionaryNative.copy(ifd_catalog())) } - private static func readRows(_ data: Data, source: String, receive: (Row) -> Void) throws { - // generate has already verified the complete byte count and both content hashes. - guard let string = String(data: data, encoding: .utf8) else { - throw IFDictionaryError("source-format", source: source, "Expected complete UTF-8 dictionary") - } - var inBody = false, sawHeader = false - var count = 0 - for (offset, rawLine) in string.split(separator: "\n", omittingEmptySubsequences: false).enumerated() { - let line = rawLine.hasSuffix("\r") ? rawLine.dropLast() : rawLine - guard line.utf8.count <= 8_192 else { - throw IFDictionaryError("source-format", source: source, line: offset + 1, "Line exceeds 8192 bytes") - } - if !inBody { - if line == "---" { sawHeader = true } - if line == "...", sawHeader { inBody = true } - continue - } - if line.trimmingCharacters(in: .whitespaces).isEmpty || line.trimmingCharacters(in: .whitespaces).hasPrefix("#") { continue } - let fields = line.split(separator: "\t", omittingEmptySubsequences: false) - guard fields.count == 3 else { - throw IFDictionaryError("source-format", source: source, line: offset + 1, "Expected text, reading and integer weight") + package static func generate(inputs: [IFDictionaryInput], corrections: Data = Data(), + catalog: [IFDictionarySourceSpec] = IFDictionaryCatalog.sources) throws -> IFDictionaryGeneration { + let encoder = JSONEncoder() + let catalogBuffer = DictionaryBuffer(try encoder.encode(catalog)) + let correctionBuffer = DictionaryBuffer(corrections) + var buffers = [DictionaryBuffer]() + var raw = [IFDInput]() + for input in inputs { + let receipt = DictionaryBuffer(try encoder.encode(input.receipt)) + let data = DictionaryBuffer(input.data) + buffers += [receipt, data] + raw.append(.init(receipt: receipt.borrowed, data: data.borrowed)) + } + let result = withExtendedLifetime((buffers, catalogBuffer, correctionBuffer)) { + raw.withUnsafeBufferPointer { + ifd_generate(catalogBuffer.borrowed, $0.baseAddress, $0.count, correctionBuffer.borrowed)! } - receive(try row(fields, source: source, line: offset + 1)) - count += 1 - guard count <= 3_000_000 else { throw IFDictionaryError("source-format", source: source, "Too many records") } } - guard inBody, count > 0 else { throw IFDictionaryError("source-format", source: source, "Missing dictionary body") } - } - - private static func row(_ fields: [Substring], source: String, line: Int) throws -> Row { - let text = String(fields[0]) - let hasHan = text.unicodeScalars.contains { scalar in - scalar.value == 0x3007 || (0x3400...0x9fff).contains(scalar.value) || (0xf900...0xfaff).contains(scalar.value) - || (0x20000...0x323af).contains(scalar.value) - } - guard hasHan, text.count <= 256, text == text.trimmingCharacters(in: .whitespacesAndNewlines), - !text.unicodeScalars.contains(where: { CharacterSet.controlCharacters.contains($0) }), - !fields[2].isEmpty, fields[2].utf8.allSatisfy({ (48...57).contains($0) }), - let weight = Int(fields[2]), weight <= maximumWeight else { - throw IFDictionaryError("source-format", source: source, line: line, "Invalid Chinese text or integer weight (0...\(maximumWeight))") + let files = try DictionaryNative.consume(result) + guard let dictionary = files[IFDictionaryCatalog.dictionaryFilename], + let metadata = files[IFDictionaryManifest.filename] else { + throw IFDictionaryError("bridge-output", "Missing generated dictionary or manifest") } - do { return Row(key: Key(text: text, reading: try normalizedReading(String(fields[1]))), weight: weight) } - catch { throw IFDictionaryError("invalid-reading", source: source, line: line, "Expected explicit, toneless Pinyin syllables") } - } - - private static func parseCorrections(_ data: Data) throws -> [Row] { - guard data.count <= 1_048_576, let string = String(data: data, encoding: .utf8) else { - throw IFDictionaryError("correction-format", "Expected UTF-8 corrections under 1 MiB") - } - var rows = [Row](), seen = Set() - for (offset, rawLine) in string.split(separator: "\n", omittingEmptySubsequences: false).enumerated() { - let line = rawLine.hasSuffix("\r") ? rawLine.dropLast() : rawLine - if line.trimmingCharacters(in: .whitespaces).isEmpty || line.trimmingCharacters(in: .whitespaces).hasPrefix("#") { continue } - let fields = line.split(separator: "\t", omittingEmptySubsequences: false) - guard fields.count == 4, !fields[3].trimmingCharacters(in: .whitespaces).isEmpty else { - throw IFDictionaryError("correction-format", line: offset + 1, "Expected text, reading, replacement weight and reason") - } - let entry = try row(Array(fields.prefix(3)), source: "corrections", line: offset + 1) - guard seen.insert(entry.key).inserted else { - throw IFDictionaryError("correction-duplicate", line: offset + 1, "Duplicate term/reading correction") - } - rows.append(entry) - } - return rows + let manifest = try JSONDecoder().decode(IFDictionaryManifest.self, from: metadata) + return IFDictionaryGeneration(dictionary: dictionary, manifest: manifest) } - private static func bucket(_ reading: String) -> Int { min(5, reading.split(separator: " ").count) } - private static func medianMultiplier(_ values: [Double]) -> Double { - let sorted = values.sorted(), middle = values.count / 2 - let median = sorted.count.isMultiple(of: 2) ? (sorted[middle - 1] + sorted[middle]) / 2 : sorted[middle] - return exp(median) - } - private static func calibrate(source: [Key: Int], baseline: [Key: Int], name: String) throws -> IFDictionaryCalibration { - var groups = Array(repeating: [Double](), count: 5) - for (key, weight) in source where weight > 0 { - if let reference = baseline[key], reference > 0 { - groups[bucket(key.reading) - 1].append(log(Double(reference) / Double(weight))) - } - } - let all = groups.flatMap { $0 } - guard !all.isEmpty else { throw IFDictionaryError("calibration-empty", source: name, "No positive-weight overlap with Frost") } - let overall = medianMultiplier(all) - let buckets = groups.enumerated().map { index, values in - IFDictionaryCalibrationBucket(syllables: index + 1, pairCount: values.count, - multiplier: values.count >= 100 ? medianMultiplier(values) : overall, usedOverall: values.count < 100) + package static func validate(_ input: IFDictionaryInput) throws { + let receipt = DictionaryBuffer(try JSONEncoder().encode(input.receipt)) + let data = DictionaryBuffer(input.data) + let result = withExtendedLifetime((receipt, data)) { + ifd_validate(.init(receipt: receipt.borrowed, data: data.borrowed))! } - return IFDictionaryCalibration(sourceGroup: name, pairCount: all.count, overallMultiplier: overall, buckets: buckets) + _ = try DictionaryNative.consume(result) } } -/// Build preparation and the isolated update worker compile the same native algebra. package enum IFSpellingGenerator { package static func generate(dictionary: Data) throws -> [String: Data] { - let syllables = try IFDictionaryGenerator.spellingSyllables(in: dictionary) - var schemas = [String: Data]() - for profile in 0..<32 { - let name = "inkflow_spelling_\(profile)" - let rules = try algebra(syllables: syllables, profile: profile) - let schema = """ - # Generated by IFSpellingGenerator. Compilation dependency only. - schema: - schema_id: \(name) - name: InkFlow spelling \(profile) - version: '1.1' - translator: - dictionary: pinyin_simp - prism: \(name) - speller: - algebra: - \(rules.map { " - \($0)" }.joined(separator: "\n")) - - """ - schemas[name + ".schema.yaml"] = Data(schema.utf8) - } - return schemas + let buffer = DictionaryBuffer(dictionary) + let result = withExtendedLifetime(buffer) { ifd_spelling(buffer.borrowed)! } + return try DictionaryNative.consume(result) } - package static func write(dictionary: Data, to directory: URL) throws { let schemas = try generate(dictionary: dictionary) for name in schemas.keys.sorted() { try schemas[name]!.write(to: directory.appendingPathComponent(name), options: .atomic) } } - - private static func algebra(syllables: Set, profile: Int) throws -> [String] { - // One list owns normal equivalent spellings and explicitly enabled fuzzy pairs. - // These simple derivations also define the legal full-spelling collision set. - var equivalents = [("^([nl])ue$", "$1ve"), ("^([jqxy])u", "$1v")] - for (initial, bit) in [("z", 4), ("c", 8), ("s", 16)] where profile & bit != 0 { - equivalents += [("^\(initial)h", initial), ("^\(initial)([^h])", initial + "h$1")] - } - var rules = equivalents.map { "derive/\($0.0)/\($0.1)/" } - if profile & 1 != 0 { - rules += ["abbrev/^([a-z]).+$/$1/", "abbrev/^([zcs]h).+$/$1/"] - } - if profile & 2 != 0 { - var legal = syllables - for (pattern, replacement) in equivalents { - let expression = try NSRegularExpression(pattern: pattern) - legal.formUnion(legal.map { - expression.stringByReplacingMatches(in: $0, range: NSRange($0.startIndex..., in: $0), - withTemplate: replacement) - }) - } - // Only copies carry typo transformations. Rime retains fuzzy properties, - // ordered combinations and the original paths. Guard the final aliases, - // then remove the temporary marker before compiling the prism. - rules += [ - "derive/^(.*)$/~$1/", - "fuzz/^~([zcs])h/~h$1/", - "fuzz/^~([bpmfdtnlgkhjqxrzcsyw])(?=[a-z])/~$1$1/", - "fuzz/^~(.*)([aeiou])ng$/~$1$2gn/", - "fuzz/^~(.*)ao$/~$1oa/", - "fuzz/^~(.*)([iu])a(o|ng?)?$/~$1a$2$3/", - "erase/^~(\(legal.sorted().joined(separator: "|")))$/", - "xform/^~//", - ] - } - return rules - } } diff --git a/Core/Sources/InkFlowDomain/DictionaryModels.swift b/Core/Sources/InkFlowDomain/DictionaryModels.swift index f335e7a..46081d9 100644 --- a/Core/Sources/InkFlowDomain/DictionaryModels.swift +++ b/Core/Sources/InkFlowDomain/DictionaryModels.swift @@ -136,15 +136,5 @@ package enum IFDictionaryCatalog { package static let correctionsFilename = "chinese-overrides.tsv" package static let maximumSourceBytes = 128 * 1024 * 1024 package static let initialEntryCount = 965_919 - package static let sources: [IFDictionarySourceSpec] = [ - .init(id: "frost-8105", group: "frost", name: "白霜 · 字表", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts/8105.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "9cbfe2acf59bb3124993d4b8b3b271f8e246b72d", pinnedSHA256: "5a6bb545d07140406208728aeed70706e84279daeee132f10b069cd6387a042a", pinnedByteCount: 99_367), - .init(id: "frost-base", group: "frost", name: "白霜 · 基础", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts/base.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "973beb1203cfd3340b4829d984f193feaf72e3cf", pinnedSHA256: "9067f23b4505e57f5e380b154e22fb2fbbf2111ab0611f8bc6e8b82efa90b891", pinnedByteCount: 9_937_947), - .init(id: "frost-ext", group: "frost", name: "白霜 · 扩展", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts/ext.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "6b8a6c84c6c7b638f099b6d2a5a91f35662fb767", pinnedSHA256: "44b78e4feb8a3b302298844061626aa4e2194901b8ed4ee6acec84261b6aa90d", pinnedByteCount: 7_711_375), - .init(id: "frost-idiom", group: "frost", name: "白霜 · 成语与诗句", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts_cell/idiom.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "c9bf62a4de8b10efa6aec1dca56fe3e423bf240e", pinnedSHA256: "341d777d4fd077ddb534e47f8ba27a8dd80d085fd0b847548a97a6666d644113", pinnedByteCount: 1_665_339), - .init(id: "ice-base", group: "ice", name: "雾凇 · 基础", repository: "iDvel/rime-ice", branch: "main", path: "cn_dicts/base.dict.yaml", pinnedCommit: "fbb516b2786e4d5444383706d13c31c2e4d10c08", pinnedBlobSHA: "af59fe3a2259ed91ae642aab09422599a0557017", pinnedSHA256: "6c594bbd03425600aa36894b713f3d268bd2a11099833a312160249e0a3f0082", pinnedByteCount: 16_620_279), - .init(id: "ice-ext", group: "ice", name: "雾凇 · 扩展", repository: "iDvel/rime-ice", branch: "main", path: "cn_dicts/ext.dict.yaml", pinnedCommit: "fbb516b2786e4d5444383706d13c31c2e4d10c08", pinnedBlobSHA: "0a3d5aa7e1bb1dc73f8d73448a1986031ae6819e", pinnedSHA256: "5435dd8b75d6eb688787a25b6e302867152ef401bec7b263136ab1e2a2ecf4ae", pinnedByteCount: 11_923_397), - .init(id: "frost-computer", group: "specialty", name: "白霜 · 计算机", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts_cell/computer.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "0a9592d65d3bfd15116b3b48a45268a81b2716fa", pinnedSHA256: "a92fe61d48b53d1d20f1e66be4ca83ac2e8be0caa1fa3f383c5c15de6cb7d5ea", pinnedByteCount: 1_030), - .init(id: "frost-exthot", group: "specialty", name: "白霜 · 网络热词", repository: "gaboolic/rime-frost", branch: "master", path: "cn_dicts_cell/exthot.dict.yaml", pinnedCommit: "19167adfe67fcba2f65c336117557639ff254ddb", pinnedBlobSHA: "caa7b822e2e994262ec660d3416ba174b151cc74", pinnedSHA256: "d5f8bda70bb621f82ed85e4e8dbe8386c81effb856ff82ebb5c708d18ae993c1", pinnedByteCount: 50_277), - .init(id: "legacy", group: "legacy", name: "旧版拼音 · 兼容增量", repository: "rime/rime-pinyin-simp", branch: "master", path: "pinyin_simp.dict.yaml", pinnedCommit: "0c6861ef7420ee780270ca6d993d18d4101049d0", pinnedBlobSHA: "6f2e996d2792416cb7f41bb49967a1dec7060c92", pinnedSHA256: "e341598343a0f0f2035bb1aafc34a7f3bb7887deeecb3f60796262aaa2983e6b", pinnedByteCount: 1_266_216) - ] + package static let sources = IFDictionaryGenerator.catalog() } diff --git a/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift b/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift index b5a55a3..d1d26b2 100644 --- a/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift +++ b/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift @@ -15,9 +15,9 @@ extension DictionaryGeneratorTests { for test in cases { let (inputs, catalog) = fixture(test.bodies ?? [:], header: test.header ?? "---\nimport_tables: [ignored]\n...\n") do { - let generated = try IFDictionaryGenerator.generate(inputs: inputs, + let generated = try IFReferenceDictionaryGenerator.generate(inputs: inputs, corrections: Data((test.corrections ?? "").utf8), catalog: catalog) - let schemas = try IFSpellingGenerator.generate(dictionary: generated.dictionary) + let schemas = try IFReferenceSpellingGenerator.generate(dictionary: generated.dictionary) results[test.name] = [ "manifest": try JSONSerialization.jsonObject(with: generated.manifest.encoded()), "spellingSHA256": schemas.mapValues { IFDictionaryHash.sha256($0) } diff --git a/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift b/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift index c1b9604..5cd2cd3 100644 --- a/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift +++ b/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift @@ -40,8 +40,10 @@ struct DictionaryGeneratorTests { } try spellingGeneration() expect(IFDictionaryHash.gitBlob(Data("hello\n".utf8)) == "ce013625030ba8dba906f756967f9e9ca394464a", "Git blob includes byte-count header") - expect(try IFDictionaryGenerator.normalizedReading(" LÜ\u{a0}SE ") == "lv se", "Pinyin whitespace, case, ü normalization") - expect(try IFDictionaryGenerator.normalizedReading("LU\u{308} SE") == "lv se", "Decomposed ü normalizes") + for reading in [" LÜ\u{a0}SE ", "LU\u{308} SE"] { + expect(text(try generate(["frost-8105": "绿\t\(reading)\t100\n"])).contains("绿\tlv se\t100\n"), + "Pinyin whitespace, case and composed/decomposed ü normalize through Rust") + } let union = try generate([ "frost-8105": "甲\tjia\t100\n绿\tlü\t0\n行\txing\t22\n", "frost-base": "甲\tJIA\t900\n行\thang\t33\n〇\tling\t0\n", @@ -172,8 +174,17 @@ struct DictionaryGeneratorTests { expect(result.manifest.entryCount == IFDictionaryCatalog.initialEntryCount, "Initial pinned source coverage") expect(try Data(contentsOf: destination.appendingPathComponent(IFDictionaryCatalog.dictionaryFilename)) == result.dictionary, "Build CLI and shared runtime module generate identical bytes") - expect(try Data(contentsOf: destination.appendingPathComponent(IFDictionaryManifest.filename)) == result.manifest.encoded(), - "Build CLI and shared runtime module generate identical metadata") + expect(try JSONDecoder().decode(IFDictionaryManifest.self, + from: Data(contentsOf: destination.appendingPathComponent(IFDictionaryManifest.filename))) == result.manifest, + "Build CLI and shared runtime module generate equivalent metadata") + let reference = try IFReferenceDictionaryGenerator.generate(inputs: inputs, corrections: corrections) + expect(reference.dictionary == result.dictionary, "Production Rust retains Swift dictionary bytes") + expect(reference.manifest == result.manifest, "Production Rust retains pinned Swift manifest values") + let referenceDirectory = destination.deletingLastPathComponent().appendingPathComponent("test-chinese-reference") + try FileManager.default.createDirectory(at: referenceDirectory, withIntermediateDirectories: true) + try reference.dictionary.write(to: referenceDirectory.appendingPathComponent(IFDictionaryCatalog.dictionaryFilename)) + try reference.manifest.encoded().write(to: referenceDirectory.appendingPathComponent(IFDictionaryManifest.filename)) + try IFReferenceSpellingGenerator.write(dictionary: reference.dictionary, to: referenceDirectory) for (name, bytes) in try IFSpellingGenerator.generate(dictionary: result.dictionary) { expect(try Data(contentsOf: destination.appendingPathComponent(name)) == bytes, "Build CLI and runtime spelling bytes match: \(name)") } diff --git a/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift b/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift new file mode 100644 index 0000000..949d859 --- /dev/null +++ b/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift @@ -0,0 +1,286 @@ +import Foundation +import InkFlowDomain + +/// Reads only the tab-separated body. Remote headers, imports and executable configuration are never interpreted. +package enum IFReferenceDictionaryGenerator { + package static let maximumWeight = Int(Int32.max) + private struct Key: Hashable { let text: String; let reading: String } + private struct Row { let key: Key; let weight: Int } + + package static func generate(inputs: [IFDictionaryInput], corrections: Data = Data(), + catalog: [IFDictionarySourceSpec] = IFDictionaryCatalog.sources) throws -> IFDictionaryGeneration { + let expected = catalog.map(\.id) + guard inputs.map(\.receipt.id) == expected else { + throw IFDictionaryError("source-set", "Expected every catalog source exactly once, in precedence order") + } + var groups = [String: [Key: Int]]() + var receipts = [IFDictionarySourceReceipt]() + for (spec, input) in zip(catalog, inputs) { + try validate(input) + guard input.receipt.repository == spec.repository, input.receipt.path == spec.path else { + throw IFDictionaryError("source-location", source: spec.id, "Source repository or path differs from the catalog") + } + if !spec.isUpdatable { + guard input.receipt.commit == spec.pinnedCommit, input.receipt.blobSHA == spec.pinnedBlobSHA, + input.receipt.sha256 == spec.pinnedSHA256 else { + throw IFDictionaryError("legacy-changed", source: spec.id, "Compatibility source must remain pinned") + } + } + let rows = try parse(input.data, source: spec.id) + var receipt = input.receipt + receipt.recordCount = rows.count + receipts.append(receipt) + var group = groups.removeValue(forKey: spec.group) ?? [:] + for row in rows where group[row.key] == nil { group[row.key] = row.weight } + groups[spec.group] = group + } + guard let frost = groups["frost"] else { throw IFDictionaryError("source-set", "Missing Frost baseline") } + var union = frost + var calibrations = [IFDictionaryCalibration]() + for groupName in ["ice", "legacy"] { + guard let group = groups[groupName] else { + throw IFDictionaryError("source-set", source: groupName, "Missing source group") + } + let calibration = try calibrate(source: group, baseline: frost, name: groupName) + calibrations.append(calibration) + for (key, weight) in group where union[key] == nil { + let factor = calibration.buckets[bucket(key.reading) - 1].multiplier + let mapped = Double(weight) * factor + // Saturate only mapped source weights at Rime's signed integer boundary. + union[key] = weight == 0 ? 0 : max(1, Int(min(Double(maximumWeight), mapped.rounded()))) + } + } + // Specialty dictionaries only fill gaps after the established calibrated union. + // They cannot change existing weights or the overlap used for calibration. + for (key, weight) in groups["specialty"] ?? [:] where union[key] == nil { union[key] = weight } + for row in try parseCorrections(corrections) { union[row.key] = row.weight } + let keys = union.keys.sorted { + $0.text == $1.text ? $0.reading.utf8.lexicographicallyPrecedes($1.reading.utf8) + : $0.text.utf8.lexicographicallyPrecedes($1.text.utf8) + } + var body = String() + body.reserveCapacity(union.count * 36) + for key in keys { body += "\(key.text)\t\(key.reading)\t\(union[key]!)\n" } + let contentHash = IFDictionaryHash.sha256(Data(body.utf8)) + let versionHash = IFDictionaryHash.sha256(Data("recipe:\(IFDictionaryCatalog.recipeVersion)\n\(body)".utf8)) + let version = "r\(IFDictionaryCatalog.recipeVersion)-\(versionHash)" + let header = """ + # Generated by InkFlow from Rime Frost, Rime Ice and pinned pinyin_simp. + # Baseline weights are retained; specialty data fills gaps. See bundled Licenses. + --- + name: pinyin_simp + version: '\(version)' + sort: by_weight + use_preset_vocabulary: false + ... + + """ + let dictionary = Data((header + body).utf8) + let manifest = IFDictionaryManifest(formatVersion: 1, recipeVersion: IFDictionaryCatalog.recipeVersion, + contentVersion: version, entryCount: union.count, contentSHA256: contentHash, + dictionarySHA256: IFDictionaryHash.sha256(dictionary), correctionsSHA256: IFDictionaryHash.sha256(corrections), + sources: receipts, calibrations: calibrations) + return IFDictionaryGeneration(dictionary: dictionary, manifest: manifest) + } + + package static func validate(_ input: IFDictionaryInput) throws { + let receipt = input.receipt + guard IFDictionaryHash.isHex(receipt.commit, length: 40), IFDictionaryHash.isHex(receipt.blobSHA, length: 40), + IFDictionaryHash.isHex(receipt.sha256, length: 64) else { + throw IFDictionaryError("invalid-receipt", source: receipt.id, "Invalid commit or content identifier") + } + guard !input.data.isEmpty, input.data.count <= IFDictionaryCatalog.maximumSourceBytes, + input.data.count == receipt.byteCount else { + throw IFDictionaryError("source-size", source: receipt.id, "Received \(input.data.count) bytes; expected \(receipt.byteCount)") + } + guard IFDictionaryHash.gitBlob(input.data) == receipt.blobSHA, + IFDictionaryHash.sha256(input.data) == receipt.sha256 else { + throw IFDictionaryError("source-checksum", source: receipt.id, "Git blob or SHA-256 verification failed") + } + } + + package static func normalizedReading(_ value: String) throws -> String { + let code = value.precomposedStringWithCanonicalMapping.lowercased().replacingOccurrences(of: "ü", with: "v") + let syllables = code.split(whereSeparator: \.isWhitespace) + guard !syllables.isEmpty, syllables.count <= 128, + syllables.allSatisfy({ !$0.isEmpty && $0.utf8.count <= 8 && $0.utf8.allSatisfy { (97...122).contains($0) } }) else { + throw IFDictionaryError("invalid-reading", "Expected explicit, toneless Pinyin syllables") + } + return syllables.joined(separator: " ") + } + + private static func parse(_ data: Data, source: String) throws -> [Row] { + var rows = [Row]() + try readRows(data, source: source) { rows.append($0) } + return rows + } + + package static func spellingSyllables(in dictionary: Data) throws -> Set { + var syllables = Set() + try readRows(dictionary, source: "generated-spelling") { row in + syllables.formUnion(row.key.reading.split(separator: " ").map(String.init)) + } + return syllables + } + + private static func readRows(_ data: Data, source: String, receive: (Row) -> Void) throws { + // generate has already verified the complete byte count and both content hashes. + guard let string = String(data: data, encoding: .utf8) else { + throw IFDictionaryError("source-format", source: source, "Expected complete UTF-8 dictionary") + } + var inBody = false, sawHeader = false + var count = 0 + for (offset, rawLine) in string.split(separator: "\n", omittingEmptySubsequences: false).enumerated() { + let line = rawLine.hasSuffix("\r") ? rawLine.dropLast() : rawLine + guard line.utf8.count <= 8_192 else { + throw IFDictionaryError("source-format", source: source, line: offset + 1, "Line exceeds 8192 bytes") + } + if !inBody { + if line == "---" { sawHeader = true } + if line == "...", sawHeader { inBody = true } + continue + } + if line.trimmingCharacters(in: .whitespaces).isEmpty || line.trimmingCharacters(in: .whitespaces).hasPrefix("#") { continue } + let fields = line.split(separator: "\t", omittingEmptySubsequences: false) + guard fields.count == 3 else { + throw IFDictionaryError("source-format", source: source, line: offset + 1, "Expected text, reading and integer weight") + } + receive(try row(fields, source: source, line: offset + 1)) + count += 1 + guard count <= 3_000_000 else { throw IFDictionaryError("source-format", source: source, "Too many records") } + } + guard inBody, count > 0 else { throw IFDictionaryError("source-format", source: source, "Missing dictionary body") } + } + + private static func row(_ fields: [Substring], source: String, line: Int) throws -> Row { + let text = String(fields[0]) + let hasHan = text.unicodeScalars.contains { scalar in + scalar.value == 0x3007 || (0x3400...0x9fff).contains(scalar.value) || (0xf900...0xfaff).contains(scalar.value) + || (0x20000...0x323af).contains(scalar.value) + } + guard hasHan, text.count <= 256, text == text.trimmingCharacters(in: .whitespacesAndNewlines), + !text.unicodeScalars.contains(where: { CharacterSet.controlCharacters.contains($0) }), + !fields[2].isEmpty, fields[2].utf8.allSatisfy({ (48...57).contains($0) }), + let weight = Int(fields[2]), weight <= maximumWeight else { + throw IFDictionaryError("source-format", source: source, line: line, "Invalid Chinese text or integer weight (0...\(maximumWeight))") + } + do { return Row(key: Key(text: text, reading: try normalizedReading(String(fields[1]))), weight: weight) } + catch { throw IFDictionaryError("invalid-reading", source: source, line: line, "Expected explicit, toneless Pinyin syllables") } + } + + private static func parseCorrections(_ data: Data) throws -> [Row] { + guard data.count <= 1_048_576, let string = String(data: data, encoding: .utf8) else { + throw IFDictionaryError("correction-format", "Expected UTF-8 corrections under 1 MiB") + } + var rows = [Row](), seen = Set() + for (offset, rawLine) in string.split(separator: "\n", omittingEmptySubsequences: false).enumerated() { + let line = rawLine.hasSuffix("\r") ? rawLine.dropLast() : rawLine + if line.trimmingCharacters(in: .whitespaces).isEmpty || line.trimmingCharacters(in: .whitespaces).hasPrefix("#") { continue } + let fields = line.split(separator: "\t", omittingEmptySubsequences: false) + guard fields.count == 4, !fields[3].trimmingCharacters(in: .whitespaces).isEmpty else { + throw IFDictionaryError("correction-format", line: offset + 1, "Expected text, reading, replacement weight and reason") + } + let entry = try row(Array(fields.prefix(3)), source: "corrections", line: offset + 1) + guard seen.insert(entry.key).inserted else { + throw IFDictionaryError("correction-duplicate", line: offset + 1, "Duplicate term/reading correction") + } + rows.append(entry) + } + return rows + } + + private static func bucket(_ reading: String) -> Int { min(5, reading.split(separator: " ").count) } + private static func medianMultiplier(_ values: [Double]) -> Double { + let sorted = values.sorted(), middle = values.count / 2 + let median = sorted.count.isMultiple(of: 2) ? (sorted[middle - 1] + sorted[middle]) / 2 : sorted[middle] + return exp(median) + } + private static func calibrate(source: [Key: Int], baseline: [Key: Int], name: String) throws -> IFDictionaryCalibration { + var groups = Array(repeating: [Double](), count: 5) + for (key, weight) in source where weight > 0 { + if let reference = baseline[key], reference > 0 { + groups[bucket(key.reading) - 1].append(log(Double(reference) / Double(weight))) + } + } + let all = groups.flatMap { $0 } + guard !all.isEmpty else { throw IFDictionaryError("calibration-empty", source: name, "No positive-weight overlap with Frost") } + let overall = medianMultiplier(all) + let buckets = groups.enumerated().map { index, values in + IFDictionaryCalibrationBucket(syllables: index + 1, pairCount: values.count, + multiplier: values.count >= 100 ? medianMultiplier(values) : overall, usedOverall: values.count < 100) + } + return IFDictionaryCalibration(sourceGroup: name, pairCount: all.count, overallMultiplier: overall, buckets: buckets) + } +} + +/// Build preparation and the isolated update worker compile the same native algebra. +package enum IFReferenceSpellingGenerator { + package static func generate(dictionary: Data) throws -> [String: Data] { + let syllables = try IFReferenceDictionaryGenerator.spellingSyllables(in: dictionary) + var schemas = [String: Data]() + for profile in 0..<32 { + let name = "inkflow_spelling_\(profile)" + let rules = try algebra(syllables: syllables, profile: profile) + let schema = """ + # Generated by IFSpellingGenerator. Compilation dependency only. + schema: + schema_id: \(name) + name: InkFlow spelling \(profile) + version: '1.1' + translator: + dictionary: pinyin_simp + prism: \(name) + speller: + algebra: + \(rules.map { " - \($0)" }.joined(separator: "\n")) + + """ + schemas[name + ".schema.yaml"] = Data(schema.utf8) + } + return schemas + } + + package static func write(dictionary: Data, to directory: URL) throws { + let schemas = try generate(dictionary: dictionary) + for name in schemas.keys.sorted() { + try schemas[name]!.write(to: directory.appendingPathComponent(name), options: .atomic) + } + } + + private static func algebra(syllables: Set, profile: Int) throws -> [String] { + // One list owns normal equivalent spellings and explicitly enabled fuzzy pairs. + // These simple derivations also define the legal full-spelling collision set. + var equivalents = [("^([nl])ue$", "$1ve"), ("^([jqxy])u", "$1v")] + for (initial, bit) in [("z", 4), ("c", 8), ("s", 16)] where profile & bit != 0 { + equivalents += [("^\(initial)h", initial), ("^\(initial)([^h])", initial + "h$1")] + } + var rules = equivalents.map { "derive/\($0.0)/\($0.1)/" } + if profile & 1 != 0 { + rules += ["abbrev/^([a-z]).+$/$1/", "abbrev/^([zcs]h).+$/$1/"] + } + if profile & 2 != 0 { + var legal = syllables + for (pattern, replacement) in equivalents { + let expression = try NSRegularExpression(pattern: pattern) + legal.formUnion(legal.map { + expression.stringByReplacingMatches(in: $0, range: NSRange($0.startIndex..., in: $0), + withTemplate: replacement) + }) + } + // Only copies carry typo transformations. Rime retains fuzzy properties, + // ordered combinations and the original paths. Guard the final aliases, + // then remove the temporary marker before compiling the prism. + rules += [ + "derive/^(.*)$/~$1/", + "fuzz/^~([zcs])h/~h$1/", + "fuzz/^~([bpmfdtnlgkhjqxrzcsyw])(?=[a-z])/~$1$1/", + "fuzz/^~(.*)([aeiou])ng$/~$1$2gn/", + "fuzz/^~(.*)ao$/~$1oa/", + "fuzz/^~(.*)([iu])a(o|ng?)?$/~$1a$2$3/", + "erase/^~(\(legal.sorted().joined(separator: "|")))$/", + "xform/^~//", + ] + } + return rules + } +} diff --git a/Core/config/chinese-sources.json b/Core/config/chinese-sources.json new file mode 100644 index 0000000..af2de65 --- /dev/null +++ b/Core/config/chinese-sources.json @@ -0,0 +1,110 @@ +[ + { + "branch" : "master", + "group" : "frost", + "id" : "frost-8105", + "name" : "白霜 · 字表", + "path" : "cn_dicts/8105.dict.yaml", + "pinnedBlobSHA" : "9cbfe2acf59bb3124993d4b8b3b271f8e246b72d", + "pinnedByteCount" : 99367, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "5a6bb545d07140406208728aeed70706e84279daeee132f10b069cd6387a042a", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "master", + "group" : "frost", + "id" : "frost-base", + "name" : "白霜 · 基础", + "path" : "cn_dicts/base.dict.yaml", + "pinnedBlobSHA" : "973beb1203cfd3340b4829d984f193feaf72e3cf", + "pinnedByteCount" : 9937947, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "9067f23b4505e57f5e380b154e22fb2fbbf2111ab0611f8bc6e8b82efa90b891", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "master", + "group" : "frost", + "id" : "frost-ext", + "name" : "白霜 · 扩展", + "path" : "cn_dicts/ext.dict.yaml", + "pinnedBlobSHA" : "6b8a6c84c6c7b638f099b6d2a5a91f35662fb767", + "pinnedByteCount" : 7711375, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "44b78e4feb8a3b302298844061626aa4e2194901b8ed4ee6acec84261b6aa90d", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "master", + "group" : "frost", + "id" : "frost-idiom", + "name" : "白霜 · 成语与诗句", + "path" : "cn_dicts_cell/idiom.dict.yaml", + "pinnedBlobSHA" : "c9bf62a4de8b10efa6aec1dca56fe3e423bf240e", + "pinnedByteCount" : 1665339, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "341d777d4fd077ddb534e47f8ba27a8dd80d085fd0b847548a97a6666d644113", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "main", + "group" : "ice", + "id" : "ice-base", + "name" : "雾凇 · 基础", + "path" : "cn_dicts/base.dict.yaml", + "pinnedBlobSHA" : "af59fe3a2259ed91ae642aab09422599a0557017", + "pinnedByteCount" : 16620279, + "pinnedCommit" : "fbb516b2786e4d5444383706d13c31c2e4d10c08", + "pinnedSHA256" : "6c594bbd03425600aa36894b713f3d268bd2a11099833a312160249e0a3f0082", + "repository" : "iDvel/rime-ice" + }, + { + "branch" : "main", + "group" : "ice", + "id" : "ice-ext", + "name" : "雾凇 · 扩展", + "path" : "cn_dicts/ext.dict.yaml", + "pinnedBlobSHA" : "0a3d5aa7e1bb1dc73f8d73448a1986031ae6819e", + "pinnedByteCount" : 11923397, + "pinnedCommit" : "fbb516b2786e4d5444383706d13c31c2e4d10c08", + "pinnedSHA256" : "5435dd8b75d6eb688787a25b6e302867152ef401bec7b263136ab1e2a2ecf4ae", + "repository" : "iDvel/rime-ice" + }, + { + "branch" : "master", + "group" : "specialty", + "id" : "frost-computer", + "name" : "白霜 · 计算机", + "path" : "cn_dicts_cell/computer.dict.yaml", + "pinnedBlobSHA" : "0a9592d65d3bfd15116b3b48a45268a81b2716fa", + "pinnedByteCount" : 1030, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "a92fe61d48b53d1d20f1e66be4ca83ac2e8be0caa1fa3f383c5c15de6cb7d5ea", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "master", + "group" : "specialty", + "id" : "frost-exthot", + "name" : "白霜 · 网络热词", + "path" : "cn_dicts_cell/exthot.dict.yaml", + "pinnedBlobSHA" : "caa7b822e2e994262ec660d3416ba174b151cc74", + "pinnedByteCount" : 50277, + "pinnedCommit" : "19167adfe67fcba2f65c336117557639ff254ddb", + "pinnedSHA256" : "d5f8bda70bb621f82ed85e4e8dbe8386c81effb856ff82ebb5c708d18ae993c1", + "repository" : "gaboolic/rime-frost" + }, + { + "branch" : "master", + "group" : "legacy", + "id" : "legacy", + "name" : "旧版拼音 · 兼容增量", + "path" : "pinyin_simp.dict.yaml", + "pinnedBlobSHA" : "6f2e996d2792416cb7f41bb49967a1dec7060c92", + "pinnedByteCount" : 1266216, + "pinnedCommit" : "0c6861ef7420ee780270ca6d993d18d4101049d0", + "pinnedSHA256" : "e341598343a0f0f2035bb1aafc34a7f3bb7887deeecb3f60796262aaa2983e6b", + "repository" : "rime/rime-pinyin-simp" + } +] \ No newline at end of file diff --git a/Core/scripts/build-dictionary-generator.sh b/Core/scripts/build-dictionary-generator.sh index a7056e0..e2a595e 100644 --- a/Core/scripts/build-dictionary-generator.sh +++ b/Core/scripts/build-dictionary-generator.sh @@ -1,6 +1,15 @@ #!/bin/bash set -euo pipefail cd "$(dirname "$0")/../.." -binary=build/dictionary-generator -source Core/scripts/swift-package.sh -build_core_product dictionary-generator "$binary" release +export CARGO_TARGET_DIR="$PWD/build/dictionary/cargo" +cargo build --locked --release --manifest-path Core/Portable/dictionary/Cargo.toml +mkdir -p build/dictionary/include +cp -p "$CARGO_TARGET_DIR/release/inkflow-dictionary" build/dictionary-generator +# SwiftPM tracks this C include, so a changed Rust archive also relinks its callers. +identity=$(shasum -a 256 "$CARGO_TARGET_DIR/release/libinkflow_dictionary.a" | cut -d ' ' -f 1) +printf '#define INKFLOW_DICTIONARY_BUILD_ID "%s"\n' "$identity" > build/dictionary/include/build_identity.h.tmp +if ! cmp -s build/dictionary/include/build_identity.h.tmp build/dictionary/include/build_identity.h; then + mv build/dictionary/include/build_identity.h.tmp build/dictionary/include/build_identity.h +else + rm build/dictionary/include/build_identity.h.tmp +fi diff --git a/Core/scripts/prepare-rime.sh b/Core/scripts/prepare-rime.sh index 533d3ec..1c73631 100644 --- a/Core/scripts/prepare-rime.sh +++ b/Core/scripts/prepare-rime.sh @@ -14,7 +14,7 @@ cleanup() { [[ -z "$staging" ]] || rm -rf "$staging" } trap cleanup EXIT -for input_root in Core/Package.swift Core/Sources Core/Tools Core/scripts schemas Core/config Core/Data \ +for input_root in Core/Package.swift Core/Sources Core/Tools Core/scripts Core/Portable/dictionary schemas Core/config Core/Data \ build/deps/rime-pinyin-simp-* build/deps/rime-easy-en-* build/dictionary-sources; do [[ -e "$input_root" ]] || continue find "$input_root" -type f -print diff --git a/Core/scripts/prepare-spelling.sh b/Core/scripts/prepare-spelling.sh index c96b576..f3ac583 100644 --- a/Core/scripts/prepare-spelling.sh +++ b/Core/scripts/prepare-spelling.sh @@ -6,4 +6,7 @@ destination=${1:?Usage: prepare-spelling.sh DESTINATION} # spelling and the context-ranking index after dictionary changes, using the same # generator as the update worker. bash Core/scripts/build-dictionary-generator.sh -build/dictionary-generator spelling "$destination/pinyin_simp.dict.yaml" "$destination" +staging=$(mktemp -d build/.spelling.XXXXXX) +trap 'rm -rf "$staging"' EXIT +build/dictionary-generator spelling "$destination/pinyin_simp.dict.yaml" "$staging/generated" +cp "$staging/generated/"*.schema.yaml "$destination/" diff --git a/Core/scripts/swift-package.sh b/Core/scripts/swift-package.sh index 7fe079d..0fe464f 100755 --- a/Core/scripts/swift-package.sh +++ b/Core/scripts/swift-package.sh @@ -3,6 +3,7 @@ build_core_product() { local product="$1" output="$2" configuration="${3:-debug}" [[ "$configuration" == debug || "$configuration" == release ]] || return 2 + bash Core/scripts/build-dictionary-generator.sh || return $? local scratch="$PWD/build/core-swiftpm" local args=(--package-path "$PWD/Core" --disable-sandbox --scratch-path "$scratch" --cache-path "$scratch/cache" diff --git a/Core/scripts/test-prepare-rime.sh b/Core/scripts/test-prepare-rime.sh index 1476a91..401dad7 100755 --- a/Core/scripts/test-prepare-rime.sh +++ b/Core/scripts/test-prepare-rime.sh @@ -60,12 +60,14 @@ awk -F '\t' ' fixture=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-rime-policy.XXXXXX") trap 'rm -rf "$fixture"' EXIT -mkdir -p "$fixture/Core/Tools/DictionaryGeneratorTool" "$fixture/Core/Sources/InkFlowDomain" "$fixture/Core/scripts" "$fixture/Core/config" "$fixture/Core/Data" "$fixture/schemas" \ +mkdir -p "$fixture/Core/Portable/dictionary/src" "$fixture/Core/Tools/DictionaryGeneratorTool" "$fixture/Core/Sources/InkFlowDomain" "$fixture/Core/scripts" "$fixture/Core/config" "$fixture/Core/Data" "$fixture/schemas" \ "$fixture/build/deps/rime-pinyin-simp-fixture" "$fixture/build/deps/rime-easy-en-fixture" cp Core/scripts/prepare-rime.sh Core/scripts/prepare-spelling.sh "$fixture/Core/scripts/" printf 'fixture package\n' > "$fixture/Core/Package.swift" printf 'fixture entry\n' > "$fixture/Core/Tools/DictionaryGeneratorTool/main.swift" printf 'fixture generator implementation\n' > "$fixture/Core/Sources/InkFlowDomain/DictionaryGenerator.swift" +printf 'fixture Rust generator\n' > "$fixture/Core/Portable/dictionary/src/lib.rs" +printf 'fixture source catalog\n' > "$fixture/Core/config/chinese-sources.json" printf 'fixture build generator\n' > "$fixture/Core/scripts/build-dictionary-generator.sh" printf 'fixture SwiftPM wrapper\n' > "$fixture/Core/scripts/swift-package.sh" : > "$fixture/Core/Data/english-technology.tsv" @@ -78,7 +80,7 @@ cd "$(dirname "$0")/../.." mkdir -p "$1" cp build/deps/rime-pinyin-simp-fixture/pinyin_simp.dict.yaml "$1/" shasum -a 256 Core/Package.swift Core/Tools/DictionaryGeneratorTool/main.swift Core/Sources/InkFlowDomain/DictionaryGenerator.swift \ - Core/scripts/build-dictionary-generator.sh Core/scripts/swift-package.sh | shasum -a 256 | awk '{print $1}' \ + Core/scripts/build-dictionary-generator.sh Core/scripts/swift-package.sh Core/Portable/dictionary/src/lib.rs Core/config/chinese-sources.json | shasum -a 256 | awk '{print $1}' \ > "$1/dictionary-manifest.json" STUB # Spelling must run after the generated Chinese dictionary has been copied. The @@ -184,7 +186,7 @@ bash "$fixture/macOS/scripts/prepare-rime.sh" "$fixture/wrapper-output" diff -r "$fixture/output" "$fixture/wrapper-output" [[ $(find "$fixture/build/rime-cache" -name complete -type f | wc -l | tr -d ' ') == "$cache_count" ]] receipt=$(cat "$fixture/output/dictionary-manifest.json") -for changed in Core/Sources/InkFlowDomain/DictionaryGenerator.swift Core/Tools/DictionaryGeneratorTool/main.swift Core/scripts/build-dictionary-generator.sh; do +for changed in Core/Sources/InkFlowDomain/DictionaryGenerator.swift Core/Tools/DictionaryGeneratorTool/main.swift Core/scripts/build-dictionary-generator.sh Core/Portable/dictionary/src/lib.rs Core/config/chinese-sources.json; do cp "$fixture/$changed" "$fixture/source-before" printf '\nchanged closure\n' >> "$fixture/$changed" generate diff --git a/Package.swift b/Package.swift index 30555e8..a79786b 100644 --- a/Package.swift +++ b/Package.swift @@ -152,7 +152,13 @@ let package = Package( .target(name: "InkFlowEngineTestSupport", dependencies: ["InkFlowDomain", "InkFlowRime", "InkFlowCoreTestSupport", "InkFlowRankingTestSupport"], path: "Core/Tests/InkFlowEngineTestSupport", swiftSettings: strictSwiftSettings), .target(name: "InkFlowCoreTestSupport", dependencies: ["InkFlowDomain", "InkFlowRime"], path: "Core/Tests/InkFlowCoreTestSupport", swiftSettings: strictSwiftSettings), .target(name: "InkFlowRime", dependencies: ["InkFlowDomain", "CRime"], path: "Core/Sources/InkFlowRime", swiftSettings: strictSwiftSettings, linkerSettings: [.linkedLibrary("sqlite3")] + rimeLinkerSettings), - .target(name: "InkFlowDomain", path: "Core/Sources/InkFlowDomain", swiftSettings: strictSwiftSettings), + .target(name: "InkFlowDictionary", path: "Core/Sources/InkFlowDictionary", + publicHeadersPath: "include", + cSettings: [.unsafeFlags(["-I\(packageRoot)/build/dictionary/include"])], + linkerSettings: [.unsafeFlags(["-L\(packageRoot)/build/dictionary/cargo/release"]), + .linkedLibrary("inkflow_dictionary"), .linkedLibrary("resolv"), .linkedLibrary("iconv")]), + .target(name: "InkFlowDomain", dependencies: ["InkFlowDictionary"], path: "Core/Sources/InkFlowDomain", + swiftSettings: strictSwiftSettings), .target( name: "CRime", dependencies: ["InkFlowRimeNative"], diff --git a/macOS/Tools/QualityBuildMetadata.swift b/macOS/Tools/QualityBuildMetadata.swift index ed62900..260faac 100644 --- a/macOS/Tools/QualityBuildMetadata.swift +++ b/macOS/Tools/QualityBuildMetadata.swift @@ -131,8 +131,8 @@ struct QualityBuildMetadataTool { } private static func isBuildInput(_ path: String) -> Bool { - if ["Package.swift", "Package.resolved", "Core/Package.swift", "macOS/Info.plist"].contains(path) { return true } - let prefixes = ["Core/Sources/", "Core/Tools/", "Core/scripts/", "macOS/Sources/", "macOS/Quality/", "macOS/SwiftPM/", "macOS/DictionaryWorker/", "macOS/DictionaryTool/", "macOS/Tools/", "macOS/Resources/", + if ["Package.swift", "Package.resolved", "Core/Package.swift", "rust-toolchain.toml", "macOS/Info.plist"].contains(path) { return true } + let prefixes = ["Core/Portable/dictionary/", "Core/Sources/", "Core/Tools/", "Core/scripts/", "macOS/Sources/", "macOS/Quality/", "macOS/SwiftPM/", "macOS/DictionaryWorker/", "macOS/DictionaryTool/", "macOS/Tools/", "macOS/Resources/", "macOS/Design/", "Core/Data/", "Core/config/", "macOS/Licenses/", "schemas/"] if prefixes.contains(where: path.hasPrefix) { return true } let scripts = ["build.sh", "build-number.sh", "build-summary.sh", "build-icon.sh", "build-dictionary-generator.sh", "build-dictionary-worker.sh", diff --git a/macOS/scripts/swift-package.sh b/macOS/scripts/swift-package.sh index 148a775..d55299c 100644 --- a/macOS/scripts/swift-package.sh +++ b/macOS/scripts/swift-package.sh @@ -14,6 +14,7 @@ build_swift_product() { echo "SwiftPM configuration must be debug or release." >&2 return 2 } + bash Core/scripts/build-dictionary-generator.sh || return $? # Every consumer must resolve CRime's headers during explicit module scanning. local build_args=(--disable-sandbox --cache-path "$swiftpm_cache" --config-path "$swiftpm_config" --security-path "$swiftpm_security" --scratch-path "$swiftpm_scratch" diff --git a/macOS/scripts/test-quality-metadata.sh b/macOS/scripts/test-quality-metadata.sh index 35f88a3..715e84d 100644 --- a/macOS/scripts/test-quality-metadata.sh +++ b/macOS/scripts/test-quality-metadata.sh @@ -277,7 +277,8 @@ before=$("$tool" "$identity_repo" --build-snapshot) printf 'documentation only\n' >> "$identity_repo/README.md" [[ "$("$tool" "$identity_repo" --build-snapshot)" == "$before" ]] for resource in Core/Data/english-wordfreq.tsv Core/config/english.conf Core/scripts/prepare-rime.sh \ - Core/scripts/build-dictionary-generator.sh; do + Core/scripts/build-dictionary-generator.sh Core/Portable/dictionary/src/lib.rs \ + Core/Portable/dictionary/Cargo.lock Core/config/chinese-sources.json rust-toolchain.toml; do cp "$identity_repo/$resource" "$fixture/resource-before" printf '\n# changed shared resource input\n' >> "$identity_repo/$resource" snapshot=$("$tool" "$identity_repo" --build-snapshot) diff --git a/scripts/mac-remote.py b/scripts/mac-remote.py index 6f31b4c..33ea6b9 100644 --- a/scripts/mac-remote.py +++ b/scripts/mac-remote.py @@ -22,6 +22,7 @@ "engine", "engine-basic", "engine-options", "engine-english", "engine-context", "engine-custom-phrases", "controller", "quality-baseline", "ai-learning", "voice-lexicon", "preparation", "dictionary-generator", "quality-metadata", + "dictionary-source", "dictionary-store", "dictionary-worker", "dictionary-activation", "deployment", } From 2e2e1c695cf285191d80189f9508737c3f0ec28c Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 07:43:35 +0800 Subject: [PATCH 04/21] Keep Swift generator reference access within the test target --- .../DictionaryGeneratorTests/SwiftReferenceGenerator.swift | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift b/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift index 949d859..dd9e30d 100644 --- a/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift +++ b/Core/Tests/DictionaryGeneratorTests/SwiftReferenceGenerator.swift @@ -1,5 +1,5 @@ import Foundation -import InkFlowDomain +@testable import InkFlowDomain /// Reads only the tab-separated body. Remote headers, imports and executable configuration are never interpreted. package enum IFReferenceDictionaryGenerator { From e80071b28937b53f57173fe5604e8c26c2ecea71 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 08:00:29 +0800 Subject: [PATCH 05/21] Prepare target-native resources and package Rust dictionary notices --- .../references/rules.md | 2 +- Core/Portable/README.md | 9 +- Core/Portable/dictionary/README.md | 21 ++-- .../dictionary/include/inkflow_dictionary.h | 2 + Core/Portable/dictionary/src/ffi.rs | 10 ++ Core/Portable/native/bridge.cpp | 23 +++- Core/Portable/native/bridge.h | 3 +- Core/Portable/prepare-resources.sh | 31 ++++++ Core/Portable/src/bin/prepare-resources.rs | 104 ++++++++++++++++++ Core/Portable/src/ffi.rs | 7 +- Core/Portable/src/lib.rs | 20 +++- Core/Portable/tests/runtime.rs | 26 ++++- .../InkFlowDomain/DictionaryGenerator.swift | 2 + .../InkFlowDomain/DictionaryModels.swift | 4 +- Core/scripts/dictionary-notices.py | 38 +++++++ Core/scripts/prepare-rime.sh | 2 +- Core/scripts/resource-dependencies.sh | 27 +++++ NOTICE | 4 +- docs/portable-core-migration.md | 2 +- docs/remote-mac-baseline.md | 7 +- docs/shared-dictionary-preparation.md | 11 +- macOS/DEPENDENCIES.md | 2 +- macOS/DICTIONARIES.md | 8 +- macOS/Quality/ranking-sources.txt | 3 + macOS/scripts/build.sh | 1 + macOS/scripts/check-bundle.sh | 4 + macOS/scripts/dependencies.sh | 8 +- scripts/mac-remote.py | 4 +- scripts/tests/test_mac_remote.py | 4 +- 29 files changed, 346 insertions(+), 43 deletions(-) create mode 100644 Core/Portable/prepare-resources.sh create mode 100644 Core/Portable/src/bin/prepare-resources.rs create mode 100644 Core/scripts/dictionary-notices.py create mode 100644 Core/scripts/resource-dependencies.sh diff --git a/.agents/skills/inkflow-mixed-input-maintenance/references/rules.md b/.agents/skills/inkflow-mixed-input-maintenance/references/rules.md index cea8872..b5c2887 100644 --- a/.agents/skills/inkflow-mixed-input-maintenance/references/rules.md +++ b/.agents/skills/inkflow-mixed-input-maintenance/references/rules.md @@ -31,7 +31,7 @@ The implemented baseline. Recheck the sources linked from the [skill entrypoint] 7. **Mixed decoding is native, filtering is customized.** The supplemental dictionary imports `pinyin_simp`; Rime's `script_translator` performs segmentation and sentence generation with completion disabled. Lua emits only candidates reaching the current segment end and containing both ASCII letters and non-ASCII bytes. Public rows still require every contiguous ASCII run in the output to be an exact entry in the admitted mixed dictionary, cached only for the current input. This prevents separately admitted fragments such as `WOMEN` and `S` from reconstructing an excluded `WOMENS`. The byte test assumes the dictionary's Chinese/English contents; it is not a general Han-script detector. Partial results and pure Chinese/English results are filtered out. Unknown Latin tails are not appended as a mixed fallback. The translator qualities `-0.5` for mixed and `-1` for standalone English shape the native stream but are not the final ordering policy for equivalent page candidates. 8. **Personal mixed decoding uses real native spans.** `InkFlowRimeNative` registers `inkflow_mixed_personal` after every Rime initialization. For input up to `mixed_personal/max_input_length` (64 bytes by default), a one-row native predictive lookup prunes impossible prefixes; non-predictive lookup retains every exact, positively confirmed display variant from `inkflow_shared_english.userdb`. A real personal edge is inserted between native dictionary/syllabifier graphs of its actual surroundings; native `Poet` chooses a complete path through that edge. No input rewriting, invented syllable IDs, personal prefix completion, first-N lexicon scan, rebuild or new store is used. Mixed candidates retain component provenance and original input cursor spans. Their distinct read-only language has no corresponding Memory, so neither Chinese nor English learns the assembled sentence. The translator-lifetime reader never changes transactions during queries. 9. **Ambiguity and boundaries use native spelling evidence.** An unmarked lowercase personal code is excluded if it is one letter or a complete sequence of normal native Pinyin spellings. Actual uppercase input supplies explicit English intent; stored uppercase display alone does not. Native normal spans in the uncut input reject cuts inside reachable surrounding syllables. Abbreviation, correction and fuzzy paths are not evidence for this rejection. Both surroundings need complete native coverage. Personal display is one printable nonspace ASCII run, including aliases such as `cpp` → `C++`; adjacent public ASCII output cannot join it. Public ASCII runs remain independently admitted. All exact display variants remain distinct. Preedit derives from real input slices and native formatting, and candidate text retains exact stored display. Over-limit input uses the ordinary Chinese/public paths. This general policy replaces the former personal short-code blacklist. -10. **Compilation and Pinyin.** The dependency-only `easy_en` schema compiles literal English independently of Pinyin algebra. The mixed and primary schemas retain the `n/l` `ue`→`ve` and `j/q/x/y` `u`→`v` spelling derivations. `prepare-spelling.sh` and the dictionary update worker share `IFSpellingGenerator` to generate 32 dependency-only Chinese prisms for abbreviation, typo tolerance and three independent fuzzy pairs; all reuse the original Chinese table/user dictionary. Automatic typo aliases cannot occupy legal full syllables from the generated dictionary and the profile's normal/fuzzy equivalents. Native algebra guards the final aliases after combined typo transformations; legal-to-legal typos lose that automatic alias, while explicit fuzzy pairs remain available. The main schema includes the generated profile 2 algebra. The mixed schema has no generic abbreviation, so English words are not abbreviated. Abbreviation can add full-coverage Chinese candidates ahead of existing mixed results without changing ranking policy; test recall and exact selection instead of a fixed mixed first position. The main speller accepts upper/lower ASCII letters and declares space/apostrophe delimiters. These remain native Pinyin delimiters, not an English phrase-input API; `xi'an` continues to produce 西安. +10. **Compilation and Pinyin.** The dependency-only `easy_en` schema compiles literal English independently of Pinyin algebra. The mixed and primary schemas retain the `n/l` `ue`→`ve` and `j/q/x/y` `u`→`v` spelling derivations. `prepare-spelling.sh` and the dictionary update worker share the Rust spelling generator under `Core/Portable/dictionary/` (the worker calls it through `IFSpellingGenerator`) to generate 32 dependency-only Chinese prisms for abbreviation, typo tolerance and three independent fuzzy pairs; all reuse the original Chinese table/user dictionary. Automatic typo aliases cannot occupy legal full syllables from the generated dictionary and the profile's normal/fuzzy equivalents. Native algebra guards the final aliases after combined typo transformations; legal-to-legal typos lose that automatic alias, while explicit fuzzy pairs remain available. The main schema includes the generated profile 2 algebra. The mixed schema has no generic abbreviation, so English words are not abbreviated. Abbreviation can add full-coverage Chinese candidates ahead of existing mixed results without changing ranking policy; test recall and exact selection instead of a fixed mixed first position. The main speller accepts upper/lower ASCII letters and declares space/apostrophe delimiters. These remain native Pinyin delimiters, not an English phrase-input API; `xi'an` continues to produce 西安. 11. **Canonical English learning is isolated.** The original Chinese dictionary/translator and `pinyin_simp.userdb` remain independent. Standalone English uses Rime `Phrase`/`ShadowCandidate` selection learning in the named `inkflow_shared_english.userdb`. Records keep the exact displayed text with a lowercase normalized Latin lookup code, so `Hello` selected through `Hello` is recalled by `hello` without changing its display. Mixed decoding keeps its own user dictionary disabled and only reads exact committed records from that shared namespace; displaying or selecting a mixed candidate does not update the record. Only selecting a learnable standalone English candidate updates this namespace. Display, raw Return, cancellation, immediate undo, and ordinary Chinese selection do not learn English. 12. **Selection and editing share the existing engine/controller.** Paging, number/space selection, backspace, cursor edits, cancellation and session isolation are not separate English features. Swift stores a display-to-native permutation and applies every selection through its original native index. A selected standalone English completion learns its canonical full dictionary code, not the typed prefix. Any selected prefix bypasses final ranking; clearing or ending a composition discards its bounded preceding context. Unknown metadata preserves native order, and partial candidates remain selectable in their original slots. 13. **Explicit ASCII mode is native passthrough.** It does not consult either dictionary and remains available for excluded words. A standalone left Shift press-release toggles the requested mode; using left Shift as a modifier and pressing right Shift do not toggle. Shift and menu changes wait for the current composition to commit or cancel. ASCII punctuation is literal and does not overwrite the persisted Chinese punctuation preference. The compilation-only English/mixed/spelling schemas do not add separate selectable input sources. diff --git a/Core/Portable/README.md b/Core/Portable/README.md index 7aa93f5..c463c6d 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -1,6 +1,6 @@ # Desktop Rust/Rime probe -Minimal Rust runtime over librime for [#34](https://github.com/nervouna/InkFlow/issues/34). The Swift engine still ships. +Minimal Rust runtime over librime for [#34](https://github.com/nervouna/InkFlow/issues/34). The Swift session/ranking/learning engine still ships. ## Build and test @@ -12,6 +12,9 @@ From the checkout root: bash Core/Portable/test.sh # Send a committed revision to the dedicated Mac checkout: python3 scripts/mac-remote.py portable +# Compile and smoke-test full production resources in isolated directories: +bash Core/Portable/prepare-resources.sh +python3 scripts/mac-remote.py resources ``` The first build downloads checksum-verified source archives. Subsequent builds reuse them under `build/portable/`; native outputs and Cargo artifacts stay there too. Set `CMAKE_BUILD_PARALLEL_LEVEL` to change the native build parallelism (default 4). Do not run two builds in the same checkout. Remove `build/portable/` for a clean rebuild or after changing compiler/SDK/architecture. @@ -30,7 +33,7 @@ The test copies the tiny checked-in source fixture into a fresh temporary direct - One process-wide Rust mutex serializes **all** Rime calls, including initialization, deployment, reads, destruction, and finalization. Sessions can move between threads. No GUI event loop, Swift actor, or thread affinity is required. Another engine must not call librime outside this lock in the same process; old/new comparisons need separate processes. - There is at most one runtime. A second initialization returns `AlreadyRunning`. Sessions retain an `Arc` to the runtime, so dropping its public handle cannot finalize live sessions. The last session/runtime owner finalizes Rime. A poisoned lock fails subsequent normal operations; destructors still attempt cleanup without panicking. -- Callers supply existing shared-resource and writable user directories. Initialization, source deployment, and session creation are setup work. Deployment is rejected while sessions exist. The wrapper does not prepare resources, download, access SQLite, or record telemetry from `process_key`. +- Callers supply existing shared-resource and writable user directories. Initialization, source deployment, and session creation are setup work. `Runtime::with_cache` separates target-native cache files from personal data. `prepare` runs Rime's schema-list maintenance and checks its completion notification. Preparation and deployment are rejected while sessions exist. The wrapper does not prepare resources, download, access SQLite, or record telemetry from `process_key`. - All text and paths crossing this ABI are NUL-terminated UTF-8. Embedded NUL and non-UTF-8 paths are rejected. Snapshot caret/selection offsets count **bytes in the returned UTF-8 preedit**. Rust validates character boundaries; frontends must convert to UTF-16 or other platform units. Candidate text and comments are independent strings. - Keys are Rime/X11 keysyms with Rime modifier masks, not macOS virtual key codes or Linux hardware scan codes. `process_key` returns whether Rime handled the event. There is no surrounding-text or mobile editing API yet. `change_page` delegates paging to Rime. `select_candidate` accepts a zero-based current-page index and the latest snapshot from that session. It rejects snapshots from other sessions, superseded snapshots, and out-of-range indices before calling Rime. Keys, clears, native selection attempts, and page changes invalidate selection tokens even if Rime does not handle the operation. Cloned snapshots retain their token; changing public display fields cannot change the native candidate count used for validation. - Snapshots are copied into Rust-owned strings and vectors while the lock is held. They remain valid after another event, another snapshot, session destruction, and runtime teardown. Page and highlighted indices are zero-based native menu metadata; an empty menu has no selectable entry regardless of those fields. @@ -38,4 +41,4 @@ The test copies the tiny checked-in source fixture into a fresh temporary direct - Internal C callers must pass valid borrowed pointers and zero-initialized outputs. Status 0 is success, -1 native failure, -2 invalid session/schema, and -3 a caught C++ exception. A false/unhandled key is not an error. Native snapshots and commit buffers must be freed with their matching ABI functions, never Rust's allocator. Snapshot free clears the struct; it also accepts a zeroed snapshot. - Every throwing C++ entry point catches exceptions before returning to Rust. The ABI has no callbacks into Rust, so Rust unwinding cannot cross it. Invalid pointers remain programmer errors, and allocator aborts/native crashes are not recoverable status codes. A native exception can leave engine state partially changed; callers must stop using the runtime and restart it. -See also the [dictionary generator comparison](dictionary/README.md) and, before packaging, [the distribution review](licenses.md). +See also the [shared dictionary generator](dictionary/README.md), which supplies Chinese source generation and spelling profiles for build preparation and the Swift update worker, and, before packaging, [the distribution review](licenses.md). diff --git a/Core/Portable/dictionary/README.md b/Core/Portable/dictionary/README.md index 147706a..1faceca 100644 --- a/Core/Portable/dictionary/README.md +++ b/Core/Portable/dictionary/README.md @@ -1,6 +1,6 @@ -# Rust dictionary generator comparison +# Shared Rust dictionary generator -This crate ports the Chinese dictionary generator and all 32 spelling profiles for #35. It is an isolated library and command-line tool, with no Rime, Swift, GUI, or network dependency in generation itself. The shipping preparation scripts and dictionary update worker still use Swift. +This crate ports the Chinese dictionary generator and all 32 spelling profiles for #35. It is an isolated library and command-line tool, with no Rime, Swift, GUI, or network dependency in generation itself. The preparation scripts run its CLI; the Swift dictionary update worker calls the same library through its C ABI. ## Verify @@ -13,17 +13,16 @@ python3 scripts/mac-remote.py dictionary On Linux, the script tests Rust against the recorded Swift contract and pinned corpus. On macOS, it first runs the existing Swift dictionary-generator tests, exports a fresh contract, then compares Rust against those live results and the recorded fixtures. Both use the repository's Rust toolchain and this crate's `Cargo.lock`. -Preparation may download the pinned public dictionary sources. Downloads go under ignored `build/dictionary-parity/inputs/`; byte counts and SHA-256 are checked before use, then Rust verifies the Git blob hash and SHA-256 again. Existing verified Mac source caches can be reused. There are no downloads from the library or CLI. All generated output and Cargo artifacts stay under `build/dictionary-parity/`. The remote runner returns the catalog, reference results, Swift corpus summary (`corpus.json`), actual Rust corpus summary (`rust-corpus.json`), logs, and its usual exact-revision receipt. +Preparation may download the pinned public dictionary sources. Downloads go under ignored `build/dictionary-parity/inputs/`; byte counts and SHA-256 are checked before use, then Rust verifies the Git blob hash and SHA-256 again. Existing verified Mac source caches can be reused. There are no downloads from the library or CLI. Comparison outputs stay under `build/dictionary-parity/`; the shared Cargo artifacts live under `build/dictionary/cargo/`. The remote runner returns the catalog, reference results, Swift corpus summary (`corpus.json`), actual Rust corpus summary (`rust-corpus.json`), logs, and its usual exact-revision receipt. -The CLI takes an explicit source catalog and creates a new output directory: +The CLI uses the compiled production catalog and creates a new output directory. Tests may supply an explicit catalog before the source-directory argument: ```sh -build/dictionary-parity/cargo/release/inkflow-dictionary generate \ - Core/Portable/dictionary/fixtures/catalog.json \ +build/dictionary/cargo/release/inkflow-dictionary generate \ build/dictionary-parity/inputs build/dictionary-parity/inputs/legacy.yaml \ Core/config/chinese-overrides.tsv build/dictionary-parity/new-dictionary -build/dictionary-parity/cargo/release/inkflow-dictionary spelling \ +build/dictionary/cargo/release/inkflow-dictionary spelling \ build/dictionary-parity/new-dictionary/pinyin_simp.dict.yaml \ build/dictionary-parity/new-spelling ``` @@ -32,7 +31,7 @@ It refuses an existing output directory. Validation completes before creating ou ## In-process preparation boundary -`include/inkflow_dictionary.h` and its Clang module map expose an experimental C ABI from `libinkflow_dictionary.a`. It calls the same Rust generation, receipt validation, and spelling functions as the CLI. It does not launch a process, touch the filesystem, or initialize Rime. Call it from preparation/update workers, never from key handling. Shipping callers have not switched to this ABI yet. +`include/inkflow_dictionary.h` and its Clang module map expose an experimental C ABI from `libinkflow_dictionary.a`. It calls the same Rust generation, receipt validation, and spelling functions as the CLI. It does not launch a process, touch the filesystem, or initialize Rime. Call it from preparation/update workers, never from key handling. `Core/Sources/InkFlowDomain/DictionaryGenerator.swift` is the Swift production wrapper. The old Swift algorithm is retained only in the dictionary test target. Inputs are borrowed pointer/length buffers. Catalog and receipt buffers contain UTF-8 JSON; dictionary and correction inputs remain raw bytes so validation can reject malformed text. The boundary accepts at most 64 inputs, 1 MiB of catalog/corrections, 16 KiB per receipt, and the existing 128 MiB per source. Malformed transport inputs return `bridge-input` or `bridge-json`; domain failures retain their code, source, and line. Recoverable Rust panics return `bridge-panic`. Invalid foreign pointers, allocator aborts, and process crashes are outside that guarantee. @@ -42,7 +41,7 @@ The test script compiles and links a C consumer on both desktops. On macOS it al ## Reference and compatibility contract -`fixtures/cases.json` supplies 43 inputs to both implementations. The Swift test executable exports `catalog.json` and `reference.json`; Rust does not maintain another hand-written source catalog. The authoritative production catalog remains `IFDictionaryCatalog` in Swift. The copied catalog is a pinned fixture; the Mac check rejects drift from the live catalog. +`Core/config/chinese-sources.json` is the authoritative production catalog. Rust embeds it; `IFDictionaryCatalog` reads it through the ABI without filesystem access. Rust also owns the recipe version and maximum source size. `fixtures/cases.json` supplies 43 inputs to both implementations. The test-only Swift reference exports `catalog.json` and `reference.json`; the copied catalog is a pinned fixture, and checks reject drift from the production catalog. `reference.json` records parser errors, complete provenance manifests, dictionary hashes, and the hashes of all 32 spelling profiles for successful cases. `corpus.json` records the Swift output for the complete pinned source set and current Chinese corrections. Both fixture generation and actual output comparison use isolated build directories. @@ -59,8 +58,8 @@ Generated schema headers retain the existing `IFSpellingGenerator` wording for b ## Scope and dependencies -Replacing the Swift generator must keep one production preparation path, including the update worker. +Production callers share one generator; deployment, session policy, and personal-data migration still require their own verification. The Rust dependency versions and checksums are in `Cargo.lock`. Hashing uses RustCrypto SHA-1/SHA-256; SHA-1 is used only for the existing Git blob identity, alongside SHA-256 verification. Serde handles the explicit catalog/manifest format. Unicode normalization, segmentation, and general-category tables implement the Swift text contract. -The dependency metadata lists MIT/Apache-2.0 alternatives for most crates, Apache-2.0 for `unicode-general-category`, MIT for `generic-array` and `zmij`, and an additional Unicode-3.0 requirement for the build-time `unicode-ident` crate. The tool is not packaged in the app. +The dependency metadata lists MIT/Apache-2.0 alternatives for most crates, Apache-2.0 for `unicode-general-category`, MIT for `generic-array` and `zmij`, and an additional Unicode-3.0 requirement for the build-time `unicode-ident` crate. The app and worker statically link the library. `Core/scripts/dictionary-notices.py` collects the locked crate license files and the toolchain's standard-library notices for app packaging. The [dictionary distribution review](../licenses.md) still applies to the data. diff --git a/Core/Portable/dictionary/include/inkflow_dictionary.h b/Core/Portable/dictionary/include/inkflow_dictionary.h index 2aa349e..666fc96 100644 --- a/Core/Portable/dictionary/include/inkflow_dictionary.h +++ b/Core/Portable/dictionary/include/inkflow_dictionary.h @@ -26,6 +26,8 @@ typedef struct IFDResult IFDResult; /* Immutable compiled catalog, valid for the process lifetime; do not free. */ IFDBytes ifd_catalog(void); +uint32_t ifd_recipe_version(void); +size_t ifd_maximum_source_bytes(void); IFDResult* ifd_generate(IFDBytes catalog, const IFDInput* inputs, size_t count, IFDBytes corrections); IFDResult* ifd_spelling(IFDBytes dictionary); diff --git a/Core/Portable/dictionary/src/ffi.rs b/Core/Portable/dictionary/src/ffi.rs index fc98e46..3c74302 100644 --- a/Core/Portable/dictionary/src/ffi.rs +++ b/Core/Portable/dictionary/src/ffi.rs @@ -75,6 +75,16 @@ pub extern "C" fn ifd_catalog() -> Bytes { Bytes::borrowed(crate::CATALOG_JSON) } +#[unsafe(no_mangle)] +pub extern "C" fn ifd_recipe_version() -> u32 { + crate::RECIPE_VERSION +} + +#[unsafe(no_mangle)] +pub extern "C" fn ifd_maximum_source_bytes() -> usize { + crate::MAX_SOURCE_BYTES +} + #[unsafe(no_mangle)] pub unsafe extern "C" fn ifd_generate( catalog: Bytes, diff --git a/Core/Portable/native/bridge.cpp b/Core/Portable/native/bridge.cpp index f49cf20..f50937e 100644 --- a/Core/Portable/native/bridge.cpp +++ b/Core/Portable/native/bridge.cpp @@ -2,6 +2,7 @@ #include "InkFlowRimeNative.h" #include #include +#include #include #include #include @@ -10,6 +11,12 @@ namespace { RimeApi* api() { return rime_get_api(); } +std::atomic deployment_status{0}; +void deployment(void*, RimeSessionId, const char* type, const char* value) { + if (std::strcmp(type, "deploy")) return; + if (!std::strcmp(value, "success")) deployment_status.store(1); + if (!std::strcmp(value, "failure")) deployment_status.store(-1); +} struct Snapshot { std::string preedit; std::vector texts; @@ -33,12 +40,14 @@ struct Commit { const char* text(const char* value) { return value ? value : ""; } } -extern "C" int ifp_initialize(const char* shared, const char* user) { +extern "C" int ifp_initialize(const char* shared, const char* user, const char* cache) { try { RimeTraits traits{}; RIME_STRUCT_INIT(RimeTraits, traits); traits.shared_data_dir = shared; traits.user_data_dir = user; + traits.staging_dir = cache; + traits.prebuilt_data_dir = cache; traits.distribution_name = "InkFlow portable probe"; traits.distribution_code_name = "inkflow-portable"; traits.distribution_version = "0"; @@ -65,6 +74,18 @@ extern "C" int ifp_initialize(const char* shared, const char* user) { extern "C" int ifp_finalize(void) { try { api()->finalize(); return 0; } catch (...) { return -3; } } +extern "C" int ifp_prepare(void) { + try { + deployment_status.store(0); + api()->set_notification_handler(deployment, nullptr); + if (api()->start_maintenance(1)) api()->join_maintenance_thread(); + api()->set_notification_handler(nullptr, nullptr); + return deployment_status.load() == 1 ? 0 : -1; + } catch (...) { + try { api()->set_notification_handler(nullptr, nullptr); } catch (...) {} + return -3; + } +} extern "C" int ifp_deploy(const char* schema_path) { try { return api()->deploy_schema(schema_path) ? 0 : -1; } catch (...) { return -3; } diff --git a/Core/Portable/native/bridge.h b/Core/Portable/native/bridge.h index 28e1dac..0e11e65 100644 --- a/Core/Portable/native/bridge.h +++ b/Core/Portable/native/bridge.h @@ -27,7 +27,8 @@ typedef struct { int last_page; void* owner; } IFPSnapshot; -int ifp_initialize(const char* shared, const char* user); +int ifp_initialize(const char* shared, const char* user, const char* cache); +int ifp_prepare(void); int ifp_finalize(void); /* Preparation is explicit and must run before interactive sessions. */ int ifp_deploy(const char* schema_path); diff --git a/Core/Portable/prepare-resources.sh b/Core/Portable/prepare-resources.sh new file mode 100644 index 0000000..f8ec733 --- /dev/null +++ b/Core/Portable/prepare-resources.sh @@ -0,0 +1,31 @@ +#!/bin/bash +set -euo pipefail +cd "$(dirname "$0")/../.." +bash Core/scripts/resource-dependencies.sh +bash Core/scripts/prepare-chinese.sh --sources-only +python3 Core/Portable/build-native.py +mkdir -p build/portable +work=$(mktemp -d "$PWD/build/portable/resources.XXXXXX") +bash Core/scripts/prepare-rime.sh "$work/shared" +export CARGO_TARGET_DIR="$PWD/build/portable/cargo" +cargo run --locked --release --manifest-path Core/Portable/Cargo.toml --bin prepare-resources -- \ + "$work/shared" "$work/prepared" +python3 - "$work" <<'PY' +import hashlib +import json +from pathlib import Path +import platform +import sys +root = Path(sys.argv[1]) +def hashes(directory): + return {str(p.relative_to(directory)): hashlib.sha256(p.read_bytes()).hexdigest() + for p in sorted(directory.rglob('*')) if p.is_file()} +report = {'platform': platform.system(), 'architecture': platform.machine(), + 'nativeBuild': json.loads(Path('build/portable/native-build.json').read_text()), + 'sourceSHA256': hashes(root / 'shared'), + 'cacheSHA256': hashes(root / 'prepared/cache'), + 'manifest': json.loads((root / 'shared/dictionary-manifest.json').read_text())} +(root / 'resources.json').write_text(json.dumps(report, ensure_ascii=False, indent=2) + '\n') +print(f'PASS production resources: {root}') +PY +if [[ $# -eq 1 ]]; then cp "$work/resources.json" "$1/resources.json"; fi diff --git a/Core/Portable/src/bin/prepare-resources.rs b/Core/Portable/src/bin/prepare-resources.rs new file mode 100644 index 0000000..e7948ea --- /dev/null +++ b/Core/Portable/src/bin/prepare-resources.rs @@ -0,0 +1,104 @@ +use inkflow_rime::Runtime; +use std::{error::Error, fs, path::PathBuf}; + +type Result = std::result::Result>; + +fn run() -> Result<()> { + let args: Vec<_> = std::env::args_os().skip(1).collect(); + if args.len() != 2 { + return Err("Usage: prepare-resources SHARED_SOURCE_DIRECTORY NEW_WORK_DIRECTORY".into()); + } + let shared = PathBuf::from(&args[0]).canonicalize()?; + let root = PathBuf::from(&args[1]); + fs::create_dir(&root)?; + let root = root.canonicalize()?; + let cache = root.join("cache"); + let compiler = root.join("compile-user"); + let probe = root.join("probe-user"); + for path in [&cache, &compiler, &probe] { + fs::create_dir(path)?; + } + { + let mut runtime = Runtime::with_cache(&shared, &compiler, &cache)?; + runtime.prepare()?; + } + let mut required = vec![ + "default.yaml".to_owned(), + "inkflow_pinyin.schema.yaml".to_owned(), + "pinyin_simp.table.bin".to_owned(), + "pinyin_simp.prism.bin".to_owned(), + "easy_en.table.bin".to_owned(), + "inkflow_mixed.table.bin".to_owned(), + ]; + for profile in 0..32 { + required.push(format!("inkflow_spelling_{profile}.schema.yaml")); + required.push(format!("inkflow_spelling_{profile}.prism.bin")); + } + for file in &required { + if fs::metadata(cache.join(file))?.len() == 0 { + return Err(format!("Empty compiled resource: {file}").into()); + } + } + { + let runtime = Runtime::with_cache(&shared, &probe, &cache)?; + let mut session = runtime.session("inkflow_pinyin")?; + // Match the existing worker's smoke cases; no production ranking claim. + for (input, target, first_only) in [ + ("xiehouyu", "歇后语", true), + ("suranqijing", "肃然起敬", true), + ("email", "email", false), + ("wofaleemail", "我发了email", false), + ("nihao", "👋", false), + ] { + session.clear()?; + for key in input.bytes() { + session.process_key(i32::from(key), 0)?; + } + let mut found = false; + let mut seen = 0; + while seen < 1000 { + let snapshot = session.snapshot()?; + if let Some(index) = snapshot + .candidates + .iter() + .take(1000 - seen) + .position(|c| c.text == target) + && (!first_only || index == 0) + { + session.select_candidate(&snapshot, index)?; + if session.take_commit()?.as_deref() != Some(target) + || session.take_commit()?.is_some() + { + return Err(format!("Commit probe failed: {input}").into()); + } + found = true; + break; + } + seen += snapshot.candidates.len().max(1); + if first_only || snapshot.last_page || !session.change_page(false)? { + break; + } + } + if !found { + return Err(format!("Candidate probe failed: {input}").into()); + } + } + session.clear()?; + } + fs::remove_dir_all(compiler)?; + fs::remove_dir_all(probe)?; + fs::write( + root.join("complete"), + b"target-native compilation and five isolated worker smoke cases passed\n", + )?; + println!( + "PASS target-native production cache, 32 spelling profiles, Chinese/English/mixed/Emoji selection and one-shot commits" + ); + Ok(()) +} +fn main() { + if let Err(error) = run() { + eprintln!("Resource preparation failed: {error}"); + std::process::exit(1); + } +} diff --git a/Core/Portable/src/ffi.rs b/Core/Portable/src/ffi.rs index 2d80c5c..f21bbd5 100644 --- a/Core/Portable/src/ffi.rs +++ b/Core/Portable/src/ffi.rs @@ -35,7 +35,12 @@ impl Drop for Snapshot { } unsafe extern "C" { - pub fn ifp_initialize(shared: *const c_char, user: *const c_char) -> c_int; + pub fn ifp_initialize( + shared: *const c_char, + user: *const c_char, + cache: *const c_char, + ) -> c_int; + pub fn ifp_prepare() -> c_int; pub fn ifp_finalize() -> c_int; pub fn ifp_deploy(schema: *const c_char) -> c_int; pub fn ifp_session_create(schema: *const c_char, session: *mut usize) -> c_int; diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index 10dc8bd..c38ddce 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -61,6 +61,7 @@ struct RuntimeOwner { // Keep traits' borrowed paths alive through native finalization. _shared: CString, _user: CString, + _cache: CString, } impl Drop for RuntimeOwner { fn drop(&mut self) { @@ -79,18 +80,25 @@ impl Runtime { /// Paths must name prepared shared resources and an isolated writable user directory. /// Initialization and deployment belong off the interactive key-event path. pub fn new(shared: &Path, user: &Path) -> Result { + Self::with_cache(shared, user, &user.join("build")) + } + + /// Use an explicit target-native cache, separate from writable personal data. + pub fn with_cache(shared: &Path, user: &Path, cache: &Path) -> Result { + let cache = path(cache)?; let shared = path(shared)?; let user = path(user)?; let mut active = lock()?; if *active { return Err(Error::AlreadyRunning); } - check(unsafe { ffi::ifp_initialize(shared.as_ptr(), user.as_ptr()) })?; + check(unsafe { ffi::ifp_initialize(shared.as_ptr(), user.as_ptr(), cache.as_ptr()) })?; *active = true; Ok(Self { owner: Arc::new(RuntimeOwner { _shared: shared, _user: user, + _cache: cache, }), }) } @@ -105,6 +113,16 @@ impl Runtime { check(unsafe { ffi::ifp_deploy(schema.as_ptr()) }) } + /// Compile the configured schema list and its dependencies through Rime maintenance. + /// This is preparation work and is rejected while any sessions exist. + pub fn prepare(&mut self) -> Result<()> { + let _lock = lock()?; + if Arc::strong_count(&self.owner) != 1 { + return Err(Error::SessionsActive); + } + check(unsafe { ffi::ifp_prepare() }) + } + pub fn session(&self, schema: &str) -> Result { let schema = string(schema)?; let _lock = lock()?; diff --git a/Core/Portable/tests/runtime.rs b/Core/Portable/tests/runtime.rs index 53018b3..66fecf9 100644 --- a/Core/Portable/tests/runtime.rs +++ b/Core/Portable/tests/runtime.rs @@ -58,6 +58,7 @@ fn desktop_runtime_contract() { )); assert!(matches!(runtime.session("missing"), Err(Error::Native(-2)))); let mut session = runtime.session("probe").unwrap(); + assert_eq!(runtime.prepare(), Err(Error::SessionsActive)); assert_eq!( runtime.deploy(&fixture.0.join("shared/probe.schema.yaml")), Err(Error::SessionsActive) @@ -228,7 +229,30 @@ fn desktop_runtime_contract() { assert_eq!(session.take_commit().unwrap().as_deref(), Some("你好")); drop(session); drop(runtime); + let cache = fixture.0.join("separate-cache"); + let compiler = fixture.0.join("compile-user"); + let serving = fixture.0.join("serving-user"); + for directory in [&cache, &compiler, &serving] { + fs::create_dir(directory).unwrap(); + } + let mut prepared = Runtime::with_cache(&fixture.0.join("shared"), &compiler, &cache).unwrap(); + prepared.prepare().unwrap(); + assert!(cache.join("probe.table.bin").exists()); + assert!(!compiler.join("build/probe.table.bin").exists()); + drop(prepared); + let prepared = Runtime::with_cache(&fixture.0.join("shared"), &serving, &cache).unwrap(); + let mut prepared_session = prepared.session("probe").unwrap(); + for key in b"nihao" { + prepared_session.process_key(i32::from(*key), 0).unwrap(); + } + assert_eq!( + prepared_session.snapshot().unwrap().candidates[0].text, + "你好" + ); + assert!(!serving.join("build/probe.table.bin").exists()); + drop(prepared_session); + drop(prepared); println!( - "PASS lifetime, target deployment, Lua, UTF-8 ownership, commits, snapshot-safe selection, paging, serialized sessions, restart" + "PASS explicit-cache preparation/serving, lifetime, target deployment, Lua, UTF-8 ownership, commits, snapshot-safe selection, paging, serialized sessions, restart" ); } diff --git a/Core/Sources/InkFlowDomain/DictionaryGenerator.swift b/Core/Sources/InkFlowDomain/DictionaryGenerator.swift index 7f4142d..97eac83 100644 --- a/Core/Sources/InkFlowDomain/DictionaryGenerator.swift +++ b/Core/Sources/InkFlowDomain/DictionaryGenerator.swift @@ -42,6 +42,8 @@ private enum DictionaryNative { /// Calls the shared Rust generator in-process, outside interactive input handling. package enum IFDictionaryGenerator { + package static let recipeVersion = Int(ifd_recipe_version()) + package static let maximumSourceBytes = ifd_maximum_source_bytes() package static func catalog() -> [IFDictionarySourceSpec] { // This immutable JSON is compiled into the same verified Rust library. try! JSONDecoder().decode([IFDictionarySourceSpec].self, from: DictionaryNative.copy(ifd_catalog())) diff --git a/Core/Sources/InkFlowDomain/DictionaryModels.swift b/Core/Sources/InkFlowDomain/DictionaryModels.swift index 46081d9..bde4b14 100644 --- a/Core/Sources/InkFlowDomain/DictionaryModels.swift +++ b/Core/Sources/InkFlowDomain/DictionaryModels.swift @@ -129,12 +129,12 @@ package enum IFDictionaryHash { } package enum IFDictionaryCatalog { - package static let recipeVersion = 3 + package static let recipeVersion = IFDictionaryGenerator.recipeVersion package static let dictionaryFilename = "pinyin_simp.dict.yaml" package static let contextIndexFilename = "pinyin_simp.context.bin" package static let legacyFilename = "legacy-pinyin-simp.dict.yaml" package static let correctionsFilename = "chinese-overrides.tsv" - package static let maximumSourceBytes = 128 * 1024 * 1024 + package static let maximumSourceBytes = IFDictionaryGenerator.maximumSourceBytes package static let initialEntryCount = 965_919 package static let sources = IFDictionaryGenerator.catalog() } diff --git a/Core/scripts/dictionary-notices.py b/Core/scripts/dictionary-notices.py new file mode 100644 index 0000000..b707828 --- /dev/null +++ b/Core/scripts/dictionary-notices.py @@ -0,0 +1,38 @@ +#!/usr/bin/env python3 +"""Collect notices for the locked dictionary library and its Rust standard library.""" +import json +from pathlib import Path +import shutil +import subprocess +import sys + +root = Path(__file__).resolve().parents[2] +output = Path(sys.argv[1]).resolve() +metadata = json.loads(subprocess.check_output([ + 'cargo', 'metadata', '--locked', '--offline', '--format-version', '1', + '--manifest-path', str(root / 'Core/Portable/dictionary/Cargo.toml'), +], cwd=root)) +sections = ['Rust dictionary dependencies\n\nIncludes build-time dependencies from the locked Cargo graph.\n'] +for package in sorted(metadata['packages'], key=lambda p: p['name']): + if package['name'] == 'inkflow-dictionary': + continue + directory = Path(package['manifest_path']).parent + notices = sorted(p for p in directory.iterdir() if p.is_file() and + p.name.lower().startswith(('license', 'copying', 'copyright', 'notice'))) + if not notices: + raise SystemExit(f"Missing license files: {package['name']}") + sections.append(f"\n{'=' * 72}\n{package['name']} {package['version']}\n" + f"License expression: {package['license']}\n" + f"Source: {package.get('repository') or package['source']}\n") + for notice in notices: + sections.append(f'\n--- {notice.name} ---\n{notice.read_text()}\n') +sysroot = Path(subprocess.check_output(['rustc', '--print', 'sysroot'], cwd=root, text=True).strip()) +standard = sysroot / 'share/doc/rust' +if not (standard / 'COPYRIGHT-library.html').is_file() or not (standard / 'licenses').is_dir(): + raise SystemExit('Rust standard-library copyright and license files are required for packaging') +output.mkdir(parents=True, exist_ok=False) +(output / 'dictionary-crates.txt').write_text(''.join(sections)) +(output / 'toolchain.txt').write_text(subprocess.check_output(['rustc', '--version'], cwd=root, text=True)) +shutil.copyfile(standard / 'COPYRIGHT-library.html', output / 'COPYRIGHT-library.html') +shutil.copytree(standard / 'licenses', output / 'licenses') +print(f'Collected dictionary and Rust standard-library notices in {output}') diff --git a/Core/scripts/prepare-rime.sh b/Core/scripts/prepare-rime.sh index 1c73631..33e396b 100644 --- a/Core/scripts/prepare-rime.sh +++ b/Core/scripts/prepare-rime.sh @@ -14,7 +14,7 @@ cleanup() { [[ -z "$staging" ]] || rm -rf "$staging" } trap cleanup EXIT -for input_root in Core/Package.swift Core/Sources Core/Tools Core/scripts Core/Portable/dictionary schemas Core/config Core/Data \ +for input_root in rust-toolchain.toml Core/Package.swift Core/Sources Core/Tools Core/scripts Core/Portable/dictionary schemas Core/config Core/Data \ build/deps/rime-pinyin-simp-* build/deps/rime-easy-en-* build/dictionary-sources; do [[ -e "$input_root" ]] || continue find "$input_root" -type f -print diff --git a/Core/scripts/resource-dependencies.sh b/Core/scripts/resource-dependencies.sh new file mode 100644 index 0000000..badb78c --- /dev/null +++ b/Core/scripts/resource-dependencies.sh @@ -0,0 +1,27 @@ +#!/bin/bash +set -euo pipefail +cd "$(dirname "$0")/../.." +mkdir -p build/deps +fetch() { + local name="$1" sha="$2" url="$3" + if [[ ! -f "build/deps/$name" ]]; then + curl --fail --location --proto '=https' --connect-timeout 20 --max-time 180 --retry 2 \ + "$url" -o "build/deps/$name.part" + mv "build/deps/$name.part" "build/deps/$name" + fi + printf '%s %s\n' "$sha" "build/deps/$name" | shasum -a 256 -c - +} +fetch pinyin.tar.gz 46f37114a7929ecc01003a236803c8b1e5198382e6a21f83fae036604a6b08bf https://codeload.github.com/rime/rime-pinyin-simp/tar.gz/0c6861ef7420ee780270ca6d993d18d4101049d0 +fetch english.tar.gz 59226ae1bb6da00d8808a0094439271225ac4f533d30cf9150ac482383895461 https://codeload.github.com/BlindingDark/rime-easy-en/tar.gz/54a4a07289412efc54134092c0d945f895a71ed3 +fetch emoji.txt 09e29b83ad367ea273e9ab438e572a7621649d93b36924ead28852762d2898b1 https://raw.githubusercontent.com/iDvel/rime-ice/fbb516b2786e4d5444383706d13c31c2e4d10c08/opencc/emoji.txt +# Restore source files from verified archives if extraction is missing or changed. +for entry in \ + 'pinyin.tar.gz:rime-pinyin-simp-0c6861ef7420ee780270ca6d993d18d4101049d0/pinyin_simp.dict.yaml' \ + 'english.tar.gz:rime-easy-en-54a4a07289412efc54134092c0d945f895a71ed3/easy_en.dict.yaml'; do + archive=${entry%%:*} + member=${entry#*:} + expected=$(tar -xOf "build/deps/$archive" "$member" | shasum -a 256 | cut -d ' ' -f 1) + if ! printf '%s %s\n' "$expected" "build/deps/$member" | shasum -a 256 -c - >/dev/null 2>&1; then + tar -xzf "build/deps/$archive" -C build/deps + fi +done diff --git a/NOTICE b/NOTICE index 745e7ce..07c1549 100644 --- a/NOTICE +++ b/NOTICE @@ -12,7 +12,9 @@ requirements. Third-party license texts and attribution notices are in macOS/Licenses/; individual source files may contain additional notices. Source provenance and modifications are described in macOS/DEPENDENCIES.md and the associated -data documentation. +data documentation. Packaged builds also include the locked Rust dictionary +crate notices and Rust standard-library notices under Resources/Licenses/Rust/; +Core/scripts/dictionary-notices.py collects them from the verified build inputs. The dictionaries compiled into the app derive from GPL-3.0, LGPL-3.0, Apache-2.0 and CC BY-SA 4.0 data listed in diff --git a/docs/portable-core-migration.md b/docs/portable-core-migration.md index ef2ce73..cc3624e 100644 --- a/docs/portable-core-migration.md +++ b/docs/portable-core-migration.md @@ -21,7 +21,7 @@ The existing code already separates much of the engine from the macOS frontend: - `schemas/`: Rime configuration, Lua modules, and OpenCC resources. - `Core/Tests/` and `Core/Fixtures/QualityBaseline/`: existing regression coverage and behavioral reference material. -Shared Swift code still imports Apple-specific facilities, and its build scripts assume macOS binaries and toolchains. Dictionary data, configuration, and source preparation now live under `Core/`; the [shared preparation recipe](shared-dictionary-preparation.md) still uses the Swift generator through the Mac toolchain. +Shared Swift code still imports Apple-specific facilities, and its build scripts assume macOS binaries and toolchains. Dictionary data, configuration, and source preparation now live under `Core/`; the [shared preparation recipe](shared-dictionary-preparation.md) uses the Rust generator on both desktops, and the Swift update worker calls it in-process. ## Architecture diff --git a/docs/remote-mac-baseline.md b/docs/remote-mac-baseline.md index 4b3c8f9..29ebd41 100644 --- a/docs/remote-mac-baseline.md +++ b/docs/remote-mac-baseline.md @@ -12,6 +12,7 @@ python3 scripts/mac-remote.py test engine controller python3 scripts/mac-remote.py baseline python3 scripts/mac-remote.py portable python3 scripts/mac-remote.py dictionary +python3 scripts/mac-remote.py resources ``` `--revision COMMIT` defaults to `HEAD`. Local uncommitted changes are never sent. The runner prints the resolved commit and evidence directory. Use `--host ALIAS` and `--remote-root PATH` to override the defaults, `tanaris` and `~/Develop/Projects/inkflow-remote`. @@ -26,7 +27,7 @@ Build caches stay under the dedicated checkout's ignored `build/`. Do not share - An unowned or dirty remote checkout is rejected. This includes untracked, non-ignored files. The runner does not reset, clean, or stash someone else's changes. - A directory lock prevents overlapping runner operations on the same checkout. Do not run builds manually in that checkout while the runner owns it. -- The runner allows only build, baseline, the portable runtime probe, the dictionary generator comparison, and a small set of focused test units. It rejects `test all` and arbitrary commands. +- The runner allows only build, baseline, the portable runtime probe, the dictionary generator comparison, target-native resource preparation, and a small set of focused test units. It rejects `test all` and arbitrary commands. - Build runs the ordinary build script and one fast bundle check. No sudo or signing-key transfer is needed. - Each action records exact commands, elapsed time, exit status, host/toolchain details, and stdout/stderr logs. Failed actions retain evidence and stop before later steps. - Results are copied to `build/mac-remote/RUN_ID/` on Linux. Cleanup requires a successful copy and a complete `run.json` matching every request field and the worker exit status, with valid start/completion timestamps. Missing, malformed, or mismatched receipts leave remote evidence in place and return a nonzero status. @@ -44,6 +45,10 @@ The remote report contains `00-portable.log`, `native-build.json`, and the norma `dictionary` runs `Core/Portable/dictionary/test.sh`. It exercises the existing Swift generator, exports its contract, and compares the Rust generator against fresh Swift outputs and recorded fixtures. It also compiles C and Swift consumers of the in-process Rust dictionary ABI. It returns `catalog.json`, `reference.json`, the Swift/Rust summaries `corpus.json` and `rust-corpus.json`, the Swift ABI consumer's `swift-ffi-corpus.json`, plus `00-dictionary.log` and the normal receipt. See [the comparison contract](../Core/Portable/dictionary/README.md). +## Target-native production resources + +`resources` runs `Core/Portable/prepare-resources.sh`. It uses the shared resource recipe and source-built runtime to compile the full schema list and all 32 spelling profiles, then runs the existing worker's Chinese, English, mixed, and Emoji smoke cases in a separate temporary user directory. The runner returns `resources.json` with source/cache hashes, dictionary metadata, and the native build manifest. Compiled caches stay on their target machine. This does not install a frontend or establish full session/ranking parity. + ## Baseline recipe `baseline` runs `Core/scripts/capture-migration-baseline.sh` in a clean checkout. It: diff --git a/docs/shared-dictionary-preparation.md b/docs/shared-dictionary-preparation.md index f4453b8..e3d0ade 100644 --- a/docs/shared-dictionary-preparation.md +++ b/docs/shared-dictionary-preparation.md @@ -6,7 +6,8 @@ Dictionary source data and build-time policy live under `Core/`. The recipe has | --- | --- | | Static English frequency snapshot and technical spellings | `Core/Data/` | | English admission/scaling and exact corrections; Chinese corrections | `Core/config/` | -| Chinese source catalog, merge/calibration rules, and spelling profiles | `Core/Sources/InkFlowDomain/DictionaryModels.swift` and `DictionaryGenerator.swift` | +| Chinese source catalog | `Core/config/chinese-sources.json`, embedded in the Rust library and read by Swift through its C ABI | +| Merge/calibration rules and spelling profiles | `Core/Portable/dictionary/src/`; Swift preparation/update callers delegate through `DictionaryGenerator.swift` | | Downloaded, verified dictionary sources | Ignored `build/deps/` and `build/dictionary-sources/` | | English admission, mixed dictionary derivation, and resource assembly | `Core/scripts/prepare-rime.sh` | | Chinese source verification and generation | `Core/scripts/prepare-chinese.sh` | @@ -17,9 +18,11 @@ The optional wordfreq snapshot tool stays under `macOS/scripts/` and writes to ` ## Commands -On the Mac, after preparing the pinned dependencies: +On Linux or macOS: ```sh +bash Core/scripts/resource-dependencies.sh +bash Core/scripts/prepare-chinese.sh --sources-only bash Core/scripts/prepare-rime.sh build/shared-rime ``` @@ -36,4 +39,6 @@ The shell fixtures use small stand-in generators to isolate admission and delive ## Migration boundary -Production still builds the Swift generator through the Mac toolchain (`Core/scripts/build-dictionary-generator.sh`). The [Rust generator comparison](../Core/Portable/dictionary/README.md) matches it on Linux and macOS but has not replaced it; target-native preparation is part of [#35](https://github.com/nervouna/InkFlow/issues/35). +`Core/scripts/build-dictionary-generator.sh` builds the Rust CLI and static library. The Swift dictionary preparation/update worker calls that library in-process. The former Swift implementation exists only in the dictionary test target for [compatibility comparisons](../Core/Portable/dictionary/README.md). SwiftPM tracks a generated archive-identity header so changes in Rust relink native callers. + +`bash Core/Portable/prepare-resources.sh` prepares source resources, compiles them with the source-built target runtime, and runs the existing worker smoke cases in separate temporary user directories. It retains source/cache hashes in `resources.json`. The cache is not a portable personal-data backup. diff --git a/macOS/DEPENDENCIES.md b/macOS/DEPENDENCIES.md index 221a57e..5570004 100644 --- a/macOS/DEPENDENCIES.md +++ b/macOS/DEPENDENCIES.md @@ -1,6 +1,6 @@ # Runtime dependencies -`macOS/scripts/dependencies.sh` pins source URLs and SHA-256 digests. Archives live only in ignored `build/deps`. Chinese dictionary sources are pinned in `Core/Sources/InkFlowDomain/DictionaryModels.swift`; see [Chinese dictionary generation](DICTIONARIES.md). English admission and mixed-input policy are in [the mixed-input rules](../.agents/skills/inkflow-mixed-input-maintenance/references/rules.md). License texts ship from `Licenses/`. +`macOS/scripts/dependencies.sh` pins native source URLs and SHA-256 digests; `Core/scripts/resource-dependencies.sh` owns the shared resource archives. Archives live only in ignored `build/deps`. Chinese source pins live in `Core/config/chinese-sources.json`; see [Chinese dictionary generation](DICTIONARIES.md). `Core/Portable/dictionary/` builds the Rust generator used by both the preparation CLI and the Swift update worker. Rust is pinned by `rust-toolchain.toml` and crate versions by its `Cargo.lock`. App packaging collects the locked crate and Rust standard-library notices into `Resources/Licenses/Rust/` through `Core/scripts/dictionary-notices.py`. English admission and mixed-input policy are in [the mixed-input rules](../.agents/skills/inkflow-mixed-input-maintenance/references/rules.md). License texts ship from `Licenses/`. - **Sparkle 2.10.0**, pinned in `Package.resolved` (revision `eef1a539a373c1f1a320624b1130fc5de7b2e100`, binary checksum `17e28312b8e18ab7cdbbe09a6fb28cc55a5479ec6c371dbc07cdecd2a14fd959`). Used only by InkFlowApp. Releases add an app-only ZIP for Sparkle beside the Installer DMG; the pinned `generate_appcast` signs it with EdDSA and writes `appcast.xml` (no deltas). The public key is in Info.plist; the private key stays in the local Keychain account `io.damao.inputmethod.inkflow` and never enters source or release artifacts. The feed is the latest GitHub Release's `appcast.xml`. - **librime 1.17.0**, revision `33e78140250125871856cdc5b42ddc6a5fcd3cd4`: official universal macOS binary, linking only system libraries. Plugins load from the sibling `rime-plugins` directory. diff --git a/macOS/DICTIONARIES.md b/macOS/DICTIONARIES.md index 14214c8..e71dd08 100644 --- a/macOS/DICTIONARIES.md +++ b/macOS/DICTIONARIES.md @@ -14,10 +14,10 @@ count and update actions; provenance and generation details live here. | InkFlow additions | Curated technology and Internet terms plus explicit corrections | [`Core/config/chinese-overrides.tsv`](../Core/config/chinese-overrides.tsv) | | Technical English | Admitted technology terms and abbreviations | [`Core/Data/TECHNOLOGY.md`](../Core/Data/TECHNOLOGY.md) | -`DictionaryModels.swift` is the source of truth for source commits, file paths, byte +`Core/config/chinese-sources.json` is the source of truth for source commits, file paths, byte sizes, Git blob IDs and SHA-256 values. `prepare-chinese.sh` downloads only these explicit -text files into ignored `build/dictionary-sources`. It runs the same -`IFDictionaryGenerator.generate` used by the dictionary preparation worker. +text files into ignored `build/dictionary-sources`. The CLI and the worker's +`IFDictionaryGenerator.generate` wrapper call the same Rust implementation. The baseline merge order is Frost `8105`, `base`, `ext`, `idiom`, then Ice `base`, `ext`, then the fixed legacy `pinyin_simp` source. Specialty sources then fill @@ -167,7 +167,7 @@ running and reopening resumes its current progress. The pane has no source switches. It reports whether the combined dictionary is ready, its Chinese term/reading count and one context-sensitive action: check, update, retry or restore. Busy states expose no duplicate action. Internal content versions, source -commits and immutable URLs remain in this document and `DictionaryModels.swift`, not +commits and immutable URLs remain in this document and the shared source catalog, not in Settings. Recoverable failures state that current input remains available; engine failure instead offers restoration. Both retain selectable technical detail behind a collapsed disclosure. diff --git a/macOS/Quality/ranking-sources.txt b/macOS/Quality/ranking-sources.txt index 4a87d3f..44ce907 100644 --- a/macOS/Quality/ranking-sources.txt +++ b/macOS/Quality/ranking-sources.txt @@ -3,6 +3,9 @@ macOS/Sources/CandidatePresentation.swift macOS/Sources/Context.swift macOS/Sources/CustomPhrases.swift Core/Sources/InkFlowDomain/DictionaryGenerator.swift +Core/Portable/dictionary/src/lib.rs +Core/Portable/dictionary/src/model.rs +Core/Portable/dictionary/src/spelling.rs Core/Sources/InkFlowRimeNative/InkFlowRimeNative.cpp Core/Sources/InkFlowRime/Engine.swift Core/Sources/InkFlowRime/EngineAI.swift diff --git a/macOS/scripts/build.sh b/macOS/scripts/build.sh index 2b26d93..5110b53 100755 --- a/macOS/scripts/build.sh +++ b/macOS/scripts/build.sh @@ -38,6 +38,7 @@ cp build/deps/dist/lib/librime.1.17.0.dylib "$app/Contents/Frameworks/librime.1. cp build/deps/dist/lib/rime-plugins/librime-lua.dylib "$app/Contents/Frameworks/rime-plugins/librime-lua.dylib" bash macOS/scripts/prepare-rime.sh "$app/Contents/Resources/Rime" cp macOS/Licenses/* "$app/Contents/Resources/Licenses/" +python3 Core/scripts/dictionary-notices.py "$app/Contents/Resources/Licenses/Rust" build_swift_product InkFlow "$app/Contents/MacOS/InkFlow" debug sparkle_framework_source="$swiftpm_scratch/artifacts/sparkle/Sparkle/Sparkle.xcframework/macos-arm64_x86_64/Sparkle.framework" [[ -d "$sparkle_framework_source" && -s "$sparkle_framework_source/Versions/B/Sparkle" ]] || { diff --git a/macOS/scripts/check-bundle.sh b/macOS/scripts/check-bundle.sh index 5c6c888..f105039 100755 --- a/macOS/scripts/check-bundle.sh +++ b/macOS/scripts/check-bundle.sh @@ -37,6 +37,10 @@ for link in Autoupdate Headers Modules PrivateHeaders Resources Sparkle Updater. done [[ "$(readlink "$sparkle_framework/Versions/Current")" == B ]] [[ "$(readlink "$sparkle_framework/Sparkle")" == Versions/Current/Sparkle ]] +[[ -s "$app/Contents/Resources/Licenses/Rust/dictionary-crates.txt" ]] +[[ -s "$app/Contents/Resources/Licenses/Rust/COPYRIGHT-library.html" ]] +[[ -s "$app/Contents/Resources/Licenses/Rust/licenses/MIT.txt" ]] +[[ -s "$app/Contents/Resources/Licenses/Rust/licenses/Apache-2.0.txt" ]] [[ -s "$app/Contents/Resources/Licenses/sparkle.txt" ]] cmp macOS/Licenses/sparkle.txt "$app/Contents/Resources/Licenses/sparkle.txt" sparkle_dependency='@rpath/Sparkle.framework/Versions/B/Sparkle' diff --git a/macOS/scripts/dependencies.sh b/macOS/scripts/dependencies.sh index 9599d57..ec4a7e1 100755 --- a/macOS/scripts/dependencies.sh +++ b/macOS/scripts/dependencies.sh @@ -2,6 +2,7 @@ set -euo pipefail cd "$(dirname "$0")/../.." mkdir -p build/deps +bash Core/scripts/resource-dependencies.sh fetch() { local name="$1" sha="$2" url="$3" if [[ ! -f "build/deps/$name" ]]; then @@ -11,9 +12,6 @@ fetch() { echo "$sha build/deps/$name" | shasum -a 256 -c - } fetch librime.tar.bz2 11d8dc663c6ec06d5ccb6111ba664a9e7b631b703ac6acd07cffbac664021850 https://github.com/rime/librime/releases/download/1.17.0/rime-33e7814-macOS-universal.tar.bz2 -fetch pinyin.tar.gz 46f37114a7929ecc01003a236803c8b1e5198382e6a21f83fae036604a6b08bf https://codeload.github.com/rime/rime-pinyin-simp/tar.gz/0c6861ef7420ee780270ca6d993d18d4101049d0 -fetch english.tar.gz 59226ae1bb6da00d8808a0094439271225ac4f533d30cf9150ac482383895461 https://codeload.github.com/BlindingDark/rime-easy-en/tar.gz/54a4a07289412efc54134092c0d945f895a71ed3 -fetch emoji.txt 09e29b83ad367ea273e9ab438e572a7621649d93b36924ead28852762d2898b1 https://raw.githubusercontent.com/iDvel/rime-ice/fbb516b2786e4d5444383706d13c31c2e4d10c08/opencc/emoji.txt # The native translator links the existing runtime. Its C++ declarations and # template types must come from the same release, not host Homebrew headers. fetch librime-source.tar.gz d3f48c2c58f718402229031d8d95fde9cac07ababa8fecf7d18b91946f27fee6 https://codeload.github.com/rime/librime/tar.gz/33e78140250125871856cdc5b42ddc6a5fcd3cd4 @@ -22,8 +20,6 @@ fetch boost_1_89_0.tar.bz2 85a33fa22621b4f314f8e85e1a5e2a9363d22e4f4992925d4bb3b stamp=build/deps/.extracted.sha256 fingerprint=$(printf '%s\n' \ 'librime 11d8dc663c6ec06d5ccb6111ba664a9e7b631b703ac6acd07cffbac664021850' \ - 'pinyin 46f37114a7929ecc01003a236803c8b1e5198382e6a21f83fae036604a6b08bf' \ - 'english 59226ae1bb6da00d8808a0094439271225ac4f533d30cf9150ac482383895461' \ 'librime-source d3f48c2c58f718402229031d8d95fde9cac07ababa8fecf7d18b91946f27fee6' \ 'librime-native-deps dfe6047e87be271963d7466bd1a6e3d9e660c30e5e73e4bb94e8782c0a6ac8df' \ 'boost 85a33fa22621b4f314f8e85e1a5e2a9363d22e4f4992925d4bb3bc631b5a0c7a' | shasum -a 256 | awk '{print $1}') @@ -40,8 +36,6 @@ if [[ "$outputs_valid" == true && -f "$stamp" && "$(cat "$stamp")" == "$fingerpr exit 0 fi tar -xjf build/deps/librime.tar.bz2 -C build/deps -tar -xzf build/deps/pinyin.tar.gz -C build/deps -tar -xzf build/deps/english.tar.gz -C build/deps mkdir -p "$native/librime" "$native/deps" "$native/boost" "$native/generated/rime" tar -xzf build/deps/librime-source.tar.gz -C "$native/librime" --strip-components=1 tar -xjf build/deps/librime-native-deps.tar.bz2 -C "$native/deps" include diff --git a/scripts/mac-remote.py b/scripts/mac-remote.py index 33ea6b9..145f7fe 100644 --- a/scripts/mac-remote.py +++ b/scripts/mac-remote.py @@ -49,6 +49,8 @@ def commands(action, units, report): return [["bash", "Core/scripts/capture-migration-baseline.sh", str(report)]] if action == "portable": return [["bash", "Core/Portable/test.sh", str(report)]] + if action == "resources": + return [["bash", "Core/Portable/prepare-resources.sh", str(report)]] if action == "dictionary": return [["bash", "Core/Portable/dictionary/test.sh", str(report)]] raise ValueError(f"Unsupported action: {action}") @@ -225,7 +227,7 @@ def main(): if len(sys.argv) == 3 and sys.argv[1] == "--worker": return worker(Path(sys.argv[2])) parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("action", choices=["build", "test", "baseline", "portable", "dictionary"]) + parser.add_argument("action", choices=["build", "test", "baseline", "portable", "dictionary", "resources"]) parser.add_argument("units", nargs="*") parser.add_argument("--host", default="tanaris") parser.add_argument("--remote-root", default="~/Develop/Projects/inkflow-remote") diff --git a/scripts/tests/test_mac_remote.py b/scripts/tests/test_mac_remote.py index 31ba495..e0e5e3a 100644 --- a/scripts/tests/test_mac_remote.py +++ b/scripts/tests/test_mac_remote.py @@ -100,11 +100,13 @@ def test_action_allowlist(self): [["bash", "macOS/scripts/test.sh", "preparation", "dictionary-generator", "quality-metadata"]]) self.assertEqual(remote.commands("portable", [], self.root), [["bash", "Core/Portable/test.sh", str(self.root)]]) + self.assertEqual(remote.commands("resources", [], self.root), + [["bash", "Core/Portable/prepare-resources.sh", str(self.root)]]) self.assertEqual(remote.commands("dictionary", [], self.root), [["bash", "Core/Portable/dictionary/test.sh", str(self.root)]]) for action, units in [("install", []), ("test", ["all"]), ("test", []), ("test", ["engine; touch bad"]), ("build", ["engine"]), - ("portable", ["engine"]), ("dictionary", ["engine"])]: + ("portable", ["engine"]), ("dictionary", ["engine"]), ("resources", ["engine"])]: with self.assertRaises(ValueError): remote.commands(action, units, self.root) From 3d793c75816e12c16f73712a59e6f09bec25b7ce Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 08:40:39 +0800 Subject: [PATCH 06/21] Port equivalent-span ranking and capture the Swift reference --- Core/Portable/Cargo.lock | 121 + Core/Portable/Cargo.toml | 7 + Core/Portable/fixtures/ranking-cases.json | 4292 ++++++++++++++++++++ Core/Portable/ranking-reference.sh | 9 + Core/Portable/src/lib.rs | 1 + Core/Portable/src/ranking.rs | 295 ++ Core/Portable/tests/RankingReference.swift | 34 + scripts/mac-remote.py | 4 +- scripts/tests/test_mac_remote.py | 4 +- 9 files changed, 4765 insertions(+), 2 deletions(-) create mode 100644 Core/Portable/fixtures/ranking-cases.json create mode 100644 Core/Portable/ranking-reference.sh create mode 100644 Core/Portable/src/ranking.rs create mode 100644 Core/Portable/tests/RankingReference.swift diff --git a/Core/Portable/Cargo.lock b/Core/Portable/Cargo.lock index 2641c0f..82d6ba9 100644 --- a/Core/Portable/Cargo.lock +++ b/Core/Portable/Cargo.lock @@ -5,3 +5,124 @@ version = 4 [[package]] name = "inkflow-rime" version = "0.1.0" +dependencies = [ + "serde_json", + "unicode-normalization", + "unicode-segmentation", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "proc-macro2" +version = "1.0.107" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9" +dependencies = [ + "unicode-ident", +] + +[[package]] +name = "quote" +version = "1.0.47" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001" +dependencies = [ + "proc-macro2", +] + +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "syn" +version = "3.0.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee" +dependencies = [ + "proc-macro2", + "quote", + "unicode-ident", +] + +[[package]] +name = "tinyvec" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fd3ca314f692efd6c868f8408f53fe444634a845f96c028b97d35f6a1f79f0ee" + +[[package]] +name = "unicode-ident" +version = "1.0.26" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954" + +[[package]] +name = "unicode-normalization" +version = "0.1.25" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8" +dependencies = [ + "tinyvec", +] + +[[package]] +name = "unicode-segmentation" +version = "1.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Core/Portable/Cargo.toml b/Core/Portable/Cargo.toml index 035f52e..2f2abca 100644 --- a/Core/Portable/Cargo.toml +++ b/Core/Portable/Cargo.toml @@ -7,3 +7,10 @@ build = "build.rs" [lib] path = "src/lib.rs" + +[dependencies] +unicode-normalization = "=0.1.25" +unicode-segmentation = "=1.13.3" + +[dev-dependencies] +serde_json = "=1.0.151" diff --git a/Core/Portable/fixtures/ranking-cases.json b/Core/Portable/fixtures/ranking-cases.json new file mode 100644 index 0000000..98098ed --- /dev/null +++ b/Core/Portable/fixtures/ranking-cases.json @@ -0,0 +1,4292 @@ +[ + { + "name": "long-crossing", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "餐", + "参" + ], + "context": "准备午", + "metadata": "0,3;0,3,n,1,0,n;0,3,n,1,0,n;0,3,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "frequency-and-stability", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "餐", + "参" + ], + "context": "午", + "metadata": "0,3;0,3,n,1,0,n;0,3,n,1,0,n;0,3,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "partial-slot", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "餐", + "参" + ], + "context": "午", + "metadata": "0,3;0,3,n,1,0,n;0,1,n,1,0,n;0,3,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "unequal-length", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "你好", + "你" + ], + "context": "迷", + "metadata": "0,2;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-日常", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "日常", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-正在使用 Swift ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "正在使用 Swift ", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-代码", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "代码", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-代码 ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "代码 ", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-Swift 日日日日日日日日日日日日日日日日", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "Swift 日日日日日日日日日日日日日日日日", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-Św", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "Św", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-👩‍💻版本", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "👩‍💻版本", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "context-準備午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "从哦的新", + "Codex", + "Codex CLI", + "才" + ], + "context": "準備午", + "metadata": "0,4;0,5,n,1,0,n;0,5,a,1,1,e;0,5,a,0,0,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-n-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-a-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-m-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-n-o-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,n;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-n-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-a-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-m-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-e-o-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,e;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-n-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-a-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-m-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-m-o-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,m;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-n-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,n,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-a-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,a,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-m-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,m,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-0-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,0,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-0-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-0-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,0,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-1-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-1-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,1,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-2-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-2-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,2,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-3-午", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "午", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "matrix-c-o-1-3-CLI ", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex", + "餐", + "我用Codex", + "!" + ], + "context": "CLI ", + "metadata": "0,5;0,5,n,1,0,n;0,5,o,1,3,c;0,5,n,1,0,n;0,5,m,1,2,m;0,5,o,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-268", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": null, + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-269", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-270", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-271", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "00,2;0,5,n,1,0,n;0,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-272", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "9,2;0,5,n,1,0,n;0,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-273", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;0,5,a,1,1,e;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-274", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;-1,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-275", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;+0,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-276", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;0,0,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-277", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;5,2,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-278", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;0,9,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-279", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;0,999999999999999999999999,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-280", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n;0,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "bad-metadata-281", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "Codex" + ], + "context": "Swift ", + "metadata": "0,2;0,5,n,1,0,n; 0,5,a,1,1,e", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-282", + "dictionary": "午餐\twu can\t100\r\n午参\twu can\t100\r\n午惨\twu can\t50\r\n准备午参\tzhun bei wu can\t1\r\n迷你\tmi ni\t70\r\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-283", + "dictionary": "午餐\twu can\t100\r\n午参\twu can\t100\r\n午惨\twu can\t50\r\n准备午参\tzhun bei wu can\t1\r\n迷你\tmi ni\t70\r\n\n午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-284", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n\n午餐\twu can\t200\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-285", + "dictionary": "午餐\t\t100\n午参\twu can\t+200\n午惨\twu can\t-0\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-286", + "dictionary": "午餐\twu can\t9223372036854775808\n午参\twu can\t-1\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-287", + "dictionary": "神龍\tshen long\t100\n神龍\tshen long\t200\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "dictionary-288", + "dictionary": "午餐️\twu can\t1000\n午参\twu can\t1\n", + "candidates": [ + "惨", + "餐", + "参", + "龍" + ], + "context": "午", + "metadata": "0,4;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "canonical-han", + "dictionary": "神龍\tshen long\t100\n", + "candidates": [ + "龙", + "龍" + ], + "context": "神", + "metadata": "0,2;0,5,n,1,0,n;0,5,n,1,0,n", + "offset": 0, + "inputLength": 8 + }, + { + "name": "empty", + "dictionary": "午餐\twu can\t100\n午参\twu can\t100\n午惨\twu can\t50\n准备午参\tzhun bei wu can\t1\n迷你\tmi ni\t70\n", + "candidates": [], + "context": "午", + "metadata": null, + "offset": 0, + "inputLength": 8 + } +] diff --git a/Core/Portable/ranking-reference.sh b/Core/Portable/ranking-reference.sh new file mode 100644 index 0000000..e7839dd --- /dev/null +++ b/Core/Portable/ranking-reference.sh @@ -0,0 +1,9 @@ +#!/bin/bash +set -euo pipefail +cd "$(dirname "$0")/../.." +mkdir -p build/portable +xcrun swiftc -package-name InkFlow \ + Core/Sources/InkFlowDomain/CandidateRanking.swift \ + Core/Portable/tests/RankingReference.swift \ + -o build/portable/ranking-reference +build/portable/ranking-reference Core/Portable/fixtures/ranking-cases.json "$1/ranking-reference.json" diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index c38ddce..2ccbe33 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -1,5 +1,6 @@ //! Minimal synchronous desktop Rime host. No frontend or InkFlow ranking policy. mod ffi; +pub mod ranking; use std::{ ffi::{CStr, CString, c_char}, diff --git a/Core/Portable/src/ranking.rs b/Core/Portable/src/ranking.rs new file mode 100644 index 0000000..0affbce --- /dev/null +++ b/Core/Portable/src/ranking.rs @@ -0,0 +1,295 @@ +//! Equivalent-span ordering, matching InkFlowDomain/CandidateRanking.swift. +use std::{collections::HashMap, ops::Range}; +use unicode_normalization::UnicodeNormalization; +use unicode_segmentation::UnicodeSegmentation; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CandidateClass { + NonAscii, + Ascii, + Mixed, + Other, +} +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Source { + Native, + English, + Mixed, + Custom, +} +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Metadata { + pub coverage: Range, + pub class: CandidateClass, + pub exact: bool, + pub personal: u8, + pub source: Source, +} + +pub fn parse_metadata( + value: &str, + offset: usize, + count: usize, + input_length: usize, +) -> Option> { + if !(1..=9).contains(&count) || input_length == 0 { + return None; + } + let mut rows = value.split(';'); + if rows.next()? != format!("{offset},{count}") { + return None; + } + let mut result = Vec::with_capacity(count); + for row in rows { + if result.len() == count { + return None; + } + let fields: Vec<_> = row.split(',').collect(); + if fields.len() != 6 + || fields[..2] + .iter() + .any(|v| v.is_empty() || !v.bytes().all(|b| b.is_ascii_digit())) + { + return None; + } + let start = fields[0].parse::().ok()?; + let end = fields[1].parse::().ok()?; + if start >= end || end as usize > input_length { + return None; + } + let class = match fields[2] { + "n" => CandidateClass::NonAscii, + "a" => CandidateClass::Ascii, + "m" => CandidateClass::Mixed, + "o" => CandidateClass::Other, + _ => return None, + }; + let exact = match fields[3] { + "1" => true, + "0" => false, + _ => return None, + }; + let personal = match fields[4] { + "0" => 0, + "1" => 1, + "2" => 2, + "3" => 3, + _ => return None, + }; + let source = match fields[5] { + "n" => Source::Native, + "e" => Source::English, + "m" => Source::Mixed, + "c" => Source::Custom, + _ => return None, + }; + if personal > 0 && !(exact && matches!(source, Source::English | Source::Mixed)) { + return None; + } + result.push(Metadata { + coverage: start as usize..end as usize, + class, + exact, + personal, + source, + }); + } + (result.len() == count).then_some(result) +} + +fn is_han(grapheme: &str) -> bool { + let mut chars = grapheme.chars(); + matches!( + chars.next(), + Some( + '\u{3400}'..='\u{4dbf}' + | '\u{4e00}'..='\u{9fff}' + | '\u{f900}'..='\u{faff}' + | '\u{20000}'..='\u{2fa1f}' + | '\u{30000}'..='\u{323af}', + ) + ) && chars.next().is_none() +} +fn all_han(text: &str) -> bool { + text.graphemes(true).all(is_han) +} +fn normalized(text: &str) -> String { + text.nfc().collect() +} +fn consistent(candidate: &str, row: &Metadata) -> bool { + let letter = candidate.bytes().any(|b| b.is_ascii_alphabetic()); + let non_ascii = !candidate.is_ascii(); + let class = match (letter, non_ascii) { + (true, false) => CandidateClass::Ascii, + (true, true) => CandidateClass::Mixed, + (false, true) => CandidateClass::NonAscii, + (false, false) => CandidateClass::Other, + }; + row.class == class + && match row.source { + Source::English => class == CandidateClass::Ascii, + Source::Mixed => class == CandidateClass::Mixed, + Source::Native | Source::Custom => row.personal == 0, + } +} +fn technical(text: &str) -> bool { + let bounded: String = text + .graphemes(true) + .rev() + .take(16) + .collect::>() + .into_iter() + .rev() + .collect(); + let mut run = 0; + for scalar in bounded.chars() { + run = if scalar.is_ascii_alphanumeric() { + run + 1 + } else { + 0 + }; + if run >= 2 { + return true; + } + } + let canonical = normalized(&bounded); + ["代码", "编程", "开发", "命令", "终端", "接口", "版本"] + .iter() + .any(|suffix| canonical.ends_with(suffix)) +} + +pub struct ContextRanker { + frequencies: HashMap, + longest: usize, +} +impl ContextRanker { + /// Build before accepting key events. Ranking itself does not access storage. + pub fn from_dictionary(text: &str) -> Self { + let mut frequencies = HashMap::::new(); + let mut longest = 0; + let mut start = 0; + // Swift splits LF Characters, not the LF inside a CRLF grapheme. + for (end, grapheme) in text + .grapheme_indices(true) + .chain(std::iter::once((text.len(), "\n"))) + { + if grapheme != "\n" { + continue; + } + let line = &text[start..end]; + start = end + 1; + let fields: Vec<_> = line.split('\t').filter(|field| !field.is_empty()).collect(); + if fields.len() < 3 { + continue; + } + let Ok(frequency) = fields[2].parse::() else { + continue; + }; + let phrase = fields[0]; + let count = phrase.graphemes(true).count(); + if frequency < 0 || !(2..=8).contains(&count) || !all_han(phrase) { + continue; + } + let entry = frequencies.entry(normalized(phrase)).or_default(); + *entry = (*entry).max(frequency); + longest = longest.max(count); + } + Self { + frequencies, + longest, + } + } + + pub fn order( + &self, + candidates: &[String], + preceding: &str, + metadata: Option<&[Metadata]>, + ) -> Vec { + let original: Vec<_> = (0..candidates.len()).collect(); + let Some(rows) = metadata else { + return original; + }; + if candidates.is_empty() + || rows.len() != candidates.len() + || rows.iter().any(|r| r.coverage.is_empty() || r.personal > 3) + || candidates.iter().zip(rows).any(|(c, r)| !consistent(c, r)) + { + return original; + } + let eligible: Vec<_> = original + .iter() + .copied() + .filter(|&i| rows[i].coverage == rows[0].coverage) + .collect(); + let prefix: Vec<_> = preceding + .graphemes(true) + .rev() + .take(self.longest.saturating_sub(1)) + .take_while(|g| is_han(g)) + .collect(); + let first_length = candidates[0].graphemes(true).count(); + let scores: Vec<_> = candidates + .iter() + .enumerate() + .map(|(i, candidate)| { + if !eligible.contains(&i) + || prefix.is_empty() + || candidate.graphemes(true).count() != first_length + || !all_han(candidate) + { + return (0, 0); + } + for count in (1..=prefix.len()).rev() { + let phrase: String = prefix[..count] + .iter() + .rev() + .copied() + .chain(std::iter::once(candidate.as_str())) + .collect(); + if let Some(&frequency) = self.frequencies.get(&normalized(&phrase)) { + return (count, frequency); + } + } + (0, 0) + }) + .collect(); + let leading_han = eligible + .iter() + .copied() + .filter(|&i| all_han(&candidates[i])) + .min_by_key(|&i| (std::cmp::Reverse(scores[i]), i)); + let is_technical = technical(preceding); + let tier = |i: usize| { + let row = &rows[i]; + if row.source == Source::Custom { + 0 + } else if is_technical && row.exact && row.personal > 0 && row.source == Source::English + { + 1 + } else if Some(i) == leading_han { + 2 + } else if row.exact && matches!(row.source, Source::English | Source::Mixed) { + 3 + } else if row.exact { + 4 + } else { + 5 + } + }; + let mut ranked = eligible.clone(); + ranked.sort_by_key(|&i| { + ( + tier(i), + std::cmp::Reverse(rows[i].personal), + std::cmp::Reverse(scores[i]), + i, + ) + }); + let mut result = original; + for (slot, candidate) in eligible.into_iter().zip(ranked) { + result[slot] = candidate; + } + result + } +} diff --git a/Core/Portable/tests/RankingReference.swift b/Core/Portable/tests/RankingReference.swift new file mode 100644 index 0000000..e3d7462 --- /dev/null +++ b/Core/Portable/tests/RankingReference.swift @@ -0,0 +1,34 @@ +import Foundation + +@main +struct RankingReference { + static func main() throws { + let args = CommandLine.arguments + precondition(args.count == 3) + let cases = try JSONSerialization.jsonObject(with: Data(contentsOf: URL(fileURLWithPath: args[1]))) as! [[String: Any]] + let directory = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) + try FileManager.default.createDirectory(at: directory, withIntermediateDirectories: true) + defer { try? FileManager.default.removeItem(at: directory) } + let file = directory.appendingPathComponent("dictionary.yaml") + var results: [String: Any] = [:] + for item in cases { + try (item["dictionary"] as! String).write(to: file, atomically: true, encoding: .utf8) + let ranker = try IFContextRanker(dictionary: file.path) + let candidates = item["candidates"] as! [String] + let metadata = (item["metadata"] as? String).flatMap { + IFContextRanker.parseMetadata($0, offset: item["offset"] as! Int, + count: candidates.count, inputLength: item["inputLength"] as! Int) + } + let parsed = metadata.map { rows in rows.map { row in + "\(row.coverage.lowerBound),\(row.coverage.upperBound),\(row.candidateClass.rawValue),\(row.exact ? 1 : 0),\(row.personalBucket),\(row.source.rawValue)" + }} + results[item["name"] as! String] = [ + "parsed": parsed as Any? ?? NSNull(), + "order": ranker.order(candidates, precedingText: item["context"] as! String, metadata: metadata) + ] + } + let output = try JSONSerialization.data(withJSONObject: results, options: [.prettyPrinted, .sortedKeys]) + try output.write(to: URL(fileURLWithPath: args[2])) + print("PASS Swift ranking reference: \(cases.count) cases") + } +} diff --git a/scripts/mac-remote.py b/scripts/mac-remote.py index 145f7fe..cb62059 100644 --- a/scripts/mac-remote.py +++ b/scripts/mac-remote.py @@ -49,6 +49,8 @@ def commands(action, units, report): return [["bash", "Core/scripts/capture-migration-baseline.sh", str(report)]] if action == "portable": return [["bash", "Core/Portable/test.sh", str(report)]] + if action == "ranking-reference": + return [["bash", "Core/Portable/ranking-reference.sh", str(report)]] if action == "resources": return [["bash", "Core/Portable/prepare-resources.sh", str(report)]] if action == "dictionary": @@ -227,7 +229,7 @@ def main(): if len(sys.argv) == 3 and sys.argv[1] == "--worker": return worker(Path(sys.argv[2])) parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("action", choices=["build", "test", "baseline", "portable", "dictionary", "resources"]) + parser.add_argument("action", choices=["build", "test", "baseline", "portable", "dictionary", "resources", "ranking-reference"]) parser.add_argument("units", nargs="*") parser.add_argument("--host", default="tanaris") parser.add_argument("--remote-root", default="~/Develop/Projects/inkflow-remote") diff --git a/scripts/tests/test_mac_remote.py b/scripts/tests/test_mac_remote.py index e0e5e3a..2971bab 100644 --- a/scripts/tests/test_mac_remote.py +++ b/scripts/tests/test_mac_remote.py @@ -100,13 +100,15 @@ def test_action_allowlist(self): [["bash", "macOS/scripts/test.sh", "preparation", "dictionary-generator", "quality-metadata"]]) self.assertEqual(remote.commands("portable", [], self.root), [["bash", "Core/Portable/test.sh", str(self.root)]]) + self.assertEqual(remote.commands("ranking-reference", [], self.root), + [["bash", "Core/Portable/ranking-reference.sh", str(self.root)]]) self.assertEqual(remote.commands("resources", [], self.root), [["bash", "Core/Portable/prepare-resources.sh", str(self.root)]]) self.assertEqual(remote.commands("dictionary", [], self.root), [["bash", "Core/Portable/dictionary/test.sh", str(self.root)]]) for action, units in [("install", []), ("test", ["all"]), ("test", []), ("test", ["engine; touch bad"]), ("build", ["engine"]), - ("portable", ["engine"]), ("dictionary", ["engine"]), ("resources", ["engine"])]: + ("portable", ["engine"]), ("dictionary", ["engine"]), ("resources", ["engine"]), ("ranking-reference", ["engine"])]: with self.assertRaises(ValueError): remote.commands(action, units, self.root) From a302e2072fc0c5ae817c5373ca89bc01aed35611 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 08:45:19 +0800 Subject: [PATCH 07/21] Verify 291 ranking cases against the unchanged Swift implementation --- Core/Portable/fixtures/ranking-reference.json | 3604 +++++++++++++++++ Core/Portable/test.sh | 11 + Core/Portable/tests/ranking.rs | 62 + 3 files changed, 3677 insertions(+) create mode 100644 Core/Portable/fixtures/ranking-reference.json create mode 100644 Core/Portable/tests/ranking.rs diff --git a/Core/Portable/fixtures/ranking-reference.json b/Core/Portable/fixtures/ranking-reference.json new file mode 100644 index 0000000..121f268 --- /dev/null +++ b/Core/Portable/fixtures/ranking-reference.json @@ -0,0 +1,3604 @@ +{ + "bad-metadata-268" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-269" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-270" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-271" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-272" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-273" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-274" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-275" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-276" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-277" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-278" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-279" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-280" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "bad-metadata-281" : { + "order" : [ + 0, + 1 + ], + "parsed" : null + }, + "canonical-han" : { + "order" : [ + 1, + 0 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "context-👩‍💻版本" : { + "order" : [ + 1, + 0, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-Św" : { + "order" : [ + 0, + 1, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-Swift 日日日日日日日日日日日日日日日日" : { + "order" : [ + 0, + 1, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-代码" : { + "order" : [ + 1, + 0, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-代码 " : { + "order" : [ + 0, + 1, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-日常" : { + "order" : [ + 0, + 1, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-正在使用 Swift " : { + "order" : [ + 1, + 0, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "context-準備午" : { + "order" : [ + 0, + 1, + 3, + 2 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,a,0,0,e", + "0,5,n,1,0,n" + ] + }, + "dictionary-282" : { + "order" : [ + 0, + 1, + 2, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-283" : { + "order" : [ + 1, + 2, + 0, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-284" : { + "order" : [ + 1, + 2, + 0, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-285" : { + "order" : [ + 2, + 0, + 1, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-286" : { + "order" : [ + 0, + 1, + 2, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-287" : { + "order" : [ + 0, + 1, + 2, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "dictionary-288" : { + "order" : [ + 2, + 0, + 1, + 3 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + }, + "empty" : { + "order" : [ + + ], + "parsed" : null + }, + "frequency-and-stability" : { + "order" : [ + 1, + 2, + 0 + ], + "parsed" : [ + "0,3,n,1,0,n", + "0,3,n,1,0,n", + "0,3,n,1,0,n" + ] + }, + "long-crossing" : { + "order" : [ + 2, + 1, + 0 + ], + "parsed" : [ + "0,3,n,1,0,n", + "0,3,n,1,0,n", + "0,3,n,1,0,n" + ] + }, + "matrix-c-a-0-0-CLI " : { + "order" : [ + 1, + 0, + 3, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-a-0-0-午" : { + "order" : [ + 1, + 2, + 3, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-a-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-0-CLI " : { + "order" : [ + 1, + 0, + 3, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-a-1-0-午" : { + "order" : [ + 1, + 2, + 3, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-a-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-a-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-m-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-m-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-m-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-m-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-m-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-n-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-n-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-n-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-n-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-n-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-o-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-o-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-o-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,c", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-c-o-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-c-o-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-0-CLI " : { + "order" : [ + 0, + 3, + 2, + 4, + 1 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-0-0-午" : { + "order" : [ + 2, + 3, + 0, + 4, + 1 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-a-1-0-CLI " : { + "order" : [ + 0, + 3, + 1, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-0-午" : { + "order" : [ + 2, + 3, + 1, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-1-CLI " : { + "order" : [ + 1, + 0, + 3, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-1-午" : { + "order" : [ + 2, + 3, + 1, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-2-CLI " : { + "order" : [ + 1, + 0, + 3, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-2-午" : { + "order" : [ + 2, + 1, + 3, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-3-CLI " : { + "order" : [ + 1, + 0, + 3, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-a-1-3-午" : { + "order" : [ + 2, + 1, + 3, + 0, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-m-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-m-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-n-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-n-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-e-o-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,1,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,2,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-e-o-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,3,e", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-a-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-a-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-m-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-m-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-n-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-n-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-m-o-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,1,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,2,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-m-o-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,3,m", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-a-0-0-CLI " : { + "order" : [ + 0, + 3, + 2, + 4, + 1 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-a-0-0-午" : { + "order" : [ + 2, + 3, + 0, + 4, + 1 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-a-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-0-CLI " : { + "order" : [ + 0, + 3, + 1, + 2, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-a-1-0-午" : { + "order" : [ + 2, + 3, + 0, + 1, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,a,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-a-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-a-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-m-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-m-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-m-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,m,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-m-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-m-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-n-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-n-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-n-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-n-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-n-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-o-0-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,0,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-o-0-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-0-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-0-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-o-1-0-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,o,1,0,n", + "0,5,n,1,0,n", + "0,5,m,1,2,m", + "0,5,o,1,0,n" + ] + }, + "matrix-n-o-1-1-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-1-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-2-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-2-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-3-CLI " : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "matrix-n-o-1-3-午" : { + "order" : [ + 0, + 1, + 2, + 3, + 4 + ], + "parsed" : null + }, + "partial-slot" : { + "order" : [ + 2, + 1, + 0 + ], + "parsed" : [ + "0,3,n,1,0,n", + "0,1,n,1,0,n", + "0,3,n,1,0,n" + ] + }, + "unequal-length" : { + "order" : [ + 0, + 1 + ], + "parsed" : [ + "0,5,n,1,0,n", + "0,5,n,1,0,n" + ] + } +} \ No newline at end of file diff --git a/Core/Portable/test.sh b/Core/Portable/test.sh index 0aed083..1abc414 100644 --- a/Core/Portable/test.sh +++ b/Core/Portable/test.sh @@ -3,7 +3,18 @@ set -euo pipefail cd "$(dirname "$0")/../.." export CARGO_TARGET_DIR="$PWD/build/portable/cargo" python3 Core/Portable/build-native.py +if [[ $(uname -s) == Darwin ]]; then + bash Core/Portable/ranking-reference.sh build/portable + python3 - <<'PY' +import json +from pathlib import Path +assert json.loads(Path('build/portable/ranking-reference.json').read_text()) == json.loads( + Path('Core/Portable/fixtures/ranking-reference.json').read_text()) +print('PASS fresh Swift ranking reference matches recorded cases') +PY +fi cargo test --locked --manifest-path Core/Portable/Cargo.toml -- --nocapture if [[ $# -eq 1 ]]; then cp build/portable/native-build.json "$1/native-build.json" + if [[ $(uname -s) == Darwin ]]; then cp build/portable/ranking-reference.json "$1/ranking-reference.json"; fi fi diff --git a/Core/Portable/tests/ranking.rs b/Core/Portable/tests/ranking.rs new file mode 100644 index 0000000..f8d0a5d --- /dev/null +++ b/Core/Portable/tests/ranking.rs @@ -0,0 +1,62 @@ +use inkflow_rime::ranking::{CandidateClass, ContextRanker, Source, parse_metadata}; +use serde_json::{Value, json}; + +#[test] +fn swift_ranking_reference() { + let cases: Vec = + serde_json::from_str(include_str!("../fixtures/ranking-cases.json")).unwrap(); + let expected: Value = + serde_json::from_str(include_str!("../fixtures/ranking-reference.json")).unwrap(); + assert_eq!(cases.len(), expected.as_object().unwrap().len()); + for case in &cases { + let candidates: Vec = serde_json::from_value(case["candidates"].clone()).unwrap(); + let metadata = case["metadata"].as_str().and_then(|value| { + parse_metadata( + value, + case["offset"].as_u64().unwrap() as usize, + candidates.len(), + case["inputLength"].as_u64().unwrap() as usize, + ) + }); + let parsed = metadata.as_ref().map(|rows| { + rows.iter() + .map(|row| { + let class = match row.class { + CandidateClass::NonAscii => "n", + CandidateClass::Ascii => "a", + CandidateClass::Mixed => "m", + CandidateClass::Other => "o", + }; + let source = match row.source { + Source::Native => "n", + Source::English => "e", + Source::Mixed => "m", + Source::Custom => "c", + }; + format!( + "{},{},{},{},{},{}", + row.coverage.start, + row.coverage.end, + class, + u8::from(row.exact), + row.personal, + source + ) + }) + .collect::>() + }); + let ranker = ContextRanker::from_dictionary(case["dictionary"].as_str().unwrap()); + let order = ranker.order( + &candidates, + case["context"].as_str().unwrap(), + metadata.as_deref(), + ); + assert_eq!( + json!({"parsed": parsed, "order": order}), + expected[case["name"].as_str().unwrap()], + "{}", + case["name"] + ); + } + println!("PASS {} Swift/Rust ranking cases", cases.len()); +} From 50f448a684c20d957e5315e7588d44fd03c848d1 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Mon, 5 Oct 2026 08:49:39 +0800 Subject: [PATCH 08/21] Record ranking parity and add read-only Linux target inspection --- Core/Portable/README.md | 4 ++++ Linux/scripts/preflight.sh | 32 ++++++++++++++++++++++++++++++++ docs/remote-mac-baseline.md | 5 +++++ 3 files changed, 41 insertions(+) create mode 100644 Linux/scripts/preflight.sh diff --git a/Core/Portable/README.md b/Core/Portable/README.md index c463c6d..3f84193 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -27,6 +27,10 @@ OpenCC 1.1.9 requests C++14, but the pinned marisa headers require C++17. The bu The test copies the tiny checked-in source fixture into a fresh temporary directory, compiles it using the target runtime, and removes the directory after runtime teardown. It checks Chinese composition, one-shot commits, a non-ASCII Lua filter, snapshot ownership, serialized sessions across threads, runtime/session lifetime, and restart. +## Ranking comparison + +`src/ranking.rs` ports the existing Swift equivalent-span ranker, including strict metadata parsing, Han phrase context, personal-strength limits, custom-source priority, and native-order fallback. Its 291 synthetic cases compare against the unchanged Swift source. Mac runtime tests regenerate the reference before checking Rust; Linux tests use the recorded results. Dictionary indexing belongs in preparation, while ordering uses only memory. This module is not yet wired into session selection or key handling, so the runtime below still exposes native order. + ## Interface contract `native/bridge.h` is the experimental internal C ABI between Rust and the native runtime. It uses Rime's public C API for lifecycle, deployment, input, context, and commits. C++ internals are confined to the existing extension and its registration check. The Rust library exposes safe `Runtime` and `Session` handles; a stable frontend-facing exported Rust C ABI is deferred until session policy is migrated. diff --git a/Linux/scripts/preflight.sh b/Linux/scripts/preflight.sh new file mode 100644 index 0000000..c7176df --- /dev/null +++ b/Linux/scripts/preflight.sh @@ -0,0 +1,32 @@ +#!/bin/bash +# Read-only target inspection. Run as the desktop user, including over SSH. +set -euo pipefail +printf 'Architecture: '; uname -m +printf 'OS:\n' +awk -F= '$1 ~ /^(ID|VERSION_ID|PRETTY_NAME|BUILD_ID)$/ { print }' /etc/os-release +printf 'C library: '; getconf GNU_LIBC_VERSION || true +printf 'Current shell session: %s / %s\n' "${XDG_SESSION_TYPE:-unknown}" "${XDG_CURRENT_DESKTOP:-unknown}" +if command -v loginctl >/dev/null; then + sessions=$(loginctl show-user "$(id -u)" -p Sessions --value 2>/dev/null || true) + for session in $sessions; do + printf 'User session:\n' + loginctl show-session "$session" -p Type -p Desktop -p Remote -p State || true + done +fi +for process in Hyprland kwin_wayland kwin_x11 fcitx5; do + if pgrep -u "$(id -u)" -x "$process" >/dev/null; then printf 'Running: %s\n' "$process"; fi +done +if command -v fcitx5 >/dev/null; then fcitx5 -v; else printf 'Fcitx5: not on PATH\n'; fi +if command -v findmnt >/dev/null; then + printf 'Root/home mount types and VFS flags:\n' + findmnt -n -o TARGET,FSTYPE,VFS-OPTIONS --target / + findmnt -n -o TARGET,FSTYPE,VFS-OPTIONS --target "$HOME" +fi +if command -v flatpak >/dev/null; then + flatpak --version + printf 'Flatpak platforms:\n' + flatpak list --runtime --columns=application,branch | grep -E '^org\.(gnome|kde|freedesktop)\.Platform[[:space:]]' || true +else + printf 'Flatpak: not on PATH\n' +fi +printf 'No installation, activation, or configuration changes performed.\n' diff --git a/docs/remote-mac-baseline.md b/docs/remote-mac-baseline.md index 29ebd41..665ac9d 100644 --- a/docs/remote-mac-baseline.md +++ b/docs/remote-mac-baseline.md @@ -13,6 +13,7 @@ python3 scripts/mac-remote.py baseline python3 scripts/mac-remote.py portable python3 scripts/mac-remote.py dictionary python3 scripts/mac-remote.py resources +python3 scripts/mac-remote.py ranking-reference ``` `--revision COMMIT` defaults to `HEAD`. Local uncommitted changes are never sent. The runner prints the resolved commit and evidence directory. Use `--host ALIAS` and `--remote-root PATH` to override the defaults, `tanaris` and `~/Develop/Projects/inkflow-remote`. @@ -45,6 +46,10 @@ The remote report contains `00-portable.log`, `native-build.json`, and the norma `dictionary` runs `Core/Portable/dictionary/test.sh`. It exercises the existing Swift generator, exports its contract, and compares the Rust generator against fresh Swift outputs and recorded fixtures. It also compiles C and Swift consumers of the in-process Rust dictionary ABI. It returns `catalog.json`, `reference.json`, the Swift/Rust summaries `corpus.json` and `rust-corpus.json`, the Swift ABI consumer's `swift-ffi-corpus.json`, plus `00-dictionary.log` and the normal receipt. See [the comparison contract](../Core/Portable/dictionary/README.md). +## Ranking reference + +`ranking-reference` compiles the unchanged Swift ranker with a standalone fixture reader and returns `ranking-reference.json`. It does not initialize Rime or inspect user data. The `portable` action also regenerates this reference on the Mac, compares it with the recorded fixture, and runs the Rust ranking tests. These compare the pure ordering policy, not full engine behavior. + ## Target-native production resources `resources` runs `Core/Portable/prepare-resources.sh`. It uses the shared resource recipe and source-built runtime to compile the full schema list and all 32 spelling profiles, then runs the existing worker's Chinese, English, mixed, and Emoji smoke cases in a separate temporary user directory. The runner returns `resources.json` with source/cache hashes, dictionary metadata, and the native build manifest. Compiled caches stay on their target machine. This does not install a frontend or establish full session/ranking parity. From a5c679cc4cf79aa53fd9093bb9e7475cee4ba5c9 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Thu, 8 Oct 2026 23:02:22 +0800 Subject: [PATCH 09/21] Generate the context-ranking index from the Rust dictionary generator Rebasing onto main brought in the prepared context index (#60): the Swift generator's `spelling` step wrote `pinyin_simp.context.bin`, but on this branch that step runs the Rust CLI, so built resources lacked the index and `IFContextRanker` could not start. Port `IFContextRanker.buildIndex` to the dictionary crate and return it from `spelling` beside the 32 schema files, so the build CLI, the Swift update worker and the C ABI share one implementation. The Swift reference export and recorded fixtures now carry the index hash; parity checks compare its bytes against the retained Swift builder. Also fold main's other changes into the moved pieces: drop the Sogou source from the embedded catalog, split `IFDictionaryError` so the standalone ranking reference still compiles, and point the corresponding- source bundle at `Core/config/chinese-sources.json` and `Core/scripts/resource-dependencies.sh`. Refs #35 --- Core/Portable/dictionary/README.md | 8 +- Core/Portable/dictionary/fixtures/corpus.json | 3 +- .../dictionary/fixtures/reference.json | 54 +++++--- Core/Portable/dictionary/src/context.rs | 116 ++++++++++++++++++ Core/Portable/dictionary/src/ffi.rs | 2 +- Core/Portable/dictionary/src/lib.rs | 2 + Core/Portable/dictionary/src/spelling.rs | 4 +- Core/Portable/dictionary/tests/cli.rs | 2 +- Core/Portable/dictionary/tests/contract.rs | 2 +- Core/Portable/dictionary/tests/native.c | 8 +- Core/Portable/dictionary/verify.py | 10 +- Core/Portable/ranking-reference.sh | 1 + Core/Portable/tests/RankingReference.swift | 6 +- .../InkFlowDomain/DictionaryError.swift | 15 +++ .../InkFlowDomain/DictionaryModels.swift | 13 -- .../DictionaryToolBootstrap.swift | 2 - .../InkFlowRime/DictionaryPreparation.swift | 3 - .../DictionaryGenerationReference.swift | 3 +- .../DictionaryGeneratorTests.swift | 4 +- Core/scripts/prepare-spelling.sh | 2 +- macOS/scripts/dictionary-source-bundle.sh | 16 +-- macOS/scripts/test-source-bundle.sh | 2 +- 22 files changed, 210 insertions(+), 68 deletions(-) create mode 100644 Core/Portable/dictionary/src/context.rs create mode 100644 Core/Sources/InkFlowDomain/DictionaryError.swift diff --git a/Core/Portable/dictionary/README.md b/Core/Portable/dictionary/README.md index 1faceca..09ffcfe 100644 --- a/Core/Portable/dictionary/README.md +++ b/Core/Portable/dictionary/README.md @@ -1,6 +1,6 @@ # Shared Rust dictionary generator -This crate ports the Chinese dictionary generator and all 32 spelling profiles for #35. It is an isolated library and command-line tool, with no Rime, Swift, GUI, or network dependency in generation itself. The preparation scripts run its CLI; the Swift dictionary update worker calls the same library through its C ABI. +This crate ports the Chinese dictionary generator, all 32 spelling profiles and the context-ranking index for #35. It is an isolated library and command-line tool, with no Rime, Swift, GUI, or network dependency in generation itself. The preparation scripts run its CLI; the Swift dictionary update worker calls the same library through its C ABI. ## Verify @@ -35,7 +35,7 @@ It refuses an existing output directory. Validation completes before creating ou Inputs are borrowed pointer/length buffers. Catalog and receipt buffers contain UTF-8 JSON; dictionary and correction inputs remain raw bytes so validation can reject malformed text. The boundary accepts at most 64 inputs, 1 MiB of catalog/corrections, 16 KiB per receipt, and the existing 128 MiB per source. Malformed transport inputs return `bridge-input` or `bridge-json`; domain failures retain their code, source, and line. Recoverable Rust panics return `bridge-panic`. Invalid foreign pointers, allocator aborts, and process crashes are outside that guarantee. -Every operation returns an owned opaque result. Successful generation supplies the dictionary and manifest; spelling supplies 32 named schema files; receipt validation supplies no files. Failure supplies error JSON and no files. Output pointers and names are length-delimited, not NUL-terminated, and remain valid until the caller frees the result. Result accessors require a live non-null handle. Independent calls share no mutable generator state. +Every operation returns an owned opaque result. Successful generation supplies the dictionary and manifest; spelling supplies 32 named schema files plus `pinyin_simp.context.bin`; receipt validation supplies no files. Failure supplies error JSON and no files. Output pointers and names are length-delimited, not NUL-terminated, and remain valid until the caller frees the result. Result accessors require a live non-null handle. Independent calls share no mutable generator state. The test script compiles and links a C consumer on both desktops. On macOS it also builds a standalone Swift consumer, releases input storage before reading results, and compares the complete corpus and all spelling bytes through the ABI against the Rust CLI and original Swift implementation. `swift-ffi-corpus.json` retains that consumer's actual summary. This proves the preparation boundary, not shipping worker sandbox, packaging, or resource activation behavior. @@ -43,11 +43,11 @@ The test script compiles and links a C consumer on both desktops. On macOS it al `Core/config/chinese-sources.json` is the authoritative production catalog. Rust embeds it; `IFDictionaryCatalog` reads it through the ABI without filesystem access. Rust also owns the recipe version and maximum source size. `fixtures/cases.json` supplies 43 inputs to both implementations. The test-only Swift reference exports `catalog.json` and `reference.json`; the copied catalog is a pinned fixture, and checks reject drift from the production catalog. -`reference.json` records parser errors, complete provenance manifests, dictionary hashes, and the hashes of all 32 spelling profiles for successful cases. `corpus.json` records the Swift output for the complete pinned source set and current Chinese corrections. Both fixture generation and actual output comparison use isolated build directories. +`reference.json` records parser errors, complete provenance manifests, dictionary hashes, and the hashes of all 32 spelling profiles and the context index for successful cases. `corpus.json` records the Swift output for the complete pinned source set and current Chinese corrections. Both fixture generation and actual output comparison use isolated build directories. The checks cover normalized readings, canonical Unicode key equality with the first display spelling retained, source/group precedence, specialty gap filling, zero weights, log-median calibration, the 100-pair bucket boundary, rounding/saturation, corrections, line/text/reading limits, and malformed input. Separate Rust tests cover receipts, invalid UTF-8, CLI replacement refusal, and output ownership. Source headers remain uninterpreted text; only the tab-separated body is read. -Dictionary bytes and spelling-profile bytes must match exactly. Manifest fields must match after JSON decoding, except calibration multipliers allow relative/absolute error up to `1e-12` for platform math libraries. This tolerance does not apply to weights, hashes, counts, or content versions. JSON spacing and floating-point number spelling are not compatibility requirements. +Dictionary, spelling-profile and context-index bytes must match exactly. Manifest fields must match after JSON decoding, except calibration multipliers allow relative/absolute error up to `1e-12` for platform math libraries. This tolerance does not apply to weights, hashes, counts, or content versions. JSON spacing and floating-point number spelling are not compatibility requirements. Two details are deliberately preserved: diff --git a/Core/Portable/dictionary/fixtures/corpus.json b/Core/Portable/dictionary/fixtures/corpus.json index 05b8101..b8e05d5 100644 --- a/Core/Portable/dictionary/fixtures/corpus.json +++ b/Core/Portable/dictionary/fixtures/corpus.json @@ -217,6 +217,7 @@ "inkflow_spelling_6.schema.yaml": "b70d5fed5129ec8e2b20cf0acddee47cda7d8861966b2ae09ab7d0ce15cdfd6d", "inkflow_spelling_7.schema.yaml": "9d7b1440ef40296fa0a3cad9ec3ef592fe2d6a6b55a7af9e1242dcef325f15f3", "inkflow_spelling_8.schema.yaml": "3082148163842c0c9697bee00e6a48ff8302b59ecac19d4d1f22df9f4eeee015", - "inkflow_spelling_9.schema.yaml": "f171ec24c1dfafdc0f7e5d6df6e03894838bb331dcafbc29a6ae387728f41cef" + "inkflow_spelling_9.schema.yaml": "f171ec24c1dfafdc0f7e5d6df6e03894838bb331dcafbc29a6ae387728f41cef", + "pinyin_simp.context.bin": "2436df5882611b82a563505cea96cb4e5334e1c26a1a493bbaf01a9d16f38899" } } diff --git a/Core/Portable/dictionary/fixtures/reference.json b/Core/Portable/dictionary/fixtures/reference.json index bb41bf2..1b43d02 100644 --- a/Core/Portable/dictionary/fixtures/reference.json +++ b/Core/Portable/dictionary/fixtures/reference.json @@ -239,7 +239,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "8d794603b5915833f6fa0d25a0cec75f8f2abd3d985ccfea28ec439c97cb35c8", - "inkflow_spelling_31.schema.yaml" : "261df9c4bd17e9b2bf97f9b998939eccd90d198bf0d7acf94b749701a9f13d69" + "inkflow_spelling_31.schema.yaml" : "261df9c4bd17e9b2bf97f9b998939eccd90d198bf0d7acf94b749701a9f13d69", + "pinyin_simp.context.bin" : "1180e709baba4c9b81043ab681cc14c048e031454781c19eb904c27d012e6a81" } }, "bucket-100" : { @@ -461,7 +462,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "8d794603b5915833f6fa0d25a0cec75f8f2abd3d985ccfea28ec439c97cb35c8", - "inkflow_spelling_31.schema.yaml" : "261df9c4bd17e9b2bf97f9b998939eccd90d198bf0d7acf94b749701a9f13d69" + "inkflow_spelling_31.schema.yaml" : "261df9c4bd17e9b2bf97f9b998939eccd90d198bf0d7acf94b749701a9f13d69", + "pinyin_simp.context.bin" : "1180e709baba4c9b81043ab681cc14c048e031454781c19eb904c27d012e6a81" } }, "canonical-text" : { @@ -683,7 +685,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "96bf666b0d1830539edce91bcf82fdb610a7f1e3fd40532a7fee9cc15eb7b059", - "inkflow_spelling_31.schema.yaml" : "1bf963f74b72dddc0d606ff1ab93b55811bca014766307194f95730f80bd02f2" + "inkflow_spelling_31.schema.yaml" : "1bf963f74b72dddc0d606ff1ab93b55811bca014766307194f95730f80bd02f2", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "control-text" : { @@ -925,7 +928,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "duplicate-correction" : { @@ -1173,7 +1177,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "d0d7dbb67a3a0a3a6c729683473864b2c6dfcf67113fd5cb40eaeb0f2913a987", - "inkflow_spelling_31.schema.yaml" : "4ae2404bf2afb12e40c2dc098c7b1ac3738868f6a48b3b954b307bedce4be044" + "inkflow_spelling_31.schema.yaml" : "4ae2404bf2afb12e40c2dc098c7b1ac3738868f6a48b3b954b307bedce4be044", + "pinyin_simp.context.bin" : "98b2de510d629305ea66655737913036aed901aecc8dba9b626c23c0c6fbd994" } }, "float-weight" : { @@ -1409,7 +1414,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "half-rounding" : { @@ -1631,7 +1637,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "977743b897399c6ed8304d1fce38c4da4ec35e1ac4a08931b9afffc5a8ed2a2f", - "inkflow_spelling_31.schema.yaml" : "00bc08eb30533178517665c496c3174fc86f15a4817d1a34b675de3a8d7926b6" + "inkflow_spelling_31.schema.yaml" : "00bc08eb30533178517665c496c3174fc86f15a4817d1a34b675de3a8d7926b6", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "long-header" : { @@ -1860,7 +1867,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "long-syllable" : { @@ -2121,7 +2129,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "no-han" : { @@ -2350,7 +2359,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "94b6c1aa0056480b214eee2351ec266eefc22fb54533846be213b6427432757c", - "inkflow_spelling_31.schema.yaml" : "09b1bb6566d25c63752aa4032d46cd39405cdafcc3674bd4781bae48217234ad" + "inkflow_spelling_31.schema.yaml" : "09b1bb6566d25c63752aa4032d46cd39405cdafcc3674bd4781bae48217234ad", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "overrides" : { @@ -2572,7 +2582,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "efb11e8b03c2168b896986fc5b2284c3f3365f4be22ede3dfed6b09f7e1facdc", - "inkflow_spelling_31.schema.yaml" : "9f041f23922fc1aa5c52b5602844805a3f7a769d5131385173f74845b78f51b9" + "inkflow_spelling_31.schema.yaml" : "9f041f23922fc1aa5c52b5602844805a3f7a769d5131385173f74845b78f51b9", + "pinyin_simp.context.bin" : "ea782068c991d2cfa834d96dac58356098046e2b7054402f8ee246537d3393d9" } }, "percent" : { @@ -2801,7 +2812,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "977743b897399c6ed8304d1fce38c4da4ec35e1ac4a08931b9afffc5a8ed2a2f", - "inkflow_spelling_31.schema.yaml" : "00bc08eb30533178517665c496c3174fc86f15a4817d1a34b675de3a8d7926b6" + "inkflow_spelling_31.schema.yaml" : "00bc08eb30533178517665c496c3174fc86f15a4817d1a34b675de3a8d7926b6", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "source-comment" : { @@ -3023,7 +3035,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "space-text" : { @@ -3252,7 +3265,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "8187520b64b9514577a6ac97d0c891a1324f5140f66d0f91d4244c431eb3b6f7", - "inkflow_spelling_31.schema.yaml" : "220221bebbe6f47ddfa263bd0e74fcfc2b4f772247846a3f376b4a452292327b" + "inkflow_spelling_31.schema.yaml" : "220221bebbe6f47ddfa263bd0e74fcfc2b4f772247846a3f376b4a452292327b", + "pinyin_simp.context.bin" : "f7d3c6b146a5bf18d682814e824bcfb599a0843df2c9954bef52621df62d55f6" } }, "syllable-boundary" : { @@ -3474,7 +3488,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "cc0fdef927513de910c4c67094c85392a781ce1497b3fdf835107736ffba09cf", - "inkflow_spelling_31.schema.yaml" : "86556f389c78f3f98194571f0763b72b46ba1ffd4be3d800ccef8604a339cb29" + "inkflow_spelling_31.schema.yaml" : "86556f389c78f3f98194571f0763b72b46ba1ffd4be3d800ccef8604a339cb29", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "syllable-over-limit" : { @@ -3703,7 +3718,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "text-over-limit" : { @@ -3932,7 +3948,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "47217f43c944341b88b15f58ba29e8fc1f17d4b4d2367dba8b0eac0d8d0f3226", - "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b" + "inkflow_spelling_31.schema.yaml" : "4980f12dc1d69db31f0e9b2654eb82c9021521af99b7b33580a1b5e3c0959c7b", + "pinyin_simp.context.bin" : "27afb7471b72ef2bee89a4932cacf53f2192606065e1ee3c9a4ce3b5c16079d1" } }, "union" : { @@ -4154,7 +4171,8 @@ "inkflow_spelling_28.schema.yaml" : "55060136d333ea1fed27c3c678e09b21e3d965ce19ceef44205fffd81fd21495", "inkflow_spelling_29.schema.yaml" : "e76a05006ef24dd3df435bd5f29bac0ed2a78b9dbadee1be17d21224ca3b9fc4", "inkflow_spelling_30.schema.yaml" : "efff2776afc5d823862523fc19000b65b4df8571bfeb9e6583a16951edb1bebe", - "inkflow_spelling_31.schema.yaml" : "77100ab1c52a0f8a9a5648439d2b5c19b1dfd42ba4555313134dbbada265af22" + "inkflow_spelling_31.schema.yaml" : "77100ab1c52a0f8a9a5648439d2b5c19b1dfd42ba4555313134dbbada265af22", + "pinyin_simp.context.bin" : "9c7b3b2248508de0ecdcd7a0ce99bf6c68f0deb69cbea3a430ae1c0b0d8c2fba" } }, "weight-overflow" : { diff --git a/Core/Portable/dictionary/src/context.rs b/Core/Portable/dictionary/src/context.rs new file mode 100644 index 0000000..96c73a7 --- /dev/null +++ b/Core/Portable/dictionary/src/context.rs @@ -0,0 +1,116 @@ +//! Context-ranking index, byte-identical to `IFContextRanker.buildIndex` in +//! InkFlowDomain/CandidateRanking.swift: a sorted, fixed-width table per phrase +//! length so the ranker reads it whole and never parses the dictionary. +use crate::{Error, Result}; +use std::collections::HashMap; +use unicode_normalization::UnicodeNormalization; +use unicode_segmentation::UnicodeSegmentation; + +pub const CONTEXT_INDEX_FILENAME: &str = "pinyin_simp.context.bin"; +const MAGIC: &[u8] = b"IFCX"; +const VERSION: u32 = 1; +const PHRASE_LIMIT: usize = 8; + +pub fn context_index(dictionary: &[u8]) -> Result> { + let text = std::str::from_utf8(dictionary).map_err(|_| Error::new("context-index"))?; + let mut frequencies: HashMap, u64> = HashMap::new(); + let mut longest = 0; + for line in text.split('\n') { + let fields: Vec<_> = line.split('\t').filter(|f| !f.is_empty()).collect(); + if fields.len() < 3 { + continue; + } + let Ok(frequency) = fields[2].parse::() else { + continue; + }; + if frequency < 0 { + continue; + } + let Some(key) = encode(fields[0]) else { + continue; + }; + let entry = frequencies.entry(key).or_insert(0); + *entry = (*entry).max(frequency as u64); + longest = longest.max(fields[0].graphemes(true).count()); + } + let mut tables = vec![Vec::new(); PHRASE_LIMIT - 1]; + for (key, frequency) in frequencies { + tables[key.len() / 3 - 2].push((key, frequency)); + } + let mut data = Vec::new(); + data.extend_from_slice(MAGIC); + for value in [VERSION, longest as u32] + .into_iter() + .chain(tables.iter().map(|table| table.len() as u32)) + { + data.extend_from_slice(&value.to_le_bytes()); + } + for table in &mut tables { + table.sort(); + for (key, frequency) in table.iter() { + data.extend_from_slice(key); + data.extend_from_slice(&frequency.to_le_bytes()); + } + } + Ok(data) +} + +/// Three bytes per canonical scalar for a phrase of 2–8 Han graphemes; `None` otherwise. +fn encode(phrase: &str) -> Option> { + let mut count = 0; + for grapheme in phrase.graphemes(true) { + let mut scalars = grapheme.chars(); + match (scalars.next(), scalars.next()) { + (Some(scalar), None) if is_han(scalar) => count += 1, + _ => return None, + } + } + if !(2..=PHRASE_LIMIT).contains(&count) { + return None; + } + let mut bytes = Vec::with_capacity(count * 3); + for scalar in phrase.nfc() { + let value = scalar as u32; + bytes.extend_from_slice(&[(value >> 16) as u8, (value >> 8) as u8, value as u8]); + } + Some(bytes) +} + +fn is_han(scalar: char) -> bool { + matches!(scalar as u32, + 0x3400..=0x4DBF | 0x4E00..=0x9FFF | 0xF900..=0xFAFF | 0x20000..=0x2FA1F | 0x30000..=0x323AF) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn fields(index: &[u8]) -> Vec { + index[4..40] + .chunks(4) + .map(|c| u32::from_le_bytes(c.try_into().unwrap())) + .collect() + } + + #[test] + fn sorted_fixed_width_tables_per_phrase_length() { + let index = context_index( + "---\nname: x\n...\n你好\tni hao\t5\n你好\tni hao\t9\n世界你好\tshi jie ni hao\t2\n\ + 一\tyi\t3\nabc\tabc\t4\n坏\tbad\n负\tfu\t-1\n九个汉字九个汉字九\t...\t1\n" + .as_bytes(), + ) + .unwrap(); + assert_eq!(&index[..4], b"IFCX"); + assert_eq!(fields(&index), [1, 4, 1, 0, 1, 0, 0, 0, 0]); + let record = &index[40..40 + 2 * 3 + 8]; + assert_eq!(&record[..6], &[0x00, 0x4F, 0x60, 0x00, 0x59, 0x7D]); + assert_eq!(u64::from_le_bytes(record[6..].try_into().unwrap()), 9); + assert_eq!(index.len(), 40 + (6 + 8) + (12 + 8)); + } + + #[test] + fn empty_dictionary_has_only_a_header() { + assert_eq!(context_index(b"").unwrap().len(), 40); + assert_eq!(context_index(&[0xFF]).unwrap_err().code, "context-index"); + } +} diff --git a/Core/Portable/dictionary/src/ffi.rs b/Core/Portable/dictionary/src/ffi.rs index 3c74302..d57bbbf 100644 --- a/Core/Portable/dictionary/src/ffi.rs +++ b/Core/Portable/dictionary/src/ffi.rs @@ -308,7 +308,7 @@ mod tests { let result = ifd_spelling(Bytes::borrowed(&source)); drop(source); assert_eq!(ifd_result_error(result).len, 0); - assert_eq!(ifd_result_count(result), 32); + assert_eq!(ifd_result_count(result), 33); for (index, (name, data)) in expected.iter().enumerate() { assert_eq!( ifd_result_name(result, index).read(1024).unwrap(), diff --git a/Core/Portable/dictionary/src/lib.rs b/Core/Portable/dictionary/src/lib.rs index ea776f4..19bda54 100644 --- a/Core/Portable/dictionary/src/lib.rs +++ b/Core/Portable/dictionary/src/lib.rs @@ -1,7 +1,9 @@ //! Offline dictionary generation. Inputs and provenance are supplied by the caller. +mod context; mod ffi; mod model; mod spelling; +pub use context::{CONTEXT_INDEX_FILENAME, context_index}; pub use model::*; pub use spelling::spelling; diff --git a/Core/Portable/dictionary/src/spelling.rs b/Core/Portable/dictionary/src/spelling.rs index 93a0ad7..dac7ec0 100644 --- a/Core/Portable/dictionary/src/spelling.rs +++ b/Core/Portable/dictionary/src/spelling.rs @@ -1,6 +1,7 @@ -use crate::{Result, read_rows}; +use crate::{CONTEXT_INDEX_FILENAME, Result, context_index, read_rows}; use std::collections::{BTreeMap, BTreeSet}; +/// Every resource derived from the generated dictionary: 32 spelling profiles and the context index. pub fn spelling(dictionary: &[u8]) -> Result>> { let mut syllables = BTreeSet::new(); read_rows(dictionary, "generated-spelling", |key, _| { @@ -88,5 +89,6 @@ pub fn spelling(dictionary: &[u8]) -> Result>> { ); schemas.insert(format!("{name}.schema.yaml"), schema.into_bytes()); } + schemas.insert(CONTEXT_INDEX_FILENAME.to_owned(), context_index(dictionary)?); Ok(schemas) } diff --git a/Core/Portable/dictionary/tests/cli.rs b/Core/Portable/dictionary/tests/cli.rs index 2c5ee16..847f06e 100644 --- a/Core/Portable/dictionary/tests/cli.rs +++ b/Core/Portable/dictionary/tests/cli.rs @@ -105,5 +105,5 @@ fn cli_checks_inputs_before_publishing_and_refuses_replacement() { .output() .unwrap(); assert!(output.status.success()); - assert_eq!(fs::read_dir(root.join("spelling")).unwrap().count(), 32); + assert_eq!(fs::read_dir(root.join("spelling")).unwrap().count(), 33); } diff --git a/Core/Portable/dictionary/tests/contract.rs b/Core/Portable/dictionary/tests/contract.rs index 586b949..ac0e1d7 100644 --- a/Core/Portable/dictionary/tests/contract.rs +++ b/Core/Portable/dictionary/tests/contract.rs @@ -117,7 +117,7 @@ fn swift_reference_contract() { } } println!( - "PASS {} Swift/Rust contract cases, manifests and 32 spelling profiles", + "PASS {} Swift/Rust contract cases, manifests, 32 spelling profiles and context indexes", cases.len() ); } diff --git a/Core/Portable/dictionary/tests/native.c b/Core/Portable/dictionary/tests/native.c index c98662c..f9dbc69 100644 --- a/Core/Portable/dictionary/tests/native.c +++ b/Core/Portable/dictionary/tests/native.c @@ -13,21 +13,21 @@ int main(void) { IFDResult* spelling = ifd_spelling(bytes(dictionary)); memset(dictionary, 0, sizeof(dictionary)); assert(ifd_result_error(spelling).len == 0); - assert(ifd_result_count(spelling) == 32); - for (size_t i = 0; i < 32; ++i) { + assert(ifd_result_count(spelling) == 33); + for (size_t i = 0; i < 33; ++i) { IFDBytes name = ifd_result_name(spelling, i); IFDBytes data = ifd_result_data(spelling, i); assert(name.len > 0 && data.len > 0); assert(memchr(name.data, '/', name.len) == NULL); } - assert(ifd_result_name(spelling, 32).len == 0); + assert(ifd_result_name(spelling, 33).len == 0); assert(ifd_result_data(spelling, SIZE_MAX).len == 0); IFDResult* malformed = ifd_generate(bytes("[]"), NULL, 1, bytes("")); assert(ifd_result_error(malformed).len > 0); assert(ifd_result_count(malformed) == 0); ifd_result_free(malformed); - assert(ifd_result_count(spelling) == 32); + assert(ifd_result_count(spelling) == 33); ifd_result_free(spelling); const uint8_t invalid_utf8[] = {0xff, 0x00}; diff --git a/Core/Portable/dictionary/verify.py b/Core/Portable/dictionary/verify.py index c580b10..c41c537 100644 --- a/Core/Portable/dictionary/verify.py +++ b/Core/Portable/dictionary/verify.py @@ -57,7 +57,8 @@ def inputs(catalog): def snapshot(dictionary, schemas): manifest = json.loads((dictionary / 'dictionary-manifest.json').read_text()) assert manifest['dictionarySHA256'] == digest(dictionary / 'pinyin_simp.dict.yaml') - return dict(manifest=manifest, spellingSHA256={p.name: digest(p) for p in sorted(schemas.glob('*.schema.yaml'))}) + return dict(manifest=manifest, spellingSHA256={p.name: digest(p) for p in sorted(schemas.iterdir()) + if p.name.endswith('.schema.yaml') or p.name == 'pinyin_simp.context.bin'}) def main(): @@ -77,7 +78,7 @@ def main(): str(ROOT / 'Core/config/chinese-overrides.tsv'), str(generated)], check=True) subprocess.run([str(args.binary), 'spelling', str(generated / 'pinyin_simp.dict.yaml'), str(schemas)], check=True) actual = snapshot(generated, schemas) - assert len(actual['spellingSHA256']) == 32 + assert len(actual['spellingSHA256']) == 33 if args.swift: assert (generated / 'pinyin_simp.dict.yaml').read_bytes() == (args.swift / 'pinyin_simp.dict.yaml').read_bytes(), 'Dictionary bytes differ from Swift' expected = snapshot(args.swift, args.swift) @@ -88,7 +89,7 @@ def main(): args.report.mkdir(parents=True, exist_ok=True) (args.report / 'corpus.json').write_text(json.dumps(expected, ensure_ascii=False, indent=2) + '\n') (args.report / 'rust-corpus.json').write_text(json.dumps(actual, ensure_ascii=False, indent=2) + '\n') - print(f"PASS pinned corpus: {actual['manifest']['entryCount']} rows, dictionary and all 32 profiles match Swift") + print(f"PASS pinned corpus: {actual['manifest']['entryCount']} rows, dictionary, all 32 profiles and the context index match Swift") if args.bridge: native_dictionary, native_schemas = root / 'native-dictionary', root / 'native-spelling' subprocess.run([str(args.bridge), 'generate', str(args.catalog), str(source), str(source / 'legacy.yaml'), @@ -100,8 +101,9 @@ def main(): assert (native_dictionary / 'pinyin_simp.dict.yaml').read_bytes() == (generated / 'pinyin_simp.dict.yaml').read_bytes() for file in schemas.glob('*.schema.yaml'): assert (native_schemas / file.name).read_bytes() == file.read_bytes(), file.name + assert (native_schemas / 'pinyin_simp.context.bin').read_bytes() == (schemas / 'pinyin_simp.context.bin').read_bytes() (args.report / 'swift-ffi-corpus.json').write_text(json.dumps(native, ensure_ascii=False, indent=2) + '\n') - print('PASS Swift in-process Rust ABI: complete dictionary, manifest, and 32 spelling profiles') + print('PASS Swift in-process Rust ABI: complete dictionary, manifest, 32 spelling profiles and context index') if args.swift: for name in ('catalog.json', 'reference.json', 'corpus.json'): equivalent(json.loads((args.report / name).read_text()), json.loads((FIXTURES / name).read_text()), name) diff --git a/Core/Portable/ranking-reference.sh b/Core/Portable/ranking-reference.sh index e7839dd..13de8d8 100644 --- a/Core/Portable/ranking-reference.sh +++ b/Core/Portable/ranking-reference.sh @@ -4,6 +4,7 @@ cd "$(dirname "$0")/../.." mkdir -p build/portable xcrun swiftc -package-name InkFlow \ Core/Sources/InkFlowDomain/CandidateRanking.swift \ + Core/Sources/InkFlowDomain/DictionaryError.swift \ Core/Portable/tests/RankingReference.swift \ -o build/portable/ranking-reference build/portable/ranking-reference Core/Portable/fixtures/ranking-cases.json "$1/ranking-reference.json" diff --git a/Core/Portable/tests/RankingReference.swift b/Core/Portable/tests/RankingReference.swift index e3d7462..ad00e20 100644 --- a/Core/Portable/tests/RankingReference.swift +++ b/Core/Portable/tests/RankingReference.swift @@ -9,11 +9,11 @@ struct RankingReference { let directory = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) try FileManager.default.createDirectory(at: directory, withIntermediateDirectories: true) defer { try? FileManager.default.removeItem(at: directory) } - let file = directory.appendingPathComponent("dictionary.yaml") + let file = directory.appendingPathComponent("pinyin_simp.context.bin") var results: [String: Any] = [:] for item in cases { - try (item["dictionary"] as! String).write(to: file, atomically: true, encoding: .utf8) - let ranker = try IFContextRanker(dictionary: file.path) + try IFContextRanker.buildIndex(dictionary: Data((item["dictionary"] as! String).utf8)).write(to: file) + let ranker = try IFContextRanker(index: file.path) let candidates = item["candidates"] as! [String] let metadata = (item["metadata"] as? String).flatMap { IFContextRanker.parseMetadata($0, offset: item["offset"] as! Int, diff --git a/Core/Sources/InkFlowDomain/DictionaryError.swift b/Core/Sources/InkFlowDomain/DictionaryError.swift new file mode 100644 index 0000000..c6eb74f --- /dev/null +++ b/Core/Sources/InkFlowDomain/DictionaryError.swift @@ -0,0 +1,15 @@ +import Foundation + +/// Shared by dictionary generation and the context ranker; compiled standalone by Core/Portable/ranking-reference.sh. +package struct IFDictionaryError: Error, LocalizedError, Sendable { + package let code: String + package let source: String? + package let line: Int? + package let detail: String + package init(_ code: String, source: String? = nil, line: Int? = nil, _ detail: String) { + self.code = code; self.source = source; self.line = line; self.detail = detail + } + package var errorDescription: String? { + [code, source, line.map { "line \($0)" }, detail].compactMap { $0 }.joined(separator: ": ") + } +} diff --git a/Core/Sources/InkFlowDomain/DictionaryModels.swift b/Core/Sources/InkFlowDomain/DictionaryModels.swift index bde4b14..0734a3a 100644 --- a/Core/Sources/InkFlowDomain/DictionaryModels.swift +++ b/Core/Sources/InkFlowDomain/DictionaryModels.swift @@ -102,19 +102,6 @@ package struct IFDictionaryGeneration: Sendable { package let manifest: IFDictionaryManifest } -package struct IFDictionaryError: Error, LocalizedError, Sendable { - package let code: String - package let source: String? - package let line: Int? - package let detail: String - package init(_ code: String, source: String? = nil, line: Int? = nil, _ detail: String) { - self.code = code; self.source = source; self.line = line; self.detail = detail - } - package var errorDescription: String? { - [code, source, line.map { "line \($0)" }, detail].compactMap { $0 }.joined(separator: ": ") - } -} - package enum IFDictionaryHash { package static func sha256(_ data: Data) -> String { SHA256.hash(data: data).map { String(format: "%02x", $0) }.joined() } package static func gitBlob(_ data: Data) -> String { diff --git a/Core/Sources/InkFlowDomain/DictionaryToolBootstrap.swift b/Core/Sources/InkFlowDomain/DictionaryToolBootstrap.swift index 51395d6..7cc24fa 100644 --- a/Core/Sources/InkFlowDomain/DictionaryToolBootstrap.swift +++ b/Core/Sources/InkFlowDomain/DictionaryToolBootstrap.swift @@ -25,8 +25,6 @@ package enum IFDictionaryToolBootstrap { let output = URL(fileURLWithPath: arguments[2]) try FileManager.default.createDirectory(at: output, withIntermediateDirectories: true) try IFSpellingGenerator.write(dictionary: dictionary, to: output) - try IFContextRanker.buildIndex(dictionary: dictionary) - .write(to: output.appendingPathComponent(IFDictionaryCatalog.contextIndexFilename), options: .atomic) } else { throw IFDictionaryError("arguments", "Usage: dictionary-generator sources | generate SOURCES LEGACY CORRECTIONS OUTPUT | spelling DICTIONARY OUTPUT") } diff --git a/Core/Sources/InkFlowRime/DictionaryPreparation.swift b/Core/Sources/InkFlowRime/DictionaryPreparation.swift index d6efc92..a08154f 100644 --- a/Core/Sources/InkFlowRime/DictionaryPreparation.swift +++ b/Core/Sources/InkFlowRime/DictionaryPreparation.swift @@ -72,9 +72,6 @@ package enum IFDictionaryPreparation { // Keep replacement staging inside the sandbox's candidate directory. try IFDictionaryFiles.atomicWrite(spelling[name]!, to: shared.appendingPathComponent(name)) } - operation = "generate-context-index" - try IFDictionaryFiles.atomicWrite(IFContextRanker.buildIndex(dictionary: dictionary), - to: shared.appendingPathComponent(IFDictionaryCatalog.contextIndexFilename)) operation = "create-isolated-directories" let cache = try IFDictionaryFiles.child("cache", in: root) let compiler = try IFDictionaryFiles.child("compile-user", in: root) diff --git a/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift b/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift index d1d26b2..d9242bc 100644 --- a/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift +++ b/Core/Tests/DictionaryGeneratorTests/DictionaryGenerationReference.swift @@ -17,7 +17,8 @@ extension DictionaryGeneratorTests { do { let generated = try IFReferenceDictionaryGenerator.generate(inputs: inputs, corrections: Data((test.corrections ?? "").utf8), catalog: catalog) - let schemas = try IFReferenceSpellingGenerator.generate(dictionary: generated.dictionary) + var schemas = try IFReferenceSpellingGenerator.generate(dictionary: generated.dictionary) + schemas[IFDictionaryCatalog.contextIndexFilename] = try IFContextRanker.buildIndex(dictionary: generated.dictionary) results[test.name] = [ "manifest": try JSONSerialization.jsonObject(with: generated.manifest.encoded()), "spellingSHA256": schemas.mapValues { IFDictionaryHash.sha256($0) } diff --git a/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift b/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift index 5cd2cd3..5f9ae02 100644 --- a/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift +++ b/Core/Tests/DictionaryGeneratorTests/DictionaryGeneratorTests.swift @@ -132,7 +132,7 @@ struct DictionaryGeneratorTests { static func spellingGeneration() throws { let data = Data("---\nname: pinyin_simp\n...\n来俩\tlai lia\t1\n女略\tnu lue\t2\n居\tju\t3\n赞\tzan\t4\n包\tbao\t5\n".utf8) let schemas = try IFSpellingGenerator.generate(dictionary: data) - expect(schemas.count == 32, "Every preference profile has its own native prism schema") + expect(schemas.count == 33, "Every preference profile has its own native prism schema, plus the context index") expect(try IFSpellingGenerator.generate(dictionary: data) == schemas, "Deterministic spelling generation") let reordered = Data("---\nname: pinyin_simp\n...\n# source-only change\n包\tbao\t99\n赞\tzan\t4\n居\tju\t3\n女略\tnu lue\t2\n来俩\tlai lia\t1\n".utf8) expect(try IFSpellingGenerator.generate(dictionary: reordered) == schemas, "Order, weights and comments do not change spelling") @@ -185,6 +185,8 @@ struct DictionaryGeneratorTests { try reference.dictionary.write(to: referenceDirectory.appendingPathComponent(IFDictionaryCatalog.dictionaryFilename)) try reference.manifest.encoded().write(to: referenceDirectory.appendingPathComponent(IFDictionaryManifest.filename)) try IFReferenceSpellingGenerator.write(dictionary: reference.dictionary, to: referenceDirectory) + try IFContextRanker.buildIndex(dictionary: reference.dictionary) + .write(to: referenceDirectory.appendingPathComponent(IFDictionaryCatalog.contextIndexFilename)) for (name, bytes) in try IFSpellingGenerator.generate(dictionary: result.dictionary) { expect(try Data(contentsOf: destination.appendingPathComponent(name)) == bytes, "Build CLI and runtime spelling bytes match: \(name)") } diff --git a/Core/scripts/prepare-spelling.sh b/Core/scripts/prepare-spelling.sh index f3ac583..7ab0f69 100644 --- a/Core/scripts/prepare-spelling.sh +++ b/Core/scripts/prepare-spelling.sh @@ -9,4 +9,4 @@ bash Core/scripts/build-dictionary-generator.sh staging=$(mktemp -d build/.spelling.XXXXXX) trap 'rm -rf "$staging"' EXIT build/dictionary-generator spelling "$destination/pinyin_simp.dict.yaml" "$staging/generated" -cp "$staging/generated/"*.schema.yaml "$destination/" +cp "$staging/generated/"*.schema.yaml "$staging/generated/pinyin_simp.context.bin" "$destination/" diff --git a/macOS/scripts/dictionary-source-bundle.sh b/macOS/scripts/dictionary-source-bundle.sh index fc733ca..6682228 100755 --- a/macOS/scripts/dictionary-source-bundle.sh +++ b/macOS/scripts/dictionary-source-bundle.sh @@ -54,10 +54,10 @@ build/dictionary-generator sources | while IFS=$'\t' read -r id sha bytes url; d record "Chinese dictionary source $id" "$(license_for "$repository")" "$url" "upstream/dictionary-sources/$id.yaml" done # Pinned archives and files that dependencies.sh downloads and verifies. -deps=$(sed -nE 's/^fetch ([^ ]+) ([0-9a-f]{64}) (https:[^ ]+)$/\1 \2 \3/p' macOS/scripts/dependencies.sh) +deps=$(sed -nE 's/^fetch ([^ ]+) ([0-9a-f]{64}) (https:[^ ]+)$/\1 \2 \3/p' Core/scripts/resource-dependencies.sh) for file in pinyin.tar.gz english.tar.gz emoji.txt; do url=$(printf '%s\n' "$deps" | awk -v f="$file" '$1 == f {print $3}') - [[ -n "$url" ]] || { echo "dependencies.sh no longer fetches $file." >&2; exit 1; } + [[ -n "$url" ]] || { echo "resource-dependencies.sh no longer fetches $file." >&2; exit 1; } cp "build/deps/$file" "$root/upstream/deps/$file" case "$file" in pinyin.tar.gz) record 'Legacy compatibility dictionary (rime/rime-pinyin-simp repository archive)' 'Apache-2.0 (LICENSES/pinyin-simp.txt)' "$url" "upstream/deps/$file" ;; @@ -77,7 +77,7 @@ for opencc in STPhrases.txt STCharacters.txt; do done record 'InkFlow Chinese corrections and curated additions' 'Apache-2.0 (inkflow/LICENSE)' "https://github.com/nervouna/InkFlow/tree/$commit" inkflow/Core/config/chinese-overrides.tsv record 'InkFlow English admission policy' 'Apache-2.0 (inkflow/LICENSE)' "https://github.com/nervouna/InkFlow/tree/$commit" inkflow/Core/config/english-overrides.tsv -record 'InkFlow dictionary generator and pinned source catalog' 'Apache-2.0 (inkflow/LICENSE)' "https://github.com/nervouna/InkFlow/tree/$commit" inkflow/Core/Sources/InkFlowDomain/DictionaryModels.swift +record 'InkFlow pinned source catalog' 'Apache-2.0 (inkflow/LICENSE)' "https://github.com/nervouna/InkFlow/tree/$commit" inkflow/Core/config/chinese-sources.json cat > "$root/README.md" < Date: Thu, 8 Oct 2026 23:37:55 +0800 Subject: [PATCH 10/21] Port session policy to the Rust engine and compare it with the Swift baseline Add `engine.rs`: `Engine` owns one runtime, the prepared context index and Rime's native custom_phrase.txt; `InputSession` ports `IFEngine`'s ordering and selection by displayed index, digit/Up/Down policy, preceding-text context, candidate counts and input preferences applied by replacing schema nodes at a composition boundary, ASCII mode, shared idle phrase reloads and commit text preserved across a settings reload. Port `InputPreferences`, `CustomPhrase` and the property-channel framing; load the prepared `pinyin_simp.context.bin` in the ranker. The bridge gains input, highlight, commit, option, channel and schema-patch entry points over Rime's public C API. Quality recording stays outside the engine: an optional observer receives each completed mutation with the displayed snapshots before and after, after the policy lock is released. `tests/parity.rs` runs on prepared production resources (`parity.sh`): all 21 quality-baseline samples reproduce `baseline.json`, including learning after a runtime restart, and the ported Swift engine regressions pass for basic cases, context ranking, custom phrases, input settings and ASCII boundaries. Refs #35 --- Core/Portable/README.md | 23 +- Core/Portable/native/bridge.cpp | 97 ++- Core/Portable/native/bridge.h | 15 + Core/Portable/parity.sh | 15 + Core/Portable/src/channel.rs | 112 ++++ Core/Portable/src/engine.rs | 1063 ++++++++++++++++++++++++++++++ Core/Portable/src/ffi.rs | 19 + Core/Portable/src/lib.rs | 143 +++- Core/Portable/src/phrases.rs | 143 ++++ Core/Portable/src/preferences.rs | 183 +++++ Core/Portable/src/ranking.rs | 92 ++- Core/Portable/tests/parity.rs | 909 +++++++++++++++++++++++++ 12 files changed, 2801 insertions(+), 13 deletions(-) create mode 100755 Core/Portable/parity.sh create mode 100644 Core/Portable/src/channel.rs create mode 100644 Core/Portable/src/engine.rs create mode 100644 Core/Portable/src/phrases.rs create mode 100644 Core/Portable/src/preferences.rs create mode 100644 Core/Portable/tests/parity.rs diff --git a/Core/Portable/README.md b/Core/Portable/README.md index 3f84193..86cc233 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -1,6 +1,6 @@ -# Desktop Rust/Rime probe +# Desktop Rust/Rime runtime -Minimal Rust runtime over librime for [#34](https://github.com/nervouna/InkFlow/issues/34). The Swift session/ranking/learning engine still ships. +Rust runtime over librime for [#34](https://github.com/nervouna/InkFlow/issues/34) and the shared session policy for [#35](https://github.com/nervouna/InkFlow/issues/35). The Swift engine still ships on macOS; the Rust engine is compared against it here. ## Build and test @@ -15,6 +15,8 @@ python3 scripts/mac-remote.py portable # Compile and smoke-test full production resources in isolated directories: bash Core/Portable/prepare-resources.sh python3 scripts/mac-remote.py resources +# Compare the Rust engine with the recorded Swift behavior on those resources: +bash Core/Portable/parity.sh [build/portable/resources.XXXXXX] ``` The first build downloads checksum-verified source archives. Subsequent builds reuse them under `build/portable/`; native outputs and Cargo artifacts stay there too. Set `CMAKE_BUILD_PARALLEL_LEVEL` to change the native build parallelism (default 4). Do not run two builds in the same checkout. Remove `build/portable/` for a clean rebuild or after changing compiler/SDK/architecture. @@ -27,19 +29,28 @@ OpenCC 1.1.9 requests C++14, but the pinned marisa headers require C++17. The bu The test copies the tiny checked-in source fixture into a fresh temporary directory, compiles it using the target runtime, and removes the directory after runtime teardown. It checks Chinese composition, one-shot commits, a non-ASCII Lua filter, snapshot ownership, serialized sessions across threads, runtime/session lifetime, and restart. -## Ranking comparison +## Session policy and old/new comparison -`src/ranking.rs` ports the existing Swift equivalent-span ranker, including strict metadata parsing, Han phrase context, personal-strength limits, custom-source priority, and native-order fallback. Its 291 synthetic cases compare against the unchanged Swift source. Mac runtime tests regenerate the reference before checking Rust; Linux tests use the recorded results. Dictionary indexing belongs in preparation, while ordering uses only memory. This module is not yet wired into session selection or key handling, so the runtime below still exposes native order. +`src/engine.rs` ports `IFEngine`: `Engine` owns one runtime, the prepared context index and the custom-phrase file; `InputSession` owns one input context. Rime keeps editing, segmentation, lookup, paging and its own learning. The engine adds equivalent-span ordering and selection by displayed index, digit and Up/Down policy, the preceding-text context, custom phrases through Rime's native `custom_phrase.txt` (written on settings changes, reloaded by every session at a shared idle), input preferences and candidate counts applied by replacing schema nodes at a composition boundary, ASCII mode, and commit text preserved across a settings reload. `src/preferences.rs`, `src/phrases.rs` and `src/channel.rs` port the matching Swift domain code. + +`src/ranking.rs` ports the Swift equivalent-span ranker, including strict metadata parsing, Han phrase context, personal-strength limits, custom-source priority, and native-order fallback; it reads the prepared `pinyin_simp.context.bin` whole into memory and never touches storage from a key event. Its 291 synthetic cases compare against the unchanged Swift source. Mac runtime tests regenerate the reference before checking Rust; Linux tests use the recorded results. + +`tests/parity.rs` compares the Rust engine with the Swift engine on prepared production resources and isolated user directories (`parity.sh`; the tests skip without `INKFLOW_PORTABLE_RESOURCES`): + +- The 21 samples of [the quality baseline](../Fixtures/QualityBaseline/README.md) reproduce `baseline.json` exactly: first page, target rank, paging, selection, one-shot commits, five learning commits, and the learned state after the runtime restarts on the same user directory. +- `swift_engine_regressions` carries the expectations of the Swift engine regressions for basic cases, context ranking, custom phrases, input settings and ASCII boundaries, plus the stale-selection and observer contracts. + +Quality recording stays outside the engine. `Engine::set_observer` installs a callback that receives every completed mutation with the displayed snapshots before and after, delivered after the policy lock is released; it cannot change input or hold a key event. The engine itself performs no telemetry, network, SQLite or disk work from a key event other than the phrase-file reload at a shared idle. Personal-learning management and portable personal-data import are not ported yet. ## Interface contract `native/bridge.h` is the experimental internal C ABI between Rust and the native runtime. It uses Rime's public C API for lifecycle, deployment, input, context, and commits. C++ internals are confined to the existing extension and its registration check. The Rust library exposes safe `Runtime` and `Session` handles; a stable frontend-facing exported Rust C ABI is deferred until session policy is migrated. -- One process-wide Rust mutex serializes **all** Rime calls, including initialization, deployment, reads, destruction, and finalization. Sessions can move between threads. No GUI event loop, Swift actor, or thread affinity is required. Another engine must not call librime outside this lock in the same process; old/new comparisons need separate processes. +- One process-wide Rust mutex serializes **all** Rime calls, including initialization, deployment, reads, destruction, and finalization. The engine's policy lock serializes every `InputSession` of an `Engine` and nests outside that mutex. Sessions can move between threads. No GUI event loop, Swift actor, or thread affinity is required. Another engine must not call librime outside this lock in the same process; old/new comparisons need separate processes. - There is at most one runtime. A second initialization returns `AlreadyRunning`. Sessions retain an `Arc` to the runtime, so dropping its public handle cannot finalize live sessions. The last session/runtime owner finalizes Rime. A poisoned lock fails subsequent normal operations; destructors still attempt cleanup without panicking. - Callers supply existing shared-resource and writable user directories. Initialization, source deployment, and session creation are setup work. `Runtime::with_cache` separates target-native cache files from personal data. `prepare` runs Rime's schema-list maintenance and checks its completion notification. Preparation and deployment are rejected while sessions exist. The wrapper does not prepare resources, download, access SQLite, or record telemetry from `process_key`. - All text and paths crossing this ABI are NUL-terminated UTF-8. Embedded NUL and non-UTF-8 paths are rejected. Snapshot caret/selection offsets count **bytes in the returned UTF-8 preedit**. Rust validates character boundaries; frontends must convert to UTF-16 or other platform units. Candidate text and comments are independent strings. -- Keys are Rime/X11 keysyms with Rime modifier masks, not macOS virtual key codes or Linux hardware scan codes. `process_key` returns whether Rime handled the event. There is no surrounding-text or mobile editing API yet. `change_page` delegates paging to Rime. `select_candidate` accepts a zero-based current-page index and the latest snapshot from that session. It rejects snapshots from other sessions, superseded snapshots, and out-of-range indices before calling Rime. Keys, clears, native selection attempts, and page changes invalidate selection tokens even if Rime does not handle the operation. Cloned snapshots retain their token; changing public display fields cannot change the native candidate count used for validation. +- Keys are Rime/X11 keysyms with Rime modifier masks, not macOS virtual key codes or Linux hardware scan codes. `process_key` returns whether Rime handled the event. There is no surrounding-text or mobile editing API yet. `change_page` delegates paging to Rime. `select_candidate` accepts a zero-based current-page index and the latest snapshot from that session. It rejects snapshots from other sessions, superseded snapshots, and out-of-range indices before calling Rime. Keys, clears, native selection attempts, and page changes invalidate selection tokens even if Rime does not handle the operation. Cloned snapshots retain their token; changing public display fields cannot change the native candidate count used for validation. `InputSession::select` and `highlight` apply the same rule to displayed indices: the snapshot must be the session's latest ordered page, and any mutation supersedes it, while repeated reads keep one identity. - Snapshots are copied into Rust-owned strings and vectors while the lock is held. They remain valid after another event, another snapshot, session destruction, and runtime teardown. Page and highlighted indices are zero-based native menu metadata; an empty menu has no selectable entry regardless of those fields. - `take_commit` consumes Rime's pending commit once and returns `None` when empty. Call it after each key or candidate selection before processing the next action. The prototype does not queue multiple uncollected commits. On an allocation or UTF-8 conversion error, a retrieved commit may already have been consumed; stop the session rather than replaying the key. - Internal C callers must pass valid borrowed pointers and zero-initialized outputs. Status 0 is success, -1 native failure, -2 invalid session/schema, and -3 a caught C++ exception. A false/unhandled key is not an error. Native snapshots and commit buffers must be freed with their matching ABI functions, never Rust's allocator. Snapshot free clears the struct; it also accepts a zeroed snapshot. diff --git a/Core/Portable/native/bridge.cpp b/Core/Portable/native/bridge.cpp index f50937e..5208fbf 100644 --- a/Core/Portable/native/bridge.cpp +++ b/Core/Portable/native/bridge.cpp @@ -32,6 +32,11 @@ struct Config { RimeConfig value{}; ~Config() { if (value.ptr) api()->config_close(&value); } }; +char* duplicate(const char* value) { + char* copy = static_cast(std::malloc(std::strlen(value) + 1)); + if (copy) std::strcpy(copy, value); + return copy; +} struct Commit { RimeCommit value{}; Commit() { RIME_STRUCT_INIT(RimeCommit, value); } @@ -172,12 +177,98 @@ extern "C" int ifp_take_commit(uintptr_t session, char** result) { if (!api()->find_session(session)) return -2; Commit commit; if (!api()->get_commit(session, &commit.value)) return 0; - const char* value = text(commit.value.text); - char* copy = static_cast(std::malloc(std::strlen(value) + 1)); + char* copy = duplicate(text(commit.value.text)); if (!copy) return -1; - std::strcpy(copy, value); *result = copy; return 0; } catch (...) { return -3; } } extern "C" void ifp_string_free(char* value) { std::free(value); } +extern "C" int ifp_get_input(uintptr_t session, char** result) { + try { + if (!api()->find_session(session)) return -2; + char* copy = duplicate(text(api()->get_input(session))); + if (!copy) return -1; + *result = copy; + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_highlight_candidate(uintptr_t session, size_t index, int* handled) { + try { + if (!api()->find_session(session)) return -2; + *handled = api()->highlight_candidate_on_current_page(session, index) ? 1 : 0; + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_commit_composition(uintptr_t session, int* handled) { + try { + if (!api()->find_session(session)) return -2; + *handled = api()->commit_composition(session) ? 1 : 0; + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_set_option(uintptr_t session, const char* option, int value) { + try { + if (!api()->find_session(session)) return -2; + api()->set_option(session, option, value != 0); + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_get_option(uintptr_t session, const char* option, int* value) { + try { + if (!api()->find_session(session)) return -2; + *value = api()->get_option(session, option) ? 1 : 0; + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_call(uintptr_t session, const char* request, size_t capacity, char** result) { + try { + if (!api()->find_session(session)) return -2; + api()->set_property(session, "inkflow_result", ""); + api()->set_property(session, "inkflow_request", request); + api()->set_property(session, "inkflow_request", ""); + std::string buffer(capacity + 1, '\0'); + const bool read = api()->get_property(session, "inkflow_result", buffer.data(), buffer.size() - 1); + api()->set_property(session, "inkflow_result", ""); + if (!read) return 0; + // A reply that fills the buffer may be truncated; fail closed. + const size_t end = buffer.find('\0'); + if (end >= buffer.size() - 1) return 0; + char* copy = duplicate(buffer.c_str()); + if (!copy) return -1; + *result = copy; + return 0; + } catch (...) { return -3; } +} +extern "C" int ifp_apply_schema_patch(uintptr_t session, const char* schema, const char* yaml, + const char* const* paths, size_t count, int* outcome) { + try { + if (!api()->find_session(session)) return -2; + Config config; + if (!api()->schema_open(schema, &config.value)) { *outcome = 1; return -1; } + Config patch; + if (!api()->config_init(&patch.value)) { *outcome = 2; return -1; } + if (!api()->config_load_string(&patch.value, yaml)) { *outcome = 3; return -1; } + // Replace whole nodes and restore them synchronously. Other sessions retain their + // component configuration, while this idle session loads one coherent snapshot. + std::vector> originals; + bool patched = true; + for (size_t i = 0; i < count; ++i) { + RimeConfig original{}, replacement{}; + if (!api()->config_get_item(&config.value, paths[i], &original)) { patched = false; break; } + originals.emplace_back(paths[i], original); + if (!api()->config_get_item(&patch.value, paths[i], &replacement)) { patched = false; break; } + const bool changed = api()->config_set_item(&config.value, paths[i], &replacement); + api()->config_close(&replacement); + if (!changed) { patched = false; break; } + } + const bool selected = patched && api()->select_schema(session, schema); + bool restored = true; + for (auto it = originals.rbegin(); it != originals.rend(); ++it) { + if (!api()->config_set_item(&config.value, it->first, &it->second)) restored = false; + api()->config_close(&it->second); + } + *outcome = (patched ? 1 : 0) | (selected ? 2 : 0) | (restored ? 4 : 0); + return 0; + } catch (...) { return -3; } +} diff --git a/Core/Portable/native/bridge.h b/Core/Portable/native/bridge.h index 0e11e65..fd165b3 100644 --- a/Core/Portable/native/bridge.h +++ b/Core/Portable/native/bridge.h @@ -43,6 +43,21 @@ void ifp_snapshot_free(IFPSnapshot* snapshot); /* A successful empty read returns NULL. A non-NULL commit is consumed once. */ int ifp_take_commit(uintptr_t session, char** commit); void ifp_string_free(char* string); +/* Raw composition input; an empty composition yields an empty string. */ +int ifp_get_input(uintptr_t session, char** input); +int ifp_highlight_candidate(uintptr_t session, size_t index, int* handled); +int ifp_commit_composition(uintptr_t session, int* handled); +int ifp_set_option(uintptr_t session, const char* option, int value); +int ifp_get_option(uintptr_t session, const char* option, int* value); +/* One synchronous property-channel round trip. NULL result: unanswered or a reply + * that would not fit in capacity bytes. Properties never retain request or reply. */ +int ifp_call(uintptr_t session, const char* request, size_t capacity, char** result); +/* Replace whole configuration nodes of the schema with the patch's, reselect the + * schema in this session, then restore the originals. outcome: bit 0 patched, + * bit 1 selected, bit 2 restored. -1 with outcome 1/2/3: schema open, patch + * init or patch load failed before any change. */ +int ifp_apply_schema_patch(uintptr_t session, const char* schema, const char* yaml, + const char* const* paths, size_t count, int* outcome); #ifdef __cplusplus } #endif diff --git a/Core/Portable/parity.sh b/Core/Portable/parity.sh new file mode 100755 index 0000000..d51344f --- /dev/null +++ b/Core/Portable/parity.sh @@ -0,0 +1,15 @@ +#!/bin/bash +# Old/new comparison on prepared production resources: the Swift quality baseline and the +# ported engine regressions run against the Rust engine. Pass an existing resources +# directory (from prepare-resources.sh) to skip preparation. +set -euo pipefail +cd "$(dirname "$0")/../.." +resources=${1:-} +if [[ -z "$resources" ]]; then + resources=$(bash Core/Portable/prepare-resources.sh | tee /dev/stderr | sed -n 's/^PASS production resources: //p') +fi +[[ -f "$resources/prepared/complete" ]] || { echo "Not a prepared resources directory: $resources" >&2; exit 2; } +# cargo test runs from the manifest directory. +resources=$(cd "$resources" && pwd) +export CARGO_TARGET_DIR="$PWD/build/portable/cargo" +INKFLOW_PORTABLE_RESOURCES="$resources" cargo test --locked --manifest-path Core/Portable/Cargo.toml --test parity -- --nocapture diff --git a/Core/Portable/src/channel.rs b/Core/Portable/src/channel.rs new file mode 100644 index 0000000..096bffc --- /dev/null +++ b/Core/Portable/src/channel.rs @@ -0,0 +1,112 @@ +//! The only host–Lua bridge: one request property that the Lua modules observe +//! synchronously, answered in one result property. Protocol version 1: +//! +//! ```text +//! request = "1\t\t\t..." fields carry no tab, CR or LF +//! result = "1\t\t\n" status: ok, failed, unknown or conflict +//! ``` +pub const VERSION: &str = "1"; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Reply { + pub status: String, + pub body: String, +} + +/// `None` when a field would break the line-oriented framing. +pub fn request(op: &str, fields: &[&str]) -> Option { + if op.is_empty() + || !op.bytes().all(|b| b.is_ascii_lowercase() || b == b'_') + || fields + .iter() + .any(|f| f.bytes().any(|b| matches!(b, b'\t' | b'\n' | b'\r'))) + { + return None; + } + let mut request = format!("{VERSION}\t{op}"); + for field in fields { + request.push('\t'); + request.push_str(field); + } + Some(request) +} + +/// `None` for another protocol version, another operation or a malformed header. +pub fn reply(result: &str, op: &str) -> Option { + let (header, body) = match result.split_once('\n') { + Some((header, body)) => (header, body), + None => (result, ""), + }; + let fields: Vec<_> = header.split('\t').collect(); + if fields.len() != 3 + || fields[0] != VERSION + || fields[1] != op + || fields[2].is_empty() + || !fields[2].bytes().all(|b| b.is_ascii_lowercase()) + { + return None; + } + Some(Reply { + status: fields[2].to_owned(), + body: body.to_owned(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn swift_framing_rules() { + assert_eq!( + request("input_coverage", &["9", "3"]).as_deref(), + Some("1\tinput_coverage\t9\t3") + ); + assert_eq!( + request("learning_invalidate", &[]).as_deref(), + Some("1\tlearning_invalidate") + ); + for (op, fields) in [ + ("", vec!["a"]), + ("Input", vec![]), + ("ai-learning", vec![]), + ("ai_learning", vec!["a\tb"]), + ("ai_learning", vec!["a\nb"]), + ("ai_learning", vec!["a\rb"]), + ] { + assert_eq!(request(op, &fields), None, "{op} {fields:?}"); + } + assert_eq!( + reply("1\tvoice_lexicon\tok\n你好\tni hao\t2\n", "voice_lexicon"), + Some(Reply { + status: "ok".into(), + body: "你好\tni hao\t2\n".into() + }) + ); + assert_eq!( + reply("1\tai_learning\tfailed", "ai_learning").map(|r| r.status), + Some("failed".into()) + ); + assert_eq!( + reply("1\tai_learning\tfailed\n", "ai_learning").map(|r| r.body), + Some(String::new()) + ); + assert_eq!( + reply("1\tinput_coverage\tok\n9,3;0,4,n,1,0,n", "input_coverage").map(|r| r.body), + Some("9,3;0,4,n,1,0,n".into()) + ); + for result in [ + "", + "ok", + "2\tai_learning\tok", + "1\tai_readings\tok", + "1\tai_learning\t", + "1\tai_learning\tOK", + "1\tai_learning\tok\textra", + "1\tai_learning", + "\n1\tai_learning\tok", + ] { + assert_eq!(reply(result, "ai_learning"), None, "{result:?}"); + } + } +} diff --git a/Core/Portable/src/engine.rs b/Core/Portable/src/engine.rs new file mode 100644 index 0000000..81be0ba --- /dev/null +++ b/Core/Portable/src/engine.rs @@ -0,0 +1,1063 @@ +//! Session coordination and InkFlow policy, matching InkFlowRime/Engine.swift. +//! +//! Rime keeps editing, segmentation, lookup, paging and its own learning. This layer owns +//! equivalent-span ordering and selection, digit/arrow key policy, custom phrases, input +//! preferences, candidate counts and safe configuration boundaries. One policy lock +//! serializes every session of an engine; the native engine lock nests inside it. +//! +//! Quality recording stays outside: an optional [`Observer`] receives each mutation with +//! the displayed snapshots before and after, after the policy lock is released, so it can +//! neither change input nor block a key event on storage. +use crate::{ + Error, Result, Runtime, Session, + phrases::{CustomPhrase, PhraseError, phrase_tsv}, + preferences::InputPreferences, + ranking::{ContextRanker, Metadata, parse_metadata}, +}; +use std::{ + collections::BTreeMap, + fs, + ops::Range, + path::{Path, PathBuf}, + sync::{Arc, Mutex, MutexGuard}, +}; +use unicode_segmentation::UnicodeSegmentation; + +pub const SCHEMA: &str = "inkflow_pinyin"; +pub const PHRASE_FILE: &str = "custom_phrase.txt"; +/// Preceding-text graphemes retained for context ranking. +pub const CONTEXT_LIMIT: usize = 16; +const PATCHED_NODES: [&str; 4] = ["menu", "translator", "key_binder", "punctuator"]; + +/// A configuration that could not be applied. The code is stable; frontends localize it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ConfigurationError { + Phrases(PhraseError), + /// custom_phrase.txt could not be written; the file keeps its previous content. + PhraseWrite, + /// This session could not reload the schema after a phrase change; it retries on use. + SessionReload, + /// The compiled spelling profile is missing from the prepared cache. + MissingPrism, + SchemaOpen, + PatchInit, + PatchLoad, + /// The temporary schema configuration could not be restored; restart the engine. + SchemaRestore, + /// The schema did not accept the patched configuration. + SchemaApply, +} + +impl ConfigurationError { + pub fn code(self) -> &'static str { + match self { + ConfigurationError::Phrases(error) => error.code(), + ConfigurationError::PhraseWrite => "phrase-write", + ConfigurationError::SessionReload => "session-reload", + ConfigurationError::MissingPrism => "missing-prism", + ConfigurationError::SchemaOpen => "schema-open", + ConfigurationError::PatchInit => "patch-init", + ConfigurationError::PatchLoad => "patch-load", + ConfigurationError::SchemaRestore => "schema-restore", + ConfigurationError::SchemaApply => "schema-apply", + } + } +} + +#[derive(Debug)] +pub enum EngineError { + Native(Error), + /// A required prepared resource is missing; the path is relative to its root. + MissingResource(String), + Configuration(ConfigurationError), + ContextIndex(String), + Io(std::io::Error), +} + +impl std::fmt::Display for EngineError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + EngineError::Native(error) => write!(f, "native: {error}"), + EngineError::MissingResource(path) => write!(f, "missing resource {path}"), + EngineError::Configuration(error) => write!(f, "configuration: {}", error.code()), + EngineError::ContextIndex(error) => write!(f, "context index: {error}"), + EngineError::Io(error) => write!(f, "io: {error}"), + } + } +} +impl std::error::Error for EngineError {} +impl From for EngineError { + fn from(error: Error) -> Self { + EngineError::Native(error) + } +} +impl From for EngineError { + fn from(error: std::io::Error) -> Self { + EngineError::Io(error) + } +} + +/// Prepared resources and the isolated writable user directory of one engine. +pub struct Configuration { + pub shared: PathBuf, + pub user: PathBuf, + /// Target-native compiled cache. `None` compiles into `user/build` before any session. + pub cache: Option, + /// The prepared `pinyin_simp.context.bin`; `None` keeps Rime's native order. + pub context_index: Option, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Candidate { + pub text: String, + pub comment: String, + /// Position in Rime's current page; quality records keep both ranks. + pub native_index: usize, + /// Bounded ranking evidence for this candidate when ordering applied. + pub evidence: Option, +} + +/// The displayed page in InkFlow order. Offsets count UTF-8 bytes of `preedit`. +#[derive(Debug, Clone)] +pub struct Snapshot { + token: Arc<()>, + pub input: String, + pub preedit: String, + pub caret_bytes: usize, + pub selection_bytes: Range, + pub candidates: Vec, + pub page: i32, + /// Display index of Rime's highlighted candidate. + pub highlighted: usize, + pub last_page: bool, +} + +impl PartialEq for Snapshot { + fn eq(&self, other: &Self) -> bool { + self.input == other.input + && self.preedit == other.preedit + && self.caret_bytes == other.caret_bytes + && self.selection_bytes == other.selection_bytes + && self.candidates == other.candidates + && self.page == other.page + && self.highlighted == other.highlighted + && self.last_page == other.last_page + } +} +impl Eq for Snapshot {} + +impl Snapshot { + /// Once a segment is selected, the immediate prefix is inside the mark. + pub fn has_selected_prefix(&self) -> bool { + self.selection_bytes.start > 0 + } + pub fn texts(&self) -> Vec<&str> { + self.candidates.iter().map(|c| c.text.as_str()).collect() + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Action { + Key { key: i32, modifiers: i32 }, + Select(usize), + Highlight(usize), + Commit, + Clear, + ToggleAscii, +} + +/// One completed mutation of a session, delivered outside the policy lock. +#[derive(Debug, Clone)] +pub struct Mutation { + pub session: u64, + pub action: Action, + pub handled: bool, + pub before: Snapshot, + pub after: Snapshot, +} + +pub type Observer = dyn Fn(&Mutation) + Send + Sync; + +struct Core { + session: Option, + lost: bool, + candidate_count: usize, + requested_count: usize, + requested_input: InputPreferences, + input: Option, + saved_ascii: bool, + buffered_commit: String, + error: Option, + preceding: String, + ordered_content: Option, + order: Vec, + metadata: Option>, + token: Arc<()>, +} + +impl Core { + fn new(session: Session) -> Self { + Self { + session: Some(session), + lost: false, + candidate_count: 5, + requested_count: 5, + requested_input: InputPreferences::default(), + input: None, + saved_ascii: false, + buffered_commit: String::new(), + error: None, + preceding: String::new(), + ordered_content: None, + order: Vec::new(), + metadata: None, + token: Arc::new(()), + } + } + fn available(&self) -> bool { + self.session.is_some() && !self.lost + } + fn reset_ordering(&mut self) { + self.order.clear(); + self.metadata = None; + self.ordered_content = None; + self.token = Arc::new(()); + } + fn detach(&mut self) { + self.session = None; + self.reset_ordering(); + self.preceding.clear(); + } +} + +struct State { + /// Custom phrases are Rime's native custom_phrase.txt in the user directory. The file is + /// written only when settings change; every live session reloads it at the next shared idle. + requested_phrases: Vec, + requested_tsv: String, + file_tsv: String, + file_error: bool, + loaded_tsv: String, + loaded_phrases: Vec, + sessions: BTreeMap, + next: u64, +} + +pub struct Engine { + state: Mutex, + observer: Mutex>>, + runtime: Runtime, + ranker: Option, + user: PathBuf, + compiled: PathBuf, +} + +fn content(raw: &crate::Snapshot) -> crate::Snapshot { + let mut content = raw.clone(); + content.highlighted = 0; + content +} + +impl Engine { + /// Initialization, compilation and the first phrase-file sync are setup work. + pub fn new(configuration: Configuration) -> std::result::Result, EngineError> { + fs::create_dir_all(&configuration.user)?; + let ranker = match &configuration.context_index { + Some(path) => Some( + ContextRanker::from_index(fs::read(path)?).map_err(EngineError::ContextIndex)?, + ), + None => None, + }; + let compiled = configuration + .cache + .clone() + .unwrap_or_else(|| configuration.user.join("build")); + let mut runtime = + Runtime::with_cache(&configuration.shared, &configuration.user, &compiled)?; + if configuration.cache.is_none() { + runtime.prepare()?; + } + for file in + ["pinyin_simp.table.bin", "pinyin_simp.prism.bin"] + .into_iter() + .map(str::to_owned) + .chain((0..32).map(|mask| { + format!("{}.prism.bin", crate::preferences::spelling_profile(mask)) + })) + { + if !compiled.join(&file).is_file() { + return Err(EngineError::MissingResource(file)); + } + } + let file = configuration.user.join(PHRASE_FILE); + let file_tsv = fs::read_to_string(&file).unwrap_or_else(|_| phrase_tsv(&[])); + let mut state = State { + requested_phrases: Vec::new(), + requested_tsv: phrase_tsv(&[]), + file_tsv, + file_error: false, + loaded_tsv: String::new(), + loaded_phrases: Vec::new(), + sessions: BTreeMap::new(), + next: 1, + }; + // No session exists yet: whatever the file holds now is what they all load. + if let Some(error) = sync_phrase_file(&mut state, &file) { + return Err(EngineError::Configuration(error)); + } + state.loaded_tsv = state.file_tsv.clone(); + state.loaded_phrases = state.requested_phrases.clone(); + Ok(Arc::new(Self { + state: Mutex::new(state), + observer: Mutex::new(None), + runtime, + ranker, + user: configuration.user, + compiled, + })) + } + + /// Replace the observer. It runs after each mutation, outside every engine lock. + pub fn set_observer(&self, observer: Option>) { + *self.observer.lock().unwrap_or_else(|e| e.into_inner()) = observer; + } + + pub fn context_ranking_ready(&self) -> bool { + self.ranker.is_some() + } + + fn lock(&self) -> Result> { + self.state.lock().map_err(|_| Error::Poisoned) + } + + pub fn session(self: &Arc) -> Result { + let mut state = self.lock()?; + let id = state.next; + state.next += 1; + let core = Core::new(self.runtime.session(SCHEMA)?); + state.sessions.insert(id, core); + let session = InputSession { + engine: self.clone(), + id, + }; + if let Err(error) = self.restore(&mut state, id) { + state.sessions.remove(&id); + return Err(error); + } + Ok(session) + } + + fn restore(&self, state: &mut State, id: u64) -> Result<()> { + let core = state.sessions.get_mut(&id).unwrap(); + core.lost = false; + if core.session.is_none() { + core.session = Some(self.runtime.session(SCHEMA)?); + } + core.candidate_count = 5; + core.input = None; + core.error = None; + core.reset_ordering(); + core.preceding.clear(); + self.apply_configuration_if_idle(state, id)?; + let core = state.sessions.get_mut(&id).unwrap(); + if let Some(error) = core.error { + core.detach(); + return Err(Error::Configuration(error.code())); + } + self.apply_runtime_options(core) + } + + /// A session that could not be recreated during a phrase reload is retried on its next use. + fn recover(&self, state: &mut State, id: u64) -> Result<()> { + if !state.sessions[&id].lost { + return Ok(()); + } + if self.restore(state, id).is_err() { + let core = state.sessions.get_mut(&id).unwrap(); + core.detach(); + core.lost = true; + } + Ok(()) + } + + fn raw(core: &mut Core) -> Result { + match &mut core.session { + Some(session) if !core.lost => session.snapshot(), + _ => Ok(crate::Snapshot { + token: Arc::new(()), + preedit: String::new(), + caret_bytes: 0, + selection_bytes: 0..0, + candidates: Vec::new(), + page: 0, + highlighted: 0, + last_page: true, + }), + } + } + + fn display(core: &mut Core) -> Result { + let raw = Self::raw(core)?; + let input = match &mut core.session { + Some(session) if !core.lost => session.input()?, + _ => String::new(), + }; + let order: Vec = if core.order.len() == raw.candidates.len() { + core.order.clone() + } else { + (0..raw.candidates.len()).collect() + }; + let candidates = order + .iter() + .map(|&native| Candidate { + text: raw.candidates[native].text.clone(), + comment: raw.candidates[native].comment.clone(), + native_index: native, + evidence: core + .metadata + .as_ref() + .and_then(|rows| rows.get(native).cloned()), + }) + .collect(); + let highlighted = usize::try_from(raw.highlighted) + .ok() + .and_then(|native| order.iter().position(|&i| i == native)) + .unwrap_or(0); + Ok(Snapshot { + token: core.token.clone(), + input, + preedit: raw.preedit, + caret_bytes: raw.caret_bytes, + selection_bytes: raw.selection_bytes, + candidates, + page: raw.page, + highlighted, + last_page: raw.last_page, + }) + } + + fn read_commit(core: &mut Core) -> Result { + match &mut core.session { + Some(session) if !core.lost => Ok(session.take_commit()?.unwrap_or_default()), + _ => Ok(String::new()), + } + } + + fn all_sessions_idle(state: &mut State) -> Result { + for core in state.sessions.values_mut() { + let commit = Self::read_commit(core)?; + core.buffered_commit.push_str(&commit); + if !Self::raw(core)?.preedit.is_empty() || !core.buffered_commit.is_empty() { + return Ok(false); + } + } + Ok(true) + } + + fn apply_configuration_if_idle(&self, state: &mut State, id: u64) -> Result<()> { + self.recover(state, id)?; + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(()); + } + if state.loaded_tsv != state.file_tsv { + // Rime shares one loaded custom_phrase table among every session that holds it, so + // the phrases change for all sessions at once. A composing session keeps its + // snapshot and retries on its next idle key; its count still applies below. + let core = state.sessions.get_mut(&id).unwrap(); + if Self::raw(core)?.preedit.is_empty() && Self::all_sessions_idle(state)? { + return self.reload_sessions(state, id); + } + } else if state.requested_tsv == state.loaded_tsv { + state.loaded_phrases = state.requested_phrases.clone(); + } + let core = state.sessions.get_mut(&id).unwrap(); + if core.candidate_count == core.requested_count + && core.input.as_ref() == Some(&core.requested_input) + { + if Self::raw(core)?.preedit.is_empty() { + self.apply_runtime_options(core)?; + } + return Ok(()); + } + if !Self::raw(core)?.preedit.is_empty() { + return Ok(()); + } + match self.recreate_schema(core)? { + Ok(()) => core.error = None, + Err(error) => core.error = Some(error), + } + Ok(()) + } + + /// The one deploy after a phrase change: release every session's dictionary handles first, + /// so the next session reads the current file instead of the cached table, then recreate + /// each session and reapply its own settings. All sessions are idle, so nothing is lost. + fn reload_sessions(&self, state: &mut State, caller: u64) -> Result<()> { + let mut ids: Vec = state + .sessions + .iter() + .filter(|(_, core)| core.available()) + .map(|(id, _)| *id) + .collect(); + if !ids.contains(&caller) { + ids.push(caller); + } + for id in &ids { + state.sessions.get_mut(id).unwrap().session = None; + } + state.loaded_tsv = state.file_tsv.clone(); + state.loaded_phrases = state.requested_phrases.clone(); + for id in ids { + let core = state.sessions.get_mut(&id).unwrap(); + match self.runtime.session(SCHEMA) { + Ok(session) => core.session = Some(session), + Err(_) => { + // Siblings already loaded the current file; this session retries alone on its next use. + core.detach(); + core.lost = true; + core.error = Some(ConfigurationError::SessionReload); + continue; + } + } + core.candidate_count = 5; + core.input = None; + core.reset_ordering(); + self.apply_configuration_if_idle(state, id)?; + } + Ok(()) + } + + fn recreate_schema( + &self, + core: &mut Core, + ) -> Result> { + let profile = core.requested_input.spelling_profile(); + if !self.compiled.join(format!("{profile}.prism.bin")).is_file() { + return Ok(Err(ConfigurationError::MissingPrism)); + } + let yaml = format!( + "{}\nmenu:\n page_size: {}\ntranslator:\n dictionary: pinyin_simp\n prism: {profile}\n preedit_format: ['xform/([nl])v/$1ü/', 'xform/([jqxy])v/$1u/']", + core.requested_input.schema_patch(), + core.requested_count + ); + // select_schema resets the commit buffer as well as the schema. Preserve completed text + // even if settings arrive before the frontend has drained the previous key's commit. + let commit = Self::read_commit(core)?; + core.buffered_commit.push_str(&commit); + let session = core.session.as_mut().unwrap(); + let outcome = match session.apply_schema_patch(SCHEMA, &yaml, &PATCHED_NODES)? { + Ok(outcome) => outcome, + Err(crate::PatchFailure::SchemaOpen) => return Ok(Err(ConfigurationError::SchemaOpen)), + Err(crate::PatchFailure::PatchInit) => return Ok(Err(ConfigurationError::PatchInit)), + Err(crate::PatchFailure::PatchLoad) => return Ok(Err(ConfigurationError::PatchLoad)), + }; + let loaded = outcome.patched && outcome.selected; + if loaded { + core.candidate_count = core.requested_count; + core.input = Some(core.requested_input.clone()); + } + self.apply_runtime_options(core)?; + if !outcome.restored { + return Ok(Err(ConfigurationError::SchemaRestore)); + } + if !loaded { + return Ok(Err(ConfigurationError::SchemaApply)); + } + Ok(Ok(())) + } + + fn apply_runtime_options(&self, core: &mut Core) -> Result<()> { + let Some(input) = core.input.clone() else { + return Ok(()); + }; + let ascii = core.saved_ascii; + let Some(session) = core.session.as_mut().filter(|_| !core.lost) else { + return Ok(()); + }; + use crate::preferences::InputOption; + session.set_option("ascii_mode", ascii)?; + session.set_option( + "ascii_punct", + ascii || input.get(InputOption::EnglishPunctuation), + )?; + session.set_option("emoji_suggestion", input.get(InputOption::Emoji))?; + session.set_option("traditional", input.get(InputOption::Traditional)) + } + + fn input_ranking_metadata( + core: &mut Core, + page: i32, + count: usize, + input_length: usize, + ) -> Result>> { + let Ok(page) = usize::try_from(page) else { + return Ok(None); + }; + let Some(offset) = page.checked_mul(core.candidate_count) else { + return Ok(None); + }; + if !(1..=9).contains(&count) { + return Ok(None); + } + let session = core.session.as_mut().unwrap(); + let reply = session.call( + "input_coverage", + &[&offset.to_string(), &count.to_string()], + 512, + )?; + Ok(reply + .filter(|reply| reply.status == "ok") + .and_then(|reply| parse_metadata(&reply.body, offset, count, input_length))) + } + + fn update_ordering(&self, state: &mut State, id: u64, force: bool) -> Result<()> { + let loaded_phrases = &state.loaded_phrases; + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(()); + } + let raw = Self::raw(core)?; + let current = content(&raw); + if !force && core.ordered_content.as_ref() == Some(¤t) { + return Ok(()); + } + // Explicit custom codes keep the user's ordered phrases ahead of ordinary words. + // Use the applied snapshot so deferred settings cannot change a live composition. + let input = core.session.as_mut().unwrap().input()?; + let has_custom_code = loaded_phrases.iter().any(|phrase| phrase.code == input); + // Once a segment is selected, the immediate prefix is inside the mark. + // Leave these remaining candidates to Rime instead of applying older document text. + core.order = (0..raw.candidates.len()).collect(); + core.metadata = None; + if raw.selection_bytes.start == 0 && !has_custom_code { + let metadata = + Self::input_ranking_metadata(core, raw.page, raw.candidates.len(), input.len())?; + if let Some(ranker) = &self.ranker { + let texts: Vec = raw.candidates.iter().map(|c| c.text.clone()).collect(); + core.order = ranker.order(&texts, &core.preceding, metadata.as_deref()); + } + core.metadata = metadata; + } + if let Some(&first) = core.order.first() { + core.session.as_mut().unwrap().highlight(first)?; + } + core.ordered_content = Some(content(&Self::raw(core)?)); + if raw.preedit.is_empty() { + core.preceding.clear(); + } + core.token = Arc::new(()); + Ok(()) + } + + fn perform_key(&self, state: &mut State, id: u64, key: i32, modifiers: i32) -> Result { + let core = state.sessions.get_mut(&id).unwrap(); + let displayed = Self::display(core)?; + if modifiers == 0 && (49..=57).contains(&key) { + let index = (key - 49) as usize; + if index < displayed.candidates.len() { + self.select_display(state, id, index)?; + return Ok(true); + } + } + if modifiers == 0 && (key == 0xff52 || key == 0xff54) && !displayed.candidates.is_empty() { + let delta = if key == 0xff54 { 1 } else { -1 }; + return self.move_highlight(state, id, delta, key); + } + let handled = core.session.as_mut().unwrap().process_key(key, modifiers)?; + self.update_ordering(state, id, false)?; + Ok(handled) + } + + fn select_display(&self, state: &mut State, id: u64, index: usize) -> Result { + let core = state.sessions.get_mut(&id).unwrap(); + let Some(&native) = core.order.get(index) else { + return Ok(false); + }; + let handled = core.session.as_mut().unwrap().select_native(native)?; + self.update_ordering(state, id, false)?; + Ok(handled) + } + + fn highlight_display(core: &mut Core, index: usize) -> Result<()> { + let Some(&native) = core.order.get(index) else { + return Ok(()); + }; + core.session.as_mut().unwrap().highlight(native)?; + // Highlighting a partial choice can change Rime's preedit and cursor. + // Record the new content without overriding explicit user navigation. + core.ordered_content = Some(content(&Self::raw(core)?)); + Ok(()) + } + + fn move_highlight(&self, state: &mut State, id: u64, delta: isize, key: i32) -> Result { + let core = state.sessions.get_mut(&id).unwrap(); + let before = Self::display(core)?; + let next = before.highlighted as isize + delta; + if next >= 0 && (next as usize) < core.order.len() { + Self::highlight_display(core, next as usize)?; + return Ok(true); + } + // Let Rime decide whether another page exists, starting at its native edge. + let edge = if delta > 0 { core.order.len() - 1 } else { 0 }; + let session = core.session.as_mut().unwrap(); + session.highlight(edge)?; + let handled = session.process_key(key, 0)?; + self.update_ordering(state, id, false)?; + let core = state.sessions.get_mut(&id).unwrap(); + if Self::display(core)?.page != before.page { + if delta < 0 { + let last = core.order.len().saturating_sub(1); + Self::highlight_display(core, last)?; + } + } else if !core.order.is_empty() { + Self::highlight_display(core, before.highlighted)?; + } + Ok(handled) + } +} + +/// Returns the failure; the file keeps its previous content on failure. +fn sync_phrase_file(state: &mut State, file: &Path) -> Option { + if state.requested_tsv == state.file_tsv { + return None; + } + if state.file_error { + return Some(ConfigurationError::PhraseWrite); + } + if write_private(file, state.requested_tsv.as_bytes()).is_err() { + state.file_error = true; + return Some(ConfigurationError::PhraseWrite); + } + state.file_tsv = state.requested_tsv.clone(); + None +} + +fn write_private(file: &Path, bytes: &[u8]) -> std::io::Result<()> { + let temporary = file.with_extension("txt.tmp"); + fs::write(&temporary, bytes)?; + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + fs::set_permissions(&temporary, fs::Permissions::from_mode(0o600))?; + } + fs::rename(&temporary, file) +} + +/// One input context's session. Every method is synchronous and thread-agnostic. +pub struct InputSession { + engine: Arc, + id: u64, +} + +impl InputSession { + pub fn id(&self) -> u64 { + self.id + } + pub fn engine(&self) -> &Arc { + &self.engine + } + + fn mutate( + &self, + action: Action, + handled: impl Fn(&T) -> bool, + body: impl FnOnce(&mut State, u64) -> Result, + ) -> Result { + let observer = self + .engine + .observer + .lock() + .map_err(|_| Error::Poisoned)? + .clone(); + let mut state = self.engine.lock()?; + let before = match &observer { + Some(_) => Some(Engine::display(state.sessions.get_mut(&self.id).unwrap())?), + None => None, + }; + let result = body(&mut state, self.id)?; + let Some(observer) = observer else { + return Ok(result); + }; + let after = Engine::display(state.sessions.get_mut(&self.id).unwrap())?; + drop(state); + observer(&Mutation { + session: self.id, + action, + handled: handled(&result), + before: before.unwrap(), + after, + }); + Ok(result) + } + + pub fn available(&self) -> bool { + self.engine + .lock() + .map(|state| state.sessions[&self.id].available()) + .unwrap_or(false) + } + + /// Keys and modifiers use Rime/X11 keysyms and masks. Digits select displayed candidates; + /// Up/Down move the displayed highlight; everything else goes to Rime. + pub fn key(&self, key: i32, modifiers: i32) -> Result { + let engine = self.engine.clone(); + self.mutate( + Action::Key { key, modifiers }, + |h| *h, + |state, id| { + engine.recover(state, id)?; + if !state.sessions[&id].available() { + return Ok(false); + } + // Apply existing idle configuration before capturing the values this key actually uses. + engine.apply_configuration_if_idle(state, id)?; + engine.perform_key(state, id, key, modifiers) + }, + ) + } + + /// Select a displayed candidate of `snapshot`, which must be this session's latest. + pub fn select(&self, snapshot: &Snapshot, index: usize) -> Result { + let engine = self.engine.clone(); + let token = snapshot.token.clone(); + self.mutate( + Action::Select(index), + |h| *h, + |state, id| { + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(false); + } + if !Arc::ptr_eq(&core.token, &token) { + return Err(Error::StaleSnapshot); + } + if index >= core.order.len() { + return Err(Error::InvalidCandidate); + } + engine.select_display(state, id, index) + }, + ) + } + + pub fn highlight(&self, snapshot: &Snapshot, index: usize) -> Result<()> { + let token = snapshot.token.clone(); + self.mutate( + Action::Highlight(index), + |_| true, + |state, id| { + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(()); + } + if !Arc::ptr_eq(&core.token, &token) { + return Err(Error::StaleSnapshot); + } + if index >= core.order.len() { + return Err(Error::InvalidCandidate); + } + Engine::highlight_display(core, index) + }, + ) + } + + /// Commit the composition as Rime would for a commit key. + pub fn commit(&self) -> Result { + let engine = self.engine.clone(); + self.mutate( + Action::Commit, + |h| *h, + |state, id| { + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(false); + } + let handled = core.session.as_mut().unwrap().commit_composition()?; + engine.update_ordering(state, id, false)?; + Ok(handled) + }, + ) + } + + pub fn clear(&self) -> Result<()> { + let engine = self.engine.clone(); + self.mutate( + Action::Clear, + |_| true, + |state, id| { + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(()); + } + core.session.as_mut().unwrap().clear()?; + engine.update_ordering(state, id, false) + }, + ) + } + + /// Drain completed text once, including text preserved across a settings reload. + pub fn take_commit(&self) -> Result { + let mut state = self.engine.lock()?; + let core = state.sessions.get_mut(&self.id).unwrap(); + let commit = Engine::read_commit(core)?; + let text = std::mem::take(&mut core.buffered_commit) + &commit; + Ok(text) + } + + /// Read-only; repeated reads return the same selection identity until a mutation. + pub fn snapshot(&self) -> Result { + let mut state = self.engine.lock()?; + Engine::display(state.sessions.get_mut(&self.id).unwrap()) + } + + /// Bounded document text before the caret; applied to the live composition. + pub fn set_preceding_text(&self, text: &str) -> Result<()> { + let graphemes: Vec<&str> = text.graphemes(true).collect(); + let start = graphemes.len().saturating_sub(CONTEXT_LIMIT); + let prefix: String = graphemes[start..].concat(); + let mut state = self.engine.lock()?; + let core = state.sessions.get_mut(&self.id).unwrap(); + if prefix == core.preceding { + return Ok(()); + } + core.preceding = prefix; + if Engine::raw(core)?.preedit.is_empty() { + return Ok(()); + } + self.engine.update_ordering(&mut state, self.id, true) + } + + pub fn set_candidate_count(&self, count: usize) -> Result<()> { + let (phrases, input) = { + let state = self.engine.lock()?; + ( + state.requested_phrases.clone(), + state.sessions[&self.id].requested_input.clone(), + ) + }; + self.set_configuration(count, &phrases, Some(&input)) + } + + /// Settings apply at this session's next idle boundary; phrases apply to every idle session. + pub fn set_configuration( + &self, + candidate_count: usize, + phrases: &[CustomPhrase], + input: Option<&InputPreferences>, + ) -> Result<()> { + let mut state = self.engine.lock()?; + if let Err(error) = CustomPhrase::validate(phrases) { + state.sessions.get_mut(&self.id).unwrap().error = + Some(ConfigurationError::Phrases(error)); + return Ok(()); + } + if phrases != state.requested_phrases.as_slice() { + state.requested_phrases = phrases.to_vec(); + state.requested_tsv = phrase_tsv(phrases); + state.file_error = false; + } + // The settings path writes the file; key events only ever reload it. Frontends push + // the same settings on every refresh, so a failed write is retried only when the + // phrases change or the engine restarts, and the whole configuration waits. + if let Some(error) = sync_phrase_file(&mut state, &self.engine.user.join(PHRASE_FILE)) { + state.sessions.get_mut(&self.id).unwrap().error = Some(error); + return Ok(()); + } + let core = state.sessions.get_mut(&self.id).unwrap(); + core.requested_count = if (3..=9).contains(&candidate_count) { + candidate_count + } else { + 5 + }; + if let Some(input) = input { + core.requested_input = input.clone(); + } + core.error = None; + self.engine.apply_configuration_if_idle(&mut state, self.id) + } + + pub fn configuration_error(&self) -> Option { + self.engine + .lock() + .ok() + .and_then(|state| state.sessions[&self.id].error) + } + + pub fn candidate_count(&self) -> usize { + self.engine + .lock() + .map(|state| state.sessions[&self.id].candidate_count) + .unwrap_or(5) + } + + /// The preferences applied to this session, once a composition boundary accepted them. + pub fn input_preferences(&self) -> Option { + self.engine + .lock() + .ok() + .and_then(|state| state.sessions[&self.id].input.clone()) + } + + pub fn requested_ascii_mode(&self) -> bool { + self.engine + .lock() + .map(|state| state.sessions[&self.id].saved_ascii) + .unwrap_or(false) + } + + /// Rime's live ASCII switch, or the requested value while the session is unavailable. + pub fn ascii_mode(&self) -> Result { + let mut state = self.engine.lock()?; + let core = state.sessions.get_mut(&self.id).unwrap(); + match core.session.as_mut().filter(|_| !core.lost) { + Some(session) => session.option("ascii_mode"), + None => Ok(core.saved_ascii), + } + } + + /// Requested ASCII mode; it starts after the current composition finishes. + pub fn set_ascii_mode(&self, value: bool) -> Result<()> { + let mut state = self.engine.lock()?; + state.sessions.get_mut(&self.id).unwrap().saved_ascii = value; + self.engine.apply_configuration_if_idle(&mut state, self.id) + } + + pub fn toggle_ascii_mode(&self) -> Result { + let engine = self.engine.clone(); + self.mutate( + Action::ToggleAscii, + |h| *h, + |state, id| { + let core = state.sessions.get_mut(&id).unwrap(); + if !core.available() { + return Ok(false); + } + core.saved_ascii = !core.saved_ascii; + engine.apply_configuration_if_idle(state, id)?; + Ok(true) + }, + ) + } + + /// Bounded Lua channel request for optional features; never from a key callback. + pub fn call( + &self, + op: &str, + fields: &[&str], + capacity: usize, + ) -> Result> { + let mut state = self.engine.lock()?; + let core = state.sessions.get_mut(&self.id).unwrap(); + match core.session.as_mut().filter(|_| !core.lost) { + Some(session) => session.call(op, fields, capacity), + None => Ok(None), + } + } +} + +impl Drop for InputSession { + fn drop(&mut self) { + let mut state = self.engine.state.lock().unwrap_or_else(|e| e.into_inner()); + state.sessions.remove(&self.id); + } +} diff --git a/Core/Portable/src/ffi.rs b/Core/Portable/src/ffi.rs index f21bbd5..2099857 100644 --- a/Core/Portable/src/ffi.rs +++ b/Core/Portable/src/ffi.rs @@ -58,4 +58,23 @@ unsafe extern "C" { pub fn ifp_snapshot_free(snapshot: *mut Snapshot); pub fn ifp_take_commit(session: usize, commit: *mut *mut c_char) -> c_int; pub fn ifp_string_free(string: *mut c_char); + pub fn ifp_get_input(session: usize, input: *mut *mut c_char) -> c_int; + pub fn ifp_highlight_candidate(session: usize, index: usize, handled: *mut c_int) -> c_int; + pub fn ifp_commit_composition(session: usize, handled: *mut c_int) -> c_int; + pub fn ifp_set_option(session: usize, option: *const c_char, value: c_int) -> c_int; + pub fn ifp_get_option(session: usize, option: *const c_char, value: *mut c_int) -> c_int; + pub fn ifp_call( + session: usize, + request: *const c_char, + capacity: usize, + result: *mut *mut c_char, + ) -> c_int; + pub fn ifp_apply_schema_patch( + session: usize, + schema: *const c_char, + yaml: *const c_char, + paths: *const *const c_char, + count: usize, + outcome: *mut c_int, + ) -> c_int; } diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index 2ccbe33..775ae4f 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -1,5 +1,10 @@ -//! Minimal synchronous desktop Rime host. No frontend or InkFlow ranking policy. +//! Synchronous desktop Rime host: the native `Runtime`/`Session` layer plus the +//! InkFlow session policy in [`engine`]. +pub mod channel; +pub mod engine; mod ffi; +pub mod phrases; +pub mod preferences; pub mod ranking; use std::{ @@ -19,6 +24,8 @@ pub enum Error { InvalidCandidate, Poisoned, Native(i32), + /// A session could not apply its configuration; the code names the failure. + Configuration(&'static str), } impl std::fmt::Display for Error { @@ -247,6 +254,140 @@ impl Session { value.map(Some) } } +/// Outcome of replacing schema nodes for one session. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct PatchOutcome { + pub patched: bool, + pub selected: bool, + pub restored: bool, +} + +/// Why a schema patch could not start; nothing was changed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum PatchFailure { + SchemaOpen, + PatchInit, + PatchLoad, +} + +impl Session { + fn owned_string(raw: *mut c_char) -> Result { + if raw.is_null() { + return Ok(String::new()); + } + let value = unsafe { copy(raw) }; + unsafe { ffi::ifp_string_free(raw) }; + value + } + + /// Rime's raw composition input. + pub(crate) fn input(&mut self) -> Result { + let _lock = lock()?; + let mut raw = std::ptr::null_mut(); + check(unsafe { ffi::ifp_get_input(self.id, &mut raw) })?; + Self::owned_string(raw) + } + + pub(crate) fn highlight(&mut self, index: usize) -> Result { + let _lock = lock()?; + self.published = None; + let mut handled = 0; + check(unsafe { ffi::ifp_highlight_candidate(self.id, index, &mut handled) })?; + Ok(handled != 0) + } + + /// Select a native index without snapshot identity; the policy layer validates its own. + pub(crate) fn select_native(&mut self, index: usize) -> Result { + let _lock = lock()?; + self.published = None; + Ok(unsafe { ffi::ifp_select_candidate(self.id, index) } == 0) + } + + pub(crate) fn commit_composition(&mut self) -> Result { + let _lock = lock()?; + self.published = None; + let mut handled = 0; + check(unsafe { ffi::ifp_commit_composition(self.id, &mut handled) })?; + Ok(handled != 0) + } + + pub(crate) fn set_option(&mut self, option: &str, value: bool) -> Result<()> { + let option = string(option)?; + let _lock = lock()?; + check(unsafe { ffi::ifp_set_option(self.id, option.as_ptr(), i32::from(value)) }) + } + + pub(crate) fn option(&mut self, option: &str) -> Result { + let option = string(option)?; + let _lock = lock()?; + let mut value = 0; + check(unsafe { ffi::ifp_get_option(self.id, option.as_ptr(), &mut value) })?; + Ok(value != 0) + } + + /// One synchronous round trip through this session's Lua modules. `None` when no module + /// answers, the reply does not fit `capacity` bytes, or the request cannot be framed. + pub(crate) fn call( + &mut self, + op: &str, + fields: &[&str], + capacity: usize, + ) -> Result> { + let Some(request) = channel::request(op, fields) else { + return Ok(None); + }; + let request = string(&request)?; + let _lock = lock()?; + let mut raw = std::ptr::null_mut(); + check(unsafe { ffi::ifp_call(self.id, request.as_ptr(), capacity, &mut raw) })?; + if raw.is_null() { + return Ok(None); + } + Ok(channel::reply(&Self::owned_string(raw)?, op)) + } + + pub(crate) fn apply_schema_patch( + &mut self, + schema: &str, + yaml: &str, + paths: &[&str], + ) -> Result> { + let schema = string(schema)?; + let yaml = string(yaml)?; + let paths = paths + .iter() + .map(|p| string(p)) + .collect::>>()?; + let pointers: Vec<_> = paths.iter().map(|p| p.as_ptr()).collect(); + let _lock = lock()?; + self.published = None; + let mut outcome = 0; + let status = unsafe { + ffi::ifp_apply_schema_patch( + self.id, + schema.as_ptr(), + yaml.as_ptr(), + pointers.as_ptr(), + pointers.len(), + &mut outcome, + ) + }; + if status == -1 { + return Ok(Err(match outcome { + 1 => PatchFailure::SchemaOpen, + 2 => PatchFailure::PatchInit, + _ => PatchFailure::PatchLoad, + })); + } + check(status)?; + Ok(Ok(PatchOutcome { + patched: outcome & 1 != 0, + selected: outcome & 2 != 0, + restored: outcome & 4 != 0, + })) + } +} + impl Drop for Session { fn drop(&mut self) { let _lock = ENGINE.lock().unwrap_or_else(|e| e.into_inner()); diff --git a/Core/Portable/src/phrases.rs b/Core/Portable/src/phrases.rs new file mode 100644 index 0000000..f53e667 --- /dev/null +++ b/Core/Portable/src/phrases.rs @@ -0,0 +1,143 @@ +//! Custom phrases, matching InkFlowDomain/CustomPhrase.swift. Validation failures carry a +//! stable code; frontends own the localized message. +use std::collections::HashSet; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct CustomPhrase { + /// Opaque frontend identity (a UUID on macOS); unique within one configuration. + pub id: String, + pub code: String, + pub text: String, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum PhraseError { + /// The code must be non-empty ASCII letters a–z. + InvalidCode, + /// The text contains a newline, tab or other control character. + ControlCharacter, + EmptyText, + /// A phrase differs from its validated form or repeats an identity. + InvalidData, + /// The same code and text appear twice. + Duplicate, +} + +impl PhraseError { + pub fn code(self) -> &'static str { + match self { + PhraseError::InvalidCode => "invalid-code", + PhraseError::ControlCharacter => "control-character", + PhraseError::EmptyText => "empty-text", + PhraseError::InvalidData => "invalid-data", + PhraseError::Duplicate => "duplicate", + } + } +} + +impl CustomPhrase { + pub fn validated(id: &str, code: &str, text: &str) -> Result { + let code = code.trim().to_ascii_lowercase(); + if code.is_empty() || !code.bytes().all(|b| b.is_ascii_lowercase()) { + return Err(PhraseError::InvalidCode); + } + if text + .chars() + .any(|c| c.is_control() || matches!(c, '\u{2028}' | '\u{2029}')) + { + return Err(PhraseError::ControlCharacter); + } + let text = text.trim(); + if text.is_empty() { + return Err(PhraseError::EmptyText); + } + Ok(Self { + id: id.to_owned(), + code, + text: text.to_owned(), + }) + } + + pub fn validate(phrases: &[CustomPhrase]) -> Result<(), PhraseError> { + let mut ids = HashSet::new(); + let mut pairs = HashSet::new(); + for phrase in phrases { + let valid = Self::validated(&phrase.id, &phrase.code, &phrase.text) + .map_err(|_| PhraseError::InvalidData)?; + if valid != *phrase || !ids.insert(&phrase.id) { + return Err(PhraseError::InvalidData); + } + if !pairs.insert((&phrase.code, &phrase.text)) { + return Err(PhraseError::Duplicate); + } + } + Ok(()) + } +} + +/// Rime's custom_phrase.txt: text, code, weight. Earlier phrases outrank later ones. +pub fn phrase_tsv(phrases: &[CustomPhrase]) -> String { + // librime's TSV reader otherwise treats phrases beginning with '#' as comments. + let mut tsv = String::from("# no comment\n"); + for (index, phrase) in phrases.iter().enumerate() { + tsv.push_str(&format!( + "{}\t{}\t{}\n", + phrase.text, + phrase.code, + phrases.len() - index + )); + } + tsv +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn swift_validation_rules() { + let phrase = CustomPhrase::validated("a", " Dz ", " 地址 ").unwrap(); + assert_eq!((phrase.code.as_str(), phrase.text.as_str()), ("dz", "地址")); + assert_eq!( + CustomPhrase::validated("a", "", "x"), + Err(PhraseError::InvalidCode) + ); + assert_eq!( + CustomPhrase::validated("a", "x\ty", "x"), + Err(PhraseError::InvalidCode) + ); + assert_eq!( + CustomPhrase::validated("a", "x", "a\tb"), + Err(PhraseError::ControlCharacter) + ); + assert_eq!( + CustomPhrase::validated("a", "x", " "), + Err(PhraseError::EmptyText) + ); + let raw = CustomPhrase { + id: "a".into(), + code: "x\ty".into(), + text: "invalid".into(), + }; + assert_eq!( + CustomPhrase::validate(&[raw]), + Err(PhraseError::InvalidData) + ); + let twice = CustomPhrase::validated("b", "dz", "地址").unwrap(); + assert_eq!( + CustomPhrase::validate(&[phrase.clone(), twice]), + Err(PhraseError::Duplicate) + ); + let same_id = CustomPhrase::validated("a", "dz", "其他").unwrap(); + assert_eq!( + CustomPhrase::validate(&[phrase.clone(), same_id]), + Err(PhraseError::InvalidData) + ); + assert_eq!(phrase_tsv(&[]), "# no comment\n"); + let second = CustomPhrase::validated("c", "dz", "# no comment").unwrap(); + assert_eq!( + phrase_tsv(&[phrase, second]), + "# no comment\n地址\tdz\t2\n# no comment\tdz\t1\n" + ); + } +} diff --git a/Core/Portable/src/preferences.rs b/Core/Portable/src/preferences.rs new file mode 100644 index 0000000..a77df99 --- /dev/null +++ b/Core/Portable/src/preferences.rs @@ -0,0 +1,183 @@ +//! Input preferences, matching InkFlowDomain/InputPreferences.swift. + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub enum InputOption { + Abbreviation, + TypoTolerance, + FuzzyZ, + FuzzyC, + FuzzyS, + Emoji, + BracketPaging, + MinusEqualPaging, + EnglishPunctuation, + CornerQuotes, + MiddleDot, + FullwidthPipe, + IdeographicComma, + Traditional, +} + +impl InputOption { + pub const ALL: [InputOption; 14] = [ + InputOption::Abbreviation, + InputOption::TypoTolerance, + InputOption::FuzzyZ, + InputOption::FuzzyC, + InputOption::FuzzyS, + InputOption::Emoji, + InputOption::BracketPaging, + InputOption::MinusEqualPaging, + InputOption::EnglishPunctuation, + InputOption::CornerQuotes, + InputOption::MiddleDot, + InputOption::FullwidthPipe, + InputOption::IdeographicComma, + InputOption::Traditional, + ]; + + /// The Swift raw value; the quality configuration records options by this name. + pub fn name(self) -> &'static str { + match self { + InputOption::Abbreviation => "abbreviation", + InputOption::TypoTolerance => "typoTolerance", + InputOption::FuzzyZ => "fuzzyZ", + InputOption::FuzzyC => "fuzzyC", + InputOption::FuzzyS => "fuzzyS", + InputOption::Emoji => "emoji", + InputOption::BracketPaging => "bracketPaging", + InputOption::MinusEqualPaging => "minusEqualPaging", + InputOption::EnglishPunctuation => "englishPunctuation", + InputOption::CornerQuotes => "cornerQuotes", + InputOption::MiddleDot => "middleDot", + InputOption::FullwidthPipe => "fullwidthPipe", + InputOption::IdeographicComma => "ideographicComma", + InputOption::Traditional => "traditional", + } + } + + pub fn default_value(self) -> bool { + !matches!( + self, + InputOption::FuzzyZ + | InputOption::FuzzyC + | InputOption::FuzzyS + | InputOption::EnglishPunctuation + | InputOption::Traditional + ) + } +} + +/// A value snapshot belongs to one composition, even while persisted preferences change. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct InputPreferences { + values: [bool; 14], +} + +impl Default for InputPreferences { + fn default() -> Self { + Self { + values: InputOption::ALL.map(InputOption::default_value), + } + } +} + +impl InputPreferences { + fn slot(option: InputOption) -> usize { + InputOption::ALL.iter().position(|o| *o == option).unwrap() + } + + pub fn get(&self, option: InputOption) -> bool { + self.values[Self::slot(option)] + } + + pub fn with(mut self, option: InputOption, value: bool) -> Self { + self.values[Self::slot(option)] = value; + self + } + + pub fn recorded_values(&self) -> Vec<(&'static str, bool)> { + InputOption::ALL + .iter() + .map(|o| (o.name(), self.get(*o))) + .collect() + } + + pub fn spelling_profile(&self) -> String { + let bits = [ + InputOption::Abbreviation, + InputOption::TypoTolerance, + InputOption::FuzzyZ, + InputOption::FuzzyC, + InputOption::FuzzyS, + ]; + let mask = bits.iter().enumerate().fold(0, |mask, (bit, option)| { + mask | (usize::from(self.get(*option)) << bit) + }); + spelling_profile(mask) + } + + /// Replace entire config nodes so Rime's existing components retain their own snapshots. + pub fn schema_patch(&self) -> String { + let mut bindings = Vec::new(); + if self.get(InputOption::BracketPaging) { + bindings.push("{ when: has_menu, accept: bracketleft, send: Page_Up }"); + bindings.push("{ when: has_menu, accept: bracketright, send: Page_Down }"); + } + if self.get(InputOption::MinusEqualPaging) { + bindings.push("{ when: has_menu, accept: minus, send: Page_Up }"); + bindings.push("{ when: has_menu, accept: equal, send: Page_Down }"); + } + let pick = + |option, on: &'static str, off: &'static str| if self.get(option) { on } else { off }; + format!( + "key_binder:\n bindings: [{}]\npunctuator:\n half_shape:\n ',': ','\n '.': '。'\n '?': '?'\n '!': '!'\n ':': ':'\n ';': ';'\n '(': '('\n ')': ')'\n '{{': '{}'\n '}}': '{}'\n '[': '【'\n ']': '】'\n '<': '《'\n '>': '》'\n '\\': '{}'\n '|': '{}'\n '`': '{}'\n '~': '~'\n '$': '¥'\n '^': '……'\n '_': '——'\n '\"': {{ pair: ['“', '”'] }}\n \"'\": {{ pair: ['‘', '’'] }}", + bindings.join(", "), + pick(InputOption::CornerQuotes, "「", "{"), + pick(InputOption::CornerQuotes, "」", "}"), + pick(InputOption::IdeographicComma, "、", "\\"), + pick(InputOption::FullwidthPipe, "|", "|"), + pick(InputOption::MiddleDot, "·", "`"), + ) + } +} + +pub fn spelling_profile(mask: usize) -> String { + format!("inkflow_spelling_{mask}") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn swift_defaults_and_profiles() { + let defaults = InputPreferences::default(); + assert_eq!(defaults.spelling_profile(), "inkflow_spelling_3"); + assert!(!defaults.get(InputOption::Traditional) && defaults.get(InputOption::Emoji)); + let all = defaults + .clone() + .with(InputOption::FuzzyZ, true) + .with(InputOption::FuzzyC, true) + .with(InputOption::FuzzyS, true); + assert_eq!(all.spelling_profile(), "inkflow_spelling_31"); + let patch = defaults.schema_patch(); + assert!(patch.starts_with("key_binder:\n bindings: [{ when: has_menu, accept: bracketleft, send: Page_Up }, { when: has_menu, accept: bracketright, send: Page_Down }, { when: has_menu, accept: minus, send: Page_Up }, { when: has_menu, accept: equal, send: Page_Down }]\npunctuator:\n half_shape:\n ',': ','\n")); + assert!(patch.contains(" '{': '「'\n '}': '」'\n '[': '【'")); + assert!(patch.contains(" '\\': '、'\n '|': '|'\n '`': '·'\n '~': '~'")); + let plain = defaults + .with(InputOption::CornerQuotes, false) + .with(InputOption::IdeographicComma, false) + .with(InputOption::FullwidthPipe, false) + .with(InputOption::MiddleDot, false) + .with(InputOption::BracketPaging, false) + .with(InputOption::MinusEqualPaging, false) + .schema_patch(); + assert!(plain.contains(" bindings: []\n")); + assert!( + plain.contains(" '{': '{'\n '}': '}'\n") + && plain.contains(" '\\': '\\'\n '|': '|'\n '`': '`'\n") + ); + assert!(plain.ends_with(" '\"': { pair: ['“', '”'] }\n \"'\": { pair: ['‘', '’'] }")); + } +} diff --git a/Core/Portable/src/ranking.rs b/Core/Portable/src/ranking.rs index 0affbce..89e9605 100644 --- a/Core/Portable/src/ranking.rs +++ b/Core/Portable/src/ranking.rs @@ -158,11 +158,94 @@ fn technical(text: &str) -> bool { .any(|suffix| canonical.ends_with(suffix)) } +/// Exact Han phrase frequencies, read whole into memory before key events. +enum Phrases { + Map(HashMap), + /// The prepared `IFCX` index: one sorted fixed-width table per phrase length, + /// keyed by three bytes per NFC scalar so byte order is table order. + Index { + data: Vec, + tables: [(usize, usize); PHRASE_LIMIT - 1], + }, +} +const PHRASE_LIMIT: usize = 8; +const INDEX_HEADER: usize = 12 + (PHRASE_LIMIT - 1) * 4; + +fn encode(text: &str) -> Vec { + text.nfc() + .flat_map(|scalar| { + let value = scalar as u32; + [(value >> 16) as u8, (value >> 8) as u8, value as u8] + }) + .collect() +} + +impl Phrases { + fn frequency(&self, phrase: &str) -> Option { + match self { + Phrases::Map(map) => map.get(&normalized(phrase)).copied(), + Phrases::Index { data, tables } => { + let key = encode(phrase); + let length = key.len() / 3; + if !(2..=PHRASE_LIMIT).contains(&length) { + return None; + } + let (offset, count) = tables[length - 2]; + let record = length * 3 + 8; + let (mut low, mut high) = (0, count); + while low < high { + let mid = (low + high) / 2; + let start = offset + mid * record; + match data[start..start + key.len()].cmp(&key[..]) { + std::cmp::Ordering::Less => low = mid + 1, + std::cmp::Ordering::Greater => high = mid, + std::cmp::Ordering::Equal => { + let bytes = &data[start + key.len()..start + record]; + return Some(u64::from_le_bytes(bytes.try_into().unwrap()) as i64); + } + } + } + None + } + } + } +} + pub struct ContextRanker { - frequencies: HashMap, + frequencies: Phrases, longest: usize, } impl ContextRanker { + /// Load the prepared `pinyin_simp.context.bin` written by dictionary preparation. + pub fn from_index(data: Vec) -> Result { + if data.len() < INDEX_HEADER || &data[..4] != b"IFCX" { + return Err("Expected an InkFlow context index".into()); + } + let field = |position: usize| { + u32::from_le_bytes(data[4 * position..4 * position + 4].try_into().unwrap()) as usize + }; + let mut tables = [(0, 0); PHRASE_LIMIT - 1]; + let mut offset = INDEX_HEADER; + for length in 2..=PHRASE_LIMIT { + let count = field(length + 1); + tables[length - 2] = (offset, count); + offset = offset + .checked_add( + count + .checked_mul(length * 3 + 8) + .ok_or("Oversized context index")?, + ) + .ok_or("Oversized context index")?; + } + if field(1) != 1 || field(2) > PHRASE_LIMIT || offset != data.len() { + return Err("Unsupported or truncated context index".into()); + } + Ok(Self { + longest: field(2), + frequencies: Phrases::Index { data, tables }, + }) + } + /// Build before accepting key events. Ranking itself does not access storage. pub fn from_dictionary(text: &str) -> Self { let mut frequencies = HashMap::::new(); @@ -195,7 +278,7 @@ impl ContextRanker { longest = longest.max(count); } Self { - frequencies, + frequencies: Phrases::Map(frequencies), longest, } } @@ -241,13 +324,16 @@ impl ContextRanker { return (0, 0); } for count in (1..=prefix.len()).rev() { + if count + first_length > PHRASE_LIMIT { + continue; + } let phrase: String = prefix[..count] .iter() .rev() .copied() .chain(std::iter::once(candidate.as_str())) .collect(); - if let Some(&frequency) = self.frequencies.get(&normalized(&phrase)) { + if let Some(frequency) = self.frequencies.frequency(&phrase) { return (count, frequency); } } diff --git a/Core/Portable/tests/parity.rs b/Core/Portable/tests/parity.rs new file mode 100644 index 0000000..c2304cd --- /dev/null +++ b/Core/Portable/tests/parity.rs @@ -0,0 +1,909 @@ +//! Old/new comparison against the recorded Swift quality baseline: candidate order, +//! paging, selection, commits and learning after restart on equivalent isolated fixtures. +//! Needs prepared production resources: set INKFLOW_PORTABLE_RESOURCES to a directory +//! holding `shared/` and `prepared/cache/` from `Core/Portable/prepare-resources.sh`. +use inkflow_rime::{ + engine::{Configuration, Engine, InputSession}, + preferences::InputPreferences, +}; +use serde_json::{Value, json}; +use std::{ + fs, + path::{Path, PathBuf}, + sync::Arc, +}; + +// Rime is process-global: one parity test runs at a time. +static RUNTIME: std::sync::Mutex<()> = std::sync::Mutex::new(()); + +fn resources() -> Option<(PathBuf, std::sync::MutexGuard<'static, ()>)> { + let path = std::env::var_os("INKFLOW_PORTABLE_RESOURCES").map(PathBuf::from)?; + Some((path, RUNTIME.lock().unwrap_or_else(|e| e.into_inner()))) +} + +fn scratch(name: &str) -> PathBuf { + let root = std::env::temp_dir().join(format!("inkflow-parity-{}-{name}", std::process::id())); + if root.exists() { + fs::remove_dir_all(&root).unwrap(); + } + fs::create_dir_all(&root).unwrap(); + root +} + +fn engine(resources: &Path, user: &Path) -> Arc { + Engine::new(Configuration { + shared: resources.join("shared"), + user: user.to_path_buf(), + cache: Some(resources.join("prepared/cache")), + context_index: Some(resources.join("shared/pinyin_simp.context.bin")), + }) + .unwrap() +} + +fn configured(engine: &Arc) -> InputSession { + let session = engine.session().unwrap(); + session + .set_configuration(9, &[], Some(&InputPreferences::default())) + .unwrap(); + assert_eq!(session.configuration_error(), None); + session.set_preceding_text("").unwrap(); + session +} + +/// The Swift runner's fixed recipe: type, record the first page, find and select the target. +fn observe(session: &InputSession, input: &str, target: &str, page_limit: usize) -> Value { + session.clear().unwrap(); + session.set_preceding_text("").unwrap(); + for byte in input.bytes() { + let modifiers = i32::from(byte.is_ascii_uppercase()); + assert!( + session.key(i32::from(byte), modifiers).unwrap(), + "unhandled {input}" + ); + assert!( + session.take_commit().unwrap().is_empty(), + "early commit {input}" + ); + } + let first_page = session.snapshot().unwrap(); + let (mut offset, mut paging, mut exhausted, mut rank, mut committed) = + (0, 0, false, None, None); + for page in 0..page_limit { + let snapshot = session.snapshot().unwrap(); + if let Some(index) = snapshot.candidates.iter().position(|c| c.text == target) { + rank = Some(offset + index + 1); + session.select(&snapshot, index).unwrap(); + let text = session.take_commit().unwrap(); + assert!( + text == target && session.snapshot().unwrap().preedit.is_empty(), + "selection did not consume complete input: {input}" + ); + committed = Some(text); + break; + } + if page + 1 == page_limit { + break; + } + session.key(0xff56, 0).unwrap(); + if session.snapshot().unwrap().page == snapshot.page { + exhausted = true; + break; + } + paging += 1; + offset += snapshot.candidates.len(); + } + session.clear().unwrap(); + let texts: Vec<&str> = first_page.texts(); + // Swift's Codable omits absent optionals. + let mut observation = json!({ + "first": texts.first().copied().unwrap_or(""), + "topThree": texts.iter().take(3).collect::>(), + "targetRank": rank, + "inputOperations": input.len(), + "pagingOperations": paging, + "selectionOperations": if committed.is_some() { 1 } else { 0 }, + "totalOperations": committed.as_ref().map(|_| input.len() + paging + 1), + "committedText": committed, + "searchExhausted": exhausted, + }); + observation + .as_object_mut() + .unwrap() + .retain(|_, value| !value.is_null()); + observation +} + +#[test] +fn swift_quality_baseline() { + let Some((resources, _runtime)) = resources() else { + println!("SKIP quality-baseline parity: INKFLOW_PORTABLE_RESOURCES is not set"); + return; + }; + let fixtures = Path::new(env!("CARGO_MANIFEST_DIR")).join("../Fixtures/QualityBaseline"); + let corpus: Value = + serde_json::from_str(&fs::read_to_string(fixtures.join("corpus.json")).unwrap()).unwrap(); + let baseline: Value = + serde_json::from_str(&fs::read_to_string(fixtures.join("baseline.json")).unwrap()).unwrap(); + assert_eq!(baseline["corpus"], corpus); + assert_eq!(corpus["candidateCount"], 9); + let page_limit = corpus["pageLimit"].as_u64().unwrap() as usize; + let selections = corpus["learningSelections"].as_u64().unwrap() as usize; + let expected: Vec<(&str, bool)> = InputPreferences::default().recorded_values(); + for (name, value) in expected { + assert_eq!(baseline["inputOptions"][name], json!(value), "{name}"); + } + let users = scratch("quality-baseline"); + let mut results = Vec::new(); + let mut drift = Vec::new(); + for (index, sample) in corpus["samples"].as_array().unwrap().iter().enumerate() { + let id = sample["id"].as_str().unwrap(); + let input = sample["input"].as_str().unwrap(); + let target = sample["target"].as_str().unwrap(); + let user = users.join(id); + let (initial, training) = { + let engine = engine(&resources, &user); + let session = configured(&engine); + let initial = observe(&session, input, target, page_limit); + let mut training = 0; + if !initial["committedText"].is_null() { + training = 1; + for _ in 1..selections { + let repeat = observe(&session, input, target, page_limit); + assert_eq!( + repeat["committedText"], + json!(target), + "learning target disappeared: {id}" + ); + training += 1; + } + } + (initial, training) + }; + let engine = engine(&resources, &user); + let learned = observe(&configured(&engine), input, target, page_limit); + drop(engine); + let result = json!({"sample": sample, "initial": initial, "trainingCommits": training, "learned": learned}); + let recorded = &baseline["results"][index]; + println!( + "{id}: rank {} → {}; training commits {}{}", + initial["targetRank"], + learned["targetRank"], + training, + if *recorded == result { "" } else { " DRIFT" } + ); + if *recorded != result { + drift.push(id.to_owned()); + } + results.push(result); + } + let output = + Path::new(env!("CARGO_MANIFEST_DIR")).join("../../build/portable/quality-baseline.json"); + fs::write( + &output, + serde_json::to_string_pretty(&json!({"results": results})).unwrap(), + ) + .unwrap(); + fs::remove_dir_all(&users).unwrap(); + assert!( + drift.is_empty(), + "quality baseline drift: {}; inspect {}", + drift.join(", "), + output.display() + ); + println!( + "PASS Swift/Rust quality baseline: {} samples, initial and persisted learned states", + results.len() + ); +} + +fn type_keys(session: &InputSession, text: &str) { + for byte in text.bytes() { + session.key(i32::from(byte), 0).unwrap(); + } +} + +fn texts(session: &InputSession) -> Vec { + session + .snapshot() + .unwrap() + .texts() + .iter() + .map(|s| s.to_string()) + .collect() +} + +fn all_candidates(session: &InputSession) -> Vec { + let mut all = Vec::new(); + for _ in 0..1000 { + let snapshot = session.snapshot().unwrap(); + all.extend(snapshot.texts().iter().map(|s| s.to_string())); + session.key(0xff56, 0).unwrap(); + if session.snapshot().unwrap().page == snapshot.page { + for _ in 0..snapshot.page { + session.key(0xff55, 0).unwrap(); + } + assert_eq!(session.snapshot().unwrap().page, 0); + return all; + } + } + panic!("candidate enumeration must reach the final page"); +} + +/// Expectations copied from Core/Tests/InkFlowEngineTestSupport/EngineRegression.swift: +/// runCases, contextReranking, contextCustomPhrasePriority, customPhrases and inputSettings. +#[test] +fn swift_engine_regressions() { + use inkflow_rime::{Error, engine::Action, phrases::CustomPhrase, preferences::InputOption}; + let Some((resources, _runtime)) = resources() else { + println!("SKIP engine regression parity: INKFLOW_PORTABLE_RESOURCES is not set"); + return; + }; + let user = scratch("engine-regressions"); + let engine = engine(&resources, &user); + let mutations = Arc::new(std::sync::Mutex::new(Vec::new())); + let log = mutations.clone(); + engine.set_observer(Some(Arc::new(move |mutation| { + log.lock().unwrap().push(( + mutation.action.clone(), + mutation.handled, + mutation.before.input.clone(), + mutation.after.input.clone(), + )); + }))); + + // runCases: basic composition, cancel, backspace, paging, digits, counts and session isolation. + let a = engine.session().unwrap(); + let b = engine.session().unwrap(); + type_keys(&a, "nihao"); + assert!(texts(&a).contains(&"你好".to_string())); + assert!(b.snapshot().unwrap().preedit.is_empty()); + let shown = a.snapshot().unwrap(); + a.select(&shown, 0).unwrap(); + assert_eq!(a.take_commit().unwrap(), "你好"); + assert_eq!( + a.select(&shown, 0), + Err(Error::StaleSnapshot), + "a delayed selection cannot reuse a superseded page" + ); + type_keys(&a, "zhongguo"); + assert!(texts(&a).contains(&"中国".to_string())); + a.key(0xff1b, 0).unwrap(); + assert!(a.snapshot().unwrap().preedit.is_empty()); + type_keys(&a, "ni"); + a.key(0xff08, 0).unwrap(); + assert_eq!(a.snapshot().unwrap().preedit, "n"); + a.clear().unwrap(); + type_keys(&a, "shi"); + let first = texts(&a); + assert_eq!(first.len(), 5); + assert!(a.key(0xff56, 0).unwrap()); + assert_eq!(a.snapshot().unwrap().page, 1); + let second = texts(&a); + assert!(second.len() == 5 && second != first); + a.key(0xff55, 0).unwrap(); + assert!(a.snapshot().unwrap().page == 0 && texts(&a) == first); + assert!(a.key(50, 0).unwrap()); + assert_eq!(a.take_commit().unwrap(), first[1]); + type_keys(&a, "shi"); + a.key(0xff56, 0).unwrap(); + let second = texts(&a); + assert!(a.key(53, 0).unwrap()); + assert_eq!(a.take_commit().unwrap(), second[4]); + type_keys(&a, "nihao"); + a.key(32, 0).unwrap(); + assert_eq!(a.take_commit().unwrap(), "你好"); + a.set_ascii_mode(true).unwrap(); + assert!(!a.key(97, 0).unwrap()); + a.set_ascii_mode(false).unwrap(); + type_keys(&a, "nihao"); + a.commit().unwrap(); + assert_eq!(a.take_commit().unwrap(), "你好"); + a.clear().unwrap(); + type_keys(&a, "shi"); + let before = a.snapshot().unwrap(); + a.set_candidate_count(9).unwrap(); + assert!( + a.snapshot().unwrap() == before && a.take_commit().unwrap().is_empty(), + "a composing session keeps its page size" + ); + a.clear().unwrap(); + type_keys(&a, "shi"); + assert_eq!(texts(&a).len(), 9); + a.key(0xff56, 0).unwrap(); + let nine = texts(&a); + assert_eq!(nine.len(), 9); + assert!(a.key(57, 0).unwrap()); + assert_eq!(a.take_commit().unwrap(), nine[8]); + { + let fresh = engine.session().unwrap(); + type_keys(&fresh, "shi"); + assert_eq!(texts(&fresh).len(), 5); + fresh.clear().unwrap(); + fresh.set_candidate_count(9).unwrap(); + type_keys(&fresh, "shi"); + assert_eq!(texts(&fresh).len(), 9); + } + b.clear().unwrap(); + type_keys(&b, "shi"); + assert_eq!(texts(&b).len(), 5, "every session keeps its own count"); + b.clear().unwrap(); + a.set_candidate_count(3).unwrap(); + type_keys(&a, "shi"); + assert_eq!(texts(&a).len(), 3); + a.clear().unwrap(); + + // contextReranking: bundled phrases, stable fallback, arrows, partial selection. + let prepared = |prefix: &str, input: &str, count: usize| { + let session = engine.session().unwrap(); + session.set_candidate_count(count).unwrap(); + session.set_preceding_text(prefix).unwrap(); + type_keys(&session, input); + session + }; + let coverage = prepared("什么", "neng", 9); + assert_eq!( + texts(&coverage)[0], + "能", + "context must not promote partial 呢 above complete 能" + ); + assert!(coverage.key(32, 0).unwrap()); + assert!( + coverage.take_commit().unwrap() == "能" && coverage.snapshot().unwrap().preedit.is_empty() + ); + for count in [3, 5, 9] { + let contextual = prepared("什么", "neng", count); + let native = prepared("", "neng", count); + assert_eq!( + texts(&contextual), + texts(&native), + "same-length partial choices retain native order" + ); + let first_page = texts(&contextual); + assert!(contextual.key(0xff56, 0).unwrap() && contextual.snapshot().unwrap().page == 1); + assert!(contextual.key(0xff55, 0).unwrap() && texts(&contextual) == first_page); + let partial = first_page + .iter() + .position(|c| c == "呢") + .expect("呢 on the neng first page"); + let page = contextual.snapshot().unwrap(); + contextual.highlight(&page, partial).unwrap(); + contextual.set_preceding_text("什么").unwrap(); + assert_eq!( + contextual.snapshot().unwrap().highlighted, + partial, + "an unchanged prefix preserves explicit selection" + ); + assert!(contextual.key(32, 0).unwrap()); + assert!( + contextual.take_commit().unwrap().is_empty() + && contextual.snapshot().unwrap().preedit == "呢ng" + ); + let page = native.snapshot().unwrap(); + native.select(&page, partial).unwrap(); + assert_eq!( + texts(&contextual), + texts(&native), + "selected-prefix bypass preserves native remaining candidates" + ); + assert!(contextual.snapshot().unwrap().has_selected_prefix()); + } + for (prefix, input, expected) in [ + ("准备午", "can", "餐"), + ("正式宣", "bu", "布"), + ("最新软", "jian", "件"), + ("非常感", "xie", "谢"), + ] { + let session = prepared(prefix, input, 9); + let snapshot = session.snapshot().unwrap(); + assert_eq!( + snapshot.texts()[0], + expected, + "bundled dictionary context {prefix} + {input}" + ); + assert_eq!(snapshot.highlighted, 0); + assert!(session.key(32, 0).unwrap()); + assert_eq!(session.take_commit().unwrap(), expected); + } + for count in [3, 5, 9] { + let session = engine.session().unwrap(); + session.set_candidate_count(count).unwrap(); + type_keys(&session, "can"); + let original = texts(&session); + session.set_preceding_text("准备午").unwrap(); + let ranked = session.snapshot().unwrap(); + assert!(ranked.texts()[0] == "餐" && ranked.candidates.len() == count); + let mut rest: Vec<&str> = ranked.texts()[1..].to_vec(); + let mut others: Vec<&str> = original + .iter() + .map(|s| s.as_str()) + .filter(|s| *s != "餐") + .collect(); + assert_eq!(rest, others, "other candidates keep their native order"); + rest.sort(); + others.sort(); + assert_eq!( + session.snapshot().unwrap(), + ranked, + "snapshot reads must be side-effect free" + ); + assert!(session.key(0xff54, 0).unwrap()); + assert_eq!(session.snapshot().unwrap().highlighted, 1); + session.set_preceding_text("准备午").unwrap(); + assert_eq!(session.snapshot().unwrap().highlighted, 1); + assert!(session.key(32, 0).unwrap()); + assert_eq!(session.take_commit().unwrap(), ranked.texts()[1]); + } + for count in [3, 5, 9] { + let session = prepared("多", "can", count); + let candidates = texts(&session); + assert_eq!(candidates.len(), count); + assert!( + candidates[0].chars().all(|c| c as u32 > 127), + "context keeps Chinese first for short English conflicts" + ); + let index = candidates + .iter() + .position(|c| c == "can") + .expect("exact short English on the first page"); + assert!( + index < 3, + "exact short English stays in the top three: {candidates:?}" + ); + let page = session.snapshot().unwrap(); + session.select(&page, index).unwrap(); + assert!( + session.take_commit().unwrap() == "can" + && session.snapshot().unwrap().preedit.is_empty() + ); + } + for prefix in [ + "", + "完全无关", + "准备午,", + "准备午 ", + "准备午\n", + "准备午😀", + ] { + assert_eq!( + texts(&prepared(prefix, "can", 5)), + texts(&prepared("", "can", 5)), + "{prefix:?}" + ); + } + for index in 0..5 { + for digit in [false, true] { + let session = prepared("准备午", "can", 5); + let page = session.snapshot().unwrap(); + let expected = page.texts()[index].to_string(); + if digit { + assert!(session.key(49 + index as i32, 0).unwrap()); + } else { + session.select(&page, index).unwrap(); + } + assert!( + session.take_commit().unwrap() == expected + && session.snapshot().unwrap().preedit.is_empty() + ); + } + } + for (action, expected) in [ + ("commit", "餐"), + ("space", "餐"), + ("comma", "餐,"), + ("return", "can"), + ] { + let session = prepared("准备午", "can", 5); + match action { + "commit" => { + session.commit().unwrap(); + } + "space" => assert!(session.key(32, 0).unwrap()), + "comma" => assert!(session.key(44, 0).unwrap()), + _ => assert!(session.key(0xff0d, 0).unwrap()), + } + assert_eq!(session.take_commit().unwrap(), expected, "{action}"); + assert!(session.snapshot().unwrap().preedit.is_empty()); + } + let session = prepared("正式宣", "bu", 5); + let first = texts(&session); + assert!(session.key(0xff56, 0).unwrap() && session.snapshot().unwrap().page == 1); + assert_ne!(texts(&session), first); + assert!(session.key(0xff55, 0).unwrap() && texts(&session) == first); + let page = session.snapshot().unwrap(); + session.highlight(&page, first.len() - 1).unwrap(); + assert!(session.key(0xff54, 0).unwrap()); + let moved = session.snapshot().unwrap(); + assert!( + moved.page == 1 && moved.highlighted == 0, + "Down at the page edge continues on the next page" + ); + assert!(session.key(0xff52, 0).unwrap()); + let back = session.snapshot().unwrap(); + assert!( + back.page == 0 && back.highlighted == first.len() - 1, + "Up at the page start returns to the previous page's end" + ); + session.clear().unwrap(); + type_keys(&session, "can"); + assert_eq!( + texts(&session), + texts(&prepared("", "can", 5)), + "clearing discards the old prefix" + ); + session.set_preceding_text("准备午").unwrap(); + assert_eq!(texts(&session)[0], "餐"); + assert!(session.key(0xff08, 0).unwrap()); + assert_eq!(session.snapshot().unwrap().preedit, "ca"); + assert!(session.key(0xff1b, 0).unwrap()); + assert!(session.snapshot().unwrap().preedit.is_empty()); + let partial = prepared("迷", "nihao", 5); + let plain = prepared("", "nihao", 5); + assert_eq!( + texts(&partial), + texts(&plain), + "do not promote the shorter 你 through 迷你" + ); + let index = texts(&partial).iter().position(|c| c == "你").unwrap(); + partial.select(&partial.snapshot().unwrap(), index).unwrap(); + plain.select(&plain.snapshot().unwrap(), index).unwrap(); + assert!( + partial.take_commit().unwrap().is_empty() + && !partial.snapshot().unwrap().preedit.is_empty() + ); + assert!(partial.snapshot().unwrap().has_selected_prefix()); + assert_eq!(texts(&partial), texts(&plain)); + partial.select(&partial.snapshot().unwrap(), 0).unwrap(); + assert_eq!(partial.take_commit().unwrap(), "你好"); + + // Phrase reloads wait for every live session to be idle; release the composing ones. + drop((a, b, coverage, session, partial, plain)); + + // contextCustomPhrasePriority and customPhrases: priority, deferred reload, pending commits. + let session = engine.session().unwrap(); + let phrase = CustomPhrase::validated("p1", "can", "残").unwrap(); + session + .set_configuration(5, std::slice::from_ref(&phrase), None) + .unwrap(); + assert_eq!(session.configuration_error(), None); + session.set_preceding_text("准备午").unwrap(); + type_keys(&session, "can"); + assert_eq!( + texts(&session)[0], + "残", + "an exact custom phrase retains priority over contextual 餐" + ); + session.set_configuration(3, &[], None).unwrap(); + assert_eq!( + texts(&session)[0], + "残", + "deferred deletion preserves the current custom phrase snapshot" + ); + assert!(session.key(32, 0).unwrap()); + assert_eq!(session.take_commit().unwrap(), "残"); + session.set_preceding_text("准备午").unwrap(); + type_keys(&session, "can"); + assert_eq!( + texts(&session)[0], + "餐", + "context ranking resumes after the custom code is deleted at idle" + ); + assert!(session.key(32, 0).unwrap()); + assert_eq!(session.take_commit().unwrap(), "餐"); + let phrases: Vec = ["地址甲", "地址乙", "地址丙", "地址丁", "地址戊", "地址己"] + .iter() + .enumerate() + .map(|(i, text)| CustomPhrase::validated(&format!("dz{i}"), "dz", text).unwrap()) + .chain([ + CustomPhrase::validated("n1", "nihao", "您好朋友").unwrap(), + CustomPhrase::validated("n2", "nihao", "你好").unwrap(), + CustomPhrase::validated("bq", "bq", "#标签").unwrap(), + ]) + .collect(); + let a = engine.session().unwrap(); + let b = engine.session().unwrap(); + a.set_configuration(3, &phrases, None).unwrap(); + b.set_configuration(3, &phrases, None).unwrap(); + assert_eq!(a.configuration_error(), None); + type_keys(&a, "bq"); + assert_eq!( + texts(&a)[0], + "#标签", + "hash text survives Rime's TSV comment rule" + ); + a.clear().unwrap(); + type_keys(&a, "nihao"); + assert_eq!( + texts(&a)[..2], + ["您好朋友", "你好"], + "custom phrases precede ordinary candidates" + ); + assert_eq!( + all_candidates(&a).iter().filter(|c| *c == "你好").count(), + 1, + "no duplicate suggestions" + ); + a.clear().unwrap(); + type_keys(&a, "dza"); + assert!(!texts(&a).contains(&"地址甲".to_string())); + a.clear().unwrap(); + type_keys(&a, "dz"); + assert_eq!(texts(&a), ["地址甲", "地址乙", "地址丙"]); + a.key(0xff56, 0).unwrap(); + assert!(a.snapshot().unwrap().page == 1 && texts(&a) == ["地址丁", "地址戊", "地址己"]); + a.key(50, 0).unwrap(); + assert_eq!(a.take_commit().unwrap(), "地址戊"); + type_keys(&a, "dz"); + let old = a.snapshot().unwrap(); + let changed = CustomPhrase::validated("dz0", "dz", "更新地址").unwrap(); + a.set_configuration(5, &[changed.clone()], None).unwrap(); + a.set_configuration(3, &[changed.clone()], None).unwrap(); + assert!( + a.snapshot().unwrap() == old + && a.take_commit().unwrap().is_empty() + && a.candidate_count() == 3 + ); + a.key(32, 0).unwrap(); + // Applying settings between a completed composition and draining its commit must not lose text. + a.set_configuration(3, &[changed.clone()], None).unwrap(); + assert_eq!(a.take_commit().unwrap(), "地址甲"); + type_keys(&a, "dz"); + assert!(texts(&a)[0] == "更新地址" && a.candidate_count() == 3); + a.clear().unwrap(); + type_keys(&b, "dz"); + assert!( + texts(&b)[0] == "更新地址" && b.candidate_count() == 3, + "every idle session reloads the saved phrases together" + ); + b.clear().unwrap(); + { + let fresh = engine.session().unwrap(); + fresh + .set_configuration(3, &[changed.clone()], None) + .unwrap(); + type_keys(&fresh, "dz"); + assert_eq!(texts(&fresh)[0], "更新地址"); + } + a.set_ascii_mode(true).unwrap(); + a.set_configuration(9, &[], None).unwrap(); + assert!( + !a.key(97, 0).unwrap(), + "schema reload must preserve ASCII mode" + ); + a.set_ascii_mode(false).unwrap(); + type_keys(&a, "dz"); + assert!(!texts(&a).contains(&"更新地址".to_string())); + a.clear().unwrap(); + let invalid = CustomPhrase { + id: "x".into(), + code: "x\ty".into(), + text: "invalid".into(), + }; + a.set_configuration(5, &[invalid], None).unwrap(); + assert!(a.configuration_error().is_some()); + a.set_configuration(5, &[], None).unwrap(); + assert_eq!(a.configuration_error(), None); + assert_eq!( + fs::read_to_string(user.join("custom_phrase.txt")).unwrap(), + "# no comment\n", + "Rime's native custom_phrase.txt holds the saved phrases" + ); + + // inputSettings: spelling profiles, fuzzy pairs, traditional/emoji, punctuation and paging. + let session = engine.session().unwrap(); + let defaults = InputPreferences::default(); + let configure = |preferences: &InputPreferences| { + session + .set_configuration(9, &[], Some(preferences)) + .unwrap(); + assert_eq!(session.configuration_error(), None); + }; + let contains = |input: &str, expected: &str| { + session.clear().unwrap(); + type_keys(&session, input); + for _ in 0..100 { + let before = session.snapshot().unwrap(); + if before.texts().contains(&expected) { + return true; + } + session.key(0xff56, 0).unwrap(); + if before.page == session.snapshot().unwrap().page { + break; + } + } + false + }; + let exact = defaults + .clone() + .with(InputOption::Abbreviation, false) + .with(InputOption::TypoTolerance, false); + configure(&defaults); + assert!(contains("hlw", "互联网") && contains("zhguo", "中国")); + configure(&exact); + assert!(!contains("hlw", "互联网"), "abbreviation off"); + assert!(session.input_preferences() == Some(exact.clone()) && session.candidate_count() == 9); + for (option, input, output) in [ + (InputOption::FuzzyZ, "zongguo", "中国"), + (InputOption::FuzzyC, "canpin", "产品"), + (InputOption::FuzzyS, "sanghai", "上海"), + ] { + configure(&exact); + assert!(!contains(input, output), "fuzzy off {input}"); + configure(&exact.clone().with(option, true)); + assert!(contains(input, output), "fuzzy on {input}"); + } + configure(&exact); + assert!(!contains("nnihao", "你好"), "typo off"); + configure(&exact.clone().with(InputOption::TypoTolerance, true)); + assert!(contains("nnihao", "你好") && contains("hzidao", "知道")); + session.clear().unwrap(); + configure(&defaults); + type_keys(&session, "hulianwang"); + let before = session.snapshot().unwrap(); + let traditional = defaults + .clone() + .with(InputOption::Traditional, true) + .with(InputOption::Emoji, false); + session + .set_configuration(9, &[], Some(&traditional)) + .unwrap(); + assert!( + session.snapshot().unwrap() == before && session.take_commit().unwrap().is_empty(), + "pending options preserve the composing snapshot" + ); + session.key(32, 0).unwrap(); + session + .set_configuration(9, &[], Some(&traditional)) + .unwrap(); + assert_eq!( + session.take_commit().unwrap(), + "互联网", + "settings reload preserves a pending simplified commit" + ); + assert!( + contains("hulianwang", "互聯網"), + "traditional next composition" + ); + session.clear().unwrap(); + configure(&defaults.clone().with(InputOption::Traditional, true)); + assert!(contains("weixiao", "😊"), "traditional retains Emoji"); + configure(&traditional); + assert!(!contains("weixiao", "😊"), "Emoji off"); + for (option, inputs, outputs) in [ + (InputOption::CornerQuotes, "{}", "「」"), + (InputOption::MiddleDot, "`", "·"), + (InputOption::FullwidthPipe, "|", "|"), + (InputOption::IdeographicComma, "\\", "、"), + ] { + for enabled in [false, true] { + session.clear().unwrap(); + configure(&defaults.clone().with(option, enabled)); + let expected: Vec = if enabled { + outputs.chars().collect() + } else { + inputs.chars().collect() + }; + for (input, output) in inputs.chars().zip(expected) { + let handled = session.key(input as i32, 0).unwrap(); + let commit = session.take_commit().unwrap(); + assert_eq!( + if handled { commit } else { input.to_string() }, + output.to_string(), + "{option:?} = {enabled}" + ); + } + } + } + session.clear().unwrap(); + configure(&defaults.clone().with(InputOption::EnglishPunctuation, true)); + for input in "{},.`|\\".chars() { + let handled = session.key(input as i32, 0).unwrap(); + let output = session.take_commit().unwrap(); + assert_eq!( + if handled { output } else { input.to_string() }, + input.to_string(), + "English punctuation {input}" + ); + } + for (option, previous, next) in [ + (InputOption::BracketPaging, 91, 93), + (InputOption::MinusEqualPaging, 45, 61), + ] { + for enabled in [false, true] { + session.clear().unwrap(); + configure(&defaults.clone().with(option, enabled)); + type_keys(&session, "shi"); + session.key(next, 0).unwrap(); + assert_eq!( + session.snapshot().unwrap().page, + i32::from(enabled), + "independent paging {option:?} = {enabled}" + ); + if enabled { + session.key(previous, 0).unwrap(); + assert_eq!(session.snapshot().unwrap().page, 0); + } + session.take_commit().unwrap(); + } + } + session.clear().unwrap(); + configure(&defaults); + type_keys(&session, "nihao"); + let composing = session.snapshot().unwrap(); + session.set_ascii_mode(true).unwrap(); + assert!( + session.requested_ascii_mode() + && !session.ascii_mode().unwrap() + && session.snapshot().unwrap() == composing + ); + session.key(32, 0).unwrap(); + assert_eq!(session.take_commit().unwrap(), "你好"); + assert!( + !session.key(97, 0).unwrap() && session.ascii_mode().unwrap(), + "ASCII starts only after the old composition finishes" + ); + session.set_ascii_mode(false).unwrap(); + assert!(session.key(44, 0).unwrap()); + assert_eq!(session.take_commit().unwrap(), ","); + let retained = engine.session().unwrap(); + type_keys(&retained, "hlw"); + let retained_state = retained.snapshot().unwrap(); + retained.set_configuration(9, &[], Some(&exact)).unwrap(); + configure(&exact); + assert_eq!( + retained.snapshot().unwrap(), + retained_state, + "another live session keeps its composing prism" + ); + assert!( + retained.input_preferences() == Some(defaults.clone()) + && session.input_preferences() == Some(exact.clone()) + ); + retained.clear().unwrap(); + type_keys(&retained, "hlw"); + assert!( + !texts(&retained).contains(&"互联网".to_string()), + "deferred abbreviation off applies after cancellation" + ); + retained.clear().unwrap(); + let prism = resources.join(format!( + "prepared/cache/{}.prism.bin", + exact.spelling_profile() + )); + let hidden = prism.with_extension("test-backup"); + configure(&defaults); + fs::rename(&prism, &hidden).unwrap(); + session.set_configuration(9, &[], Some(&exact)).unwrap(); + let failed = session.configuration_error() + == Some(inkflow_rime::engine::ConfigurationError::MissingPrism) + && session.input_preferences() == Some(defaults.clone()); + fs::rename(&hidden, &prism).unwrap(); + assert!(failed, "a missing prism must not mark settings applied"); + configure(&exact); + assert_eq!(session.input_preferences(), Some(exact)); + + let recorded = mutations.lock().unwrap(); + assert!( + recorded + .iter() + .any(|(action, handled, before, after)| *action + == Action::Key { + key: 32, + modifiers: 0 + } + && *handled + && before == "nihao" + && after.is_empty()), + "the observer sees each mutation with the input before and after" + ); + assert!( + recorded + .iter() + .any(|(action, _, _, _)| matches!(action, Action::Select(_))) + ); + drop(recorded); + drop(engine); + fs::remove_dir_all(&user).unwrap(); + println!( + "PASS Swift engine regressions on the Rust engine: basic cases, context ranking, custom phrases, input settings, ASCII boundaries, stale selection, observer" + ); +} From faee8df870f51d7e14b21da525a4612104b4b46c Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Thu, 8 Oct 2026 23:50:25 +0800 Subject: [PATCH 11/21] Port personal-learning management and portable personal data to the Rust core `Engine::personal_learning_entries`, `delete_personal_learning` and `undo_personal_learning` port `EnginePersonalLearning.swift` over the same Lua `learning_manage` channel. They run only at the idle boundary after Rime's native undo window and carry a learning revision that commits, Backspace or Delete, new sessions and invalidation supersede. `personal.rs` defines the portable backup: it reads and writes the macOS format-1 document, keeps the three user dictionaries, custom phrases, the candidate count and input options, discloses skipped macOS-only preferences and rejects unknown fields. Export snapshots closed dictionaries through the existing native helper on a copy; import restores snapshots into staging, moves the originals aside, installs the new databases and restores the originals on any failure; `recover` finishes an interrupted import. Nothing runs while an engine is initialized in the process. The parity test trains on production resources, manages entries (busy, conflict, delete, undo), exports and imports between isolated directories, imports a macOS-format document with synthetic rows, and checks that rejected snapshots leave every dictionary unchanged. Refs #35 --- Core/Portable/Cargo.toml | 4 +- Core/Portable/README.md | 9 +- Core/Portable/native/bridge.cpp | 4 + Core/Portable/native/bridge.h | 4 + Core/Portable/src/engine.rs | 259 ++++++++++++++- Core/Portable/src/ffi.rs | 6 + Core/Portable/src/lib.rs | 26 ++ Core/Portable/src/personal.rs | 557 ++++++++++++++++++++++++++++++++ Core/Portable/tests/parity.rs | 274 ++++++++++++++++ 9 files changed, 1126 insertions(+), 17 deletions(-) create mode 100644 Core/Portable/src/personal.rs diff --git a/Core/Portable/Cargo.toml b/Core/Portable/Cargo.toml index 2f2abca..52bbb43 100644 --- a/Core/Portable/Cargo.toml +++ b/Core/Portable/Cargo.toml @@ -9,8 +9,6 @@ build = "build.rs" path = "src/lib.rs" [dependencies] +serde_json = "=1.0.151" unicode-normalization = "=0.1.25" unicode-segmentation = "=1.13.3" - -[dev-dependencies] -serde_json = "=1.0.151" diff --git a/Core/Portable/README.md b/Core/Portable/README.md index 86cc233..f1df621 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -39,8 +39,15 @@ The test copies the tiny checked-in source fixture into a fresh temporary direct - The 21 samples of [the quality baseline](../Fixtures/QualityBaseline/README.md) reproduce `baseline.json` exactly: first page, target rank, paging, selection, one-shot commits, five learning commits, and the learned state after the runtime restarts on the same user directory. - `swift_engine_regressions` carries the expectations of the Swift engine regressions for basic cases, context ranking, custom phrases, input settings and ASCII boundaries, plus the stale-selection and observer contracts. +- `swift_personal_learning_and_data` covers learning management at the idle boundary, export/import between isolated user directories, a macOS-format document with disclosed unsupported preferences, and rejected snapshots rolling back. -Quality recording stays outside the engine. `Engine::set_observer` installs a callback that receives every completed mutation with the displayed snapshots before and after, delivered after the policy lock is released; it cannot change input or hold a key event. The engine itself performs no telemetry, network, SQLite or disk work from a key event other than the phrase-file reload at a shared idle. Personal-learning management and portable personal-data import are not ported yet. +Quality recording stays outside the engine. `Engine::set_observer` installs a callback that receives every completed mutation with the displayed snapshots before and after, delivered after the policy lock is released; it cannot change input or hold a key event. The engine itself performs no telemetry, network, SQLite or disk work from a key event other than the phrase-file reload at a shared idle. + +## Personal learning and portable personal data + +`Engine::personal_learning_entries`, `delete_personal_learning` and `undo_personal_learning` port `EnginePersonalLearning.swift` over the same Lua `learning_manage` channel: they run only when every session is idle and Rime's native undo window (4 s after the last commit) has passed, otherwise `Busy`; entries and undo tokens carry a learning revision that any commit, Backspace/Delete, new session or invalidation supersedes (`Conflict`). As on macOS, the newest commit may still be in Rime's pending transaction until the next transaction or teardown, and a native restore revives a deleted count with one confirmation. Never call this API from a key callback. + +`src/personal.rs` defines the portable backup: `Backup::from_json`/`to_json` read and write the macOS format-1 document (`format`, `rime`, `settings`, `dictionaries`). Supported data are the three Rime user dictionaries (`pinyin_simp`, `inkflow_shared_english`, `inkflow_voice_alias`), custom phrases, the candidate count and the input options. macOS-only preferences (shortcuts, font size, layout, thunder mode, voice rules) are skipped and listed in `Backup::unsupported`; unknown fields are rejected like on macOS. Written documents carry neutral values for those fields; macOS additionally requires the three fuzzy options to agree and exactly one paging pair. `personal::export` snapshots closed dictionaries through the existing native helper on a copy; `personal::import` restores snapshots into a staging directory, then moves originals aside and installs the new databases, restoring the originals on any failure; `personal::recover` finishes an interrupted import. `None` in a backup is explicit absence and removes the local dictionary. None of this runs while an engine is initialized in the process (`EngineActive`); frontends stop the engine first and apply the returned settings through `set_configuration` afterwards. Backups never imply that live databases can be shared between engines. ## Interface contract diff --git a/Core/Portable/native/bridge.cpp b/Core/Portable/native/bridge.cpp index 5208fbf..adde829 100644 --- a/Core/Portable/native/bridge.cpp +++ b/Core/Portable/native/bridge.cpp @@ -272,3 +272,7 @@ extern "C" int ifp_apply_schema_patch(uintptr_t session, const char* schema, con return 0; } catch (...) { return -3; } } +extern "C" int ifp_personal_data_snapshot(const char* root, const char* name, const char* file, int restore) { + // IFPersonalDataSnapshot catches every exception and finalizes its own Rime instance. + return IFPersonalDataSnapshot(root, name, file, restore) == 0 ? 0 : -1; +} diff --git a/Core/Portable/native/bridge.h b/Core/Portable/native/bridge.h index fd165b3..1df9a6d 100644 --- a/Core/Portable/native/bridge.h +++ b/Core/Portable/native/bridge.h @@ -58,6 +58,10 @@ int ifp_call(uintptr_t session, const char* request, size_t capacity, char** res * init or patch load failed before any change. */ int ifp_apply_schema_patch(uintptr_t session, const char* schema, const char* yaml, const char* const* paths, size_t count, int* outcome); +/* Back up (restore == 0) or restore a closed user dictionary `root/.userdb` to/from the + * TSV snapshot `file`. Runs its own Rime instance: only while no runtime is initialized. + * -1: the snapshot or database was rejected; nothing is reported about why. */ +int ifp_personal_data_snapshot(const char* root, const char* name, const char* file, int restore); #ifdef __cplusplus } #endif diff --git a/Core/Portable/src/engine.rs b/Core/Portable/src/engine.rs index 81be0ba..939bc9b 100644 --- a/Core/Portable/src/engine.rs +++ b/Core/Portable/src/engine.rs @@ -241,6 +241,76 @@ struct State { loaded_phrases: Vec, sessions: BTreeMap, next: u64, + /// Invalidates management snapshots/undo after any observed learning or reset. + learning_revision: u64, + last_commit: Option, +} + +impl State { + fn observe_learning(&mut self) { + self.learning_revision = self.learning_revision.wrapping_add(1); + } + /// Native Rime allows undo while time(NULL) - transaction_time <= 3 seconds. + fn can_read_learning(&self) -> bool { + self.last_commit + .is_none_or(|at| at.elapsed() >= std::time::Duration::from_secs(4)) + } +} + +/// One personal-learning record of the shared English or voice-alias namespace. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct LearningEntry { + pub source: LearningSource, + pub code: String, + pub text: String, + pub commits: u32, + revision: u64, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum LearningSource { + English, + Voice, +} + +impl LearningSource { + fn name(self) -> &'static str { + match self { + LearningSource::English => "english", + LearningSource::Voice => "voice", + } + } +} + +impl LearningEntry { + pub fn id(&self) -> String { + format!("{}\t{}\t{}", self.source.name(), self.code, self.text) + } + fn fields(&self) -> [String; 4] { + [ + self.source.name().to_owned(), + self.code.clone(), + self.text.clone(), + self.commits.to_string(), + ] + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct LearningUndo { + pub entry: LearningEntry, + revision: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum LearningError { + /// A session is composing, a commit is undrained, or the native undo window is open. + Busy, + Unavailable, + /// The live state changed since the entry or undo was captured. + Conflict, + TooLarge, + Native(Error), } pub struct Engine { @@ -300,6 +370,8 @@ impl Engine { loaded_phrases: Vec::new(), sessions: BTreeMap::new(), next: 1, + learning_revision: 0, + last_commit: None, }; // No session exists yet: whatever the file holds now is what they all load. if let Some(error) = sync_phrase_file(&mut state, &file) { @@ -334,6 +406,7 @@ impl Engine { let mut state = self.lock()?; let id = state.next; state.next += 1; + state.observe_learning(); let core = Core::new(self.runtime.session(SCHEMA)?); state.sessions.insert(id, core); let session = InputSession { @@ -436,16 +509,25 @@ impl Engine { }) } - fn read_commit(core: &mut Core) -> Result { - match &mut core.session { - Some(session) if !core.lost => Ok(session.take_commit()?.unwrap_or_default()), - _ => Ok(String::new()), + /// Drains Rime's commit; a non-empty commit may have trained the user dictionary. + fn read_commit(state: &mut State, id: u64) -> Result { + let core = state.sessions.get_mut(&id).unwrap(); + let text = match &mut core.session { + Some(session) if !core.lost => session.take_commit()?.unwrap_or_default(), + _ => String::new(), + }; + if !text.is_empty() { + state.observe_learning(); + state.last_commit = Some(std::time::Instant::now()); } + Ok(text) } fn all_sessions_idle(state: &mut State) -> Result { - for core in state.sessions.values_mut() { - let commit = Self::read_commit(core)?; + let ids: Vec = state.sessions.keys().copied().collect(); + for id in ids { + let commit = Self::read_commit(state, id)?; + let core = state.sessions.get_mut(&id).unwrap(); core.buffered_commit.push_str(&commit); if !Self::raw(core)?.preedit.is_empty() || !core.buffered_commit.is_empty() { return Ok(false); @@ -483,10 +565,8 @@ impl Engine { if !Self::raw(core)?.preedit.is_empty() { return Ok(()); } - match self.recreate_schema(core)? { - Ok(()) => core.error = None, - Err(error) => core.error = Some(error), - } + let outcome = self.recreate_schema(state, id)?; + state.sessions.get_mut(&id).unwrap().error = outcome.err(); Ok(()) } @@ -530,8 +610,10 @@ impl Engine { fn recreate_schema( &self, - core: &mut Core, + state: &mut State, + id: u64, ) -> Result> { + let core = state.sessions.get_mut(&id).unwrap(); let profile = core.requested_input.spelling_profile(); if !self.compiled.join(format!("{profile}.prism.bin")).is_file() { return Ok(Err(ConfigurationError::MissingPrism)); @@ -543,7 +625,8 @@ impl Engine { ); // select_schema resets the commit buffer as well as the schema. Preserve completed text // even if settings arrive before the frontend has drained the previous key's commit. - let commit = Self::read_commit(core)?; + let commit = Self::read_commit(state, id)?; + let core = state.sessions.get_mut(&id).unwrap(); core.buffered_commit.push_str(&commit); let session = core.session.as_mut().unwrap(); let outcome = match session.apply_schema_patch(SCHEMA, &yaml, &PATCHED_NODES)? { @@ -665,6 +748,10 @@ impl Engine { return self.move_highlight(state, id, delta, key); } let handled = core.session.as_mut().unwrap().process_key(key, modifiers)?; + // Rime can undo learning while leaving Backspace unhandled for the client. + if key == 0xff08 || key == 0xffff { + state.observe_learning(); + } self.update_ordering(state, id, false)?; Ok(handled) } @@ -717,6 +804,152 @@ impl Engine { } } +impl Engine { + /// No filesystem enumeration and no secondary store: the active Rime memories own both + /// namespaces. This lifecycle API must never run in a key callback. + pub fn personal_learning_entries( + &self, + ) -> std::result::Result, LearningError> { + let (reply, revision) = self.learning_request(&["list"])?; + if reply.status != "ok" || !(reply.body.is_empty() || reply.body.ends_with('\n')) { + return Err(LearningError::Unavailable); + } + let lines: Vec<&str> = reply.body.split('\n').collect(); + let lines = &lines[..lines.len() - 1]; + if lines.len() > 8192 { + return Err(LearningError::TooLarge); + } + let mut ids = std::collections::HashSet::new(); + let mut entries = Vec::with_capacity(lines.len()); + for line in lines { + let parts: Vec<&str> = line.split('\t').collect(); + let [source, code, text, count] = parts[..] else { + return Err(LearningError::Unavailable); + }; + let source = match source { + "english" => LearningSource::English, + "voice" => LearningSource::Voice, + _ => return Err(LearningError::Unavailable), + }; + let commits: u32 = count.parse().map_err(|_| LearningError::Unavailable)?; + if !(1..=64).contains(&code.len()) + || !code.bytes().all(|b| b.is_ascii_lowercase()) + || !(1..=256).contains(&text.len()) + || text.chars().any(char::is_control) + || commits == 0 + { + return Err(LearningError::Unavailable); + } + let entry = LearningEntry { + source, + code: code.to_owned(), + text: text.to_owned(), + commits, + revision, + }; + if !ids.insert(entry.id()) { + return Err(LearningError::Unavailable); + } + entries.push(entry); + } + entries.sort_by_key(LearningEntry::id); + Ok(entries) + } + + pub fn delete_personal_learning( + &self, + entry: &LearningEntry, + ) -> std::result::Result { + if entry.revision != self.learning_revision()? { + return Err(LearningError::Conflict); + } + if entry.commits >= i32::MAX as u32 { + return Err(LearningError::TooLarge); + } + let fields = entry.fields(); + let mut request = vec!["delete"]; + request.extend(fields.iter().map(String::as_str)); + self.learning_mutation(&request)?; + Ok(LearningUndo { + entry: entry.clone(), + revision: self.learning_revision()?, + }) + } + + pub fn undo_personal_learning( + &self, + undo: &LearningUndo, + ) -> std::result::Result<(), LearningError> { + if undo.revision != self.learning_revision()? { + return Err(LearningError::Conflict); + } + let fields = undo.entry.fields(); + let mut request = vec!["restore"]; + request.extend(fields.iter().map(String::as_str)); + self.learning_mutation(&request) + } + + /// Drop every session's cached learning view after external changes to the dictionaries. + pub fn invalidate_personal_learning(&self) -> Result<()> { + let mut state = self.lock()?; + for core in state.sessions.values_mut() { + if let Some(session) = core.session.as_mut().filter(|_| !core.lost) { + session.call("learning_invalidate", &[], 512)?; + } + } + state.observe_learning(); + Ok(()) + } + + fn learning_revision(&self) -> std::result::Result { + Ok(self + .lock() + .map_err(LearningError::Native)? + .learning_revision) + } + + fn learning_mutation(&self, fields: &[&str]) -> std::result::Result<(), LearningError> { + let status = self.learning_request(fields)?.0.status; + // A failed native write may have applied part of the operation. Do not retain a + // stale snapshot or allow an older undo after any attempt. + self.invalidate_personal_learning() + .map_err(LearningError::Native)?; + match status.as_str() { + "ok" => Ok(()), + "conflict" => Err(LearningError::Conflict), + _ => Err(LearningError::Unavailable), + } + } + + fn learning_request( + &self, + fields: &[&str], + ) -> std::result::Result<(crate::channel::Reply, u64), LearningError> { + let mut state = self.lock().map_err(LearningError::Native)?; + if !Self::all_sessions_idle(&mut state).map_err(LearningError::Native)? + || !state.can_read_learning() + { + return Err(LearningError::Busy); + } + let revision = state.learning_revision; + let live = state.sessions.values_mut().find(|core| core.available()); + let mut temporary = None; + let session = match live { + Some(core) => core.session.as_mut().unwrap(), + None => temporary.insert( + self.runtime + .session(SCHEMA) + .map_err(LearningError::Native)?, + ), + }; + let reply = session + .call("learning_manage", fields, 3 * 1024 * 1024) + .map_err(LearningError::Native)? + .ok_or(LearningError::Unavailable)?; + Ok((reply, revision)) + } +} + /// Returns the failure; the file keeps its previous content on failure. fn sync_phrase_file(state: &mut State, file: &Path) -> Option { if state.requested_tsv == state.file_tsv { @@ -898,8 +1131,8 @@ impl InputSession { /// Drain completed text once, including text preserved across a settings reload. pub fn take_commit(&self) -> Result { let mut state = self.engine.lock()?; + let commit = Engine::read_commit(&mut state, self.id)?; let core = state.sessions.get_mut(&self.id).unwrap(); - let commit = Engine::read_commit(core)?; let text = std::mem::take(&mut core.buffered_commit) + &commit; Ok(text) } diff --git a/Core/Portable/src/ffi.rs b/Core/Portable/src/ffi.rs index 2099857..d9a4bdc 100644 --- a/Core/Portable/src/ffi.rs +++ b/Core/Portable/src/ffi.rs @@ -77,4 +77,10 @@ unsafe extern "C" { count: usize, outcome: *mut c_int, ) -> c_int; + pub fn ifp_personal_data_snapshot( + root: *const c_char, + name: *const c_char, + file: *const c_char, + restore: c_int, + ) -> c_int; } diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index 775ae4f..c36b3c9 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -3,6 +3,7 @@ pub mod channel; pub mod engine; mod ffi; +pub mod personal; pub mod phrases; pub mod preferences; pub mod ranking; @@ -65,6 +66,31 @@ unsafe fn copy(value: *const c_char) -> Result { .map_err(|_| Error::InvalidUtf8) } +/// Back up or restore one closed user dictionary through the native helper, which runs its +/// own Rime instance; rejected while a runtime is initialized in this process. +pub(crate) fn personal_data_snapshot( + root: &Path, + name: &str, + file: &Path, + restore: bool, +) -> Result { + let root = path(root)?; + let name = string(name)?; + let file = path(file)?; + let active = lock()?; + if *active { + return Err(Error::AlreadyRunning); + } + Ok(unsafe { + ffi::ifp_personal_data_snapshot( + root.as_ptr(), + name.as_ptr(), + file.as_ptr(), + i32::from(restore), + ) + } == 0) +} + struct RuntimeOwner { // Keep traits' borrowed paths alive through native finalization. _shared: CString, diff --git a/Core/Portable/src/personal.rs b/Core/Portable/src/personal.rs new file mode 100644 index 0000000..838186e --- /dev/null +++ b/Core/Portable/src/personal.rs @@ -0,0 +1,557 @@ +//! Portable personal data: the macOS format-1 backup document, user-dictionary snapshots +//! through the native helper, and a rollback-first install into an isolated user directory. +//! +//! Supported data: the three Rime user dictionaries, custom phrases, the candidate count and +//! input options. macOS-only preferences (shortcuts, font size, layout, voice rules) are +//! skipped and listed in [`Backup::unsupported`]. Nothing here runs while an engine is +//! initialized in the process; callers stop the engine first. +use crate::{ + Error, + phrases::CustomPhrase, + preferences::{InputOption, InputPreferences}, +}; +use serde_json::{Map, Value, json}; +use std::{ + collections::BTreeMap, + fs, + path::{Path, PathBuf}, + time::{SystemTime, UNIX_EPOCH}, +}; + +pub const DICTIONARIES: [&str; 3] = [ + "pinyin_simp", + "inkflow_shared_english", + "inkflow_voice_alias", +]; +pub const RIME_VERSION: &str = "1.17.0"; +pub const MAXIMUM_DOCUMENT_BYTES: usize = 128 * 1024 * 1024; +const MAXIMUM_SNAPSHOT_BYTES: usize = 32 * 1024 * 1024; +const SNAPSHOT_HEADER: &str = "# Rime user dictionary\n"; +const UNSUPPORTED_INTEGERS: [&str; 3] = ["fontSize", "vertical", "thunderMode"]; + +#[derive(Debug)] +pub enum PersonalError { + /// Another format, Rime version or dictionary set. + Incompatible, + /// A field this core does not know; macOS rejects these too. + UnknownFields, + Settings, + Phrases(crate::phrases::PhraseError), + /// A dictionary snapshot breaks the TSV contract. + Snapshot, + /// The native helper rejected a snapshot or database; the user directory is unchanged. + NativeSnapshot, + /// An earlier import was interrupted; call [`recover`] before importing again. + RecoveryRequired, + /// An engine is still initialized in this process. + EngineActive, + Io(std::io::Error), +} + +impl std::fmt::Display for PersonalError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + PersonalError::Phrases(error) => write!(f, "phrases: {}", error.code()), + PersonalError::Io(error) => write!(f, "io: {error}"), + other => write!(f, "{other:?}"), + } + } +} +impl std::error::Error for PersonalError {} +impl From for PersonalError { + fn from(error: std::io::Error) -> Self { + PersonalError::Io(error) + } +} + +/// The portable view of a macOS personal backup document. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Backup { + pub rime: String, + pub candidate_count: usize, + pub input: InputPreferences, + pub phrases: Vec, + /// `None` is explicit absence: importing removes the local dictionary. + pub dictionaries: BTreeMap>, + /// Document keys that were present but skipped because they are not portable. + pub unsupported: Vec, +} + +fn validate_snapshot(snapshot: &str) -> Result<(), PersonalError> { + if snapshot.len() > MAXIMUM_SNAPSHOT_BYTES + || !snapshot.starts_with(SNAPSHOT_HEADER) + || !snapshot.ends_with('\n') + || snapshot.contains('\0') + || snapshot.contains('\r') + { + return Err(PersonalError::Snapshot); + } + Ok(()) +} + +fn keys(object: &Map) -> Vec<&str> { + object.keys().map(String::as_str).collect() +} + +fn complete(names: &mut Vec<&str>) -> bool { + names.sort_unstable(); + let mut expected = DICTIONARIES; + expected.sort_unstable(); + *names == expected +} + +impl Backup { + /// Parse a format-1 document. Unknown fields are rejected; macOS-only fields are disclosed. + pub fn from_json(bytes: &[u8]) -> Result { + if bytes.len() > MAXIMUM_DOCUMENT_BYTES { + return Err(PersonalError::Incompatible); + } + let document: Value = + serde_json::from_slice(bytes).map_err(|_| PersonalError::Incompatible)?; + let object = document.as_object().ok_or(PersonalError::Incompatible)?; + if object.keys().any(|key| { + !matches!( + key.as_str(), + "format" | "rime" | "settings" | "dictionaries" + ) + }) { + return Err(PersonalError::UnknownFields); + } + let rime = object["rime"].as_str().ok_or(PersonalError::Incompatible)?; + if object["format"] != json!(1) || rime != RIME_VERSION { + return Err(PersonalError::Incompatible); + } + let snapshots = object["dictionaries"] + .as_object() + .ok_or(PersonalError::Incompatible)?; + if !complete(&mut keys(snapshots)) { + return Err(PersonalError::Incompatible); + } + let mut dictionaries = BTreeMap::new(); + for (name, snapshot) in snapshots { + let snapshot = match snapshot { + Value::Null => None, + Value::String(text) => { + validate_snapshot(text)?; + Some(text.clone()) + } + _ => return Err(PersonalError::Incompatible), + }; + dictionaries.insert(name.clone(), snapshot); + } + let settings = object["settings"] + .as_object() + .ok_or(PersonalError::Settings)?; + if settings.keys().any(|key| { + !matches!( + key.as_str(), + "integers" | "shortcuts" | "phrases" | "voiceRules" + ) + }) { + return Err(PersonalError::UnknownFields); + } + let integers = settings["integers"] + .as_object() + .ok_or(PersonalError::Settings)?; + let integer = |key: &str| { + integers + .get(key) + .and_then(Value::as_i64) + .ok_or(PersonalError::Settings) + }; + let candidate_count = integer("candidateCount")?; + if !(3..=9).contains(&candidate_count) { + return Err(PersonalError::Settings); + } + let mut input = InputPreferences::default(); + for option in InputOption::ALL { + input = input.with( + option, + match integer(&format!("input.{}", option.name()))? { + 0 => false, + 1 => true, + _ => return Err(PersonalError::Settings), + }, + ); + } + let mut unsupported = Vec::new(); + for key in integers.keys() { + let known = key == "candidateCount" + || key + .strip_prefix("input.") + .is_some_and(|name| InputOption::ALL.iter().any(|o| o.name() == name)); + if UNSUPPORTED_INTEGERS.contains(&key.as_str()) { + unsupported.push(format!("settings.integers.{key}")); + } else if !known { + return Err(PersonalError::UnknownFields); + } + } + // Neutral values (unbound shortcuts, no rules) are what this core writes; only real + // macOS preferences are disclosed as skipped. + let bound = settings + .get("shortcuts") + .and_then(Value::as_object) + .is_some_and(|map| map.values().any(|binding| !binding["keyCode"].is_null())); + if bound { + unsupported.push("settings.shortcuts".into()); + } + if settings + .get("voiceRules") + .and_then(Value::as_array) + .is_some_and(|rules| !rules.is_empty()) + { + unsupported.push("settings.voiceRules".into()); + } + let mut phrases = Vec::new(); + for phrase in settings + .get("phrases") + .and_then(Value::as_array) + .ok_or(PersonalError::Settings)? + { + let object = phrase.as_object().ok_or(PersonalError::Settings)?; + if object + .keys() + .any(|key| !matches!(key.as_str(), "id" | "code" | "text")) + { + return Err(PersonalError::UnknownFields); + } + let field = |key: &str| { + object + .get(key) + .and_then(Value::as_str) + .ok_or(PersonalError::Settings) + }; + phrases.push(CustomPhrase { + id: field("id")?.to_owned(), + code: field("code")?.to_owned(), + text: field("text")?.to_owned(), + }); + } + CustomPhrase::validate(&phrases).map_err(PersonalError::Phrases)?; + Ok(Self { + rime: rime.to_owned(), + candidate_count: candidate_count as usize, + input, + phrases, + dictionaries, + unsupported, + }) + } + + /// Encode a document macOS can read: unsupported preferences take their neutral values. + /// macOS additionally requires the three fuzzy options to agree and exactly one paging pair. + pub fn to_json(&self) -> Result, PersonalError> { + let mut names: Vec<&str> = self.dictionaries.keys().map(String::as_str).collect(); + if !complete(&mut names) || !(3..=9).contains(&self.candidate_count) { + return Err(PersonalError::Incompatible); + } + for snapshot in self.dictionaries.values().flatten() { + validate_snapshot(snapshot)?; + } + CustomPhrase::validate(&self.phrases).map_err(PersonalError::Phrases)?; + let mut integers = Map::new(); + integers.insert("candidateCount".into(), json!(self.candidate_count)); + integers.insert("fontSize".into(), json!(14)); + integers.insert("vertical".into(), json!(0)); + integers.insert("thunderMode".into(), json!(0)); + for option in InputOption::ALL { + integers.insert( + format!("input.{}", option.name()), + json!(i32::from(self.input.get(option))), + ); + } + let none = json!({"keyCode": null, "modifierBits": 0, "keyLabel": ""}); + let shortcuts: Map = [ + "inputMode", + "punctuation", + "script", + "voiceHold", + "voiceToggle", + ] + .into_iter() + .map(|action| (action.to_owned(), none.clone())) + .collect(); + let phrases: Vec = self + .phrases + .iter() + .map(|p| json!({"id": p.id, "code": p.code, "text": p.text})) + .collect(); + let document = json!({ + "format": 1, + "rime": RIME_VERSION, + "settings": {"integers": integers, "shortcuts": shortcuts, "phrases": phrases, "voiceRules": []}, + "dictionaries": self.dictionaries, + }); + let bytes = serde_json::to_vec(&document).map_err(|_| PersonalError::Incompatible)?; + if bytes.len() > MAXIMUM_DOCUMENT_BYTES { + return Err(PersonalError::Incompatible); + } + Ok(bytes) + } +} + +fn staging(user: &Path) -> Result { + let nanos = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_nanos()) + .unwrap_or(0); + let root = user + .join("PersonalData/staging") + .join(format!("{}-{nanos}", std::process::id())); + fs::create_dir_all(root.join("databases"))?; + Ok(root) +} + +fn copy_tree(from: &Path, to: &Path) -> std::io::Result<()> { + fs::create_dir(to)?; + for entry in fs::read_dir(from)? { + let entry = entry?; + let target = to.join(entry.file_name()); + if entry.file_type()?.is_dir() { + copy_tree(&entry.path(), &target)?; + } else { + fs::copy(entry.path(), target)?; + } + } + Ok(()) +} + +fn snapshot(root: &Path, name: &str, file: &Path, restore: bool) -> Result<(), PersonalError> { + match crate::personal_data_snapshot(root, name, file, restore) { + Ok(true) => Ok(()), + Ok(false) => Err(PersonalError::NativeSnapshot), + Err(Error::AlreadyRunning) => Err(PersonalError::EngineActive), + Err(_) => Err(PersonalError::NativeSnapshot), + } +} + +/// Snapshot the user directory's closed dictionaries; an absent dictionary is `None`. +pub fn export(user: &Path) -> Result>, PersonalError> { + let root = staging(user)?; + let result = (|| { + let mut dictionaries = BTreeMap::new(); + for name in DICTIONARIES { + let live = user.join(format!("{name}.userdb")); + if !live.is_dir() { + dictionaries.insert(name.to_owned(), None); + continue; + } + // Work on a copy so the live database is never opened by the helper. + copy_tree(&live, &root.join(format!("databases/{name}.userdb")))?; + let file = root.join(format!("{name}.userdb.txt")); + snapshot(&root.join("databases"), name, &file, false)?; + let text = fs::read_to_string(&file)?; + validate_snapshot(&text)?; + dictionaries.insert(name.to_owned(), Some(text)); + } + Ok(dictionaries) + })(); + let _ = fs::remove_dir_all(&root); + result +} + +fn transaction(user: &Path) -> PathBuf { + user.join("PersonalData/transaction") +} + +/// Replace the user directory's dictionaries with the snapshots. Originals are kept until every +/// move succeeds and restored on failure; an interrupted run is finished by [`recover`]. +pub fn import( + user: &Path, + dictionaries: &BTreeMap>, +) -> Result<(), PersonalError> { + let mut names: Vec<&str> = dictionaries.keys().map(String::as_str).collect(); + if !complete(&mut names) { + return Err(PersonalError::Incompatible); + } + if transaction(user).exists() { + return Err(PersonalError::RecoveryRequired); + } + let root = staging(user)?; + let staged = (|| { + for (name, snapshot_text) in dictionaries { + let Some(text) = snapshot_text else { continue }; + validate_snapshot(text)?; + let file = root.join(format!("{name}.userdb.txt")); + fs::write(&file, text)?; + snapshot(&root.join("databases"), name, &file, true)?; + } + Ok(()) + })(); + if let Err(error) = staged { + let _ = fs::remove_dir_all(&root); + return Err(error); + } + let transaction = transaction(user); + fs::create_dir_all(transaction.join("old"))?; + let originals: Map = DICTIONARIES + .iter() + .map(|name| { + ( + name.to_string(), + json!(user.join(format!("{name}.userdb")).is_dir()), + ) + }) + .collect(); + fs::write( + transaction.join("journal.json"), + serde_json::to_vec(&json!({"phase": "applying", "originals": originals})).unwrap(), + )?; + fs::rename(root.join("databases"), transaction.join("new"))?; + let _ = fs::remove_dir_all(&root); + let installed = install(user, &transaction); + if installed.is_err() { + rollback(user, &transaction)?; + return installed; + } + let _ = fs::remove_dir_all(&transaction); + Ok(()) +} + +fn install(user: &Path, transaction: &Path) -> Result<(), PersonalError> { + for name in DICTIONARIES { + let database = format!("{name}.userdb"); + let live = user.join(&database); + if live.exists() { + fs::rename(&live, transaction.join("old").join(&database))?; + } + let new = transaction.join("new").join(&database); + if new.exists() { + fs::rename(&new, &live)?; + } + } + Ok(()) +} + +fn rollback(user: &Path, transaction: &Path) -> Result<(), PersonalError> { + let journal: Value = serde_json::from_slice(&fs::read(transaction.join("journal.json"))?) + .map_err(|_| PersonalError::RecoveryRequired)?; + for name in DICTIONARIES { + let database = format!("{name}.userdb"); + let live = user.join(&database); + let old = transaction.join("old").join(&database); + let had_original = journal["originals"][name] == json!(true); + if old.exists() { + if live.exists() { + fs::remove_dir_all(&live)?; + } + fs::rename(&old, &live)?; + } else if !had_original && live.exists() { + fs::remove_dir_all(&live)?; + } else if had_original && !live.exists() { + return Err(PersonalError::RecoveryRequired); + } + } + fs::remove_dir_all(transaction)?; + Ok(()) +} + +/// Restore the originals of an interrupted import. No-op without a pending transaction. +pub fn recover(user: &Path) -> Result<(), PersonalError> { + let transaction = transaction(user); + if !transaction.exists() { + return Ok(()); + } + rollback(user, &transaction) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn fixture(name: &str) -> String { + format!( + "# Rime user dictionary\n#@/db_name\t{name}\n#@/db_type\tuserdb\n#@/rime_version\t1.17.0\n#@/tick\t99\nni hao \t你好\tc=7 d=0.123456789 t=42\n" + ) + } + + #[test] + fn macos_document_round_trip() { + let document = json!({ + "format": 1, "rime": "1.17.0", + "settings": { + "integers": {"candidateCount": 9, "fontSize": 18, "vertical": 1, "thunderMode": 0, + "input.abbreviation": 1, "input.typoTolerance": 0, "input.fuzzyZ": 1, "input.fuzzyC": 1, "input.fuzzyS": 1, + "input.emoji": 0, "input.bracketPaging": 0, "input.minusEqualPaging": 1, "input.englishPunctuation": 1, + "input.cornerQuotes": 0, "input.middleDot": 1, "input.fullwidthPipe": 1, "input.ideographicComma": 0, "input.traditional": 1}, + "shortcuts": {"inputMode": {"keyCode": 60, "modifierBits": 131072, "keyLabel": "右 Shift"}}, + "phrases": [{"id": "A", "code": "dz", "text": "地址"}], + "voiceRules": [{"anything": true}] + }, + "dictionaries": {"pinyin_simp": fixture("pinyin_simp"), "inkflow_shared_english": null, "inkflow_voice_alias": fixture("inkflow_voice_alias")} + }); + let backup = Backup::from_json(&serde_json::to_vec(&document).unwrap()).unwrap(); + assert_eq!(backup.candidate_count, 9); + assert!( + backup.input.get(InputOption::FuzzyZ) + && !backup.input.get(InputOption::Emoji) + && backup.input.get(InputOption::Traditional) + ); + assert!( + !backup.input.get(InputOption::TypoTolerance) + && backup.input.get(InputOption::MinusEqualPaging) + ); + assert_eq!( + backup.phrases, + vec![CustomPhrase::validated("A", "dz", "地址").unwrap()] + ); + assert_eq!(backup.dictionaries["inkflow_shared_english"], None); + assert_eq!( + backup.unsupported, + [ + "settings.integers.fontSize", + "settings.integers.thunderMode", + "settings.integers.vertical", + "settings.shortcuts", + "settings.voiceRules" + ] + ); + let encoded = backup.to_json().unwrap(); + let again = Backup::from_json(&encoded).unwrap(); + assert!( + again.candidate_count == 9 + && again.input == backup.input + && again.phrases == backup.phrases + && again.dictionaries == backup.dictionaries + ); + let reencoded: Value = serde_json::from_slice(&encoded).unwrap(); + assert_eq!(reencoded["settings"]["integers"]["fontSize"], json!(14)); + assert_eq!( + reencoded["settings"]["shortcuts"]["voiceHold"], + json!({"keyCode": null, "modifierBits": 0, "keyLabel": ""}) + ); + for (mutate, expected) in [ + (json!({"credentials": "never"}), "UnknownFields"), + (json!({"format": 2}), "Incompatible"), + (json!({"rime": "1.16.0"}), "Incompatible"), + ( + json!({"dictionaries": {"pinyin_simp": null}}), + "Incompatible", + ), + ( + json!({"dictionaries": {"pinyin_simp": "no header\n", "inkflow_shared_english": null, "inkflow_voice_alias": null}}), + "Snapshot", + ), + ( + json!({"settings": {"integers": {"candidateCount": 2}, "phrases": []}}), + "Settings", + ), + ( + json!({"settings": {"integers": {"candidateCount": 5}, "phrases": []}}), + "Settings", + ), + (json!({"settings.integers.input.extra": 1}), "UnknownFields"), + ] { + let mut broken = document.clone(); + for (key, value) in mutate.as_object().unwrap() { + if key == "settings.integers.input.extra" { + broken["settings"]["integers"]["input.extra"] = value.clone(); + } else { + broken[key] = value.clone(); + } + } + let error = Backup::from_json(&serde_json::to_vec(&broken).unwrap()).unwrap_err(); + assert_eq!(format!("{error:?}"), expected, "{mutate}"); + } + } +} diff --git a/Core/Portable/tests/parity.rs b/Core/Portable/tests/parity.rs index c2304cd..53222ca 100644 --- a/Core/Portable/tests/parity.rs +++ b/Core/Portable/tests/parity.rs @@ -907,3 +907,277 @@ fn swift_engine_regressions() { "PASS Swift engine regressions on the Rust engine: basic cases, context ranking, custom phrases, input settings, ASCII boundaries, stale selection, observer" ); } + +/// Expectations from Core/Sources/InkFlowRime/EnginePersonalLearning.swift and +/// macOS/Tests/PersonalDataTests.swift: management at the idle boundary, portable +/// backup/import on isolated user directories, and rollback on a rejected snapshot. +#[test] +fn swift_personal_learning_and_data() { + use inkflow_rime::engine::{LearningError, LearningSource}; + use inkflow_rime::personal::{self, Backup, DICTIONARIES, PersonalError}; + let Some((resources, _runtime)) = resources() else { + println!("SKIP personal data parity: INKFLOW_PORTABLE_RESOURCES is not set"); + return; + }; + let root = scratch("personal-data"); + let source = root.join("source"); + let trained = engine(&resources, &source); + let session = configured(&trained); + assert_eq!( + trained.personal_learning_entries().unwrap(), + vec![], + "a fresh user directory has no learning" + ); + for _ in 0..5 { + type_keys(&session, "email"); + let page = session.snapshot().unwrap(); + let index = page + .texts() + .iter() + .position(|c| *c == "email") + .expect("email candidate"); + session.select(&page, index).unwrap(); + assert_eq!(session.take_commit().unwrap(), "email"); + } + type_keys(&session, "emai"); + assert_eq!( + trained.personal_learning_entries(), + Err(LearningError::Busy), + "composition blocks management" + ); + session.clear().unwrap(); + assert_eq!( + trained.personal_learning_entries(), + Err(LearningError::Busy), + "the native undo window blocks management" + ); + std::thread::sleep(std::time::Duration::from_secs(4)); + let entries = trained.personal_learning_entries().unwrap(); + let email = entries + .iter() + .find(|e| e.source == LearningSource::English && e.code == "email" && e.text == "email") + .expect("learned email"); + // Rime batches user-dictionary writes per transaction; the newest commit can still be + // pending until the next transaction or teardown, exactly as with the Swift engine. + assert!(matches!(email.commits, 4 | 5), "{entries:?}"); + let commits = email.commits; + let undo = trained.delete_personal_learning(email).unwrap(); + assert_eq!( + trained.delete_personal_learning(email), + Err(LearningError::Conflict), + "a stale entry cannot be deleted twice" + ); + assert!( + trained + .personal_learning_entries() + .unwrap() + .iter() + .all(|e| e.code != "email"), + "deleted learning disappears" + ); + trained.undo_personal_learning(&undo).unwrap(); + assert_eq!( + trained.undo_personal_learning(&undo), + Err(LearningError::Conflict) + ); + let restored = trained.personal_learning_entries().unwrap(); + // Native restore revives the tombstoned count with one confirmation, as on macOS. + assert!( + restored + .iter() + .any(|e| e.code == "email" && e.commits >= commits), + "undo restores the entry: {restored:?}" + ); + type_keys(&session, "nihao"); + session.key(32, 0).unwrap(); + assert_eq!(session.take_commit().unwrap(), "你好"); + assert_eq!( + trained.delete_personal_learning(&restored[0]), + Err(LearningError::Conflict), + "a commit invalidates captured entries" + ); + assert_eq!( + trained.personal_learning_entries(), + Err(LearningError::Busy), + "a new commit reopens the undo window" + ); + std::thread::sleep(std::time::Duration::from_secs(4)); + assert!(trained.personal_learning_entries().is_ok()); + drop(session); + drop(trained); + + // Export the trained directory and import it into a fresh one. + fs::remove_dir_all(source.join("inkflow_voice_alias.userdb")).unwrap(); + let dictionaries = personal::export(&source).unwrap(); + assert!( + dictionaries["pinyin_simp"].is_some() && dictionaries["inkflow_shared_english"].is_some() + ); + assert_eq!( + dictionaries["inkflow_voice_alias"], None, + "an absent dictionary exports as absent" + ); + let backup = Backup { + rime: personal::RIME_VERSION.into(), + candidate_count: 7, + input: InputPreferences::default(), + phrases: vec![inkflow_rime::phrases::CustomPhrase::validated("p", "dz", "地址").unwrap()], + dictionaries, + unsupported: vec![], + }; + let document = backup.to_json().unwrap(); + let parsed = Backup::from_json(&document).unwrap(); + assert!( + parsed.candidate_count == 7 + && parsed.phrases == backup.phrases + && parsed.dictionaries == backup.dictionaries + ); + let target = root.join("target"); + fs::create_dir_all(&target).unwrap(); + personal::import(&target, &parsed.dictionaries).unwrap(); + assert!( + target.join("pinyin_simp.userdb").is_dir() + && !target.join("inkflow_voice_alias.userdb").exists() + ); + { + let engine = engine(&resources, &target); + let session = configured(&engine); + session + .set_configuration(parsed.candidate_count, &parsed.phrases, Some(&parsed.input)) + .unwrap(); + let imported = engine.personal_learning_entries().unwrap(); + assert!( + imported.iter().any(|e| e.source == LearningSource::English + && e.code == "email" + && e.commits >= commits), + "English learning survives import: {imported:?}" + ); + type_keys(&session, "dz"); + assert_eq!( + texts(&session)[0], + "地址", + "imported phrases apply through settings" + ); + session.clear().unwrap(); + assert_eq!( + personal::export(&target).unwrap_err().to_string(), + PersonalError::EngineActive.to_string(), + "personal data never runs beside a live engine" + ); + } + assert_eq!( + personal::export(&target).unwrap()["inkflow_shared_english"], + parsed.dictionaries["inkflow_shared_english"], + "import preserves the snapshot rows" + ); + + // A macOS-format document with synthetic rows restores into existing data, and a rejected + // snapshot leaves the originals in place. + let synthetic = |name: &str, row: &str| { + format!( + "# Rime user dictionary\n#@/db_name\t{name}\n#@/db_type\tuserdb\n#@/rime_version\t1.17.0\n#@/user_id\tsynthetic-source\n#@/tick\t99\n{row}\tc=7 d=0.123456789 t=42\nce shi \t测试\tc=-3 d=0.0000123 t=17\n" + ) + }; + let mut macos: serde_json::Map = serde_json::from_slice(&document).unwrap(); + macos["dictionaries"] = json!({ + "pinyin_simp": synthetic("pinyin_simp", "ni hao \t你好"), + "inkflow_shared_english": synthetic("inkflow_shared_english", "backupword \tBackupWord"), + "inkflow_voice_alias": synthetic("inkflow_voice_alias", "beifen \t备份别名"), + }); + macos["settings"]["shortcuts"]["inputMode"] = + json!({"keyCode": 60, "modifierBits": 131072, "keyLabel": "右 Shift"}); + let restored = Backup::from_json(&serde_json::to_vec(&macos).unwrap()).unwrap(); + assert_eq!( + restored.unsupported, + [ + "settings.integers.fontSize", + "settings.integers.thunderMode", + "settings.integers.vertical", + "settings.shortcuts" + ] + ); + personal::import(&target, &restored.dictionaries).unwrap(); + { + let engine = engine(&resources, &target); + let _session = configured(&engine); + let learned = engine.personal_learning_entries().unwrap(); + assert!( + learned.iter().any(|e| e.source == LearningSource::English + && e.text == "BackupWord" + && e.commits == 7), + "English learning survives destination identity: {learned:?}" + ); + assert!( + learned.iter().any(|e| e.source == LearningSource::Voice + && e.text == "备份别名" + && e.commits == 7), + "voice learning survives destination identity" + ); + assert!( + learned.iter().all(|e| e.code != "email"), + "import replaces the previous dictionaries" + ); + } + let before = personal::export(&target).unwrap(); + let mut broken = restored.dictionaries.clone(); + broken.insert( + "pinyin_simp".into(), + Some(synthetic("pinyin_simp", "ni hao \t你好") + "x \tx\tc=NaN d=1 t=1\n"), + ); + assert!(matches!( + personal::import(&target, &broken), + Err(PersonalError::NativeSnapshot) + )); + assert_eq!( + personal::export(&target).unwrap(), + before, + "a rejected import leaves every dictionary unchanged" + ); + for suffix in [ + "malformed\n", + "x \tx\tc=1 d=inf t=1\n", + "x \tx\tc=1 d=1 t=-1\n", + "ni hao \t你好\tc=1 d=1 t=1\n", + "#unknown\n", + ] { + let mut broken = restored.dictionaries.clone(); + broken.insert( + "pinyin_simp".into(), + Some(synthetic("pinyin_simp", "ni hao \t你好") + suffix), + ); + assert!( + matches!( + personal::import(&target, &broken), + Err(PersonalError::NativeSnapshot) + ), + "{suffix:?}" + ); + } + let mut absent = restored.dictionaries.clone(); + for name in DICTIONARIES { + absent.insert(name.into(), None); + } + personal::import(&target, &absent).unwrap(); + assert!( + DICTIONARIES + .iter() + .all(|name| !target.join(format!("{name}.userdb")).exists()), + "absence in the backup removes the local dictionaries" + ); + assert_eq!(personal::export(&target).unwrap(), absent); + fs::create_dir_all(target.join("PersonalData/transaction/old/pinyin_simp.userdb")).unwrap(); + fs::write(target.join("PersonalData/transaction/journal.json"), r#"{"phase":"applying","originals":{"pinyin_simp":true,"inkflow_shared_english":false,"inkflow_voice_alias":false}}"#).unwrap(); + assert!(matches!( + personal::import(&target, &absent), + Err(PersonalError::RecoveryRequired) + )); + personal::recover(&target).unwrap(); + assert!( + target.join("pinyin_simp.userdb").is_dir() + && !target.join("PersonalData/transaction").exists(), + "recovery restores the original" + ); + fs::remove_dir_all(&root).unwrap(); + println!( + "PASS personal learning and data: idle-boundary management, delete/undo/conflict, export/import on isolated directories, macOS document import with disclosure, rejected snapshots roll back" + ); +} From 000f2335e56478686afba117eeef68970d7f22e3 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Thu, 8 Oct 2026 23:54:59 +0800 Subject: [PATCH 12/21] Capture Rust engine performance with the recorded macOS headless protocol Add `performance-baseline`, the Rust counterpart of the Swift headless capture, and `performance.sh`, which runs five fresh release processes on prepared production resources and prints the distributions beside the approved review limits. Record a same-host comparison with the Swift engine in the README: key latency matches, startup and peak RSS are lower, and every approved limit holds. Refs #35 --- Core/Portable/README.md | 18 +++ Core/Portable/performance.sh | 52 ++++++++ Core/Portable/src/bin/performance-baseline.rs | 117 ++++++++++++++++++ 3 files changed, 187 insertions(+) create mode 100755 Core/Portable/performance.sh create mode 100644 Core/Portable/src/bin/performance-baseline.rs diff --git a/Core/Portable/README.md b/Core/Portable/README.md index f1df621..b302e95 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -17,6 +17,8 @@ bash Core/Portable/prepare-resources.sh python3 scripts/mac-remote.py resources # Compare the Rust engine with the recorded Swift behavior on those resources: bash Core/Portable/parity.sh [build/portable/resources.XXXXXX] +# Capture key latency and memory with the recorded macOS headless protocol: +bash Core/Portable/performance.sh build/portable/resources.XXXXXX [OUTPUT_DIRECTORY] ``` The first build downloads checksum-verified source archives. Subsequent builds reuse them under `build/portable/`; native outputs and Cargo artifacts stay there too. Set `CMAKE_BUILD_PARALLEL_LEVEL` to change the native build parallelism (default 4). Do not run two builds in the same checkout. Remove `build/portable/` for a clean rebuild or after changing compiler/SDK/architecture. @@ -43,6 +45,22 @@ The test copies the tiny checked-in source fixture into a fresh temporary direct Quality recording stays outside the engine. `Engine::set_observer` installs a callback that receives every completed mutation with the displayed snapshots before and after, delivered after the policy lock is released; it cannot change input or hold a key event. The engine itself performs no telemetry, network, SQLite or disk work from a key event other than the phrase-file reload at a shared idle. +## Performance + +`performance.sh` runs `src/bin/performance-baseline.rs`, the Rust counterpart of `Core/Tests/PerformanceBaseline/PerformanceBaseline.swift`: five fresh release processes, an absent user directory each, the 21-sample quality corpus typed five times with nine candidates, no commits, context and custom phrases empty. Each key operation is `key` + `take_commit` + `snapshot`; startup covers engine creation (prepared cache, context index, Rime start) and the first configured session. Memory is `getrusage` peak RSS. The summary prints the distributions beside the [approved review limits](../Fixtures/MigrationBaseline/README.md#approved-review-limits) without failing on them. + +Same host (Apple M1 Pro, 32 GiB, macOS 27.0.1), same protocol, both engines on freshly prepared production resources, at faee8df. The Swift column ran `performance-baseline` in release from this checkout; the recorded M5 Pro baseline is not comparable to either. + +| Measurement | Swift engine | Rust engine | Approved limit | +| --- | ---: | ---: | ---: | +| Headless startup median, ms | 304.61 | 150.58 | 2,300 | +| Peak RSS after input max, MiB | 58.95 | 51.69 | 344 | +| First-pass key median / p95 / p99, ms | 0.609 / 1.952 / 2.543 | 0.585 / 1.856 / 2.309 | p99 2.6 | +| Repeated-pass key median / p95 / p99, ms | 0.582 / 1.879 / 2.289 | 0.547 / 1.759 / 2.153 | p95 1.8, p99 2.3 | +| First key of each process, ms | 13.35–15.33 | 14.19–20.24 | 25 | + +Key latency is the same Rime work in both engines; the Rust engine's lower startup and RSS come from mapping no Swift runtime and reading only the prepared index. The one 20 ms first key was the first process after a rebuild (cold file cache); the other four match Swift. Recapture both columns on the same machine after toolchain or OS changes. + ## Personal learning and portable personal data `Engine::personal_learning_entries`, `delete_personal_learning` and `undo_personal_learning` port `EnginePersonalLearning.swift` over the same Lua `learning_manage` channel: they run only when every session is idle and Rime's native undo window (4 s after the last commit) has passed, otherwise `Busy`; entries and undo tokens carry a learning revision that any commit, Backspace/Delete, new session or invalidation supersedes (`Conflict`). As on macOS, the newest commit may still be in Rime's pending transaction until the next transaction or teardown, and a native restore revives a deleted count with one confirmation. Never call this API from a key callback. diff --git a/Core/Portable/performance.sh b/Core/Portable/performance.sh new file mode 100755 index 0000000..ab5ff35 --- /dev/null +++ b/Core/Portable/performance.sh @@ -0,0 +1,52 @@ +#!/bin/bash +# Capture the Rust engine with the recorded macOS headless protocol: five fresh release +# processes, absent user directories, the quality corpus typed five times. Reports the +# distributions beside the approved review limits; it does not fail on them. +set -euo pipefail +cd "$(dirname "$0")/../.." +resources=${1:?Usage: performance.sh RESOURCES_DIRECTORY [OUTPUT_DIRECTORY]} +[[ -f "$resources/prepared/complete" ]] || { echo "Not a prepared resources directory: $resources" >&2; exit 2; } +resources=$(cd "$resources" && pwd) +output=${2:-build/portable/performance} +mkdir -p "$output"; output=$(cd "$output" && pwd) +export CARGO_TARGET_DIR="$PWD/build/portable/cargo" +cargo build --locked --release --manifest-path Core/Portable/Cargo.toml --bin performance-baseline +scratch=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-portable-performance.XXXXXX") +trap 'rm -rf "$scratch"' EXIT +for trial in 1 2 3 4 5; do + build/portable/cargo/release/performance-baseline "$resources" "$scratch/user-$trial" \ + Core/Fixtures/QualityBaseline/corpus.json "$output/performance-$trial.json" 2>/dev/null +done +python3 - "$output" <<'PY' +import json, math, sys +from pathlib import Path +output = Path(sys.argv[1]) +trials = [json.loads((output / f"performance-{trial}.json").read_text()) for trial in range(1, 6)] +def distribution(values): + ordered = sorted(values) + return {"count": len(ordered), "median": round(float(__import__('statistics').median(ordered)), 3), + "p95": round(ordered[math.ceil(len(ordered) * 0.95) - 1], 3), + "p99": round(ordered[math.ceil(len(ordered) * 0.99) - 1], 3), "max": round(ordered[-1], 3)} +keys = [key for trial in trials for key in trial["timings"]] +summary = { + "protocol": "Core/Tests/PerformanceBaseline/PerformanceBaseline.swift on the Rust engine: release build, five processes, five corpus passes, absent user directory, no commits", + "startupMilliseconds": distribution([t["startupMilliseconds"] for t in trials]), + "peakResidentMiBAfterInput": distribution([t["peakResidentBytesAfterInput"] / 1048576 for t in trials]), + "firstPassKeyMilliseconds": distribution([k["milliseconds"] for k in keys if k["pass"] == 0]), + "repeatedPassKeyMilliseconds": distribution([k["milliseconds"] for k in keys if k["pass"] > 0]), + "firstKeyPerProcessMilliseconds": [round(t["timings"][0]["milliseconds"], 3) for t in trials], +} +# Core/Fixtures/MigrationBaseline/README.md, approved review limits. +limits = [("startup median", summary["startupMilliseconds"]["median"], 2300), + ("peak RSS after input max MiB", summary["peakResidentMiBAfterInput"]["max"], 344), + ("repeated-pass key p95", summary["repeatedPassKeyMilliseconds"]["p95"], 1.8), + ("repeated-pass key p99", summary["repeatedPassKeyMilliseconds"]["p99"], 2.3), + ("first-pass key p99", summary["firstPassKeyMilliseconds"]["p99"], 2.6), + ("max first key", max(summary["firstKeyPerProcessMilliseconds"]), 25)] +summary["approvedLimits"] = [{"metric": m, "value": v, "limit": l, "withinLimit": v <= l} for m, v, l in limits] +(output / "summary.json").write_text(json.dumps(summary, indent=2) + "\n") +for name in ["startupMilliseconds", "peakResidentMiBAfterInput", "firstPassKeyMilliseconds", "repeatedPassKeyMilliseconds", "firstKeyPerProcessMilliseconds"]: + print(f"{name}: {json.dumps(summary[name])}") +for item in summary["approvedLimits"]: + print(f"{'ok ' if item['withinLimit'] else 'OVER'} {item['metric']}: {item['value']} (limit {item['limit']})") +PY diff --git a/Core/Portable/src/bin/performance-baseline.rs b/Core/Portable/src/bin/performance-baseline.rs new file mode 100644 index 0000000..f38e4a0 --- /dev/null +++ b/Core/Portable/src/bin/performance-baseline.rs @@ -0,0 +1,117 @@ +//! Headless key-latency and memory capture of the Rust engine, following the protocol of +//! Core/Tests/PerformanceBaseline/PerformanceBaseline.swift: one fresh process per trial, +//! an absent user directory, the quality corpus typed five times without commits. +use inkflow_rime::{ + engine::{Configuration, Engine}, + preferences::InputPreferences, +}; +use serde_json::{Value, json}; +use std::{error::Error, fs, path::PathBuf, time::Instant}; + +#[repr(C)] +struct Rusage { + utime: [i64; 2], + stime: [i64; 2], + maxrss: i64, + rest: [i64; 13], +} +unsafe extern "C" { + fn getrusage(who: i32, usage: *mut Rusage) -> i32; +} +/// Process peak RSS in bytes (Darwin reports bytes, Linux KiB). +fn peak_resident_bytes() -> Result> { + let mut usage = Rusage { + utime: [0; 2], + stime: [0; 2], + maxrss: 0, + rest: [0; 13], + }; + if unsafe { getrusage(0, &mut usage) } != 0 { + return Err("getrusage failed".into()); + } + Ok(if cfg!(target_os = "macos") { + usage.maxrss + } else { + usage.maxrss * 1024 + }) +} + +fn run() -> Result<(), Box> { + let args: Vec<_> = std::env::args_os().skip(1).map(PathBuf::from).collect(); + let [resources, user, corpus, output] = args.as_slice() else { + return Err("Usage: performance-baseline RESOURCES ABSENT_USER CORPUS OUTPUT".into()); + }; + if user.exists() { + return Err("User directory must start absent".into()); + } + let corpus: Value = serde_json::from_slice(&fs::read(corpus)?)?; + let count = corpus["candidateCount"].as_u64().ok_or("corpus")? as usize; + let samples = corpus["samples"].as_array().ok_or("corpus")?; + if count != 9 || samples.is_empty() { + return Err("Unsupported corpus".into()); + } + let start = Instant::now(); + let engine = Engine::new(Configuration { + shared: resources.join("shared"), + user: user.clone(), + cache: Some(resources.join("prepared/cache")), + context_index: Some(resources.join("shared/pinyin_simp.context.bin")), + })?; + let session = engine.session()?; + session.set_configuration(count, &[], Some(&InputPreferences::default()))?; + if session.configuration_error().is_some() { + return Err("Configuration failed".into()); + } + session.set_preceding_text("")?; + let startup = start.elapsed().as_secs_f64() * 1000.0; + let startup_resident = peak_resident_bytes()?; + let mut timings = Vec::new(); + for pass in 0..5 { + for sample in samples { + let id = sample["id"].as_str().ok_or("sample")?; + let input = sample["input"].as_str().ok_or("sample")?; + session.clear()?; + session.set_preceding_text("")?; + for (offset, byte) in input.bytes().enumerate() { + let began = Instant::now(); + let handled = session.key(i32::from(byte), i32::from(byte.is_ascii_uppercase()))?; + let commit = session.take_commit()?; + let snapshot = session.snapshot()?; + let elapsed = began.elapsed().as_secs_f64() * 1000.0; + if !handled || !commit.is_empty() || snapshot.preedit.is_empty() { + return Err(format!("Unexpected input result in {id} at {offset}").into()); + } + timings.push( + json!({"pass": pass, "sample": id, "offset": offset, "milliseconds": elapsed}), + ); + } + } + } + session.clear()?; + let report = json!({ + "formatVersion": 1, + "startupMilliseconds": startup, + "peakResidentBytesAfterStartup": startup_resident, + "peakResidentBytesAfterInput": peak_resident_bytes()?, + "inputOptions": InputPreferences::default() + .recorded_values() + .into_iter() + .map(|(name, value)| (name.to_owned(), json!(value))) + .collect::>(), + "candidateCount": count, + "timings": timings, + }); + fs::write(output, serde_json::to_vec_pretty(&report)?)?; + println!( + "PASS performance baseline: {} key operations; startup_ms={startup}", + report["timings"].as_array().unwrap().len() + ); + Ok(()) +} + +fn main() { + if let Err(error) = run() { + eprintln!("FAIL performance baseline: {error}"); + std::process::exit(1); + } +} From bf31bc38061a4ce6dff1e7db8cf4a7c0237127bb Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 00:14:56 +0800 Subject: [PATCH 13/21] Export the frontend-facing C ABI of the Rust engine Add src/abi.rs and include/inkflow_rime.h over Engine/InputSession: engine and session lifetime, Rime keysym events, immutable snapshots with UTF-8 byte offsets, token-identified selection and highlight, paging, commit draining, preceding text, configuration (candidate count, input option bitmask, custom phrases), ASCII mode and the mutation observer as a C callback delivered outside the engine lock. Every export catches panics, writes outputs only on success and reports a thread-local message; handles have no thread affinity. Build the crate as cdylib and staticlib beside the rlib. abi.sh compiles tests/abi.c against the shared library: the fixture part drives the real runtime lifetime and resource validation with the probe schema, the production part needs prepared resources. test.sh runs the former and parity.sh both. Refs #36 --- Core/Portable/Cargo.toml | 1 + Core/Portable/README.md | 4 +- Core/Portable/abi.sh | 23 + Core/Portable/include/inkflow_rime.h | 192 ++++++++ Core/Portable/parity.sh | 1 + Core/Portable/src/abi.rs | 700 +++++++++++++++++++++++++++ Core/Portable/src/lib.rs | 1 + Core/Portable/test.sh | 1 + Core/Portable/tests/abi.c | 225 +++++++++ 9 files changed, 1147 insertions(+), 1 deletion(-) create mode 100755 Core/Portable/abi.sh create mode 100644 Core/Portable/include/inkflow_rime.h create mode 100644 Core/Portable/src/abi.rs create mode 100644 Core/Portable/tests/abi.c diff --git a/Core/Portable/Cargo.toml b/Core/Portable/Cargo.toml index 52bbb43..f35a8af 100644 --- a/Core/Portable/Cargo.toml +++ b/Core/Portable/Cargo.toml @@ -7,6 +7,7 @@ build = "build.rs" [lib] path = "src/lib.rs" +crate-type = ["rlib", "cdylib", "staticlib"] [dependencies] serde_json = "=1.0.151" diff --git a/Core/Portable/README.md b/Core/Portable/README.md index b302e95..0851747 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -69,7 +69,9 @@ Key latency is the same Rime work in both engines; the Rust engine's lower start ## Interface contract -`native/bridge.h` is the experimental internal C ABI between Rust and the native runtime. It uses Rime's public C API for lifecycle, deployment, input, context, and commits. C++ internals are confined to the existing extension and its registration check. The Rust library exposes safe `Runtime` and `Session` handles; a stable frontend-facing exported Rust C ABI is deferred until session policy is migrated. +`native/bridge.h` is the experimental internal C ABI between Rust and the native runtime. It uses Rime's public C API for lifecycle, deployment, input, context, and commits. C++ internals are confined to the existing extension and its registration check. The Rust library exposes safe `Runtime` and `Session` handles. + +`include/inkflow_rime.h` is the frontend-facing C ABI over `Engine`/`InputSession` (`src/abi.rs`, ABI version 1), built as `cdylib` and `staticlib` beside the rlib. It covers engine and session lifetime, keys, immutable snapshots with UTF-8 byte offsets, token-identified selection and highlight, paging, commit draining, preceding text, configuration (candidate count, input options as a bitmask, custom phrases), ASCII mode and the mutation observer as a C callback. Every export catches panics and returns a status with a thread-local message; outputs are written only on success; there is no thread affinity. `abi.sh` builds both libraries and runs `tests/abi.c`: the fixture part exercises the real runtime lifetime and resource validation with the tiny probe schema (the engine itself needs the production schema), and the production part needs prepared resources; `test.sh` runs the former, `parity.sh` both. Personal-learning management and personal-data import are not exported yet. - One process-wide Rust mutex serializes **all** Rime calls, including initialization, deployment, reads, destruction, and finalization. The engine's policy lock serializes every `InputSession` of an `Engine` and nests outside that mutex. Sessions can move between threads. No GUI event loop, Swift actor, or thread affinity is required. Another engine must not call librime outside this lock in the same process; old/new comparisons need separate processes. - There is at most one runtime. A second initialization returns `AlreadyRunning`. Sessions retain an `Arc` to the runtime, so dropping its public handle cannot finalize live sessions. The last session/runtime owner finalizes Rime. A poisoned lock fails subsequent normal operations; destructors still attempt cleanup without panicking. diff --git a/Core/Portable/abi.sh b/Core/Portable/abi.sh new file mode 100755 index 0000000..582d3fd --- /dev/null +++ b/Core/Portable/abi.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# Build the exported C ABI (cdylib + staticlib) and run the C consumer. Without a +# prepared resources directory only the fixture and argument checks run. +set -euo pipefail +cd "$(dirname "$0")/../.." +resources=${1:-} +if [[ -n "$resources" ]]; then + [[ -f "$resources/prepared/complete" ]] || { echo "Not a prepared resources directory: $resources" >&2; exit 2; } + resources=$(cd "$resources" && pwd) +fi +export CARGO_TARGET_DIR="$PWD/build/portable/cargo" +cargo build --locked --manifest-path Core/Portable/Cargo.toml +lib="$CARGO_TARGET_DIR/debug" +[[ -f "$lib/libinkflow_rime.a" ]] +scratch=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-abi.XXXXXX") +trap 'rm -rf "$scratch"' EXIT +mkdir -p "$scratch/fixture/shared/lua" "$scratch/work" +for file in default.yaml probe.schema.yaml probe.dict.yaml lua/probe.lua; do + cp "Core/Portable/fixtures/$file" "$scratch/fixture/shared/$file" +done +cc -std=c11 -Wall -Wextra -Werror -I Core/Portable/include Core/Portable/tests/abi.c \ + -L"$lib" -linkflow_rime -Wl,-rpath,"$lib" -o "$scratch/abi-test" +"$scratch/abi-test" "$scratch/fixture" "$scratch/work" ${resources:+"$resources"} diff --git a/Core/Portable/include/inkflow_rime.h b/Core/Portable/include/inkflow_rime.h new file mode 100644 index 0000000..476c61e --- /dev/null +++ b/Core/Portable/include/inkflow_rime.h @@ -0,0 +1,192 @@ +#ifndef INKFLOW_RIME_H +#define INKFLOW_RIME_H +#include +#include +#ifdef __cplusplus +extern "C" { +#endif + +/* Frontend-facing ABI of the InkFlow Rust/Rime engine (ABI version 1). + * + * Handles: IFREngine, IFRSession and IFRSnapshot are opaque. Each is created by one + * function and released by its matching destroy or free function, which accepts NULL. Owned + * strings from ifr_session_take_commit are released with ifr_string_free. Borrowed + * strings (snapshot accessors, ifr_last_error, configuration errors) stay valid while + * their owner lives; ifr_last_error is valid on the calling thread until its next + * failing call. Every string crossing the ABI is NUL-terminated UTF-8; embedded NUL + * and non-UTF-8 input are rejected with IFR_INVALID_ARGUMENT. + * + * Status: functions returning IFRStatus write their outputs only on IFR_OK and set + * ifr_last_error otherwise. Accessors that cannot fail return neutral values for NULL + * or out-of-range arguments. No Rust panic unwinds across the ABI: a caught panic is + * IFR_PANIC and the engine should be destroyed. Invalid pointers remain undefined. + * + * Threads: there is no thread affinity. One process-wide lock serializes every Rime + * call and one policy lock serializes every session of an engine, so all handles may + * be used from any thread, one call at a time per handle. Rime is process-global: a + * second live engine returns IFR_ALREADY_RUNNING. Sessions keep their engine alive; + * ifr_engine_destroy only drops the caller's reference, and Rime finalizes when the + * last session or engine reference goes. + * + * Offsets: ifr_snapshot_caret/selection_* count bytes of the returned UTF-8 preedit + * and always fall on character boundaries. Frontends convert to their own units. + * + * Keys: Rime/X11 keysyms with Rime modifier masks (Shift 1<<0, Control 1<<2, + * Alt 1<<3, Super 1<<26, Release 1<<30). Digits select displayed candidates, Up/Down + * move the highlight, Page_Up/Page_Down page; everything else goes to Rime. + * + * Candidate selection and highlight take the snapshot they were decided on. A + * snapshot superseded by any mutation returns IFR_STALE_SNAPSHOT; an index outside its + * page returns IFR_INVALID_CANDIDATE. Snapshots are copies and outlive the session. + * + * Commits: ifr_session_take_commit drains completed text exactly once. Call it after + * every key, selection, commit or configuration change before the next action. + * + * Nothing here touches the network or telemetry. Engine creation, configuration + * changes and the custom-phrase reload at a shared idle are the only disk work; + * keys never read storage. The observer runs on the mutating thread after the engine + * lock is released; it must not call back into the same session and cannot block + * input on storage without blocking that thread. */ + +typedef int32_t IFRStatus; +enum { + IFR_OK = 0, + IFR_PANIC = 1, + IFR_INVALID_ARGUMENT = 2, + IFR_NATIVE = 3, /* librime failed; stop using the engine and recreate it. */ + IFR_POISONED = 4, /* An earlier panic poisoned a lock; recreate the engine. */ + IFR_ALREADY_RUNNING = 5, /* Another engine (or its sessions) is still alive. */ + IFR_SESSIONS_ACTIVE = 6, + IFR_RESOURCE = 7, /* A prepared resource or the context index is missing/invalid. */ + IFR_CONFIGURATION = 8, /* A session could not apply its configuration. */ + IFR_IO = 9, + IFR_STALE_SNAPSHOT = 10, + IFR_INVALID_CANDIDATE = 11 +}; + +typedef struct IFREngine IFREngine; +typedef struct IFRSession IFRSession; +typedef struct IFRSnapshot IFRSnapshot; + +typedef struct { + const char* shared; /* Prepared shared resources. */ + const char* user; /* Isolated writable user directory; created if missing. */ + const char* cache; /* Target-native compiled cache, or NULL to compile into user/build. */ + const char* context_index; /* Prepared pinyin_simp.context.bin, or NULL for Rime's order. */ +} IFREngineConfig; + +typedef struct { + const char* id; /* Opaque frontend identity, unique within one configuration. */ + const char* code; + const char* text; +} IFRPhrase; + +/* Input options as a bitmask; ifr_input_options_default() gives the product defaults. */ +enum { + IFR_OPTION_ABBREVIATION = 1u << 0, + IFR_OPTION_TYPO_TOLERANCE = 1u << 1, + IFR_OPTION_FUZZY_Z = 1u << 2, + IFR_OPTION_FUZZY_C = 1u << 3, + IFR_OPTION_FUZZY_S = 1u << 4, + IFR_OPTION_EMOJI = 1u << 5, + IFR_OPTION_BRACKET_PAGING = 1u << 6, + IFR_OPTION_MINUS_EQUAL_PAGING = 1u << 7, + IFR_OPTION_ENGLISH_PUNCTUATION = 1u << 8, + IFR_OPTION_CORNER_QUOTES = 1u << 9, + IFR_OPTION_MIDDLE_DOT = 1u << 10, + IFR_OPTION_FULLWIDTH_PIPE = 1u << 11, + IFR_OPTION_IDEOGRAPHIC_COMMA = 1u << 12, + IFR_OPTION_TRADITIONAL = 1u << 13 +}; + +enum { + IFR_ACTION_KEY = 0, + IFR_ACTION_SELECT = 1, + IFR_ACTION_HIGHLIGHT = 2, + IFR_ACTION_COMMIT = 3, + IFR_ACTION_CLEAR = 4, + IFR_ACTION_TOGGLE_ASCII = 5 +}; + +/* One completed mutation. Snapshots are borrowed for the callback only. */ +typedef struct { + uint64_t session; + int action; + int32_t key; /* IFR_ACTION_KEY */ + int32_t modifiers; /* IFR_ACTION_KEY */ + size_t index; /* IFR_ACTION_SELECT / IFR_ACTION_HIGHLIGHT */ + int handled; + const IFRSnapshot* before; + const IFRSnapshot* after; +} IFRMutation; + +typedef void (*IFRObserver)(void* user_data, const IFRMutation* mutation); + +uint32_t ifr_abi_version(void); +const char* ifr_last_error(void); +uint32_t ifr_input_options_default(void); + +/* Engine creation initializes Rime, checks the prepared cache and loads the context + * index: setup work, never on the key path. */ +IFRStatus ifr_engine_create(const IFREngineConfig* config, IFREngine** engine); +void ifr_engine_destroy(IFREngine* engine); +int ifr_engine_context_ranking_ready(const IFREngine* engine); +/* NULL removes the observer. The callback and user_data must be usable from any + * thread that mutates a session of this engine. */ +IFRStatus ifr_engine_set_observer(const IFREngine* engine, IFRObserver callback, void* user_data); + +IFRStatus ifr_session_create(const IFREngine* engine, IFRSession** session); +void ifr_session_destroy(IFRSession* session); +uint64_t ifr_session_id(const IFRSession* session); +int ifr_session_available(const IFRSession* session); + +/* handled is 1 when the engine consumed the key. Output pointers may be NULL. */ +IFRStatus ifr_session_key(const IFRSession* session, int32_t key, int32_t modifiers, int* handled); +IFRStatus ifr_session_snapshot(const IFRSession* session, IFRSnapshot** snapshot); +void ifr_snapshot_free(IFRSnapshot* snapshot); +const char* ifr_snapshot_input(const IFRSnapshot* snapshot); /* Rime's raw input. */ +const char* ifr_snapshot_preedit(const IFRSnapshot* snapshot); +size_t ifr_snapshot_caret(const IFRSnapshot* snapshot); +size_t ifr_snapshot_selection_start(const IFRSnapshot* snapshot); +size_t ifr_snapshot_selection_end(const IFRSnapshot* snapshot); +size_t ifr_snapshot_candidate_count(const IFRSnapshot* snapshot); +const char* ifr_snapshot_candidate_text(const IFRSnapshot* snapshot, size_t index); +const char* ifr_snapshot_candidate_comment(const IFRSnapshot* snapshot, size_t index); +int32_t ifr_snapshot_page(const IFRSnapshot* snapshot); +size_t ifr_snapshot_highlighted(const IFRSnapshot* snapshot); /* Display index. */ +int ifr_snapshot_last_page(const IFRSnapshot* snapshot); + +IFRStatus ifr_session_select(const IFRSession* session, const IFRSnapshot* snapshot, + size_t index, int* handled); +IFRStatus ifr_session_highlight(const IFRSession* session, const IFRSnapshot* snapshot, + size_t index); +IFRStatus ifr_session_change_page(const IFRSession* session, int backward, int* handled); +IFRStatus ifr_session_commit(const IFRSession* session, int* handled); +IFRStatus ifr_session_clear(const IFRSession* session); +/* *commit receives an owned string, or NULL when nothing completed. */ +IFRStatus ifr_session_take_commit(const IFRSession* session, char** commit); +void ifr_string_free(char* string); + +/* Bounded document text before the caret; only its last graphemes are kept. */ +IFRStatus ifr_session_set_preceding_text(const IFRSession* session, const char* text); +/* candidate_count outside 3..9 falls back to 5. options NULL keeps the current + * preferences. Phrases are validated; failures are reported by + * ifr_session_configuration_error, not by the status. Settings apply at this + * session's next composition boundary; phrases apply to every idle session. */ +IFRStatus ifr_session_set_configuration(const IFRSession* session, size_t candidate_count, + const IFRPhrase* phrases, size_t phrase_count, + const uint32_t* options); +/* A stable code (invalid-code, duplicate, phrase-write, missing-prism, ...) or NULL. */ +const char* ifr_session_configuration_error(const IFRSession* session); +size_t ifr_session_candidate_count(const IFRSession* session); +/* Returns 1 and writes the applied options once a composition boundary accepted them. */ +int ifr_session_input_options(const IFRSession* session, uint32_t* options); + +IFRStatus ifr_session_set_ascii_mode(const IFRSession* session, int value); +IFRStatus ifr_session_ascii_mode(const IFRSession* session, int* value); +IFRStatus ifr_session_toggle_ascii_mode(const IFRSession* session, int* handled); + +#ifdef __cplusplus +} +#endif +#endif diff --git a/Core/Portable/parity.sh b/Core/Portable/parity.sh index d51344f..e669e6b 100755 --- a/Core/Portable/parity.sh +++ b/Core/Portable/parity.sh @@ -13,3 +13,4 @@ fi resources=$(cd "$resources" && pwd) export CARGO_TARGET_DIR="$PWD/build/portable/cargo" INKFLOW_PORTABLE_RESOURCES="$resources" cargo test --locked --manifest-path Core/Portable/Cargo.toml --test parity -- --nocapture +bash Core/Portable/abi.sh "$resources" diff --git a/Core/Portable/src/abi.rs b/Core/Portable/src/abi.rs new file mode 100644 index 0000000..30700fc --- /dev/null +++ b/Core/Portable/src/abi.rs @@ -0,0 +1,700 @@ +//! The frontend-facing C ABI over [`crate::engine`]. `include/inkflow_rime.h` is the +//! hand-maintained contract; keep both in step. +//! +//! Every export catches panics and returns a status; nothing unwinds across the ABI. +//! Handles are opaque boxes, outputs are written only on `IFR_OK`, and every call is +//! serialized by the engine's own locks, so handles may be used from any thread. +use crate::{ + Error, + engine::{ + Action, Configuration, ConfigurationError, Engine, EngineError, InputSession, Mutation, + Snapshot, + }, + phrases::CustomPhrase, + preferences::{InputOption, InputPreferences}, +}; +use std::{ + cell::RefCell, + ffi::{CStr, CString, c_char, c_int, c_void}, + panic::{AssertUnwindSafe, catch_unwind}, + path::PathBuf, + ptr, + sync::Arc, +}; + +pub const ABI_VERSION: u32 = 1; + +pub type Status = i32; +pub const OK: Status = 0; +pub const PANIC: Status = 1; +pub const INVALID_ARGUMENT: Status = 2; +pub const NATIVE: Status = 3; +pub const POISONED: Status = 4; +pub const ALREADY_RUNNING: Status = 5; +pub const SESSIONS_ACTIVE: Status = 6; +pub const RESOURCE: Status = 7; +pub const CONFIGURATION: Status = 8; +pub const IO: Status = 9; +pub const STALE_SNAPSHOT: Status = 10; +pub const INVALID_CANDIDATE: Status = 11; + +pub struct Failure { + status: Status, + message: String, +} +impl Failure { + fn new(status: Status, message: impl Into) -> Self { + Self { + status, + message: message.into(), + } + } + fn argument(message: &str) -> Self { + Self::new(INVALID_ARGUMENT, message) + } +} +impl From for Failure { + fn from(error: Error) -> Self { + let status = match error { + Error::AlreadyRunning => ALREADY_RUNNING, + Error::SessionsActive => SESSIONS_ACTIVE, + Error::InvalidString => INVALID_ARGUMENT, + Error::InvalidUtf8 | Error::InvalidOffsets | Error::Native(_) => NATIVE, + Error::StaleSnapshot => STALE_SNAPSHOT, + Error::InvalidCandidate => INVALID_CANDIDATE, + Error::Poisoned => POISONED, + Error::Configuration(_) => CONFIGURATION, + }; + Self::new(status, error.to_string()) + } +} +impl From for Failure { + fn from(error: EngineError) -> Self { + let message = error.to_string(); + let status = match error { + EngineError::Native(error) => return Failure::from(error), + EngineError::MissingResource(_) | EngineError::ContextIndex(_) => RESOURCE, + EngineError::Configuration(_) => CONFIGURATION, + EngineError::Io(_) => IO, + }; + Self::new(status, message) + } +} + +thread_local! { + static LAST_ERROR: RefCell = RefCell::new(CString::default()); +} + +/// Run one export body: panics and failures become a status and a thread-local message. +fn guard(work: impl FnOnce() -> Result<(), Failure>) -> Status { + let failure = match catch_unwind(AssertUnwindSafe(work)) { + Ok(Ok(())) => return OK, + Ok(Err(failure)) => failure, + Err(_) => Failure::new(PANIC, "panic"), + }; + LAST_ERROR.with(|last| { + *last.borrow_mut() = CString::new(failure.message.replace('\0', "?")).unwrap(); + }); + failure.status +} + +unsafe fn text<'a>(pointer: *const c_char, name: &str) -> Result<&'a str, Failure> { + if pointer.is_null() { + return Err(Failure::argument(&format!("{name} is NULL"))); + } + unsafe { CStr::from_ptr(pointer) } + .to_str() + .map_err(|_| Failure::argument(&format!("{name} is not UTF-8"))) +} + +unsafe fn optional_path(pointer: *const c_char, name: &str) -> Result, Failure> { + if pointer.is_null() { + return Ok(None); + } + Ok(Some(PathBuf::from(unsafe { text(pointer, name) }?))) +} + +unsafe fn reference<'a, T>(pointer: *const T, name: &str) -> Result<&'a T, Failure> { + // Handles come from this module's boxes; the caller keeps them alive for the call. + unsafe { pointer.as_ref() }.ok_or_else(|| Failure::argument(&format!("{name} is NULL"))) +} + +unsafe fn write(pointer: *mut T, value: T) { + if !pointer.is_null() { + unsafe { ptr::write(pointer, value) }; + } +} + +fn c_string(value: &str) -> Result { + CString::new(value).map_err(|_| Failure::new(NATIVE, "embedded NUL in engine text")) +} + +/// Input options as a bitmask in `InputOption::ALL` order. +fn options_from_mask(mask: u32) -> InputPreferences { + InputOption::ALL + .iter() + .enumerate() + .fold(InputPreferences::default(), |prefs, (bit, option)| { + prefs.with(*option, mask & (1 << bit) != 0) + }) +} + +fn mask_from_options(input: &InputPreferences) -> u32 { + InputOption::ALL + .iter() + .enumerate() + .fold(0, |mask, (bit, option)| mask | (u32::from(input.get(*option)) << bit)) +} + +fn configuration_code(error: ConfigurationError) -> &'static CStr { + use crate::phrases::PhraseError; + match error { + ConfigurationError::Phrases(PhraseError::InvalidCode) => c"invalid-code", + ConfigurationError::Phrases(PhraseError::ControlCharacter) => c"control-character", + ConfigurationError::Phrases(PhraseError::EmptyText) => c"empty-text", + ConfigurationError::Phrases(PhraseError::InvalidData) => c"invalid-data", + ConfigurationError::Phrases(PhraseError::Duplicate) => c"duplicate", + ConfigurationError::PhraseWrite => c"phrase-write", + ConfigurationError::SessionReload => c"session-reload", + ConfigurationError::MissingPrism => c"missing-prism", + ConfigurationError::SchemaOpen => c"schema-open", + ConfigurationError::PatchInit => c"patch-init", + ConfigurationError::PatchLoad => c"patch-load", + ConfigurationError::SchemaRestore => c"schema-restore", + ConfigurationError::SchemaApply => c"schema-apply", + } +} + +#[repr(C)] +pub struct EngineConfig { + pub shared: *const c_char, + pub user: *const c_char, + pub cache: *const c_char, + pub context_index: *const c_char, +} + +#[repr(C)] +pub struct Phrase { + pub id: *const c_char, + pub code: *const c_char, + pub text: *const c_char, +} + +/// An immutable display snapshot with NUL-terminated copies of every string. +pub struct SnapshotHandle { + snapshot: Snapshot, + input: CString, + preedit: CString, + candidates: Vec<(CString, CString)>, +} + +impl SnapshotHandle { + fn new(snapshot: Snapshot) -> Result { + let input = c_string(&snapshot.input)?; + let preedit = c_string(&snapshot.preedit)?; + let candidates = snapshot + .candidates + .iter() + .map(|c| Ok((c_string(&c.text)?, c_string(&c.comment)?))) + .collect::, Failure>>()?; + Ok(Self { + snapshot, + input, + preedit, + candidates, + }) + } +} + +pub const ACTION_KEY: c_int = 0; +pub const ACTION_SELECT: c_int = 1; +pub const ACTION_HIGHLIGHT: c_int = 2; +pub const ACTION_COMMIT: c_int = 3; +pub const ACTION_CLEAR: c_int = 4; +pub const ACTION_TOGGLE_ASCII: c_int = 5; + +#[repr(C)] +pub struct MutationRecord { + pub session: u64, + pub action: c_int, + pub key: i32, + pub modifiers: i32, + pub index: usize, + pub handled: c_int, + pub before: *const SnapshotHandle, + pub after: *const SnapshotHandle, +} + +pub type ObserverCallback = unsafe extern "C" fn(user_data: *mut c_void, mutation: *const MutationRecord); + +struct Observer { + callback: ObserverCallback, + user_data: *mut c_void, +} +// The caller promises the callback and its data may be used from any mutating thread. +unsafe impl Send for Observer {} +unsafe impl Sync for Observer {} + +impl Observer { + fn deliver(&self, mutation: &Mutation) { + let (Ok(before), Ok(after)) = ( + SnapshotHandle::new(mutation.before.clone()), + SnapshotHandle::new(mutation.after.clone()), + ) else { + return; + }; + let (action, key, modifiers, index) = match mutation.action { + Action::Key { key, modifiers } => (ACTION_KEY, key, modifiers, 0), + Action::Select(index) => (ACTION_SELECT, 0, 0, index), + Action::Highlight(index) => (ACTION_HIGHLIGHT, 0, 0, index), + Action::Commit => (ACTION_COMMIT, 0, 0, 0), + Action::Clear => (ACTION_CLEAR, 0, 0, 0), + Action::ToggleAscii => (ACTION_TOGGLE_ASCII, 0, 0, 0), + }; + let record = MutationRecord { + session: mutation.session, + action, + key, + modifiers, + index, + handled: c_int::from(mutation.handled), + before: &before, + after: &after, + }; + unsafe { (self.callback)(self.user_data, &record) }; + } +} + +#[unsafe(no_mangle)] +pub extern "C" fn ifr_abi_version() -> u32 { + ABI_VERSION +} + +#[unsafe(no_mangle)] +pub extern "C" fn ifr_last_error() -> *const c_char { + LAST_ERROR.with(|last| last.borrow().as_ptr()) +} + +#[unsafe(no_mangle)] +pub extern "C" fn ifr_input_options_default() -> u32 { + mask_from_options(&InputPreferences::default()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_engine_create( + config: *const EngineConfig, + engine: *mut *mut Arc, +) -> Status { + guard(|| { + let config = unsafe { reference(config, "config") }?; + if engine.is_null() { + return Err(Failure::argument("engine is NULL")); + } + let configuration = Configuration { + shared: PathBuf::from(unsafe { text(config.shared, "shared") }?), + user: PathBuf::from(unsafe { text(config.user, "user") }?), + cache: unsafe { optional_path(config.cache, "cache") }?, + context_index: unsafe { optional_path(config.context_index, "context_index") }?, + }; + let created = Engine::new(configuration)?; + unsafe { write(engine, Box::into_raw(Box::new(created))) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_engine_destroy(engine: *mut Arc) { + if !engine.is_null() { + let _ = catch_unwind(AssertUnwindSafe(|| drop(unsafe { Box::from_raw(engine) }))); + } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_engine_context_ranking_ready(engine: *const Arc) -> c_int { + let mut ready = 0; + guard(|| { + ready = c_int::from(unsafe { reference(engine, "engine") }?.context_ranking_ready()); + Ok(()) + }); + ready +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_engine_set_observer( + engine: *const Arc, + callback: Option, + user_data: *mut c_void, +) -> Status { + guard(|| { + let engine = unsafe { reference(engine, "engine") }?; + engine.set_observer(callback.map(|callback| { + let observer = Observer { + callback, + user_data, + }; + Arc::new(move |mutation: &Mutation| observer.deliver(mutation)) + as Arc + })); + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_create( + engine: *const Arc, + session: *mut *mut InputSession, +) -> Status { + guard(|| { + let engine = unsafe { reference(engine, "engine") }?; + if session.is_null() { + return Err(Failure::argument("session is NULL")); + } + let created = engine.session()?; + unsafe { write(session, Box::into_raw(Box::new(created))) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_destroy(session: *mut InputSession) { + if !session.is_null() { + let _ = catch_unwind(AssertUnwindSafe(|| drop(unsafe { Box::from_raw(session) }))); + } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_id(session: *const InputSession) -> u64 { + let mut id = 0; + guard(|| { + id = unsafe { reference(session, "session") }?.id(); + Ok(()) + }); + id +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_available(session: *const InputSession) -> c_int { + let mut available = 0; + guard(|| { + available = c_int::from(unsafe { reference(session, "session") }?.available()); + Ok(()) + }); + available +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_key( + session: *const InputSession, + key: i32, + modifiers: i32, + handled: *mut c_int, +) -> Status { + guard(|| { + let result = unsafe { reference(session, "session") }?.key(key, modifiers)?; + unsafe { write(handled, c_int::from(result)) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_snapshot( + session: *const InputSession, + snapshot: *mut *mut SnapshotHandle, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + if snapshot.is_null() { + return Err(Failure::argument("snapshot is NULL")); + } + let handle = SnapshotHandle::new(session.snapshot()?)?; + unsafe { write(snapshot, Box::into_raw(Box::new(handle))) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_free(snapshot: *mut SnapshotHandle) { + if !snapshot.is_null() { + let _ = catch_unwind(AssertUnwindSafe(|| drop(unsafe { Box::from_raw(snapshot) }))); + } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_input(snapshot: *const SnapshotHandle) -> *const c_char { + unsafe { snapshot.as_ref() }.map_or(ptr::null(), |s| s.input.as_ptr()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_preedit(snapshot: *const SnapshotHandle) -> *const c_char { + unsafe { snapshot.as_ref() }.map_or(ptr::null(), |s| s.preedit.as_ptr()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_caret(snapshot: *const SnapshotHandle) -> usize { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.snapshot.caret_bytes) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_selection_start(snapshot: *const SnapshotHandle) -> usize { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.snapshot.selection_bytes.start) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_selection_end(snapshot: *const SnapshotHandle) -> usize { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.snapshot.selection_bytes.end) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_candidate_count(snapshot: *const SnapshotHandle) -> usize { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.candidates.len()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_candidate_text( + snapshot: *const SnapshotHandle, + index: usize, +) -> *const c_char { + unsafe { snapshot.as_ref() } + .and_then(|s| s.candidates.get(index)) + .map_or(ptr::null(), |(text, _)| text.as_ptr()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_candidate_comment( + snapshot: *const SnapshotHandle, + index: usize, +) -> *const c_char { + unsafe { snapshot.as_ref() } + .and_then(|s| s.candidates.get(index)) + .map_or(ptr::null(), |(_, comment)| comment.as_ptr()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_page(snapshot: *const SnapshotHandle) -> i32 { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.snapshot.page) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_highlighted(snapshot: *const SnapshotHandle) -> usize { + unsafe { snapshot.as_ref() }.map_or(0, |s| s.snapshot.highlighted) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_snapshot_last_page(snapshot: *const SnapshotHandle) -> c_int { + unsafe { snapshot.as_ref() }.map_or(1, |s| c_int::from(s.snapshot.last_page)) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_select( + session: *const InputSession, + snapshot: *const SnapshotHandle, + index: usize, + handled: *mut c_int, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + let snapshot = unsafe { reference(snapshot, "snapshot") }?; + let result = session.select(&snapshot.snapshot, index)?; + unsafe { write(handled, c_int::from(result)) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_highlight( + session: *const InputSession, + snapshot: *const SnapshotHandle, + index: usize, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + let snapshot = unsafe { reference(snapshot, "snapshot") }?; + Ok(session.highlight(&snapshot.snapshot, index)?) + }) +} + +/// Paging is Rime's Page_Up/Page_Down through the same key policy as a frontend key. +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_change_page( + session: *const InputSession, + backward: c_int, + handled: *mut c_int, +) -> Status { + let key = if backward != 0 { 0xff55 } else { 0xff56 }; + unsafe { ifr_session_key(session, key, 0, handled) } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_commit( + session: *const InputSession, + handled: *mut c_int, +) -> Status { + guard(|| { + let result = unsafe { reference(session, "session") }?.commit()?; + unsafe { write(handled, c_int::from(result)) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_clear(session: *const InputSession) -> Status { + guard(|| Ok(unsafe { reference(session, "session") }?.clear()?)) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_take_commit( + session: *const InputSession, + commit: *mut *mut c_char, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + if commit.is_null() { + return Err(Failure::argument("commit is NULL")); + } + let text = session.take_commit()?; + let owned = if text.is_empty() { + ptr::null_mut() + } else { + c_string(&text)?.into_raw() + }; + unsafe { write(commit, owned) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_string_free(string: *mut c_char) { + if !string.is_null() { + drop(unsafe { CString::from_raw(string) }); + } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_set_preceding_text( + session: *const InputSession, + text: *const c_char, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + let text = unsafe { self::text(text, "text") }?; + Ok(session.set_preceding_text(text)?) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_set_configuration( + session: *const InputSession, + candidate_count: usize, + phrases: *const Phrase, + phrase_count: usize, + options: *const u32, +) -> Status { + guard(|| { + let session = unsafe { reference(session, "session") }?; + if phrase_count != 0 && phrases.is_null() { + return Err(Failure::argument("phrases is NULL")); + } + if phrase_count > 65_536 { + return Err(Failure::argument("too many phrases")); + } + let mut owned = Vec::with_capacity(phrase_count); + for index in 0..phrase_count { + let phrase = unsafe { &*phrases.add(index) }; + owned.push(CustomPhrase { + id: unsafe { text(phrase.id, "phrase id") }?.to_owned(), + code: unsafe { text(phrase.code, "phrase code") }?.to_owned(), + text: unsafe { text(phrase.text, "phrase text") }?.to_owned(), + }); + } + let input = unsafe { options.as_ref() }.map(|mask| options_from_mask(*mask)); + Ok(session.set_configuration(candidate_count, &owned, input.as_ref())?) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_configuration_error( + session: *const InputSession, +) -> *const c_char { + let mut code = ptr::null(); + guard(|| { + if let Some(error) = unsafe { reference(session, "session") }?.configuration_error() { + code = configuration_code(error).as_ptr(); + } + Ok(()) + }); + code +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_candidate_count(session: *const InputSession) -> usize { + let mut count = 5; + guard(|| { + count = unsafe { reference(session, "session") }?.candidate_count(); + Ok(()) + }); + count +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_input_options( + session: *const InputSession, + options: *mut u32, +) -> c_int { + let mut applied = 0; + guard(|| { + if let Some(input) = unsafe { reference(session, "session") }?.input_preferences() { + unsafe { write(options, mask_from_options(&input)) }; + applied = 1; + } + Ok(()) + }); + applied +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_set_ascii_mode( + session: *const InputSession, + value: c_int, +) -> Status { + guard(|| Ok(unsafe { reference(session, "session") }?.set_ascii_mode(value != 0)?)) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_ascii_mode( + session: *const InputSession, + value: *mut c_int, +) -> Status { + guard(|| { + let result = unsafe { reference(session, "session") }?.ascii_mode()?; + unsafe { write(value, c_int::from(result)) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_session_toggle_ascii_mode( + session: *const InputSession, + handled: *mut c_int, +) -> Status { + guard(|| { + let result = unsafe { reference(session, "session") }?.toggle_ascii_mode()?; + unsafe { write(handled, c_int::from(result)) }; + Ok(()) + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn option_masks_round_trip() { + let defaults = InputPreferences::default(); + assert_eq!(options_from_mask(ifr_input_options_default()), defaults); + let traditional = defaults.with(InputOption::Traditional, true); + let mask = mask_from_options(&traditional); + assert_eq!(mask, ifr_input_options_default() | (1 << 13)); + assert_eq!(options_from_mask(mask), traditional); + assert_eq!(configuration_code(ConfigurationError::PhraseWrite), c"phrase-write"); + } +} diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index c36b3c9..4af4107 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -1,5 +1,6 @@ //! Synchronous desktop Rime host: the native `Runtime`/`Session` layer plus the //! InkFlow session policy in [`engine`]. +pub mod abi; pub mod channel; pub mod engine; mod ffi; diff --git a/Core/Portable/test.sh b/Core/Portable/test.sh index 1abc414..085ee00 100644 --- a/Core/Portable/test.sh +++ b/Core/Portable/test.sh @@ -14,6 +14,7 @@ print('PASS fresh Swift ranking reference matches recorded cases') PY fi cargo test --locked --manifest-path Core/Portable/Cargo.toml -- --nocapture +bash Core/Portable/abi.sh if [[ $# -eq 1 ]]; then cp build/portable/native-build.json "$1/native-build.json" if [[ $(uname -s) == Darwin ]]; then cp build/portable/ranking-reference.json "$1/ranking-reference.json"; fi diff --git a/Core/Portable/tests/abi.c b/Core/Portable/tests/abi.c new file mode 100644 index 0000000..d420b6f --- /dev/null +++ b/Core/Portable/tests/abi.c @@ -0,0 +1,225 @@ +/* C consumer of the frontend ABI. Arguments: FIXTURE_ROOT SCRATCH [RESOURCES]. + * FIXTURE_ROOT holds shared/ with the tiny probe schema; RESOURCES holds prepared + * production resources (shared/ and prepared/cache/). */ +#include "inkflow_rime.h" +#include +#include +#include +#include + +static char buffer[8][4096]; +static const char* path(int slot, const char* root, const char* tail) { + snprintf(buffer[slot], sizeof(buffer[slot]), "%s/%s", root, tail); + return buffer[slot]; +} + +static void expect(IFRStatus status, IFRStatus wanted, const char* what) { + if (status != wanted) { + fprintf(stderr, "%s: status %d (wanted %d): %s\n", what, status, wanted, ifr_last_error()); + exit(1); + } +} + +static void type_text(const IFRSession* session, const char* text) { + for (const char* p = text; *p; ++p) { + int handled = 0; + expect(ifr_session_key(session, *p, 0, &handled), IFR_OK, "key"); + assert(handled == 1); + } +} + +static char* take(const IFRSession* session) { + char* commit = (char*)1; + expect(ifr_session_take_commit(session, &commit), IFR_OK, "take_commit"); + return commit; +} + +typedef struct { + int count; + uint64_t session; + int bad; +} Observed; + +static void observe(void* user_data, const IFRMutation* mutation) { + Observed* observed = (Observed*)user_data; + observed->count++; + if (mutation->session != observed->session || !mutation->before || !mutation->after || + !ifr_snapshot_preedit(mutation->before) || !ifr_snapshot_preedit(mutation->after)) { + observed->bad++; + } +} + +static void fixture_checks(const char* fixture, const char* scratch) { + IFREngine* engine = NULL; + IFREngineConfig config = {path(0, fixture, "shared"), path(1, scratch, "fixture-user"), NULL, NULL}; + assert(ifr_abi_version() == 1); + expect(ifr_engine_create(NULL, &engine), IFR_INVALID_ARGUMENT, "null config"); + expect(ifr_engine_create(&config, NULL), IFR_INVALID_ARGUMENT, "null out"); + /* The probe schema compiles through the real runtime; the engine then rejects the + * missing production dictionary and finalizes Rime before returning. */ + expect(ifr_engine_create(&config, &engine), IFR_RESOURCE, "fixture engine"); + assert(engine == NULL && strstr(ifr_last_error(), "pinyin_simp") != NULL); + expect(ifr_engine_create(&config, &engine), IFR_RESOURCE, "fixture engine again"); + config.context_index = path(2, scratch, "missing.context.bin"); + expect(ifr_engine_create(&config, &engine), IFR_IO, "missing context index"); + config.shared = "bad\xff"; + expect(ifr_engine_create(&config, &engine), IFR_INVALID_ARGUMENT, "non-UTF-8 path"); + expect(ifr_session_key(NULL, 97, 0, NULL), IFR_INVALID_ARGUMENT, "null session"); + assert(ifr_session_id(NULL) == 0 && ifr_snapshot_candidate_count(NULL) == 0); + assert(ifr_snapshot_candidate_text(NULL, 0) == NULL); + ifr_engine_destroy(NULL); + ifr_session_destroy(NULL); + ifr_snapshot_free(NULL); + ifr_string_free(NULL); + puts("PASS C ABI: argument validation, fixture runtime lifetime and resource checks"); +} + +static IFREngine* production(const char* resources, const char* user) { + IFREngine* engine = NULL; + IFREngineConfig config = {path(3, resources, "shared"), user, path(4, resources, "prepared/cache"), + path(5, resources, "shared/pinyin_simp.context.bin")}; + expect(ifr_engine_create(&config, &engine), IFR_OK, "engine"); + assert(engine != NULL && ifr_engine_context_ranking_ready(engine) == 1); + return engine; +} + +static void resource_checks(const char* resources, const char* scratch) { + const char* user = path(6, scratch, "user"); + IFREngine* engine = production(resources, user); + IFREngine* second = NULL; + IFREngineConfig duplicate = {path(3, resources, "shared"), user, NULL, NULL}; + expect(ifr_engine_create(&duplicate, &second), IFR_ALREADY_RUNNING, "second engine"); + + IFRSession* session = NULL; + expect(ifr_session_create(engine, &session), IFR_OK, "session"); + Observed observed = {0, ifr_session_id(session), 0}; + assert(observed.session != 0 && ifr_session_available(session) == 1); + expect(ifr_engine_set_observer(engine, observe, &observed), IFR_OK, "observer"); + + IFRPhrase phrases[] = {{"a", "dz", "地址"}}; + uint32_t options = ifr_input_options_default(); + expect(ifr_session_set_configuration(session, 9, phrases, 1, &options), IFR_OK, "configure"); + assert(ifr_session_configuration_error(session) == NULL); + assert(ifr_session_candidate_count(session) == 9); + uint32_t applied = 0; + assert(ifr_session_input_options(session, &applied) == 1 && applied == options); + IFRPhrase invalid[] = {{"b", "", "x"}}; + expect(ifr_session_set_configuration(session, 9, invalid, 1, NULL), IFR_OK, "invalid phrases"); + assert(strcmp(ifr_session_configuration_error(session), "invalid-data") == 0); + expect(ifr_session_set_configuration(session, 9, phrases, 1, NULL), IFR_OK, "reconfigure"); + assert(ifr_session_configuration_error(session) == NULL); + + /* Composition, UTF-8 byte offsets, token-identified selection and commit draining. */ + assert(take(session) == NULL); + type_text(session, "nihao"); + IFRSnapshot* shown = NULL; + expect(ifr_session_snapshot(session, &shown), IFR_OK, "snapshot"); + assert(strcmp(ifr_snapshot_input(shown), "nihao") == 0); + assert(strcmp(ifr_snapshot_preedit(shown), "ni hao") == 0); + assert(ifr_snapshot_caret(shown) == strlen("ni hao")); + assert(ifr_snapshot_selection_start(shown) == 0 && ifr_snapshot_selection_end(shown) == strlen("ni hao")); + assert(ifr_snapshot_candidate_count(shown) == 9 && ifr_snapshot_page(shown) == 0); + assert(strcmp(ifr_snapshot_candidate_text(shown, 0), "你好") == 0); + assert(ifr_snapshot_candidate_comment(shown, 0) != NULL); + assert(ifr_snapshot_candidate_text(shown, 9) == NULL && ifr_snapshot_highlighted(shown) == 0); + int handled = 0; + expect(ifr_session_select(session, shown, 9, &handled), IFR_INVALID_CANDIDATE, "out of range"); + expect(ifr_session_select(session, shown, 0, &handled), IFR_OK, "select"); + assert(handled == 1); + char* commit = take(session); + assert(commit && strcmp(commit, "你好") == 0); + ifr_string_free(commit); + assert(take(session) == NULL); + expect(ifr_session_select(session, shown, 0, &handled), IFR_STALE_SNAPSHOT, "stale select"); + expect(ifr_session_highlight(session, shown, 0), IFR_STALE_SNAPSHOT, "stale highlight"); + ifr_snapshot_free(shown); + + /* Paging, digit selection and highlight through the policy layer. */ + type_text(session, "shi"); + IFRSnapshot* first = NULL; + expect(ifr_session_snapshot(session, &first), IFR_OK, "first page"); + assert(ifr_snapshot_last_page(first) == 0); + expect(ifr_session_change_page(session, 0, &handled), IFR_OK, "page down"); + IFRSnapshot* next = NULL; + expect(ifr_session_snapshot(session, &next), IFR_OK, "second page"); + assert(handled == 1 && ifr_snapshot_page(next) == 1); + assert(strcmp(ifr_snapshot_candidate_text(next, 0), ifr_snapshot_candidate_text(first, 0)) != 0); + ifr_snapshot_free(next); + expect(ifr_session_change_page(session, 1, &handled), IFR_OK, "page up"); + expect(ifr_session_snapshot(session, &next), IFR_OK, "back on first page"); + assert(ifr_snapshot_page(next) == 0); + expect(ifr_session_highlight(session, next, 2), IFR_OK, "highlight"); + ifr_snapshot_free(next); + expect(ifr_session_snapshot(session, &next), IFR_OK, "highlighted"); + assert(ifr_snapshot_highlighted(next) == 2); + expect(ifr_session_key(session, 0xff54, 0, &handled), IFR_OK, "Down"); + ifr_snapshot_free(next); + expect(ifr_session_snapshot(session, &next), IFR_OK, "moved"); + assert(handled == 1 && ifr_snapshot_highlighted(next) == 3); + ifr_snapshot_free(next); + expect(ifr_session_key(session, '2', 0, &handled), IFR_OK, "digit"); + commit = take(session); + assert(handled == 1 && commit && strcmp(commit, ifr_snapshot_candidate_text(first, 1)) == 0); + ifr_string_free(commit); + ifr_snapshot_free(first); + + /* Custom phrases, preceding text, commit, clear and ASCII mode. */ + type_text(session, "dz"); + expect(ifr_session_snapshot(session, &shown), IFR_OK, "phrase page"); + assert(strcmp(ifr_snapshot_candidate_text(shown, 0), "地址") == 0); + ifr_snapshot_free(shown); + expect(ifr_session_clear(session), IFR_OK, "clear"); + expect(ifr_session_snapshot(session, &shown), IFR_OK, "cleared"); + assert(ifr_snapshot_preedit(shown)[0] == 0 && ifr_snapshot_candidate_count(shown) == 0); + ifr_snapshot_free(shown); + expect(ifr_session_set_preceding_text(session, "中国"), IFR_OK, "preceding"); + expect(ifr_session_set_preceding_text(session, "bad\xff"), IFR_INVALID_ARGUMENT, "bad preceding"); + type_text(session, "nihao"); + expect(ifr_session_commit(session, &handled), IFR_OK, "commit"); + commit = take(session); + assert(handled == 1 && commit && strcmp(commit, "你好") == 0); + ifr_string_free(commit); + int ascii = 1; + expect(ifr_session_ascii_mode(session, &ascii), IFR_OK, "ascii read"); + assert(ascii == 0); + expect(ifr_session_set_ascii_mode(session, 1), IFR_OK, "ascii on"); + expect(ifr_session_key(session, 'a', 0, &handled), IFR_OK, "ascii key"); + assert(handled == 0); + expect(ifr_session_ascii_mode(session, &ascii), IFR_OK, "ascii read"); + assert(ascii == 1); + expect(ifr_session_toggle_ascii_mode(session, &handled), IFR_OK, "toggle"); + expect(ifr_session_ascii_mode(session, &ascii), IFR_OK, "ascii read"); + assert(handled == 1 && ascii == 0); + assert(observed.count > 10 && observed.bad == 0); + expect(ifr_engine_set_observer(engine, NULL, NULL), IFR_OK, "remove observer"); + int before = observed.count; + type_text(session, "ni"); + assert(observed.count == before); + expect(ifr_session_clear(session), IFR_OK, "clear"); + + /* The session keeps the engine alive after the caller drops its reference. */ + ifr_engine_destroy(engine); + type_text(session, "nihao"); + expect(ifr_session_snapshot(session, &shown), IFR_OK, "after engine release"); + assert(strcmp(ifr_snapshot_candidate_text(shown, 0), "你好") == 0); + ifr_snapshot_free(shown); + expect(ifr_engine_create(&duplicate, &second), IFR_ALREADY_RUNNING, "still running"); + ifr_session_destroy(session); + engine = production(resources, user); + ifr_engine_destroy(engine); + puts("PASS C ABI: engine/session lifetime, keys, snapshots, selection, paging, commits, configuration"); +} + +int main(int argc, char** argv) { + if (argc < 3) { + fputs("usage: abi-test FIXTURE_ROOT SCRATCH [RESOURCES]\n", stderr); + return 2; + } + fixture_checks(argv[1], argv[2]); + if (argc > 3) { + resource_checks(argv[3], argv[2]); + } else { + puts("SKIP C ABI production checks: no resources directory"); + } + return 0; +} From 735f8c4788c3a0b35f0d18dbf4ee4e21df74e3b4 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 00:26:19 +0800 Subject: [PATCH 14/21] Add the Fcitx5 addon skeleton over the engine C ABI Linux/fcitx5 holds a CMake project, addon and input-method descriptors, and an InputMethodEngine that keeps one engine session per input context. Keys reach the engine synchronously with Rime keysyms and translated modifier masks; every mutation drains the commit exactly once and rebuilds the preedit (client or panel) and candidate list from a fresh snapshot using UTF-8 byte offsets. The candidate list routes clicks, paging and cursor moves back through the ABI with the snapshot they were shown from, so late actions on a superseded page are rejected. Reset and focus changes clear the composition, switching input methods commits it, password and sensitive contexts never compose, and surrounding text is read only behind the client's capability when a composition starts. Resource and user paths follow XDG. bridge.h keeps the platform-independent helpers (paths, preedit layout, modifiers, bounded preceding text) with a test that runs anywhere; test.sh also compiles the addon syntax-only against an fcitx5 source tree. The CMake build and runtime behavior are unverified until they run on the Linux host, as the README states. Refs #36 --- Linux/fcitx5/CMakeLists.txt | 47 ++++++ Linux/fcitx5/README.md | 34 ++++ Linux/fcitx5/inkflow-pinyin.conf | 7 + Linux/fcitx5/inkflow.conf.in | 9 ++ Linux/fcitx5/src/bridge.h | 141 +++++++++++++++++ Linux/fcitx5/src/candidates.cpp | 65 ++++++++ Linux/fcitx5/src/candidates.h | 57 +++++++ Linux/fcitx5/src/engine.cpp | 244 +++++++++++++++++++++++++++++ Linux/fcitx5/src/engine.h | 70 +++++++++ Linux/fcitx5/test.sh | 23 +++ Linux/fcitx5/tests/bridge_test.cpp | 62 ++++++++ 11 files changed, 759 insertions(+) create mode 100644 Linux/fcitx5/CMakeLists.txt create mode 100644 Linux/fcitx5/README.md create mode 100644 Linux/fcitx5/inkflow-pinyin.conf create mode 100644 Linux/fcitx5/inkflow.conf.in create mode 100644 Linux/fcitx5/src/bridge.h create mode 100644 Linux/fcitx5/src/candidates.cpp create mode 100644 Linux/fcitx5/src/candidates.h create mode 100644 Linux/fcitx5/src/engine.cpp create mode 100644 Linux/fcitx5/src/engine.h create mode 100755 Linux/fcitx5/test.sh create mode 100644 Linux/fcitx5/tests/bridge_test.cpp diff --git a/Linux/fcitx5/CMakeLists.txt b/Linux/fcitx5/CMakeLists.txt new file mode 100644 index 0000000..cd09fd2 --- /dev/null +++ b/Linux/fcitx5/CMakeLists.txt @@ -0,0 +1,47 @@ +cmake_minimum_required(VERSION 3.20) +project(inkflow-fcitx5 VERSION 0.1.0 LANGUAGES C CXX) +if(NOT CMAKE_SYSTEM_NAME STREQUAL "Linux") + message(FATAL_ERROR "The Fcitx5 addon builds on Linux only") +endif() +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) +set(CMAKE_CXX_VISIBILITY_PRESET hidden) + +find_package(Fcitx5Core REQUIRED) + +# The Rust engine is built first: cargo build --release in Core/Portable (see README). +get_filename_component(INKFLOW_ROOT "${CMAKE_CURRENT_SOURCE_DIR}/../.." ABSOLUTE) +set(INKFLOW_CARGO_DIR "${INKFLOW_ROOT}/build/portable/cargo/release" CACHE PATH "Cargo release output of Core/Portable") +set(INKFLOW_NATIVE_PREFIX "${INKFLOW_ROOT}/build/portable/prefix" CACHE PATH "Native prefix from build-native.py") +set(INKFLOW_RESOURCES "" CACHE PATH "Prepared resources (shared/ and prepared/cache/) to install; empty skips") +foreach(file "${INKFLOW_CARGO_DIR}/libinkflow_rime.a" "${INKFLOW_NATIVE_PREFIX}/lib/librime.so") + if(NOT EXISTS "${file}") + message(FATAL_ERROR "Missing ${file}; build the Rust engine first") + endif() +endforeach() + +add_library(inkflow MODULE src/engine.cpp src/candidates.cpp) +target_include_directories(inkflow PRIVATE "${INKFLOW_ROOT}/Core/Portable/include" src) +target_compile_options(inkflow PRIVATE -Wall -Wextra -Werror) +# The staticlib bundles the Rust crate and the C++ bridge; librime stays shared. +target_link_libraries(inkflow PRIVATE Fcitx5::Core + "${INKFLOW_CARGO_DIR}/libinkflow_rime.a" "${INKFLOW_NATIVE_PREFIX}/lib/librime.so" + stdc++ dl pthread m) +set(INKFLOW_LIBDIR "${CMAKE_INSTALL_PREFIX}/lib/inkflow") +set_target_properties(inkflow PROPERTIES INSTALL_RPATH "${INKFLOW_LIBDIR}" BUILD_WITH_INSTALL_RPATH OFF) + +configure_file(inkflow.conf.in inkflow.conf @ONLY) +install(TARGETS inkflow DESTINATION "${FCITX_INSTALL_ADDONDIR}") +install(FILES "${CMAKE_CURRENT_BINARY_DIR}/inkflow.conf" DESTINATION "${FCITX_INSTALL_PKGDATADIR}/addon") +install(FILES inkflow-pinyin.conf DESTINATION "${FCITX_INSTALL_PKGDATADIR}/inputmethod") +file(GLOB INKFLOW_RIME_LIBS "${INKFLOW_NATIVE_PREFIX}/lib/librime.so*") +install(FILES ${INKFLOW_RIME_LIBS} DESTINATION "${INKFLOW_LIBDIR}") +if(INKFLOW_RESOURCES) + install(DIRECTORY "${INKFLOW_RESOURCES}/shared" "${INKFLOW_RESOURCES}/prepared" + DESTINATION "${CMAKE_INSTALL_PREFIX}/share/inkflow/rime") +endif() + +enable_testing() +add_executable(bridge_test tests/bridge_test.cpp) +target_include_directories(bridge_test PRIVATE src) +add_test(NAME bridge_test COMMAND bridge_test) diff --git a/Linux/fcitx5/README.md b/Linux/fcitx5/README.md new file mode 100644 index 0000000..14cf086 --- /dev/null +++ b/Linux/fcitx5/README.md @@ -0,0 +1,34 @@ +# Fcitx5 adapter + +Minimal Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://github.com/nervouna/InkFlow/issues/36). Targets Omarchy/Hyprland and Steam Deck Desktop Mode (KDE Plasma); GNOME/IBus is deferred. This is scaffolding for integration work, not daily-use support: the pieces below are unverified until they run on the Linux host. + +## What it does + +- `src/engine.cpp`: one engine per addon (`ifr_engine_create` on load, from XDG paths), one engine session per Fcitx5 input context (an `InputContextProperty`). Keys go to `ifr_session_key` synchronously on Fcitx5's thread with Rime keysyms and translated modifier masks; releases carry Rime's release mask. Every mutation ends in `refresh`: the commit is drained exactly once with `ifr_session_take_commit` and delivered with `commitString`, then the preedit (client preedit when the client declares `Preedit`, otherwise the panel) and candidate list are rebuilt from a fresh snapshot. Preedit highlight and caret come from the snapshot's UTF-8 byte offsets, which are Fcitx5's units too. +- `src/candidates.cpp`: digit selection and Up/Down are the engine's own key policy; mouse clicks and Fcitx5's paging/cursor calls go back through the ABI with the snapshot they were shown from, so a late click on a superseded page is rejected and only redraws. +- Focus and reset: a reset or focus change clears the composition; switching input methods commits it first (what the macOS frontend does on deactivation). Password and `Sensitive` contexts never compose and get raw keys, so nothing can be learned from them. +- Surrounding text: document context is read only when a composition starts and only when the client declares `SurroundingText` with valid text; otherwise the engine gets no context and keeps Rime's native order. +- `src/bridge.h`: the pure helpers (XDG paths, preedit layout, modifier translation, bounded preceding text) with `tests/bridge_test.cpp`, which runs on any platform. + +Resources are found at `INKFLOW_RESOURCES` (a prepared directory from `Core/Portable/prepare-resources.sh`, holding `shared/` and `prepared/cache/`), else the first `inkflow/rime` under `XDG_DATA_HOME` then `XDG_DATA_DIRS` that is prepared. User data lives in `$XDG_DATA_HOME/inkflow/rime` (default `~/.local/share/inkflow/rime`). Without resources the addon loads, logs a warning and passes every key through. + +## Build on Linux + +Requirements: the repository's Rust toolchain, CMake 3.20+, Ninja, a C++20 compiler, Python 3, and the Fcitx5 development files (`fcitx5` on Arch/SteamOS, `libfcitx5core-dev` plus `extra-cmake-modules` on Ubuntu). From the checkout root: + +```sh +python3 Core/Portable/build-native.py +CARGO_TARGET_DIR=$PWD/build/portable/cargo cargo build --locked --release --manifest-path Core/Portable/Cargo.toml +bash Core/Portable/prepare-resources.sh # prints PASS production resources: build/portable/resources.XXXXXX +cmake -S Linux/fcitx5 -B build/fcitx5 -G Ninja -DCMAKE_INSTALL_PREFIX=/usr/local \ + -DINKFLOW_RESOURCES=$PWD/build/portable/resources.XXXXXX +cmake --build build/fcitx5 && ctest --test-dir build/fcitx5 +``` + +The addon links the crate's `staticlib` (which bundles the C++ bridge) and the pinned `librime.so`, installed under `/lib/inkflow` with a matching rpath. `cmake --install build/fcitx5` places the addon in Fcitx5's addon directory, its `addon/` and `inputmethod/` descriptors in Fcitx5's data directory, and the prepared resources under `/share/inkflow/rime`. Installing and enabling the input method change the user's input configuration: do that only with explicit approval, and keep another input method enabled while testing. + +`bash Linux/fcitx5/test.sh` runs the helper tests anywhere; with `FCITX5_SOURCE=` it also compiles the addon syntax-only against those headers (export headers stubbed), which is how it was checked from macOS against fcitx5 5.1.14. + +## Unverified from macOS + +The CMake build, `Fcitx5::Core` linkage, addon loading, and all runtime behavior (preedit rendering, candidate window, commit delivery, focus and sensitive-field handling under Hyprland and KDE Plasma, Flatpak clients) have not run. Fcitx5 5.1.14 headers need C++20; older Fcitx5 releases on SteamOS may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). Packaging, configuration and personal-data import are later slices. diff --git a/Linux/fcitx5/inkflow-pinyin.conf b/Linux/fcitx5/inkflow-pinyin.conf new file mode 100644 index 0000000..76ae1a0 --- /dev/null +++ b/Linux/fcitx5/inkflow-pinyin.conf @@ -0,0 +1,7 @@ +[InputMethod] +Name=InkFlow Pinyin +Icon=fcitx-pinyin +Label=墨 +LangCode=zh_CN +Addon=inkflow +Configurable=False diff --git a/Linux/fcitx5/inkflow.conf.in b/Linux/fcitx5/inkflow.conf.in new file mode 100644 index 0000000..e2a24ce --- /dev/null +++ b/Linux/fcitx5/inkflow.conf.in @@ -0,0 +1,9 @@ +[Addon] +Name=InkFlow +Comment=InkFlow Pinyin on the shared Rust/Rime core +Category=InputMethod +Library=libinkflow +Type=SharedLibrary +OnDemand=True +Configurable=False +Version=@PROJECT_VERSION@ diff --git a/Linux/fcitx5/src/bridge.h b/Linux/fcitx5/src/bridge.h new file mode 100644 index 0000000..891f834 --- /dev/null +++ b/Linux/fcitx5/src/bridge.h @@ -0,0 +1,141 @@ +// Pure helpers between Fcitx5 and the engine ABI: XDG paths, preedit layout from UTF-8 +// byte offsets, key-state translation and bounded surrounding text. No Fcitx5 or GLib +// dependency, so tests/bridge_test.cpp runs on any platform. +#ifndef INKFLOW_FCITX5_BRIDGE_H +#define INKFLOW_FCITX5_BRIDGE_H +#include +#include +#include +#include + +namespace inkflow { + +struct Paths { + std::string shared; + std::string cache; + std::string context_index; + std::string user; +}; + +using Env = std::function; +using Exists = std::function; + +inline std::string env_or(const Env& env, const char* name, const std::string& fallback) { + const char* value = env(name); + return value && *value ? value : fallback; +} + +inline std::vector split(const std::string& text, char separator) { + std::vector parts; + std::string::size_type start = 0; + while (start <= text.size()) { + auto end = text.find(separator, start); + if (end == std::string::npos) end = text.size(); + if (end > start) parts.push_back(text.substr(start, end - start)); + start = end + 1; + } + return parts; +} + +// Resources: INKFLOW_RESOURCES (a prepared directory holding shared/ and prepared/cache/), +// otherwise the first /inkflow/rime of XDG_DATA_HOME then XDG_DATA_DIRS that holds +// shared/. User data: XDG_DATA_HOME/inkflow/rime. Empty strings mean "not found". +inline Paths resolve_paths(const Env& env, const Exists& exists) { + Paths paths; + std::string home = env_or(env, "HOME", ""); + std::string data_home = env_or(env, "XDG_DATA_HOME", home + "/.local/share"); + paths.user = data_home + "/inkflow/rime"; + std::vector roots; + std::string override = env_or(env, "INKFLOW_RESOURCES", ""); + if (!override.empty()) roots.push_back(override); + roots.push_back(data_home + "/inkflow/rime"); + for (const auto& dir : split(env_or(env, "XDG_DATA_DIRS", "/usr/local/share:/usr/share"), ':')) { + roots.push_back(dir + "/inkflow/rime"); + } + for (const auto& root : roots) { + if (!exists(root + "/shared") || !exists(root + "/prepared/complete")) continue; + paths.shared = root + "/shared"; + paths.cache = root + "/prepared/cache"; + paths.context_index = paths.shared + "/pinyin_simp.context.bin"; + break; + } + return paths; +} + +struct PreeditSegment { + std::string text; + bool highlighted; +}; + +struct PreeditLayout { + std::vector segments; + std::size_t cursor; +}; + +// Offsets are UTF-8 bytes of `preedit` on character boundaries (the ABI guarantees it); +// anything inconsistent collapses to an unhighlighted preedit with the caret at its end. +inline PreeditLayout preedit_layout(const std::string& preedit, std::size_t caret, + std::size_t selection_start, std::size_t selection_end) { + PreeditLayout layout{{}, preedit.size()}; + bool valid = caret <= preedit.size() && selection_start <= selection_end && + selection_end <= preedit.size(); + if (!valid) { + if (!preedit.empty()) layout.segments.push_back({preedit, false}); + return layout; + } + layout.cursor = caret; + if (selection_start > 0) layout.segments.push_back({preedit.substr(0, selection_start), false}); + if (selection_end > selection_start) { + layout.segments.push_back({preedit.substr(selection_start, selection_end - selection_start), true}); + } + if (selection_end < preedit.size()) layout.segments.push_back({preedit.substr(selection_end), false}); + return layout; +} + +// Fcitx5 KeyState bits that Rime understands, as Rime masks. Super (Mod4) and GTK's +// virtual Super both become Rime's Super mask; a release adds Rime's release mask. +inline std::uint32_t rime_modifiers(std::uint32_t fcitx_states, bool release) { + const std::uint32_t shift = 1u << 0, lock = 1u << 1, control = 1u << 2, alt = 1u << 3; + const std::uint32_t super = 1u << 6, super2 = 1u << 26; + std::uint32_t mask = fcitx_states & (shift | lock | control | alt); + if (fcitx_states & (super | super2)) mask |= 1u << 26; + if (release) mask |= 1u << 30; + return mask; +} + +inline std::size_t utf8_sequence_length(unsigned char lead) { + if (lead < 0x80) return 1; + if ((lead & 0xE0) == 0xC0) return 2; + if ((lead & 0xF0) == 0xE0) return 3; + if ((lead & 0xF8) == 0xF0) return 4; + return 0; +} + +// Byte offsets of each code point start plus the end; empty when the text is not UTF-8. +inline std::vector code_point_offsets(const std::string& text) { + std::vector offsets; + std::size_t index = 0; + while (index < text.size()) { + offsets.push_back(index); + std::size_t length = utf8_sequence_length(static_cast(text[index])); + if (length == 0 || index + length > text.size()) return {}; + for (std::size_t i = 1; i < length; ++i) { + if ((static_cast(text[index + i]) & 0xC0) != 0x80) return {}; + } + index += length; + } + offsets.push_back(text.size()); + return offsets; +} + +// The last `limit` code points before `cursor` (a code-point index, as Fcitx5 reports it). +// Invalid UTF-8 or an out-of-range cursor yields no context rather than wrong context. +inline std::string preceding_text(const std::string& text, std::size_t cursor, std::size_t limit) { + auto offsets = code_point_offsets(text); + if (offsets.empty() || cursor + 1 > offsets.size()) return ""; + std::size_t first = cursor > limit ? cursor - limit : 0; + return text.substr(offsets[first], offsets[cursor] - offsets[first]); +} + +} // namespace inkflow +#endif diff --git a/Linux/fcitx5/src/candidates.cpp b/Linux/fcitx5/src/candidates.cpp new file mode 100644 index 0000000..ebe5623 --- /dev/null +++ b/Linux/fcitx5/src/candidates.cpp @@ -0,0 +1,65 @@ +#include "candidates.h" + +#include "engine.h" + +namespace inkflow { + +Candidate::Candidate(Engine* engine, SnapshotRef snapshot, std::size_t index) + : fcitx::CandidateWord(fcitx::Text(ifr_snapshot_candidate_text(snapshot.get(), index))), + engine_(engine), snapshot_(std::move(snapshot)), index_(index) { + const char* comment = ifr_snapshot_candidate_comment(snapshot_.get(), index_); + if (comment && *comment) setComment(fcitx::Text(comment)); +} + +void Candidate::select(fcitx::InputContext* ic) const { + engine_->selectCandidate(ic, snapshot_, index_); +} + +CandidateList::CandidateList(Engine* engine, fcitx::InputContext* ic, SnapshotRef snapshot) + : engine_(engine), ic_(ic), snapshot_(std::move(snapshot)) { + setPageable(this); + setCursorMovable(this); + std::size_t count = ifr_snapshot_candidate_count(snapshot_.get()); + for (std::size_t index = 0; index < count; ++index) { + labels_.emplace_back(std::to_string(index + 1) + ". "); + words_.push_back(std::make_unique(engine_, snapshot_, index)); + } +} + +const fcitx::Text& CandidateList::label(int index) const { return labels_.at(index); } +const fcitx::CandidateWord& CandidateList::candidate(int index) const { return *words_.at(index); } +int CandidateList::size() const { return static_cast(words_.size()); } +int CandidateList::cursorIndex() const { + return static_cast(ifr_snapshot_highlighted(snapshot_.get())); +} +fcitx::CandidateLayoutHint CandidateList::layoutHint() const { + return fcitx::CandidateLayoutHint::Horizontal; +} +bool CandidateList::hasPrev() const { return ifr_snapshot_page(snapshot_.get()) > 0; } +bool CandidateList::hasNext() const { return !ifr_snapshot_last_page(snapshot_.get()); } +bool CandidateList::usedNextBefore() const { return false; } + +// Each navigation call ends by asking the engine to rebuild the panel, which destroys +// this list; nothing of it is touched afterwards. +void CandidateList::prev() { + Engine* engine = engine_; + fcitx::InputContext* ic = ic_; + engine->changePage(ic, true); +} +void CandidateList::next() { + Engine* engine = engine_; + fcitx::InputContext* ic = ic_; + engine->changePage(ic, false); +} +void CandidateList::prevCandidate() { + Engine* engine = engine_; + fcitx::InputContext* ic = ic_; + engine->moveHighlight(ic, true); +} +void CandidateList::nextCandidate() { + Engine* engine = engine_; + fcitx::InputContext* ic = ic_; + engine->moveHighlight(ic, false); +} + +} // namespace inkflow diff --git a/Linux/fcitx5/src/candidates.h b/Linux/fcitx5/src/candidates.h new file mode 100644 index 0000000..73adfa7 --- /dev/null +++ b/Linux/fcitx5/src/candidates.h @@ -0,0 +1,57 @@ +// Fcitx5 candidate list over one engine snapshot. Navigation goes back through the +// engine, which rebuilds the panel; every selection carries the snapshot it was shown +// from so a late click on a superseded page is rejected by the engine. +#ifndef INKFLOW_FCITX5_CANDIDATES_H +#define INKFLOW_FCITX5_CANDIDATES_H +#include +#include +#include +#include +#include + +#include "inkflow_rime.h" + +namespace inkflow { + +class Engine; +using SnapshotRef = std::shared_ptr; + +class Candidate final : public fcitx::CandidateWord { +public: + Candidate(Engine* engine, SnapshotRef snapshot, std::size_t index); + void select(fcitx::InputContext* ic) const override; + +private: + Engine* engine_; + SnapshotRef snapshot_; + std::size_t index_; +}; + +class CandidateList final : public fcitx::CandidateList, + public fcitx::PageableCandidateList, + public fcitx::CursorMovableCandidateList { +public: + CandidateList(Engine* engine, fcitx::InputContext* ic, SnapshotRef snapshot); + const fcitx::Text& label(int index) const override; + const fcitx::CandidateWord& candidate(int index) const override; + int size() const override; + int cursorIndex() const override; + fcitx::CandidateLayoutHint layoutHint() const override; + bool hasPrev() const override; + bool hasNext() const override; + void prev() override; + void next() override; + bool usedNextBefore() const override; + void prevCandidate() override; + void nextCandidate() override; + +private: + Engine* engine_; + fcitx::InputContext* ic_; + SnapshotRef snapshot_; + std::vector labels_; + std::vector> words_; +}; + +} // namespace inkflow +#endif diff --git a/Linux/fcitx5/src/engine.cpp b/Linux/fcitx5/src/engine.cpp new file mode 100644 index 0000000..487408e --- /dev/null +++ b/Linux/fcitx5/src/engine.cpp @@ -0,0 +1,244 @@ +#include "engine.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +#include "bridge.h" + +namespace inkflow { + +FCITX_DEFINE_LOG_CATEGORY(inkflow_log, "inkflow"); + +namespace { +// The engine keeps 16 graphemes; a few more code points cover combining sequences. +constexpr std::size_t kPrecedingCodePoints = 32; + +bool directory_exists(const std::string& path) { + struct stat info{}; + return ::stat(path.c_str(), &info) == 0; +} +} // namespace + +Engine::Engine(fcitx::Instance* instance) + : instance_(instance), factory_([this](fcitx::InputContext&) { + auto* state = new State; + if (!engine_) return state; + IFRStatus status = ifr_session_create(engine_, &state->session); + if (status != IFR_OK) { + INKFLOW_WARN() << "session create failed: " << status << " " << ifr_last_error(); + state->session = nullptr; + return state; + } + uint32_t options = ifr_input_options_default(); + ifr_session_set_configuration(state->session, 5, nullptr, 0, &options); + if (const char* error = ifr_session_configuration_error(state->session)) { + INKFLOW_WARN() << "configuration error: " << error; + } + return state; + }) { + Paths paths = resolve_paths([](const char* name) { return std::getenv(name); }, directory_exists); + if (paths.shared.empty()) { + INKFLOW_WARN() << "no prepared resources under XDG data directories; keys pass through"; + } else { + IFREngineConfig config = {paths.shared.c_str(), paths.user.c_str(), paths.cache.c_str(), + paths.context_index.c_str()}; + IFRStatus status = ifr_engine_create(&config, &engine_); + if (status != IFR_OK) { + INKFLOW_WARN() << "engine create failed: " << status << " " << ifr_last_error(); + engine_ = nullptr; + } else { + INKFLOW_INFO() << "engine ready: " << paths.shared << " user " << paths.user; + } + } + instance_->inputContextManager().registerProperty("inkflowState", &factory_); +} + +Engine::~Engine() { + // The factory unregisters itself and destroys every session after this body; the + // last of those references finalizes Rime. + ifr_engine_destroy(engine_); +} + +State* Engine::state(fcitx::InputContext* ic) { return ic->propertyFor(&factory_); } + +bool Engine::sensitive(const fcitx::InputContext* ic) const { + return ic->capabilityFlags().testAny( + fcitx::CapabilityFlags{fcitx::CapabilityFlag::Password, fcitx::CapabilityFlag::Sensitive}); +} + +void Engine::failed(fcitx::InputContext* ic, State* state, const char* what, IFRStatus status) { + INKFLOW_WARN() << what << " failed: " << status << " " << ifr_last_error(); + clear(ic, state); +} + +// Document context is captured only when a composition starts and only through the +// client's declared surrounding-text capability; anything else leaves native order. +void Engine::refreshContext(fcitx::InputContext* ic, State* state) { + std::string preceding; + if (!sensitive(ic) && ic->capabilityFlags().test(fcitx::CapabilityFlag::SurroundingText)) { + const auto& surrounding = ic->surroundingText(); + if (surrounding.isValid()) { + preceding = preceding_text(surrounding.text(), surrounding.cursor(), kPrecedingCodePoints); + } + } + ifr_session_set_preceding_text(state->session, preceding.c_str()); +} + +void Engine::keyEvent(const fcitx::InputMethodEntry&, fcitx::KeyEvent& event) { + auto* ic = event.inputContext(); + State* st = state(ic); + if (!st->session) return; + if (sensitive(ic)) { + // Never compose or learn in password and sensitive fields; the client gets raw keys. + if (st->composing()) clear(ic, st); + return; + } + const fcitx::Key& key = event.rawKey(); + if (!st->composing()) refreshContext(ic, st); + int handled = 0; + IFRStatus status = ifr_session_key(st->session, static_cast(key.sym()), + static_cast(rime_modifiers(key.states(), event.isRelease())), + &handled); + if (status != IFR_OK) { + failed(ic, st, "key", status); + return; + } + refresh(ic, st); + if (handled) event.filterAndAccept(); +} + +void Engine::activate(const fcitx::InputMethodEntry&, fcitx::InputContextEvent& event) { + State* st = state(event.inputContext()); + if (st->session) refresh(event.inputContext(), st); +} + +void Engine::deactivate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) { + auto* ic = event.inputContext(); + State* st = state(ic); + // Switching input methods keeps what was typed, as the macOS frontend does on + // deactivation; focus changes drop the composition because the target is gone. + if (st->session && st->composing() && + event.type() == fcitx::EventType::InputContextSwitchInputMethod) { + int handled = 0; + IFRStatus status = ifr_session_commit(st->session, &handled); + if (status != IFR_OK) { + failed(ic, st, "commit", status); + return; + } + refresh(ic, st); + } + reset(entry, event); +} + +void Engine::reset(const fcitx::InputMethodEntry&, fcitx::InputContextEvent& event) { + State* st = state(event.inputContext()); + if (st->session) clear(event.inputContext(), st); +} + +void Engine::selectCandidate(fcitx::InputContext* ic, const SnapshotRef& snapshot, std::size_t index) { + State* st = state(ic); + if (!st->session) return; + int handled = 0; + IFRStatus status = ifr_session_select(st->session, snapshot.get(), index, &handled); + if (status == IFR_STALE_SNAPSHOT || status == IFR_INVALID_CANDIDATE) { + INKFLOW_INFO() << "ignored late candidate action: " << status; + refresh(ic, st); + return; + } + if (status != IFR_OK) { + failed(ic, st, "select", status); + return; + } + refresh(ic, st); +} + +void Engine::changePage(fcitx::InputContext* ic, bool backward) { + State* st = state(ic); + if (!st->session) return; + IFRStatus status = ifr_session_change_page(st->session, backward ? 1 : 0, nullptr); + if (status != IFR_OK) { + failed(ic, st, "page", status); + return; + } + refresh(ic, st); +} + +void Engine::moveHighlight(fcitx::InputContext* ic, bool backward) { + State* st = state(ic); + if (!st->session) return; + IFRStatus status = ifr_session_key(st->session, backward ? 0xff52 : 0xff54, 0, nullptr); + if (status != IFR_OK) { + failed(ic, st, "highlight", status); + return; + } + refresh(ic, st); +} + +void Engine::refresh(fcitx::InputContext* ic, State* st) { + char* commit = nullptr; + IFRStatus status = ifr_session_take_commit(st->session, &commit); + if (status != IFR_OK) { + INKFLOW_WARN() << "take_commit failed: " << status << " " << ifr_last_error(); + } else if (commit) { + ic->commitString(commit); + ifr_string_free(commit); + } + IFRSnapshot* raw = nullptr; + status = ifr_session_snapshot(st->session, &raw); + auto& panel = ic->inputPanel(); + panel.reset(); + if (status != IFR_OK) { + INKFLOW_WARN() << "snapshot failed: " << status << " " << ifr_last_error(); + st->snapshot.reset(); + } else { + st->snapshot = SnapshotRef(raw, ifr_snapshot_free); + const char* preedit = ifr_snapshot_preedit(raw); + if (preedit && *preedit) { + PreeditLayout layout = preedit_layout(preedit, ifr_snapshot_caret(raw), + ifr_snapshot_selection_start(raw), + ifr_snapshot_selection_end(raw)); + fcitx::Text text; + for (const auto& segment : layout.segments) { + fcitx::TextFormatFlags flags = fcitx::TextFormatFlag::Underline; + if (segment.highlighted) flags |= fcitx::TextFormatFlag::HighLight; + text.append(segment.text, flags); + } + text.setCursor(static_cast(layout.cursor)); + if (ic->capabilityFlags().test(fcitx::CapabilityFlag::Preedit)) { + panel.setClientPreedit(text); + } else { + panel.setPreedit(text); + } + } + if (ifr_snapshot_candidate_count(raw) > 0) { + panel.setCandidateList(std::make_unique(this, ic, st->snapshot)); + } + } + ic->updatePreedit(); + ic->updateUserInterface(fcitx::UserInterfaceComponent::InputPanel); +} + +void Engine::clear(fcitx::InputContext* ic, State* st) { + IFRStatus status = ifr_session_clear(st->session); + if (status != IFR_OK) { + INKFLOW_WARN() << "clear failed: " << status << " " << ifr_last_error(); + } + refresh(ic, st); +} + +} // namespace inkflow + +#ifdef FCITX_ADDON_FACTORY_V2 +FCITX_ADDON_FACTORY_V2(inkflow, inkflow::Factory); +#else +FCITX_ADDON_FACTORY(inkflow::Factory); +#endif diff --git a/Linux/fcitx5/src/engine.h b/Linux/fcitx5/src/engine.h new file mode 100644 index 0000000..542d8ab --- /dev/null +++ b/Linux/fcitx5/src/engine.h @@ -0,0 +1,70 @@ +// Fcitx5 input-method engine over the InkFlow C ABI. One engine per addon, one engine +// session per input context. Keys are processed synchronously on Fcitx5's thread with +// no event-loop dependency; the engine never touches network, SQLite or telemetry. +#ifndef INKFLOW_FCITX5_ENGINE_H +#define INKFLOW_FCITX5_ENGINE_H +#include +#include +#include +#include +#include +#include +#include + +#include "candidates.h" +#include "inkflow_rime.h" + +namespace inkflow { + +FCITX_DECLARE_LOG_CATEGORY(inkflow_log); +#define INKFLOW_WARN() FCITX_LOGC(::inkflow::inkflow_log, Warn) +#define INKFLOW_INFO() FCITX_LOGC(::inkflow::inkflow_log, Info) + +class State final : public fcitx::InputContextProperty { +public: + ~State() override { ifr_session_destroy(session); } + IFRSession* session = nullptr; + SnapshotRef snapshot; + bool composing() const { + return snapshot && ifr_snapshot_preedit(snapshot.get()) && *ifr_snapshot_preedit(snapshot.get()); + } +}; + +class Engine final : public fcitx::InputMethodEngineV2 { +public: + explicit Engine(fcitx::Instance* instance); + ~Engine() override; + + void keyEvent(const fcitx::InputMethodEntry& entry, fcitx::KeyEvent& event) override; + void activate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; + void deactivate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; + void reset(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; + + // Called by the candidate list; each rebuilds the panel. + void selectCandidate(fcitx::InputContext* ic, const SnapshotRef& snapshot, std::size_t index); + void changePage(fcitx::InputContext* ic, bool backward); + void moveHighlight(fcitx::InputContext* ic, bool backward); + +private: + State* state(fcitx::InputContext* ic); + bool sensitive(const fcitx::InputContext* ic) const; + void refreshContext(fcitx::InputContext* ic, State* state); + // Drain the commit exactly once, then rebuild preedit and candidates from a fresh snapshot. + void refresh(fcitx::InputContext* ic, State* state); + void clear(fcitx::InputContext* ic, State* state); + void failed(fcitx::InputContext* ic, State* state, const char* what, IFRStatus status); + + fcitx::Instance* instance_; + IFREngine* engine_ = nullptr; + fcitx::FactoryFor factory_; +}; + +class Factory final : public fcitx::AddonFactory { +public: + fcitx::AddonInstance* create(fcitx::AddonManager* manager) override { + return new Engine(manager->instance()); + } +}; + +} // namespace inkflow +#endif diff --git a/Linux/fcitx5/test.sh b/Linux/fcitx5/test.sh new file mode 100755 index 0000000..cb81176 --- /dev/null +++ b/Linux/fcitx5/test.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# Platform-independent checks of the Fcitx5 adapter: the pure bridge helpers, and a +# syntax-only compile of the addon when FCITX5_SOURCE points at an fcitx5 source tree +# (its generated export headers are stubbed). The real build runs on Linux via CMake. +set -euo pipefail +cd "$(dirname "$0")/../.." +scratch=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-fcitx5.XXXXXX") +trap 'rm -rf "$scratch"' EXIT +c++ -std=c++17 -Wall -Wextra -Werror -I Linux/fcitx5/src Linux/fcitx5/tests/bridge_test.cpp -o "$scratch/bridge_test" +"$scratch/bridge_test" +if [[ -n "${FCITX5_SOURCE:-}" ]]; then + mkdir -p "$scratch/exports/fcitx" "$scratch/exports/fcitx-utils" "$scratch/exports/fcitx-config" + for pair in fcitx:fcitxcore:FCITXCORE fcitx-utils:fcitxutils:FCITXUTILS fcitx-config:fcitxconfig:FCITXCONFIG; do + IFS=: read -r dir base macro <<<"$pair" + printf '#define %s_EXPORT\n#define %s_NO_EXPORT\n#define %s_DEPRECATED\n#define %s_DEPRECATED_EXPORT\n' \ + "$macro" "$macro" "$macro" "$macro" > "$scratch/exports/$dir/${base}_export.h" + done + for file in engine.cpp candidates.cpp; do + c++ -std=c++20 -fsyntax-only -Wall -Wextra -Werror -I Core/Portable/include -I Linux/fcitx5/src \ + -I "$scratch/exports" -I "$FCITX5_SOURCE/src/lib" "Linux/fcitx5/src/$file" + done + echo "PASS fcitx5 addon syntax against $FCITX5_SOURCE" +fi diff --git a/Linux/fcitx5/tests/bridge_test.cpp b/Linux/fcitx5/tests/bridge_test.cpp new file mode 100644 index 0000000..7954390 --- /dev/null +++ b/Linux/fcitx5/tests/bridge_test.cpp @@ -0,0 +1,62 @@ +#include "bridge.h" +#include +#include +#include +#include + +using namespace inkflow; + +int main() { + std::map env = {{"HOME", "/home/u"}}; + auto getenv = [&](const char* name) -> const char* { + auto it = env.find(name); + return it == env.end() ? nullptr : it->second.c_str(); + }; + std::set present = {"/usr/share/inkflow/rime/shared", "/usr/share/inkflow/rime/prepared/complete"}; + auto exists = [&](const std::string& path) { return present.count(path) > 0; }; + Paths paths = resolve_paths(getenv, exists); + assert(paths.user == "/home/u/.local/share/inkflow/rime"); + assert(paths.shared == "/usr/share/inkflow/rime/shared"); + assert(paths.cache == "/usr/share/inkflow/rime/prepared/cache"); + assert(paths.context_index == "/usr/share/inkflow/rime/shared/pinyin_simp.context.bin"); + env["XDG_DATA_HOME"] = "/data"; + env["XDG_DATA_DIRS"] = "/opt/share:/usr/share"; + present.insert("/opt/share/inkflow/rime/shared"); + paths = resolve_paths(getenv, exists); + assert(paths.user == "/data/inkflow/rime" && paths.shared == "/usr/share/inkflow/rime/shared"); + present.insert("/opt/share/inkflow/rime/prepared/complete"); + assert(resolve_paths(getenv, exists).shared == "/opt/share/inkflow/rime/shared"); + env["INKFLOW_RESOURCES"] = "/tmp/res"; + present.insert("/tmp/res/shared"); + present.insert("/tmp/res/prepared/complete"); + assert(resolve_paths(getenv, exists).cache == "/tmp/res/prepared/cache"); + present.clear(); + assert(resolve_paths(getenv, exists).shared.empty()); + + PreeditLayout layout = preedit_layout("ni hao", 6, 0, 6); + assert(layout.cursor == 6 && layout.segments.size() == 1 && layout.segments[0].highlighted); + layout = preedit_layout("你好 ma", 6, 7, 9); + assert(layout.segments.size() == 2 && layout.segments[0].text == "你好 " && + layout.segments[1].text == "ma" && layout.segments[1].highlighted && !layout.segments[0].highlighted); + layout = preedit_layout("abc", 1, 1, 2); + assert(layout.segments.size() == 3 && layout.segments[2].text == "c" && layout.cursor == 1); + layout = preedit_layout("abc", 9, 0, 0); + assert(layout.segments.size() == 1 && !layout.segments[0].highlighted && layout.cursor == 3); + assert(preedit_layout("", 0, 0, 0).segments.empty()); + + assert(rime_modifiers(0, false) == 0); + assert(rime_modifiers(1u << 0 | 1u << 2, false) == ((1u << 0) | (1u << 2))); + assert(rime_modifiers(1u << 6, false) == (1u << 26)); + assert(rime_modifiers(1u << 26, true) == ((1u << 26) | (1u << 30))); + assert(rime_modifiers(1u << 4 | 1u << 31, false) == 0); + + assert(preceding_text("中文abc", 5, 16) == "中文abc"); + assert(preceding_text("中文abc", 5, 2) == "bc"); + assert(preceding_text("中文abc", 2, 16) == "中文"); + assert(preceding_text("中文abc", 0, 16) == ""); + assert(preceding_text("中文abc", 6, 16) == ""); + assert(preceding_text("\xff\xfe", 1, 16) == ""); + assert(preceding_text("a\xe4\xb8", 2, 16) == ""); + std::puts("PASS fcitx5 bridge helpers: XDG paths, preedit layout, key states, preceding text"); + return 0; +} From 477d9d5ed0dadbb64593b6d675b5ad24d64e7349 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 00:33:18 +0800 Subject: [PATCH 15/21] Add the Fcitx5 configuration surface and the personal-data import entry point Export the macOS backup document through the C ABI: ifr_backup_parse validates a format-1 document and exposes its candidate count, input options, phrases and the non-portable preferences it skips; ifr_backup_import and ifr_personal_recover run personal::import and recover, and refuse while any engine is initialized in the process. Backup::from_json now reports a missing field as Incompatible instead of panicking on the map index; the C consumer found it through the catch_unwind boundary. The Fcitx5 addon gains conf/inkflow.conf (candidates per page, the fourteen input options under their macOS names, custom phrases as code=text) applied to every live session and to sessions created on first use, and an ImportBackup path consumed once: the addon destroys every session and the engine, imports with rollback, writes the backup's settings into the configuration and recreates the engine. Refs #36 --- Core/Portable/README.md | 2 +- Core/Portable/include/inkflow_rime.h | 31 +++- Core/Portable/src/abi.rs | 183 +++++++++++++++++++- Core/Portable/src/lib.rs | 4 + Core/Portable/src/personal.rs | 25 ++- Core/Portable/tests/abi.c | 51 ++++++ Linux/fcitx5/README.md | 4 +- Linux/fcitx5/inkflow-pinyin.conf | 2 +- Linux/fcitx5/inkflow.conf.in | 2 +- Linux/fcitx5/src/bridge.h | 9 + Linux/fcitx5/src/config.h | 37 ++++ Linux/fcitx5/src/engine.cpp | 241 +++++++++++++++++++++++---- Linux/fcitx5/src/engine.h | 36 +++- Linux/fcitx5/tests/bridge_test.cpp | 6 +- 14 files changed, 575 insertions(+), 58 deletions(-) create mode 100644 Linux/fcitx5/src/config.h diff --git a/Core/Portable/README.md b/Core/Portable/README.md index 0851747..2ce5db4 100644 --- a/Core/Portable/README.md +++ b/Core/Portable/README.md @@ -71,7 +71,7 @@ Key latency is the same Rime work in both engines; the Rust engine's lower start `native/bridge.h` is the experimental internal C ABI between Rust and the native runtime. It uses Rime's public C API for lifecycle, deployment, input, context, and commits. C++ internals are confined to the existing extension and its registration check. The Rust library exposes safe `Runtime` and `Session` handles. -`include/inkflow_rime.h` is the frontend-facing C ABI over `Engine`/`InputSession` (`src/abi.rs`, ABI version 1), built as `cdylib` and `staticlib` beside the rlib. It covers engine and session lifetime, keys, immutable snapshots with UTF-8 byte offsets, token-identified selection and highlight, paging, commit draining, preceding text, configuration (candidate count, input options as a bitmask, custom phrases), ASCII mode and the mutation observer as a C callback. Every export catches panics and returns a status with a thread-local message; outputs are written only on success; there is no thread affinity. `abi.sh` builds both libraries and runs `tests/abi.c`: the fixture part exercises the real runtime lifetime and resource validation with the tiny probe schema (the engine itself needs the production schema), and the production part needs prepared resources; `test.sh` runs the former, `parity.sh` both. Personal-learning management and personal-data import are not exported yet. +`include/inkflow_rime.h` is the frontend-facing C ABI over `Engine`/`InputSession` (`src/abi.rs`, ABI version 1), built as `cdylib` and `staticlib` beside the rlib. It covers engine and session lifetime, keys, immutable snapshots with UTF-8 byte offsets, token-identified selection and highlight, paging, commit draining, preceding text, configuration (candidate count, input options as a bitmask, custom phrases), ASCII mode and the mutation observer as a C callback. Every export catches panics and returns a status with a thread-local message; outputs are written only on success; there is no thread affinity. `abi.sh` builds both libraries and runs `tests/abi.c`: the fixture part exercises the real runtime lifetime and resource validation with the tiny probe schema (the engine itself needs the production schema), and the production part needs prepared resources; `test.sh` runs the former, `parity.sh` both. Personal-data import is exported as `ifr_backup_parse`/`ifr_backup_import`/`ifr_personal_recover`, which refuse to run while any engine is initialized; personal-learning management is not exported yet. - One process-wide Rust mutex serializes **all** Rime calls, including initialization, deployment, reads, destruction, and finalization. The engine's policy lock serializes every `InputSession` of an `Engine` and nests outside that mutex. Sessions can move between threads. No GUI event loop, Swift actor, or thread affinity is required. Another engine must not call librime outside this lock in the same process; old/new comparisons need separate processes. - There is at most one runtime. A second initialization returns `AlreadyRunning`. Sessions retain an `Arc` to the runtime, so dropping its public handle cannot finalize live sessions. The last session/runtime owner finalizes Rime. A poisoned lock fails subsequent normal operations; destructors still attempt cleanup without panicking. diff --git a/Core/Portable/include/inkflow_rime.h b/Core/Portable/include/inkflow_rime.h index 476c61e..305a4de 100644 --- a/Core/Portable/include/inkflow_rime.h +++ b/Core/Portable/include/inkflow_rime.h @@ -61,12 +61,20 @@ enum { IFR_CONFIGURATION = 8, /* A session could not apply its configuration. */ IFR_IO = 9, IFR_STALE_SNAPSHOT = 10, - IFR_INVALID_CANDIDATE = 11 + IFR_INVALID_CANDIDATE = 11, + IFR_ENGINE_ACTIVE = 12, /* Personal-data work needs every engine destroyed first. */ + IFR_INCOMPATIBLE = 13, /* Another backup format, Rime version or dictionary set. */ + IFR_UNKNOWN_FIELDS = 14, + IFR_SETTINGS = 15, + IFR_PHRASES = 16, + IFR_SNAPSHOT = 17, /* A dictionary snapshot breaks the TSV contract. */ + IFR_RECOVERY_REQUIRED = 18 /* An earlier import was interrupted; call ifr_personal_recover. */ }; typedef struct IFREngine IFREngine; typedef struct IFRSession IFRSession; typedef struct IFRSnapshot IFRSnapshot; +typedef struct IFRBackup IFRBackup; typedef struct { const char* shared; /* Prepared shared resources. */ @@ -186,6 +194,27 @@ IFRStatus ifr_session_set_ascii_mode(const IFRSession* session, int value); IFRStatus ifr_session_ascii_mode(const IFRSession* session, int* value); IFRStatus ifr_session_toggle_ascii_mode(const IFRSession* session, int* handled); +/* Portable personal data: the macOS format-1 backup document. Parsing validates the + * document and keeps its portable settings; macOS-only preferences are listed as + * unsupported. Importing replaces the user directory's three Rime dictionaries with + * the backup's (absence removes the local one) with rollback on failure, and is + * rejected with IFR_ENGINE_ACTIVE while any engine is initialized in the process: + * destroy every session and engine first, import, then recreate the engine and apply + * the backup's settings through ifr_session_set_configuration. Phrase strings are + * borrowed from the backup. Never call these from the key path. */ +IFRStatus ifr_backup_parse(const uint8_t* bytes, size_t length, IFRBackup** backup); +void ifr_backup_free(IFRBackup* backup); +size_t ifr_backup_candidate_count(const IFRBackup* backup); +uint32_t ifr_backup_input_options(const IFRBackup* backup); +size_t ifr_backup_phrase_count(const IFRBackup* backup); +/* Returns 1 and fills phrase, or 0 when index is out of range. */ +int ifr_backup_phrase(const IFRBackup* backup, size_t index, IFRPhrase* phrase); +size_t ifr_backup_unsupported_count(const IFRBackup* backup); +const char* ifr_backup_unsupported(const IFRBackup* backup, size_t index); +IFRStatus ifr_backup_import(const IFRBackup* backup, const char* user); +/* Finish an interrupted import in `user`; a no-op without a pending transaction. */ +IFRStatus ifr_personal_recover(const char* user); + #ifdef __cplusplus } #endif diff --git a/Core/Portable/src/abi.rs b/Core/Portable/src/abi.rs index 30700fc..785cc51 100644 --- a/Core/Portable/src/abi.rs +++ b/Core/Portable/src/abi.rs @@ -10,6 +10,7 @@ use crate::{ Action, Configuration, ConfigurationError, Engine, EngineError, InputSession, Mutation, Snapshot, }, + personal::{self, Backup, PersonalError}, phrases::CustomPhrase, preferences::{InputOption, InputPreferences}, }; @@ -17,8 +18,8 @@ use std::{ cell::RefCell, ffi::{CStr, CString, c_char, c_int, c_void}, panic::{AssertUnwindSafe, catch_unwind}, - path::PathBuf, - ptr, + path::{Path, PathBuf}, + ptr, slice, sync::Arc, }; @@ -37,6 +38,13 @@ pub const CONFIGURATION: Status = 8; pub const IO: Status = 9; pub const STALE_SNAPSHOT: Status = 10; pub const INVALID_CANDIDATE: Status = 11; +pub const ENGINE_ACTIVE: Status = 12; +pub const INCOMPATIBLE: Status = 13; +pub const UNKNOWN_FIELDS: Status = 14; +pub const SETTINGS: Status = 15; +pub const PHRASES: Status = 16; +pub const SNAPSHOT: Status = 17; +pub const RECOVERY_REQUIRED: Status = 18; pub struct Failure { status: Status, @@ -81,6 +89,23 @@ impl From for Failure { } } +impl From for Failure { + fn from(error: PersonalError) -> Self { + let status = match error { + PersonalError::Incompatible => INCOMPATIBLE, + PersonalError::UnknownFields => UNKNOWN_FIELDS, + PersonalError::Settings => SETTINGS, + PersonalError::Phrases(_) => PHRASES, + PersonalError::Snapshot => SNAPSHOT, + PersonalError::NativeSnapshot => NATIVE, + PersonalError::RecoveryRequired => RECOVERY_REQUIRED, + PersonalError::EngineActive => ENGINE_ACTIVE, + PersonalError::Io(_) => IO, + }; + Self::new(status, error.to_string()) + } +} + thread_local! { static LAST_ERROR: RefCell = RefCell::new(CString::default()); } @@ -143,7 +168,9 @@ fn mask_from_options(input: &InputPreferences) -> u32 { InputOption::ALL .iter() .enumerate() - .fold(0, |mask, (bit, option)| mask | (u32::from(input.get(*option)) << bit)) + .fold(0, |mask, (bit, option)| { + mask | (u32::from(input.get(*option)) << bit) + }) } fn configuration_code(error: ConfigurationError) -> &'static CStr { @@ -225,7 +252,8 @@ pub struct MutationRecord { pub after: *const SnapshotHandle, } -pub type ObserverCallback = unsafe extern "C" fn(user_data: *mut c_void, mutation: *const MutationRecord); +pub type ObserverCallback = + unsafe extern "C" fn(user_data: *mut c_void, mutation: *const MutationRecord); struct Observer { callback: ObserverCallback, @@ -415,7 +443,9 @@ pub unsafe extern "C" fn ifr_session_snapshot( #[unsafe(no_mangle)] pub unsafe extern "C" fn ifr_snapshot_free(snapshot: *mut SnapshotHandle) { if !snapshot.is_null() { - let _ = catch_unwind(AssertUnwindSafe(|| drop(unsafe { Box::from_raw(snapshot) }))); + let _ = catch_unwind(AssertUnwindSafe(|| { + drop(unsafe { Box::from_raw(snapshot) }) + })); } } @@ -683,6 +713,144 @@ pub unsafe extern "C" fn ifr_session_toggle_ascii_mode( }) } +/// A parsed macOS backup with NUL-terminated copies of its portable settings. +pub struct BackupHandle { + backup: Backup, + phrases: Vec<(CString, CString, CString)>, + unsupported: Vec, +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_parse( + bytes: *const u8, + length: usize, + backup: *mut *mut BackupHandle, +) -> Status { + guard(|| { + if backup.is_null() || (length != 0 && bytes.is_null()) { + return Err(Failure::argument("backup or bytes is NULL")); + } + if length > personal::MAXIMUM_DOCUMENT_BYTES { + return Err(Failure::new(INCOMPATIBLE, "document too large")); + } + let document = if length == 0 { + &[][..] + } else { + unsafe { slice::from_raw_parts(bytes, length) } + }; + let parsed = Backup::from_json(document)?; + let phrases = parsed + .phrases + .iter() + .map(|p| Ok((c_string(&p.id)?, c_string(&p.code)?, c_string(&p.text)?))) + .collect::, Failure>>()?; + let unsupported = parsed + .unsupported + .iter() + .map(|key| c_string(key)) + .collect::, Failure>>()?; + let handle = BackupHandle { + backup: parsed, + phrases, + unsupported, + }; + unsafe { write(backup, Box::into_raw(Box::new(handle))) }; + Ok(()) + }) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_free(backup: *mut BackupHandle) { + if !backup.is_null() { + let _ = catch_unwind(AssertUnwindSafe(|| drop(unsafe { Box::from_raw(backup) }))); + } +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_candidate_count(backup: *const BackupHandle) -> usize { + unsafe { backup.as_ref() }.map_or(5, |b| b.backup.candidate_count) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_input_options(backup: *const BackupHandle) -> u32 { + unsafe { backup.as_ref() }.map_or_else( + || ifr_input_options_default(), + |b| mask_from_options(&b.backup.input), + ) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_phrase_count(backup: *const BackupHandle) -> usize { + unsafe { backup.as_ref() }.map_or(0, |b| b.phrases.len()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_phrase( + backup: *const BackupHandle, + index: usize, + phrase: *mut Phrase, +) -> c_int { + let Some((id, code, text)) = unsafe { backup.as_ref() }.and_then(|b| b.phrases.get(index)) + else { + return 0; + }; + unsafe { + write( + phrase, + Phrase { + id: id.as_ptr(), + code: code.as_ptr(), + text: text.as_ptr(), + }, + ) + }; + 1 +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_unsupported_count(backup: *const BackupHandle) -> usize { + unsafe { backup.as_ref() }.map_or(0, |b| b.unsupported.len()) +} + +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_unsupported( + backup: *const BackupHandle, + index: usize, +) -> *const c_char { + unsafe { backup.as_ref() } + .and_then(|b| b.unsupported.get(index)) + .map_or(ptr::null(), |key| key.as_ptr()) +} + +/// Replace the user directory's dictionaries with the backup's. Rejected while any engine +/// is initialized in this process; settings are applied by the frontend afterwards. +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_backup_import( + backup: *const BackupHandle, + user: *const c_char, +) -> Status { + guard(|| { + let backup = unsafe { reference(backup, "backup") }?; + let user = PathBuf::from(unsafe { text(user, "user") }?); + if crate::engine_active() { + return Err(Failure::new(ENGINE_ACTIVE, "an engine is initialized")); + } + Ok(personal::import(&user, &backup.backup.dictionaries)?) + }) +} + +/// Finish an interrupted import; a no-op without a pending transaction. +#[unsafe(no_mangle)] +pub unsafe extern "C" fn ifr_personal_recover(user: *const c_char) -> Status { + guard(|| { + let user = PathBuf::from(unsafe { text(user, "user") }?); + if crate::engine_active() { + return Err(Failure::new(ENGINE_ACTIVE, "an engine is initialized")); + } + Ok(personal::recover(Path::new(&user))?) + }) +} + #[cfg(test)] mod tests { use super::*; @@ -695,6 +863,9 @@ mod tests { let mask = mask_from_options(&traditional); assert_eq!(mask, ifr_input_options_default() | (1 << 13)); assert_eq!(options_from_mask(mask), traditional); - assert_eq!(configuration_code(ConfigurationError::PhraseWrite), c"phrase-write"); + assert_eq!( + configuration_code(ConfigurationError::PhraseWrite), + c"phrase-write" + ); } } diff --git a/Core/Portable/src/lib.rs b/Core/Portable/src/lib.rs index 4af4107..f63fc49 100644 --- a/Core/Portable/src/lib.rs +++ b/Core/Portable/src/lib.rs @@ -43,6 +43,10 @@ static ENGINE: Mutex = Mutex::new(false); fn lock() -> Result> { ENGINE.lock().map_err(|_| Error::Poisoned) } +/// Whether a runtime is initialized in this process (personal-data work must wait). +pub(crate) fn engine_active() -> bool { + ENGINE.lock().map(|active| *active).unwrap_or(true) +} fn check(code: i32) -> Result<()> { if code == 0 { Ok(()) diff --git a/Core/Portable/src/personal.rs b/Core/Portable/src/personal.rs index 838186e..b4ba751 100644 --- a/Core/Portable/src/personal.rs +++ b/Core/Portable/src/personal.rs @@ -117,11 +117,11 @@ impl Backup { }) { return Err(PersonalError::UnknownFields); } - let rime = object["rime"].as_str().ok_or(PersonalError::Incompatible)?; - if object["format"] != json!(1) || rime != RIME_VERSION { + let rime = document["rime"].as_str().ok_or(PersonalError::Incompatible)?; + if document["format"] != json!(1) || rime != RIME_VERSION { return Err(PersonalError::Incompatible); } - let snapshots = object["dictionaries"] + let snapshots = document["dictionaries"] .as_object() .ok_or(PersonalError::Incompatible)?; if !complete(&mut keys(snapshots)) { @@ -139,7 +139,7 @@ impl Backup { }; dictionaries.insert(name.clone(), snapshot); } - let settings = object["settings"] + let settings = document["settings"] .as_object() .ok_or(PersonalError::Settings)?; if settings.keys().any(|key| { @@ -150,7 +150,7 @@ impl Backup { }) { return Err(PersonalError::UnknownFields); } - let integers = settings["integers"] + let integers = settings.get("integers").unwrap_or(&Value::Null) .as_object() .ok_or(PersonalError::Settings)?; let integer = |key: &str| { @@ -465,6 +465,21 @@ mod tests { ) } + #[test] + fn missing_fields_are_incompatible_not_panics() { + for document in [ + "{}", + r#"{"format":1}"#, + r#"{"format":1,"rime":"1.17.0","settings":{},"dictionaries":{}}"#, + r#"{"format":1,"rime":"1.17.0","settings":{"phrases":[]},"dictionaries":{"pinyin_simp":null,"inkflow_shared_english":null,"inkflow_voice_alias":null}}"#, + ] { + assert!(matches!( + Backup::from_json(document.as_bytes()), + Err(PersonalError::Incompatible | PersonalError::Settings) + )); + } + } + #[test] fn macos_document_round_trip() { let document = json!({ diff --git a/Core/Portable/tests/abi.c b/Core/Portable/tests/abi.c index d420b6f..c067de3 100644 --- a/Core/Portable/tests/abi.c +++ b/Core/Portable/tests/abi.c @@ -49,6 +49,49 @@ static void observe(void* user_data, const IFRMutation* mutation) { } } +static const char* backup_json = + "{\"format\":1,\"rime\":\"1.17.0\",\"settings\":{\"integers\":{\"candidateCount\":7,\"fontSize\":18," + "\"input.abbreviation\":1,\"input.typoTolerance\":1,\"input.fuzzyZ\":0,\"input.fuzzyC\":0," + "\"input.fuzzyS\":0,\"input.emoji\":1,\"input.bracketPaging\":1,\"input.minusEqualPaging\":1," + "\"input.englishPunctuation\":0,\"input.cornerQuotes\":1,\"input.middleDot\":1," + "\"input.fullwidthPipe\":1,\"input.ideographicComma\":1,\"input.traditional\":1}," + "\"phrases\":[{\"id\":\"p1\",\"code\":\"dz\",\"text\":\"地址\"}]}," + "\"dictionaries\":{\"pinyin_simp\":null,\"inkflow_shared_english\":null,\"inkflow_voice_alias\":null}}"; + +static IFRBackup* parsed_backup(void) { + IFRBackup* backup = NULL; + expect(ifr_backup_parse((const uint8_t*)backup_json, strlen(backup_json), &backup), IFR_OK, "backup parse"); + assert(backup && ifr_backup_candidate_count(backup) == 7); + assert(ifr_backup_input_options(backup) == (ifr_input_options_default() | IFR_OPTION_TRADITIONAL)); + assert(ifr_backup_phrase_count(backup) == 1); + IFRPhrase phrase = {NULL, NULL, NULL}; + assert(ifr_backup_phrase(backup, 0, &phrase) == 1 && strcmp(phrase.code, "dz") == 0 && + strcmp(phrase.text, "地址") == 0 && strcmp(phrase.id, "p1") == 0); + assert(ifr_backup_phrase(backup, 1, &phrase) == 0); + assert(ifr_backup_unsupported_count(backup) == 1 && + strcmp(ifr_backup_unsupported(backup, 0), "settings.integers.fontSize") == 0); + assert(ifr_backup_unsupported(backup, 1) == NULL); + return backup; +} + +static void backup_checks(const char* scratch) { + IFRBackup* backup = NULL; + expect(ifr_backup_parse(NULL, 1, &backup), IFR_INVALID_ARGUMENT, "null bytes"); + expect(ifr_backup_parse((const uint8_t*)"{}", 2, &backup), IFR_INCOMPATIBLE, "empty document"); + const char* unknown = "{\"format\":1,\"rime\":\"1.17.0\",\"settings\":{},\"dictionaries\":{},\"x\":1}"; + expect(ifr_backup_parse((const uint8_t*)unknown, strlen(unknown), &backup), IFR_UNKNOWN_FIELDS, "unknown field"); + backup = parsed_backup(); + /* No engine is live: importing absent dictionaries into a fresh directory succeeds. */ + const char* user = path(7, scratch, "import-user"); + expect(ifr_backup_import(backup, user), IFR_OK, "import"); + expect(ifr_personal_recover(user), IFR_OK, "recover"); + expect(ifr_backup_import(backup, "bad\xff"), IFR_INVALID_ARGUMENT, "bad user"); + ifr_backup_free(backup); + ifr_backup_free(NULL); + assert(ifr_backup_candidate_count(NULL) == 5 && ifr_backup_phrase_count(NULL) == 0); + puts("PASS C ABI: backup parsing, disclosure, import and recovery on an isolated directory"); +} + static void fixture_checks(const char* fixture, const char* scratch) { IFREngine* engine = NULL; IFREngineConfig config = {path(0, fixture, "shared"), path(1, scratch, "fixture-user"), NULL, NULL}; @@ -197,6 +240,10 @@ static void resource_checks(const char* resources, const char* scratch) { assert(observed.count == before); expect(ifr_session_clear(session), IFR_OK, "clear"); + /* Personal-data import waits for every engine reference, including live sessions. */ + IFRBackup* backup = parsed_backup(); + expect(ifr_backup_import(backup, user), IFR_ENGINE_ACTIVE, "import while live"); + /* The session keeps the engine alive after the caller drops its reference. */ ifr_engine_destroy(engine); type_text(session, "nihao"); @@ -204,7 +251,10 @@ static void resource_checks(const char* resources, const char* scratch) { assert(strcmp(ifr_snapshot_candidate_text(shown, 0), "你好") == 0); ifr_snapshot_free(shown); expect(ifr_engine_create(&duplicate, &second), IFR_ALREADY_RUNNING, "still running"); + expect(ifr_backup_import(backup, user), IFR_ENGINE_ACTIVE, "import with a live session"); ifr_session_destroy(session); + expect(ifr_backup_import(backup, user), IFR_OK, "import after teardown"); + ifr_backup_free(backup); engine = production(resources, user); ifr_engine_destroy(engine); puts("PASS C ABI: engine/session lifetime, keys, snapshots, selection, paging, commits, configuration"); @@ -216,6 +266,7 @@ int main(int argc, char** argv) { return 2; } fixture_checks(argv[1], argv[2]); + backup_checks(argv[2]); if (argc > 3) { resource_checks(argv[3], argv[2]); } else { diff --git a/Linux/fcitx5/README.md b/Linux/fcitx5/README.md index 14cf086..82d591f 100644 --- a/Linux/fcitx5/README.md +++ b/Linux/fcitx5/README.md @@ -8,6 +8,8 @@ Minimal Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](htt - `src/candidates.cpp`: digit selection and Up/Down are the engine's own key policy; mouse clicks and Fcitx5's paging/cursor calls go back through the ABI with the snapshot they were shown from, so a late click on a superseded page is rejected and only redraws. - Focus and reset: a reset or focus change clears the composition; switching input methods commits it first (what the macOS frontend does on deactivation). Password and `Sensitive` contexts never compose and get raw keys, so nothing can be learned from them. - Surrounding text: document context is read only when a composition starts and only when the client declares `SurroundingText` with valid text; otherwise the engine gets no context and keeps Rime's native order. +- Configuration (`src/config.h`, `~/.config/fcitx5/conf/inkflow.conf`, also through `fcitx5-configtool`): candidates per page (3–9), the fourteen input options under their macOS names, and custom phrases as `code=text` entries. Settings reach every live session through `ifr_session_set_configuration` and apply at the session's next composition boundary; new sessions are configured when they are created on first use. +- Personal-data import (`ImportBackup=/path/to/backup.json` in that file, then reload with `fcitx5-remote -r` or apply in the configuration tool): the addon consumes the request first so a bad file cannot repeat, parses the macOS format-1 document, logs the non-portable preferences it skips, destroys every session and the engine (Rime finalizes), imports the three dictionaries with rollback on failure, writes the backup's candidate count, options and phrases into the configuration, and recreates the engine. Compositions in progress are lost, which is why the entry point is explicit. An interrupted earlier import is recovered first. - `src/bridge.h`: the pure helpers (XDG paths, preedit layout, modifier translation, bounded preceding text) with `tests/bridge_test.cpp`, which runs on any platform. Resources are found at `INKFLOW_RESOURCES` (a prepared directory from `Core/Portable/prepare-resources.sh`, holding `shared/` and `prepared/cache/`), else the first `inkflow/rime` under `XDG_DATA_HOME` then `XDG_DATA_DIRS` that is prepared. User data lives in `$XDG_DATA_HOME/inkflow/rime` (default `~/.local/share/inkflow/rime`). Without resources the addon loads, logs a warning and passes every key through. @@ -31,4 +33,4 @@ The addon links the crate's `staticlib` (which bundles the C++ bridge) and the p ## Unverified from macOS -The CMake build, `Fcitx5::Core` linkage, addon loading, and all runtime behavior (preedit rendering, candidate window, commit delivery, focus and sensitive-field handling under Hyprland and KDE Plasma, Flatpak clients) have not run. Fcitx5 5.1.14 headers need C++20; older Fcitx5 releases on SteamOS may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). Packaging, configuration and personal-data import are later slices. +The CMake build, `Fcitx5::Core` linkage, addon loading, and all runtime behavior (preedit rendering, candidate window, commit delivery, focus and sensitive-field handling under Hyprland and KDE Plasma, Flatpak clients) have not run. Fcitx5 5.1.14 headers need C++20; older Fcitx5 releases on SteamOS may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). The configuration surface and the import entry point compile against the headers but have not run. Packaging is a later slice. diff --git a/Linux/fcitx5/inkflow-pinyin.conf b/Linux/fcitx5/inkflow-pinyin.conf index 76ae1a0..c64e4d2 100644 --- a/Linux/fcitx5/inkflow-pinyin.conf +++ b/Linux/fcitx5/inkflow-pinyin.conf @@ -4,4 +4,4 @@ Icon=fcitx-pinyin Label=墨 LangCode=zh_CN Addon=inkflow -Configurable=False +Configurable=True diff --git a/Linux/fcitx5/inkflow.conf.in b/Linux/fcitx5/inkflow.conf.in index e2a24ce..4ed75e6 100644 --- a/Linux/fcitx5/inkflow.conf.in +++ b/Linux/fcitx5/inkflow.conf.in @@ -5,5 +5,5 @@ Category=InputMethod Library=libinkflow Type=SharedLibrary OnDemand=True -Configurable=False +Configurable=True Version=@PROJECT_VERSION@ diff --git a/Linux/fcitx5/src/bridge.h b/Linux/fcitx5/src/bridge.h index 891f834..9b13019 100644 --- a/Linux/fcitx5/src/bridge.h +++ b/Linux/fcitx5/src/bridge.h @@ -137,5 +137,14 @@ inline std::string preceding_text(const std::string& text, std::size_t cursor, s return text.substr(offsets[first], offsets[cursor] - offsets[first]); } +// One configured custom phrase, "code=text"; the engine validates both parts. +inline bool parse_phrase(const std::string& entry, std::string& code, std::string& text) { + auto separator = entry.find('='); + if (separator == std::string::npos) return false; + code = entry.substr(0, separator); + text = entry.substr(separator + 1); + return !code.empty() && !text.empty(); +} + } // namespace inkflow #endif diff --git a/Linux/fcitx5/src/config.h b/Linux/fcitx5/src/config.h new file mode 100644 index 0000000..259e9c2 --- /dev/null +++ b/Linux/fcitx5/src/config.h @@ -0,0 +1,37 @@ +// Minimum daily-use configuration: ~/.config/fcitx5/conf/inkflow.conf. Option names +// follow the macOS preference keys so a backup's settings map one to one. +#ifndef INKFLOW_FCITX5_CONFIG_H +#define INKFLOW_FCITX5_CONFIG_H +#include +#include + +#include +#include + +namespace inkflow { + +FCITX_CONFIGURATION( + Config, + fcitx::Option candidateCount{this, "CandidateCount", "Candidates per page", + 5, fcitx::IntConstrain(3, 9)}; + fcitx::Option abbreviation{this, "Abbreviation", "Abbreviated pinyin", true}; + fcitx::Option typoTolerance{this, "TypoTolerance", "Typo tolerance", true}; + fcitx::Option fuzzyZ{this, "FuzzyZ", "Fuzzy z/zh", false}; + fcitx::Option fuzzyC{this, "FuzzyC", "Fuzzy c/ch", false}; + fcitx::Option fuzzyS{this, "FuzzyS", "Fuzzy s/sh", false}; + fcitx::Option emoji{this, "Emoji", "Emoji suggestions", true}; + fcitx::Option bracketPaging{this, "BracketPaging", "Page with [ and ]", true}; + fcitx::Option minusEqualPaging{this, "MinusEqualPaging", "Page with - and =", true}; + fcitx::Option englishPunctuation{this, "EnglishPunctuation", "English punctuation", false}; + fcitx::Option cornerQuotes{this, "CornerQuotes", "Corner quotes for braces", true}; + fcitx::Option middleDot{this, "MiddleDot", "Middle dot for backquote", true}; + fcitx::Option fullwidthPipe{this, "FullwidthPipe", "Fullwidth pipe", true}; + fcitx::Option ideographicComma{this, "IdeographicComma", "Ideographic comma for backslash", true}; + fcitx::Option traditional{this, "Traditional", "Traditional Chinese output", false}; + fcitx::Option> customPhrases{this, "CustomPhrases", + "Custom phrases, one code=text per entry"}; + fcitx::Option importBackup{this, "ImportBackup", + "Path of a macOS personal backup to import once"};); + +} // namespace inkflow +#endif diff --git a/Linux/fcitx5/src/engine.cpp b/Linux/fcitx5/src/engine.cpp index 487408e..85ead5c 100644 --- a/Linux/fcitx5/src/engine.cpp +++ b/Linux/fcitx5/src/engine.cpp @@ -1,7 +1,7 @@ #include "engine.h" +#include #include -#include #include #include #include @@ -11,14 +11,17 @@ #include #include - -#include "bridge.h" +#include +#include +#include +#include namespace inkflow { FCITX_DEFINE_LOG_CATEGORY(inkflow_log, "inkflow"); namespace { +constexpr const char* kConfigFile = "conf/inkflow.conf"; // The engine keeps 16 graphemes; a few more code points cover combining sequences. constexpr std::size_t kPrecedingCodePoints = 32; @@ -29,37 +32,12 @@ bool directory_exists(const std::string& path) { } // namespace Engine::Engine(fcitx::Instance* instance) - : instance_(instance), factory_([this](fcitx::InputContext&) { - auto* state = new State; - if (!engine_) return state; - IFRStatus status = ifr_session_create(engine_, &state->session); - if (status != IFR_OK) { - INKFLOW_WARN() << "session create failed: " << status << " " << ifr_last_error(); - state->session = nullptr; - return state; - } - uint32_t options = ifr_input_options_default(); - ifr_session_set_configuration(state->session, 5, nullptr, 0, &options); - if (const char* error = ifr_session_configuration_error(state->session)) { - INKFLOW_WARN() << "configuration error: " << error; - } - return state; - }) { - Paths paths = resolve_paths([](const char* name) { return std::getenv(name); }, directory_exists); - if (paths.shared.empty()) { - INKFLOW_WARN() << "no prepared resources under XDG data directories; keys pass through"; - } else { - IFREngineConfig config = {paths.shared.c_str(), paths.user.c_str(), paths.cache.c_str(), - paths.context_index.c_str()}; - IFRStatus status = ifr_engine_create(&config, &engine_); - if (status != IFR_OK) { - INKFLOW_WARN() << "engine create failed: " << status << " " << ifr_last_error(); - engine_ = nullptr; - } else { - INKFLOW_INFO() << "engine ready: " << paths.shared << " user " << paths.user; - } - } + : instance_(instance), + paths_(resolve_paths([](const char* name) { return std::getenv(name); }, directory_exists)), + factory_([](fcitx::InputContext&) { return new State; }) { + createEngine(); instance_->inputContextManager().registerProperty("inkflowState", &factory_); + reloadConfig(); } Engine::~Engine() { @@ -68,8 +46,199 @@ Engine::~Engine() { ifr_engine_destroy(engine_); } +void Engine::createEngine() { + if (engine_) return; + if (paths_.shared.empty()) { + INKFLOW_WARN() << "no prepared resources under XDG data directories; keys pass through"; + return; + } + IFREngineConfig config = {paths_.shared.c_str(), paths_.user.c_str(), paths_.cache.c_str(), + paths_.context_index.c_str()}; + IFRStatus status = ifr_engine_create(&config, &engine_); + if (status != IFR_OK) { + INKFLOW_WARN() << "engine create failed: " << status << " " << ifr_last_error(); + engine_ = nullptr; + return; + } + INKFLOW_INFO() << "engine ready: " << paths_.shared << " user " << paths_.user; +} + +// Every session and the engine reference go, so Rime finalizes before this returns. +void Engine::destroyEngine() { + instance_->inputContextManager().foreach([this](fcitx::InputContext* ic) { + State* st = state(ic); + if (st->session) { + st->release(); + ic->inputPanel().reset(); + if (ic->hasFocus()) { + ic->updatePreedit(); + ic->updateUserInterface(fcitx::UserInterfaceComponent::InputPanel); + } + } + return true; + }); + ifr_engine_destroy(engine_); + engine_ = nullptr; +} + State* Engine::state(fcitx::InputContext* ic) { return ic->propertyFor(&factory_); } +IFRSession* Engine::session(State* st) { + if (st->session || !engine_) return st->session; + IFRStatus status = ifr_session_create(engine_, &st->session); + if (status != IFR_OK) { + INKFLOW_WARN() << "session create failed: " << status << " " << ifr_last_error(); + st->session = nullptr; + return nullptr; + } + applyConfiguration(st->session); + return st->session; +} + +uint32_t Engine::inputOptions() const { + struct Bit { + const fcitx::Option& option; + uint32_t mask; + }; + const Bit bits[] = { + {config_.abbreviation, IFR_OPTION_ABBREVIATION}, + {config_.typoTolerance, IFR_OPTION_TYPO_TOLERANCE}, + {config_.fuzzyZ, IFR_OPTION_FUZZY_Z}, + {config_.fuzzyC, IFR_OPTION_FUZZY_C}, + {config_.fuzzyS, IFR_OPTION_FUZZY_S}, + {config_.emoji, IFR_OPTION_EMOJI}, + {config_.bracketPaging, IFR_OPTION_BRACKET_PAGING}, + {config_.minusEqualPaging, IFR_OPTION_MINUS_EQUAL_PAGING}, + {config_.englishPunctuation, IFR_OPTION_ENGLISH_PUNCTUATION}, + {config_.cornerQuotes, IFR_OPTION_CORNER_QUOTES}, + {config_.middleDot, IFR_OPTION_MIDDLE_DOT}, + {config_.fullwidthPipe, IFR_OPTION_FULLWIDTH_PIPE}, + {config_.ideographicComma, IFR_OPTION_IDEOGRAPHIC_COMMA}, + {config_.traditional, IFR_OPTION_TRADITIONAL}, + }; + uint32_t mask = 0; + for (const Bit& bit : bits) { + if (*bit.option) mask |= bit.mask; + } + return mask; +} + +void Engine::applyConfiguration(IFRSession* session) { + std::vector strings; + const auto& entries = *config_.customPhrases; + strings.reserve(entries.size() * 3); + std::vector phrases; + for (std::size_t index = 0; index < entries.size(); ++index) { + std::string code, text; + if (!parse_phrase(entries[index], code, text)) { + INKFLOW_WARN() << "skipping custom phrase without code=text: " << entries[index]; + continue; + } + strings.push_back(std::to_string(index)); + strings.push_back(std::move(code)); + strings.push_back(std::move(text)); + } + for (std::size_t i = 0; i + 2 < strings.size(); i += 3) { + phrases.push_back({strings[i].c_str(), strings[i + 1].c_str(), strings[i + 2].c_str()}); + } + uint32_t options = inputOptions(); + IFRStatus status = ifr_session_set_configuration(session, static_cast(*config_.candidateCount), + phrases.data(), phrases.size(), &options); + if (status != IFR_OK) { + INKFLOW_WARN() << "set_configuration failed: " << status << " " << ifr_last_error(); + } else if (const char* error = ifr_session_configuration_error(session)) { + INKFLOW_WARN() << "configuration not applied: " << error; + } +} + +void Engine::applyConfigurationToSessions() { + instance_->inputContextManager().foreach([this](fcitx::InputContext* ic) { + State* st = state(ic); + if (st->session) applyConfiguration(st->session); + return true; + }); +} + +void Engine::saveConfig() { fcitx::safeSaveAsIni(config_, kConfigFile); } + +void Engine::reloadConfig() { + fcitx::readAsIni(config_, kConfigFile); + if (!config_.importBackup->empty()) { + importBackup(*config_.importBackup); + } + applyConfigurationToSessions(); +} + +void Engine::setConfig(const fcitx::RawConfig& raw) { + config_.load(raw, true); + saveConfig(); + if (!config_.importBackup->empty()) { + importBackup(*config_.importBackup); + } + applyConfigurationToSessions(); +} + +void Engine::importBackup(const std::string& file) { + // The request is consumed whatever happens, so a bad file cannot repeat on every reload. + config_.importBackup.setValue(std::string()); + saveConfig(); + std::ifstream input(file, std::ios::binary); + if (!input) { + INKFLOW_WARN() << "backup not readable: " << file; + return; + } + std::vector bytes((std::istreambuf_iterator(input)), std::istreambuf_iterator()); + IFRBackup* backup = nullptr; + IFRStatus status = ifr_backup_parse(bytes.data(), bytes.size(), &backup); + if (status != IFR_OK) { + INKFLOW_WARN() << "backup rejected: " << status << " " << ifr_last_error(); + return; + } + for (std::size_t i = 0; i < ifr_backup_unsupported_count(backup); ++i) { + INKFLOW_INFO() << "backup preference not portable, skipped: " << ifr_backup_unsupported(backup, i); + } + destroyEngine(); + status = ifr_backup_import(backup, paths_.user.c_str()); + if (status == IFR_RECOVERY_REQUIRED) { + INKFLOW_WARN() << "finishing an interrupted import first"; + if (ifr_personal_recover(paths_.user.c_str()) == IFR_OK) { + status = ifr_backup_import(backup, paths_.user.c_str()); + } + } + if (status != IFR_OK) { + INKFLOW_WARN() << "import failed, user data unchanged: " << status << " " << ifr_last_error(); + } else { + config_.candidateCount.setValue(static_cast(ifr_backup_candidate_count(backup))); + uint32_t options = ifr_backup_input_options(backup); + config_.abbreviation.setValue((options & IFR_OPTION_ABBREVIATION) != 0); + config_.typoTolerance.setValue((options & IFR_OPTION_TYPO_TOLERANCE) != 0); + config_.fuzzyZ.setValue((options & IFR_OPTION_FUZZY_Z) != 0); + config_.fuzzyC.setValue((options & IFR_OPTION_FUZZY_C) != 0); + config_.fuzzyS.setValue((options & IFR_OPTION_FUZZY_S) != 0); + config_.emoji.setValue((options & IFR_OPTION_EMOJI) != 0); + config_.bracketPaging.setValue((options & IFR_OPTION_BRACKET_PAGING) != 0); + config_.minusEqualPaging.setValue((options & IFR_OPTION_MINUS_EQUAL_PAGING) != 0); + config_.englishPunctuation.setValue((options & IFR_OPTION_ENGLISH_PUNCTUATION) != 0); + config_.cornerQuotes.setValue((options & IFR_OPTION_CORNER_QUOTES) != 0); + config_.middleDot.setValue((options & IFR_OPTION_MIDDLE_DOT) != 0); + config_.fullwidthPipe.setValue((options & IFR_OPTION_FULLWIDTH_PIPE) != 0); + config_.ideographicComma.setValue((options & IFR_OPTION_IDEOGRAPHIC_COMMA) != 0); + config_.traditional.setValue((options & IFR_OPTION_TRADITIONAL) != 0); + std::vector entries; + for (std::size_t i = 0; i < ifr_backup_phrase_count(backup); ++i) { + IFRPhrase phrase = {nullptr, nullptr, nullptr}; + if (ifr_backup_phrase(backup, i, &phrase)) { + entries.push_back(std::string(phrase.code) + "=" + phrase.text); + } + } + config_.customPhrases.setValue(entries); + saveConfig(); + INKFLOW_INFO() << "imported personal data from " << file; + } + ifr_backup_free(backup); + createEngine(); +} + bool Engine::sensitive(const fcitx::InputContext* ic) const { return ic->capabilityFlags().testAny( fcitx::CapabilityFlags{fcitx::CapabilityFlag::Password, fcitx::CapabilityFlag::Sensitive}); @@ -96,12 +265,12 @@ void Engine::refreshContext(fcitx::InputContext* ic, State* state) { void Engine::keyEvent(const fcitx::InputMethodEntry&, fcitx::KeyEvent& event) { auto* ic = event.inputContext(); State* st = state(ic); - if (!st->session) return; if (sensitive(ic)) { // Never compose or learn in password and sensitive fields; the client gets raw keys. - if (st->composing()) clear(ic, st); + if (st->session && st->composing()) clear(ic, st); return; } + if (!session(st)) return; const fcitx::Key& key = event.rawKey(); if (!st->composing()) refreshContext(ic, st); int handled = 0; @@ -118,7 +287,7 @@ void Engine::keyEvent(const fcitx::InputMethodEntry&, fcitx::KeyEvent& event) { void Engine::activate(const fcitx::InputMethodEntry&, fcitx::InputContextEvent& event) { State* st = state(event.inputContext()); - if (st->session) refresh(event.inputContext(), st); + if (session(st)) refresh(event.inputContext(), st); } void Engine::deactivate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) { diff --git a/Linux/fcitx5/src/engine.h b/Linux/fcitx5/src/engine.h index 542d8ab..131ded8 100644 --- a/Linux/fcitx5/src/engine.h +++ b/Linux/fcitx5/src/engine.h @@ -1,6 +1,7 @@ // Fcitx5 input-method engine over the InkFlow C ABI. One engine per addon, one engine -// session per input context. Keys are processed synchronously on Fcitx5's thread with -// no event-loop dependency; the engine never touches network, SQLite or telemetry. +// session per input context, created on first use. Keys are processed synchronously on +// Fcitx5's thread with no event-loop dependency; the engine never touches network, +// SQLite or telemetry. #ifndef INKFLOW_FCITX5_ENGINE_H #define INKFLOW_FCITX5_ENGINE_H #include @@ -11,7 +12,11 @@ #include #include +#include + +#include "bridge.h" #include "candidates.h" +#include "config.h" #include "inkflow_rime.h" namespace inkflow { @@ -22,12 +27,17 @@ FCITX_DECLARE_LOG_CATEGORY(inkflow_log); class State final : public fcitx::InputContextProperty { public: - ~State() override { ifr_session_destroy(session); } - IFRSession* session = nullptr; - SnapshotRef snapshot; + ~State() override { release(); } + void release() { + snapshot.reset(); + ifr_session_destroy(session); + session = nullptr; + } bool composing() const { return snapshot && ifr_snapshot_preedit(snapshot.get()) && *ifr_snapshot_preedit(snapshot.get()); } + IFRSession* session = nullptr; + SnapshotRef snapshot; }; class Engine final : public fcitx::InputMethodEngineV2 { @@ -39,6 +49,9 @@ class Engine final : public fcitx::InputMethodEngineV2 { void activate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; void deactivate(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; void reset(const fcitx::InputMethodEntry& entry, fcitx::InputContextEvent& event) override; + void reloadConfig() override; + const fcitx::Configuration* getConfig() const override { return &config_; } + void setConfig(const fcitx::RawConfig& raw) override; // Called by the candidate list; each rebuilds the panel. void selectCandidate(fcitx::InputContext* ic, const SnapshotRef& snapshot, std::size_t index); @@ -47,14 +60,27 @@ class Engine final : public fcitx::InputMethodEngineV2 { private: State* state(fcitx::InputContext* ic); + // The input context's engine session, created and configured on first use. + IFRSession* session(State* state); bool sensitive(const fcitx::InputContext* ic) const; void refreshContext(fcitx::InputContext* ic, State* state); // Drain the commit exactly once, then rebuild preedit and candidates from a fresh snapshot. void refresh(fcitx::InputContext* ic, State* state); void clear(fcitx::InputContext* ic, State* state); void failed(fcitx::InputContext* ic, State* state, const char* what, IFRStatus status); + void createEngine(); + void destroyEngine(); + uint32_t inputOptions() const; + void applyConfiguration(IFRSession* session); + void applyConfigurationToSessions(); + void saveConfig(); + // The explicit personal-data entry point: stops the engine, imports, restarts, and + // writes the backup's settings into the configuration. Never runs on the key path. + void importBackup(const std::string& file); fcitx::Instance* instance_; + Paths paths_; + Config config_; IFREngine* engine_ = nullptr; fcitx::FactoryFor factory_; }; diff --git a/Linux/fcitx5/tests/bridge_test.cpp b/Linux/fcitx5/tests/bridge_test.cpp index 7954390..ab0380b 100644 --- a/Linux/fcitx5/tests/bridge_test.cpp +++ b/Linux/fcitx5/tests/bridge_test.cpp @@ -57,6 +57,10 @@ int main() { assert(preceding_text("中文abc", 6, 16) == ""); assert(preceding_text("\xff\xfe", 1, 16) == ""); assert(preceding_text("a\xe4\xb8", 2, 16) == ""); - std::puts("PASS fcitx5 bridge helpers: XDG paths, preedit layout, key states, preceding text"); + std::string code, text; + assert(parse_phrase("dz=地址", code, text) && code == "dz" && text == "地址"); + assert(parse_phrase("a=b=c", code, text) && code == "a" && text == "b=c"); + assert(!parse_phrase("nothing", code, text) && !parse_phrase("=x", code, text) && !parse_phrase("x=", code, text)); + std::puts("PASS fcitx5 bridge helpers: XDG paths, preedit layout, key states, preceding text, phrases"); return 0; } From 8a4c9688b03986bf1f3de93a5120a9a92a6f8885 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 19:32:17 +0800 Subject: [PATCH 16/21] fix(linux): install and verify the Fcitx5 addon on ARM64 Omarchy Use the available SHA-256 tool and an explicit addon library path for user-local installs. Prevent raw preedit commits on focus loss, retain the backup path while consuming an import request, and omit phrase text from warnings. Add isolated installed-addon tests covering commits, focus, sensitive fields, configuration and backup import. Refs #36. --- Core/scripts/build-dictionary-generator.sh | 4 +- Core/scripts/prepare-chinese.sh | 12 +- Core/scripts/prepare-rime.sh | 8 +- Core/scripts/resource-dependencies.sh | 8 +- Linux/fcitx5/README.md | 24 +++- Linux/fcitx5/inkflow.conf.in | 2 +- Linux/fcitx5/src/engine.cpp | 7 +- Linux/fcitx5/src/engine.h | 2 +- Linux/fcitx5/test-installed.sh | 16 +++ Linux/fcitx5/tests/installed_smoke.py | 158 +++++++++++++++++++++ 10 files changed, 219 insertions(+), 22 deletions(-) create mode 100644 Linux/fcitx5/test-installed.sh create mode 100644 Linux/fcitx5/tests/installed_smoke.py diff --git a/Core/scripts/build-dictionary-generator.sh b/Core/scripts/build-dictionary-generator.sh index e2a595e..169cbfa 100644 --- a/Core/scripts/build-dictionary-generator.sh +++ b/Core/scripts/build-dictionary-generator.sh @@ -1,12 +1,14 @@ #!/bin/bash set -euo pipefail cd "$(dirname "$0")/../.." +sha256=(shasum -a 256) +if command -v sha256sum >/dev/null; then sha256=(sha256sum); fi export CARGO_TARGET_DIR="$PWD/build/dictionary/cargo" cargo build --locked --release --manifest-path Core/Portable/dictionary/Cargo.toml mkdir -p build/dictionary/include cp -p "$CARGO_TARGET_DIR/release/inkflow-dictionary" build/dictionary-generator # SwiftPM tracks this C include, so a changed Rust archive also relinks its callers. -identity=$(shasum -a 256 "$CARGO_TARGET_DIR/release/libinkflow_dictionary.a" | cut -d ' ' -f 1) +identity=$("${sha256[@]}" "$CARGO_TARGET_DIR/release/libinkflow_dictionary.a" | cut -d ' ' -f 1) printf '#define INKFLOW_DICTIONARY_BUILD_ID "%s"\n' "$identity" > build/dictionary/include/build_identity.h.tmp if ! cmp -s build/dictionary/include/build_identity.h.tmp build/dictionary/include/build_identity.h; then mv build/dictionary/include/build_identity.h.tmp build/dictionary/include/build_identity.h diff --git a/Core/scripts/prepare-chinese.sh b/Core/scripts/prepare-chinese.sh index cdc8b3f..33eb27a 100644 --- a/Core/scripts/prepare-chinese.sh +++ b/Core/scripts/prepare-chinese.sh @@ -1,6 +1,8 @@ #!/bin/bash set -euo pipefail cd "$(dirname "$0")/../.." +sha256=(shasum -a 256) +if command -v sha256sum >/dev/null; then sha256=(sha256sum); fi mode=generate case "${1:-}" in --sources-only) mode=sources-only; destination="" ;; @@ -19,28 +21,28 @@ while IFS=$'\t' read -r identifier sha bytes url; do if [[ ! -f "$source_file" ]]; then curl --fail --location --proto '=https' --connect-timeout 20 --max-time 180 --retry 2 \ --max-filesize "$bytes" "$url" -o "$staging/$identifier.yaml" - printf '%s %s\n' "$sha" "$staging/$identifier.yaml" | shasum -a 256 -c - + printf '%s %s\n' "$sha" "$staging/$identifier.yaml" | "${sha256[@]}" -c - [[ $(wc -c < "$staging/$identifier.yaml") -eq $bytes ]] mv "$staging/$identifier.yaml" "$source_file" fi - printf '%s %s\n' "$sha" "$source_file" | shasum -a 256 -c - > /dev/null + printf '%s %s\n' "$sha" "$source_file" | "${sha256[@]}" -c - > /dev/null done < "$staging/sources.tsv" [[ "$mode" == generate ]] || exit 0 mkdir -p "$destination" # Cache is only a build optimization. Every raw input is verified above and the key # includes the executable, exact legacy bytes and local correction rules. -shasum -a 256 build/dictionary-generator build/dictionary-sources/*.yaml "${legacy[0]}" \ +"${sha256[@]}" build/dictionary-generator build/dictionary-sources/*.yaml "${legacy[0]}" \ Core/config/chinese-overrides.tsv > "$staging/inputs.sha256" cache=build/generated-chinese cache_valid=false if [[ -s "$cache/outputs.sha256" ]]; then - if (cd "$cache" && shasum -a 256 -c outputs.sha256 > /dev/null 2>&1); then cache_valid=true; fi + if (cd "$cache" && "${sha256[@]}" -c outputs.sha256 > /dev/null 2>&1); then cache_valid=true; fi fi if [[ ! -f "$cache/inputs.sha256" ]] || ! cmp -s "$staging/inputs.sha256" "$cache/inputs.sha256" \ || ! $cache_valid; then build/dictionary-generator generate build/dictionary-sources "${legacy[0]}" Core/config/chinese-overrides.tsv "$staging/generated" cp "$staging/inputs.sha256" "$staging/generated/inputs.sha256" - (cd "$staging/generated" && shasum -a 256 pinyin_simp.dict.yaml dictionary-manifest.json > outputs.sha256) + (cd "$staging/generated" && "${sha256[@]}" pinyin_simp.dict.yaml dictionary-manifest.json > outputs.sha256) mkdir -p "$cache" cp "$staging/generated/"* "$cache/" fi diff --git a/Core/scripts/prepare-rime.sh b/Core/scripts/prepare-rime.sh index 33e396b..ecc1563 100644 --- a/Core/scripts/prepare-rime.sh +++ b/Core/scripts/prepare-rime.sh @@ -1,6 +1,8 @@ #!/bin/bash set -euo pipefail cd "$(dirname "$0")/../.." +sha256=(shasum -a 256) +if command -v sha256sum >/dev/null; then sha256=(sha256sum); fi requested_destination=${1:?Usage: prepare-rime.sh DESTINATION} source Core/config/english.conf mkdir -p build/rime-cache @@ -18,9 +20,9 @@ for input_root in rust-toolchain.toml Core/Package.swift Core/Sources Core/Tools build/deps/rime-pinyin-simp-* build/deps/rime-easy-en-* build/dictionary-sources; do [[ -e "$input_root" ]] || continue find "$input_root" -type f -print -done | LC_ALL=C sort | while IFS= read -r file; do shasum -a 256 "$file"; done > "$manifest" -[[ ! -f build/deps/emoji.txt ]] || shasum -a 256 build/deps/emoji.txt >> "$manifest" -input_digest=$(shasum -a 256 "$manifest" | awk '{print $1}') +done | LC_ALL=C sort | while IFS= read -r file; do "${sha256[@]}" "$file"; done > "$manifest" +[[ ! -f build/deps/emoji.txt ]] || "${sha256[@]}" build/deps/emoji.txt >> "$manifest" +input_digest=$("${sha256[@]}" "$manifest" | awk '{print $1}') cache="$PWD/build/rime-cache/$input_digest" if [[ ! -f "$cache/complete" || ! -d "$cache/content" ]]; then cache_build=$(mktemp -d "$PWD/build/rime-cache/build.$input_digest.XXXXXX") diff --git a/Core/scripts/resource-dependencies.sh b/Core/scripts/resource-dependencies.sh index badb78c..c644912 100644 --- a/Core/scripts/resource-dependencies.sh +++ b/Core/scripts/resource-dependencies.sh @@ -1,6 +1,8 @@ #!/bin/bash set -euo pipefail cd "$(dirname "$0")/../.." +sha256=(shasum -a 256) +if command -v sha256sum >/dev/null; then sha256=(sha256sum); fi mkdir -p build/deps fetch() { local name="$1" sha="$2" url="$3" @@ -9,7 +11,7 @@ fetch() { "$url" -o "build/deps/$name.part" mv "build/deps/$name.part" "build/deps/$name" fi - printf '%s %s\n' "$sha" "build/deps/$name" | shasum -a 256 -c - + printf '%s %s\n' "$sha" "build/deps/$name" | "${sha256[@]}" -c - } fetch pinyin.tar.gz 46f37114a7929ecc01003a236803c8b1e5198382e6a21f83fae036604a6b08bf https://codeload.github.com/rime/rime-pinyin-simp/tar.gz/0c6861ef7420ee780270ca6d993d18d4101049d0 fetch english.tar.gz 59226ae1bb6da00d8808a0094439271225ac4f533d30cf9150ac482383895461 https://codeload.github.com/BlindingDark/rime-easy-en/tar.gz/54a4a07289412efc54134092c0d945f895a71ed3 @@ -20,8 +22,8 @@ for entry in \ 'english.tar.gz:rime-easy-en-54a4a07289412efc54134092c0d945f895a71ed3/easy_en.dict.yaml'; do archive=${entry%%:*} member=${entry#*:} - expected=$(tar -xOf "build/deps/$archive" "$member" | shasum -a 256 | cut -d ' ' -f 1) - if ! printf '%s %s\n' "$expected" "build/deps/$member" | shasum -a 256 -c - >/dev/null 2>&1; then + expected=$(tar -xOf "build/deps/$archive" "$member" | "${sha256[@]}" | cut -d ' ' -f 1) + if ! printf '%s %s\n' "$expected" "build/deps/$member" | "${sha256[@]}" -c - >/dev/null 2>&1; then tar -xzf "build/deps/$archive" -C build/deps fi done diff --git a/Linux/fcitx5/README.md b/Linux/fcitx5/README.md index 82d591f..d7fcb22 100644 --- a/Linux/fcitx5/README.md +++ b/Linux/fcitx5/README.md @@ -1,12 +1,12 @@ # Fcitx5 adapter -Minimal Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://github.com/nervouna/InkFlow/issues/36). Targets Omarchy/Hyprland and Steam Deck Desktop Mode (KDE Plasma); GNOME/IBus is deferred. This is scaffolding for integration work, not daily-use support: the pieces below are unverified until they run on the Linux host. +Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://github.com/nervouna/InkFlow/issues/36). Targets Omarchy/Hyprland and Steam Deck Desktop Mode (KDE Plasma); GNOME/IBus is deferred. The addon has been built, installed and activated on ARM64 Omarchy with Fcitx5 5.1.23. Application-level typing checks and Steam Deck validation remain open. ## What it does - `src/engine.cpp`: one engine per addon (`ifr_engine_create` on load, from XDG paths), one engine session per Fcitx5 input context (an `InputContextProperty`). Keys go to `ifr_session_key` synchronously on Fcitx5's thread with Rime keysyms and translated modifier masks; releases carry Rime's release mask. Every mutation ends in `refresh`: the commit is drained exactly once with `ifr_session_take_commit` and delivered with `commitString`, then the preedit (client preedit when the client declares `Preedit`, otherwise the panel) and candidate list are rebuilt from a fresh snapshot. Preedit highlight and caret come from the snapshot's UTF-8 byte offsets, which are Fcitx5's units too. - `src/candidates.cpp`: digit selection and Up/Down are the engine's own key policy; mouse clicks and Fcitx5's paging/cursor calls go back through the ABI with the snapshot they were shown from, so a late click on a superseded page is rejected and only redraws. -- Focus and reset: a reset or focus change clears the composition; switching input methods commits it first (what the macOS frontend does on deactivation). Password and `Sensitive` contexts never compose and get raw keys, so nothing can be learned from them. +- Focus and reset: a reset or focus change clears the composition; switching input methods commits it first (what the macOS frontend does on deactivation). Client preedit uses `DontCommit` so Fcitx5 does not commit raw Pinyin before the focus-out handler runs. Password and `Sensitive` contexts never compose and get raw keys, so nothing can be learned from them. - Surrounding text: document context is read only when a composition starts and only when the client declares `SurroundingText` with valid text; otherwise the engine gets no context and keeps Rime's native order. - Configuration (`src/config.h`, `~/.config/fcitx5/conf/inkflow.conf`, also through `fcitx5-configtool`): candidates per page (3–9), the fourteen input options under their macOS names, and custom phrases as `code=text` entries. Settings reach every live session through `ifr_session_set_configuration` and apply at the session's next composition boundary; new sessions are configured when they are created on first use. - Personal-data import (`ImportBackup=/path/to/backup.json` in that file, then reload with `fcitx5-remote -r` or apply in the configuration tool): the addon consumes the request first so a bad file cannot repeat, parses the macOS format-1 document, logs the non-portable preferences it skips, destroys every session and the engine (Rime finalizes), imports the three dictionaries with rollback on failure, writes the backup's candidate count, options and phrases into the configuration, and recreates the engine. Compositions in progress are lost, which is why the entry point is explicit. An interrupted earlier import is recovered first. @@ -16,7 +16,7 @@ Resources are found at `INKFLOW_RESOURCES` (a prepared directory from `Core/Port ## Build on Linux -Requirements: the repository's Rust toolchain, CMake 3.20+, Ninja, a C++20 compiler, Python 3, and the Fcitx5 development files (`fcitx5` on Arch/SteamOS, `libfcitx5core-dev` plus `extra-cmake-modules` on Ubuntu). From the checkout root: +Requirements: the repository's Rust toolchain, CMake 3.31.6 and Ninja 1.11.1.4 for the pinned native build, a C++20 compiler, Python 3, and the Fcitx5 development files (`fcitx5` on Arch/SteamOS, `libfcitx5core-dev` plus `extra-cmake-modules` on Ubuntu). Resource preparation uses `sha256sum` when available, otherwise `shasum -a 256`. From the checkout root: ```sh python3 Core/Portable/build-native.py @@ -31,6 +31,20 @@ The addon links the crate's `staticlib` (which bundles the C++ bridge) and the p `bash Linux/fcitx5/test.sh` runs the helper tests anywhere; with `FCITX5_SOURCE=` it also compiles the addon syntax-only against those headers (export headers stubbed), which is how it was checked from macOS against fcitx5 5.1.14. -## Unverified from macOS +## User-local installation and installed-addon test -The CMake build, `Fcitx5::Core` linkage, addon loading, and all runtime behavior (preedit rendering, candidate window, commit delivery, focus and sensitive-field handling under Hyprland and KDE Plasma, Flatpak clients) have not run. Fcitx5 5.1.14 headers need C++20; older Fcitx5 releases on SteamOS may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). The configuration surface and the import entry point compile against the headers but have not run. Packaging is a later slice. +For an installation without sudo, configure with `-DCMAKE_INSTALL_PREFIX="$HOME/.local" -DFCITX_INSTALL_USE_FCITX_SYS_PATHS=OFF`, then build and install with CMake. The generated addon descriptor names its absolute library path, so Fcitx5 can load it without a global `FCITX_ADDON_DIRS` override. Back up `~/.config/fcitx5/` before enabling InkFlow. Keep the existing keyboard and input-method entries; add `inkflow-pinyin` through Fcitx5 configuration and restart the user's Fcitx5 service. On Omarchy that service is `omarchy-fcitx5.service`. + +The installed-addon test requires `dbus-run-session` and Python with `dbus-next==0.2.3`: + +```sh +PYTHON=/path/to/test-venv/bin/python bash Linux/fcitx5/test-installed.sh "$HOME/.local" +``` + +It starts a private Fcitx5 daemon on a separate D-Bus with temporary configuration and personal data. It checks real addon loading, preedit/candidate signals, mouse and space selection, exactly-once commits, focus/reset cancellation, sensitive-field pass-through, configuration reload, custom phrases, text-free warnings, and explicit backup import with engine recreation. It does not type into desktop applications or import into live user dictionaries. + +## Target verification + +On ARM64 Arch Linux/Omarchy with Hyprland, Fcitx5 5.1.23 and GCC 16.1.1, the native build, Rust runtime tests, 291 ranking cases, production resource preparation, 21-sample learned baseline, engine/personal-data parity, C ABI consumers, addon build, helper tests and installed-addon test passed. The desktop service loaded InkFlow and reported `inkflow-pinyin` as selected; the existing US keyboard and Pinyin entries were retained. + +Actual application rendering and typing still need a user check. Steam Deck/KDE Plasma, older Fcitx5 versions, Flatpak clients, distribution packaging and OS-update persistence have not been verified. Fcitx5 5.1.14 headers need C++20; older releases may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). diff --git a/Linux/fcitx5/inkflow.conf.in b/Linux/fcitx5/inkflow.conf.in index 4ed75e6..f05e436 100644 --- a/Linux/fcitx5/inkflow.conf.in +++ b/Linux/fcitx5/inkflow.conf.in @@ -2,7 +2,7 @@ Name=InkFlow Comment=InkFlow Pinyin on the shared Rust/Rime core Category=InputMethod -Library=libinkflow +Library=@FCITX_INSTALL_ADDONDIR@/libinkflow Type=SharedLibrary OnDemand=True Configurable=True diff --git a/Linux/fcitx5/src/engine.cpp b/Linux/fcitx5/src/engine.cpp index 85ead5c..095697e 100644 --- a/Linux/fcitx5/src/engine.cpp +++ b/Linux/fcitx5/src/engine.cpp @@ -131,7 +131,7 @@ void Engine::applyConfiguration(IFRSession* session) { for (std::size_t index = 0; index < entries.size(); ++index) { std::string code, text; if (!parse_phrase(entries[index], code, text)) { - INKFLOW_WARN() << "skipping custom phrase without code=text: " << entries[index]; + INKFLOW_WARN() << "skipping custom phrase without code=text at index " << index; continue; } strings.push_back(std::to_string(index)); @@ -178,7 +178,7 @@ void Engine::setConfig(const fcitx::RawConfig& raw) { applyConfigurationToSessions(); } -void Engine::importBackup(const std::string& file) { +void Engine::importBackup(std::string file) { // The request is consumed whatever happens, so a bad file cannot repeat on every reload. config_.importBackup.setValue(std::string()); saveConfig(); @@ -377,7 +377,8 @@ void Engine::refresh(fcitx::InputContext* ic, State* st) { ifr_snapshot_selection_end(raw)); fcitx::Text text; for (const auto& segment : layout.segments) { - fcitx::TextFormatFlags flags = fcitx::TextFormatFlag::Underline; + fcitx::TextFormatFlags flags{fcitx::TextFormatFlag::Underline, + fcitx::TextFormatFlag::DontCommit}; if (segment.highlighted) flags |= fcitx::TextFormatFlag::HighLight; text.append(segment.text, flags); } diff --git a/Linux/fcitx5/src/engine.h b/Linux/fcitx5/src/engine.h index 131ded8..39a783e 100644 --- a/Linux/fcitx5/src/engine.h +++ b/Linux/fcitx5/src/engine.h @@ -76,7 +76,7 @@ class Engine final : public fcitx::InputMethodEngineV2 { void saveConfig(); // The explicit personal-data entry point: stops the engine, imports, restarts, and // writes the backup's settings into the configuration. Never runs on the key path. - void importBackup(const std::string& file); + void importBackup(std::string file); fcitx::Instance* instance_; Paths paths_; diff --git a/Linux/fcitx5/test-installed.sh b/Linux/fcitx5/test-installed.sh new file mode 100644 index 0000000..a420fad --- /dev/null +++ b/Linux/fcitx5/test-installed.sh @@ -0,0 +1,16 @@ +#!/bin/bash +# Requires dbus-next in PYTHON; never connects to the desktop's D-Bus or user data. +set -euo pipefail +cd "$(dirname "$0")/../.." +prefix=${1:-$HOME/.local} +scratch=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-installed.XXXXXX") +trap 'rm -rf "$scratch"' EXIT +mkdir -p "$scratch/config/fcitx5" "$scratch/data" "$scratch/cache" "$scratch/runtime" +chmod 700 "$scratch/runtime" +printf '[Groups/0]\nName=Default\nDefault Layout=us\nDefaultIM=inkflow-pinyin\n\n[Groups/0/Items/0]\nName=keyboard-us\n\n[Groups/0/Items/1]\nName=inkflow-pinyin\n\n[GroupOrder]\n0=Default\n' > "$scratch/config/fcitx5/profile" +env -u DISPLAY -u WAYLAND_DISPLAY -u FCITX_CONFIG_HOME -u FCITX_DATA_HOME \ + INKFLOW_ISOLATED_TEST=1 INKFLOW_RESOURCES="$prefix/share/inkflow/rime" \ + XDG_CONFIG_HOME="$scratch/config" XDG_DATA_HOME="$scratch/data" \ + XDG_CACHE_HOME="$scratch/cache" XDG_RUNTIME_DIR="$scratch/runtime" \ + FCITX_DATA_DIRS="$prefix/share/fcitx5:/usr/share/fcitx5" \ + dbus-run-session -- "${PYTHON:-python3}" Linux/fcitx5/tests/installed_smoke.py diff --git a/Linux/fcitx5/tests/installed_smoke.py b/Linux/fcitx5/tests/installed_smoke.py new file mode 100644 index 0000000..b2673fc --- /dev/null +++ b/Linux/fcitx5/tests/installed_smoke.py @@ -0,0 +1,158 @@ +#!/usr/bin/env python3 +"""Exercise an installed addon on a private D-Bus with disposable personal data.""" +import asyncio +import json +import os +from pathlib import Path +import sys + +from dbus_next.aio import MessageBus + + +async def main(): + assert os.environ.get("INKFLOW_ISOLATED_TEST") == "1" + log = Path(os.environ["XDG_CONFIG_HOME"]) / "fcitx.log" + with log.open("w") as output: + daemon = await asyncio.create_subprocess_exec( + "fcitx5", "-k", "--disable", "all", "--enable", + "dbus,dbusfrontend,keyboard,inkflow", stdout=output, stderr=output, + ) + bus = await MessageBus().connect() + context = controller = None + try: + tree = await bus.introspect("org.freedesktop.DBus", "/org/freedesktop/DBus") + names = bus.get_proxy_object( + "org.freedesktop.DBus", "/org/freedesktop/DBus", tree + ).get_interface("org.freedesktop.DBus") + for _ in range(100): + if await names.call_name_has_owner("org.fcitx.Fcitx5"): + tree = await bus.introspect("org.fcitx.Fcitx5", "/controller") + break + if daemon.returncode is not None: + raise RuntimeError("Fcitx5 exited before registering D-Bus") + await asyncio.sleep(0.1) + else: + raise RuntimeError("Fcitx5 did not register D-Bus") + controller = bus.get_proxy_object("org.fcitx.Fcitx5", "/controller", tree).get_interface( + "org.fcitx.Fcitx.Controller1" + ) + tree = await bus.introspect("org.fcitx.Fcitx5", "/org/freedesktop/portal/inputmethod") + frontend = bus.get_proxy_object( + "org.fcitx.Fcitx5", "/org/freedesktop/portal/inputmethod", tree + ).get_interface("org.fcitx.Fcitx.InputMethod1") + path, _ = await frontend.call_create_input_context([["program", "inkflow-isolated-smoke"]]) + tree = await bus.introspect("org.fcitx.Fcitx5", path) + context = bus.get_proxy_object("org.fcitx.Fcitx5", path, tree).get_interface( + "org.fcitx.Fcitx.InputContext1" + ) + commits, preedits, panels = [], [], [] + context.on_commit_string(lambda text: commits.append(text)) + context.on_update_formatted_preedit(lambda text, cursor: preedits.append((text, cursor))) + + def panel(preedit, cursor, upper, lower, candidates, index, layout, previous, following): + panels.append(candidates) + + context.on_update_client_side_ui(panel) + capabilities = (1 << 1) | (1 << 4) | (1 << 39) + await context.call_set_capability(capabilities) + await context.call_focus_in() + await controller.call_set_current_im("inkflow-pinyin") + await controller.call_activate() + assert await controller.call_current_input_method() == "inkflow-pinyin" + + async def key(code, release=False): + return await context.call_process_key_event(code, 0, 0, release, 0) + + async def type_text(text): + for char in text: + assert await key(ord(char)), f"unhandled fixture key: {char}" + await key(ord(char), True) + + await type_text("nihao") + assert any(parts for parts, _ in preedits), "no preedit received" + assert panels and panels[-1], "no candidates received" + assert panels[-1][0][1] == "你好", panels[-1] + await context.call_select_candidate(0) + assert commits == ["你好"], commits + assert not preedits[-1][0], "preedit survived selection" + print("PASS installed addon: preedit, candidate panel, click selection, one commit") + + await type_text("nihao") + assert await key(32) + await key(32, True) + assert commits == ["你好", "你好"], commits + await type_text("nihao") + await context.call_reset() + assert not preedits[-1][0] + await type_text("nihao") + await context.call_focus_out() + await context.call_focus_in() + assert not preedits[-1][0] + assert commits == ["你好", "你好"], commits + print("PASS installed addon: space selection, key releases, reset and focus cancellation") + + for flag in (1 << 3, 1 << 36): + await context.call_set_capability(capabilities | flag) + assert not await key(ord("n")), "sensitive key was consumed" + assert commits == ["你好", "你好"] + await context.call_set_capability(capabilities) + await controller.call_set_current_im("inkflow-pinyin") + await controller.call_activate() + await type_text("nihao") + await context.call_reset() + print("PASS installed addon: password/sensitive pass-through and recovery") + + config = Path(os.environ["XDG_CONFIG_HOME"]) / "fcitx5/conf/inkflow.conf" + config.parent.mkdir(exist_ok=True) + marker = "fixture-phrase-must-not-appear-in-logs" + config.write_text(f"CandidateCount=7\n\n[CustomPhrases]\n0={marker}\n1=zzcs=安装测试\n") + await controller.call_reload_addon_config("inkflow") + await type_text("zzcs") + assert panels[-1][0][1] == "安装测试", panels[-1] + await context.call_select_candidate(0) + assert commits[-1] == "安装测试" + assert marker not in log.read_text() + print("PASS installed addon: configuration reload, custom phrase, text-free warning") + + backup = config.parent / "fixture-backup.json" + backup.write_text(json.dumps({ + "format": 1, "rime": "1.17.0", + "settings": { + "integers": { + "candidateCount": 7, "fontSize": 18, + "input.abbreviation": 1, "input.typoTolerance": 1, + "input.fuzzyZ": 0, "input.fuzzyC": 0, "input.fuzzyS": 0, + "input.emoji": 1, "input.bracketPaging": 1, + "input.minusEqualPaging": 1, "input.englishPunctuation": 0, + "input.cornerQuotes": 1, "input.middleDot": 1, + "input.fullwidthPipe": 1, "input.ideographicComma": 1, + "input.traditional": 0, + }, + "phrases": [{"id": "p1", "code": "zzbk", "text": "备份测试"}], + }, + "dictionaries": {"pinyin_simp": None, "inkflow_shared_english": None, + "inkflow_voice_alias": None}, + })) + config.write_text(f"ImportBackup={backup}\n") + await controller.call_reload_addon_config("inkflow") + assert f"ImportBackup={backup}" not in config.read_text() + await type_text("zzbk") + assert panels[-1][0][1] == "备份测试", panels[-1] + await context.call_select_candidate(0) + assert commits[-1] == "备份测试" + assert "imported personal data from" in log.read_text() + print("PASS installed addon: explicit backup import, engine restart, restored phrase") + finally: + if context: + await context.call_destroy_ic() + if controller: + await controller.call_exit() + elif daemon.returncode is None: + daemon.terminate() + await asyncio.wait_for(daemon.wait(), 10) + bus.disconnect() + if sys.exc_info()[0]: + print(log.read_text(), file=sys.stderr) + + +asyncio.run(asyncio.wait_for(main(), 60)) From 9d76112b2b738c6918811885124c3cb82d77d362 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 19:57:48 +0800 Subject: [PATCH 17/21] feat(linux): package and manage user-local Fcitx5 installations Stage relocatable native libraries, target resources, notices and corresponding sources. Install immutable releases with checksum verification, service-aware recovery, rollback, manual-registration restoration and data-preserving uninstall. Add isolated installer and real-package lifecycle checks. Keep Steam Deck validation and the macOS cutover out of this iteration. Refs #36. --- .github/workflows/ci.yml | 6 +- Core/Portable/build-native.py | 3 +- Linux/README.md | 77 +++++++ Linux/fcitx5/CMakeLists.txt | 17 +- Linux/fcitx5/README.md | 12 +- Linux/fcitx5/src/bridge.h | 1 + Linux/fcitx5/test-installed.sh | 6 +- Linux/fcitx5/tests/bridge_test.cpp | 4 + Linux/scripts/install.py | 343 +++++++++++++++++++++++++++++ Linux/scripts/package-notices.py | 214 ++++++++++++++++++ Linux/scripts/package.sh | 46 ++++ Linux/scripts/test_install.py | 302 +++++++++++++++++++++++++ Linux/scripts/test_package.py | 77 +++++++ 13 files changed, 1095 insertions(+), 13 deletions(-) create mode 100644 Linux/README.md create mode 100644 Linux/scripts/install.py create mode 100644 Linux/scripts/package-notices.py create mode 100755 Linux/scripts/package.sh create mode 100644 Linux/scripts/test_install.py create mode 100644 Linux/scripts/test_package.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6e4dc58..2e2a9ba 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -67,10 +67,14 @@ jobs: run: bash Core/Portable/dictionary/test.sh scripts: - name: Remote runner tests + name: Script and frontend helper tests runs-on: ubuntu-24.04 timeout-minutes: 10 steps: - uses: actions/checkout@v5 - name: Test scripts/mac-remote.py run: python3 -B scripts/tests/test_mac_remote.py + - name: Test Linux installer and frontend helpers + run: | + python3 -B Linux/scripts/test_install.py + bash Linux/fcitx5/test.sh diff --git a/Core/Portable/build-native.py b/Core/Portable/build-native.py index ed96166..b33fd46 100644 --- a/Core/Portable/build-native.py +++ b/Core/Portable/build-native.py @@ -83,7 +83,8 @@ def cmake(name, *options, source=None): '-DINSTALL_PRIVATE_HEADERS=ON', '-DENABLE_TIMESTAMP=OFF', '-DCMAKE_DISABLE_FIND_PACKAGE_Gflags=ON', f'-DBOOST_ROOT={PREFIX}', f'-DBoost_INCLUDE_DIR={PREFIX}/include', - f'-DCMAKE_INSTALL_RPATH={PREFIX}/lib') + # Shared librime and prefix/bin tools both need relocatable lookup. + f'-DCMAKE_INSTALL_RPATH={PREFIX / "lib" if sys.platform == "darwin" else "$ORIGIN;$ORIGIN/../lib"}') cmake('bridge', source=HERE / 'native') cache = (BUILD / 'cmake-bridge/CMakeCache.txt').read_text().splitlines() compiler = next(line.split('=', 1)[1] for line in cache diff --git a/Linux/README.md b/Linux/README.md new file mode 100644 index 0000000..ed887cb --- /dev/null +++ b/Linux/README.md @@ -0,0 +1,77 @@ +# Linux installation + +InkFlow uses Fcitx5 on Linux. ARM64 Omarchy/Hyprland is working; Steam Deck, KDE Plasma and Flatpak application checks are deferred. This installs the offline Rust/Rime engine. Linux AI and voice features are outside this build. + +## Build a package + +Use a committed checkout and the [native build requirements](../Core/Portable/README.md#build-and-test), plus the Fcitx5 development files (`fcitx5` on Arch, `libfcitx5core-dev` on Ubuntu). Python 3.11 or newer, `ldd`, `pgrep` and an existing Fcitx5 installation are required on the installation target. + +```sh +bash Linux/scripts/package.sh +# Or reuse production resources prepared on this target: +bash Linux/scripts/package.sh /absolute/path/to/prepared-resources +``` + +The result is `build/linux/package/`. It contains the addon, pinned librime, compiled resources, notices, dictionary corresponding source, `install.py`, and `package.json` with file hashes and the source revision. Packaging downloads pinned inputs if they are not already cached; it does not install or change the desktop. + +Build on the target distribution and CPU architecture. The package uses the target's Fcitx5, ICU, C/C++ runtime and other system libraries; it is not a distribution-independent binary. The installer checks architecture, hashes and library dependencies before changing anything. Hashes detect damaged files, not a malicious publisher: install packages only from a trusted source. + +## Install or upgrade + +Keep another input method enabled and finish any current composition before running the installer. Installation briefly restarts an active Fcitx5 user service. + +```sh +python3 Linux/scripts/install.py install build/linux/package +# The package also carries the installer: +python3 /path/to/package/install.py install /path/to/package +``` + +No sudo or network access is used during installation. The installer recognizes `omarchy-fcitx5.service`, `fcitx5.service` and `plasma-fcitx5.service`. For another service, append `--service NAME`. If Fcitx5 is running without a recognized service, quit it first; the installer refuses to overwrite files beneath a running unmanaged daemon. When no daemon is running, start Fcitx5 after installing. + +The installer adds **InkFlow Pinyin** while preserving the existing input-method entries and default. Select it in Fcitx5, or run: + +```sh +fcitx5-remote -s inkflow-pinyin +``` + +Installing a new package with the same command upgrades the installation. Reinstalling the same package is a no-op. Releases are stored separately and a `current` symlink selects the active one; an update never overwrites a loaded library. The previous managed release stays available for rollback. + +## Roll back or uninstall + +```sh +python3 Linux/scripts/install.py status +python3 Linux/scripts/install.py rollback +python3 Linux/scripts/install.py uninstall +``` + +Rollback switches to the previous managed package without restoring an old copy of personal data. There is no previous managed release after the first install. When replacing a manual installation, the installer saves its original registration separately; `python3 Linux/scripts/install.py restore-manual` restores those descriptors and the original release pointer without replacing the current profile or personal data. + +If a write or service restart fails, the installer restores the prior descriptors, profile and release pointer automatically. If restoration itself fails, it leaves Fcitx5 stopped and retains the recovery record. Fix the reported filesystem or service error, then run `python3 Linux/scripts/install.py recover` before retrying. Recovery remembers and restarts the service that was active before an interrupted operation. + +Uninstall removes managed releases, InkFlow's Fcitx5 descriptors and its profile entries. It keeps personal dictionaries, custom phrases and settings. If InkFlow was the default input method, another remaining entry becomes the default. Files from an earlier manual CMake installation are not owned or removed by this installer. + +## Files + +Paths below use the default XDG directories; `XDG_DATA_HOME` and `XDG_CONFIG_HOME` are supported. + +| Path | Contents | +| --- | --- | +| `~/.local/share/inkflow/releases/` | Managed packages, including resources, notices and source | +| `~/.local/share/inkflow/current` | Active package symlink | +| `~/.local/share/inkflow/installation.json` | Current and previous package IDs | +| `~/.local/share/inkflow/manual-installation.json` | Prior manual registration, when one existed | +| `~/.local/share/inkflow/rime/` | Personal dictionaries; never removed by the installer | +| `~/.local/share/fcitx5/addon/inkflow.conf` | Addon descriptor pointing at the active package | +| `~/.local/share/fcitx5/inputmethod/inkflow-pinyin.conf` | Input-method entry | +| `~/.config/fcitx5/conf/inkflow.conf` | Options, custom phrases and explicit backup-import request | + +## Focused checks + +```sh +python3 -B Linux/scripts/test_install.py +bash Linux/fcitx5/test.sh +PYTHON=/path/to/python-with-dbus-next bash Linux/fcitx5/test-installed.sh +PYTHON=/path/to/python-with-dbus-next python3 -B Linux/scripts/test_package.py build/linux/package +``` + +The installer tests use temporary directories and a fake service. The real-package test installs into a temporary prefix, upgrades to a test-only package identity containing the same binaries, rolls back, and uninstalls. It loads the addon after each switch and checks that personal files survive. The installed-addon tests use a private D-Bus, disposable configuration and personal data; they do not type into desktop applications. See [the adapter README](fcitx5/README.md) for coverage and the remaining target checks. diff --git a/Linux/fcitx5/CMakeLists.txt b/Linux/fcitx5/CMakeLists.txt index cd09fd2..48ff83e 100644 --- a/Linux/fcitx5/CMakeLists.txt +++ b/Linux/fcitx5/CMakeLists.txt @@ -8,6 +8,9 @@ set(CMAKE_CXX_STANDARD_REQUIRED ON) set(CMAKE_CXX_VISIBILITY_PRESET hidden) find_package(Fcitx5Core REQUIRED) +# Package staging uses relative paths regardless of the host Fcitx installation. +set(INKFLOW_ADDONDIR "${FCITX_INSTALL_ADDONDIR}" CACHE STRING "Fcitx addon install directory") +set(INKFLOW_PKGDATADIR "${FCITX_INSTALL_PKGDATADIR}" CACHE STRING "Fcitx data install directory") # The Rust engine is built first: cargo build --release in Core/Portable (see README). get_filename_component(INKFLOW_ROOT "${CMAKE_CURRENT_SOURCE_DIR}/../.." ABSOLUTE) @@ -27,21 +30,23 @@ target_compile_options(inkflow PRIVATE -Wall -Wextra -Werror) target_link_libraries(inkflow PRIVATE Fcitx5::Core "${INKFLOW_CARGO_DIR}/libinkflow_rime.a" "${INKFLOW_NATIVE_PREFIX}/lib/librime.so" stdc++ dl pthread m) -set(INKFLOW_LIBDIR "${CMAKE_INSTALL_PREFIX}/lib/inkflow") -set_target_properties(inkflow PROPERTIES INSTALL_RPATH "${INKFLOW_LIBDIR}" BUILD_WITH_INSTALL_RPATH OFF) +set(INKFLOW_LIBDIR "lib/inkflow") +set_target_properties(inkflow PROPERTIES INSTALL_RPATH "$ORIGIN/../inkflow" BUILD_WITH_INSTALL_RPATH OFF) configure_file(inkflow.conf.in inkflow.conf @ONLY) -install(TARGETS inkflow DESTINATION "${FCITX_INSTALL_ADDONDIR}") -install(FILES "${CMAKE_CURRENT_BINARY_DIR}/inkflow.conf" DESTINATION "${FCITX_INSTALL_PKGDATADIR}/addon") -install(FILES inkflow-pinyin.conf DESTINATION "${FCITX_INSTALL_PKGDATADIR}/inputmethod") +install(TARGETS inkflow DESTINATION "${INKFLOW_ADDONDIR}") +install(FILES "${CMAKE_CURRENT_BINARY_DIR}/inkflow.conf" DESTINATION "${INKFLOW_PKGDATADIR}/addon") +install(FILES inkflow-pinyin.conf DESTINATION "${INKFLOW_PKGDATADIR}/inputmethod") file(GLOB INKFLOW_RIME_LIBS "${INKFLOW_NATIVE_PREFIX}/lib/librime.so*") install(FILES ${INKFLOW_RIME_LIBS} DESTINATION "${INKFLOW_LIBDIR}") if(INKFLOW_RESOURCES) install(DIRECTORY "${INKFLOW_RESOURCES}/shared" "${INKFLOW_RESOURCES}/prepared" - DESTINATION "${CMAKE_INSTALL_PREFIX}/share/inkflow/rime") + DESTINATION "share/inkflow/rime") endif() enable_testing() add_executable(bridge_test tests/bridge_test.cpp) target_include_directories(bridge_test PRIVATE src) +# Release builds must still execute the assert-based helper checks. +target_compile_options(bridge_test PRIVATE -UNDEBUG) add_test(NAME bridge_test COMMAND bridge_test) diff --git a/Linux/fcitx5/README.md b/Linux/fcitx5/README.md index d7fcb22..2e98c04 100644 --- a/Linux/fcitx5/README.md +++ b/Linux/fcitx5/README.md @@ -1,6 +1,6 @@ # Fcitx5 adapter -Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://github.com/nervouna/InkFlow/issues/36). Targets Omarchy/Hyprland and Steam Deck Desktop Mode (KDE Plasma); GNOME/IBus is deferred. The addon has been built, installed and activated on ARM64 Omarchy with Fcitx5 5.1.23. Application-level typing checks and Steam Deck validation remain open. +Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://github.com/nervouna/InkFlow/issues/36). Targets Omarchy/Hyprland and Steam Deck Desktop Mode (KDE Plasma); GNOME/IBus is deferred. The addon has been built, installed and activated on ARM64 Omarchy with Fcitx5 5.1.23, and the user confirmed typing works. Steam Deck validation is deferred. ## What it does @@ -12,7 +12,7 @@ Fcitx5 input-method addon over the shared Rust/Rime engine for [#36](https://git - Personal-data import (`ImportBackup=/path/to/backup.json` in that file, then reload with `fcitx5-remote -r` or apply in the configuration tool): the addon consumes the request first so a bad file cannot repeat, parses the macOS format-1 document, logs the non-portable preferences it skips, destroys every session and the engine (Rime finalizes), imports the three dictionaries with rollback on failure, writes the backup's candidate count, options and phrases into the configuration, and recreates the engine. Compositions in progress are lost, which is why the entry point is explicit. An interrupted earlier import is recovered first. - `src/bridge.h`: the pure helpers (XDG paths, preedit layout, modifier translation, bounded preceding text) with `tests/bridge_test.cpp`, which runs on any platform. -Resources are found at `INKFLOW_RESOURCES` (a prepared directory from `Core/Portable/prepare-resources.sh`, holding `shared/` and `prepared/cache/`), else the first `inkflow/rime` under `XDG_DATA_HOME` then `XDG_DATA_DIRS` that is prepared. User data lives in `$XDG_DATA_HOME/inkflow/rime` (default `~/.local/share/inkflow/rime`). Without resources the addon loads, logs a warning and passes every key through. +Resources are found at `INKFLOW_RESOURCES` (a prepared directory from `Core/Portable/prepare-resources.sh`, holding `shared/` and `prepared/cache/`), then the managed installation at `$XDG_DATA_HOME/inkflow/current/share/inkflow/rime`, then the first prepared `inkflow/rime` under `XDG_DATA_HOME` or `XDG_DATA_DIRS`. User data lives in `$XDG_DATA_HOME/inkflow/rime` (default `~/.local/share/inkflow/rime`). Without resources the addon loads, logs a warning and passes every key through. ## Build on Linux @@ -33,7 +33,11 @@ The addon links the crate's `staticlib` (which bundles the C++ bridge) and the p ## User-local installation and installed-addon test -For an installation without sudo, configure with `-DCMAKE_INSTALL_PREFIX="$HOME/.local" -DFCITX_INSTALL_USE_FCITX_SYS_PATHS=OFF`, then build and install with CMake. The generated addon descriptor names its absolute library path, so Fcitx5 can load it without a global `FCITX_ADDON_DIRS` override. Back up `~/.config/fcitx5/` before enabling InkFlow. Keep the existing keyboard and input-method entries; add `inkflow-pinyin` through Fcitx5 configuration and restart the user's Fcitx5 service. On Omarchy that service is `omarchy-fcitx5.service`. +Use `Linux/scripts/package.sh` to build a package directory with the addon, target-native resources, notices, corresponding dictionary sources, and a SHA-256 file manifest. See [Linux installation](../README.md) for packaging and install commands. + +`Linux/scripts/install.py` installs that directory without sudo or network access. It keeps immutable releases under `$XDG_DATA_HOME/inkflow/releases/` and switches the `current` symlink during upgrades. The generated addon descriptor names its absolute library path, so no global `FCITX_ADDON_DIRS` override is needed. Personal dictionaries remain in `$XDG_DATA_HOME/inkflow/rime`; configuration and custom phrases remain in `~/.config/fcitx5/conf/inkflow.conf`. + +Install, rollback and uninstall briefly stop and restart an active Fcitx5 user service. The installer recognizes `omarchy-fcitx5.service`, `fcitx5.service` and `plasma-fcitx5.service`; pass `--service NAME` for another service, or quit Fcitx5 first. It preserves other input-method entries and keeps a keyboard fallback. Ordinary builds and tests never install or activate the addon. The installed-addon test requires `dbus-run-session` and Python with `dbus-next==0.2.3`: @@ -47,4 +51,4 @@ It starts a private Fcitx5 daemon on a separate D-Bus with temporary configurati On ARM64 Arch Linux/Omarchy with Hyprland, Fcitx5 5.1.23 and GCC 16.1.1, the native build, Rust runtime tests, 291 ranking cases, production resource preparation, 21-sample learned baseline, engine/personal-data parity, C ABI consumers, addon build, helper tests and installed-addon test passed. The desktop service loaded InkFlow and reported `inkflow-pinyin` as selected; the existing US keyboard and Pinyin entries were retained. -Actual application rendering and typing still need a user check. Steam Deck/KDE Plasma, older Fcitx5 versions, Flatpak clients, distribution packaging and OS-update persistence have not been verified. Fcitx5 5.1.14 headers need C++20; older releases may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). +The user confirmed the Omarchy installation works. Steam Deck/KDE Plasma, older Fcitx5 versions, Flatpak clients and OS-update persistence have not been verified; Steam Deck work is deferred at the user's request. Fcitx5 5.1.14 headers need C++20; older releases may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). diff --git a/Linux/fcitx5/src/bridge.h b/Linux/fcitx5/src/bridge.h index 9b13019..5fc5a97 100644 --- a/Linux/fcitx5/src/bridge.h +++ b/Linux/fcitx5/src/bridge.h @@ -48,6 +48,7 @@ inline Paths resolve_paths(const Env& env, const Exists& exists) { std::vector roots; std::string override = env_or(env, "INKFLOW_RESOURCES", ""); if (!override.empty()) roots.push_back(override); + roots.push_back(data_home + "/inkflow/current/share/inkflow/rime"); roots.push_back(data_home + "/inkflow/rime"); for (const auto& dir : split(env_or(env, "XDG_DATA_DIRS", "/usr/local/share:/usr/share"), ':')) { roots.push_back(dir + "/inkflow/rime"); diff --git a/Linux/fcitx5/test-installed.sh b/Linux/fcitx5/test-installed.sh index a420fad..bbc37a3 100644 --- a/Linux/fcitx5/test-installed.sh +++ b/Linux/fcitx5/test-installed.sh @@ -3,13 +3,17 @@ set -euo pipefail cd "$(dirname "$0")/../.." prefix=${1:-$HOME/.local} +resources="$prefix/share/inkflow/rime" +if [[ -f "$prefix/share/inkflow/current/share/inkflow/rime/prepared/complete" ]]; then + resources="$prefix/share/inkflow/current/share/inkflow/rime" +fi scratch=$(mktemp -d "${TMPDIR:-/tmp}/inkflow-installed.XXXXXX") trap 'rm -rf "$scratch"' EXIT mkdir -p "$scratch/config/fcitx5" "$scratch/data" "$scratch/cache" "$scratch/runtime" chmod 700 "$scratch/runtime" printf '[Groups/0]\nName=Default\nDefault Layout=us\nDefaultIM=inkflow-pinyin\n\n[Groups/0/Items/0]\nName=keyboard-us\n\n[Groups/0/Items/1]\nName=inkflow-pinyin\n\n[GroupOrder]\n0=Default\n' > "$scratch/config/fcitx5/profile" env -u DISPLAY -u WAYLAND_DISPLAY -u FCITX_CONFIG_HOME -u FCITX_DATA_HOME \ - INKFLOW_ISOLATED_TEST=1 INKFLOW_RESOURCES="$prefix/share/inkflow/rime" \ + INKFLOW_ISOLATED_TEST=1 INKFLOW_RESOURCES="$resources" \ XDG_CONFIG_HOME="$scratch/config" XDG_DATA_HOME="$scratch/data" \ XDG_CACHE_HOME="$scratch/cache" XDG_RUNTIME_DIR="$scratch/runtime" \ FCITX_DATA_DIRS="$prefix/share/fcitx5:/usr/share/fcitx5" \ diff --git a/Linux/fcitx5/tests/bridge_test.cpp b/Linux/fcitx5/tests/bridge_test.cpp index ab0380b..b056977 100644 --- a/Linux/fcitx5/tests/bridge_test.cpp +++ b/Linux/fcitx5/tests/bridge_test.cpp @@ -26,6 +26,10 @@ int main() { assert(paths.user == "/data/inkflow/rime" && paths.shared == "/usr/share/inkflow/rime/shared"); present.insert("/opt/share/inkflow/rime/prepared/complete"); assert(resolve_paths(getenv, exists).shared == "/opt/share/inkflow/rime/shared"); + present.insert("/data/inkflow/current/share/inkflow/rime/shared"); + present.insert("/data/inkflow/current/share/inkflow/rime/prepared/complete"); + assert(resolve_paths(getenv, exists).shared == "/data/inkflow/current/share/inkflow/rime/shared"); + assert(resolve_paths(getenv, exists).user == "/data/inkflow/rime"); env["INKFLOW_RESOURCES"] = "/tmp/res"; present.insert("/tmp/res/shared"); present.insert("/tmp/res/prepared/complete"); diff --git a/Linux/scripts/install.py b/Linux/scripts/install.py new file mode 100644 index 0000000..6c43a33 --- /dev/null +++ b/Linux/scripts/install.py @@ -0,0 +1,343 @@ +#!/usr/bin/env python3 +"""Install a local InkFlow package without root or changes to personal dictionaries.""" +import argparse +import base64 +import configparser +from contextlib import contextmanager +import fcntl +import hashlib +import io +import json +import os +from pathlib import Path, PurePosixPath +import platform +import re +import shutil +import subprocess +import tempfile + + +RELEASE = re.compile(r"[0-9a-f]{12}-[0-9a-f]{12}\Z") +REQUIRED = { + "lib/fcitx5/libinkflow.so", "lib/inkflow/librime.so.1", + "share/fcitx5/inputmethod/inkflow-pinyin.conf", + "share/inkflow/rime/prepared/complete", + "share/inkflow/rime/shared/pinyin_simp.context.bin", +} +CACHE = "share/inkflow/rime/prepared/cache/" +SHARED = "share/inkflow/rime/shared/" +REQUIRED |= {CACHE + name for name in ("default.yaml", "inkflow_pinyin.schema.yaml", + "easy_en.schema.yaml", "inkflow_mixed.schema.yaml")} +REQUIRED |= {CACHE + name + suffix for name in ("pinyin_simp", "easy_en", "inkflow_mixed") + for suffix in (".table.bin", ".prism.bin", ".reverse.bin")} +REQUIRED |= {CACHE + f"inkflow_spelling_{mask}" + suffix for mask in range(32) + for suffix in (".prism.bin", ".schema.yaml")} +REQUIRED |= {SHARED + "lua/" + name + ".lua" for name in ( + "inkflow_ai_learning", "inkflow_channel", "inkflow_english", "inkflow_input_coverage", + "inkflow_mixed", "inkflow_short_conflict")} +REQUIRED |= {SHARED + "opencc/" + name for name in ( + "STCharacters.txt", "STPhrases.txt", "inkflow_emoji.json", "inkflow_s2t.json", "emoji.txt")} + + +def validate(package): + manifest_path = package / "package.json" + if manifest_path.is_symlink(): + raise ValueError("Package manifest must not be a symlink") + raw = manifest_path.read_bytes() + manifest = json.loads(raw) + if manifest.get("format") != 1 or manifest.get("architecture") != platform.machine(): + raise ValueError("Unsupported package format or CPU architecture") + if not re.fullmatch(r"[0-9a-f]{40}", manifest.get("revision", "")): + raise ValueError("Package must identify its source revision") + files = manifest.get("files", {}) + if not isinstance(files, dict) or not REQUIRED <= files.keys(): + raise ValueError("Package is missing required files") + actual = set() + for path in package.rglob("*"): + if path.is_symlink(): + raise ValueError(f"Package symlinks are not supported: {path}") + if path.is_file() and path != manifest_path: + actual.add(path.relative_to(package).as_posix()) + if actual != files.keys(): + raise ValueError("Package inventory does not match its manifest") + for name, digest in files.items(): + relative = PurePosixPath(name) + if relative.is_absolute() or ".." in relative.parts or str(relative) != name: + raise ValueError(f"Invalid package path: {name}") + if hashlib.sha256((package / name).read_bytes()).hexdigest() != digest: + raise ValueError(f"Package checksum mismatch: {name}") + identifier = manifest["revision"][:12] + "-" + hashlib.sha256(raw).hexdigest()[:12] + return identifier + + +def check_libraries(package): + result = subprocess.run(["ldd", "-r", str(package / "lib/fcitx5/libinkflow.so")], + text=True, capture_output=True) + if result.returncode or any(s in result.stdout + result.stderr for s in ("not found", "undefined symbol")): + raise RuntimeError("Fcitx5 or native dependencies are missing:\n" + result.stdout + result.stderr) + + +def atomic_write(path, data): + path.parent.mkdir(parents=True, exist_ok=True) + fd, temporary = tempfile.mkstemp(prefix=".inkflow-", dir=path.parent) + try: + with os.fdopen(fd, "wb") as output: + output.write(data) + output.flush() + os.fsync(output.fileno()) + os.replace(temporary, path) + finally: + Path(temporary).unlink(missing_ok=True) + + +def atomic_link(path, target): + temporary = path.with_name(path.name + ".new") + temporary.unlink(missing_ok=True) + temporary.symlink_to(target) + os.replace(temporary, path) + + +def edit_profile(path, enable): + config = configparser.ConfigParser(interpolation=None) + config.optionxform = str + if path.exists(): + config.read_string(path.read_text()) + groups = [s for s in config.sections() if re.fullmatch(r"Groups/[0-9]+", s)] + if not groups and enable: + groups = ["Groups/0"] + config[groups[0]] = {"Name": "Default", "Default Layout": "us", "DefaultIM": "keyboard-us"} + config["GroupOrder"] = {"0": "Default"} + for group in groups: + items = [s for s in config.sections() if re.fullmatch(re.escape(group) + r"/Items/[0-9]+", s)] + entries = [dict(config[s]) for s in sorted(items, key=lambda s: int(s.rsplit("/", 1)[1])) + if enable or config[s].get("Name") != "inkflow-pinyin"] + if not any(e.get("Name", "").startswith("keyboard-") for e in entries): + entries.insert(0, {"Name": "keyboard-us", "Layout": ""}) + if enable and not any(e.get("Name") == "inkflow-pinyin" for e in entries): + entries.append({"Name": "inkflow-pinyin", "Layout": ""}) + elif not enable and config[group].get("DefaultIM") == "inkflow-pinyin": + config[group]["DefaultIM"] = entries[0]["Name"] + for item in items: + config.remove_section(item) + for index, entry in enumerate(entries): + config[f"{group}/Items/{index}"] = entry + text = io.StringIO() + config.write(text, space_around_delimiters=False) + return text.getvalue().encode() + + +class Desktop: + def __init__(self, service=None, resume=False): + self.service = service if resume else None + candidates = [service] if service else ["omarchy-fcitx5.service", "fcitx5.service", "plasma-fcitx5.service"] + if not self.service: + for candidate in candidates: + if subprocess.run(["systemctl", "--user", "is-active", "--quiet", candidate], + stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL).returncode == 0: + self.service = candidate + break + if self.running() and not self.service: + raise RuntimeError("Quit Fcitx5 first, or pass --service NAME for its active user service") + + @staticmethod + def running(): + return subprocess.run(["pgrep", "-u", str(os.getuid()), "-x", "fcitx5"], + stdout=subprocess.DEVNULL).returncode == 0 + + def stop(self): + if self.service: + subprocess.run(["systemctl", "--user", "stop", self.service], check=False) + if self.running(): + raise RuntimeError("Fcitx5 is still running; stop it before retrying or recovering") + + def start(self): + if self.service: + subprocess.run(["systemctl", "--user", "start", self.service], check=True) + + +class Installer: + def __init__(self, data, config): + self.root = data / "inkflow" + self.releases = self.root / "releases" + self.paths = { + "current": self.root / "current", + "state": self.root / "installation.json", + "addon": data / "fcitx5/addon/inkflow.conf", + "entry": data / "fcitx5/inputmethod/inkflow-pinyin.conf", + "profile": config / "fcitx5/profile", + } + self.journal = self.root / "installation-backup.json" + self.manual = self.root / "manual-installation.json" + + @contextmanager + def locked(self): + self.root.mkdir(parents=True, exist_ok=True, mode=0o700) + with (self.root / ".installation.lock").open("w") as lock: + fcntl.flock(lock, fcntl.LOCK_EX) + yield + + def state(self): + state = json.loads(self.paths["state"].read_text()) if self.paths["state"].exists() else {} + for name in ("current", "previous"): + if state.get(name) is not None and not RELEASE.fullmatch(state[name]): + raise ValueError("Invalid installation state") + return state + + def snapshot(self): + result = {} + for name, path in self.paths.items(): + if path.is_symlink(): + result[name] = {"link": os.readlink(path)} + elif path.exists(): + result[name] = {"bytes": base64.b64encode(path.read_bytes()).decode()} + else: + result[name] = None + return result + + def restore(self, backup): + for name, value in backup.items(): + path = self.paths[name] + if value is None: + path.unlink(missing_ok=True) + elif "link" in value: + atomic_link(path, value["link"]) + else: + atomic_write(path, base64.b64decode(value["bytes"])) + + @contextmanager + def transaction(self, desktop): + if self.journal.exists(): + raise RuntimeError("An interrupted operation needs the recover command first") + backup = self.snapshot() + record = {"files": backup, "service": desktop.service} + atomic_write(self.journal, json.dumps(record).encode()) + try: + desktop.stop() + # Fcitx5 can flush its profile during shutdown. + backup = self.snapshot() + record["files"] = backup + atomic_write(self.journal, json.dumps(record).encode()) + yield + desktop.start() + self.journal.unlink() + except BaseException: + desktop.stop() + self.restore(backup) + desktop.start() + self.journal.unlink() + raise + + def recover(self, desktop): + if not self.journal.exists(): + raise RuntimeError("No interrupted operation to recover") + desktop.stop() + self.restore(json.loads(self.journal.read_text())["files"]) + desktop.start() + self.journal.unlink() + + def restore_manual(self, desktop): + if not self.manual.exists(): + raise RuntimeError("No saved manual-installation registration") + with self.transaction(desktop): + self.restore(json.loads(self.manual.read_text())) + print("Restored the prior manual registration; profile and personal data were kept.") + + def activate(self, identifier, previous): + atomic_link(self.paths["current"], "releases/" + identifier) + library = self.paths["current"] / "lib/fcitx5/libinkflow" + descriptor = ("[Addon]\nName=InkFlow\nComment=InkFlow Pinyin\nCategory=InputMethod\n" + f"Library={library}\nType=SharedLibrary\nOnDemand=True\nConfigurable=True\n") + atomic_write(self.paths["addon"], descriptor.encode()) + entry = self.releases / identifier / "share/fcitx5/inputmethod/inkflow-pinyin.conf" + atomic_write(self.paths["entry"], entry.read_bytes()) + atomic_write(self.paths["profile"], edit_profile(self.paths["profile"], True)) + atomic_write(self.paths["state"], json.dumps({"current": identifier, "previous": previous}).encode()) + + def install(self, package, desktop): + if self.journal.exists(): + raise RuntimeError("An interrupted operation needs the recover command first") + identifier = validate(package) + check_libraries(package) + state = self.state() + if state.get("current") == identifier: + validate(self.releases / identifier) + print("Already installed:", identifier) + return + self.releases.mkdir(parents=True, exist_ok=True) + target = self.releases / identifier + if not target.exists(): + staging = Path(tempfile.mkdtemp(prefix=".stage-", dir=self.releases)) + try: + shutil.copytree(package, staging, dirs_exist_ok=True) + if validate(staging) != identifier: + raise ValueError("Package changed while being copied") + staging.rename(target) + finally: + if staging.exists(): + shutil.rmtree(staging) + elif validate(target) != identifier: + raise ValueError("Installed release does not match its name") + with self.transaction(desktop): + if not state and self.paths["addon"].exists() and self.paths["entry"].exists(): + baseline = {name: value for name, value in self.snapshot().items() if name != "profile"} + atomic_write(self.manual, json.dumps(baseline).encode()) + self.activate(identifier, state.get("current")) + print("Installed:", identifier) + + def rollback(self, desktop): + state = self.state() + previous = state.get("previous") + if not previous: + raise RuntimeError("No previous managed release; use uninstall for a first installation") + package = self.releases / previous + if validate(package) != previous: + raise ValueError("Previous release is damaged") + check_libraries(package) + with self.transaction(desktop): + self.activate(previous, state["current"]) + print("Restored:", previous) + + def uninstall(self, desktop): + state = self.state() + if not state: + raise RuntimeError("No managed installation") + with self.transaction(desktop): + atomic_write(self.paths["profile"], edit_profile(self.paths["profile"], False)) + for name in ("current", "addon", "entry", "state"): + self.paths[name].unlink(missing_ok=True) + shutil.rmtree(self.releases) + print("Uninstalled. Personal dictionaries and conf/inkflow.conf were kept.") + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("command", choices=["install", "rollback", "uninstall", "recover", "restore-manual", "status"]) + parser.add_argument("package", nargs="?", type=Path) + parser.add_argument("--service", help="existing Fcitx5 user service, stopped and restarted during changes") + args = parser.parse_args() + if (args.command == "install") != (args.package is not None): + parser.error("Only install takes a package directory") + home = Path.home() + data = Path(os.environ.get("XDG_DATA_HOME", home / ".local/share")).absolute() + config = Path(os.environ.get("XDG_CONFIG_HOME", home / ".config")).absolute() + installer = Installer(data, config) + with installer.locked(): + if args.command == "status": + print(json.dumps(installer.state(), indent=2)) + else: + service = args.service + if args.command == "recover" and installer.journal.exists(): + service = json.loads(installer.journal.read_text())["service"] or service + desktop = Desktop(service, resume=args.command == "recover") + if args.command == "install": + installer.install(args.package.resolve(), desktop) + else: + getattr(installer, args.command.replace("-", "_"))(desktop) + + +if __name__ == "__main__": + try: + main() + except (OSError, ValueError, RuntimeError, subprocess.CalledProcessError) as error: + raise SystemExit(f"{error}\nIf installation-backup.json remains, fix the error and run recover; " + "Fcitx5 may be stopped until recovery succeeds.") diff --git a/Linux/scripts/package-notices.py b/Linux/scripts/package-notices.py new file mode 100644 index 0000000..9eec701 --- /dev/null +++ b/Linux/scripts/package-notices.py @@ -0,0 +1,214 @@ +#!/usr/bin/env python3 +"""Complete a Linux staging tree with notices, corresponding source and hashes.""" +import hashlib +import json +from pathlib import Path +import platform +import re +import shutil +import subprocess +import sys + +ROOT = Path(__file__).resolve().parents[2] +BUILD = ROOT / 'build/portable' + + +def command(*args): + return subprocess.check_output(args, cwd=ROOT, text=True).strip() + + +def digest(path): + with path.open('rb') as stream: + return hashlib.file_digest(stream, 'sha256').hexdigest() + + +def copy(source, destination, expected=None): + if expected and digest(source) != expected: + raise SystemExit(f'Checksum mismatch: {source}') + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copyfile(source, destination) + + +def hashes(directory): + return {p.relative_to(directory).as_posix(): digest(p) + for p in sorted(directory.rglob('*')) if p.is_file()} + + +def check_resources(resources, reused): + if not (resources / 'prepared/complete').is_file(): + raise SystemExit('Missing target-native resource completion marker') + native = json.loads((BUILD / 'native-build.json').read_text()) + shared = hashes(resources / 'shared') + cache = hashes(resources / 'prepared/cache') + manifest = json.loads((resources / 'shared/dictionary-manifest.json').read_text()) + if reused: + report_path = resources / 'resources.json' + if not report_path.is_file(): + raise SystemExit('Reused resources require resources.json; run package.sh without arguments to rebuild') + report = json.loads(report_path.read_text()) + expected = hashes(ROOT / 'build/linux/package-shared') + old_native = report.get('nativeBuild', {}) + if (report.get('platform') != 'Linux' or + report.get('architecture') != platform.machine() or + any(old_native.get(key) != native.get(key) + for key in ('platform', 'machine', 'sources', 'revision')) or + report.get('sourceSHA256') != shared or + report.get('cacheSHA256') != cache or + report.get('manifest') != manifest or shared != expected): + raise SystemExit('Reused resources do not match this Linux target/revision/recipe; rebuild without arguments') + else: + report = {'platform': 'Linux', 'architecture': platform.machine(), + 'nativeBuild': native, 'sourceSHA256': shared, + 'cacheSHA256': cache, 'manifest': manifest} + (resources / 'resources.json').write_text(json.dumps(report, indent=2) + '\n') + + +def main(): + package = Path(sys.argv[1]).resolve() + if command('git', 'status', '--porcelain', '--untracked-files=no'): + raise SystemExit('Commit tracked changes before packaging') + revision = command('git', 'rev-parse', 'HEAD') + for name in ('lib/fcitx5/libinkflow.so', 'lib/inkflow/librime.so', + 'lib/inkflow/librime.so.1', + 'share/inkflow/rime/prepared/complete', + 'share/inkflow/rime/shared/pinyin_simp.context.bin', + 'share/fcitx5/addon/inkflow.conf', + 'share/fcitx5/inputmethod/inkflow-pinyin.conf', + 'share/inkflow/rime/shared/dictionary-manifest.json'): + if not (package / name).is_file(): + raise SystemExit(f'Missing staged payload: {name}') + copy(ROOT / 'Linux/scripts/install.py', package / 'install.py') + notices = package / 'share/inkflow/Licenses' + source = package / 'share/inkflow/SOURCE' + source.mkdir(parents=True) + notices.mkdir(parents=True) + # Full committed project: recipe, modifications, data snapshots and licenses. + subprocess.run(['git', 'archive', '--format=tar', '-o', str(source / 'inkflow.tar'), + revision], cwd=ROOT, check=True) + for name in ('LICENSE', 'NOTICE'): + copy(ROOT / name, notices / ('InkFlow-' + name)) + for name in ('chinese-dictionaries-NOTICE.txt', 'easy-en-GPL-3.0.txt', + 'easy-en-LGPL-3.0.txt', 'pinyin-simp.txt', 'rime-frost.txt', + 'rime-ice.txt', 'technology-english-NOTICE.txt', 'wordfreq.txt', 'opencc.txt'): + copy(ROOT / 'macOS/Licenses' / name, notices / 'dictionary' / name) + + lock = json.loads((ROOT / 'Core/Portable/native-sources.lock.json').read_text()) + for name, item in lock.items(): + archive = name + ('.tar.bz2' if name == 'boost' else '.tar.gz') + copy(BUILD / archive, source / 'native' / archive, item['sha256']) + native = BUILD / 'sources' + required = ['rime/LICENSE', 'rime/include/COPYING.darts-clone', 'rime/include/utf8.h', + 'rime-lua/LICENSE', 'rime-lua/src/lib/lauxlib-compat.c', + 'lua/lua5.4/lua.h', 'boost/LICENSE_1_0.txt', 'glog/COPYING', + 'leveldb/LICENSE', 'marisa/COPYING.md', 'yaml/LICENSE', 'opencc/LICENSE'] + utf8 = 'rime-lua/src/lib/lutf8lib-compat.c' + if not (native / utf8).is_file(): + utf8 = 'rime-lua/src/lib/lutf8lib.c' + required.append(utf8) + for name in required: + copy(native / name, notices / 'native' / name) + # Preserve embedded header notices too (RapidJSON, UTF8-CPP, Darts-clone). + for directory in ('rime/include', 'opencc/deps'): + for path in sorted((native / directory).rglob('*')): + if path.is_file() and (path.suffix in ('.h', '.hpp') or + path.name.lower().startswith(('license', 'copying', 'notice'))): + copy(path, notices / 'native' / path.relative_to(native)) + + recipe = (ROOT / 'Core/scripts/resource-dependencies.sh').read_text() + inputs = re.findall(r'^fetch (\S+) ([0-9a-f]{64}) (https:\S+)$', recipe, re.M) + if len(inputs) != 3: + raise SystemExit('Review changed dictionary download recipe') + for name, sha, _ in inputs: + copy(ROOT / 'build/deps' / name, source / 'deps' / name, sha) + catalog = json.loads((ROOT / 'Core/config/chinese-sources.json').read_text()) + for item in catalog: + name = item['id'] + '.yaml' + copy(ROOT / 'build/dictionary-sources' / name, + source / 'dictionary-sources' / name, item['pinnedSHA256']) + + # Both locked Rust graphs are used: engine and dictionary generator. + for crate in ('Core/Portable', 'Core/Portable/dictionary'): + metadata = json.loads(command('cargo', 'metadata', '--locked', '--offline', + '--format-version', '1', '--manifest-path', + str(ROOT / crate / 'Cargo.toml'))) + for entry in metadata['packages']: + if entry['source'] is None: + continue # Local source and its license are in inkflow.tar. + directory = Path(entry['manifest_path']).parent + key = entry['name'] + '-' + entry['version'] + destination = source / 'rust' / key + if not destination.exists(): + shutil.copytree(directory, destination, ignore=shutil.ignore_patterns('.git')) + files = [p for p in directory.rglob('*') if p.is_file() and + p.name.lower().startswith(('license', 'copying', 'copyright', 'notice'))] + if not files: + raise SystemExit(f'Missing Rust notices: {key}') + for path in files: + copy(path, notices / 'Rust' / key / path.relative_to(directory)) + standard = Path(command('rustc', '--print', 'sysroot')) / 'share/doc/rust' + copy(standard / 'COPYRIGHT-library.html', notices / 'Rust/COPYRIGHT-library.html') + shutil.copytree(standard / 'licenses', notices / 'Rust/standard-library-licenses') + (notices / 'Rust/toolchain.txt').write_text(command('rustc', '--version', '--verbose') + '\n') + copy(BUILD / 'native-build.json', source / 'native-build.json') + (source / 'REBUILD.txt').write_text(f'''InkFlow revision {revision} +Build scripts require Git metadata. Start with a clean checkout at this revision: + git clone https://github.com/nervouna/InkFlow.git inkflow-rebuild + cd inkflow-rebuild + git checkout --detach {revision} +Alternatively, to use only the bundled project source, extract inkflow.tar into +an empty directory, enter it, and create a local provenance commit: + git init + git add . + git -c user.name='Source rebuild' -c user.email='rebuild@localhost' commit -m 'Rebuild bundled InkFlow source from {revision}' +This local commit has a different revision from the original above; rebuilt +metadata identifies that local commit, not the original release revision. +From either checkout, restore the bundled inputs (replace /path/to/SOURCE): + mkdir -p build/portable build/deps build/dictionary-sources + cp /path/to/SOURCE/native/* build/portable/ + cp /path/to/SOURCE/deps/* build/deps/ + cp /path/to/SOURCE/dictionary-sources/* build/dictionary-sources/ + python3 Core/Portable/build-native.py + bash Core/scripts/resource-dependencies.sh + bash Core/scripts/prepare-chinese.sh --sources-only + bash Core/scripts/prepare-rime.sh build/linux/resources/shared + CARGO_TARGET_DIR="$PWD/build/portable/cargo" cargo run --locked --release --manifest-path Core/Portable/Cargo.toml --bin prepare-resources -- build/linux/resources/shared build/linux/resources/prepared +To build the full package instead, after restoring the inputs above run: + bash Linux/scripts/package.sh +The packaging recipe requires clean tracked Git source and records its HEAD. +The source archive itself has no .git directory, hence the initialization step +when rebuilding from the archive rather than the original checkout. +Native archives are checksum pinned. OpenCC's C++17/cstdint adjustments are in +build-native.py. Rust dependency sources are in rust/; Cargo.lock records origins +and checksums. Cargo/toolchain and system build prerequisites may require network +access; this is source delivery, not an offline build environment. +Externally supplied prepared resources require a resources.json report matching +this Linux architecture, native pins and revision. Shared resource bytes must match +the current recipe; cache hashes must match the report. The report is a build +provenance record, not independent proof of how compiled cache bytes were produced. +''') + dependencies = ('Requires compatible host Linux libc, libstdc++, libgcc and Fcitx5 ' + '(including their transitive system dependencies). These are not bundled. ' + 'Architecture alone does not imply ABI compatibility; build on the oldest ' + 'supported target and check ELF dependencies on each target.') + (package / 'share/inkflow/SYSTEM-DEPENDENCIES.txt').write_text(dependencies + '\n') + # Dereference installed library aliases so every delivered byte is hashed. + for path in sorted(package.rglob('*')): + if path.is_symlink(): + if not path.is_file(): + raise SystemExit(f'Unsupported directory or broken symlink: {path}') + content = path.read_bytes() + path.unlink() + path.write_bytes(content) + manifest = {'format': 1, 'architecture': platform.machine(), 'revision': revision, + 'dirty': False, 'systemDependencies': dependencies, + 'files': {p.relative_to(package).as_posix(): digest(p) + for p in sorted(package.rglob('*')) + if p.is_file() and p != package / 'package.json'}} + (package / 'package.json').write_text(json.dumps(manifest, indent=2) + '\n') + + +if __name__ == '__main__': + if len(sys.argv) > 1 and sys.argv[1] == '--check-resources': + check_resources(Path(sys.argv[2]).resolve(), sys.argv[3] == 'reused') + else: + main() diff --git a/Linux/scripts/package.sh b/Linux/scripts/package.sh new file mode 100755 index 0000000..fca9f5d --- /dev/null +++ b/Linux/scripts/package.sh @@ -0,0 +1,46 @@ +#!/bin/bash +# Build a directory package only. Commit tracked changes before packaging. +# Usage: bash Linux/scripts/package.sh [PREPARED_RESOURCES] +set -euo pipefail +cd "$(dirname "$0")/../.." +[[ $(uname -s) == Linux ]] || { echo 'Linux host required.' >&2; exit 1; } +[[ $# -le 1 ]] || { echo 'Usage: package.sh [PREPARED_RESOURCES]' >&2; exit 1; } +[[ -z $(git status --porcelain --untracked-files=no) ]] || { + echo 'Commit tracked source changes before packaging for exact provenance.' >&2; exit 1; +} +export CARGO_TARGET_DIR="$PWD/build/portable/cargo" +python3 Core/Portable/build-native.py +bash Core/scripts/resource-dependencies.sh +bash Core/scripts/prepare-chinese.sh --sources-only +resources=${1:-$PWD/build/linux/resources} +resources=$(realpath -m "$resources") +if [[ $# == 0 ]]; then + bash Core/scripts/prepare-rime.sh "$resources/shared" + cargo run --locked --release --manifest-path Core/Portable/Cargo.toml --bin prepare-resources -- \ + "$resources/shared" "$resources/prepared" +else + # Compare reused resources against today's committed dictionary recipe too. + bash Core/scripts/prepare-rime.sh build/linux/package-shared +fi +python3 Linux/scripts/package-notices.py --check-resources "$resources" "${1:+reused}" +[[ -d "$resources/shared" && -d "$resources/prepared/cache" ]] || { + echo 'Resources must contain shared/ and target-native prepared/cache/.' >&2; exit 1; +} +cargo build --locked --release --manifest-path Core/Portable/Cargo.toml --lib +cmake -S Linux/fcitx5 -B build/linux/cmake -G Ninja \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr \ + -DINKFLOW_ADDONDIR=lib/fcitx5 -DINKFLOW_PKGDATADIR=share/fcitx5 \ + -DINKFLOW_RESOURCES="$resources" +cmake --build build/linux/cmake --target inkflow --parallel "${CMAKE_BUILD_PARALLEL_LEVEL:-4}" +# A failed build never replaces a previously completed package. +mkdir -p build/linux +stage=$(mktemp -d "$PWD/build/linux/stage.XXXXXX") +trap 'rm -rf "$stage"' EXIT +DESTDIR="$stage" cmake --install build/linux/cmake +cp "$resources/resources.json" "$stage/usr/share/inkflow/rime/resources.json" +# The addon loader resolves a bare library name in its configured addon paths. +sed -i 's|^Library=.*|Library=libinkflow|' "$stage/usr/share/fcitx5/addon/inkflow.conf" +python3 Linux/scripts/package-notices.py "$stage/usr" +rm -rf build/linux/package +mv "$stage/usr" build/linux/package +printf 'Package directory: %s/build/linux/package\n' "$PWD" diff --git a/Linux/scripts/test_install.py b/Linux/scripts/test_install.py new file mode 100644 index 0000000..c972af1 --- /dev/null +++ b/Linux/scripts/test_install.py @@ -0,0 +1,302 @@ +#!/usr/bin/env python3 +"""Installer tests use temporary XDG directories and never contact a desktop.""" +import configparser +import hashlib +import importlib.util +import json +from pathlib import Path +import platform +import tempfile +import unittest +from unittest.mock import patch + +spec = importlib.util.spec_from_file_location("installer", Path(__file__).with_name("install.py")) +installer = importlib.util.module_from_spec(spec) +spec.loader.exec_module(installer) + + +class Desktop: + def __init__(self): + self.stops = self.starts = 0 + self.fail_start = False + self.service = "fixture-fcitx5.service" + + def stop(self): + self.stops += 1 + + def start(self): + self.starts += 1 + if self.fail_start: + self.fail_start = False + raise RuntimeError("fixture service failed to start") + + +class InstallTests(unittest.TestCase): + def setUp(self): + self.scratch = tempfile.TemporaryDirectory() + self.addCleanup(self.scratch.cleanup) + self.root = Path(self.scratch.name) + self.manager = installer.Installer(self.root / "data", self.root / "config") + self.manager.root.mkdir(parents=True) + self.desktop = Desktop() + self.libraries = patch.object(installer, "check_libraries") + self.libraries.start() + self.addCleanup(self.libraries.stop) + profile = self.manager.paths["profile"] + profile.parent.mkdir(parents=True) + profile.write_text("[Groups/0]\nName=Default\nDefault Layout=us\nDefaultIM=pinyin\n" + "[Groups/0/Items/0]\nName=keyboard-us\n" + "[Groups/0/Items/1]\nName=pinyin\n" + "[GroupOrder]\n0=Default\n") + self.learning = self.manager.root / "rime/pinyin_simp.userdb/fixture" + self.learning.parent.mkdir(parents=True) + self.learning.write_bytes(b"personal-learning-must-survive") + self.settings = profile.parent / "conf/inkflow.conf" + self.settings.parent.mkdir() + self.settings.write_text("CandidateCount=7\n[CustomPhrases]\n0=dz=private-fixture\n") + + def package(self, revision): + package = self.root / revision + package.mkdir() + for name in installer.REQUIRED: + path = package / name + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(revision + "\n") + (package / "share/fcitx5/inputmethod/inkflow-pinyin.conf").write_text( + "[InputMethod]\nName=InkFlow Pinyin\nAddon=inkflow\n") + self.manifest(package, revision) + return package + + def manifest(self, package, revision): + manifest = {"format": 1, "architecture": platform.machine(), "revision": revision * 40, + "files": {p.relative_to(package).as_posix(): hashlib.sha256(p.read_bytes()).hexdigest() + for p in package.rglob("*") if p.is_file() and p.name != "package.json"}} + (package / "package.json").write_text(json.dumps(manifest)) + + def profile(self): + profile = configparser.ConfigParser(interpolation=None) + profile.read(self.manager.paths["profile"]) + return profile + + def assert_personal_data(self): + self.assertEqual(self.learning.read_bytes(), b"personal-learning-must-survive") + self.assertIn("private-fixture", self.settings.read_text()) + + def test_install_upgrade_rollback_and_uninstall(self): + first, second = self.package("a"), self.package("b") + first_id, second_id = installer.validate(first), installer.validate(second) + self.manager.install(first, self.desktop) + self.assertEqual(self.manager.state(), {"current": first_id, "previous": None}) + self.assertEqual(self.profile()["Groups/0"]["DefaultIM"], "pinyin") + self.assertEqual(self.profile()["Groups/0/Items/2"]["Name"], "inkflow-pinyin") + self.manager.install(second, self.desktop) + self.assertEqual(self.manager.state(), {"current": second_id, "previous": first_id}) + self.manager.rollback(self.desktop) + self.assertEqual(self.manager.paths["current"].resolve(), self.manager.releases / first_id) + self.assertEqual(self.manager.state(), {"current": first_id, "previous": second_id}) + self.manager.install(second, self.desktop) + self.manager.uninstall(self.desktop) + self.assertFalse(self.manager.paths["current"].exists()) + self.assertFalse(self.manager.paths["addon"].exists()) + self.assertFalse(self.manager.releases.exists()) + self.assertEqual(self.profile()["Groups/0/Items/1"]["Name"], "pinyin") + self.assertNotIn("inkflow-pinyin", self.manager.paths["profile"].read_text()) + self.assert_personal_data() + self.assertEqual(self.desktop.stops, self.desktop.starts) + + def test_same_package_is_idempotent(self): + package = self.package("a") + self.manager.install(package, self.desktop) + before = self.manager.snapshot() + self.manager.install(package, self.desktop) + self.assertEqual(before, self.manager.snapshot()) + self.assertEqual(self.desktop.stops, 1) + + def test_modified_package_and_architecture_are_rejected_before_stopping(self): + package = self.package("a") + (package / "lib/fcitx5/libinkflow.so").write_text("corrupt") + with self.assertRaises(ValueError): + self.manager.install(package, self.desktop) + self.manifest(package, "a") + manifest = json.loads((package / "package.json").read_text()) + manifest["architecture"] = "other-cpu" + (package / "package.json").write_text(json.dumps(manifest)) + with self.assertRaises(ValueError): + self.manager.install(package, self.desktop) + self.assertEqual(self.desktop.stops, 0) + self.assert_personal_data() + + def test_symlinks_and_extra_files_are_rejected(self): + package = self.package("a") + (package / "extra").write_text("unlisted") + with self.assertRaises(ValueError): + installer.validate(package) + (package / "extra").unlink() + (package / "escape").symlink_to(self.learning.parent, target_is_directory=True) + with self.assertRaises(ValueError): + installer.validate(package) + + def test_write_failure_restores_old_installation(self): + self.manager.install(self.package("a"), self.desktop) + before = self.manager.snapshot() + write = installer.atomic_write + failed = False + + def fail_once(path, data): + nonlocal failed + if path == self.manager.paths["entry"] and not failed: + failed = True + raise OSError("fixture full disk") + write(path, data) + + with patch.object(installer, "atomic_write", side_effect=fail_once): + with self.assertRaises(OSError): + self.manager.install(self.package("b"), self.desktop) + self.assertEqual(self.manager.snapshot(), before) + self.assertFalse(self.manager.journal.exists()) + self.assert_personal_data() + + def test_service_start_failure_restores_old_installation(self): + self.manager.install(self.package("a"), self.desktop) + before = self.manager.snapshot() + self.desktop.fail_start = True + with self.assertRaises(RuntimeError): + self.manager.install(self.package("b"), self.desktop) + self.assertEqual(self.manager.snapshot(), before) + self.assertFalse(self.manager.journal.exists()) + + def test_interrupted_install_recovers_before_retry(self): + package = self.package("a") + self.manager.install(package, self.desktop) + before = self.manager.snapshot() + self.manager.journal.write_text(json.dumps({"files": before, "service": self.desktop.service})) + self.manager.paths["addon"].write_text("interrupted") + with self.assertRaises(RuntimeError): + self.manager.install(package, self.desktop) + self.manager.recover(self.desktop) + self.assertEqual(self.manager.snapshot(), before) + self.assertFalse(self.manager.journal.exists()) + + def test_manual_registration_can_be_restored_without_replacing_later_profile_changes(self): + for name, content in (("addon", b"original custom addon"), ("entry", b"original custom entry")): + self.manager.paths[name].parent.mkdir(parents=True, exist_ok=True) + self.manager.paths[name].write_bytes(content) + self.manager.install(self.package("a"), self.desktop) + self.manager.install(self.package("b"), self.desktop) + profile = self.manager.paths["profile"] + with profile.open("a") as output: + output.write("[Groups/0/Items/3]\nName=other-ime\n") + self.manager.restore_manual(self.desktop) + self.assertEqual(self.manager.paths["addon"].read_bytes(), b"original custom addon") + self.assertEqual(self.manager.paths["entry"].read_bytes(), b"original custom entry") + self.assertIn("Name=other-ime", profile.read_text()) + self.assertFalse(self.manager.paths["current"].exists()) + self.assertEqual(self.manager.state(), {}) + self.assert_personal_data() + + def test_persistent_restore_failure_keeps_journal_and_service_stopped(self): + self.manager.install(self.package("a"), self.desktop) + starts = self.desktop.starts + write = installer.atomic_write + + def persistent_failure(path, data): + if path == self.manager.paths["addon"]: + raise OSError("fixture persistent I/O failure") + write(path, data) + + with patch.object(installer, "atomic_write", side_effect=persistent_failure): + with self.assertRaises(OSError): + self.manager.install(self.package("b"), self.desktop) + self.assertTrue(self.manager.journal.exists()) + self.assertEqual(self.desktop.starts, starts) + with self.assertRaises(OSError): + self.manager.recover(self.desktop) + self.assertEqual(self.desktop.starts, starts) + self.manager.recover(self.desktop) + self.assertFalse(self.manager.journal.exists()) + self.assertEqual(self.desktop.starts, starts + 1) + + def test_journal_records_service_before_shutdown(self): + original = self.desktop.stop + + def inspect_stop(): + record = json.loads(self.manager.journal.read_text()) + self.assertEqual(record["service"], "fixture-fcitx5.service") + self.assertIn("files", record) + original() + + with patch.object(self.desktop, "stop", side_effect=inspect_stop): + self.manager.install(self.package("a"), self.desktop) + + def test_missing_compiled_resource_is_rejected_even_with_fresh_hashes(self): + package = self.package("a") + (package / (installer.CACHE + "inkflow_spelling_31.prism.bin")).unlink() + self.manifest(package, "a") + with self.assertRaises(ValueError): + self.manager.install(package, self.desktop) + self.assertEqual(self.desktop.stops, 0) + + def test_rollback_rejects_damaged_previous_release(self): + self.manager.install(self.package("a"), self.desktop) + previous = self.manager.paths["current"].resolve() + self.manager.install(self.package("b"), self.desktop) + before = self.manager.snapshot() + (previous / "lib/fcitx5/libinkflow.so").write_text("damaged") + with self.assertRaises(ValueError): + self.manager.rollback(self.desktop) + self.assertEqual(before, self.manager.snapshot()) + + def test_uninstall_selects_keyboard_if_inkflow_was_default(self): + self.manager.install(self.package("a"), self.desktop) + profile = self.manager.paths["profile"] + profile.write_text(profile.read_text().replace("DefaultIM=pinyin", "DefaultIM=inkflow-pinyin")) + self.manager.uninstall(self.desktop) + self.assertEqual(self.profile()["Groups/0"]["DefaultIM"], "keyboard-us") + self.assert_personal_data() + + def test_second_group_and_later_user_entries_survive(self): + profile = self.manager.paths["profile"] + with profile.open("a") as stream: + stream.write("[Groups/1]\nName=Other\nDefault Layout=de\nDefaultIM=keyboard-de\n" + "[Groups/1/Items/0]\nName=keyboard-de\n") + self.manager.install(self.package("a"), self.desktop) + with profile.open("a") as stream: + stream.write("[Groups/0/Items/3]\nName=other-ime\n") + self.manager.install(self.package("b"), self.desktop) + self.manager.rollback(self.desktop) + self.assertIn("Name=other-ime", profile.read_text()) + self.manager.uninstall(self.desktop) + self.assertIn("Name=other-ime", profile.read_text()) + self.assertEqual(self.profile()["Groups/1/Items/0"]["Name"], "keyboard-de") + self.assertEqual(self.profile()["Groups/1"]["DefaultIM"], "keyboard-de") + + def test_fresh_profile_has_keyboard_fallback(self): + self.manager.paths["profile"].unlink() + self.manager.install(self.package("a"), self.desktop) + self.assertEqual(self.profile()["Groups/0/Items/0"]["Name"], "keyboard-us") + self.assertEqual(self.profile()["Groups/0/Items/1"]["Name"], "inkflow-pinyin") + + +class DesktopTests(unittest.TestCase): + def test_recovery_remembers_inactive_service(self): + with patch.object(installer.Desktop, "running", return_value=False), \ + patch.object(installer.subprocess, "run") as run: + desktop = installer.Desktop("omarchy-fcitx5.service", resume=True) + desktop.start() + run.assert_called_once_with( + ["systemctl", "--user", "start", "omarchy-fcitx5.service"], check=True) + + def test_nonzero_stop_is_safe_only_when_process_is_gone(self): + with patch.object(installer.Desktop, "running", return_value=False), \ + patch.object(installer.subprocess, "run") as run: + run.return_value.returncode = 1 + desktop = installer.Desktop("omarchy-fcitx5.service", resume=True) + desktop.stop() + with patch.object(installer.Desktop, "running", return_value=True): + with self.assertRaises(RuntimeError): + desktop.stop() + self.assertFalse(any("start" in call.args[0] for call in run.call_args_list)) + + +if __name__ == "__main__": + unittest.main() diff --git a/Linux/scripts/test_package.py b/Linux/scripts/test_package.py new file mode 100644 index 0000000..bb080fe --- /dev/null +++ b/Linux/scripts/test_package.py @@ -0,0 +1,77 @@ +#!/usr/bin/env python3 +"""Exercise a real package's install/upgrade/rollback/uninstall in temporary directories.""" +import hashlib +import importlib.util +import json +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile + +spec = importlib.util.spec_from_file_location("installer", Path(__file__).with_name("install.py")) +installer = importlib.util.module_from_spec(spec) +spec.loader.exec_module(installer) + + +class NoDesktop: + # Each addon check starts its own private daemon; never stop the user's service. + service = None + + def stop(self): + pass + + def start(self): + pass + + +def main(): + package = Path(sys.argv[1]).resolve() + root = Path(__file__).resolve().parents[2] + identifier = installer.validate(package) + with tempfile.TemporaryDirectory(prefix="inkflow-package-") as temporary: + scratch = Path(temporary) + prefix = scratch / "prefix" + manager = installer.Installer(prefix / "share", prefix / "config") + desktop = NoDesktop() + with manager.locked(): + personal = manager.root / "rime/installer-preservation-fixture" + personal.parent.mkdir() + personal.write_bytes(b"personal-data-fixture") + settings = prefix / "config/fcitx5/conf/inkflow.conf" + settings.parent.mkdir(parents=True) + settings.write_text("CandidateCount=7\n") + + def check(): + subprocess.run(["bash", "Linux/fcitx5/test-installed.sh", str(prefix)], cwd=root, check=True) + assert personal.read_bytes() == b"personal-data-fixture" + assert settings.read_text() == "CandidateCount=7\n" + + manager.install(package, desktop) + check() + # A distinct test-only package exercises replacement using the same target binaries. + newer = scratch / "upgrade-fixture" + shutil.copytree(package, newer) + marker = newer / "upgrade-fixture.txt" + marker.write_text("Test-only package identity for the upgrade exercise.\n") + manifest = json.loads((newer / "package.json").read_text()) + manifest["files"][marker.name] = hashlib.sha256(marker.read_bytes()).hexdigest() + (newer / "package.json").write_text(json.dumps(manifest, sort_keys=True)) + manager.install(newer, desktop) + assert manager.state()["previous"] == identifier + check() + manager.rollback(desktop) + assert manager.state()["current"] == identifier + check() + manager.uninstall(desktop) + assert not manager.paths["addon"].exists() + assert not manager.paths["entry"].exists() + assert not manager.paths["current"].exists() + assert not manager.releases.exists() + assert personal.read_bytes() == b"personal-data-fixture" + assert settings.read_text() == "CandidateCount=7\n" + print("PASS real package install, upgrade-fixture, rollback and uninstall; personal files preserved") + + +if __name__ == "__main__": + main() From ce3d2e228e5eec7d70b7ce2d32f9e9e54a043348 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 20:02:09 +0800 Subject: [PATCH 18/21] fix(linux): preserve recovery state across interrupted shutdowns Record shutdown separately from file changes, retain partial manual registrations, and fsync installer updates before clearing recovery information. Reject packages with incomplete runtime resources and exercise persistent restoration failures. Package the legacy dictionary from its pinned archive extraction and reject untracked source inputs. Refs #36. --- Linux/scripts/install.py | 51 +++++++++++++++++++++++++------- Linux/scripts/package-notices.py | 8 +++-- Linux/scripts/package.sh | 2 +- Linux/scripts/test_install.py | 33 +++++++++++++++++++++ 4 files changed, 80 insertions(+), 14 deletions(-) diff --git a/Linux/scripts/install.py b/Linux/scripts/install.py index 6c43a33..f3a72a3 100644 --- a/Linux/scripts/install.py +++ b/Linux/scripts/install.py @@ -77,6 +77,20 @@ def check_libraries(package): raise RuntimeError("Fcitx5 or native dependencies are missing:\n" + result.stdout + result.stderr) +def sync_directory(path): + fd = os.open(path, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(fd) + finally: + os.close(fd) + + +def durable_unlink(path): + path.unlink(missing_ok=True) + if path.parent.exists(): + sync_directory(path.parent) + + def atomic_write(path, data): path.parent.mkdir(parents=True, exist_ok=True) fd, temporary = tempfile.mkstemp(prefix=".inkflow-", dir=path.parent) @@ -86,6 +100,7 @@ def atomic_write(path, data): output.flush() os.fsync(output.fileno()) os.replace(temporary, path) + sync_directory(path.parent) finally: Path(temporary).unlink(missing_ok=True) @@ -95,6 +110,7 @@ def atomic_link(path, target): temporary.unlink(missing_ok=True) temporary.symlink_to(target) os.replace(temporary, path) + sync_directory(path.parent) def edit_profile(path, enable): @@ -172,6 +188,7 @@ def __init__(self, data, config): @contextmanager def locked(self): self.root.mkdir(parents=True, exist_ok=True, mode=0o700) + sync_directory(self.root.parent) with (self.root / ".installation.lock").open("w") as lock: fcntl.flock(lock, fcntl.LOCK_EX) yield @@ -198,7 +215,7 @@ def restore(self, backup): for name, value in backup.items(): path = self.paths[name] if value is None: - path.unlink(missing_ok=True) + durable_unlink(path) elif "link" in value: atomic_link(path, value["link"]) else: @@ -208,8 +225,8 @@ def restore(self, backup): def transaction(self, desktop): if self.journal.exists(): raise RuntimeError("An interrupted operation needs the recover command first") - backup = self.snapshot() - record = {"files": backup, "service": desktop.service} + backup = None + record = {"files": None, "service": desktop.service} atomic_write(self.journal, json.dumps(record).encode()) try: desktop.stop() @@ -219,21 +236,24 @@ def transaction(self, desktop): atomic_write(self.journal, json.dumps(record).encode()) yield desktop.start() - self.journal.unlink() + durable_unlink(self.journal) except BaseException: desktop.stop() - self.restore(backup) + if backup is not None: + self.restore(backup) desktop.start() - self.journal.unlink() + durable_unlink(self.journal) raise def recover(self, desktop): if not self.journal.exists(): raise RuntimeError("No interrupted operation to recover") desktop.stop() - self.restore(json.loads(self.journal.read_text())["files"]) + backup = json.loads(self.journal.read_text())["files"] + if backup is not None: + self.restore(backup) desktop.start() - self.journal.unlink() + durable_unlink(self.journal) def restore_manual(self, desktop): if not self.manual.exists(): @@ -264,6 +284,7 @@ def install(self, package, desktop): print("Already installed:", identifier) return self.releases.mkdir(parents=True, exist_ok=True) + sync_directory(self.root) target = self.releases / identifier if not target.exists(): staging = Path(tempfile.mkdtemp(prefix=".stage-", dir=self.releases)) @@ -271,14 +292,23 @@ def install(self, package, desktop): shutil.copytree(package, staging, dirs_exist_ok=True) if validate(staging) != identifier: raise ValueError("Package changed while being copied") + # Persist the payload before a durable current link can refer to it. + for path in staging.rglob("*"): + if path.is_file(): + with path.open("rb") as source: + os.fsync(source.fileno()) + elif path.is_dir(): + sync_directory(path) + sync_directory(staging) staging.rename(target) + sync_directory(self.releases) finally: if staging.exists(): shutil.rmtree(staging) elif validate(target) != identifier: raise ValueError("Installed release does not match its name") with self.transaction(desktop): - if not state and self.paths["addon"].exists() and self.paths["entry"].exists(): + if not state and (self.paths["addon"].exists() or self.paths["entry"].exists()): baseline = {name: value for name, value in self.snapshot().items() if name != "profile"} atomic_write(self.manual, json.dumps(baseline).encode()) self.activate(identifier, state.get("current")) @@ -304,8 +334,9 @@ def uninstall(self, desktop): with self.transaction(desktop): atomic_write(self.paths["profile"], edit_profile(self.paths["profile"], False)) for name in ("current", "addon", "entry", "state"): - self.paths[name].unlink(missing_ok=True) + durable_unlink(self.paths[name]) shutil.rmtree(self.releases) + sync_directory(self.root) print("Uninstalled. Personal dictionaries and conf/inkflow.conf were kept.") diff --git a/Linux/scripts/package-notices.py b/Linux/scripts/package-notices.py index 9eec701..acfcd75 100644 --- a/Linux/scripts/package-notices.py +++ b/Linux/scripts/package-notices.py @@ -65,7 +65,7 @@ def check_resources(resources, reused): def main(): package = Path(sys.argv[1]).resolve() - if command('git', 'status', '--porcelain', '--untracked-files=no'): + if command('git', 'status', '--porcelain', '--untracked-files=normal'): raise SystemExit('Commit tracked changes before packaging') revision = command('git', 'rev-parse', 'HEAD') for name in ('lib/fcitx5/libinkflow.so', 'lib/inkflow/librime.so', @@ -123,8 +123,10 @@ def main(): catalog = json.loads((ROOT / 'Core/config/chinese-sources.json').read_text()) for item in catalog: name = item['id'] + '.yaml' - copy(ROOT / 'build/dictionary-sources' / name, - source / 'dictionary-sources' / name, item['pinnedSHA256']) + original = ROOT / 'build/dictionary-sources' / name + if item['group'] == 'legacy': + original = ROOT / 'build/deps' / ('rime-pinyin-simp-' + item['pinnedCommit']) / item['path'] + copy(original, source / 'dictionary-sources' / name, item['pinnedSHA256']) # Both locked Rust graphs are used: engine and dictionary generator. for crate in ('Core/Portable', 'Core/Portable/dictionary'): diff --git a/Linux/scripts/package.sh b/Linux/scripts/package.sh index fca9f5d..5f96bfa 100755 --- a/Linux/scripts/package.sh +++ b/Linux/scripts/package.sh @@ -5,7 +5,7 @@ set -euo pipefail cd "$(dirname "$0")/../.." [[ $(uname -s) == Linux ]] || { echo 'Linux host required.' >&2; exit 1; } [[ $# -le 1 ]] || { echo 'Usage: package.sh [PREPARED_RESOURCES]' >&2; exit 1; } -[[ -z $(git status --porcelain --untracked-files=no) ]] || { +[[ -z $(git status --porcelain --untracked-files=normal) ]] || { echo 'Commit tracked source changes before packaging for exact provenance.' >&2; exit 1; } export CARGO_TARGET_DIR="$PWD/build/portable/cargo" diff --git a/Linux/scripts/test_install.py b/Linux/scripts/test_install.py index c972af1..d164a17 100644 --- a/Linux/scripts/test_install.py +++ b/Linux/scripts/test_install.py @@ -228,6 +228,39 @@ def inspect_stop(): with patch.object(self.desktop, "stop", side_effect=inspect_stop): self.manager.install(self.package("a"), self.desktop) + def test_recovery_before_mutations_preserves_shutdown_profile_flush(self): + self.manager.journal.write_text(json.dumps({"files": None, "service": self.desktop.service})) + profile = self.manager.paths["profile"] + profile.write_text(profile.read_text() + "[Groups/0/Items/2]\nName=recently-added-ime\n") + after_shutdown = profile.read_bytes() + self.manager.recover(self.desktop) + self.assertEqual(profile.read_bytes(), after_shutdown) + self.assertEqual(self.desktop.starts, 1) + + def test_partial_manual_registration_is_saved(self): + package = self.package("a") + for name in ("addon", "entry"): + with self.subTest(name=name): + manager = installer.Installer(self.root / name / "data", self.root / name / "config") + with manager.locked(): + path = manager.paths[name] + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("manual override") + manager.install(package, self.desktop) + manager.restore_manual(self.desktop) + self.assertEqual(path.read_text(), "manual override") + absent = "entry" if name == "addon" else "addon" + self.assertFalse(manager.paths[absent].exists()) + + def test_atomic_updates_sync_the_containing_directory(self): + path = self.manager.root / "durability-fixture" + with patch.object(installer, "sync_directory", wraps=installer.sync_directory) as sync: + installer.atomic_write(path, b"fixture") + sync.assert_called_with(path.parent) + sync.reset_mock() + installer.durable_unlink(path) + sync.assert_called_once_with(path.parent) + def test_missing_compiled_resource_is_rejected_even_with_fresh_hashes(self): package = self.package("a") (package / (installer.CACHE + "inkflow_spelling_31.prism.bin")).unlink() From dec1f3f2258ab24e62b73f5503311f76e51da0a2 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 20:04:37 +0800 Subject: [PATCH 19/21] fix(linux): prepare package resources in a fresh staging directory --- Linux/scripts/package.sh | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/Linux/scripts/package.sh b/Linux/scripts/package.sh index 5f96bfa..6d8ccd8 100755 --- a/Linux/scripts/package.sh +++ b/Linux/scripts/package.sh @@ -1,18 +1,21 @@ #!/bin/bash -# Build a directory package only. Commit tracked changes before packaging. +# Build a directory package only. Commit source changes before packaging. # Usage: bash Linux/scripts/package.sh [PREPARED_RESOURCES] set -euo pipefail cd "$(dirname "$0")/../.." [[ $(uname -s) == Linux ]] || { echo 'Linux host required.' >&2; exit 1; } [[ $# -le 1 ]] || { echo 'Usage: package.sh [PREPARED_RESOURCES]' >&2; exit 1; } [[ -z $(git status --porcelain --untracked-files=normal) ]] || { - echo 'Commit tracked source changes before packaging for exact provenance.' >&2; exit 1; + echo 'Commit source changes before packaging for exact provenance.' >&2; exit 1; } +mkdir -p build/linux +work=$(mktemp -d "$PWD/build/linux/package-work.XXXXXX") +trap 'rm -rf "$work"' EXIT export CARGO_TARGET_DIR="$PWD/build/portable/cargo" python3 Core/Portable/build-native.py bash Core/scripts/resource-dependencies.sh bash Core/scripts/prepare-chinese.sh --sources-only -resources=${1:-$PWD/build/linux/resources} +resources=${1:-$work/resources} resources=$(realpath -m "$resources") if [[ $# == 0 ]]; then bash Core/scripts/prepare-rime.sh "$resources/shared" @@ -33,9 +36,8 @@ cmake -S Linux/fcitx5 -B build/linux/cmake -G Ninja \ -DINKFLOW_RESOURCES="$resources" cmake --build build/linux/cmake --target inkflow --parallel "${CMAKE_BUILD_PARALLEL_LEVEL:-4}" # A failed build never replaces a previously completed package. -mkdir -p build/linux -stage=$(mktemp -d "$PWD/build/linux/stage.XXXXXX") -trap 'rm -rf "$stage"' EXIT +stage="$work/stage" +mkdir -p "$stage" DESTDIR="$stage" cmake --install build/linux/cmake cp "$resources/resources.json" "$stage/usr/share/inkflow/rime/resources.json" # The addon loader resolves a bare library name in its configured addon paths. From d7490e573861e88118ab9758239d46b45b37d782 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 20:12:15 +0800 Subject: [PATCH 20/21] docs(linux): record managed Omarchy installation verification --- Linux/fcitx5/README.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Linux/fcitx5/README.md b/Linux/fcitx5/README.md index 2e98c04..1e67ece 100644 --- a/Linux/fcitx5/README.md +++ b/Linux/fcitx5/README.md @@ -51,4 +51,6 @@ It starts a private Fcitx5 daemon on a separate D-Bus with temporary configurati On ARM64 Arch Linux/Omarchy with Hyprland, Fcitx5 5.1.23 and GCC 16.1.1, the native build, Rust runtime tests, 291 ranking cases, production resource preparation, 21-sample learned baseline, engine/personal-data parity, C ABI consumers, addon build, helper tests and installed-addon test passed. The desktop service loaded InkFlow and reported `inkflow-pinyin` as selected; the existing US keyboard and Pinyin entries were retained. +The managed package at `dec1f3f` passed all 20 installer tests and the real-package install, upgrade-fixture, rollback and uninstall exercise on that device. Each package switch loaded the addon on a private D-Bus; personal-data fixtures survived. The live manual installation was then upgraded, its original registration restored and compared byte-for-byte, and the managed package reinstalled. The final desktop service loaded its library and resources from the managed release with no automatic service restarts reported. ELF runtime paths are relative and do not depend on the build checkout. + The user confirmed the Omarchy installation works. Steam Deck/KDE Plasma, older Fcitx5 versions, Flatpak clients and OS-update persistence have not been verified; Steam Deck work is deferred at the user's request. Fcitx5 5.1.14 headers need C++20; older releases may need `FCITX_ADDON_FACTORY` (the fallback is compiled in). From a385c5902a403de09d3ef588ea8d793b480300d7 Mon Sep 17 00:00:00 2001 From: Xiaoyu Guan Date: Fri, 9 Oct 2026 22:02:39 +0800 Subject: [PATCH 21/21] fix(fcitx5): keep the engine off until import recovery succeeds and toggle ASCII mode with standalone left Shift - Run ifr_personal_recover before every engine start; if a pending import cannot be rolled back, keys pass through instead of Rime opening mixed dictionaries. - A standalone left Shift press and release now calls ifr_session_toggle_ascii_mode, matching the macOS frontend; any other key or Ctrl/Alt/Super disarms it. --- Linux/fcitx5/src/bridge.h | 21 +++++++++++++++++++++ Linux/fcitx5/src/engine.cpp | 23 +++++++++++++++++++++-- Linux/fcitx5/src/engine.h | 1 + Linux/fcitx5/tests/bridge_test.cpp | 7 +++++++ 4 files changed, 50 insertions(+), 2 deletions(-) diff --git a/Linux/fcitx5/src/bridge.h b/Linux/fcitx5/src/bridge.h index 5fc5a97..3e85910 100644 --- a/Linux/fcitx5/src/bridge.h +++ b/Linux/fcitx5/src/bridge.h @@ -104,6 +104,27 @@ inline std::uint32_t rime_modifiers(std::uint32_t fcitx_states, bool release) { return mask; } +// A standalone left Shift toggles ASCII mode, as on macOS: a press without Control, Alt +// or Super arms it, any other key disarms it, and the release that follows toggles. +struct ShiftToggle { + bool armed = false; + bool key(std::uint32_t keysym, std::uint32_t fcitx_states, bool release) { + const std::uint32_t shift_l = 0xffe1; + const std::uint32_t others = 1u << 2 | 1u << 3 | 1u << 6 | 1u << 26; + if (keysym != shift_l || (fcitx_states & others)) { + armed = false; + return false; + } + if (!release) { + armed = true; + return false; + } + bool toggle = armed; + armed = false; + return toggle; + } +}; + inline std::size_t utf8_sequence_length(unsigned char lead) { if (lead < 0x80) return 1; if ((lead & 0xE0) == 0xC0) return 2; diff --git a/Linux/fcitx5/src/engine.cpp b/Linux/fcitx5/src/engine.cpp index 095697e..399a76e 100644 --- a/Linux/fcitx5/src/engine.cpp +++ b/Linux/fcitx5/src/engine.cpp @@ -52,6 +52,14 @@ void Engine::createEngine() { INKFLOW_WARN() << "no prepared resources under XDG data directories; keys pass through"; return; } + // A pending import can leave old and new dictionaries mixed; restore the originals + // before Rime opens the directory, or stay off so typing cannot build on that state. + IFRStatus recovered = ifr_personal_recover(paths_.user.c_str()); + if (recovered != IFR_OK) { + INKFLOW_WARN() << "personal data needs recovery; keys pass through: " << recovered << " " + << ifr_last_error(); + return; + } IFREngineConfig config = {paths_.shared.c_str(), paths_.user.c_str(), paths_.cache.c_str(), paths_.context_index.c_str()}; IFRStatus status = ifr_engine_create(&config, &engine_); @@ -206,7 +214,7 @@ void Engine::importBackup(std::string file) { } } if (status != IFR_OK) { - INKFLOW_WARN() << "import failed, user data unchanged: " << status << " " << ifr_last_error(); + INKFLOW_WARN() << "import failed: " << status << " " << ifr_last_error(); } else { config_.candidateCount.setValue(static_cast(ifr_backup_candidate_count(backup))); uint32_t options = ifr_backup_input_options(backup); @@ -272,8 +280,19 @@ void Engine::keyEvent(const fcitx::InputMethodEntry&, fcitx::KeyEvent& event) { } if (!session(st)) return; const fcitx::Key& key = event.rawKey(); - if (!st->composing()) refreshContext(ic, st); int handled = 0; + if (st->shiftToggle.key(static_cast(key.sym()), static_cast(key.states()), + event.isRelease())) { + IFRStatus status = ifr_session_toggle_ascii_mode(st->session, &handled); + if (status != IFR_OK) { + failed(ic, st, "ascii toggle", status); + return; + } + refresh(ic, st); + event.filterAndAccept(); + return; + } + if (!st->composing()) refreshContext(ic, st); IFRStatus status = ifr_session_key(st->session, static_cast(key.sym()), static_cast(rime_modifiers(key.states(), event.isRelease())), &handled); diff --git a/Linux/fcitx5/src/engine.h b/Linux/fcitx5/src/engine.h index 39a783e..8c3cfc9 100644 --- a/Linux/fcitx5/src/engine.h +++ b/Linux/fcitx5/src/engine.h @@ -38,6 +38,7 @@ class State final : public fcitx::InputContextProperty { } IFRSession* session = nullptr; SnapshotRef snapshot; + ShiftToggle shiftToggle; }; class Engine final : public fcitx::InputMethodEngineV2 { diff --git a/Linux/fcitx5/tests/bridge_test.cpp b/Linux/fcitx5/tests/bridge_test.cpp index b056977..26d0963 100644 --- a/Linux/fcitx5/tests/bridge_test.cpp +++ b/Linux/fcitx5/tests/bridge_test.cpp @@ -54,6 +54,13 @@ int main() { assert(rime_modifiers(1u << 26, true) == ((1u << 26) | (1u << 30))); assert(rime_modifiers(1u << 4 | 1u << 31, false) == 0); + ShiftToggle shift; + assert(!shift.key(0xffe1, 0, false) && shift.key(0xffe1, 1u << 0, true)); + assert(!shift.key(0xffe1, 1u << 0, true)); + assert(!shift.key(0xffe1, 0, false) && !shift.key('A', 1u << 0, false) && !shift.key(0xffe1, 1u << 0, true)); + assert(!shift.key(0xffe1, 1u << 2, false) && !shift.key(0xffe1, 1u << 2, true)); + assert(!shift.key(0xffe2, 0, false) && !shift.key(0xffe2, 1u << 0, true)); + assert(preceding_text("中文abc", 5, 16) == "中文abc"); assert(preceding_text("中文abc", 5, 2) == "bc"); assert(preceding_text("中文abc", 2, 16) == "中文");