hololake-system-architecture/product-source/hololake-native-desktop/src-tauri/src/education_translation.rs

1541 lines
52 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// SPDX-License-Identifier: AGPL-3.0-or-later
//! 教育表格的双向动态转译层。
//!
//! 外部文件只作为只读输入。解析完成后立即转换为 HoloLake 原生列/行结构并进入
//! 当前登录账号的 SQLite 空间WebView 不接触任意本机路径。导出则从原生结构生成
//! 人类选择的交换格式,不把 Office 文档模型带进运行内核。
use crate::education_workspace::{
education_workspace_database, import_table_batch_at, validate_table_data,
EducationImportRegistration, EducationTableColumn, EducationTableRow, ImportedEducationTable,
};
use calamine::{open_workbook_auto, Data, Reader};
use encoding_rs::GBK;
use ring::digest::{digest, SHA256};
use rust_xlsxwriter::{Color, Format, Workbook};
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::fs;
use std::path::Path;
use std::time::{SystemTime, UNIX_EPOCH};
use tauri::AppHandle;
use tauri_plugin_dialog::DialogExt;
use uuid::Uuid;
const TRANSLATOR_SCHEMA: &str = "hololake.education-table-translator/v1";
const NATIVE_TABLE_SCHEMA: &str = "hololake.education-table/v1";
const IMPORT_ADAPTER: &str = "EDU-TABLE-IMPORT-ADAPTER/v1";
const EXPORT_ADAPTER: &str = "EDU-TABLE-EXPORT-ADAPTER/v1";
const MAX_IMPORT_FILE_BYTES: u64 = 25 * 1024 * 1024;
const MAX_TABLE_COLUMNS: usize = 30;
const MAX_TABLE_ROWS: usize = 1_000;
const MAX_CELL_BYTES: usize = 10_000;
const MAX_TITLE_BYTES: usize = 300;
const MODULE_NUMBER: &str = "HLP-MOD-OFFICIAL-EDUCATION-WORKBENCH-0001";
const ADAPTER: &str = "education-workbench-v1";
fn require_active(app: &AppHandle) -> Result<(), String> {
crate::module_package_runtime::require_active_module_adapter(app, MODULE_NUMBER, ADAPTER)
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableImportItem {
pub table_id: String,
pub title: String,
pub sheet_name: String,
pub column_count: usize,
pub row_count: usize,
pub leading_rows_ignored: usize,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableImportReceipt {
pub schema: &'static str,
pub state: &'static str,
pub adapter_id: &'static str,
pub import_id: String,
pub source_format: String,
pub source_filename: String,
pub source_sha256: String,
pub source_bytes: u64,
pub native_schema: &'static str,
pub content_profile: EducationContentProfile,
pub imported_tables: Vec<EducationTableImportItem>,
pub imported_at_unix_ms: u128,
pub source_preserved_read_only: bool,
pub truncated: bool,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentProfile {
pub schema: &'static str,
pub state: &'static str,
pub detected_family: String,
pub recognition_mode: String,
pub container_format: String,
pub extension_matches_container: bool,
pub page_kind: String,
pub page_count: usize,
pub non_empty_page_count: usize,
pub total_data_rows: usize,
pub total_columns: usize,
pub pages: Vec<EducationContentPageProfile>,
pub routing: Vec<EducationContentRoute>,
pub unresolved_signals: Vec<String>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationImportOutcome {
pub state: String,
pub import_receipt: Option<EducationTableImportReceipt>,
pub assistance_receipt: Option<EducationRecognitionAssistanceReceipt>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationRecognitionAssistanceReceipt {
pub schema: &'static str,
pub state: String,
pub request_id: String,
pub title: String,
pub message: String,
pub source_filename: String,
pub detected_container: String,
pub reason_code: String,
pub model_assistance_eligible: bool,
pub model_api_state: &'static str,
pub model_api_slot: &'static str,
pub requires_explicit_file_consent: bool,
pub source_preserved_read_only: bool,
pub available_learning_scopes: Vec<&'static str>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationRecognitionCapability {
pub schema: &'static str,
pub state: &'static str,
pub deterministic_adapter_ids: Vec<&'static str>,
pub model_api_slot: &'static str,
pub model_api_state: &'static str,
pub provider_binding: &'static str,
pub secret_storage_requirement: &'static str,
pub file_transfer_default: &'static str,
pub rule_update_flow: Vec<&'static str>,
pub learning_scopes: Vec<&'static str>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentPageProfile {
pub page_name: String,
pub state: String,
pub used_rows: usize,
pub used_columns: usize,
pub header_row_index: Option<usize>,
pub header_confidence: String,
pub data_rows: usize,
pub text_cells: usize,
pub numeric_cells: usize,
pub boolean_cells: usize,
pub date_cells: usize,
pub formula_cells: usize,
pub structure_kind: String,
pub focus_state: String,
pub focus_title: String,
pub primary_measure_column: Option<String>,
pub secondary_measure_columns: Vec<String>,
pub focus_confidence: String,
pub focus_reasons: Vec<String>,
}
#[derive(Clone, Debug)]
struct EducationSemanticFocus {
state: String,
title: String,
primary_measure_column: Option<String>,
secondary_measure_columns: Vec<String>,
confidence: String,
reasons: Vec<String>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentRoute {
pub page_name: String,
pub source_structure: String,
pub native_kind: String,
pub target_module: String,
pub decision: String,
}
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
pub struct ExportEducationTableInput {
pub table: EducationTableExportDraft,
pub format: String,
}
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
pub struct EducationTableExportDraft {
pub table_id: String,
pub title: String,
pub columns: Vec<EducationTableColumn>,
pub rows: Vec<EducationTableRow>,
pub revision: i64,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableExportReceipt {
pub schema: &'static str,
pub state: &'static str,
pub adapter_id: &'static str,
pub table_id: String,
pub table_revision: i64,
pub target_format: String,
pub target_filename: String,
pub bytes: u64,
pub row_count: usize,
pub column_count: usize,
pub exported_at_unix_ms: u128,
}
#[derive(Clone, Debug)]
struct ParsedSheet {
sheet_name: String,
leading_rows_ignored: usize,
table: ImportedEducationTable,
}
#[derive(Clone, Debug)]
struct ParseOutcome {
source_format: String,
content_profile: EducationContentProfile,
parsed_sheets: Vec<ParsedSheet>,
}
#[derive(Clone, Copy, Debug, Default)]
struct CellCounts {
text: usize,
numeric: usize,
boolean: usize,
date: usize,
}
#[derive(Clone, Debug, Serialize, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
struct NativeEducationTableFile {
schema: String,
title: String,
columns: Vec<EducationTableColumn>,
rows: Vec<EducationTableRow>,
}
pub async fn import_education_tables_from_dialog(
app: AppHandle,
) -> Result<Option<EducationImportOutcome>, String> {
require_active(&app)?;
let selected = app
.dialog()
.file()
.set_title("导入外部表格并转为 HoloLake 原生格式")
.add_filter(
"表格文件",
&["xlsx", "xls", "xlsm", "xlsb", "ods", "csv", "tsv", "json"],
)
.blocking_pick_file();
let Some(selected) = selected else {
return Ok(None);
};
let path = selected
.into_path()
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_PATH_INVALID: {error}"))?;
let database = education_workspace_database(&app)?;
tauri::async_runtime::spawn_blocking(move || match import_tables_from_path(&database, &path) {
Ok(receipt) => Ok(EducationImportOutcome {
state: "STAGED_UNASSIGNED".into(),
import_receipt: Some(receipt),
assistance_receipt: None,
}),
Err(error) if is_human_assistance_case(&error) => Ok(EducationImportOutcome {
state: "NEEDS_HUMAN_DECISION".into(),
import_receipt: None,
assistance_receipt: Some(assistance_receipt_for(&path, &error)),
}),
Err(error) => Err(error),
})
.await
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_JOIN_FAILED: {error}"))?
.map(Some)
}
pub fn get_education_recognition_capability(
app: AppHandle,
) -> Result<EducationRecognitionCapability, String> {
require_active(&app)?;
Ok(EducationRecognitionCapability {
schema: "hololake.education-recognition-capability/v1",
state: "DETERMINISTIC_READY_MODEL_SLOT_UNBOUND",
deterministic_adapter_ids: vec![IMPORT_ADAPTER, EXPORT_ADAPTER],
model_api_slot: "HOLOLAKE_MODEL_RECOGNITION_API/v1",
model_api_state: "NOT_CONFIGURED",
provider_binding: "USER_SELECTED_PROVIDER",
secret_storage_requirement: "OPERATING_SYSTEM_SECRET_STORE",
file_transfer_default: "DENY_UNTIL_EXPLICIT_PER_FILE_CONSENT",
rule_update_flow: vec![
"MODEL_PROPOSES_CANDIDATE",
"LOCAL_VALIDATION",
"HUMAN_CONFIRMATION",
"VERSIONED_RULE_INSTALL",
],
learning_scopes: vec!["PRIVATE_ONLY", "SHARE_ANONYMIZED_RULE"],
})
}
pub async fn export_education_table_to_dialog(
app: AppHandle,
input: ExportEducationTableInput,
) -> Result<Option<EducationTableExportReceipt>, String> {
require_active(&app)?;
validate_export_draft(&input.table)?;
let format = normalized_export_format(&input.format)?;
let extension = match format.as_str() {
"XLSX" => "xlsx",
"CSV" => "csv",
"TSV" => "tsv",
"HOLOLAKE_NATIVE" => "holotable.json",
_ => return Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into()),
};
let filename = format!("{}.{}", safe_filename(&input.table.title), extension);
let selected = app
.dialog()
.file()
.set_title("导出教育表格")
.set_file_name(filename)
.add_filter(export_filter_label(&format), &[extension])
.blocking_save_file();
let Some(selected) = selected else {
return Ok(None);
};
let path = selected
.into_path()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_PATH_INVALID: {error}"))?;
tauri::async_runtime::spawn_blocking(move || export_table_to_path(&path, &format, input.table))
.await
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_JOIN_FAILED: {error}"))?
.map(Some)
}
fn import_tables_from_path(
database: &Path,
source: &Path,
) -> Result<EducationTableImportReceipt, String> {
let metadata = fs::metadata(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE: {error}"))?;
if !metadata.is_file() || metadata.len() == 0 || metadata.len() > MAX_IMPORT_FILE_BYTES {
return Err("HOLOLAKE_EDUCATION_IMPORT_FILE_BOUNDS_INVALID".into());
}
let source_filename = source
.file_name()
.and_then(|value| value.to_str())
.ok_or_else(|| "HOLOLAKE_EDUCATION_IMPORT_FILENAME_INVALID".to_string())?
.to_string();
let source_bytes = fs::read(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE: {error}"))?;
let source_sha256 = hex_digest(&source_bytes);
let outcome = parse_source_file(source, &source_bytes)?;
if outcome.parsed_sheets.is_empty() {
return Err("HOLOLAKE_EDUCATION_IMPORT_EMPTY".into());
}
let imported_at_unix_ms = now_ms();
let profile_json = serde_json::to_string(&outcome.content_profile)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_PROFILE_INVALID: {error}"))?;
let (import_id, imported) = import_table_batch_at(
database,
outcome
.parsed_sheets
.iter()
.map(|sheet| sheet.table.clone())
.collect(),
EducationImportRegistration {
source_filename: source_filename.clone(),
source_format: outcome.source_format.clone(),
source_sha256: source_sha256.clone(),
profile_json,
imported_at_unix_ms: imported_at_unix_ms as i64,
},
)?;
let imported_tables = outcome
.parsed_sheets
.into_iter()
.zip(imported)
.map(|(sheet, table)| EducationTableImportItem {
table_id: table.table_id,
title: table.title,
sheet_name: sheet.sheet_name,
column_count: table.columns.len(),
row_count: table.rows.len(),
leading_rows_ignored: sheet.leading_rows_ignored,
})
.collect();
Ok(EducationTableImportReceipt {
schema: TRANSLATOR_SCHEMA,
state: "STAGED_UNASSIGNED",
adapter_id: IMPORT_ADAPTER,
import_id,
source_format: outcome.source_format,
source_filename,
source_sha256,
source_bytes: metadata.len(),
native_schema: NATIVE_TABLE_SCHEMA,
content_profile: outcome.content_profile,
imported_tables,
imported_at_unix_ms,
source_preserved_read_only: true,
truncated: false,
})
}
fn parse_source_file(source: &Path, bytes: &[u8]) -> Result<ParseOutcome, String> {
let filename = source
.file_name()
.and_then(|value| value.to_str())
.unwrap_or_default()
.to_ascii_lowercase();
if filename.ends_with(".holotable.json") {
let sheet = parse_native_table(bytes)?;
return Ok(outcome_from_parsed(
"HOLOLAKE_NATIVE",
"NATIVE_JSON",
true,
vec![sheet],
Vec::new(),
));
}
let extension = source
.extension()
.and_then(|value| value.to_str())
.unwrap_or_default()
.to_ascii_lowercase();
let detected_container = detect_container(bytes);
match extension.as_str() {
"csv" => parse_delimited(bytes, b',', source).map(|sheet| {
outcome_from_parsed("CSV", "DELIMITED_TEXT", true, vec![sheet], Vec::new())
}),
"tsv" => parse_delimited(bytes, b'\t', source).map(|sheet| {
outcome_from_parsed("TSV", "DELIMITED_TEXT", true, vec![sheet], Vec::new())
}),
"xlsx" | "xls" | "xlsm" | "xlsb" | "ods" => parse_workbook(
source,
&detected_workbook_format(&extension, &detected_container),
&detected_container,
extension_matches_container(&extension, &detected_container),
),
_ => Err("HOLOLAKE_EDUCATION_IMPORT_FORMAT_UNSUPPORTED".into()),
}
}
fn parse_workbook(
source: &Path,
detected_format: &str,
detected_container: &str,
extension_matches: bool,
) -> Result<ParseOutcome, String> {
let mut workbook = open_workbook_auto(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_WORKBOOK_INVALID: {error}"))?;
let stem = file_stem(source);
let sheet_names = workbook.sheet_names().to_vec();
let mut parsed = Vec::new();
let mut profiles = Vec::new();
for sheet_name in sheet_names {
let range = workbook
.worksheet_range(&sheet_name)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_SHEET_INVALID: {error}"))?;
let formula_range = workbook.worksheet_formula(&sheet_name).ok();
let formula_cells = formula_range
.as_ref()
.map(|formulas| {
formulas
.rows()
.flat_map(|row| row.iter())
.filter(|formula| !formula.trim().is_empty())
.count()
})
.unwrap_or(0);
let cell_counts = workbook_cell_counts(range.rows().flat_map(|row| row.iter()));
let matrix = range
.rows()
.map(|row| row.iter().map(cell_text).collect::<Vec<_>>())
.collect::<Vec<_>>();
if let Some((table, leading_rows_ignored)) =
matrix_to_native_table(matrix, &workbook_table_title(&stem, &sheet_name), true)?
{
profiles.push(profile_for_imported_sheet(
&sheet_name,
&table,
leading_rows_ignored,
cell_counts,
formula_cells,
));
parsed.push(ParsedSheet {
sheet_name,
leading_rows_ignored,
table,
});
} else {
profiles.push(EducationContentPageProfile {
page_name: sheet_name,
state: "EMPTY_SKIPPED".into(),
used_rows: 0,
used_columns: 0,
header_row_index: None,
header_confidence: "NONE".into(),
data_rows: 0,
text_cells: 0,
numeric_cells: 0,
boolean_cells: 0,
date_cells: 0,
formula_cells,
structure_kind: "EMPTY".into(),
focus_state: "NO_CONTENT".into(),
focus_title: "空页".into(),
primary_measure_column: None,
secondary_measure_columns: Vec::new(),
focus_confidence: "NONE".into(),
focus_reasons: vec!["页面没有可分析的数据".into()],
});
}
}
Ok(outcome_from_parsed(
detected_format,
detected_container,
extension_matches,
parsed,
profiles,
))
}
fn parse_delimited(bytes: &[u8], delimiter: u8, source: &Path) -> Result<ParsedSheet, String> {
let decoded = decode_delimited_text(bytes)?;
let mut reader = csv::ReaderBuilder::new()
.has_headers(false)
.flexible(true)
.delimiter(delimiter)
.from_reader(decoded.as_bytes());
let mut matrix = Vec::new();
for record in reader.records() {
let record = record
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_DELIMITED_INVALID: {error}"))?;
matrix.push(record.iter().map(ToString::to_string).collect());
if matrix.len() > MAX_TABLE_ROWS + 32 {
return Err("HOLOLAKE_EDUCATION_IMPORT_ROW_LIMIT_EXCEEDED".into());
}
}
let title = file_stem(source);
let (table, leading_rows_ignored) = matrix_to_native_table(matrix, &title, false)?
.ok_or_else(|| "HOLOLAKE_EDUCATION_IMPORT_EMPTY".to_string())?;
Ok(ParsedSheet {
sheet_name: title,
leading_rows_ignored,
table,
})
}
fn parse_native_table(bytes: &[u8]) -> Result<ParsedSheet, String> {
let native: NativeEducationTableFile = serde_json::from_slice(bytes)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_NATIVE_INVALID: {error}"))?;
if native.schema != NATIVE_TABLE_SCHEMA {
return Err("HOLOLAKE_EDUCATION_IMPORT_NATIVE_SCHEMA_UNSUPPORTED".into());
}
let columns = native
.columns
.into_iter()
.map(|column| EducationTableColumn {
column_id: format!("COL-{}", Uuid::new_v4()),
title: column.title,
})
.collect::<Vec<_>>();
let rows = native
.rows
.into_iter()
.map(|row| EducationTableRow {
row_id: format!("ROW-{}", Uuid::new_v4()),
cells: row.cells,
})
.collect::<Vec<_>>();
validate_table_data(&columns, &rows)?;
Ok(ParsedSheet {
sheet_name: "HoloLake 原生表格".into(),
leading_rows_ignored: 0,
table: ImportedEducationTable {
title: native.title,
columns,
rows,
},
})
}
fn outcome_from_parsed(
source_format: &str,
container_format: &str,
extension_matches_container: bool,
parsed_sheets: Vec<ParsedSheet>,
mut profiles: Vec<EducationContentPageProfile>,
) -> ParseOutcome {
if profiles.is_empty() {
profiles = parsed_sheets
.iter()
.map(|sheet| {
let counts = inferred_cell_counts(&sheet.table);
profile_for_imported_sheet(
&sheet.sheet_name,
&sheet.table,
sheet.leading_rows_ignored,
counts,
0,
)
})
.collect();
}
let routing = profiles
.iter()
.map(|page| EducationContentRoute {
page_name: page.page_name.clone(),
source_structure: page.structure_kind.clone(),
native_kind: if page.state == "READY_TO_TRANSLATE" {
"HOLOLAKE_TABLE".into()
} else {
"NONE".into()
},
target_module: if page.state == "READY_TO_TRANSLATE" {
"UNASSIGNED_CHANNEL_STAGING".into()
} else {
"NONE".into()
},
decision: if page.state == "READY_TO_TRANSLATE" {
"PROFILED_THEN_STAGED_PENDING_HUMAN_ROUTE".into()
} else {
"EMPTY_PAGE_SKIPPED_WITH_RECORD".into()
},
})
.collect::<Vec<_>>();
let non_empty_page_count = profiles
.iter()
.filter(|profile| profile.state == "READY_TO_TRANSLATE")
.count();
let total_data_rows = profiles.iter().map(|profile| profile.data_rows).sum();
let total_columns = profiles.iter().map(|profile| profile.used_columns).sum();
ParseOutcome {
source_format: source_format.into(),
content_profile: EducationContentProfile {
schema: "hololake.content-profile/v1",
state: "PROFILED_BEFORE_WRITE",
detected_family: "TABULAR_DATA".into(),
recognition_mode: "DETERMINISTIC_LOCAL".into(),
container_format: container_format.into(),
extension_matches_container,
page_kind: if profiles.len() > 1 {
"WORKSHEET".into()
} else {
"TABLE_PAGE".into()
},
page_count: profiles.len(),
non_empty_page_count,
total_data_rows,
total_columns,
pages: profiles,
routing,
unresolved_signals: profile_unresolved_signals(container_format),
},
parsed_sheets,
}
}
fn profile_unresolved_signals(container_format: &str) -> Vec<String> {
match container_format {
"OOXML_WORKBOOK_ZIP" | "XLSB_WORKBOOK_ZIP" | "OLE_COMPOUND" | "ODS_ZIP" => vec![
"SOURCE_VISUAL_FORMATTING_IS_PRESENTATION_ONLY_NOT_NATIVE_DATA".into(),
"FORMULAS_ARE_IMPORTED_AS_LAST_STORED_VALUES_UNLESS_EXPLICITLY_REBUILT".into(),
],
_ => Vec::new(),
}
}
fn is_human_assistance_case(error: &str) -> bool {
error.starts_with("HOLOLAKE_EDUCATION_IMPORT_")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_PATH_INVALID")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_FILENAME_INVALID")
}
fn assistance_receipt_for(source: &Path, reason: &str) -> EducationRecognitionAssistanceReceipt {
let filename = source
.file_name()
.and_then(|value| value.to_str())
.unwrap_or("未命名文件")
.to_string();
let bytes = fs::read(source).unwrap_or_default();
let detected_container = detect_container(&bytes);
let safety_limit = reason.contains("LIMIT_EXCEEDED")
|| reason.contains("BOUNDS_INVALID")
|| reason.contains("TOO_LARGE");
let (title, message, eligible) = if safety_limit {
(
"这个文件超出了当前模块的安全处理范围".to_string(),
"HoloLake 已停止写入并保留原文件。请先拆分文件或减少单页行列;模型辅助不会绕过本机安全上限。".to_string(),
false,
)
} else {
(
"当前版本还不能可靠识别这个文件".to_string(),
"HoloLake 没有猜测内容,也没有写入模块。配置模型接口后,可在明确授权当前文件的前提下重新识别,并把新规则保存为仅自己使用或去隐私后共享。".to_string(),
true,
)
};
EducationRecognitionAssistanceReceipt {
schema: "hololake.recognition-assistance-receipt/v1",
state: if eligible {
"MODEL_ASSISTANCE_AVAILABLE_AFTER_CONFIGURATION".into()
} else {
"LOCAL_INPUT_CHANGE_REQUIRED".into()
},
request_id: format!("EDU-RECOGNITION-{}", Uuid::new_v4()),
title,
message,
source_filename: filename,
detected_container,
reason_code: reason.to_string(),
model_assistance_eligible: eligible,
model_api_state: "NOT_CONFIGURED",
model_api_slot: "HOLOLAKE_MODEL_RECOGNITION_API/v1",
requires_explicit_file_consent: true,
source_preserved_read_only: true,
available_learning_scopes: vec!["PRIVATE_ONLY", "SHARE_ANONYMIZED_RULE"],
}
}
fn profile_for_imported_sheet(
sheet_name: &str,
table: &ImportedEducationTable,
leading_rows_ignored: usize,
counts: CellCounts,
formula_cells: usize,
) -> EducationContentPageProfile {
let focus = infer_semantic_focus(table);
EducationContentPageProfile {
page_name: sheet_name.into(),
state: "READY_TO_TRANSLATE".into(),
used_rows: table.rows.len() + 1,
used_columns: table.columns.len(),
header_row_index: Some(leading_rows_ignored),
header_confidence: if leading_rows_ignored > 0 {
"DENSE_ROW_DETECTED".into()
} else {
"FIRST_NONEMPTY_ROW".into()
},
data_rows: table.rows.len(),
text_cells: counts.text,
numeric_cells: counts.numeric,
boolean_cells: counts.boolean,
date_cells: counts.date,
formula_cells,
structure_kind: if table.columns.len() == 1 {
"SINGLE_FIELD_LIST".into()
} else {
"RECTANGULAR_TABLE".into()
},
focus_state: focus.state,
focus_title: focus.title,
primary_measure_column: focus.primary_measure_column,
secondary_measure_columns: focus.secondary_measure_columns,
focus_confidence: focus.confidence,
focus_reasons: focus.reasons,
}
}
fn infer_semantic_focus(table: &ImportedEducationTable) -> EducationSemanticFocus {
let title = table.title.to_lowercase();
let compensation_context = contains_any(&title, &["稿费", "收入", "薪酬", "结算", "营收"]);
let achievement_context = contains_any(&title, &["成绩", "考试", "测评", "学员"]);
let attendance_context = contains_any(&title, &["考勤", "出勤", "签到"]);
let mut scored = table
.columns
.iter()
.enumerate()
.filter_map(|(index, column)| {
let name = column.title.trim();
if name.is_empty() || is_sensitive_header(name) {
return None;
}
let normalized = name.to_lowercase().replace([' ', '_', '-'], "");
let mut score = semantic_measure_score(&normalized, compensation_context);
let numeric_count = table
.rows
.iter()
.filter(|row| {
parse_semantic_number(row.cells.get(index).map(String::as_str).unwrap_or(""))
.is_some()
})
.count();
if !table.rows.is_empty() {
score += ((numeric_count * 30) / table.rows.len()) as i32;
}
(score > 0).then(|| (score, name.to_string()))
})
.collect::<Vec<_>>();
scored.sort_by(|left, right| right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1)));
let primary = scored.first().cloned();
let focus_title = if compensation_context {
"稿费收入".to_string()
} else if achievement_context {
"学习成绩".to_string()
} else if attendance_context {
"出勤情况".to_string()
} else if let Some((_, column)) = &primary {
column.clone()
} else {
"重点待选择".to_string()
};
let Some((primary_score, primary_column)) = primary else {
return EducationSemanticFocus {
state: "NEEDS_HUMAN_SELECTION".into(),
title: focus_title,
primary_measure_column: None,
secondary_measure_columns: Vec::new(),
confidence: "LOW".into(),
reasons: vec!["没有发现可可靠计算的重点字段,渲染层必须请人选择".into()],
};
};
if primary_score < 70 {
return EducationSemanticFocus {
state: "NEEDS_HUMAN_SELECTION".into(),
title: "重点待选择".into(),
primary_measure_column: None,
secondary_measure_columns: scored.into_iter().take(3).map(|(_, name)| name).collect(),
confidence: "LOW".into(),
reasons: vec!["字段语义和数值覆盖率不足以自动决定主指标".into()],
};
}
EducationSemanticFocus {
state: "DETERMINISTIC_FOCUS_INFERRED".into(),
title: focus_title.clone(),
primary_measure_column: Some(primary_column.clone()),
secondary_measure_columns: scored
.into_iter()
.skip(1)
.take(3)
.map(|(_, name)| name)
.collect(),
confidence: if primary_score >= 140 {
"HIGH"
} else {
"MEDIUM"
}
.into(),
reasons: vec![
format!("表名与页名语义指向“{focus_title}"),
format!("字段“{primary_column}”同时通过语义优先级与数值覆盖率评分"),
],
}
}
fn semantic_measure_score(name: &str, compensation_context: bool) -> i32 {
if contains_any(
name,
&["序号", "编号", "账号", "电话", "手机", "id", "日期", "时间"],
) {
return -200;
}
let mut score = if contains_any(name, &["实际收入", "实收", "到账"]) {
170
} else if contains_any(name, &["当月收入", "本月收入", "稿费", "结算金额"]) {
160
} else if contains_any(name, &["累计收入", "总收入", "收入合计"]) {
150
} else if contains_any(name, &["收入", "营收", "薪酬", "金额", "合计", "总计"]) {
120
} else if contains_any(
name,
&[
"成绩",
"分数",
"得分",
"数量",
"人数",
"课时",
"时长",
"成本",
"预算",
"进度",
"完成率",
],
) {
90
} else {
0
};
if compensation_context && contains_any(name, &["收入", "稿费", "实收", "到账"]) {
score += 35;
}
score
}
fn contains_any(value: &str, candidates: &[&str]) -> bool {
candidates.iter().any(|candidate| value.contains(candidate))
}
fn is_sensitive_header(value: &str) -> bool {
contains_any(
&value.to_lowercase(),
&[
"密码",
"口令",
"密钥",
"凭据",
"secret",
"token",
"apikey",
"accesskey",
"credential",
],
)
}
fn parse_semantic_number(value: &str) -> Option<f64> {
let normalized = value
.trim()
.replace([',', '', ' ', '¥', '¥', '$', '€', '£'], "")
.trim_end_matches('%')
.to_string();
if normalized.is_empty() {
return None;
}
normalized
.parse::<f64>()
.ok()
.filter(|value| value.is_finite())
}
fn workbook_cell_counts<'a>(cells: impl Iterator<Item = &'a Data>) -> CellCounts {
let mut counts = CellCounts::default();
for cell in cells {
match cell {
Data::Int(_) | Data::Float(_) => counts.numeric += 1,
Data::Bool(_) => counts.boolean += 1,
Data::DateTime(_) | Data::DateTimeIso(_) | Data::DurationIso(_) => counts.date += 1,
Data::String(value) if !value.trim().is_empty() => counts.text += 1,
Data::Error(_) => counts.text += 1,
Data::Empty | Data::String(_) => {}
}
}
counts
}
fn inferred_cell_counts(table: &ImportedEducationTable) -> CellCounts {
let mut counts = CellCounts {
text: table.columns.len(),
..CellCounts::default()
};
for cell in table.rows.iter().flat_map(|row| row.cells.iter()) {
let value = cell.trim();
if value.is_empty() {
continue;
}
if value.parse::<f64>().is_ok() {
counts.numeric += 1;
} else if matches!(value.to_ascii_lowercase().as_str(), "true" | "false") {
counts.boolean += 1;
} else if looks_like_iso_date(value) {
counts.date += 1;
} else {
counts.text += 1;
}
}
counts
}
fn looks_like_iso_date(value: &str) -> bool {
let bytes = value.as_bytes();
bytes.len() >= 10
&& bytes[0..4].iter().all(u8::is_ascii_digit)
&& matches!(bytes[4], b'-' | b'/')
&& bytes[5..7].iter().all(u8::is_ascii_digit)
&& matches!(bytes[7], b'-' | b'/')
&& bytes[8..10].iter().all(u8::is_ascii_digit)
}
fn detect_container(bytes: &[u8]) -> String {
if bytes.starts_with(&[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]) {
return "OLE_COMPOUND".into();
}
if bytes.starts_with(b"PK\x03\x04") {
if contains_bytes(bytes, b"xl/workbook.bin") {
return "XLSB_WORKBOOK_ZIP".into();
}
if contains_bytes(bytes, b"xl/workbook.xml") {
return "OOXML_WORKBOOK_ZIP".into();
}
if contains_bytes(bytes, b"content.xml") && contains_bytes(bytes, b"mimetype") {
return "ODS_ZIP".into();
}
return "ZIP_CONTAINER".into();
}
if bytes
.iter()
.take(512)
.all(|byte| *byte == 0 || *byte == 9 || *byte == 10 || *byte == 13 || *byte >= 0x20)
{
return "TEXT_STREAM".into();
}
"UNKNOWN_BINARY".into()
}
fn detected_workbook_format(extension: &str, container: &str) -> String {
match container {
"OLE_COMPOUND" => "XLS".into(),
"XLSB_WORKBOOK_ZIP" => "XLSB".into(),
"ODS_ZIP" => "ODS".into(),
"OOXML_WORKBOOK_ZIP" if extension == "xlsm" => "XLSM".into(),
"OOXML_WORKBOOK_ZIP" => "XLSX".into(),
_ => extension.to_ascii_uppercase(),
}
}
fn extension_matches_container(extension: &str, container: &str) -> bool {
match extension {
"xls" => container == "OLE_COMPOUND",
"xlsb" => container == "XLSB_WORKBOOK_ZIP",
"ods" => container == "ODS_ZIP",
"xlsx" | "xlsm" => container == "OOXML_WORKBOOK_ZIP",
_ => false,
}
}
fn contains_bytes(haystack: &[u8], needle: &[u8]) -> bool {
!needle.is_empty()
&& haystack
.windows(needle.len())
.any(|window| window == needle)
}
fn matrix_to_native_table(
matrix: Vec<Vec<String>>,
title: &str,
detect_header: bool,
) -> Result<Option<(ImportedEducationTable, usize)>, String> {
let nonempty_rows = matrix
.iter()
.enumerate()
.filter(|(_, row)| row.iter().any(|cell| !cell.trim().is_empty()))
.map(|(index, _)| index)
.collect::<Vec<_>>();
let Some(&first_nonempty) = nonempty_rows.first() else {
return Ok(None);
};
let header_index = if detect_header {
let mut best_index = first_nonempty;
let mut best_score = 0;
for index in nonempty_rows.iter().copied().take(10) {
let score = matrix[index]
.iter()
.filter(|cell| !cell.trim().is_empty())
.count();
if score > best_score {
best_index = index;
best_score = score;
}
}
best_index
} else {
first_nonempty
};
let last_row = *nonempty_rows.last().unwrap_or(&header_index);
let first_column = matrix[header_index..=last_row]
.iter()
.filter_map(|row| row.iter().position(|cell| !cell.trim().is_empty()))
.min()
.unwrap_or(0);
let last_column = matrix[header_index..=last_row]
.iter()
.filter_map(|row| row.iter().rposition(|cell| !cell.trim().is_empty()))
.max()
.unwrap_or(first_column);
let column_count = last_column.saturating_sub(first_column) + 1;
if column_count == 0 || column_count > MAX_TABLE_COLUMNS {
return Err("HOLOLAKE_EDUCATION_IMPORT_COLUMN_LIMIT_EXCEEDED".into());
}
let header = &matrix[header_index];
let mut seen_headers: HashMap<String, usize> = HashMap::new();
let mut columns = Vec::with_capacity(column_count);
for offset in 0..column_count {
let source_title = header
.get(first_column + offset)
.map(|value| value.trim())
.unwrap_or_default();
let base = if source_title.is_empty() {
format!("字段 {}", offset + 1)
} else {
source_title.to_string()
};
if base.len() > MAX_TITLE_BYTES {
return Err("HOLOLAKE_EDUCATION_IMPORT_HEADER_TOO_LARGE".into());
}
let count = seen_headers.entry(base.clone()).or_insert(0);
*count += 1;
let title = if *count == 1 {
base
} else {
format!("{} ({})", base, count)
};
columns.push(EducationTableColumn {
column_id: format!("COL-{}", Uuid::new_v4()),
title,
});
}
let mut rows = Vec::new();
for source_row in matrix.iter().take(last_row + 1).skip(header_index + 1) {
let cells = (0..column_count)
.map(|offset| {
source_row
.get(first_column + offset)
.cloned()
.unwrap_or_default()
})
.collect::<Vec<_>>();
if cells.iter().all(|cell| cell.trim().is_empty()) {
continue;
}
if cells.iter().any(|cell| cell.len() > MAX_CELL_BYTES) {
return Err("HOLOLAKE_EDUCATION_IMPORT_CELL_TOO_LARGE".into());
}
rows.push(EducationTableRow {
row_id: format!("ROW-{}", Uuid::new_v4()),
cells,
});
if rows.len() > MAX_TABLE_ROWS {
return Err("HOLOLAKE_EDUCATION_IMPORT_ROW_LIMIT_EXCEEDED".into());
}
}
let table = ImportedEducationTable {
title: bounded_title(title)?,
columns,
rows,
};
validate_table_data(&table.columns, &table.rows)?;
Ok(Some((table, header_index)))
}
fn cell_text(cell: &Data) -> String {
match cell {
Data::DateTime(value) => value
.as_datetime()
.map(|datetime| {
if datetime.and_utc().timestamp() % 86_400 == 0 {
datetime.date().to_string()
} else {
datetime.to_string()
}
})
.unwrap_or_else(|| value.to_string()),
Data::Empty => String::new(),
_ => cell.to_string(),
}
}
fn decode_delimited_text(bytes: &[u8]) -> Result<String, String> {
if let Some(stripped) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
return String::from_utf8(stripped.to_vec())
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Some(stripped) = bytes.strip_prefix(&[0xFF, 0xFE]) {
if stripped.len() % 2 != 0 {
return Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
let values = stripped
.chunks_exact(2)
.map(|chunk| u16::from_le_bytes([chunk[0], chunk[1]]))
.collect::<Vec<_>>();
return String::from_utf16(&values)
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Some(stripped) = bytes.strip_prefix(&[0xFE, 0xFF]) {
if stripped.len() % 2 != 0 {
return Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
let values = stripped
.chunks_exact(2)
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
.collect::<Vec<_>>();
return String::from_utf16(&values)
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Ok(value) = String::from_utf8(bytes.to_vec()) {
return Ok(value);
}
let (decoded, _, had_errors) = GBK.decode(bytes);
if had_errors {
Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into())
} else {
Ok(decoded.into_owned())
}
}
fn export_table_to_path(
target: &Path,
format: &str,
table: EducationTableExportDraft,
) -> Result<EducationTableExportReceipt, String> {
validate_export_draft(&table)?;
if let Some(parent) = target.parent() {
if !parent.exists() {
return Err("HOLOLAKE_EDUCATION_EXPORT_DIRECTORY_MISSING".into());
}
}
match format {
"XLSX" => write_xlsx(target, &table)?,
"CSV" => write_delimited(target, &table, b',')?,
"TSV" => write_delimited(target, &table, b'\t')?,
"HOLOLAKE_NATIVE" => write_native_table(target, &table)?,
_ => return Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into()),
}
let metadata = fs::metadata(target)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_VERIFY_FAILED: {error}"))?;
Ok(EducationTableExportReceipt {
schema: TRANSLATOR_SCHEMA,
state: "EXPORTED_FROM_NATIVE",
adapter_id: EXPORT_ADAPTER,
table_id: table.table_id,
table_revision: table.revision,
target_format: format.to_string(),
target_filename: target
.file_name()
.and_then(|value| value.to_str())
.unwrap_or("教育表格")
.to_string(),
bytes: metadata.len(),
row_count: table.rows.len(),
column_count: table.columns.len(),
exported_at_unix_ms: now_ms(),
})
}
fn write_xlsx(target: &Path, table: &EducationTableExportDraft) -> Result<(), String> {
let mut workbook = Workbook::new();
let worksheet = workbook.add_worksheet();
worksheet
.set_name("HoloLake 数据")
.map_err(export_xlsx_error)?;
let header = Format::new()
.set_bold()
.set_font_color(Color::RGB(0xF7EDC5))
.set_background_color(Color::RGB(0x111827));
for (column_index, column) in table.columns.iter().enumerate() {
worksheet
.write_string_with_format(0, column_index as u16, &column.title, &header)
.map_err(export_xlsx_error)?;
}
for (row_index, row) in table.rows.iter().enumerate() {
for (column_index, cell) in row.cells.iter().enumerate() {
worksheet
.write_string((row_index + 1) as u32, column_index as u16, cell)
.map_err(export_xlsx_error)?;
}
}
worksheet
.set_freeze_panes(1, 0)
.map_err(export_xlsx_error)?;
worksheet.autofit();
workbook.save(target).map_err(export_xlsx_error)
}
fn write_delimited(
target: &Path,
table: &EducationTableExportDraft,
delimiter: u8,
) -> Result<(), String> {
let mut writer = csv::WriterBuilder::new()
.delimiter(delimiter)
.from_writer(Vec::new());
writer
.write_record(table.columns.iter().map(|column| column.title.as_str()))
.map_err(export_csv_error)?;
for row in &table.rows {
let protected = row
.cells
.iter()
.map(|cell| protect_spreadsheet_formula(cell))
.collect::<Vec<_>>();
writer.write_record(protected).map_err(export_csv_error)?;
}
writer
.flush()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}"))?;
let bytes = writer
.into_inner()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}"))?;
let mut with_bom = Vec::with_capacity(bytes.len() + 3);
with_bom.extend_from_slice(&[0xEF, 0xBB, 0xBF]);
with_bom.extend_from_slice(&bytes);
fs::write(target, with_bom)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_WRITE_FAILED: {error}"))
}
fn write_native_table(target: &Path, table: &EducationTableExportDraft) -> Result<(), String> {
let native = NativeEducationTableFile {
schema: NATIVE_TABLE_SCHEMA.into(),
title: table.title.clone(),
columns: table.columns.clone(),
rows: table.rows.clone(),
};
let bytes = serde_json::to_vec_pretty(&native)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_NATIVE_FAILED: {error}"))?;
fs::write(target, bytes)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_WRITE_FAILED: {error}"))
}
fn validate_export_draft(table: &EducationTableExportDraft) -> Result<(), String> {
if !table.table_id.starts_with("EDU-TABLE-") || table.revision < 1 {
return Err("HOLOLAKE_EDUCATION_EXPORT_TABLE_INVALID".into());
}
bounded_title(&table.title)?;
validate_table_data(&table.columns, &table.rows)
}
fn normalized_export_format(value: &str) -> Result<String, String> {
let value = value.trim().to_ascii_uppercase();
if matches!(value.as_str(), "XLSX" | "CSV" | "TSV" | "HOLOLAKE_NATIVE") {
Ok(value)
} else {
Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into())
}
}
fn export_filter_label(format: &str) -> &'static str {
match format {
"XLSX" => "Excel 工作簿",
"CSV" => "CSV 表格",
"TSV" => "TSV 表格",
"HOLOLAKE_NATIVE" => "HoloLake 原生表格",
_ => "表格文件",
}
}
fn protect_spreadsheet_formula(cell: &str) -> String {
if cell
.trim_start()
.chars()
.next()
.is_some_and(|character| matches!(character, '=' | '+' | '-' | '@'))
{
format!("'{cell}")
} else {
cell.to_string()
}
}
fn bounded_title(value: &str) -> Result<String, String> {
let title = value.trim();
if title.is_empty()
|| title.len() > MAX_TITLE_BYTES
|| title
.chars()
.any(|character| matches!(character, '\0' | '\r' | '\n'))
{
Err("HOLOLAKE_EDUCATION_IMPORT_TITLE_INVALID".into())
} else {
Ok(title.to_string())
}
}
fn workbook_table_title(stem: &str, sheet_name: &str) -> String {
let candidate = format!("{stem} · {sheet_name}");
if candidate.len() <= MAX_TITLE_BYTES {
candidate
} else if sheet_name.len() <= MAX_TITLE_BYTES {
sheet_name.to_string()
} else {
"导入的教育表格".into()
}
}
fn file_stem(path: &Path) -> String {
path.file_stem()
.and_then(|value| value.to_str())
.filter(|value| !value.trim().is_empty())
.unwrap_or("导入的教育表格")
.to_string()
}
fn safe_filename(value: &str) -> String {
let name = value
.chars()
.map(|character| {
if matches!(
character,
'/' | '\\' | ':' | '*' | '?' | '"' | '<' | '>' | '|'
) || character.is_control()
{
'_'
} else {
character
}
})
.collect::<String>();
let name = name.trim().trim_matches('.');
if name.is_empty() {
"教育表格".into()
} else {
name.to_string()
}
}
fn hex_digest(bytes: &[u8]) -> String {
digest(&SHA256, bytes)
.as_ref()
.iter()
.map(|byte| format!("{byte:02x}"))
.collect()
}
fn export_xlsx_error(error: rust_xlsxwriter::XlsxError) -> String {
format!("HOLOLAKE_EDUCATION_EXPORT_XLSX_FAILED: {error}")
}
fn export_csv_error(error: csv::Error) -> String {
format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}")
}
fn now_ms() -> u128 {
SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_millis()
}
#[cfg(test)]
mod tests {
use super::*;
use tempfile::tempdir;
#[test]
fn csv_round_trip_preserves_chinese_and_blocks_formula_injection() {
let directory = tempdir().unwrap();
let source = directory.path().join("学员.csv");
fs::write(&source, "姓名,状态\n冰朔,连载中\n危险,=cmd()\n").unwrap();
let database = directory.path().join("education.sqlite3");
let receipt = import_tables_from_path(&database, &source).unwrap();
assert_eq!(receipt.imported_tables.len(), 1);
let imported = crate::education_workspace::read_table_at(
&database,
&receipt.imported_tables[0].table_id,
)
.unwrap();
assert_eq!(imported.rows[0].cells, vec!["冰朔", "连载中"]);
let export = directory.path().join("export.csv");
export_table_to_path(
&export,
"CSV",
EducationTableExportDraft {
table_id: imported.table_id,
title: imported.title,
columns: imported.columns,
rows: imported.rows,
revision: imported.revision,
},
)
.unwrap();
let exported = fs::read_to_string(export).unwrap();
assert!(exported.contains("危险,'=cmd()"));
}
#[test]
fn xlsx_import_uses_dense_header_and_exports_valid_workbook() {
let directory = tempdir().unwrap();
let source = directory.path().join("稿费.xlsx");
let mut workbook = Workbook::new();
let worksheet = workbook.add_worksheet();
worksheet.write_string(0, 0, "稿费汇总").unwrap();
worksheet.write_string(2, 0, "作品").unwrap();
worksheet.write_string(2, 1, "月份").unwrap();
worksheet.write_string(2, 2, "金额").unwrap();
worksheet.write_string(3, 0, "测试作品").unwrap();
worksheet.write_string(3, 1, "2025-10").unwrap();
worksheet.write_number(3, 2, 1234.5).unwrap();
workbook.save(&source).unwrap();
let database = directory.path().join("education.sqlite3");
let receipt = import_tables_from_path(&database, &source).unwrap();
assert_eq!(receipt.imported_tables[0].leading_rows_ignored, 2);
assert_eq!(receipt.imported_tables[0].column_count, 3);
assert_eq!(receipt.imported_tables[0].row_count, 1);
assert_eq!(receipt.content_profile.pages[0].focus_title, "稿费收入");
assert_eq!(
receipt.content_profile.pages[0]
.primary_measure_column
.as_deref(),
Some("金额")
);
let imported = crate::education_workspace::read_table_at(
&database,
&receipt.imported_tables[0].table_id,
)
.unwrap();
let target = directory.path().join("roundtrip.xlsx");
export_table_to_path(
&target,
"XLSX",
EducationTableExportDraft {
table_id: imported.table_id,
title: imported.title,
columns: imported.columns,
rows: imported.rows,
revision: imported.revision,
},
)
.unwrap();
let mut roundtrip = open_workbook_auto(target).unwrap();
let range = roundtrip.worksheet_range("HoloLake 数据").unwrap();
assert_eq!(range.get_value((0, 0)).unwrap().to_string(), "作品");
assert_eq!(range.get_value((1, 2)).unwrap().to_string(), "1234.5");
}
#[test]
fn oversized_sheet_fails_instead_of_truncating() {
let mut matrix = vec![(0..31).map(|index| format!("字段 {index}")).collect()];
matrix.push((0..31).map(|_| "".to_string()).collect());
assert_eq!(
matrix_to_native_table(matrix, "超限", false).unwrap_err(),
"HOLOLAKE_EDUCATION_IMPORT_COLUMN_LIMIT_EXCEEDED"
);
}
#[test]
fn ambiguous_table_keeps_the_focus_as_a_human_choice() {
let table = ImportedEducationTable {
title: "杂项记录".into(),
columns: vec![
EducationTableColumn {
column_id: "COL-A".into(),
title: "内容甲".into(),
},
EducationTableColumn {
column_id: "COL-B".into(),
title: "内容乙".into(),
},
],
rows: vec![EducationTableRow {
row_id: "ROW-A".into(),
cells: vec!["".into(), "".into()],
}],
};
let focus = infer_semantic_focus(&table);
assert_eq!(focus.state, "NEEDS_HUMAN_SELECTION");
assert_eq!(focus.primary_measure_column, None);
}
}