hololake-system-architecture/product-source/hololake-native-desktop/src-tauri/src/education_translation.rs

1541 lines
52 KiB
Rust
Raw Normal View History

// SPDX-License-Identifier: AGPL-3.0-or-later
//! 教育表格的双向动态转译层。
//!
//! 外部文件只作为只读输入。解析完成后立即转换为 HoloLake 原生列/行结构并进入
//! 当前登录账号的 SQLite 空间WebView 不接触任意本机路径。导出则从原生结构生成
//! 人类选择的交换格式,不把 Office 文档模型带进运行内核。
use crate::education_workspace::{
education_workspace_database, import_table_batch_at, validate_table_data,
EducationImportRegistration, EducationTableColumn, EducationTableRow, ImportedEducationTable,
};
use calamine::{open_workbook_auto, Data, Reader};
use encoding_rs::GBK;
use ring::digest::{digest, SHA256};
use rust_xlsxwriter::{Color, Format, Workbook};
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::fs;
use std::path::Path;
use std::time::{SystemTime, UNIX_EPOCH};
use tauri::AppHandle;
use tauri_plugin_dialog::DialogExt;
use uuid::Uuid;
const TRANSLATOR_SCHEMA: &str = "hololake.education-table-translator/v1";
const NATIVE_TABLE_SCHEMA: &str = "hololake.education-table/v1";
const IMPORT_ADAPTER: &str = "EDU-TABLE-IMPORT-ADAPTER/v1";
const EXPORT_ADAPTER: &str = "EDU-TABLE-EXPORT-ADAPTER/v1";
const MAX_IMPORT_FILE_BYTES: u64 = 25 * 1024 * 1024;
const MAX_TABLE_COLUMNS: usize = 30;
const MAX_TABLE_ROWS: usize = 1_000;
const MAX_CELL_BYTES: usize = 10_000;
const MAX_TITLE_BYTES: usize = 300;
const MODULE_NUMBER: &str = "HLP-MOD-OFFICIAL-EDUCATION-WORKBENCH-0001";
const ADAPTER: &str = "education-workbench-v1";
fn require_active(app: &AppHandle) -> Result<(), String> {
crate::module_package_runtime::require_active_module_adapter(app, MODULE_NUMBER, ADAPTER)
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableImportItem {
pub table_id: String,
pub title: String,
pub sheet_name: String,
pub column_count: usize,
pub row_count: usize,
pub leading_rows_ignored: usize,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableImportReceipt {
pub schema: &'static str,
pub state: &'static str,
pub adapter_id: &'static str,
pub import_id: String,
pub source_format: String,
pub source_filename: String,
pub source_sha256: String,
pub source_bytes: u64,
pub native_schema: &'static str,
pub content_profile: EducationContentProfile,
pub imported_tables: Vec<EducationTableImportItem>,
pub imported_at_unix_ms: u128,
pub source_preserved_read_only: bool,
pub truncated: bool,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentProfile {
pub schema: &'static str,
pub state: &'static str,
pub detected_family: String,
pub recognition_mode: String,
pub container_format: String,
pub extension_matches_container: bool,
pub page_kind: String,
pub page_count: usize,
pub non_empty_page_count: usize,
pub total_data_rows: usize,
pub total_columns: usize,
pub pages: Vec<EducationContentPageProfile>,
pub routing: Vec<EducationContentRoute>,
pub unresolved_signals: Vec<String>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationImportOutcome {
pub state: String,
pub import_receipt: Option<EducationTableImportReceipt>,
pub assistance_receipt: Option<EducationRecognitionAssistanceReceipt>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationRecognitionAssistanceReceipt {
pub schema: &'static str,
pub state: String,
pub request_id: String,
pub title: String,
pub message: String,
pub source_filename: String,
pub detected_container: String,
pub reason_code: String,
pub model_assistance_eligible: bool,
pub model_api_state: &'static str,
pub model_api_slot: &'static str,
pub requires_explicit_file_consent: bool,
pub source_preserved_read_only: bool,
pub available_learning_scopes: Vec<&'static str>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationRecognitionCapability {
pub schema: &'static str,
pub state: &'static str,
pub deterministic_adapter_ids: Vec<&'static str>,
pub model_api_slot: &'static str,
pub model_api_state: &'static str,
pub provider_binding: &'static str,
pub secret_storage_requirement: &'static str,
pub file_transfer_default: &'static str,
pub rule_update_flow: Vec<&'static str>,
pub learning_scopes: Vec<&'static str>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentPageProfile {
pub page_name: String,
pub state: String,
pub used_rows: usize,
pub used_columns: usize,
pub header_row_index: Option<usize>,
pub header_confidence: String,
pub data_rows: usize,
pub text_cells: usize,
pub numeric_cells: usize,
pub boolean_cells: usize,
pub date_cells: usize,
pub formula_cells: usize,
pub structure_kind: String,
pub focus_state: String,
pub focus_title: String,
pub primary_measure_column: Option<String>,
pub secondary_measure_columns: Vec<String>,
pub focus_confidence: String,
pub focus_reasons: Vec<String>,
}
#[derive(Clone, Debug)]
struct EducationSemanticFocus {
state: String,
title: String,
primary_measure_column: Option<String>,
secondary_measure_columns: Vec<String>,
confidence: String,
reasons: Vec<String>,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationContentRoute {
pub page_name: String,
pub source_structure: String,
pub native_kind: String,
pub target_module: String,
pub decision: String,
}
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
pub struct ExportEducationTableInput {
pub table: EducationTableExportDraft,
pub format: String,
}
#[derive(Clone, Debug, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
pub struct EducationTableExportDraft {
pub table_id: String,
pub title: String,
pub columns: Vec<EducationTableColumn>,
pub rows: Vec<EducationTableRow>,
pub revision: i64,
}
#[derive(Clone, Debug, Serialize)]
#[serde(rename_all = "camelCase")]
pub struct EducationTableExportReceipt {
pub schema: &'static str,
pub state: &'static str,
pub adapter_id: &'static str,
pub table_id: String,
pub table_revision: i64,
pub target_format: String,
pub target_filename: String,
pub bytes: u64,
pub row_count: usize,
pub column_count: usize,
pub exported_at_unix_ms: u128,
}
#[derive(Clone, Debug)]
struct ParsedSheet {
sheet_name: String,
leading_rows_ignored: usize,
table: ImportedEducationTable,
}
#[derive(Clone, Debug)]
struct ParseOutcome {
source_format: String,
content_profile: EducationContentProfile,
parsed_sheets: Vec<ParsedSheet>,
}
#[derive(Clone, Copy, Debug, Default)]
struct CellCounts {
text: usize,
numeric: usize,
boolean: usize,
date: usize,
}
#[derive(Clone, Debug, Serialize, Deserialize)]
#[serde(rename_all = "camelCase", deny_unknown_fields)]
struct NativeEducationTableFile {
schema: String,
title: String,
columns: Vec<EducationTableColumn>,
rows: Vec<EducationTableRow>,
}
pub async fn import_education_tables_from_dialog(
app: AppHandle,
) -> Result<Option<EducationImportOutcome>, String> {
require_active(&app)?;
let selected = app
.dialog()
.file()
.set_title("导入外部表格并转为 HoloLake 原生格式")
.add_filter(
"表格文件",
&["xlsx", "xls", "xlsm", "xlsb", "ods", "csv", "tsv", "json"],
)
.blocking_pick_file();
let Some(selected) = selected else {
return Ok(None);
};
let path = selected
.into_path()
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_PATH_INVALID: {error}"))?;
let database = education_workspace_database(&app)?;
tauri::async_runtime::spawn_blocking(move || match import_tables_from_path(&database, &path) {
Ok(receipt) => Ok(EducationImportOutcome {
state: "STAGED_UNASSIGNED".into(),
import_receipt: Some(receipt),
assistance_receipt: None,
}),
Err(error) if is_human_assistance_case(&error) => Ok(EducationImportOutcome {
state: "NEEDS_HUMAN_DECISION".into(),
import_receipt: None,
assistance_receipt: Some(assistance_receipt_for(&path, &error)),
}),
Err(error) => Err(error),
})
.await
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_JOIN_FAILED: {error}"))?
.map(Some)
}
pub fn get_education_recognition_capability(
app: AppHandle,
) -> Result<EducationRecognitionCapability, String> {
require_active(&app)?;
Ok(EducationRecognitionCapability {
schema: "hololake.education-recognition-capability/v1",
state: "DETERMINISTIC_READY_MODEL_SLOT_UNBOUND",
deterministic_adapter_ids: vec![IMPORT_ADAPTER, EXPORT_ADAPTER],
model_api_slot: "HOLOLAKE_MODEL_RECOGNITION_API/v1",
model_api_state: "NOT_CONFIGURED",
provider_binding: "USER_SELECTED_PROVIDER",
secret_storage_requirement: "OPERATING_SYSTEM_SECRET_STORE",
file_transfer_default: "DENY_UNTIL_EXPLICIT_PER_FILE_CONSENT",
rule_update_flow: vec![
"MODEL_PROPOSES_CANDIDATE",
"LOCAL_VALIDATION",
"HUMAN_CONFIRMATION",
"VERSIONED_RULE_INSTALL",
],
learning_scopes: vec!["PRIVATE_ONLY", "SHARE_ANONYMIZED_RULE"],
})
}
pub async fn export_education_table_to_dialog(
app: AppHandle,
input: ExportEducationTableInput,
) -> Result<Option<EducationTableExportReceipt>, String> {
require_active(&app)?;
validate_export_draft(&input.table)?;
let format = normalized_export_format(&input.format)?;
let extension = match format.as_str() {
"XLSX" => "xlsx",
"CSV" => "csv",
"TSV" => "tsv",
"HOLOLAKE_NATIVE" => "holotable.json",
_ => return Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into()),
};
let filename = format!("{}.{}", safe_filename(&input.table.title), extension);
let selected = app
.dialog()
.file()
.set_title("导出教育表格")
.set_file_name(filename)
.add_filter(export_filter_label(&format), &[extension])
.blocking_save_file();
let Some(selected) = selected else {
return Ok(None);
};
let path = selected
.into_path()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_PATH_INVALID: {error}"))?;
tauri::async_runtime::spawn_blocking(move || export_table_to_path(&path, &format, input.table))
.await
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_JOIN_FAILED: {error}"))?
.map(Some)
}
fn import_tables_from_path(
database: &Path,
source: &Path,
) -> Result<EducationTableImportReceipt, String> {
let metadata = fs::metadata(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE: {error}"))?;
if !metadata.is_file() || metadata.len() == 0 || metadata.len() > MAX_IMPORT_FILE_BYTES {
return Err("HOLOLAKE_EDUCATION_IMPORT_FILE_BOUNDS_INVALID".into());
}
let source_filename = source
.file_name()
.and_then(|value| value.to_str())
.ok_or_else(|| "HOLOLAKE_EDUCATION_IMPORT_FILENAME_INVALID".to_string())?
.to_string();
let source_bytes = fs::read(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE: {error}"))?;
let source_sha256 = hex_digest(&source_bytes);
let outcome = parse_source_file(source, &source_bytes)?;
if outcome.parsed_sheets.is_empty() {
return Err("HOLOLAKE_EDUCATION_IMPORT_EMPTY".into());
}
let imported_at_unix_ms = now_ms();
let profile_json = serde_json::to_string(&outcome.content_profile)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_PROFILE_INVALID: {error}"))?;
let (import_id, imported) = import_table_batch_at(
database,
outcome
.parsed_sheets
.iter()
.map(|sheet| sheet.table.clone())
.collect(),
EducationImportRegistration {
source_filename: source_filename.clone(),
source_format: outcome.source_format.clone(),
source_sha256: source_sha256.clone(),
profile_json,
imported_at_unix_ms: imported_at_unix_ms as i64,
},
)?;
let imported_tables = outcome
.parsed_sheets
.into_iter()
.zip(imported)
.map(|(sheet, table)| EducationTableImportItem {
table_id: table.table_id,
title: table.title,
sheet_name: sheet.sheet_name,
column_count: table.columns.len(),
row_count: table.rows.len(),
leading_rows_ignored: sheet.leading_rows_ignored,
})
.collect();
Ok(EducationTableImportReceipt {
schema: TRANSLATOR_SCHEMA,
state: "STAGED_UNASSIGNED",
adapter_id: IMPORT_ADAPTER,
import_id,
source_format: outcome.source_format,
source_filename,
source_sha256,
source_bytes: metadata.len(),
native_schema: NATIVE_TABLE_SCHEMA,
content_profile: outcome.content_profile,
imported_tables,
imported_at_unix_ms,
source_preserved_read_only: true,
truncated: false,
})
}
fn parse_source_file(source: &Path, bytes: &[u8]) -> Result<ParseOutcome, String> {
let filename = source
.file_name()
.and_then(|value| value.to_str())
.unwrap_or_default()
.to_ascii_lowercase();
if filename.ends_with(".holotable.json") {
let sheet = parse_native_table(bytes)?;
return Ok(outcome_from_parsed(
"HOLOLAKE_NATIVE",
"NATIVE_JSON",
true,
vec![sheet],
Vec::new(),
));
}
let extension = source
.extension()
.and_then(|value| value.to_str())
.unwrap_or_default()
.to_ascii_lowercase();
let detected_container = detect_container(bytes);
match extension.as_str() {
"csv" => parse_delimited(bytes, b',', source).map(|sheet| {
outcome_from_parsed("CSV", "DELIMITED_TEXT", true, vec![sheet], Vec::new())
}),
"tsv" => parse_delimited(bytes, b'\t', source).map(|sheet| {
outcome_from_parsed("TSV", "DELIMITED_TEXT", true, vec![sheet], Vec::new())
}),
"xlsx" | "xls" | "xlsm" | "xlsb" | "ods" => parse_workbook(
source,
&detected_workbook_format(&extension, &detected_container),
&detected_container,
extension_matches_container(&extension, &detected_container),
),
_ => Err("HOLOLAKE_EDUCATION_IMPORT_FORMAT_UNSUPPORTED".into()),
}
}
fn parse_workbook(
source: &Path,
detected_format: &str,
detected_container: &str,
extension_matches: bool,
) -> Result<ParseOutcome, String> {
let mut workbook = open_workbook_auto(source)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_WORKBOOK_INVALID: {error}"))?;
let stem = file_stem(source);
let sheet_names = workbook.sheet_names().to_vec();
let mut parsed = Vec::new();
let mut profiles = Vec::new();
for sheet_name in sheet_names {
let range = workbook
.worksheet_range(&sheet_name)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_SHEET_INVALID: {error}"))?;
let formula_range = workbook.worksheet_formula(&sheet_name).ok();
let formula_cells = formula_range
.as_ref()
.map(|formulas| {
formulas
.rows()
.flat_map(|row| row.iter())
.filter(|formula| !formula.trim().is_empty())
.count()
})
.unwrap_or(0);
let cell_counts = workbook_cell_counts(range.rows().flat_map(|row| row.iter()));
let matrix = range
.rows()
.map(|row| row.iter().map(cell_text).collect::<Vec<_>>())
.collect::<Vec<_>>();
if let Some((table, leading_rows_ignored)) =
matrix_to_native_table(matrix, &workbook_table_title(&stem, &sheet_name), true)?
{
profiles.push(profile_for_imported_sheet(
&sheet_name,
&table,
leading_rows_ignored,
cell_counts,
formula_cells,
));
parsed.push(ParsedSheet {
sheet_name,
leading_rows_ignored,
table,
});
} else {
profiles.push(EducationContentPageProfile {
page_name: sheet_name,
state: "EMPTY_SKIPPED".into(),
used_rows: 0,
used_columns: 0,
header_row_index: None,
header_confidence: "NONE".into(),
data_rows: 0,
text_cells: 0,
numeric_cells: 0,
boolean_cells: 0,
date_cells: 0,
formula_cells,
structure_kind: "EMPTY".into(),
focus_state: "NO_CONTENT".into(),
focus_title: "空页".into(),
primary_measure_column: None,
secondary_measure_columns: Vec::new(),
focus_confidence: "NONE".into(),
focus_reasons: vec!["页面没有可分析的数据".into()],
});
}
}
Ok(outcome_from_parsed(
detected_format,
detected_container,
extension_matches,
parsed,
profiles,
))
}
fn parse_delimited(bytes: &[u8], delimiter: u8, source: &Path) -> Result<ParsedSheet, String> {
let decoded = decode_delimited_text(bytes)?;
let mut reader = csv::ReaderBuilder::new()
.has_headers(false)
.flexible(true)
.delimiter(delimiter)
.from_reader(decoded.as_bytes());
let mut matrix = Vec::new();
for record in reader.records() {
let record = record
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_DELIMITED_INVALID: {error}"))?;
matrix.push(record.iter().map(ToString::to_string).collect());
if matrix.len() > MAX_TABLE_ROWS + 32 {
return Err("HOLOLAKE_EDUCATION_IMPORT_ROW_LIMIT_EXCEEDED".into());
}
}
let title = file_stem(source);
let (table, leading_rows_ignored) = matrix_to_native_table(matrix, &title, false)?
.ok_or_else(|| "HOLOLAKE_EDUCATION_IMPORT_EMPTY".to_string())?;
Ok(ParsedSheet {
sheet_name: title,
leading_rows_ignored,
table,
})
}
fn parse_native_table(bytes: &[u8]) -> Result<ParsedSheet, String> {
let native: NativeEducationTableFile = serde_json::from_slice(bytes)
.map_err(|error| format!("HOLOLAKE_EDUCATION_IMPORT_NATIVE_INVALID: {error}"))?;
if native.schema != NATIVE_TABLE_SCHEMA {
return Err("HOLOLAKE_EDUCATION_IMPORT_NATIVE_SCHEMA_UNSUPPORTED".into());
}
let columns = native
.columns
.into_iter()
.map(|column| EducationTableColumn {
column_id: format!("COL-{}", Uuid::new_v4()),
title: column.title,
})
.collect::<Vec<_>>();
let rows = native
.rows
.into_iter()
.map(|row| EducationTableRow {
row_id: format!("ROW-{}", Uuid::new_v4()),
cells: row.cells,
})
.collect::<Vec<_>>();
validate_table_data(&columns, &rows)?;
Ok(ParsedSheet {
sheet_name: "HoloLake 原生表格".into(),
leading_rows_ignored: 0,
table: ImportedEducationTable {
title: native.title,
columns,
rows,
},
})
}
fn outcome_from_parsed(
source_format: &str,
container_format: &str,
extension_matches_container: bool,
parsed_sheets: Vec<ParsedSheet>,
mut profiles: Vec<EducationContentPageProfile>,
) -> ParseOutcome {
if profiles.is_empty() {
profiles = parsed_sheets
.iter()
.map(|sheet| {
let counts = inferred_cell_counts(&sheet.table);
profile_for_imported_sheet(
&sheet.sheet_name,
&sheet.table,
sheet.leading_rows_ignored,
counts,
0,
)
})
.collect();
}
let routing = profiles
.iter()
.map(|page| EducationContentRoute {
page_name: page.page_name.clone(),
source_structure: page.structure_kind.clone(),
native_kind: if page.state == "READY_TO_TRANSLATE" {
"HOLOLAKE_TABLE".into()
} else {
"NONE".into()
},
target_module: if page.state == "READY_TO_TRANSLATE" {
"UNASSIGNED_CHANNEL_STAGING".into()
} else {
"NONE".into()
},
decision: if page.state == "READY_TO_TRANSLATE" {
"PROFILED_THEN_STAGED_PENDING_HUMAN_ROUTE".into()
} else {
"EMPTY_PAGE_SKIPPED_WITH_RECORD".into()
},
})
.collect::<Vec<_>>();
let non_empty_page_count = profiles
.iter()
.filter(|profile| profile.state == "READY_TO_TRANSLATE")
.count();
let total_data_rows = profiles.iter().map(|profile| profile.data_rows).sum();
let total_columns = profiles.iter().map(|profile| profile.used_columns).sum();
ParseOutcome {
source_format: source_format.into(),
content_profile: EducationContentProfile {
schema: "hololake.content-profile/v1",
state: "PROFILED_BEFORE_WRITE",
detected_family: "TABULAR_DATA".into(),
recognition_mode: "DETERMINISTIC_LOCAL".into(),
container_format: container_format.into(),
extension_matches_container,
page_kind: if profiles.len() > 1 {
"WORKSHEET".into()
} else {
"TABLE_PAGE".into()
},
page_count: profiles.len(),
non_empty_page_count,
total_data_rows,
total_columns,
pages: profiles,
routing,
unresolved_signals: profile_unresolved_signals(container_format),
},
parsed_sheets,
}
}
fn profile_unresolved_signals(container_format: &str) -> Vec<String> {
match container_format {
"OOXML_WORKBOOK_ZIP" | "XLSB_WORKBOOK_ZIP" | "OLE_COMPOUND" | "ODS_ZIP" => vec![
"SOURCE_VISUAL_FORMATTING_IS_PRESENTATION_ONLY_NOT_NATIVE_DATA".into(),
"FORMULAS_ARE_IMPORTED_AS_LAST_STORED_VALUES_UNLESS_EXPLICITLY_REBUILT".into(),
],
_ => Vec::new(),
}
}
fn is_human_assistance_case(error: &str) -> bool {
error.starts_with("HOLOLAKE_EDUCATION_IMPORT_")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_UNREADABLE")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_PATH_INVALID")
&& !error.starts_with("HOLOLAKE_EDUCATION_IMPORT_FILENAME_INVALID")
}
fn assistance_receipt_for(source: &Path, reason: &str) -> EducationRecognitionAssistanceReceipt {
let filename = source
.file_name()
.and_then(|value| value.to_str())
.unwrap_or("未命名文件")
.to_string();
let bytes = fs::read(source).unwrap_or_default();
let detected_container = detect_container(&bytes);
let safety_limit = reason.contains("LIMIT_EXCEEDED")
|| reason.contains("BOUNDS_INVALID")
|| reason.contains("TOO_LARGE");
let (title, message, eligible) = if safety_limit {
(
"这个文件超出了当前模块的安全处理范围".to_string(),
"HoloLake 已停止写入并保留原文件。请先拆分文件或减少单页行列;模型辅助不会绕过本机安全上限。".to_string(),
false,
)
} else {
(
"当前版本还不能可靠识别这个文件".to_string(),
"HoloLake 没有猜测内容,也没有写入模块。配置模型接口后,可在明确授权当前文件的前提下重新识别,并把新规则保存为仅自己使用或去隐私后共享。".to_string(),
true,
)
};
EducationRecognitionAssistanceReceipt {
schema: "hololake.recognition-assistance-receipt/v1",
state: if eligible {
"MODEL_ASSISTANCE_AVAILABLE_AFTER_CONFIGURATION".into()
} else {
"LOCAL_INPUT_CHANGE_REQUIRED".into()
},
request_id: format!("EDU-RECOGNITION-{}", Uuid::new_v4()),
title,
message,
source_filename: filename,
detected_container,
reason_code: reason.to_string(),
model_assistance_eligible: eligible,
model_api_state: "NOT_CONFIGURED",
model_api_slot: "HOLOLAKE_MODEL_RECOGNITION_API/v1",
requires_explicit_file_consent: true,
source_preserved_read_only: true,
available_learning_scopes: vec!["PRIVATE_ONLY", "SHARE_ANONYMIZED_RULE"],
}
}
fn profile_for_imported_sheet(
sheet_name: &str,
table: &ImportedEducationTable,
leading_rows_ignored: usize,
counts: CellCounts,
formula_cells: usize,
) -> EducationContentPageProfile {
let focus = infer_semantic_focus(table);
EducationContentPageProfile {
page_name: sheet_name.into(),
state: "READY_TO_TRANSLATE".into(),
used_rows: table.rows.len() + 1,
used_columns: table.columns.len(),
header_row_index: Some(leading_rows_ignored),
header_confidence: if leading_rows_ignored > 0 {
"DENSE_ROW_DETECTED".into()
} else {
"FIRST_NONEMPTY_ROW".into()
},
data_rows: table.rows.len(),
text_cells: counts.text,
numeric_cells: counts.numeric,
boolean_cells: counts.boolean,
date_cells: counts.date,
formula_cells,
structure_kind: if table.columns.len() == 1 {
"SINGLE_FIELD_LIST".into()
} else {
"RECTANGULAR_TABLE".into()
},
focus_state: focus.state,
focus_title: focus.title,
primary_measure_column: focus.primary_measure_column,
secondary_measure_columns: focus.secondary_measure_columns,
focus_confidence: focus.confidence,
focus_reasons: focus.reasons,
}
}
fn infer_semantic_focus(table: &ImportedEducationTable) -> EducationSemanticFocus {
let title = table.title.to_lowercase();
let compensation_context = contains_any(&title, &["稿费", "收入", "薪酬", "结算", "营收"]);
let achievement_context = contains_any(&title, &["成绩", "考试", "测评", "学员"]);
let attendance_context = contains_any(&title, &["考勤", "出勤", "签到"]);
let mut scored = table
.columns
.iter()
.enumerate()
.filter_map(|(index, column)| {
let name = column.title.trim();
if name.is_empty() || is_sensitive_header(name) {
return None;
}
let normalized = name.to_lowercase().replace([' ', '_', '-'], "");
let mut score = semantic_measure_score(&normalized, compensation_context);
let numeric_count = table
.rows
.iter()
.filter(|row| {
parse_semantic_number(row.cells.get(index).map(String::as_str).unwrap_or(""))
.is_some()
})
.count();
if !table.rows.is_empty() {
score += ((numeric_count * 30) / table.rows.len()) as i32;
}
(score > 0).then(|| (score, name.to_string()))
})
.collect::<Vec<_>>();
scored.sort_by(|left, right| right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1)));
let primary = scored.first().cloned();
let focus_title = if compensation_context {
"稿费收入".to_string()
} else if achievement_context {
"学习成绩".to_string()
} else if attendance_context {
"出勤情况".to_string()
} else if let Some((_, column)) = &primary {
column.clone()
} else {
"重点待选择".to_string()
};
let Some((primary_score, primary_column)) = primary else {
return EducationSemanticFocus {
state: "NEEDS_HUMAN_SELECTION".into(),
title: focus_title,
primary_measure_column: None,
secondary_measure_columns: Vec::new(),
confidence: "LOW".into(),
reasons: vec!["没有发现可可靠计算的重点字段,渲染层必须请人选择".into()],
};
};
if primary_score < 70 {
return EducationSemanticFocus {
state: "NEEDS_HUMAN_SELECTION".into(),
title: "重点待选择".into(),
primary_measure_column: None,
secondary_measure_columns: scored.into_iter().take(3).map(|(_, name)| name).collect(),
confidence: "LOW".into(),
reasons: vec!["字段语义和数值覆盖率不足以自动决定主指标".into()],
};
}
EducationSemanticFocus {
state: "DETERMINISTIC_FOCUS_INFERRED".into(),
title: focus_title.clone(),
primary_measure_column: Some(primary_column.clone()),
secondary_measure_columns: scored
.into_iter()
.skip(1)
.take(3)
.map(|(_, name)| name)
.collect(),
confidence: if primary_score >= 140 {
"HIGH"
} else {
"MEDIUM"
}
.into(),
reasons: vec![
format!("表名与页名语义指向“{focus_title}"),
format!("字段“{primary_column}”同时通过语义优先级与数值覆盖率评分"),
],
}
}
fn semantic_measure_score(name: &str, compensation_context: bool) -> i32 {
if contains_any(
name,
&["序号", "编号", "账号", "电话", "手机", "id", "日期", "时间"],
) {
return -200;
}
let mut score = if contains_any(name, &["实际收入", "实收", "到账"]) {
170
} else if contains_any(name, &["当月收入", "本月收入", "稿费", "结算金额"]) {
160
} else if contains_any(name, &["累计收入", "总收入", "收入合计"]) {
150
} else if contains_any(name, &["收入", "营收", "薪酬", "金额", "合计", "总计"]) {
120
} else if contains_any(
name,
&[
"成绩",
"分数",
"得分",
"数量",
"人数",
"课时",
"时长",
"成本",
"预算",
"进度",
"完成率",
],
) {
90
} else {
0
};
if compensation_context && contains_any(name, &["收入", "稿费", "实收", "到账"]) {
score += 35;
}
score
}
fn contains_any(value: &str, candidates: &[&str]) -> bool {
candidates.iter().any(|candidate| value.contains(candidate))
}
fn is_sensitive_header(value: &str) -> bool {
contains_any(
&value.to_lowercase(),
&[
"密码",
"口令",
"密钥",
"凭据",
"secret",
"token",
"apikey",
"accesskey",
"credential",
],
)
}
fn parse_semantic_number(value: &str) -> Option<f64> {
let normalized = value
.trim()
.replace([',', '', ' ', '¥', '¥', '$', '€', '£'], "")
.trim_end_matches('%')
.to_string();
if normalized.is_empty() {
return None;
}
normalized
.parse::<f64>()
.ok()
.filter(|value| value.is_finite())
}
fn workbook_cell_counts<'a>(cells: impl Iterator<Item = &'a Data>) -> CellCounts {
let mut counts = CellCounts::default();
for cell in cells {
match cell {
Data::Int(_) | Data::Float(_) => counts.numeric += 1,
Data::Bool(_) => counts.boolean += 1,
Data::DateTime(_) | Data::DateTimeIso(_) | Data::DurationIso(_) => counts.date += 1,
Data::String(value) if !value.trim().is_empty() => counts.text += 1,
Data::Error(_) => counts.text += 1,
Data::Empty | Data::String(_) => {}
}
}
counts
}
fn inferred_cell_counts(table: &ImportedEducationTable) -> CellCounts {
let mut counts = CellCounts {
text: table.columns.len(),
..CellCounts::default()
};
for cell in table.rows.iter().flat_map(|row| row.cells.iter()) {
let value = cell.trim();
if value.is_empty() {
continue;
}
if value.parse::<f64>().is_ok() {
counts.numeric += 1;
} else if matches!(value.to_ascii_lowercase().as_str(), "true" | "false") {
counts.boolean += 1;
} else if looks_like_iso_date(value) {
counts.date += 1;
} else {
counts.text += 1;
}
}
counts
}
fn looks_like_iso_date(value: &str) -> bool {
let bytes = value.as_bytes();
bytes.len() >= 10
&& bytes[0..4].iter().all(u8::is_ascii_digit)
&& matches!(bytes[4], b'-' | b'/')
&& bytes[5..7].iter().all(u8::is_ascii_digit)
&& matches!(bytes[7], b'-' | b'/')
&& bytes[8..10].iter().all(u8::is_ascii_digit)
}
fn detect_container(bytes: &[u8]) -> String {
if bytes.starts_with(&[0xD0, 0xCF, 0x11, 0xE0, 0xA1, 0xB1, 0x1A, 0xE1]) {
return "OLE_COMPOUND".into();
}
if bytes.starts_with(b"PK\x03\x04") {
if contains_bytes(bytes, b"xl/workbook.bin") {
return "XLSB_WORKBOOK_ZIP".into();
}
if contains_bytes(bytes, b"xl/workbook.xml") {
return "OOXML_WORKBOOK_ZIP".into();
}
if contains_bytes(bytes, b"content.xml") && contains_bytes(bytes, b"mimetype") {
return "ODS_ZIP".into();
}
return "ZIP_CONTAINER".into();
}
if bytes
.iter()
.take(512)
.all(|byte| *byte == 0 || *byte == 9 || *byte == 10 || *byte == 13 || *byte >= 0x20)
{
return "TEXT_STREAM".into();
}
"UNKNOWN_BINARY".into()
}
fn detected_workbook_format(extension: &str, container: &str) -> String {
match container {
"OLE_COMPOUND" => "XLS".into(),
"XLSB_WORKBOOK_ZIP" => "XLSB".into(),
"ODS_ZIP" => "ODS".into(),
"OOXML_WORKBOOK_ZIP" if extension == "xlsm" => "XLSM".into(),
"OOXML_WORKBOOK_ZIP" => "XLSX".into(),
_ => extension.to_ascii_uppercase(),
}
}
fn extension_matches_container(extension: &str, container: &str) -> bool {
match extension {
"xls" => container == "OLE_COMPOUND",
"xlsb" => container == "XLSB_WORKBOOK_ZIP",
"ods" => container == "ODS_ZIP",
"xlsx" | "xlsm" => container == "OOXML_WORKBOOK_ZIP",
_ => false,
}
}
fn contains_bytes(haystack: &[u8], needle: &[u8]) -> bool {
!needle.is_empty()
&& haystack
.windows(needle.len())
.any(|window| window == needle)
}
fn matrix_to_native_table(
matrix: Vec<Vec<String>>,
title: &str,
detect_header: bool,
) -> Result<Option<(ImportedEducationTable, usize)>, String> {
let nonempty_rows = matrix
.iter()
.enumerate()
.filter(|(_, row)| row.iter().any(|cell| !cell.trim().is_empty()))
.map(|(index, _)| index)
.collect::<Vec<_>>();
let Some(&first_nonempty) = nonempty_rows.first() else {
return Ok(None);
};
let header_index = if detect_header {
let mut best_index = first_nonempty;
let mut best_score = 0;
for index in nonempty_rows.iter().copied().take(10) {
let score = matrix[index]
.iter()
.filter(|cell| !cell.trim().is_empty())
.count();
if score > best_score {
best_index = index;
best_score = score;
}
}
best_index
} else {
first_nonempty
};
let last_row = *nonempty_rows.last().unwrap_or(&header_index);
let first_column = matrix[header_index..=last_row]
.iter()
.filter_map(|row| row.iter().position(|cell| !cell.trim().is_empty()))
.min()
.unwrap_or(0);
let last_column = matrix[header_index..=last_row]
.iter()
.filter_map(|row| row.iter().rposition(|cell| !cell.trim().is_empty()))
.max()
.unwrap_or(first_column);
let column_count = last_column.saturating_sub(first_column) + 1;
if column_count == 0 || column_count > MAX_TABLE_COLUMNS {
return Err("HOLOLAKE_EDUCATION_IMPORT_COLUMN_LIMIT_EXCEEDED".into());
}
let header = &matrix[header_index];
let mut seen_headers: HashMap<String, usize> = HashMap::new();
let mut columns = Vec::with_capacity(column_count);
for offset in 0..column_count {
let source_title = header
.get(first_column + offset)
.map(|value| value.trim())
.unwrap_or_default();
let base = if source_title.is_empty() {
format!("字段 {}", offset + 1)
} else {
source_title.to_string()
};
if base.len() > MAX_TITLE_BYTES {
return Err("HOLOLAKE_EDUCATION_IMPORT_HEADER_TOO_LARGE".into());
}
let count = seen_headers.entry(base.clone()).or_insert(0);
*count += 1;
let title = if *count == 1 {
base
} else {
format!("{} ({})", base, count)
};
columns.push(EducationTableColumn {
column_id: format!("COL-{}", Uuid::new_v4()),
title,
});
}
let mut rows = Vec::new();
for source_row in matrix.iter().take(last_row + 1).skip(header_index + 1) {
let cells = (0..column_count)
.map(|offset| {
source_row
.get(first_column + offset)
.cloned()
.unwrap_or_default()
})
.collect::<Vec<_>>();
if cells.iter().all(|cell| cell.trim().is_empty()) {
continue;
}
if cells.iter().any(|cell| cell.len() > MAX_CELL_BYTES) {
return Err("HOLOLAKE_EDUCATION_IMPORT_CELL_TOO_LARGE".into());
}
rows.push(EducationTableRow {
row_id: format!("ROW-{}", Uuid::new_v4()),
cells,
});
if rows.len() > MAX_TABLE_ROWS {
return Err("HOLOLAKE_EDUCATION_IMPORT_ROW_LIMIT_EXCEEDED".into());
}
}
let table = ImportedEducationTable {
title: bounded_title(title)?,
columns,
rows,
};
validate_table_data(&table.columns, &table.rows)?;
Ok(Some((table, header_index)))
}
fn cell_text(cell: &Data) -> String {
match cell {
Data::DateTime(value) => value
.as_datetime()
.map(|datetime| {
if datetime.and_utc().timestamp() % 86_400 == 0 {
datetime.date().to_string()
} else {
datetime.to_string()
}
})
.unwrap_or_else(|| value.to_string()),
Data::Empty => String::new(),
_ => cell.to_string(),
}
}
fn decode_delimited_text(bytes: &[u8]) -> Result<String, String> {
if let Some(stripped) = bytes.strip_prefix(&[0xEF, 0xBB, 0xBF]) {
return String::from_utf8(stripped.to_vec())
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Some(stripped) = bytes.strip_prefix(&[0xFF, 0xFE]) {
if stripped.len() % 2 != 0 {
return Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
let values = stripped
.chunks_exact(2)
.map(|chunk| u16::from_le_bytes([chunk[0], chunk[1]]))
.collect::<Vec<_>>();
return String::from_utf16(&values)
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Some(stripped) = bytes.strip_prefix(&[0xFE, 0xFF]) {
if stripped.len() % 2 != 0 {
return Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
let values = stripped
.chunks_exact(2)
.map(|chunk| u16::from_be_bytes([chunk[0], chunk[1]]))
.collect::<Vec<_>>();
return String::from_utf16(&values)
.map_err(|_| "HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into());
}
if let Ok(value) = String::from_utf8(bytes.to_vec()) {
return Ok(value);
}
let (decoded, _, had_errors) = GBK.decode(bytes);
if had_errors {
Err("HOLOLAKE_EDUCATION_IMPORT_TEXT_ENCODING_UNSUPPORTED".into())
} else {
Ok(decoded.into_owned())
}
}
fn export_table_to_path(
target: &Path,
format: &str,
table: EducationTableExportDraft,
) -> Result<EducationTableExportReceipt, String> {
validate_export_draft(&table)?;
if let Some(parent) = target.parent() {
if !parent.exists() {
return Err("HOLOLAKE_EDUCATION_EXPORT_DIRECTORY_MISSING".into());
}
}
match format {
"XLSX" => write_xlsx(target, &table)?,
"CSV" => write_delimited(target, &table, b',')?,
"TSV" => write_delimited(target, &table, b'\t')?,
"HOLOLAKE_NATIVE" => write_native_table(target, &table)?,
_ => return Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into()),
}
let metadata = fs::metadata(target)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_VERIFY_FAILED: {error}"))?;
Ok(EducationTableExportReceipt {
schema: TRANSLATOR_SCHEMA,
state: "EXPORTED_FROM_NATIVE",
adapter_id: EXPORT_ADAPTER,
table_id: table.table_id,
table_revision: table.revision,
target_format: format.to_string(),
target_filename: target
.file_name()
.and_then(|value| value.to_str())
.unwrap_or("教育表格")
.to_string(),
bytes: metadata.len(),
row_count: table.rows.len(),
column_count: table.columns.len(),
exported_at_unix_ms: now_ms(),
})
}
fn write_xlsx(target: &Path, table: &EducationTableExportDraft) -> Result<(), String> {
let mut workbook = Workbook::new();
let worksheet = workbook.add_worksheet();
worksheet
.set_name("HoloLake 数据")
.map_err(export_xlsx_error)?;
let header = Format::new()
.set_bold()
.set_font_color(Color::RGB(0xF7EDC5))
.set_background_color(Color::RGB(0x111827));
for (column_index, column) in table.columns.iter().enumerate() {
worksheet
.write_string_with_format(0, column_index as u16, &column.title, &header)
.map_err(export_xlsx_error)?;
}
for (row_index, row) in table.rows.iter().enumerate() {
for (column_index, cell) in row.cells.iter().enumerate() {
worksheet
.write_string((row_index + 1) as u32, column_index as u16, cell)
.map_err(export_xlsx_error)?;
}
}
worksheet
.set_freeze_panes(1, 0)
.map_err(export_xlsx_error)?;
worksheet.autofit();
workbook.save(target).map_err(export_xlsx_error)
}
fn write_delimited(
target: &Path,
table: &EducationTableExportDraft,
delimiter: u8,
) -> Result<(), String> {
let mut writer = csv::WriterBuilder::new()
.delimiter(delimiter)
.from_writer(Vec::new());
writer
.write_record(table.columns.iter().map(|column| column.title.as_str()))
.map_err(export_csv_error)?;
for row in &table.rows {
let protected = row
.cells
.iter()
.map(|cell| protect_spreadsheet_formula(cell))
.collect::<Vec<_>>();
writer.write_record(protected).map_err(export_csv_error)?;
}
writer
.flush()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}"))?;
let bytes = writer
.into_inner()
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}"))?;
let mut with_bom = Vec::with_capacity(bytes.len() + 3);
with_bom.extend_from_slice(&[0xEF, 0xBB, 0xBF]);
with_bom.extend_from_slice(&bytes);
fs::write(target, with_bom)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_WRITE_FAILED: {error}"))
}
fn write_native_table(target: &Path, table: &EducationTableExportDraft) -> Result<(), String> {
let native = NativeEducationTableFile {
schema: NATIVE_TABLE_SCHEMA.into(),
title: table.title.clone(),
columns: table.columns.clone(),
rows: table.rows.clone(),
};
let bytes = serde_json::to_vec_pretty(&native)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_NATIVE_FAILED: {error}"))?;
fs::write(target, bytes)
.map_err(|error| format!("HOLOLAKE_EDUCATION_EXPORT_WRITE_FAILED: {error}"))
}
fn validate_export_draft(table: &EducationTableExportDraft) -> Result<(), String> {
if !table.table_id.starts_with("EDU-TABLE-") || table.revision < 1 {
return Err("HOLOLAKE_EDUCATION_EXPORT_TABLE_INVALID".into());
}
bounded_title(&table.title)?;
validate_table_data(&table.columns, &table.rows)
}
fn normalized_export_format(value: &str) -> Result<String, String> {
let value = value.trim().to_ascii_uppercase();
if matches!(value.as_str(), "XLSX" | "CSV" | "TSV" | "HOLOLAKE_NATIVE") {
Ok(value)
} else {
Err("HOLOLAKE_EDUCATION_EXPORT_FORMAT_UNSUPPORTED".into())
}
}
fn export_filter_label(format: &str) -> &'static str {
match format {
"XLSX" => "Excel 工作簿",
"CSV" => "CSV 表格",
"TSV" => "TSV 表格",
"HOLOLAKE_NATIVE" => "HoloLake 原生表格",
_ => "表格文件",
}
}
fn protect_spreadsheet_formula(cell: &str) -> String {
if cell
.trim_start()
.chars()
.next()
.is_some_and(|character| matches!(character, '=' | '+' | '-' | '@'))
{
format!("'{cell}")
} else {
cell.to_string()
}
}
fn bounded_title(value: &str) -> Result<String, String> {
let title = value.trim();
if title.is_empty()
|| title.len() > MAX_TITLE_BYTES
|| title
.chars()
.any(|character| matches!(character, '\0' | '\r' | '\n'))
{
Err("HOLOLAKE_EDUCATION_IMPORT_TITLE_INVALID".into())
} else {
Ok(title.to_string())
}
}
fn workbook_table_title(stem: &str, sheet_name: &str) -> String {
let candidate = format!("{stem} · {sheet_name}");
if candidate.len() <= MAX_TITLE_BYTES {
candidate
} else if sheet_name.len() <= MAX_TITLE_BYTES {
sheet_name.to_string()
} else {
"导入的教育表格".into()
}
}
fn file_stem(path: &Path) -> String {
path.file_stem()
.and_then(|value| value.to_str())
.filter(|value| !value.trim().is_empty())
.unwrap_or("导入的教育表格")
.to_string()
}
fn safe_filename(value: &str) -> String {
let name = value
.chars()
.map(|character| {
if matches!(
character,
'/' | '\\' | ':' | '*' | '?' | '"' | '<' | '>' | '|'
) || character.is_control()
{
'_'
} else {
character
}
})
.collect::<String>();
let name = name.trim().trim_matches('.');
if name.is_empty() {
"教育表格".into()
} else {
name.to_string()
}
}
fn hex_digest(bytes: &[u8]) -> String {
digest(&SHA256, bytes)
.as_ref()
.iter()
.map(|byte| format!("{byte:02x}"))
.collect()
}
fn export_xlsx_error(error: rust_xlsxwriter::XlsxError) -> String {
format!("HOLOLAKE_EDUCATION_EXPORT_XLSX_FAILED: {error}")
}
fn export_csv_error(error: csv::Error) -> String {
format!("HOLOLAKE_EDUCATION_EXPORT_DELIMITED_FAILED: {error}")
}
fn now_ms() -> u128 {
SystemTime::now()
.duration_since(UNIX_EPOCH)
.unwrap_or_default()
.as_millis()
}
#[cfg(test)]
mod tests {
use super::*;
use tempfile::tempdir;
#[test]
fn csv_round_trip_preserves_chinese_and_blocks_formula_injection() {
let directory = tempdir().unwrap();
let source = directory.path().join("学员.csv");
fs::write(&source, "姓名,状态\n冰朔,连载中\n危险,=cmd()\n").unwrap();
let database = directory.path().join("education.sqlite3");
let receipt = import_tables_from_path(&database, &source).unwrap();
assert_eq!(receipt.imported_tables.len(), 1);
let imported = crate::education_workspace::read_table_at(
&database,
&receipt.imported_tables[0].table_id,
)
.unwrap();
assert_eq!(imported.rows[0].cells, vec!["冰朔", "连载中"]);
let export = directory.path().join("export.csv");
export_table_to_path(
&export,
"CSV",
EducationTableExportDraft {
table_id: imported.table_id,
title: imported.title,
columns: imported.columns,
rows: imported.rows,
revision: imported.revision,
},
)
.unwrap();
let exported = fs::read_to_string(export).unwrap();
assert!(exported.contains("危险,'=cmd()"));
}
#[test]
fn xlsx_import_uses_dense_header_and_exports_valid_workbook() {
let directory = tempdir().unwrap();
let source = directory.path().join("稿费.xlsx");
let mut workbook = Workbook::new();
let worksheet = workbook.add_worksheet();
worksheet.write_string(0, 0, "稿费汇总").unwrap();
worksheet.write_string(2, 0, "作品").unwrap();
worksheet.write_string(2, 1, "月份").unwrap();
worksheet.write_string(2, 2, "金额").unwrap();
worksheet.write_string(3, 0, "测试作品").unwrap();
worksheet.write_string(3, 1, "2025-10").unwrap();
worksheet.write_number(3, 2, 1234.5).unwrap();
workbook.save(&source).unwrap();
let database = directory.path().join("education.sqlite3");
let receipt = import_tables_from_path(&database, &source).unwrap();
assert_eq!(receipt.imported_tables[0].leading_rows_ignored, 2);
assert_eq!(receipt.imported_tables[0].column_count, 3);
assert_eq!(receipt.imported_tables[0].row_count, 1);
assert_eq!(receipt.content_profile.pages[0].focus_title, "稿费收入");
assert_eq!(
receipt.content_profile.pages[0]
.primary_measure_column
.as_deref(),
Some("金额")
);
let imported = crate::education_workspace::read_table_at(
&database,
&receipt.imported_tables[0].table_id,
)
.unwrap();
let target = directory.path().join("roundtrip.xlsx");
export_table_to_path(
&target,
"XLSX",
EducationTableExportDraft {
table_id: imported.table_id,
title: imported.title,
columns: imported.columns,
rows: imported.rows,
revision: imported.revision,
},
)
.unwrap();
let mut roundtrip = open_workbook_auto(target).unwrap();
let range = roundtrip.worksheet_range("HoloLake 数据").unwrap();
assert_eq!(range.get_value((0, 0)).unwrap().to_string(), "作品");
assert_eq!(range.get_value((1, 2)).unwrap().to_string(), "1234.5");
}
#[test]
fn oversized_sheet_fails_instead_of_truncating() {
let mut matrix = vec![(0..31).map(|index| format!("字段 {index}")).collect()];
matrix.push((0..31).map(|_| "".to_string()).collect());
assert_eq!(
matrix_to_native_table(matrix, "超限", false).unwrap_err(),
"HOLOLAKE_EDUCATION_IMPORT_COLUMN_LIMIT_EXCEEDED"
);
}
#[test]
fn ambiguous_table_keeps_the_focus_as_a_human_choice() {
let table = ImportedEducationTable {
title: "杂项记录".into(),
columns: vec![
EducationTableColumn {
column_id: "COL-A".into(),
title: "内容甲".into(),
},
EducationTableColumn {
column_id: "COL-B".into(),
title: "内容乙".into(),
},
],
rows: vec![EducationTableRow {
row_id: "ROW-A".into(),
cells: vec!["".into(), "".into()],
}],
};
let focus = infer_semantic_focus(&table);
assert_eq!(focus.state, "NEEDS_HUMAN_SELECTION");
assert_eq!(focus.primary_measure_column, None);
}
}