From 833d7ce4cc549382aef0f2e8675f3be2e73ce204 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 29 Sep 2026 23:11:12 +0800 Subject: [PATCH 1/3] perf(pdf): batch table scripts and geometry evidence --- rust/docvortex-core/src/geometry_risk.rs | 1 + rust/docvortex-core/src/lib.rs | 2 +- rust/docvortex-python/src/geometry_risk.rs | 45 +++- rust/docvortex-python/src/lib.rs | 3 + rust/docvortex-python/src/snapshot.rs | 174 ++++++++++++++- src/docvortex/_compute_backend.py | 8 +- .../analyzers/native/pdf/char_geometry.py | 81 +++++-- .../analyzers/native/pdf/pipeline.py | 19 +- .../analyzers/native/pdf/table_annotations.py | 70 +++--- .../native/pdf/table_materialization.py | 21 +- .../analyzers/native/pdf/table_rules.py | 7 + .../analyzers/native/pdf/table_text_styles.py | 206 +++++++++++++++++- tests/test_native_parity.py | 5 +- tests/test_native_round4.py | 51 +++++ tests/test_native_text_snapshot.py | 49 +++++ 15 files changed, 677 insertions(+), 65 deletions(-) diff --git a/rust/docvortex-core/src/geometry_risk.rs b/rust/docvortex-core/src/geometry_risk.rs index 3f9ed903..04b58c2a 100644 --- a/rust/docvortex-core/src/geometry_risk.rs +++ b/rust/docvortex-core/src/geometry_risk.rs @@ -8,6 +8,7 @@ use crate::{ }; pub type Entry = (usize, Box4, Box4, Size, f64); +pub type RunKey = (String, u64, i32, i32, i32, String); /// 保留线性插值分位数,不复用其他模块的取整采样规则。 pub(crate) fn quantile(mut values: Vec, fraction: f64) -> f64 { diff --git a/rust/docvortex-core/src/lib.rs b/rust/docvortex-core/src/lib.rs index f850e557..1815f456 100644 --- a/rust/docvortex-core/src/lib.rs +++ b/rust/docvortex-core/src/lib.rs @@ -1,6 +1,6 @@ //! 不访问 Python 对象或 PDFium 的单线程批量计算内核。 -pub const PROTOCOL_VERSION: u32 = 27; +pub const PROTOCOL_VERSION: u32 = 28; pub mod columns; pub mod note_index; pub mod row_geometry; diff --git a/rust/docvortex-python/src/geometry_risk.rs b/rust/docvortex-python/src/geometry_risk.rs index 06590e1e..8aded5eb 100644 --- a/rust/docvortex-python/src/geometry_risk.rs +++ b/rust/docvortex-python/src/geometry_risk.rs @@ -1,6 +1,36 @@ //! 文档风险累积器绑定:只在整行输入与最终结论边界接触 Python。 -use docvortex_core::geometry_risk::{Entry, Risk}; +use docvortex_core::geometry_risk::{Entry, Risk, RunKey}; use pyo3::prelude::*; +use std::collections::HashMap; + +#[pyclass(module = "docvortex._native")] +pub(super) struct NativeGeometryRuns { + values: HashMap, +} + +#[pymethods] +impl NativeGeometryRuns { + /// 创建全文共享的 run 编号表,保证跨页同 key 继续累计。 + #[new] + fn new() -> Self { + Self { + values: HashMap::new(), + } + } + + /// 报告当前不同 run 数,仅用于测试和诊断。 + fn __len__(&self) -> usize { + self.values.len() + } +} + +impl NativeGeometryRuns { + /// 为完全相同的字体/角度/文字 key 分配稳定编号。 + pub(super) fn intern(&mut self, key: RunKey) -> usize { + let next = self.values.len(); + *self.values.entry(key).or_insert(next) + } +} #[pyclass(module = "docvortex._native")] pub(super) struct NativeGeometryRisk { @@ -30,6 +60,19 @@ impl NativeGeometryRisk { py.detach(|| self.state.add_line(page, source, height, skip_y, entries)) } + /// 供页面级自有 evidence 在 Rust 内完成 run 编号后连续累积风险。 + pub(super) fn add_prepared( + &mut self, + py: Python<'_>, + page: usize, + source: i64, + height: f64, + skip_y: bool, + entries: Vec, + ) -> Option { + py.detach(|| self.state.add_line(page, source, height, skip_y, entries)) + } + /// 返回布局与样式风险,不物化字符或字体字典。 fn finish(&self, py: Python<'_>) -> (bool, bool) { py.detach(|| self.state.finish()) diff --git a/rust/docvortex-python/src/lib.rs b/rust/docvortex-python/src/lib.rs index 83ac342c..0594163c 100644 --- a/rust/docvortex-python/src/lib.rs +++ b/rust/docvortex-python/src/lib.rs @@ -34,6 +34,7 @@ fn _native(module: &Bound<'_, PyModule>) -> PyResult<()> { module.add_class::()?; module.add_class::()?; module.add_class::()?; + module.add_class::()?; module.add_class::()?; module.add_function(wrap_pyfunction!( geometry_runs::build_geometry_style, @@ -44,6 +45,7 @@ fn _native(module: &Bound<'_, PyModule>) -> PyResult<()> { module )?)?; module.add_class::()?; + module.add_class::()?; module.add_class::()?; module.add_function(wrap_pyfunction!( classification::read_pdfium_classification, @@ -59,6 +61,7 @@ fn _native(module: &Bound<'_, PyModule>) -> PyResult<()> { module )?)?; module.add_function(wrap_pyfunction!(snapshot::text_snapshot_stats, module)?)?; + module.add_function(wrap_pyfunction!(snapshot::geometry_evidence_stats, module)?)?; module.add_function(wrap_pyfunction!( snapshot::read_pdfium_text_snapshot, module diff --git a/rust/docvortex-python/src/snapshot.rs b/rust/docvortex-python/src/snapshot.rs index 5d0fa949..4c8bb034 100644 --- a/rust/docvortex-python/src/snapshot.rs +++ b/rust/docvortex-python/src/snapshot.rs @@ -1,6 +1,9 @@ //! 同库原始读取直接进入自有 canonical 快照,只在兼容边界构造 Python 字符。 +use docvortex_core::geometry_risk::{Entry, RunKey}; use docvortex_core::{ - extraction, text_assignment, text_content, + extraction, + geometry::{self, SourceRow}, + text_assignment, text_content, text_pipeline::{self, TextChar}, text_snapshot::{ self as snapshot, Angle, Character, Font, InputCharacter, Properties, SnapshotError, @@ -157,6 +160,44 @@ pub struct NativeTextSnapshot { raw_count: usize, } +struct GeometryRecord { + source: SourceRow, + family: String, + font_size: f64, + flags: i32, + weight: i32, + script: String, + anchor: bool, +} + +#[pyclass(frozen, module = "docvortex._native")] +pub struct NativeGeometryEvidence { + records: Vec, +} + +static GEOMETRY_EVIDENCE_PREPARES: AtomicU64 = AtomicU64::new(0); +static GEOMETRY_EVIDENCE_LINES: AtomicU64 = AtomicU64::new(0); +static GEOMETRY_EVIDENCE_FALLBACKS: AtomicU64 = AtomicU64::new(0); + +/// Python float 字典语义把正负零视为同键;普通有限值按位型保持区分。 +fn geometry_number_key(value: f64) -> u64 { + if value == 0.0 { + 0 + } else { + value.to_bits() + } +} + +/// 报告页面级 geometry evidence 准备、风险行命中和明确回退次数。 +#[pyfunction] +pub fn geometry_evidence_stats() -> (u64, u64, u64) { + ( + GEOMETRY_EVIDENCE_PREPARES.load(Ordering::Relaxed), + GEOMETRY_EVIDENCE_LINES.load(Ordering::Relaxed), + GEOMETRY_EVIDENCE_FALLBACKS.load(Ordering::Relaxed), + ) +} + /// 由 Rust 短暂持有并释放 textpage,字符规范化仍复用同一个自有快照构建入口。 #[pyfunction] pub fn read_pdfium_page_text_snapshot( @@ -418,6 +459,69 @@ pub fn read_pdfium_text_snapshot( } impl NativeTextSnapshot { + /// 一次准备页面级风险输入;解释器 Unicode/round 语义只在不同文本和字体上调用。 + fn prepare_geometry_evidence_impl( + &self, + py: Python<'_>, + ) -> PyResult> { + let char_geometry = py.import("docvortex.analyzers.native.pdf.char_geometry")?; + let script_group = char_geometry.getattr("_script_group")?; + let normalize = char_geometry.getattr("_normalized_font_family")?; + let round = py.import("builtins")?.getattr("round")?; + let mut groups: HashMap = HashMap::new(); + let mut families: HashMap = HashMap::new(); + let mut font_metadata: HashMap = HashMap::new(); + let mut records = Vec::with_capacity(self.data.chars.len()); + for ch in &self.data.chars { + let font = &self.data.fonts[ch.font]; + if !font.size.is_finite() { + GEOMETRY_EVIDENCE_FALLBACKS.fetch_add(1, Ordering::Relaxed); + return Ok(None); + } + let metadata = if let Some(value) = font_metadata.get(&ch.font) { + value.clone() + } else { + let raw_name = if font.name.is_empty() { + "".to_owned() + } else { + font.name.clone() + }; + let family = if let Some(value) = families.get(&raw_name) { + value.clone() + } else { + let value: String = normalize.call1((&raw_name,))?.extract()?; + families.insert(raw_name.clone(), value.clone()); + value + }; + let rounded_size: f64 = + round.call1((font.size * 4.0,))?.extract::()? as f64 / 4.0; + let weight: i64 = round.call1((f64::from(font.weight) / 100.0,))?.extract()?; + let value = (family, rounded_size, font.flags, weight as i32); + font_metadata.insert(ch.font, value.clone()); + value + }; + let (script, anchor) = if let Some(value) = groups.get(&ch.text) { + value.clone() + } else { + let group: String = script_group.call1((&ch.text,))?.extract()?; + let value = (group.clone(), group != "other"); + groups.insert(ch.text.clone(), value.clone()); + value + }; + records.push(GeometryRecord { + source: (Some(ch.bbox), ch.loose, ch.tight, ch.origin, ch.rotation), + family: metadata.0, + font_size: metadata.1, + flags: metadata.2, + weight: metadata.3, + script, + anchor, + }); + } + GEOMETRY_EVIDENCE_PREPARES.fetch_add(1, Ordering::Relaxed); + Ok(Some(NativeGeometryEvidence { records })) + } + /// 一次物化中的字体字典共享,字符和 Bbox 独立;不会在 Rust 快照缓存 Python 可变对象。 fn geometry<'py>(&self, py: Python<'py>) -> PyResult<(Bound<'py, PyAny>, Bound<'py, PyList>)> { let bbox_type = py @@ -531,6 +635,14 @@ impl NativeTextSnapshot { #[pymethods] impl NativeTextSnapshot { + /// 一次准备全文风险使用的页面级 geometry evidence。 + fn prepare_geometry_evidence( + &self, + py: Python<'_>, + ) -> PyResult> { + self.prepare_geometry_evidence_impl(py) + } + /// 兼容属性每次创建独立输出,快照在页面关闭后仍能使用。 fn materialize_geometry<'py>(&self, py: Python<'py>) -> PyResult> { self.geometry(py).map(|(geometry, _)| geometry) @@ -1025,6 +1137,66 @@ impl NativeTextSnapshot { } } +#[pymethods] +impl NativeGeometryEvidence { + /// 把同源行成员一次转换并累积到全文风险器;None 表示本行选择参考路径。 + #[pyo3(signature = (risk, runs, page, source, height, skip_y, indices, size, angle))] + fn add_line( + &self, + py: Python<'_>, + mut risk: pyo3::PyRefMut<'_, super::geometry_risk::NativeGeometryRisk>, + mut runs: pyo3::PyRefMut<'_, super::geometry_risk::NativeGeometryRuns>, + page: usize, + source: i64, + height: f64, + skip_y: bool, + indices: Vec, + size: [f64; 2], + angle: i32, + ) -> Option { + if indices.iter().any(|&index| index >= self.records.len()) + || !size.iter().all(|v| v.is_finite()) + { + GEOMETRY_EVIDENCE_FALLBACKS.fetch_add(1, Ordering::Relaxed); + return None; + } + let rows: Vec<_> = indices + .iter() + .map(|&index| self.records[index].source) + .collect(); + let prepared = py.detach(|| geometry::source_rows(rows, size, angle)); + let mut entries: Vec = Vec::with_capacity(indices.len()); + for (record, prepared) in indices + .iter() + .map(|&index| &self.records[index]) + .zip(prepared) + { + let Some(row) = prepared else { + continue; + }; + if !record.anchor || !record.font_size.is_finite() { + continue; + } + let key: RunKey = ( + record.family.clone(), + geometry_number_key(record.font_size), + record.flags, + record.weight, + angle, + record.script.clone(), + ); + let run = runs.intern(key); + entries.push((run, row.3, row.4, row.5, record.font_size)); + } + GEOMETRY_EVIDENCE_LINES.fetch_add(1, Ordering::Relaxed); + let result = risk.add_prepared(py, page, source, height, skip_y, entries); + if result.is_none() { + GEOMETRY_EVIDENCE_FALLBACKS.fetch_add(1, Ordering::Relaxed); + } + result + } +} + /// 累计剖析中的视觉证据阶段耗时,单位为纳秒。 fn record_visual_stage_stats(values: [std::time::Duration; 6]) { let mut slot = VISUAL_STAGE_NS diff --git a/src/docvortex/_compute_backend.py b/src/docvortex/_compute_backend.py index 13dc9546..744cc6b3 100644 --- a/src/docvortex/_compute_backend.py +++ b/src/docvortex/_compute_backend.py @@ -10,7 +10,7 @@ from pathlib import Path from types import ModuleType -_PROTOCOL_VERSION = 27 +_PROTOCOL_VERSION = 28 _SELECTED_MODE = None _LOAD_FAILURE = None @@ -59,6 +59,12 @@ def backend_info() -> dict[str, str | int | None]: "native_classification_snapshot_calls": native.classification_snapshot_stats() if native is not None else 0, "native_script_snapshot_batches": native.script_snapshot_stats() if native is not None else 0, "native_span_assignment_calls": span_calls, + "native_table_script_cell_batches": table_scripts.table_script_stats() + if (table_scripts := sys.modules.get("docvortex.analyzers.native.pdf.table_text_styles")) is not None + else (0, 0, 0), + "native_geometry_evidence_stats": native.geometry_evidence_stats() + if native is not None and hasattr(native, "geometry_evidence_stats") + else (0, 0, 0), "native_span_assignment_unsupported": span_unsupported, "backend": "rust" if native is not None else "python", "protocol": getattr(native, "PROTOCOL_VERSION", None), diff --git a/src/docvortex/analyzers/native/pdf/char_geometry.py b/src/docvortex/analyzers/native/pdf/char_geometry.py index 95b0268e..d7800f9c 100644 --- a/src/docvortex/analyzers/native/pdf/char_geometry.py +++ b/src/docvortex/analyzers/native/pdf/char_geometry.py @@ -693,7 +693,7 @@ def _anchor_pair_statistics(rows, positive_source=False): return _anchor_pair_statistics_python(rows, positive_source) -def _document_requires_full_geometry(lines_by_page, geometries, page_sizes): +def _document_requires_full_geometry(lines_by_page, geometries, page_sizes, *, owned_geometry_inputs=None): """标准数值路径连续累积全文风险,自定义规则明确使用 Python 参考实现。""" from ...._compute_backend import get_native @@ -704,7 +704,12 @@ def _document_requires_full_geometry(lines_by_page, geometries, page_sizes): or any(globals()[name] is not value for name, value in _RISK_FUNCTIONS.items()) ): return _document_requires_full_geometry_python(lines_by_page, geometries, page_sizes) + if owned_geometry_inputs is not None and ( + type(owned_geometry_inputs) is not list or len(owned_geometry_inputs) != len(lines_by_page) + ): + owned_geometry_inputs = None state = native.NativeGeometryRisk() + run_table = native.NativeGeometryRuns() if hasattr(native, "NativeGeometryRuns") else None run_ids = {} for page_index, (lines, geometry, page_size) in enumerate(zip(lines_by_page, geometries, page_sizes, strict=True)): for line in lines: @@ -720,25 +725,46 @@ def _document_requires_full_geometry(lines_by_page, geometries, page_sizes): ) ): return _document_requires_full_geometry_python(lines_by_page, geometries, page_sizes) - font_metadata = _ReadOnlyFontCache() - entries = [] - for _position, text, _char_idx, char, prepared in _prepared_line_geometry( - line, geometry, page_size, anchors_only=True - ): - key, size = _font_run_key(char, line.angle, text, font_metadata) - entries.append((run_ids.setdefault(key, len(run_ids)), prepared[3], prepared[4], prepared[5], size)) - early = state.add_line( - page_index, - line.source_index, - line.effective_height, - bool( - line.angle != 0 - or line.formula_candidate_only - or line.restored_inline_cluster - or line.compact_formula_cluster - ), - entries, + skip_y = bool( + line.angle != 0 or line.formula_candidate_only or line.restored_inline_cluster or line.compact_formula_cluster ) + owned = owned_geometry_inputs[page_index] if owned_geometry_inputs is not None else None + owned_indices = None + if owned is not None and run_table is not None: + identities = owned[1] + if type(line.chars) is not list: + owned_indices = False + else: + owned_indices = [identities.get(id(char)) for char in line.chars] + if any(index is None for index in owned_indices): + owned_indices = False + if owned_indices is not False and owned is not None and run_table is not None: + early = owned[0].add_line( + state, + run_table, + page_index, + line.source_index, + line.effective_height, + skip_y, + owned_indices, + page_size, + line.angle, + ) + else: + font_metadata = _ReadOnlyFontCache() + entries = [] + for _position, text, _char_idx, char, prepared in _prepared_line_geometry( + line, geometry, page_size, anchors_only=True + ): + key, size = _font_run_key(char, line.angle, text, font_metadata) + entries.append((run_ids.setdefault(key, len(run_ids)), prepared[3], prepared[4], prepared[5], size)) + early = state.add_line( + page_index, + line.source_index, + line.effective_height, + skip_y, + entries, + ) if early is None: return _document_requires_full_geometry_python(lines_by_page, geometries, page_sizes) if early: @@ -2050,17 +2076,24 @@ def build_document_geometry_plan( lines_by_page: list[list[_LineItem]], geometries: list[PDFPageTextGeometry], page_sizes: list[tuple[float, float]], + *, + owned_geometry_inputs: list[Any] | None = None, ) -> DocumentGeometryPlan: """构建文档级 X 修复、Y trim 与 split shadow 计划。""" plan = DocumentGeometryPlan() if not any(geometry.tight_bboxes and geometry.origins for geometry in geometries): return plan - risk = _document_requires_full_geometry( - lines_by_page, - geometries, - page_sizes, - ) + if owned_geometry_inputs is None: + # 保持既有单页/测试替身签名;只有主链路显式提供页面级 evidence 才传私有参数。 + risk = _document_requires_full_geometry(lines_by_page, geometries, page_sizes) + else: + risk = _document_requires_full_geometry( + lines_by_page, + geometries, + page_sizes, + owned_geometry_inputs=owned_geometry_inputs, + ) if not risk.any: for geometry in geometries: geometry.loose_bboxes.clear() diff --git a/src/docvortex/analyzers/native/pdf/pipeline.py b/src/docvortex/analyzers/native/pdf/pipeline.py index 826a2fc7..61f20184 100644 --- a/src/docvortex/analyzers/native/pdf/pipeline.py +++ b/src/docvortex/analyzers/native/pdf/pipeline.py @@ -314,9 +314,20 @@ class _DocumentSources: page_style_lines: list[list[PDFTextStyleLine]] page_link_lines: list[list[PDFTextLinkLine]] page_owned_scripts: list[Any] = field(default_factory=list) + page_geometry_evidence: list[Any] = field(default_factory=list) page_spacing_lines: list[list[Any]] = field(default_factory=list) +def _prepare_owned_geometry_evidence(owner, chars): + """一次准备全文风险所需的页面级 Rust 几何记录和原字符身份表。""" + if not hasattr(owner, "prepare_geometry_evidence"): + return None + evidence = owner.prepare_geometry_evidence() + if evidence is None: + return None + return evidence, {id(char): index for index, char in enumerate(chars)} + + def _collect_document_sources(pdf_doc: NativePdfSource) -> _DocumentSources: """逐页收集原生证据,局部快照与字符引用在收集阶段退出时释放。""" @@ -329,6 +340,7 @@ def _collect_document_sources(pdf_doc: NativePdfSource) -> _DocumentSources: page_style_lines: list[list[PDFTextStyleLine]] = [] page_link_lines: list[list[PDFTextLinkLine]] = [] page_owned_scripts = [] + page_geometry_evidence = [] page_spacing_lines = [] for page_idx in range(pdf_doc.page_count): snapshot = pdf_doc._extract_native_page(page_idx) @@ -349,8 +361,9 @@ def _collect_document_sources(pdf_doc: NativePdfSource) -> _DocumentSources: space_before = set(native_text.tight_space_indices()) if native_text is not None else None page_spacing_lines.append(prepare_spacing_lines(lines, space_before)) chars = text_geometry.chars - # 只保留纯 Rust 脚本记录与本页身份映射,不延长 PDFium 页面或文档句柄生命周期。 + # 只保留纯 Rust 脚本/几何记录与本页身份映射,不延长 PDFium 页面或文档句柄生命周期。 page_owned_scripts.append(_prepare_owned_script_evidence(native_text, chars) if native_text is not None else None) + page_geometry_evidence.append(_prepare_owned_geometry_evidence(native_text, chars) if native_text is not None else None) drawing_lines = _coerce_pdf_drawing_lines(snapshot.drawing_lines) lines, decorative_rules = _extract_decorative_text_rules( lines, @@ -400,6 +413,7 @@ def _collect_document_sources(pdf_doc: NativePdfSource) -> _DocumentSources: page_style_lines, page_link_lines, page_owned_scripts, + page_geometry_evidence, page_spacing_lines, ) @@ -415,6 +429,7 @@ def _prepare_document_sources( [source.lines for source in sources.page_sources], sources.page_text_geometries, sources.page_sizes, + owned_geometry_inputs=sources.page_geometry_evidence, ) for page_index, source in enumerate(sources.page_sources): # 容器认领前只允许 X 修复,表格和图形认领后再启用 Y trim。 @@ -430,6 +445,7 @@ def _prepare_document_sources( owned_scripts = sources.page_owned_scripts or [None] * len(sources.page_sources) pending = deque(zip(sources.page_sources, sources.page_text_geometries, owned_scripts, strict=True)) sources.page_owned_scripts.clear() + sources.page_geometry_evidence.clear() del owned_scripts sources.page_sources.clear() sources.page_text_geometries.clear() @@ -681,6 +697,7 @@ def _prepare_page_source( candidates, tight_bboxes=tight_bboxes, origins=origins, + _owned_script_inputs=_owned_script_inputs, ) claimed_line_indices.update(caption_graphic_claims) claimed_line_indices.update(claimed_rule_code_line_indices) diff --git a/src/docvortex/analyzers/native/pdf/table_annotations.py b/src/docvortex/analyzers/native/pdf/table_annotations.py index 0de66b78..ecb6232c 100644 --- a/src/docvortex/analyzers/native/pdf/table_annotations.py +++ b/src/docvortex/analyzers/native/pdf/table_annotations.py @@ -13,6 +13,8 @@ from itertools import islice from typing import Any, Literal from ....schema import BBox +from ....document.pdf.text._contracts import Bbox as CharBbox +from ...._compute_backend import get_native from .models import _LineItem, _TableAnnotation, _TableCandidate, _VisualRow from .geometry import ( _bbox_axis_overlap_ratio, @@ -160,24 +162,14 @@ def prepared_marker_line(self, line: _LineItem, page_size: tuple[float, float], return self.marker_prepared.prepare(line, page_size, angle) -def _prepare_table_core_rows(rows, lines, metrics, marker_prepared: _PreparedMarkerCache | None = None): - """一次校验并打包走廊成员;特殊来源和对象仍交给原集合路径。""" - if metrics.native is None or len({id(row) for row in rows}) != len(rows): - return None - members = [] - source_rows = {} - for position, row in enumerate(rows): - indices = [] - for fragment in row.fragments: - index = fragment.line_index - if type(index) is not int or not -(2**63) <= index < 2**63: - return None - indices.append(index) - source_rows.setdefault(index, []).append(position) - members.append(indices) - from ....document.pdf.text import Bbox +def _prepare_marker_line_context(lines): + """同一候选上下文只校验一次 marker 输入,并冻结来源行索引。""" + from ...._compute_backend import get_native - marker_safe = all( + if get_native() is None: + return False, {} + source_lines: dict[int, list[_LineItem]] = {} + safe = all( type(line) is _LineItem and type(line.source_index) is int and type(line.text) is str @@ -188,8 +180,8 @@ def _prepare_table_core_rows(rows, lines, metrics, marker_prepared: _PreparedMar and ( char.get("bbox") is None or ( - type(char.get("bbox")) in (tuple, list, Bbox) - and len(char["bbox"].bbox if type(char["bbox"]) is Bbox else char["bbox"]) == 4 + type(char.get("bbox")) in (tuple, list, CharBbox) + and len(char["bbox"].bbox if type(char["bbox"]) is CharBbox else char["bbox"]) == 4 and all( (type(value) is float and math.isfinite(value)) or (type(value) is int and -(2**53) <= value <= 2**53) for value in char["bbox"] @@ -200,13 +192,37 @@ def _prepare_table_core_rows(rows, lines, metrics, marker_prepared: _PreparedMar ) for line in lines ) - from ...._compute_backend import get_native - - source_lines = {} - if marker_safe: + if safe: for line in lines: - if line.source_index in source_rows: - source_lines.setdefault(line.source_index, []).append(line) + source_lines.setdefault(line.source_index, []).append(line) + return safe, source_lines + + +def _prepare_table_core_rows( + rows, + lines, + metrics, + marker_prepared: _PreparedMarkerCache | None = None, + *, + marker_safe: bool | None = None, + source_lines: dict[int, list[_LineItem]] | None = None, +): + """一次校验并打包走廊成员;特殊来源和对象仍交给原集合路径。""" + if metrics.native is None or len({id(row) for row in rows}) != len(rows): + return None + members = [] + source_rows = {} + for position, row in enumerate(rows): + indices = [] + for fragment in row.fragments: + index = fragment.line_index + if type(index) is not int or not -(2**63) <= index < 2**63: + return None + indices.append(index) + source_rows.setdefault(index, []).append(position) + members.append(indices) + if marker_safe is None or source_lines is None: + marker_safe, source_lines = _prepare_marker_line_context(lines) geometry = None if all( type(row.bbox) in (tuple, list) @@ -713,8 +729,6 @@ def _prepare_marker_line( ) -> tuple[list[tuple[str, BBox]], tuple[int, tuple[str, ...]]] | None: """仅为普通来源行准备与具体标记无关的局部字形及紧凑 token。""" - from ....document.pdf.text import Bbox - if ( type(line) is not _LineItem or type(line.text) is not str @@ -730,7 +744,7 @@ def _prepare_marker_line( if type(char) is not dict or type(char.get("char")) not in (str, type(None)): return None value = char.get("bbox") - raw = value.bbox if type(value) is Bbox else value + raw = value.bbox if type(value) is CharBbox else value if raw is not None and ( type(raw) not in (tuple, list) or len(raw) != 4 diff --git a/src/docvortex/analyzers/native/pdf/table_materialization.py b/src/docvortex/analyzers/native/pdf/table_materialization.py index 0b504a2c..0d2c565f 100644 --- a/src/docvortex/analyzers/native/pdf/table_materialization.py +++ b/src/docvortex/analyzers/native/pdf/table_materialization.py @@ -50,6 +50,8 @@ def _recover_native_table_html( rectangles: tuple[NativeTableRectangle, ...] | None = None, tight_bboxes: dict[int, BBox] | None = None, origins: dict[int, tuple[float, float]] | None = None, + *, + _owned_script_inputs: tuple[Any, dict[int, int]] | None = None, ) -> str: """使用共享原生字符与绘图原语恢复高置信表格 HTML。""" @@ -61,7 +63,12 @@ def _recover_native_table_html( drawing_lines=(drawing_lines if drawing_lines is not None else coerce_native_table_rules(source.drawing_lines)), rectangles=(rectangles if rectangles is not None else coerce_native_table_rectangles(source.path_infos)), ) - recovered = recover_table_result(table_input, tight_bboxes or {}, origins or {}) + recovered = recover_table_result( + table_input, + tight_bboxes or {}, + origins or {}, + _owned_script_inputs=_owned_script_inputs, + ) return recovered[0] if recovered is not None else "" @@ -71,6 +78,7 @@ def _materialize_table_blocks( *, tight_bboxes: dict[int, BBox] | None = None, origins: dict[int, tuple[float, float]] | None = None, + _owned_script_inputs: tuple[Any, dict[int, int]] | None = None, ) -> tuple[list[dict[str, Any]], list[dict[str, Any]], set[int]]: """原子物化表体及其独立注释,仅认领整组成功输出的文本行。""" @@ -109,6 +117,7 @@ def _materialize_table_blocks( native_rectangles, tight_bboxes, origins, + _owned_script_inputs=_owned_script_inputs, ) except Exception as exc: logger.warning( @@ -541,6 +550,8 @@ def recover_table_result( table_input: NativeTableInput, tight_bboxes: dict[int, BBox], origins: dict[int, tuple[float, float]], + *, + _owned_script_inputs: tuple[Any, dict[int, int]] | None = None, ) -> tuple[str, NativeTableResult] | None: """统一表格恢复和上下标物化,保留调用方接受或回退的决策权。""" try: @@ -549,5 +560,11 @@ def recover_table_result( raise PDFTableRecoveryError(str(error)) from error if result is None: return None - content = render_native_table_html_with_scripts(result, table_input, tight_bboxes, origins) + content = render_native_table_html_with_scripts( + result, + table_input, + tight_bboxes, + origins, + _owned_script_inputs=_owned_script_inputs, + ) return content, result diff --git a/src/docvortex/analyzers/native/pdf/table_rules.py b/src/docvortex/analyzers/native/pdf/table_rules.py index e6d4608b..66a7426d 100644 --- a/src/docvortex/analyzers/native/pdf/table_rules.py +++ b/src/docvortex/analyzers/native/pdf/table_rules.py @@ -40,6 +40,7 @@ _prepare_table_note_body_metrics, _prepare_table_note_rows, _prepare_table_core_rows, + _prepare_marker_line_context, ) from .table_rows import _clip_visual_row_to_corridor @@ -133,6 +134,7 @@ class _RuleCandidateContext: grids: list | None = None grid_members: dict = field(default_factory=dict) core_indexes: dict = field(default_factory=dict) + marker_line_context: tuple[bool, dict] | None = None row_bounds: dict = field(default_factory=dict) annotation_geometry: Any = None @@ -512,6 +514,9 @@ def _build_rule_table_candidates( if defer_materialization: band_index = band_indexes[corridor_key] if corridor_key not in context.core_indexes: + if context.marker_line_context is None: + context.marker_line_context = _prepare_marker_line_context(lines) + marker_safe, marker_source_lines = context.marker_line_context # 构建阶段内部行序列冻结为 tuple,允许两个只读索引安全共享切片校验。 corridor_rows = ( band_index.rows if band_index is not None else tuple(item.row for item in corridor_cache[corridor_key]) @@ -521,6 +526,8 @@ def _build_rule_table_candidates( lines, prepared_note_body_metrics, context.marker_prepared, + marker_safe=marker_safe, + source_lines=marker_source_lines, ) core_index = context.core_indexes[corridor_key] interval = core_index.interval(accepted_rows) if core_index is not None else None diff --git a/src/docvortex/analyzers/native/pdf/table_text_styles.py b/src/docvortex/analyzers/native/pdf/table_text_styles.py index 332d0346..0be2987d 100644 --- a/src/docvortex/analyzers/native/pdf/table_text_styles.py +++ b/src/docvortex/analyzers/native/pdf/table_text_styles.py @@ -22,11 +22,26 @@ from ._table_recovery.geometry import page_bbox_to_table_local from ._table_recovery.text import build_cell_text_parts from .geometry import _bbox_union_many, _coerce_bbox -from .inline.scripts import _fraction_member_indices, _prepare_fraction_rules, _script_line_char_roles +from .inline.scripts import ( + _classify_script_runs, + _fraction_member_indices, + _prepare_fraction_rules, + _script_line_char_roles, +) from .line_merging import _merge_overlapping_inline_text_clusters from .models import _LineItem from .native_text import _fill_native_typography +_OWNED_CELL_BATCH_CHAR_LIMIT = 8192 +_owned_cell_batches = 0 +_owned_cell_lines = 0 +_owned_cell_fallbacks = 0 + + +def table_script_stats() -> tuple[int, int, int]: + """报告表格上下标 owned 批次、命中行数和整批回退数。""" + return _owned_cell_batches, _owned_cell_lines, _owned_cell_fallbacks + def _table_char_map(chars: tuple[Char, ...]) -> dict[int, Char]: """按合法 char_idx 建立页面字符查询表,重复索引保留首项。""" @@ -155,6 +170,18 @@ def _drop_shifted_word_prefixes(chars: list[dict[str, Any]], roles: list[ScriptR start = end +def _roles_by_source(chars: list[dict[str, Any]], roles: list[ScriptRole]) -> dict[int, ScriptRole]: + """按来源编号聚合角色;同源冲突沿用原实现降级为 body。""" + output: dict[int, ScriptRole] = {} + for char, role in zip(chars, roles, strict=True): + char_idx = char.get("char_idx") + if not isinstance(char_idx, int) or role == "body": + continue + previous = output.get(char_idx) + output[char_idx] = role if previous in {None, role} else "body" + return {source_index: role for source_index, role in output.items() if role != "body"} + + def _cell_script_roles( glyphs: list[NativeTableGlyph], chars_by_source: dict[int, Char], @@ -163,10 +190,12 @@ def _cell_script_roles( tight_bboxes: dict[int, BBox], origins: dict[int, tuple[float, float]], fraction_members: set[int], + *, + visual_lines: list[_LineItem] | None = None, ) -> dict[int, ScriptRole]: """在 cell 内按正文同款二维公式分段和字符几何返回稳定角色。""" - lines = _cell_visual_lines(glyphs, chars_by_source, page_size, angle) + lines = _cell_visual_lines(glyphs, chars_by_source, page_size, angle) if visual_lines is None else visual_lines if not lines: return {} segmented_lines = _merge_overlapping_inline_text_clusters(lines, page_size, []) @@ -189,6 +218,125 @@ def _cell_script_roles( return {source_index: role for source_index, role in roles_by_source.items() if role != "body"} +def _owned_cell_line_indices(line: _LineItem, identities: dict[int, int]) -> list[int] | None: + """仅普通 0 度、同源且无特殊公式标记的行可进入页面级 owned 批次。""" + if ( + type(line) is not _LineItem + or type(line.chars) is not list + or line.angle != 0 + or line.inline_math_regions + or line.compact_formula_cluster + or line.restored_inline_cluster + ): + return None + indices: list[int] = [] + for char in line.chars: + index = identities.get(id(char)) if type(char) is dict else None + if index is None: + return None + indices.append(index) + return indices + + +def _owned_cell_classify( + pending: list[tuple[tuple[int, int], list[_LineItem], list[dict[str, Any]], list[int]]], + owned_inputs: tuple[Any, dict[int, int]], + page_size: tuple[float, float], + tight_bboxes: dict[int, BBox], + origins: dict[int, tuple[float, float]], +) -> dict[tuple[int, int], dict[int, ScriptRole]]: + """把多个同源 cell 的视觉行合并成有界 owned 批次并保留原精炼规则。""" + global _owned_cell_batches, _owned_cell_lines, _owned_cell_fallbacks + evidence, identities = owned_inputs + output: dict[tuple[int, int], dict[int, ScriptRole]] = {} + cell_lines = {key: lines for key, lines, _chars, _indices in pending} + records = [] + for key, lines, _chars, cell_indices in pending: + offset = 0 + for line in lines: + width = len(line.chars) + records.append((key, line, cell_indices[offset : offset + width])) + offset += width + start = 0 + while start < len(records): + end = start + 1 + char_count = len(records[start][2]) + while end < len(records) and (char_count + len(records[end][2]) <= _OWNED_CELL_BATCH_CHAR_LIMIT or char_count == 0): + char_count += len(records[end][2]) + end += 1 + batch = records[start:end] + flat_indices = [index for _key, _line, indices in batch for index in indices] + offsets = [0] + for _key, _line, indices in batch: + offsets.append(offsets[-1] + len(indices)) + classified = evidence.classify_indices(flat_indices, offsets) + if len(classified) != len(batch) or any( + item is None or len(item) != len(indices) for item, (_key, _line, indices) in zip(classified, batch, strict=True) + ): + fallback_keys = {key for key, _line, _indices in batch} + _owned_cell_fallbacks += sum(key in cell_lines for key in fallback_keys) + for key in fallback_keys: + output[key] = _cell_script_roles_from_lines(cell_lines[key], page_size, tight_bboxes, origins) + start = end + continue + _owned_cell_batches += 1 + _owned_cell_lines += len(batch) + # 表格 cell 与正文行的大二维合并语义不同;原生先给出任何非正文角色时, + # 整 cell 回到既有参考判定,只有确定全正文的结果直接复用 owned 批次。 + fallback_keys = { + key + for (key, _line, _indices), raw_roles in zip(batch, classified, strict=True) + if any(role != 0 for role in raw_roles) + } + for key in fallback_keys: + output.pop(key, None) + _owned_cell_fallbacks += 1 + output[key] = _cell_script_roles_from_lines(cell_lines[key], page_size, tight_bboxes, origins) + for (key, line, _indices), raw_roles in zip(batch, classified, strict=True): + if key in fallback_keys: + continue + roles_by_source = output.setdefault(key, {}) + chars = line.chars + memberships = [None] * len(chars) + roles, _body_counts, _formula_flags = _classify_script_runs( + chars, + tight_bboxes, + origins, + memberships, + preclassified_native=[raw_roles], + ) + _drop_shifted_word_prefixes(chars, roles) + for char, role in zip(chars, roles, strict=True): + char_idx = char.get("char_idx") + if not isinstance(char_idx, int) or role == "body": + continue + previous = roles_by_source.get(char_idx) + roles_by_source[char_idx] = role if previous in {None, role} else "body" + start = end + for key, roles in list(output.items()): + output[key] = {source_index: role for source_index, role in roles.items() if role != "body"} + return output + + +def _cell_script_roles_from_lines( + lines: list[_LineItem], + page_size: tuple[float, float], + tight_bboxes: dict[int, BBox], + origins: dict[int, tuple[float, float]], +) -> dict[int, ScriptRole]: + """以已缓存的 cell 视觉行执行参考路径,避免 owned 不可用时重复组行。""" + return _cell_script_roles( + [], + {}, + page_size, + 0, + tight_bboxes, + origins, + set(), + visual_lines=lines, + ) + + def _render_styled_cell( cell: NativeTableCell, glyphs: list[NativeTableGlyph], @@ -254,6 +402,8 @@ def render_native_table_html_with_scripts( table_input: NativeTableInput, tight_bboxes: dict[int, BBox], origins: dict[int, tuple[float, float]], + *, + _owned_script_inputs: tuple[Any, dict[int, int]] | None = None, ) -> str: """为高置信原生表格恢复上下标,证据不足时返回原始 HTML。""" @@ -267,8 +417,17 @@ def render_native_table_html_with_scripts( # 一张表只建立一次来源索引;保持原字典推导式的后项覆盖语义。 glyph_by_source = {glyph.source_index: glyph for glyph in result.text.glyphs} cell_glyphs = {(cell.row, cell.col): _cell_glyphs(result, cell, glyph_by_source) for cell in result.cells} - cell_roles: dict[tuple[int, int], dict[int, ScriptRole]] = {} - has_missing_space = False + cell_lines = { + (cell.row, cell.col): _cell_visual_lines( + cell_glyphs[(cell.row, cell.col)], + chars_by_source, + table_input.page_size, + table_input.angle, + ) + for cell in result.cells + } + owned_pending: list[tuple[tuple[int, int], list[_LineItem], list[dict[str, Any]], list[int]]] = [] + fallback_cells: list[tuple[tuple[int, int], set[int]]] = [] for cell in result.cells: key = (cell.row, cell.col) glyphs = cell_glyphs[key] @@ -281,17 +440,54 @@ def render_native_table_html_with_scripts( table_input.angle, prepared_fraction_rules, ) + lines = cell_lines[key] + indices = None + if _owned_script_inputs is not None and table_input.angle == 0 and not fraction_members and lines: + # owned 批次必须消费与参考路径完全相同的二维合并后的行。 + segmented = _merge_overlapping_inline_text_clusters(lines, table_input.page_size, []) + packed = [ + (line, line_indices) + for line in segmented + if (line_indices := _owned_cell_line_indices(line, _owned_script_inputs[1])) is not None + ] + if len(packed) == len(segmented): + indices = [index for _line, row_indices in packed for index in row_indices] + chars = [char for line in segmented for char in line.chars] + owned_pending.append((key, segmented, chars, indices)) + if indices is None: + fallback_cells.append((key, fraction_members)) + + owned_roles = ( + _owned_cell_classify( + owned_pending, + _owned_script_inputs, + table_input.page_size, + tight_bboxes, + origins, + ) + if owned_pending + else {} + ) + cell_roles: dict[tuple[int, int], dict[int, ScriptRole]] = dict(owned_roles) + for key, fraction_members in fallback_cells: roles = _cell_script_roles( - glyphs, + cell_glyphs[key], chars_by_source, table_input.page_size, table_input.angle, tight_bboxes, origins, fraction_members, + visual_lines=cell_lines[key], ) if roles: cell_roles[key] = roles + + has_missing_space = False + for cell in result.cells: + key = (cell.row, cell.col) + glyphs = cell_glyphs[key] + roles = cell_roles.get(key, {}) if not has_missing_space: has_missing_space = any( left.visual_row == right.visual_row diff --git a/tests/test_native_parity.py b/tests/test_native_parity.py index d586853f..7c834242 100644 --- a/tests/test_native_parity.py +++ b/tests/test_native_parity.py @@ -31,6 +31,7 @@ def test_native_registration_contract(native): "build_geometry_runs": "(samples, by_line, keys, sample_type, run_type)", "text_snapshot_stats": "()", "visual_evidence_stage_stats": "()", + "geometry_evidence_stats": "()", "classification_snapshot_stats": "()", "read_pdfium_classification": "(addresses, handle, count, cjk_ranges, allowed_controls, private_range, normalize_font)", "script_snapshot_stats": "()", @@ -85,7 +86,9 @@ def test_native_registration_contract(native): "NativeTableMerger": ("docvortex._native", "()"), "NativeStyleDocument": ("docvortex._native", "(with_samples=False)"), "NativeGeometryRisk": ("docvortex._native", "()"), + "NativeGeometryRuns": ("docvortex._native", "()"), "NativeTextSnapshot": ("docvortex._native", None), + "NativeGeometryEvidence": ("docvortex._native", None), "NativeClassificationSnapshot": ("docvortex._native", None), "NativeScriptEvidence": ("docvortex._native", None), "NativeFontProvider": ( @@ -107,7 +110,7 @@ def test_native_registration_contract(native): "TableNoteMetrics": ("builtins", "(items)"), "TableRowGeometry": ("builtins", "(boxes)"), } - constants = {"PROTOCOL_VERSION": 27, "PDFIUM_RECORD_BATCH_SIZE": 1024} + constants = {"PROTOCOL_VERSION": 28, "PDFIUM_RECORD_BATCH_SIZE": 1024} assert {name for name in dir(native) if not name.startswith("__")} == functions.keys() | classes.keys() | constants.keys() for name, signature in functions.items(): function = getattr(native, name) diff --git a/tests/test_native_round4.py b/tests/test_native_round4.py index a38f22fa..762c5395 100644 --- a/tests/test_native_round4.py +++ b/tests/test_native_round4.py @@ -165,6 +165,57 @@ def test_marker_index_duplicate_sources_and_existing_cache(native): assert cache == before +def test_marker_line_context_is_reused_across_corridors(native, monkeypatch): + """同一候选上下文只做一次 marker 输入校验,多个走廊复用来源行索引。""" + from docvortex.analyzers.native.pdf import table_annotations as notes + from docvortex.analyzers.native.pdf.models import _LineItem + + lines = [ + _LineItem( + text, + (0.0, float(index), 10.0, float(index + 1)), + 0, + index, + chars=[{"char": text, "char_idx": index, "bbox": (0.0, 0.0, 1.0, 1.0)}], + ) + for index, text in enumerate(("body", "a1", "body")) + ] + calls = 0 + original = notes._prepare_marker_line_context + + def counted(values): + """仅统计上下文准备次数,不改变普通输入判定。""" + nonlocal calls + calls += 1 + return original(values) + + monkeypatch.setattr(notes, "_prepare_marker_line_context", counted) + context = rules._RuleCandidateContext( + [], lines, (100.0, 100.0), 0, 8.0, [], [], {}, notes._prepare_table_note_body_metrics(lines, (100.0, 100.0), 0) + ) + prepared = [] + for corridor in range(2): + if context.marker_line_context is None: + context.marker_line_context = notes._prepare_marker_line_context(lines) + marker_safe, source_lines = context.marker_line_context + row = SimpleNamespace( + fragments=[SimpleNamespace(line_index=1)], + bbox=(0.0, float(corridor), 10.0, float(corridor + 1)), + ) + prepared.append( + notes._prepare_table_core_rows( + [row], lines, context.body_metrics, marker_safe=marker_safe, source_lines=source_lines + ) + ) + assert calls == 1 + assert marker_safe is True + assert source_lines == {index: [line] for index, line in enumerate(lines)} + assert all(item.marker_safe and item.source_lines is source_lines for item in prepared) + unsafe = [{"char": "x", "char_idx": 0, "bbox": (0.0, 0.0, math.nan, 1.0)}] + lines[0].chars.append(unsafe) + assert notes._prepare_marker_line_context(lines) == (False, {}) + + def test_marker_queries_are_lazy_and_bounded(native, monkeypatch): """小候选只检查自身所需来源,重复查询复用位图,标记淘汰不影响判定。""" from docvortex.analyzers.native.pdf import table_annotations as notes diff --git a/tests/test_native_text_snapshot.py b/tests/test_native_text_snapshot.py index a40240f7..86a288a4 100644 --- a/tests/test_native_text_snapshot.py +++ b/tests/test_native_text_snapshot.py @@ -3,6 +3,7 @@ from contextlib import closing from dataclasses import fields, is_dataclass from io import BytesIO +from pathlib import Path import ctypes import math import pickle @@ -191,6 +192,54 @@ def test_snapshot_special_metadata_selects_reference_before_calculation(native): assert "nonstandard" in snapshot_bridge_info()["native_text_snapshot_unavailable_reason"] +def test_owned_geometry_risk_matches_line_reference(native): + """页面级 geometry evidence 与逐行参考风险结论一致,身份副本强制回退。""" + from copy import deepcopy + from docvortex.analyzers.native.pdf import char_geometry, pipeline + from docvortex.analyzers.native.pdf.native_text import _build_native_line_items_from_records + + with PDFDocument(_pdf(long=True)) as document: + private = document._extract_native_page(0) + geometry, records = private.native_text.prepare_visual_evidence(private.page_size, private.rotation, (0.0, 90.0, 270.0)) + lines = _build_native_line_items_from_records(records, private.page_size) + owned = pipeline._prepare_owned_geometry_evidence(private.native_text, geometry.chars) + args = ([lines], [geometry], [private.page_size]) + expected = char_geometry._document_requires_full_geometry(*args) + actual = char_geometry._document_requires_full_geometry(*args, owned_geometry_inputs=[owned]) + assert actual == expected + broken = deepcopy(lines) + broken[0].chars = [dict(broken[0].chars[0])] + fallback = char_geometry._document_requires_full_geometry( + [broken], [geometry], [private.page_size], owned_geometry_inputs=[owned] + ) + assert fallback == expected + + +def test_owned_table_and_geometry_evidence_match_reference(native, monkeypatch): + """表格脚本与全文几何 owned 输入必须与逐行参考输出完全一致。""" + from docvortex.analyzers.native.pdf import pipeline + from docvortex.analyzers.native.pdf.table_text_styles import table_script_stats + from docvortex._native import geometry_evidence_stats + + sample = Path(__file__).parents[1] / "demo/pdfs/demo3.pdf" + before_table = table_script_stats() + before_geometry = geometry_evidence_stats() + with PDFDocument(str(sample)) as document: + actual = pipeline._analyze_native_document(document) + with PDFDocument(str(sample)) as document: + monkeypatch.setattr(pipeline, "_prepare_owned_script_evidence", lambda *args, **kwargs: None) + monkeypatch.setattr(pipeline, "_prepare_owned_geometry_evidence", lambda *args, **kwargs: None) + expected = pipeline._analyze_native_document(document) + assert actual == expected + after_table = table_script_stats() + after_geometry = geometry_evidence_stats() + assert after_table[0] > before_table[0] + assert after_table[1] > before_table[1] + assert after_geometry[0] > before_geometry[0] + assert after_geometry[1] > before_geometry[1] + assert after_geometry[2] == before_geometry[2] + + def test_page_snapshot_bridge_loads_and_closes_textpage(native): """页面级入口不创建 Python textpage,并与兼容入口保持完整快照等价。""" with ( From 8387767878b3be47ec55b5f61e533066fabc9787 Mon Sep 17 00:00:00 2001 From: myhloli Date: Tue, 29 Sep 2026 23:12:28 +0800 Subject: [PATCH 2/3] docs(pdf): record stage 21 batching validation --- docs/rust-pdf-stage21-evidence.md | 55 +++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) create mode 100644 docs/rust-pdf-stage21-evidence.md diff --git a/docs/rust-pdf-stage21-evidence.md b/docs/rust-pdf-stage21-evidence.md new file mode 100644 index 00000000..d2b007c7 --- /dev/null +++ b/docs/rust-pdf-stage21-evidence.md @@ -0,0 +1,55 @@ +# PDF 原生内核阶段 21:表格脚本与全文几何边界批量化 + +## 实现边界 + +本阶段以 Stage 20 提交 `d504c5e` 为基线,不再继续 textpage 或 Path 迁移。私有协议从 27 升到 28;公开 `auto|python|rust` 计算选择、`auto|legacy|session` 渲染选择、公共 SDK 和输出 schema 均保持不变。 + +表格 HTML 物化复用页面级 `NativeScriptEvidence`:`_collect_document_sources()` 已生成的 owned 身份映射继续传入表格恢复链路。普通 0 度、无 fraction/inline 特殊证据且字符身份完整的 cell visual line 被合并为最多 8192 字的 Rust 批次;原生角色只要出现非 body,整 cell 回退既有 `_script_line_char_roles()` 参考路径,避免改变表格二维行合并后的精炼语义。诊断新增表格 owned 批次、命中行数和回退 cell 数。 + +表格候选的 marker 输入校验从每个 corridor 改为每个 `_RuleCandidateContext` 一次。同一方向候选组共享 marker-safe 结果和 source line 索引;重复 row、特殊字符、非有限坐标和 Python backend 仍走原参考路径。 + +新增页面级 `NativeGeometryEvidence`:Rust 一次从自有 text snapshot 准备 source/loose/tight/origin、rotation、字体 run key、字号、宽粒度文字类别和 anchor 标记。Python 主链路只传行成员索引和行元数据;`NativeGeometryRuns` 在全文内分配稳定 run ID,`NativeGeometryRisk` 连续累积布局/样式风险。解释器相关的字体族归一化、字号/字重取整和 Unicode 文字类别只在不同值上调用。身份缺失、特殊输入、旧扩展或规则替换时整本文档回退原逐行参考路径。 + +## 验证 + +- 32 PDF/299 页公开回放每份双跑;ModelJson、MiddleJson、素材哈希和诊断与 Stage 20 完整输出完全一致。 +- MinerU 当前 `dev@f504cff`:31 份 eligible Flash 文本完整输出一致;32 份实际 medium shared 完整输出一致。MinerU 工作区仅保留原有未跟踪 `examples/`,未修改其源码。 +- Rust/session 完整测试:5465 passed / 14 skipped。 +- Python/legacy 完整测试:4756 passed / 723 skipped。 +- Cargo workspace tests、Clippy `-D warnings`、rustfmt、Ruff check 和改动文件 format check 均通过。 +- ABI3 wheel 为 `docvortex-0.5.7-cp310-abi3-macosx_11_0_arm64.whl`。CPython 3.14 独立环境中 DocVortex 核心路径和 MinerU Flash 文本路径的 `auto|rust` 输出一致,协议 28,geometry evidence API 可用。源码扩展与 wheel 扩展 SHA-256 均为 `6266f409337d742153dc5bc0727d3b28f32a8f4cb210d8a5ea6e06cc650d3deb`。 + +公开正确性诊断中,表格 owned 路径执行 368 批/27,602 行/932 个 cell 回退;geometry evidence 执行 598 页/34,990 行/0 回退。32 PDF 单遍完整 cProfile 中,geometry evidence 执行 299 页/17,495 行,表格 owned 执行 184 批/13,801 行。 + +## 性能与资源 + +正式计时均为每文档一次预热、五次热运行;公开 parse 按文档先基线后候选,MinerU 按文档交替方向,RSS 使用独立进程树采样。结果仅代表本机语料与当前 PDFium 环境。 + +| 链路 | Stage 20 基线 | Stage 21 候选 | 首轮总降幅 | 首轮最大耗时比 | 首轮最大 RSS 比 | +| --- | ---: | ---: | ---: | ---: | ---: | +| DocVortex 公开 parse,32 PDF | 15.943439 s | 15.513413 s | 2.70% | 1.02077 | 1.03600 | +| MinerU Flash 文本,31 PDF | 15.746155 s | 15.304521 s | 2.80% | 1.02846 | 1.04413 | +| MinerU medium shared,32 PDF | 16.010728 s | 17.845207 s | -11.46% | 1.58188 | 1.00490 | + +shared 首轮有 4 个样本耗时超过 5%,均已按脚本自动反序复测: + +- `small_ocr.pdf`: `1.09802 → 0.98395` +- `engineering_process_restrictions_table.pdf`: `1.51380 → 0.99771` +- `pollutant_discharge_tables.pdf`: `1.58188 → 0.88237` +- `quarterly_report_financial_tables.pdf`: `1.12287 → 0.90017` + +四个样本均无持续退化。仅把这四个已触发样本替换为反序复测值作诊断汇总时,shared 总时间为 `16.174975 s → 15.777589 s`,约 2.46% 改善;该补充口径不替代首轮正式总表。所有 benchmark 完整输出均相等,公开 parse、Flash 和 shared 均无持续超过 5% 的耗时或 RSS 退化。 + +十样本表格重载焦点 profile 中: + +- 墙钟:`11.889497 s → 10.338891 s`,约 13.03%。 +- `_cell_script_roles`: `1.969974 s → 0.040009 s` +- `_prepare_table_core_rows`: `0.542125 s → 0.026794 s` +- `_document_requires_full_geometry`: `0.467145 s → 0.050283 s` +- 三项合计:`2.979244 s → 0.117086 s`,低于 `2.80 s` 门槛。 + +## 后续状态 + +原始 2 倍性能目标仍未完成。Stage 21 后的 32 PDF cProfile 主要剩余热点为 `_detect_table_candidates()` 约 `6.821 s`、`_materialize_table_blocks()` 约 `5.384 s`、`_infer_text_lanes()` 约 `3.252 s`、`build_document_geometry_plan()` 约 `3.626 s` 和 `_merge_owned_table_candidates()` 约 `2.224 s`。其中 geometry plan 的准入扫描已下降,剩余成本主要在确认布局风险后的 canonical 样本构造。Stage 22 应优先从表格候选合并/物化与 lane inference 重新 profile 后选择,不应继续扩大本轮 owned script/geometry 范围。 + +证据目录为 `output/pdf/native-kernel-20260929-stage21/`。本阶段不自动合并、发版或发布 PyPI。 From 4c771211d484d770da05f3e97d12dc45fdea0748 Mon Sep 17 00:00:00 2001 From: myhloli Date: Wed, 30 Sep 2026 00:11:13 +0800 Subject: [PATCH 3/3] fix(pdf): keep geometry run ids single-namespaced on owned fallback Owned batches intern run ids in NativeGeometryRuns while per-line fallback reallocated ids from zero in a Python dict; mixing both inside one NativeGeometryRisk accumulator merged unrelated runs and split identical ones, silently skewing the full-document risk verdict. Documents that opt into owned evidence now restart the whole calculation on the reference path as soon as any line misses, matching the documented whole-document fallback semantics. --- .../analyzers/native/pdf/char_geometry.py | 4 +++ tests/test_native_text_snapshot.py | 32 +++++++++++++++++++ 2 files changed, 36 insertions(+) diff --git a/src/docvortex/analyzers/native/pdf/char_geometry.py b/src/docvortex/analyzers/native/pdf/char_geometry.py index d7800f9c..757cdadd 100644 --- a/src/docvortex/analyzers/native/pdf/char_geometry.py +++ b/src/docvortex/analyzers/native/pdf/char_geometry.py @@ -750,6 +750,10 @@ def _document_requires_full_geometry(lines_by_page, geometries, page_sizes, *, o page_size, line.angle, ) + elif owned_geometry_inputs is not None: + # 回退行与 owned 批次分用两个从 0 起的 run 编号空间,混入同一风险器会合并无关 + # run、拆散同一 run;启用 owned 通道的文档任一行失配即整体回到参考路径。 + return _document_requires_full_geometry_python(lines_by_page, geometries, page_sizes) else: font_metadata = _ReadOnlyFontCache() entries = [] diff --git a/tests/test_native_text_snapshot.py b/tests/test_native_text_snapshot.py index 86a288a4..417ed991 100644 --- a/tests/test_native_text_snapshot.py +++ b/tests/test_native_text_snapshot.py @@ -215,6 +215,38 @@ def test_owned_geometry_risk_matches_line_reference(native): assert fallback == expected +def test_owned_geometry_fallback_restarts_reference_run_namespace(native, monkeypatch): + """启用 owned 通道后任一行失配必须整体重启参考路径,禁止两个从 0 起的 run 编号空间混用。""" + from copy import deepcopy + from docvortex.analyzers.native.pdf import char_geometry, pipeline + + with PDFDocument(_pdf(long=True)) as document: + private = document._extract_native_page(0) + geometry, records = private.native_text.prepare_visual_evidence(private.page_size, private.rotation, (0.0, 90.0, 270.0)) + lines = _build_native_line_items_from_records(records, private.page_size) + owned = pipeline._prepare_owned_geometry_evidence(private.native_text, geometry.chars) + expected = char_geometry._document_requires_full_geometry([lines], [geometry], [private.page_size]) + + reference = char_geometry._document_requires_full_geometry_python + calls = [] + + def spy(*args, **kwargs): + calls.append(args) + return reference(*args, **kwargs) + + monkeypatch.setattr(char_geometry, "_document_requires_full_geometry_python", spy) + broken = deepcopy(lines) + broken[0].chars = [dict(broken[0].chars[0])] + identity_miss = char_geometry._document_requires_full_geometry( + [broken], [geometry], [private.page_size], owned_geometry_inputs=[owned] + ) + page_missing = char_geometry._document_requires_full_geometry( + [lines], [geometry], [private.page_size], owned_geometry_inputs=[None] + ) + assert [call[0] for call in calls] == [[broken], [lines]] + assert identity_miss == expected and page_missing == expected + + def test_owned_table_and_geometry_evidence_match_reference(native, monkeypatch): """表格脚本与全文几何 owned 输入必须与逐行参考输出完全一致。""" from docvortex.analyzers.native.pdf import pipeline