diff --git a/.github/workflows/dotnet-tests.yml b/.github/workflows/dotnet-tests.yml index 1544d86577..1bc321cc32 100644 --- a/.github/workflows/dotnet-tests.yml +++ b/.github/workflows/dotnet-tests.yml @@ -2105,3 +2105,37 @@ jobs: with: name: nativeaot-win-x64 path: artifacts/aot/*.json + + dbf-contracts: + name: 'DBF contracts / ${{ matrix.os }} / ${{ matrix.framework }}' + runs-on: ${{ matrix.os }} + timeout-minutes: 15 + strategy: + fail-fast: false + matrix: + os: [ubuntu-latest, windows-latest] + framework: [net8.0, net10.0] + include: + - os: windows-latest + framework: net472 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + - uses: actions/setup-dotnet@v6 + with: + dotnet-version: ${{ env.DOTNET_VERSION }} + - name: Read independent DBF fixtures and reopen CSV/XLSX output + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $PSNativeCommandUseErrorActionPreference = $true + dotnet test OfficeIMO.Reader.Dbf.Tests/OfficeIMO.Reader.Dbf.Tests.csproj --configuration ${{ env.BUILD_CONFIGURATION }} --framework ${{ matrix.framework }} --filter 'Category!=Performance' --logger 'trx;LogFileName=dbf.trx' --results-directory TestResults/Dbf + - name: Upload test results + if: always() + uses: actions/upload-artifact@v7 + with: + name: dbf-tests-${{ matrix.os }}-${{ matrix.framework }} + path: TestResults/Dbf + if-no-files-found: ignore + retention-days: 7 diff --git a/Build/CompatibilityCatalog/ConversionApiCompileContract.cs b/Build/CompatibilityCatalog/ConversionApiCompileContract.cs index 68692cd5e5..fc96e3bf74 100644 --- a/Build/CompatibilityCatalog/ConversionApiCompileContract.cs +++ b/Build/CompatibilityCatalog/ConversionApiCompileContract.cs @@ -1,4 +1,5 @@ using OfficeIMO.Adf; +using DBAClientX.Dbf; using OfficeIMO.AsciiDoc; using OfficeIMO.Confluence; using OfficeIMO.CSV; @@ -174,6 +175,13 @@ internal static void Verify( _ = ConfluenceContentConverter.ToHtml(confluencePage); _ = ExcelDocumentCsvExtensions.ToExcelDocument(csv); _ = ExcelSheetCsvExtensions.ToCsv(excel.Sheets[0]); + using (DbfDataReader dbf = DbfDataReader.Open(source)) + using (StringWriter csvOutput = new StringWriter()) { + CsvDocument.WriteDataReader(csvOutput, dbf); + } + using (DbfDataReader dbf = DbfDataReader.Open(source)) { + _ = ExcelDocument.WriteDataReader(stream, dbf); + } _ = OfficeIMO.Markup.Word.OfficeMarkupWordConverterExtensions.ToWordDocumentResult(officeMarkup); _ = OfficeIMO.Markup.Excel.OfficeMarkupExcelConverterExtensions.ToExcelDocumentResult(officeMarkup); _ = OfficeIMO.Markup.PowerPoint.OfficeMarkupPowerPointConverterExtensions.ToPowerPointPresentationResult(officeMarkup); diff --git a/Build/CompatibilityCatalog/FormatMapProjection.cs b/Build/CompatibilityCatalog/FormatMapProjection.cs index 86d2aa506c..f8c64be968 100644 --- a/Build/CompatibilityCatalog/FormatMapProjection.cs +++ b/Build/CompatibilityCatalog/FormatMapProjection.cs @@ -66,7 +66,7 @@ private static string[] Surfaces(OfficeConversionCapability route, HashSetWithin a family, the best-known formats come first. private static readonly string[] Prominence = [ "DOCX", "DOC", "ODT", "RTF", "Pages", "Google Docs", - "XLSX", "ODS", "CSV", "Numbers", "Google Sheets", + "XLSX", "ODS", "CSV", "DBF/xBase", "Numbers", "Google Sheets", "PPTX", "ODP", "Keynote", "Google Slides", "PDF", "XPS/OpenXPS", "MHTML", "Visio", "ODG/FODG", "PNG", "SVG", "JPEG", "TIFF", "WebP", @@ -84,7 +84,7 @@ private static int Rank(string format) { private static string Family(string format) => format switch { "DOCX" or "DOC" or "ODT" or "RTF" or "Pages" or "Google Docs" => "documents", - "XLSX" or "ODS" or "CSV" or "Numbers" or "Google Sheets" => "spreadsheets", + "XLSX" or "ODS" or "CSV" or "DBF/xBase" or "Numbers" or "Google Sheets" => "spreadsheets", "PPTX" or "ODP" or "Keynote" or "Google Slides" => "presentations", "PDF" or "XPS/OpenXPS" or "MHTML" or "Visio" or "ODG/FODG" => "fixed", "PNG" or "SVG" or "JPEG" or "TIFF" or "WebP" => "images", diff --git a/Build/CompatibilityCatalog/OfficeIMO.CompatibilityCatalog.Tool.csproj b/Build/CompatibilityCatalog/OfficeIMO.CompatibilityCatalog.Tool.csproj index 59296bb956..d76821782f 100644 --- a/Build/CompatibilityCatalog/OfficeIMO.CompatibilityCatalog.Tool.csproj +++ b/Build/CompatibilityCatalog/OfficeIMO.CompatibilityCatalog.Tool.csproj @@ -7,6 +7,7 @@ false + diff --git a/Build/Test-AotProjectCoverage.ps1 b/Build/Test-AotProjectCoverage.ps1 index 369d71ffe8..3bc26b4167 100644 --- a/Build/Test-AotProjectCoverage.ps1 +++ b/Build/Test-AotProjectCoverage.ps1 @@ -48,6 +48,11 @@ $nativeTools = @( } ) $nonNativeValidated = @( + [ordered]@{ + name = 'OfficeIMO.Reader.Dbf' + classification = 'managed-cross-platform' + evidence = 'DBF ingestion is qualified through managed .NET 8 and 10 extraction and CSV/XLSX conversion tests; the DBAClientX.Dbf owner and Reader adapter are not rooted in a NativeAOT qualification host.' + } [ordered]@{ name = 'OfficeIMO.Chm' classification = 'managed-cross-platform' diff --git a/Build/project.build.json b/Build/project.build.json index 67efa19e2b..1293fdec30 100644 --- a/Build/project.build.json +++ b/Build/project.build.json @@ -99,6 +99,7 @@ "OfficeIMO.Reader.AsciiDoc": "3.4.X", "OfficeIMO.Reader.Chm": "3.4.X", "OfficeIMO.Reader.Csv": "3.4.X", + "OfficeIMO.Reader.Dbf": "3.4.X", "OfficeIMO.Reader.DocBook": "3.4.X", "OfficeIMO.Reader.Email": "3.4.X", "OfficeIMO.Reader.Excel": "3.4.X", diff --git a/Docs/Compatibility/generated/README.md b/Docs/Compatibility/generated/README.md index c5c06f4dad..5410acbd21 100644 --- a/Docs/Compatibility/generated/README.md +++ b/Docs/Compatibility/generated/README.md @@ -23,7 +23,7 @@ dotnet run --framework net8.0 --project Build/CompatibilityCatalog/OfficeIMO.Com | OfficeIMO.Excel.LegacyXls | 1 | 28 | [JSON](excel-legacy-xls.json) | [Markdown](excel-legacy-xls.md) | | OfficeIMO.Excel.Xlsb | 1 | 20 | [JSON](excel-xlsb.json) | [Markdown](excel-xlsb.md) | | OfficeIMO.PowerPoint.LegacyPpt | 1 | 56 | [JSON](powerpoint-legacy-ppt.json) | [Markdown](powerpoint-legacy-ppt.md) | -| OfficeIMO.Operations | 1 | 1394 | [JSON](package-operations.json) | [Markdown](package-operations.md) | +| OfficeIMO.Operations | 1 | 1396 | [JSON](package-operations.json) | [Markdown](package-operations.md) | | OfficeIMO.Provenance | 1 | 11 | [JSON](provenance.json) | [Markdown](provenance.md) | | OfficeIMO.ProtectedContent | 2 | 18 | [JSON](protected-content.json) | [Markdown](protected-content.md) | diff --git a/Docs/Compatibility/generated/conversion-routes.json b/Docs/Compatibility/generated/conversion-routes.json index c1152ac1e4..52701b076e 100644 --- a/Docs/Compatibility/generated/conversion-routes.json +++ b/Docs/Compatibility/generated/conversion-routes.json @@ -1561,6 +1561,46 @@ "browserAvailable":false, "agentDiscoverable":true }, + { + "id":"dbf-csv", + "source":"DBF/xBase", + "target":"CSV", + "inputKind":"File", + "sourceExtensions":[ ".dbf" ], + "targetExtension":".csv", + "packageId":"OfficeIMO.CSV", + "api":"DbfDataReader.Open(path), then CsvDocument.WriteDataReader(writer, reader, options)", + "description":"Export typed DBF rows and supported memos through DBAClientX.Dbf; binary values become Base64.", + "fidelity":"Semantic", + "textFormatting":"DataOnly", + "textFormattingContract":"DBF carries typed table values, not document typography or layout; binary values become Base64, and DBF indexes, deletion flags and native schema metadata are not exported.", + "supportLevel":"Targeted", + "supportEvidence":"Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation.", + "knownLimitations":"Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits.", + "resultContract":"void", + "browserAvailable":false, + "agentDiscoverable":true + }, + { + "id":"dbf-xlsx", + "source":"DBF/xBase", + "target":"XLSX", + "inputKind":"File", + "sourceExtensions":[ ".dbf" ], + "targetExtension":".xlsx", + "packageId":"OfficeIMO.Excel", + "api":"DbfDataReader.Open(path), then ExcelDocument.WriteDataReader(stream, reader, options)", + "description":"Create a workbook from typed DBF rows and supported memos through DBAClientX.Dbf.", + "fidelity":"Editable", + "textFormatting":"DataOnly", + "textFormattingContract":"DBF carries typed table values, not document typography or layout; binary values become Base64, and DBF indexes, deletion flags and native schema metadata are not exported.", + "supportLevel":"Targeted", + "supportEvidence":"Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation.", + "knownLimitations":"Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits.", + "resultContract":"ExcelDataSetImportResult", + "browserAvailable":false, + "agentDiscoverable":true + }, { "id":"officemarkup-docx", "source":"OfficeIMO Markup", diff --git a/Docs/Compatibility/generated/conversion-routes.md b/Docs/Compatibility/generated/conversion-routes.md index 9debe244fb..09e1a72f14 100644 --- a/Docs/Compatibility/generated/conversion-routes.md +++ b/Docs/Compatibility/generated/conversion-routes.md @@ -84,6 +84,8 @@ Schema version: 9 | confluence-html | Confluence | HTML | OfficeIMO.Confluence | Semantic | SyntaxSubset | Preserves the text styling represented by the source and destination profiles; unsupported native decoration variants, arbitrary CSS presentation, and format-specific layout are reported or simplified. | Targeted | Materialized Confluence page and body fixtures exercise ADF and storage representations through the shared ADF, Markdown, and HTML converters with combined diagnostics. | This is content-body conversion, not live page synchronization; Confluence-only macros, layouts, extensions, and presentation outside the supported ADF/storage subset may be simplified. | No | `ConfluenceContentConverter.ToHtml(page)` | ConfluenceContentConversionResult | Project a materialized Confluence page body to HTML with fidelity diagnostics. | | csv-xlsx | CSV | XLSX | OfficeIMO.Excel.Csv | Editable | DataOnly | CSV and TSV carry values, delimiters, and records only; font family, size, color, emphasis, decoration, scripts, casing metadata, and layout are intentionally not representable. | Established | Delimited-data fixtures cover streams, files, delimiter detection, typed cell import, worksheet-range export, reopen behavior, and cancellation. | CSV and TSV have no font, rich-text, formula, multi-sheet, drawing, or layout model; those workbook semantics are intentionally absent from CSV output. | No | `CsvDocument.Load(stream).ToExcelDocument(options)` | ExcelDocument | Import delimited values into an editable workbook. | | xlsx-csv | XLSX | CSV | OfficeIMO.Excel.Csv | Semantic | DataOnly | CSV and TSV carry values, delimiters, and records only; font family, size, color, emphasis, decoration, scripts, casing metadata, and layout are intentionally not representable. | Established | Delimited-data fixtures cover streams, files, delimiter detection, typed cell import, worksheet-range export, reopen behavior, and cancellation. | CSV and TSV have no font, rich-text, formula, multi-sheet, drawing, or layout model; those workbook semantics are intentionally absent from CSV output. | No | `ExcelDocument.Load(stream).Sheets[0].ToCsv(options)` | string | Export a worksheet used range as delimited values. | +| dbf-csv | DBF/xBase | CSV | OfficeIMO.CSV | Semantic | DataOnly | DBF carries typed table values, not document typography or layout; binary values become Base64, and DBF indexes, deletion flags and native schema metadata are not exported. | Targeted | Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation. | Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits. | No | `DbfDataReader.Open(path), then CsvDocument.WriteDataReader(writer, reader, options)` | void | Export typed DBF rows and supported memos through DBAClientX.Dbf; binary values become Base64. | +| dbf-xlsx | DBF/xBase | XLSX | OfficeIMO.Excel | Editable | DataOnly | DBF carries typed table values, not document typography or layout; binary values become Base64, and DBF indexes, deletion flags and native schema metadata are not exported. | Targeted | Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation. | Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits. | No | `DbfDataReader.Open(path), then ExcelDocument.WriteDataReader(stream, reader, options)` | ExcelDataSetImportResult | Create a workbook from typed DBF rows and supported memos through DBAClientX.Dbf. | | officemarkup-docx | OfficeIMO Markup | DOCX | OfficeIMO.Markup.Word | Editable | EditableEquivalent | Authors editable native typography, including family, size, color, emphasis, decoration, script, and casing, in the generated Office document; diagnostics identify profile-specific approximations and omissions. | Established | Typed profile fixtures verify parser diagnostics, editable destination artifacts, font family/size/color, bold, italic, underline variants, strike, scripts, case transforms, small caps where native, and target-specific layout. | OfficeIMO Markup is a directed authoring format rather than a lossless Office round trip; unsupported blocks and destination-only effects are diagnosed, simplified, or omitted according to exporter options. | No | `OfficeMarkupParser.Parse(markup, options).Document.ToWordDocumentResult(exportOptions)` | OfficeMarkupConversionResult | Render document-profile OfficeIMO Markup into an editable Word document. | | officemarkup-xlsx | OfficeIMO Markup | XLSX | OfficeIMO.Markup.Excel | Editable | EditableEquivalent | Authors editable native typography, including family, size, color, emphasis, decoration, script, and casing, in the generated Office document; diagnostics identify profile-specific approximations and omissions. | Established | Typed profile fixtures verify parser diagnostics, editable destination artifacts, font family/size/color, bold, italic, underline variants, strike, scripts, case transforms, small caps where native, and target-specific layout. | OfficeIMO Markup is a directed authoring format rather than a lossless Office round trip; unsupported blocks and destination-only effects are diagnosed, simplified, or omitted according to exporter options. | No | `OfficeMarkupParser.Parse(markup, options).Document.ToExcelDocumentResult(exportOptions)` | OfficeMarkupConversionResult | Render workbook-profile OfficeIMO Markup into an editable Excel workbook. | | officemarkup-pptx | OfficeIMO Markup | PPTX | OfficeIMO.Markup.PowerPoint | Editable | EditableEquivalent | Authors editable native typography, including family, size, color, emphasis, decoration, script, and casing, in the generated Office document; diagnostics identify profile-specific approximations and omissions. | Established | Typed profile fixtures verify parser diagnostics, editable destination artifacts, font family/size/color, bold, italic, underline variants, strike, scripts, case transforms, small caps where native, and target-specific layout. | OfficeIMO Markup is a directed authoring format rather than a lossless Office round trip; unsupported blocks and destination-only effects are diagnosed, simplified, or omitted according to exporter options. | No | `OfficeMarkupParser.Parse(markup, options).Document.ToPowerPointPresentationResult(exportOptions)` | OfficeMarkupPowerPointConversionResult | Render presentation-profile OfficeIMO Markup into an editable PowerPoint presentation. | diff --git a/Docs/Compatibility/generated/package-operations.json b/Docs/Compatibility/generated/package-operations.json index 6189c27304..df54cb3a78 100644 --- a/Docs/Compatibility/generated/package-operations.json +++ b/Docs/Compatibility/generated/package-operations.json @@ -772,6 +772,20 @@ "extensions":[".br",".brotli",".csv",".deflate",".gz",".gzip",".zlib"], "limitation":"" }, + { + "id":"conversion:dbf-csv", + "packageId":"OfficeIMO.CSV", + "formatId":"DBF/xBase", + "targetFormatId":"CSV", + "capabilityId":"dbf-csv", + "operation":"Convert", + "state":"Partial", + "publicApi":"DbfDataReader.Open(path), then CsvDocument.WriteDataReader(writer, reader, options)", + "evidence":"Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation.", + "sourceCatalog":"OfficeConversionCapabilityCatalog", + "extensions":[".dbf"], + "limitation":"Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits." + }, { "id":"native:chm-native:create", "packageId":"OfficeIMO.Chm", @@ -2970,6 +2984,20 @@ "extensions":[".epub"], "limitation":"Image output is intentionally flattened. Unsupported source structure, remote or missing resources, and advanced source-specific layout remain approximated or diagnosed." }, + { + "id":"conversion:dbf-xlsx", + "packageId":"OfficeIMO.Excel", + "formatId":"DBF/xBase", + "targetFormatId":"XLSX", + "capabilityId":"dbf-xlsx", + "operation":"Convert", + "state":"Partial", + "publicApi":"DbfDataReader.Open(path), then ExcelDocument.WriteDataReader(stream, reader, options)", + "evidence":"Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation.", + "sourceCatalog":"OfficeConversionCapabilityCatalog", + "extensions":[".dbf"], + "limitation":"Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits." + }, { "id":"protection:vba-signature:excel:create", "packageId":"OfficeIMO.Excel", diff --git a/Docs/Compatibility/generated/package-operations.md b/Docs/Compatibility/generated/package-operations.md index fc7a0a5300..f6638912a4 100644 --- a/Docs/Compatibility/generated/package-operations.md +++ b/Docs/Compatibility/generated/package-operations.md @@ -61,6 +61,7 @@ Each row retains its detailed owning catalog and evidence. A supported row may s | `OfficeIMO.CSV` | Csv.Native | .br, .brotli, .csv, .deflate, .gz, .gzip, .zlib | | Preserve | Partial | csv-native | OfficeIMO.NativeLifecycle | `CsvDocument.Parse / Load / Save / Validate` | OfficeIMO.CSV.Tests parsing, schema validation, editing, bounded streaming, and deterministic-write contracts::Preserve | Saving serializes the typed table with the selected dialect and normalization settings; byte-identical source preservation is not promised after parsing. | | `OfficeIMO.CSV` | Csv.Native | .br, .brotli, .csv, .deflate, .gz, .gzip, .zlib | | Inspect | Supported | csv-native | OfficeIMO.NativeLifecycle | `CsvDocument.Parse / Load / Save / Validate` | OfficeIMO.CSV.Tests parsing, schema validation, editing, bounded streaming, and deterministic-write contracts::Inspect | | | `OfficeIMO.CSV` | Csv.Native | .br, .brotli, .csv, .deflate, .gz, .gzip, .zlib | | Validate | Supported | csv-native | OfficeIMO.NativeLifecycle | `CsvDocument.Parse / Load / Save / Validate` | OfficeIMO.CSV.Tests parsing, schema validation, editing, bounded streaming, and deterministic-write contracts::Validate | | +| `OfficeIMO.CSV` | DBF/xBase | .dbf | CSV | Convert | Partial | dbf-csv | OfficeConversionCapabilityCatalog | `DbfDataReader.Open(path), then CsvDocument.WriteDataReader(writer, reader, options)` | Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation. | Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits. | | `OfficeIMO.Chm` | Chm.Native | .chm | | Create | Unsupported | chm-native | OfficeIMO.NativeLifecycle | `ChmDocument.Load / Topics / Resources / TableOfContents / Index; ChmEntry.GetBytes / OpenRead` | OfficeIMO.Chm.Tests Microsoft-compiled fixture, independent entry hashes, bounded reads, navigation and conversion contracts::Create | Read-only ITSF 2/3 archives with uncompressed or standard LZX storage. Raw entries remain exact; no CHM writer, Windows help viewer, ActiveX, script execution, merged-book loading or compiled full-text search is provided. | | `OfficeIMO.Chm` | Chm.Native | .chm | | Read | Supported | chm-native | OfficeIMO.NativeLifecycle | `ChmDocument.Load / Topics / Resources / TableOfContents / Index; ChmEntry.GetBytes / OpenRead` | OfficeIMO.Chm.Tests Microsoft-compiled fixture, independent entry hashes, bounded reads, navigation and conversion contracts::Read | Read-only ITSF 2/3 archives with uncompressed or standard LZX storage. Raw entries remain exact; no CHM writer, Windows help viewer, ActiveX, script execution, merged-book loading or compiled full-text search is provided. | | `OfficeIMO.Chm` | Chm.Native | .chm | | Edit | Unsupported | chm-native | OfficeIMO.NativeLifecycle | `ChmDocument.Load / Topics / Resources / TableOfContents / Index; ChmEntry.GetBytes / OpenRead` | OfficeIMO.Chm.Tests Microsoft-compiled fixture, independent entry hashes, bounded reads, navigation and conversion contracts::Edit | Read-only ITSF 2/3 archives with uncompressed or standard LZX storage. Raw entries remain exact; no CHM writer, Windows help viewer, ActiveX, script execution, merged-book loading or compiled full-text search is provided. | @@ -218,6 +219,7 @@ Each row retains its detailed owning catalog and evidence. A supported row may s | `OfficeIMO.Epub.Image` | EPUB | .epub | SVG | Export | Partial | epub-svg | OfficeConversionCapabilityCatalog | `EpubDocument.Load(stream, readOptions).ExportImages(format, options)` | Typed fixtures exercise the source adapter, shared SVG and raster renderer, all five image encoders, and deterministic diagnostics. | Image output is intentionally flattened. Unsupported source structure, remote or missing resources, and advanced source-specific layout remain approximated or diagnosed. | | `OfficeIMO.Epub.Image` | EPUB | .epub | TIFF | Export | Partial | epub-tiff | OfficeConversionCapabilityCatalog | `EpubDocument.Load(stream, readOptions).ExportImages(format, options)` | Typed fixtures exercise the source adapter, shared SVG and raster renderer, all five image encoders, and deterministic diagnostics. | Image output is intentionally flattened. Unsupported source structure, remote or missing resources, and advanced source-specific layout remain approximated or diagnosed. | | `OfficeIMO.Epub.Image` | EPUB | .epub | WebP | Export | Partial | epub-webp | OfficeConversionCapabilityCatalog | `EpubDocument.Load(stream, readOptions).ExportImages(format, options)` | Typed fixtures exercise the source adapter, shared SVG and raster renderer, all five image encoders, and deterministic diagnostics. | Image output is intentionally flattened. Unsupported source structure, remote or missing resources, and advanced source-specific layout remain approximated or diagnosed. | +| `OfficeIMO.Excel` | DBF/xBase | .dbf | XLSX | Convert | Partial | dbf-xlsx | OfficeConversionCapabilityCatalog | `DbfDataReader.Open(path), then ExcelDocument.WriteDataReader(stream, reader, options)` | Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation. | Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits. | | `OfficeIMO.Excel` | DOCM/XLSM/XLSB/PPTM families | .xlam, .xlsb, .xlsm, .xltm | | Create | Supported | vba-signature | OfficeIMO.ProtectedContent | `InspectVbaSignatures / SignVbaProject` | OfficeProtectionCapabilityCatalog.Current::vba-signature::create | Managed legacy, agile, and V3 VBA signature profiles are corpus-bound. | | `OfficeIMO.Excel` | DOCM/XLSM/XLSB/PPTM families | .xlam, .xlsb, .xlsm, .xltm | | Read | NotApplicable | vba-signature | OfficeIMO.ProtectedContent | `InspectVbaSignatures / SignVbaProject` | OfficeProtectionCapabilityCatalog.Current::vba-signature::open | Managed legacy, agile, and V3 VBA signature profiles are corpus-bound. | | `OfficeIMO.Excel` | DOCM/XLSM/XLSB/PPTM families | .xlam, .xlsb, .xlsm, .xltm | | Edit | Rejected | vba-signature | OfficeIMO.ProtectedContent | `InspectVbaSignatures / SignVbaProject` | OfficeProtectionCapabilityCatalog.Current::vba-signature::mutate | Managed legacy, agile, and V3 VBA signature profiles are corpus-bound. | diff --git a/Docs/ROADMAP.md b/Docs/ROADMAP.md index 17096ad3de..e224ad8250 100644 --- a/Docs/ROADMAP.md +++ b/Docs/ROADMAP.md @@ -713,7 +713,7 @@ Current native lifecycle and rendering boundaries: [OfficeIMO.Xps support matrix - [ ] Establish a provenance-recorded cross-producer corpus for every structured spreadsheet profile, including malformed/truncated records, formula-bomb limits, code pages, workbook bounds, active content, and external links; add semantic, reopen, and rendered conversion evidence for XLSX, ODS, CSV, HTML, and PDF before promoting each profile beyond its documented maturity. - [ ] Add HWP binary and HWPX readers with safe text, structure, metadata, embedded-resource, and security inspection plus DOCX, ODT, HTML, Markdown, PDF, and plain-text conversion. Keep binary HWP read/convert-only; consider bounded HWPX authoring later only when schema-version and producer-interoperability evidence justify it. - [ ] Add unencrypted MOBI and AZW-family inspection and conversion only for versions with a legally and technically supportable fixture corpus. Extract metadata, navigation, text, and safe resources into the shared ebook/Reader model; report unsupported layouts and encryption; and do not implement DRM removal or source-format writing. -- [ ] Integrate DBF/xBase through a thin `OfficeIMO.Reader.Dbf` adapter and conversion routes only after the canonical codec and tabular streaming contracts exist in DbaClientX. OfficeIMO may map rows to CSV, Excel, HTML, PDF, and Reader results, but must not own another DBF parser, memo codec, code-page engine, or database writer. +- [ ] Widen the [DbaClientX-backed DBF/xBase integration](../OfficeIMO.Reader.Dbf/README.md) with native dBASE/FoxPro application acceptance and independent fixtures for Visual FoxPro `31`/`32`, broader memo/field profiles and code pages. Qualify any additional HTML/PDF projections through the existing table/rendering owners; keep DBF parsing, memo handling and typed streaming in DbaClientX. - [ ] Broaden the [CHM read-and-convert profile](../OfficeIMO.Chm/SUPPORT.md) with additional independently compiled multilingual archives and rendered-output qualification; improve CSS scope and identifier rebasing for reflowable exports, and add preserved tagged structure and cross-topic destinations through the shared PDF owner before extending multi-topic PDF guarantees. ### Future integration decisions @@ -723,7 +723,7 @@ Current native lifecycle and rendering boundaries: [OfficeIMO.Xps support matrix ### Admission and sequencing - [ ] Extend the generated capability and conversion catalogs so every additional format declares `FullProduct`, `ReadConvert`, `SalvageRead`, or `IntentionalBoundary`, together with versions/profiles, security behavior, preservation guarantees, fixture provenance, and route maturity. -- [ ] Deliver the remaining expansion wave in ownership order: the DbaClientX-backed DBF adapter; bounded JATS; then proprietary legacy and additional ebook readers selected by real fixture availability and consumer demand. +- [ ] Deliver the remaining expansion wave in ownership order: bounded JATS; then proprietary legacy and additional ebook readers selected by real fixture availability and consumer demand. - [ ] Require hostile-input limits, path and stream parity, cancellation where applicable, reopen or independent-reader evidence, deterministic-loss reporting, and license/provenance review before publishing any new format package. A file-extension detector or text scraper alone does not qualify as format support. ## Completion rule diff --git a/MIGRATION.md b/MIGRATION.md index 0124a9646b..f9d79a8291 100644 --- a/MIGRATION.md +++ b/MIGRATION.md @@ -168,6 +168,16 @@ XPS input kinds. Use `OfficeDocumentReadResultSchema.GetJsonSchema()` for the cu artifact. Native logical order is retained in `ReaderLocation.LogicalOrder`; physical page citations remain separate. Null order values retain existing container order. +## DBF Reader identity and transport + +Register `.AddDbfHandler()` from `OfficeIMO.Reader.Dbf`, or use the all-adapters +preset, to ingest DBF/xBase tables. `ReaderInputKind.Dbf` (`29`) requires document +transport schema v13. Update exhaustive input-kind switches and transport bindings; +schemas v5 through v12 remain readable but cannot carry DBF input kinds. CHM (`27`) +and DjVu (`28`) retain their existing identities and minimum schema versions. +Memo sidecar reads require explicit opt-in through `ReaderDbfOptions` or +`ReaderAllOptions.Dbf.AllowMemoSidecarReads`. + ## DjVu Reader identity and transport Register `.AddDjVuHandler()` from `OfficeIMO.Reader.DjVu` for `.djvu` and `.djv`, diff --git a/OfficeIMO.CSV/CsvWriter.cs b/OfficeIMO.CSV/CsvWriter.cs index 9d9cdecd87..6b8ac47336 100644 --- a/OfficeIMO.CSV/CsvWriter.cs +++ b/OfficeIMO.CSV/CsvWriter.cs @@ -785,6 +785,8 @@ private static string FormatValue(object? value, CultureInfo culture, string? da return nullValue ?? string.Empty; } + if (value is byte[] bytes) return Convert.ToBase64String(bytes); + if (value is DateTime dateTime) { if (useUtc) diff --git a/OfficeIMO.CSV/README.md b/OfficeIMO.CSV/README.md index 371c91e863..cf1c035e03 100644 --- a/OfficeIMO.CSV/README.md +++ b/OfficeIMO.CSV/README.md @@ -917,6 +917,7 @@ This table is generated from the package-neutral OfficeIMO operation catalog. Th | Preserve | 0 | 1 | 0 | 0 | 0 | 0 | | Inspect | 1 | 0 | 0 | 0 | 0 | 0 | | Validate | 1 | 0 | 0 | 0 | 0 | 0 | +| Convert | 0 | 1 | 0 | 0 | 0 | 0 | The complete rows for `OfficeIMO.CSV` are published in the [generated operation contract](https://github.com/EvotecIT/OfficeIMO/blob/master/Docs/Compatibility/generated/package-operations.md). diff --git a/OfficeIMO.Chm.Tests/ReaderTests.cs b/OfficeIMO.Chm.Tests/ReaderTests.cs index 6e490349b5..116c47edbd 100644 --- a/OfficeIMO.Chm.Tests/ReaderTests.cs +++ b/OfficeIMO.Chm.Tests/ReaderTests.cs @@ -54,9 +54,12 @@ public void ReaderRetainsTopicCitationsRichTablesAndArchiveHash() { Assert.NotEmpty(result.Blocks); Assert.NotEmpty(result.Tables); Assert.NotEmpty(result.Links); Assert.Equal(result.Blocks.Count, result.Blocks.Select(block => block.Id).Distinct().Count()); Assert.Equal(ReaderInputKind.Chm, OfficeDocumentReadResultJson.Deserialize(OfficeDocumentReadResultJson.Serialize(result)).Kind); - var oldEnvelope = System.Text.Json.Nodes.JsonNode.Parse(OfficeDocumentReadResultJson.Serialize(result))!; - oldEnvelope["schemaVersion"] = 10; - string oldBinding = oldEnvelope.ToJsonString(); + result.SchemaVersion = 11; + string chmBinding = OfficeDocumentReadResultJson.Serialize(result); + Assert.Equal(ReaderInputKind.Chm, OfficeDocumentReadResultJson.Deserialize(chmBinding).Kind); + var oldBindingJson = System.Text.Json.Nodes.JsonNode.Parse(chmBinding)!; + oldBindingJson["schemaVersion"] = 10; + string oldBinding = oldBindingJson.ToJsonString(); Assert.Throws(() => OfficeDocumentReadResultJson.Deserialize(oldBinding)); } diff --git a/OfficeIMO.Core/Compatibility/OfficeConversionCapabilityCatalog.cs b/OfficeIMO.Core/Compatibility/OfficeConversionCapabilityCatalog.cs index 8f1692babc..b10b41716b 100644 --- a/OfficeIMO.Core/Compatibility/OfficeConversionCapabilityCatalog.cs +++ b/OfficeIMO.Core/Compatibility/OfficeConversionCapabilityCatalog.cs @@ -428,6 +428,8 @@ private static OfficeConversionCapability[] CreateRoutes() { Route("confluence-html", "Confluence", "HTML", OfficeConversionInputKind.ObjectModel, new[] { ".adf", ".json" }, ".html", "OfficeIMO.Confluence", "ConfluenceContentConverter.ToHtml(page)", "Project a materialized Confluence page body to HTML with fidelity diagnostics.", OfficeConversionFidelityKind.Semantic, "ConfluenceContentConversionResult"), Route("csv-xlsx", "CSV", "XLSX", OfficeConversionInputKind.File, new[] { ".csv", ".tsv" }, ".xlsx", "OfficeIMO.Excel.Csv", "CsvDocument.Load(stream).ToExcelDocument(options)", "Import delimited values into an editable workbook.", OfficeConversionFidelityKind.Editable, "ExcelDocument"), Route("xlsx-csv", "XLSX", "CSV", OfficeConversionInputKind.File, new[] { ".xlsx" }, ".csv", "OfficeIMO.Excel.Csv", "ExcelDocument.Load(stream).Sheets[0].ToCsv(options)", "Export a worksheet used range as delimited values.", OfficeConversionFidelityKind.Semantic, "string"), + Route("dbf-csv", "DBF/xBase", "CSV", OfficeConversionInputKind.File, new[] { ".dbf" }, ".csv", "OfficeIMO.CSV", "DbfDataReader.Open(path), then CsvDocument.WriteDataReader(writer, reader, options)", "Export typed DBF rows and supported memos through DBAClientX.Dbf; binary values become Base64.", OfficeConversionFidelityKind.Semantic, "void"), + Route("dbf-xlsx", "DBF/xBase", "XLSX", OfficeConversionInputKind.File, new[] { ".dbf" }, ".xlsx", "OfficeIMO.Excel", "DbfDataReader.Open(path), then ExcelDocument.WriteDataReader(stream, reader, options)", "Create a workbook from typed DBF rows and supported memos through DBAClientX.Dbf.", OfficeConversionFidelityKind.Editable, "ExcelDataSetImportResult"), Route("officemarkup-docx", "OfficeIMO Markup", "DOCX", OfficeConversionInputKind.Text, new[] { ".omd", ".office.md" }, ".docx", "OfficeIMO.Markup.Word", "OfficeMarkupParser.Parse(markup, options).Document.ToWordDocumentResult(exportOptions)", "Render document-profile OfficeIMO Markup into an editable Word document.", OfficeConversionFidelityKind.Editable, "OfficeMarkupConversionResult"), Route("officemarkup-xlsx", "OfficeIMO Markup", "XLSX", OfficeConversionInputKind.Text, new[] { ".omd", ".office.md" }, ".xlsx", "OfficeIMO.Markup.Excel", "OfficeMarkupParser.Parse(markup, options).Document.ToExcelDocumentResult(exportOptions)", "Render workbook-profile OfficeIMO Markup into an editable Excel workbook.", OfficeConversionFidelityKind.Editable, "OfficeMarkupConversionResult"), Route("officemarkup-pptx", "OfficeIMO Markup", "PPTX", OfficeConversionInputKind.Text, new[] { ".omd", ".office.md" }, ".pptx", "OfficeIMO.Markup.PowerPoint", "OfficeMarkupParser.Parse(markup, options).Document.ToPowerPointPresentationResult(exportOptions)", "Render presentation-profile OfficeIMO Markup into an editable PowerPoint presentation.", OfficeConversionFidelityKind.Editable, "OfficeMarkupPowerPointConversionResult"), @@ -625,6 +627,10 @@ private static (OfficeConversionTextFormattingKind Kind, string Contract) GetTex return (OfficeConversionTextFormattingKind.SyntaxSubset, "Preserves only emphasis, strike, script, and inline styling represented by the supported source and destination syntax profiles; arbitrary family, size, color, casing metadata, and underline variants are not portable."); } + if (source == "DBF/xBase") { + return (OfficeConversionTextFormattingKind.DataOnly, + "DBF carries typed table values, not document typography or layout; binary values become Base64, and DBF indexes, deletion flags and native schema metadata are not exported."); + } if (source == "CSV" || target == "CSV") { return (OfficeConversionTextFormattingKind.DataOnly, "CSV and TSV carry values, delimiters, and records only; font family, size, color, emphasis, decoration, scripts, casing metadata, and layout are intentionally not representable."); diff --git a/OfficeIMO.Core/Compatibility/OfficeConversionSupportAssessments.cs b/OfficeIMO.Core/Compatibility/OfficeConversionSupportAssessments.cs index 56c5229ae1..05d381b3c0 100644 --- a/OfficeIMO.Core/Compatibility/OfficeConversionSupportAssessments.cs +++ b/OfficeIMO.Core/Compatibility/OfficeConversionSupportAssessments.cs @@ -160,6 +160,9 @@ internal static OfficeConversionSupportAssessment Get(string routeId) { "csv-xlsx" or "xlsx-csv" => Established( "Delimited-data fixtures cover streams, files, delimiter detection, typed cell import, worksheet-range export, reopen behavior, and cancellation.", "CSV and TSV have no font, rich-text, formula, multi-sheet, drawing, or layout model; those workbook semantics are intentionally absent from CSV output."), + "dbf-csv" or "dbf-xlsx" => Targeted( + "Independent Python-produced dBASE III, FoxPro 2 and Visual FoxPro fixtures exercise typed rows, deleted-row filtering, DBT/FPT memos, binary Base64, CSV reopen, and streaming/buffered XLSX reopen and Open XML validation.", + "Requires DBAClientX.Dbf's bounded field profiles. Native application acceptance and Visual FoxPro 31/32 generations remain unqualified. Output omits DBF indexes, deletion flags and native schema metadata; CSV nulls need a selected marker, and XLSX follows Excel numeric precision and cell text limits."), "officemarkup-docx" or "officemarkup-xlsx" or "officemarkup-pptx" => Established( "Typed profile fixtures verify parser diagnostics, editable destination artifacts, font family/size/color, bold, italic, underline variants, strike, scripts, case transforms, small caps where native, and target-specific layout.", "OfficeIMO Markup is a directed authoring format rather than a lossless Office round trip; unsupported blocks and destination-only effects are diagnosed, simplified, or omitted according to exporter options."), diff --git a/OfficeIMO.DjVu.Tests/ReaderAdapterTests.cs b/OfficeIMO.DjVu.Tests/ReaderAdapterTests.cs index 555782b9e8..9363c4144a 100644 --- a/OfficeIMO.DjVu.Tests/ReaderAdapterTests.cs +++ b/OfficeIMO.DjVu.Tests/ReaderAdapterTests.cs @@ -93,7 +93,7 @@ public void DjVuTransportRequiresItsVersionedSchemaAndRetainsGeometry() { var rich = DjVuDocument.Load(Fixture("reader-book.djvu")).ToReadResult(); string json = OfficeDocumentReadResultJson.Serialize(rich); var restored = OfficeDocumentReadResultJson.Deserialize(json); - Assert.Equal(12, restored.SchemaVersion); + Assert.Equal(OfficeDocumentReadResultSchema.CurrentVersion, restored.SchemaVersion); Assert.Equal(ReaderInputKind.DjVu, restored.Kind); Assert.Equal(rich.Blocks[0].Region!.X, restored.Blocks[0].Region!.X); Assert.Equal(4, restored.GetTotalPageCount()); diff --git a/OfficeIMO.Excel/ExcelDocument.DirectDataSet.TableModel.cs b/OfficeIMO.Excel/ExcelDocument.DirectDataSet.TableModel.cs index 80e7db1219..e4c1a64406 100644 --- a/OfficeIMO.Excel/ExcelDocument.DirectDataSet.TableModel.cs +++ b/OfficeIMO.Excel/ExcelDocument.DirectDataSet.TableModel.cs @@ -423,7 +423,7 @@ private static bool RequiresEagerValueSnapshot(Type type) } default: try { - string text = value.ToString() ?? string.Empty; + string text = value is byte[] bytes ? System.Convert.ToBase64String(bytes) : value.ToString() ?? string.Empty; CoerceValueHelper.ValidateSharedStringLength(text, nameof(value)); return text; } catch (Exception exception) { diff --git a/OfficeIMO.Excel/ExcelDocument.DirectDataSet.Writer.Cells.cs b/OfficeIMO.Excel/ExcelDocument.DirectDataSet.Writer.Cells.cs index 79048e29b5..3a9402543b 100644 --- a/OfficeIMO.Excel/ExcelDocument.DirectDataSet.Writer.Cells.cs +++ b/OfficeIMO.Excel/ExcelDocument.DirectDataSet.Writer.Cells.cs @@ -257,6 +257,9 @@ internal static void WriteCellValue(TextWriter writer, object? value, Func1" : " t=\"b\">0"); return; diff --git a/OfficeIMO.Excel/README.md b/OfficeIMO.Excel/README.md index c0e3fb4fae..81b4325cb3 100644 --- a/OfficeIMO.Excel/README.md +++ b/OfficeIMO.Excel/README.md @@ -1756,7 +1756,7 @@ This table is generated from the package-neutral OfficeIMO operation catalog. Th | Inspect | 6 | 0 | 0 | 0 | 0 | 0 | | Validate | 4 | 0 | 0 | 0 | 0 | 2 | | Remove | 4 | 0 | 0 | 0 | 1 | 0 | -| Convert | 48 | 39 | 0 | 9 | 0 | 0 | +| Convert | 48 | 40 | 0 | 9 | 0 | 0 | | Export | 5 | 0 | 0 | 0 | 0 | 0 | The complete rows for `OfficeIMO.Excel` are published in the [generated operation contract](https://github.com/EvotecIT/OfficeIMO/blob/master/Docs/Compatibility/generated/package-operations.md). diff --git a/OfficeIMO.Reader.All/OfficeDocumentReaderBuilderAllExtensions.cs b/OfficeIMO.Reader.All/OfficeDocumentReaderBuilderAllExtensions.cs index 8a92873b81..cba06b7692 100644 --- a/OfficeIMO.Reader.All/OfficeDocumentReaderBuilderAllExtensions.cs +++ b/OfficeIMO.Reader.All/OfficeDocumentReaderBuilderAllExtensions.cs @@ -1,6 +1,7 @@ using OfficeIMO.Reader.AsciiDoc; using OfficeIMO.Reader.Csv; using OfficeIMO.Reader.Chm; +using OfficeIMO.Reader.Dbf; using OfficeIMO.Reader.DocBook; using OfficeIMO.Reader.Email; using OfficeIMO.Reader.Epub; @@ -55,6 +56,7 @@ public static OfficeDocumentReaderBuilder AddAllOfficeIMOHandlers( .AddAsciiDocHandler(configured.AsciiDoc) .AddCsvHandler(configured.Csv) .AddChmHandler(configured.Chm) + .AddDbfHandler(configured.Dbf) .AddDocBookHandler(configured.DocBook) .AddDjVuHandler(configured.DjVu) .AddEmailHandlers(configured.Email) diff --git a/OfficeIMO.Reader.All/OfficeIMO.Reader.All.csproj b/OfficeIMO.Reader.All/OfficeIMO.Reader.All.csproj index 7f97e25352..4662506fbd 100644 --- a/OfficeIMO.Reader.All/OfficeIMO.Reader.All.csproj +++ b/OfficeIMO.Reader.All/OfficeIMO.Reader.All.csproj @@ -31,6 +31,7 @@ + diff --git a/OfficeIMO.Reader.All/README.md b/OfficeIMO.Reader.All/README.md index 49fb43dbd1..a693aced7e 100644 --- a/OfficeIMO.Reader.All/README.md +++ b/OfficeIMO.Reader.All/README.md @@ -44,7 +44,7 @@ Email attachments, EPUB resources and chapter pagination, and Visio preview-versus-semantic fallback behavior follow `PdfProjectionOptions` and are reported as structured conversion evidence. -The preset adds Word, Excel, their safe legacy-word and legacy-spreadsheet families, PowerPoint, Markdown, direct email artifacts, Outlook stores and OAB address books, plus AsciiDoc, CHM, CSV/TSV, DjVu, DocBook, EPUB, HTML/MHTML, Apple Pages/Numbers/Keynote, standalone images, JSON, LaTeX, Jupyter Notebook, offline OneNote, OpenDocument, OPML, PDF, RTF, subtitles, Visio, XML, XPS/OpenXPS, YAML, and ZIP handlers. `OfficeIMO.Reader.Core` itself contains no format parser. +The preset adds Word, Excel, their safe legacy-word and legacy-spreadsheet families, PowerPoint, Markdown, direct email artifacts, Outlook stores and OAB address books, plus AsciiDoc, CHM, CSV/TSV, DBF/xBase, DjVu, DocBook, EPUB, HTML/MHTML, Apple Pages/Numbers/Keynote, standalone images, JSON, LaTeX, Jupyter Notebook, offline OneNote, OpenDocument, OPML, PDF, RTF, subtitles, Visio, XML, XPS/OpenXPS, YAML, and ZIP handlers. `OfficeIMO.Reader.Core` itself contains no format parser. DBF memo sidecars require opt-in through `ReaderAllOptions.Dbf.AllowMemoSidecarReads`. Configure a format through one options object: diff --git a/OfficeIMO.Reader.All/ReaderAllOptions.cs b/OfficeIMO.Reader.All/ReaderAllOptions.cs index a663230717..d613f1367b 100644 --- a/OfficeIMO.Reader.All/ReaderAllOptions.cs +++ b/OfficeIMO.Reader.All/ReaderAllOptions.cs @@ -26,6 +26,9 @@ public sealed class ReaderAllOptions { /// Gets or sets CSV and TSV adapter options. public Csv.CsvReadOptions? Csv { get; set; } + /// Gets or sets DBF/xBase adapter options. Memo sidecar reads are disabled by default. + public Dbf.ReaderDbfOptions? Dbf { get; set; } + /// Gets or sets DocBook adapter options. public DocBook.ReaderDocBookOptions? DocBook { get; set; } diff --git a/OfficeIMO.Reader.Core/OfficeDocumentReadResultJson.cs b/OfficeIMO.Reader.Core/OfficeDocumentReadResultJson.cs index 2e6dc77535..9bda5f0367 100644 --- a/OfficeIMO.Reader.Core/OfficeDocumentReadResultJson.cs +++ b/OfficeIMO.Reader.Core/OfficeDocumentReadResultJson.cs @@ -268,13 +268,15 @@ private static void EnsureKindSupported(int schemaVersion, ReaderInputKind kind) bool requiresVersion10 = kind == ReaderInputKind.Xps; bool requiresVersion11 = kind == ReaderInputKind.Chm; bool requiresVersion12 = kind == ReaderInputKind.DjVu; + bool requiresVersion13 = kind == ReaderInputKind.Dbf; if (!Enum.IsDefined(typeof(ReaderInputKind), kind) || schemaVersion < 6 && requiresVersion6 || schemaVersion < 7 && requiresVersion7 || schemaVersion < 8 && requiresVersion8 || schemaVersion < 10 && requiresVersion10 || schemaVersion < 11 && requiresVersion11 || - schemaVersion < 12 && requiresVersion12) { + schemaVersion < 12 && requiresVersion12 || + schemaVersion < 13 && requiresVersion13) { throw new JsonException( $"Reader input kind '{kind}' is not supported by document read result schema version {schemaVersion}."); } diff --git a/OfficeIMO.Reader.Core/OfficeDocumentReadResultSchema.cs b/OfficeIMO.Reader.Core/OfficeDocumentReadResultSchema.cs index f9703ed584..bfa96834ac 100644 --- a/OfficeIMO.Reader.Core/OfficeDocumentReadResultSchema.cs +++ b/OfficeIMO.Reader.Core/OfficeDocumentReadResultSchema.cs @@ -17,17 +17,17 @@ public static partial class OfficeDocumentReadResultSchema { /// /// Current schema version emitted and accepted by this package. /// - public const int CurrentVersion = 12; + public const int CurrentVersion = 13; /// - /// Stable JSON Schema identifier for current version 12 payloads. + /// Stable JSON Schema identifier for current version 13 payloads. /// - public const string JsonSchemaId = "urn:officeimo:schema:document-read-result:12"; + public const string JsonSchemaId = "urn:officeimo:schema:document-read-result:13"; /// /// File name used for the packaged current JSON Schema artifact. /// - public const string JsonSchemaFileName = "officeimo.document.read-result.v12.schema.json"; + public const string JsonSchemaFileName = "officeimo.document.read-result.v13.schema.json"; /// /// Returns true when a schema header can be consumed by this package. diff --git a/OfficeIMO.Reader.Core/OfficeIMO.Reader.Core.csproj b/OfficeIMO.Reader.Core/OfficeIMO.Reader.Core.csproj index 088da81441..9ae75d2378 100644 --- a/OfficeIMO.Reader.Core/OfficeIMO.Reader.Core.csproj +++ b/OfficeIMO.Reader.Core/OfficeIMO.Reader.Core.csproj @@ -28,6 +28,8 @@ + + diff --git a/OfficeIMO.Reader.Core/README.md b/OfficeIMO.Reader.Core/README.md index 76093b2534..0edab8ea2c 100644 --- a/OfficeIMO.Reader.Core/README.md +++ b/OfficeIMO.Reader.Core/README.md @@ -146,14 +146,14 @@ flattened chunks. Each item supplies its virtual `Path` and complete `Document`, forms, metadata, diagnostics and assets. Identifiers inside a child result remain local to that child. Asset payload bytes stay in memory when requested and remain excluded from JSON transport. -Document transport schema version 12 adds native DjVu input identity (`ReaderInputKind.DjVu`, 28). -Version 11 adds CHM input identity (`ReaderInputKind.Chm`, 27). Version 10 adds -XPS/OpenXPS identity and explicit native `ReaderLocation.LogicalOrder` across -physical containers; version 9 includes recursive `nestedDocuments`. -Versions 5 through 11 remain readable. Load the matching artifact with -`OfficeDocumentReadResultSchema.GetJsonSchema(version)`. Versions below 9 cannot -carry nested documents, versions below 10 cannot carry XPS input kinds, versions -below 11 cannot carry CHM input kinds, and versions below 12 cannot carry DjVu input kinds. +Document transport schema version 13 adds DBF input identity (`ReaderInputKind.Dbf`, 29). +Version 12 adds DjVu identity (`ReaderInputKind.DjVu`, 28); version 11 adds CHM identity +(`ReaderInputKind.Chm`, 27). Version 10 adds XPS/OpenXPS identity and explicit native +`ReaderLocation.LogicalOrder` across physical containers; version 9 includes recursive +`nestedDocuments`. Versions 5 through 12 remain readable. Load the matching artifact +with `OfficeDocumentReadResultSchema.GetJsonSchema(version)`. Versions below 9 cannot +carry nested documents, below 10 cannot carry XPS input kinds, below 11 cannot carry +CHM input kinds, below 12 cannot carry DjVu input kinds, and below 13 cannot carry DBF input kinds. Capability manifest version 6 includes incremental route flags and per-extension `FormatQualifications`. A qualification records a format ID, extraction maturity, profile, preservation, limitations and evidence diff --git a/OfficeIMO.Reader.Core/ReaderModels.cs b/OfficeIMO.Reader.Core/ReaderModels.cs index d3466cc8d1..6fd2499ae2 100644 --- a/OfficeIMO.Reader.Core/ReaderModels.cs +++ b/OfficeIMO.Reader.Core/ReaderModels.cs @@ -118,7 +118,9 @@ public enum ReaderInputKind { /// Compiled HTML Help archive. Chm = 27, /// DjVu scanned document with optional stored text. - DjVu = 28 + DjVu = 28, + /// DBF/xBase table. + Dbf = 29 } /// diff --git a/OfficeIMO.Reader.Core/Schemas/officeimo.document.read-result.v13.schema.json b/OfficeIMO.Reader.Core/Schemas/officeimo.document.read-result.v13.schema.json new file mode 100644 index 0000000000..9c133fef12 --- /dev/null +++ b/OfficeIMO.Reader.Core/Schemas/officeimo.document.read-result.v13.schema.json @@ -0,0 +1,243 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "urn:officeimo:schema:document-read-result:13", + "title": "OfficeIMO document read result", + "description": "Stable OfficeIMO.Reader transport envelope, schema version 13.", + "type": "object", + "required": [ + "schemaId", + "schemaVersion", + "kind", + "source", + "capabilitiesUsed", + "chunks", + "metadata", + "pages", + "blocks", + "tables", + "assets", + "links", + "forms", + "ocrCandidates", + "visuals", + "diagnostics", + "nestedDocuments" + ], + "properties": { + "schemaId": { + "const": "officeimo.document.read-result" + }, + "schemaVersion": { + "const": 13 + }, + "kind": { + "enum": [ + "Unknown", + "Word", + "Excel", + "PowerPoint", + "Markdown", + "Text", + "Pdf", + "Csv", + "Json", + "Xml", + "Html", + "Zip", + "Epub", + "Visio", + "Yaml", + "Rtf", + "OpenDocument", + "AsciiDoc", + "Latex", + "Email", + "OneNote", + "Calendar", + "VCard", + "Opml", + "DocBook", + "IWork", + "Xps", + "Chm", + "DjVu", + "Dbf" + ] + }, + "source": { + "$ref": "#/$defs/source" + }, + "capabilitiesUsed": { + "$ref": "#/$defs/stringArray" + }, + "markdown": { + "type": "string" + }, + "html": { + "type": "string" + }, + "json": { + "type": "string" + }, + "chunks": { + "$ref": "#/$defs/objectArray" + }, + "metadata": { + "$ref": "#/$defs/objectArray" + }, + "pages": { + "$ref": "#/$defs/objectArray" + }, + "blocks": { + "$ref": "#/$defs/objectArray" + }, + "tables": { + "$ref": "#/$defs/objectArray" + }, + "assets": { + "$ref": "#/$defs/objectArray" + }, + "links": { + "$ref": "#/$defs/objectArray" + }, + "forms": { + "$ref": "#/$defs/objectArray" + }, + "ocrCandidates": { + "$ref": "#/$defs/objectArray" + }, + "visuals": { + "$ref": "#/$defs/objectArray" + }, + "diagnostics": { + "type": "array", + "items": { + "$ref": "#/$defs/diagnostic" + } + }, + "nestedDocuments": { + "type": "array", + "items": { + "type": "object", + "required": [ + "path", + "document" + ], + "properties": { + "path": { + "type": "string" + }, + "document": { + "$ref": "#" + } + }, + "additionalProperties": false + } + } + }, + "additionalProperties": false, + "$defs": { + "stringArray": { + "type": "array", + "items": { + "type": "string" + } + }, + "objectArray": { + "type": "array", + "items": { + "type": "object" + } + }, + "source": { + "type": "object", + "properties": { + "path": { + "type": "string" + }, + "sourceId": { + "type": "string" + }, + "sourceHash": { + "type": "string" + }, + "lastWriteUtc": { + "type": "string", + "format": "date-time" + }, + "lengthBytes": { + "type": "integer" + }, + "title": { + "type": "string" + }, + "author": { + "type": "string" + }, + "subject": { + "type": "string" + }, + "keywords": { + "type": "string" + } + }, + "additionalProperties": false + }, + "diagnostic": { + "type": "object", + "required": [ + "severity", + "category", + "code", + "message", + "attributes" + ], + "properties": { + "severity": { + "enum": [ + "Information", + "Warning", + "Error" + ] + }, + "category": { + "enum": [ + "General", + "Detection", + "Input", + "Parsing", + "Content", + "Security", + "Ocr", + "Limit", + "Adapter" + ] + }, + "code": { + "type": "string", + "minLength": 1, + "pattern": "\\S" + }, + "message": { + "type": "string" + }, + "source": { + "type": "string" + }, + "isRecoverable": { + "type": "boolean" + }, + "location": { + "type": "object" + }, + "attributes": { + "type": "object", + "additionalProperties": { + "type": "string" + } + } + }, + "additionalProperties": false + } + } +} diff --git a/OfficeIMO.Reader.Dbf.Tests/DbfConversionTests.cs b/OfficeIMO.Reader.Dbf.Tests/DbfConversionTests.cs new file mode 100644 index 0000000000..72a5e9d483 --- /dev/null +++ b/OfficeIMO.Reader.Dbf.Tests/DbfConversionTests.cs @@ -0,0 +1,88 @@ +using System.Data; +using System.Globalization; +using DBAClientX.Dbf; +using DocumentFormat.OpenXml.Packaging; +using DocumentFormat.OpenXml.Spreadsheet; +using DocumentFormat.OpenXml.Validation; +using OfficeIMO.CSV; +using OfficeIMO.Excel; + +namespace OfficeIMO.Reader.Dbf.Tests { + public sealed class DbfConversionTests { + [Theory] + [InlineData("db3")] + [InlineData("fp")] + [InlineData("vfp")] + public void CsvUsesTheStandardTypedReaderAndPreservesMemoAndBinaryValues(string profile) { + using DbfDataReader reader = DbfDataReader.Open(DbfReaderTests.Fixture(profile + ".dbf")); + using StringWriter output = new StringWriter(CultureInfo.InvariantCulture); + CsvDocument.WriteDataReader(output, reader, new CsvSaveOptions { DateTimeFormat = "O", NullValue = "" }); + Assert.False(reader.IsClosed); + CsvRow[] rows = CsvDocument.Parse(output.ToString()).AsEnumerable().ToArray(); + Assert.Equal(2, rows.Length); + Assert.Equal(profile == "vfp" ? "Café" : "Café £", rows[0][0]); + if (profile == "vfp") { + Assert.Equal("-123.4567", rows[0][3]); + Assert.Equal("AP8B", rows[0][7]); + Assert.Equal("", rows[1][0]); + Assert.Equal("2020-02-29T12:34:56.7890000", rows[0][4]); + } else { + Assert.Equal(1234.50m, decimal.Parse((string)rows[0][1]!, CultureInfo.InvariantCulture)); + Assert.Equal("First memo\r\nSecond line: naïve", rows[0][4]); + Assert.Equal("", rows[1][1]); + } + } + + [Theory] + [InlineData("db3", false)] + [InlineData("fp", false)] + [InlineData("vfp", false)] + [InlineData("vfp", true)] + public void ExcelReopensTypedRowsThroughStreamingAndBufferedWriters(string profile, bool buffered) { + using DbfDataReader reader = DbfDataReader.Open(DbfReaderTests.Fixture(profile + ".dbf")); + using MemoryStream output = new MemoryStream(); + ExcelDataSetImportResult result = ExcelDocument.WriteDataReader(output, reader, + new ExcelTabularWriteOptions { UseSharedStrings = buffered, RequireStreaming = !buffered }); + Assert.Equal(2, result.RowCount); + Assert.False(reader.IsClosed); + using (SpreadsheetDocument package = SpreadsheetDocument.Open(output, false)) { + Assert.Empty(new OpenXmlValidator().Validate(package)); + Cell[] cells = package.WorkbookPart!.WorksheetParts.First().Worksheet!.Descendants().Skip(1).First().Elements().ToArray(); + Assert.Equal(profile == "vfp" ? 42m : 1234.50m, decimal.Parse(cells[1].CellValue!.Text, CultureInfo.InvariantCulture)); + } + output.Position = 0; + using ExcelWorkbookDataReader reopened = ExcelDocument.OpenDataReader(output); + Assert.True(reopened.Read()); + Assert.Equal(profile == "vfp" ? "Café" : "Café £", reopened.GetString(0)); + if (profile == "vfp") { + Assert.Equal("AP8B", reopened.GetString(7)); + Assert.Equal(Convert.ToBase64String(new byte[] { 0, 255, 1, 32, 65, 66, 67, 32 }), reopened.GetString(6)); + Assert.Equal(-123.4567, Convert.ToDouble(reopened.GetValue(3), CultureInfo.InvariantCulture)); + } else { + Assert.Equal("First memo\r\nSecond line: naïve", reopened.GetString(4)); + Assert.Equal(1234.50, Convert.ToDouble(reopened.GetValue(1), CultureInfo.InvariantCulture)); + } + Assert.True(reopened.Read()); + Assert.False(reopened.Read()); + } + + [Fact] + public void GenericBinaryConsumersUseBase64InsteadOfTheClrTypeName() { + DataTable table = new DataTable("Data"); + table.Columns.Add("Payload", typeof(byte[])); + table.Rows.Add(new object[] { new byte[] { 0, 255, 1 } }); + using DataTableReader csvReader = table.CreateDataReader(); + using StringWriter text = new StringWriter(CultureInfo.InvariantCulture); + CsvDocument.WriteDataReader(text, csvReader); + Assert.Equal("AP8B", CsvDocument.Parse(text.ToString()).AsEnumerable().Single()[0]); + using MemoryStream output = new MemoryStream(); + DataSet dataSet = new DataSet(); + dataSet.Tables.Add(table); + ExcelDocument.WriteDataSet(output, dataSet); + output.Position = 0; + using ExcelWorkbookDataReader reopened = ExcelDocument.OpenDataReader(output); + Assert.True(reopened.Read()); + Assert.Equal("AP8B", reopened.GetString(0)); + } + } +} diff --git a/OfficeIMO.Reader.Dbf.Tests/DbfJsonContractTests.cs b/OfficeIMO.Reader.Dbf.Tests/DbfJsonContractTests.cs new file mode 100644 index 0000000000..22abb7310d --- /dev/null +++ b/OfficeIMO.Reader.Dbf.Tests/DbfJsonContractTests.cs @@ -0,0 +1,54 @@ +using System.Text.Json; +using System.Text.Json.Nodes; +using OfficeIMO.Reader.Dbf; + +namespace OfficeIMO.Reader.Dbf.Tests { + public sealed class DbfJsonContractTests { + [Fact] + public void NativeDbfResultUsesTheSchemaThatDeclaresItsKind() { + var reader = new OfficeDocumentReaderBuilder().AddDbfHandler().Build(); + string json = reader.ReadDocumentJson(DbfReaderTests.Fixture("plain.dbf")); + using var payload = JsonDocument.Parse(json); + int version = payload.RootElement.GetProperty("schemaVersion").GetInt32(); + using var schema = JsonDocument.Parse(OfficeDocumentReadResultSchema.GetJsonSchema(version)); + Assert.Equal(13, version); + Assert.Equal(27, (int)ReaderInputKind.Chm); + Assert.Equal(28, (int)ReaderInputKind.DjVu); + Assert.Equal(29, (int)ReaderInputKind.Dbf); + Assert.Equal(version, schema.RootElement.GetProperty("properties").GetProperty("schemaVersion").GetProperty("const").GetInt32()); + Assert.Contains(payload.RootElement.GetProperty("kind").GetString(), schema.RootElement.GetProperty("properties").GetProperty("kind").GetProperty("enum").EnumerateArray().Select(value => value.GetString())); + OfficeDocumentReadResult restored = OfficeDocumentReadResultJson.Deserialize(json); + Assert.Equal(ReaderInputKind.Dbf, restored.Kind); + Assert.All(restored.Chunks, chunk => Assert.Equal(ReaderInputKind.Dbf, chunk.Kind)); + } + + [Fact] + public void OlderEnvelopesRejectDbfInRootsChunksAndNestedDocuments() { + for (int version = 5; version <= 12; version++) { + using var schema = JsonDocument.Parse(OfficeDocumentReadResultSchema.GetJsonSchema(version)); + Assert.DoesNotContain("Dbf", schema.RootElement.GetProperty("properties").GetProperty("kind").GetProperty("enum").EnumerateArray().Select(value => value.GetString())); + var result = new OfficeDocumentReadResult { SchemaVersion = version, Kind = ReaderInputKind.Dbf }; + Assert.Throws(() => result.ToJson()); + result.Kind = ReaderInputKind.Text; + string text = result.ToJson(); + Assert.Throws(() => OfficeDocumentReadResultJson.Deserialize(text.Replace("\"kind\":\"Text\"", "\"kind\":\"Dbf\""))); + result.Chunks = new[] { new ReaderChunk { Kind = ReaderInputKind.Dbf } }; + Assert.Throws(() => result.ToJson()); + result.Chunks = new[] { new ReaderChunk { Kind = ReaderInputKind.Text } }; + text = result.ToJson(); + JsonNode chunkEnvelope = JsonNode.Parse(text)!; + chunkEnvelope["chunks"]![0]!["kind"] = "Dbf"; + Assert.Throws(() => OfficeDocumentReadResultJson.Deserialize(chunkEnvelope.ToJsonString())); + if (version < 9) continue; + result.Chunks = Array.Empty(); + result.NestedDocuments = new[] { new OfficeDocumentNestedResult { Path = "table.dbf", Document = new OfficeDocumentReadResult { Kind = ReaderInputKind.Dbf } } }; + Assert.Throws(() => result.ToJson()); + result.NestedDocuments[0].Document.Kind = ReaderInputKind.Text; + text = result.ToJson(); + JsonNode nestedEnvelope = JsonNode.Parse(text)!; + nestedEnvelope["nestedDocuments"]![0]!["document"]!["kind"] = "Dbf"; + Assert.Throws(() => OfficeDocumentReadResultJson.Deserialize(nestedEnvelope.ToJsonString())); + } + } + } +} diff --git a/OfficeIMO.Reader.Dbf.Tests/DbfReaderTests.cs b/OfficeIMO.Reader.Dbf.Tests/DbfReaderTests.cs new file mode 100644 index 0000000000..851ca50fe8 --- /dev/null +++ b/OfficeIMO.Reader.Dbf.Tests/DbfReaderTests.cs @@ -0,0 +1,85 @@ +using DBAClientX.Dbf; +using OfficeIMO.Reader.Dbf; + +namespace OfficeIMO.Reader.Dbf.Tests { + public sealed class DbfReaderTests { + internal static string Fixture(string name) => Path.Combine(AppContext.BaseDirectory, "Fixtures", name); + + [Fact] + public void ReaderProjectsNativeSchemaAndValuesWithPhysicalLocations() { + var reader = new OfficeDocumentReaderBuilder().AddDbfHandler(new ReaderDbfOptions { ChunkRows = 1 }).Build(); + ReaderChunk[] chunks = reader.Read(Fixture("plain.dbf")).ToArray(); + Assert.Equal(2, chunks.Length); + Assert.Equal(ReaderInputKind.Dbf, chunks[0].Kind); + ReaderTable table = Assert.Single(chunks[0].Tables!); + Assert.Equal(new[] { "NAME", "AMOUNT", "ACTIVE" }, table.Columns); + Assert.Equal("Café |