From d564a719b1cce16c25ebea53a537393cb757f5d1 Mon Sep 17 00:00:00 2001 From: John Lambert Date: Fri, 17 Jul 2026 07:11:15 -0400 Subject: [PATCH] Add LCM Grammar JSON: a deterministic grammar export with its contract LCM Grammar JSON ({"format":"lcm-grammar","version":1}) is a deterministic, GUID-keyed JSON projection of the parser-relevant subset of a project - phonology, morphology, and lexicon - for external morphological-parser tooling: grammar verification ("does this change parse better?"), conformance fixtures, and field deployment (a ~56 MB .fwdata project exports to ~2.4 MB pretty-printed, ~250 KB gzipped, loadable by web/wasm parsers without LibLCM). It is a read-only projection: not an editing, synchronization, or storage format. The exporter, SIL.LCModel.DomainServices.GrammarJsonServices, sits next to M3ModelExportServices and follows the same pattern: one static service class, no new project, no new runtime dependency. The contract ships with the code and cannot drift from it: - doc/lcm-grammar.md - the specification: conventions (GUID keys, determinism/ordering rules incl. normative key order, omission semantics, tagged unions, never-silent skips), per-section structure, a referential-integrity section stating each reference field's resolution scope (the schema checks GUID shape, never existence), the additive-only versioning policy, and an appendix of built-in FieldWorks GUIDs (word boundary, the 17 morph types) so importers can be self-contained. - doc/lcm-grammar.schema.json - strict JSON Schema (draft-07), enforced by unit tests: empty-project and populated-project exports must validate. Determinism: owning/reference sequences keep model order (rule order, slot order, allomorph disjunctive order are semantic); unordered collections sort by GUID string; multistring values sort by writing- system tag. Two exports of the same data are byte-identical. Unrepresentable data (unknown morph types, dangling references, malformed parser-parameter XML) is skipped with a message in an optional warnings collection, never silently. An affix process with an unrepresentable input part is skipped whole, since its output mappings reference input parts by position. 17 NUnit tests (memory-only backend) cover schema validation, byte- determinism, entry/phonology/MSA/sense round-trips, GUID sort order, parser-parameter parsing, affix processes, ad hoc rules, entry-ref discrimination, and warn-on-skip behavior. The exporter was reviewed field-by-field against the format's independent reference implementation (a Rust .fwdata reader producing the same format); a cross-implementation byte-equality gate on real projects is a planned follow-up. Co-Authored-By: Claude Fable 5 --- CHANGELOG.md | 1 + README.md | 11 + doc/lcm-grammar.md | 220 +++ doc/lcm-grammar.schema.json | 744 ++++++++ .../DomainServices/GrammarJsonServices.cs | 1576 +++++++++++++++++ .../GrammarJsonServicesTests.cs | 582 ++++++ .../SIL.LCModel.Tests.csproj | 5 + 7 files changed, 3139 insertions(+) create mode 100644 doc/lcm-grammar.md create mode 100644 doc/lcm-grammar.schema.json create mode 100644 src/SIL.LCModel/DomainServices/GrammarJsonServices.cs create mode 100644 tests/SIL.LCModel.Tests/DomainServices/GrammarJsonServicesTests.cs diff --git a/CHANGELOG.md b/CHANGELOG.md index 23a4b2c6..f42160d0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,7 @@ and this project adheres to [Semantic Versioning](http://semver.org/). ### Added +- [SIL.LCModel] Add GrammarJsonServices.ExportGrammar: deterministic "LCM Grammar JSON" export of the parser-relevant subset of a project (phonology, morphology, lexicon), with the format specification and JSON Schema in doc/ (lcm-grammar.md, lcm-grammar.schema.json). - [SIL.LCModel] Add new virtual property LicenseTSS in CmPicture, to access info about the picture's copyright and license. - [SIL.LCModel] Add new virtual property CreatorTSS in CmPicture, to access info about the picture's creator. - [SIL.LCModel] Add SpecificItemAndFieldName() to ISenseOrEntry diff --git a/README.md b/README.md index 6419b977..bb269625 100644 --- a/README.md +++ b/README.md @@ -8,6 +8,17 @@ The liblcm library is the core [FieldWorks](https://github.com/sillsdev/FieldWor linguistic analyses of languages. Tools in this library provide the ability to store and interact with language and culture data, including anthropological, text corpus, and linguistics data. +## LCM Grammar JSON + +liblcm defines and exports **LCM Grammar JSON** — a deterministic, GUID-keyed JSON projection of +the parser-relevant subset of a project (phonology, morphology, lexicon) for external +morphological-parser tooling: grammar verification, conformance fixtures, and field deployment. +Export with `SIL.LCModel.DomainServices.GrammarJsonServices.ExportGrammar(cache, writer)`. The +format contract lives in this repository: the specification is +[doc/lcm-grammar.md](doc/lcm-grammar.md) and the machine-checkable schema is +[doc/lcm-grammar.schema.json](doc/lcm-grammar.schema.json) (enforced against the exporter by unit +tests). It is a read-only projection — not an editing, synchronization, or storage format. + ## Instructions 1. Install Required Software diff --git a/doc/lcm-grammar.md b/doc/lcm-grammar.md new file mode 100644 index 00000000..805e56c1 --- /dev/null +++ b/doc/lcm-grammar.md @@ -0,0 +1,220 @@ +# LCM Grammar JSON (format `lcm-grammar`, version 1) + +A deterministic, GUID-keyed JSON projection of the **parser-relevant subset** of a FieldWorks/LCM +project: phonology, morphology, and lexicon — everything a morphological parser needs, and (aside +from glosses, definitions, and names) nothing else. Produced by +`SIL.LCModel.DomainServices.GrammarJsonServices.ExportGrammar`; the machine-checkable structure is +[`lcm-grammar.schema.json`](lcm-grammar.schema.json) (validated against the exporter by +`GrammarJsonServicesTests`). + +**What it is for:** grammar verification tooling ("does this change parse better?"), conformance +fixtures for morphological-parser test suites, and field deployment (a small artifact that +web/wasm parsers can load without LibLCM — as one measured data point, a real ~56 MB `.fwdata` +project exported to roughly 2.4 MB pretty-printed, ~250 KB gzipped). + +**What it is not:** an editing format, a synchronization/merge format, or a replacement for +`.fwdata`. It is a read-only projection; LCM remains the authority. Consuming an LCM Grammar JSON +document never requires LibLCM or FieldWorks — that independence is the point. + +## 1. Conventions + +- **Envelope.** Every document is `{ "format": "lcm-grammar", "version": 1, ... }`. Consumers + should reject any other format tag or major version. +- **Naming.** camelCase field names owned by this spec — not a mirror of LCM class/property names. + The exporter source documents which LCM property each field originates from. +- **Cross-references.** Every reference to another object in the document is a FieldWorks GUID, + rendered lowercase-hyphenated (`Guid.ToString()`'s default format) — never an `Hvo` + (FieldWorks' in-session integer id, which is not durable). +- **Determinism.** Two exports of the same data are byte-identical, and an independent + implementation reading the same project must be able to reproduce the same bytes: + - Output is pretty-printed with two-space indentation. Keys within each object appear in a + fixed, normative order: the order in which the schema's `properties` lists them. + - LCM owning/reference *sequences* (`OS`/`RS` properties) keep model order — that order is + semantically meaningful (rule order, slot order, allomorph disjunctive order, ...). + - Unordered LCM *collections* (`OC`/`RC` properties, repository enumerations — notably lexical + entries) are sorted by the ordinal comparison of their lowercase GUID string. + - Multistring values are arrays of `{"ws": tag, "form": text}` sorted ordinally by + writing-system tag; empty alternatives are skipped. (Phoneme/boundary-marker + `representations` concatenate codes in model order, sorting per-code.) A `ws` tag **may + repeat** within one array (a phoneme with two codes in the same writing system, for + example) — model these fields as ordered lists of pairs, never as a map keyed by tag. +- **Optional vs. absent.** Absent optional values and empty arrays are omitted entirely, never + written as `null` or `[]` — with one exception: a lexical entry's `allomorphs` array is always + present, even when empty. An omitted field means "no data"; consumers must not distinguish + omission from emptiness. +- **Tagged unions.** Polymorphic objects (natural classes, phonological rules, pattern contexts, + compound rules, ad hoc rules, rule mappings, MSAs, entry refs) carry a `kind` discriminator. +- **Never-silent skips.** Data the format cannot represent (an unknown morph type, a dangling + reference, malformed parser-parameter XML) is skipped by the exporter with a message in its + optional warnings collection — the document itself contains only well-formed data. +- **Unicode normalization.** All exported text is NFC. (LCM holds strings NFD in memory; the raw + `.fwdata` XML at rest is NFC — exporting NFC lets an independent reader of the raw XML + reproduce the same bytes.) +- **Name fields.** Name-like multistrings export the best analysis alternative, falling back to + the best vernacular alternative (phoneme names, for example, are usually authored only in a + vernacular writing system). A field with no usable value is exported as `""` (FieldWorks' + internal `"***"` missing-value marker is normalized to `""`). + +## 2. Document structure + +Top level: `format`, `version`, `project`, `featureSystems`, `phonology`, `morphology`, +`lexicon` — all always present. See the JSON Schema for every field's exact shape; the highlights +per section: + +### `project` +`name`; `vernacularWritingSystems` / `analysisWritingSystems` (ICU tags, default writing system +first, then the remaining current writing systems in project order). + +### `featureSystems` +`phonological` and `morphosyntactic` — FieldWorks' two independent feature systems, each with +`closedFeatures` (values enumerated as guid/name/abbreviation symbols) and `complexFeatures`. +Feature *structures* appear throughout the document as +`{"values":[{"feature": guid, "value": {"kind": "closed"|"complex", "value": ...}}]}`, recursive +through complex values. A structure's features resolve against whichever feature system its host +belongs to; the two are never mixed in one structure. A complex feature's `featureType` is an +**intentionally opaque** reference to FieldWorks' feature-structure-type object, which this +format does not project — it never resolves in-document; treat it as an opaque grouping key or +ignore it. + +### `phonology` +`phonemes` (per-writing-system `representations` with FieldWorks' dotted-circle placeholder +U+25CC stripped; optional `features`, `basicIpaSymbol`), `boundaryMarkers` (excluding the built-in +word-boundary marker — see appendix), `naturalClasses` (`kind:"segments"` extensional / +`kind:"features"` intensional; `name` is the FieldWorks *abbreviation*, which is what environment +strings reference), `environments` (the raw FieldWorks environment string, +untokenized — these follow FieldWorks' phonological-environment syntax: `/` introduces the +context, `_` is the target slot, `#` a word boundary, `[...]` a natural-class *abbreviation*, +`(...)` optional material; e.g. `/[V+mid] ([preNas]) _ #`; this format carries the string +verbatim and consumers that evaluate environments own its tokenization), `rules` (rewrite and +metathesis, in `OrderNumber` order, disabled rules excluded), and `featureConstraints` +(alpha-variable slots). + +Pattern positions use the recursive `PhonContext` union: `sequence`, `iteration` (`min` is +always ≥ 0; `max` = -1 means unbounded), `segment`, `naturalClass` (with `plusVariables`/`minusVariables` alpha-variable +agreement), `boundary`, `wordBoundary`, `variable`. Rewrite rules carry `direction` +(`leftToRight` | `rightToLeft` | `simultaneous`), a structural description, and one or more +right-hand sides (structural change, left/right context, required parts of speech, +required/excluded rule features). A rewrite rule's `featureConstraintVariables` lists its +alpha-variable feature constraints **in assignment order** — consumers assign variable names +(α, β, γ, ...) in exactly this order, so the order is semantic, not cosmetic. Metathesis rules carry 0-based `leftSwitchIndex` / +`rightSwitchIndex` into their structural description. + +### `morphology` +`partsOfSpeech` (the full possibility tree, with per-POS inflection classes (recursive), +`defaultInflectionClass`, `inflectableFeatures`, stem names (feature-structure `regions`), affix +slots, and affix templates (`prefixSlots` innermost-to-outermost, `suffixSlots` in order, +`isFinal`, disabled templates included with their flag)); `compoundRules` (`endocentric` / +`exocentric`, disabled included); `adhocProhibitions` (allomorph- and morpheme-level, with +`adjacency`: `anywhere` | `somewhereToLeft` | `somewhereToRight` | `adjacentToLeft` | +`adjacentToRight`); `exceptionFeatures` (the merged registry of productivity restrictions and +possibility-typed phonological rule features); `lexEntryInflTypes` (irregularly-inflected-form +variant types, with `glossPrepend`/`glossAppend` and template `slots`); and `parserParameters` +(parsed from FieldWorks' stored XML block: `notOnClitics` **defaults to true when absent**, +`acceptUnspecifiedGraphemes` and `noDefaultCompounding` default to false, optional raw `strata` +string, optional per-compound-rule `compoundRuleMaxApplications`). + +### `lexicon` +`entries`, GUID-sorted. Each entry: `citationForm` (optional — when absent, the conventional +headword is the lexeme form's `forms`, i.e. the **last** allomorph's), `lexemeMorphType` (the +lexeme form's morph type — see appendix for the closed enum), `allomorphs` (**alternate forms +first, lexeme form last** — this order carries allomorph-selection semantics), `msas`, `senses`, +`entryRefs`. An allomorph's `isAbstract` marks an underlying/abstract form rather than a surface +form; parsers conventionally exclude abstract allomorphs from surface matching. + +- **Allomorphs** cover stem allomorphs (`environments`, `stemName`), affix allomorphs + (`environments`, `positions`, `inflectionClasses`, `msEnvFeatures`, `msEnvPartOfSpeech`), and + affix processes (`process` with `input` pattern parts and `output` mappings: + `insertNaturalClass`, `copyFromInput`, `insertSegments`, `modifyFromInput`). `copyFromInput` / + `modifyFromInput` reference input parts **positionally** (1-based index into `input`); an affix + process with an unrepresentable input part is therefore skipped whole rather than emitted with + misaligned indices. Forms may contain FieldWorks' lexical-pattern bracket notation (e.g. + `[C][V]d`) verbatim; tokenizing it is the consumer's concern. +- **MSAs** are a tagged union: `stem`, `inflectional` (empty `slots` means the affix applies + outside any template), `derivational` (from/to pairs), `unclassified`. +- **Senses** are the entry's sense tree flattened pre-order (parent before its subsenses), each + with `gloss`, `definition`, and an `msa` reference. Resolve `msa` against the **document-wide + union** of every entry's `msas`, not just the owning entry's: real projects contain stray + senses whose MSA belongs to a different entry (the exporter emits a warning when it sees one, + but carries the reference as-is). +- **Entry refs** are `variant` or `complexForm`: a ref is a `complexForm` only when it has + complex-form types and no variant types; anything else (variant types only, both kinds, or + neither) is a `variant`. `componentLexemes` may reference entries **or** senses. + +## 3. Referential integrity — what the schema does not check + +The JSON Schema validates structure and GUID *shape* only, never reference *existence*. Every +consumer must run its own resolution pass. GUIDs are globally unique across the whole document +(entries, senses, MSAs, allomorphs, phonemes, ... never collide), so build **one document-wide +guid → object index**; per-category indexes are insufficient for the starred rows below. Scope +of each reference field: + +| Reference | Resolves against | +|---|---| +| `sense.msa` * | document-wide union of every entry's `msas` (usually, but not always, the owning entry's) | +| `entryRef.componentLexemes` * | union of all entry guids **and** all sense guids | +| `entryRef.variantEntryTypes` | `morphology.lexEntryInflTypes`, **or** a plain variant-type possibility this format does not enumerate — unresolvable guids here are normal | +| `entryRef.complexEntryTypes` | complex-form-type possibilities this format does not enumerate — opaque | +| `complexFeature.featureType` | **never resolvable in-document** (see §2) | +| feature-structure `feature` / closed `value` | the host's feature system's features / that feature's `values` | +| `naturalClass.phonemes`, `PhonContext.segment.phoneme` | `phonology.phonemes` | +| `PhonContext.boundary.marker` | `phonology.boundaryMarkers` | +| `PhonContext.naturalClass`, rule-mapping `naturalClass` | `phonology.naturalClasses` | +| `plusVariables`/`minusVariables`, `featureConstraintVariables` | `phonology.featureConstraints` | +| allomorph `environments`/`positions` | `phonology.environments` | +| allomorph/MSA `stemName`, `inflectionClasses`, `partOfSpeech`, `slots` | the `morphology.partsOfSpeech` tree's stem names / inflection classes / own guids / affix slots | +| MSA/compound/rewrite-RHS `exceptionFeatures`/rule features | `morphology.exceptionFeatures` **or** the inflection-class hierarchy (both are valid targets) | +| ad hoc `primary`/`others` | all allomorph guids (allomorph kind) or all MSA guids (morpheme kind) | +| `parserParameters.compoundRuleMaxApplications[].compoundRule` | `morphology.compoundRules` | +| `copyFromInput`/`modifyFromInput` `part` | 1-based position in the **same process's** `input` array (positional, not a guid; always in range in exporter output) | + +Dangling references beyond those noted as expected indicate source-data problems; the exporter +emits a warning for every reference it knows to be stray but still carries representable data. + +## 4. Versioning and evolution + +- The schema in this directory validates exactly what the current exporter emits; exporter and + schema change together, in the same commit. +- Within major version 1, changes are **additive only**: new optional fields may appear; existing + fields never change meaning, type, or optionality. Consumers should ignore fields they do not + recognize. +- Anything that would break a faithful consumer requires a major-version bump of the `version` + field. + +Planned additive extensions (not yet present): writing-system definitions (so a document can be +imported with no source project behind it) and optional export filters with an `omits` envelope +marker. A cross-implementation byte-equality gate against an independent `.fwdata` reader is +planned to keep this spec honest. + +## 5. Appendix: built-in FieldWorks GUIDs + +These well-known objects are referenced by meaning rather than enumerated in the document. They +are constant across all FieldWorks projects. + +**Word boundary** (`PhBdryMarker`, excluded from `boundaryMarkers`; pattern contexts referencing +it are exported as `kind:"wordBoundary"`): `7db635e0-9ef3-4167-a594-12551ed89aaa`. + +**Morph types** (`MoMorphType` possibilities → the `morphType`/`lexemeMorphType` enum): + +| Enum value | GUID | +|---|---| +| `stem` | `d7f713e8-e8cf-11d3-9764-00c04f186933` | +| `boundStem` | `d7f713e7-e8cf-11d3-9764-00c04f186933` | +| `root` | `d7f713e5-e8cf-11d3-9764-00c04f186933` | +| `boundRoot` | `d7f713e4-e8cf-11d3-9764-00c04f186933` | +| `prefix` | `d7f713db-e8cf-11d3-9764-00c04f186933` | +| `suffix` | `d7f713dd-e8cf-11d3-9764-00c04f186933` | +| `infix` | `d7f713da-e8cf-11d3-9764-00c04f186933` | +| `circumfix` | `d7f713df-e8cf-11d3-9764-00c04f186933` | +| `proclitic` | `d7f713e2-e8cf-11d3-9764-00c04f186933` | +| `enclitic` | `d7f713e1-e8cf-11d3-9764-00c04f186933` | +| `clitic` | `c2d140e5-7ca9-41f4-a69a-22fc7049dd2c` | +| `particle` | `56db04bf-3d58-44cc-b292-4c8aa68538f4` | +| `phrase` | `a23b6faa-1052-4f4d-984b-4b338bdaf95f` | +| `discontigPhrase` | `0cc8c35a-cee9-434d-be58-5d29130fba5b` | +| `prefixingInterfix` | `af6537b0-7175-4387-ba6a-36547d37fb13` | +| `infixingInterfix` | `18d9b1c3-b5b6-4c07-b92c-2fe1d2281bd4` | +| `suffixingInterfix` | `3433683d-08a9-4bae-ae53-2a7798f64068` | + +FieldWorks' *simulfix* and *suprafix* morph types, `MoDerivStepMsa` analyses, and coordinate +compound rules (`MoCoordinateCompound`) have no representation in this format; each is skipped +with a warning. diff --git a/doc/lcm-grammar.schema.json b/doc/lcm-grammar.schema.json new file mode 100644 index 00000000..87bf9518 --- /dev/null +++ b/doc/lcm-grammar.schema.json @@ -0,0 +1,744 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema#", + "$id": "https://raw.githubusercontent.com/sillsdev/liblcm/master/doc/lcm-grammar.schema.json", + "title": "LCM Grammar JSON", + "description": "Deterministic, GUID-keyed JSON projection of the parser-relevant subset of a FieldWorks/LCM project (phonology, morphology, lexicon). See lcm-grammar.md. This schema validates exactly what the current exporter emits; it evolves additively with the exporter within a major version.", + "type": "object", + "required": ["format", "version", "project", "featureSystems", "phonology", "morphology", "lexicon"], + "additionalProperties": false, + "properties": { + "format": { "enum": ["lcm-grammar"] }, + "version": { "enum": [1] }, + "project": { + "type": "object", + "required": ["name"], + "additionalProperties": false, + "properties": { + "name": { "type": "string" }, + "vernacularWritingSystems": { "$ref": "#/definitions/wsTagArray" }, + "analysisWritingSystems": { "$ref": "#/definitions/wsTagArray" } + } + }, + "featureSystems": { + "type": "object", + "required": ["phonological", "morphosyntactic"], + "additionalProperties": false, + "properties": { + "phonological": { "$ref": "#/definitions/featureSystem" }, + "morphosyntactic": { "$ref": "#/definitions/featureSystem" } + } + }, + "phonology": { + "type": "object", + "additionalProperties": false, + "properties": { + "phonemes": { "type": "array", "items": { "$ref": "#/definitions/phoneme" } }, + "boundaryMarkers": { "type": "array", "items": { "$ref": "#/definitions/boundaryMarker" } }, + "naturalClasses": { "type": "array", "items": { "$ref": "#/definitions/naturalClass" } }, + "environments": { "type": "array", "items": { "$ref": "#/definitions/environment" } }, + "rules": { "type": "array", "items": { "$ref": "#/definitions/phonologicalRule" } }, + "featureConstraints": { "type": "array", "items": { "$ref": "#/definitions/featureConstraint" } } + } + }, + "morphology": { + "type": "object", + "required": ["parserParameters"], + "additionalProperties": false, + "properties": { + "partsOfSpeech": { "type": "array", "items": { "$ref": "#/definitions/partOfSpeech" } }, + "compoundRules": { "type": "array", "items": { "$ref": "#/definitions/compoundRule" } }, + "adhocProhibitions": { "type": "array", "items": { "$ref": "#/definitions/adhocProhibition" } }, + "exceptionFeatures": { "type": "array", "items": { "$ref": "#/definitions/exceptionFeature" } }, + "lexEntryInflTypes": { "type": "array", "items": { "$ref": "#/definitions/lexEntryInflType" } }, + "parserParameters": { "$ref": "#/definitions/parserParameters" } + } + }, + "lexicon": { + "type": "object", + "additionalProperties": false, + "properties": { + "entries": { "type": "array", "items": { "$ref": "#/definitions/lexEntry" } } + } + } + }, + "definitions": { + "guid": { + "type": "string", + "pattern": "^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$" + }, + "guidArray": { + "type": "array", + "minItems": 1, + "items": { "$ref": "#/definitions/guid" } + }, + "wsTagArray": { + "type": "array", + "minItems": 1, + "items": { "type": "string", "minLength": 1 } + }, + "wsForm": { + "type": "object", + "required": ["ws", "form"], + "additionalProperties": false, + "properties": { + "ws": { "type": "string", "minLength": 1 }, + "form": { "type": "string", "minLength": 1 } + } + }, + "wsFormArray": { + "type": "array", + "minItems": 1, + "items": { "$ref": "#/definitions/wsForm" } + }, + "featureSystem": { + "type": "object", + "additionalProperties": false, + "properties": { + "closedFeatures": { "type": "array", "items": { "$ref": "#/definitions/closedFeature" } }, + "complexFeatures": { "type": "array", "items": { "$ref": "#/definitions/complexFeature" } } + } + }, + "closedFeature": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" }, + "values": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" } + } + } + } + } + }, + "complexFeature": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" }, + "featureType": { "$ref": "#/definitions/guid" } + } + }, + "featureStructure": { + "type": "object", + "additionalProperties": false, + "properties": { + "values": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "required": ["feature", "value"], + "additionalProperties": false, + "properties": { + "feature": { "$ref": "#/definitions/guid" }, + "value": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "value"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["closed"] }, + "value": { "$ref": "#/definitions/guid" } + } + }, + { + "type": "object", + "required": ["kind", "value"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["complex"] }, + "value": { "$ref": "#/definitions/featureStructure" } + } + } + ] + } + } + } + } + } + }, + "phoneme": { + "type": "object", + "required": ["guid", "name"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "representations": { "$ref": "#/definitions/wsFormArray" }, + "features": { "$ref": "#/definitions/featureStructure" }, + "basicIpaSymbol": { "type": "string", "minLength": 1 } + } + }, + "boundaryMarker": { + "type": "object", + "required": ["guid", "name"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "representations": { "$ref": "#/definitions/wsFormArray" } + } + }, + "naturalClass": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid", "name"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["segments"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "phonemes": { "$ref": "#/definitions/guidArray" } + } + }, + { + "type": "object", + "required": ["kind", "guid", "name", "features"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["features"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "features": { "$ref": "#/definitions/featureStructure" } + } + } + ] + }, + "environment": { + "type": "object", + "required": ["guid", "name", "representation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "representation": { "type": "string" } + } + }, + "featureConstraint": { + "type": "object", + "required": ["guid", "feature"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "feature": { "$ref": "#/definitions/guid" } + } + }, + "ruleDirection": { "enum": ["leftToRight", "rightToLeft", "simultaneous"] }, + "phonologicalRule": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid", "name", "direction"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["rewrite"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "direction": { "$ref": "#/definitions/ruleDirection" }, + "structuralDescription": { "$ref": "#/definitions/phonContextArray" }, + "featureConstraintVariables": { "$ref": "#/definitions/guidArray" }, + "rightHandSides": { + "type": "array", + "minItems": 1, + "items": { "$ref": "#/definitions/rewriteRhs" } + } + } + }, + { + "type": "object", + "required": ["kind", "guid", "name", "direction", "leftSwitchIndex", "rightSwitchIndex"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["metathesis"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "direction": { "$ref": "#/definitions/ruleDirection" }, + "structuralDescription": { "$ref": "#/definitions/phonContextArray" }, + "leftSwitchIndex": { "type": "integer" }, + "rightSwitchIndex": { "type": "integer" } + } + } + ] + }, + "rewriteRhs": { + "type": "object", + "additionalProperties": false, + "properties": { + "structuralChange": { "$ref": "#/definitions/phonContextArray" }, + "leftContext": { "$ref": "#/definitions/phonContext" }, + "rightContext": { "$ref": "#/definitions/phonContext" }, + "requiredPartsOfSpeech": { "$ref": "#/definitions/guidArray" }, + "requiredRuleFeatures": { "$ref": "#/definitions/guidArray" }, + "excludedRuleFeatures": { "$ref": "#/definitions/guidArray" } + } + }, + "phonContextArray": { + "type": "array", + "minItems": 1, + "items": { "$ref": "#/definitions/phonContext" } + }, + "phonContext": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "members"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["sequence"] }, + "members": { "type": "array", "items": { "$ref": "#/definitions/phonContext" } } + } + }, + { + "type": "object", + "required": ["kind", "min", "max", "member"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["iteration"] }, + "min": { "type": "integer" }, + "max": { "type": "integer", "description": "-1 = unbounded" }, + "member": { "$ref": "#/definitions/phonContext" } + } + }, + { + "type": "object", + "required": ["kind", "phoneme"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["segment"] }, + "phoneme": { "$ref": "#/definitions/guid" } + } + }, + { + "type": "object", + "required": ["kind", "naturalClass"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["naturalClass"] }, + "naturalClass": { "$ref": "#/definitions/guid" }, + "plusVariables": { "$ref": "#/definitions/guidArray" }, + "minusVariables": { "$ref": "#/definitions/guidArray" } + } + }, + { + "type": "object", + "required": ["kind", "marker"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["boundary"] }, + "marker": { "$ref": "#/definitions/guid" } + } + }, + { + "type": "object", + "required": ["kind"], + "additionalProperties": false, + "properties": { "kind": { "enum": ["wordBoundary"] } } + }, + { + "type": "object", + "required": ["kind"], + "additionalProperties": false, + "properties": { "kind": { "enum": ["variable"] } } + } + ] + }, + "partOfSpeech": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" }, + "children": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/partOfSpeech" } }, + "inflectionClasses": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/inflectionClass" } }, + "defaultInflectionClass": { "$ref": "#/definitions/guid" }, + "inflectableFeatures": { "$ref": "#/definitions/guidArray" }, + "stemNames": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/stemName" } }, + "affixSlots": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/affixSlot" } }, + "affixTemplates": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/affixTemplate" } } + } + }, + "inflectionClass": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" }, + "children": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/inflectionClass" } } + } + }, + "stemName": { + "type": "object", + "required": ["guid", "name"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string", "minLength": 1 }, + "regions": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/featureStructure" } } + } + }, + "affixSlot": { + "type": "object", + "required": ["guid", "name", "optional"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "optional": { "type": "boolean" } + } + }, + "affixTemplate": { + "type": "object", + "required": ["guid", "name", "disabled", "isFinal"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "disabled": { "type": "boolean" }, + "prefixSlots": { "$ref": "#/definitions/guidArray" }, + "suffixSlots": { "$ref": "#/definitions/guidArray" }, + "isFinal": { "type": "boolean" } + } + }, + "compoundConstituent": { + "type": "object", + "additionalProperties": false, + "properties": { + "partOfSpeech": { "$ref": "#/definitions/guid" }, + "exceptionFeatures": { "$ref": "#/definitions/guidArray" } + } + }, + "compoundOutcome": { + "type": "object", + "additionalProperties": false, + "properties": { + "partOfSpeech": { "$ref": "#/definitions/guid" }, + "inflectionClass": { "$ref": "#/definitions/guid" } + } + }, + "compoundRule": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid", "name", "disabled", "headLast", "left", "right", "overriding"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["endocentric"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "disabled": { "type": "boolean" }, + "headLast": { "type": "boolean" }, + "left": { "$ref": "#/definitions/compoundConstituent" }, + "right": { "$ref": "#/definitions/compoundConstituent" }, + "overriding": { "$ref": "#/definitions/compoundOutcome" } + } + }, + { + "type": "object", + "required": ["kind", "guid", "name", "disabled", "left", "right", "to"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["exocentric"] }, + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "disabled": { "type": "boolean" }, + "left": { "$ref": "#/definitions/compoundConstituent" }, + "right": { "$ref": "#/definitions/compoundConstituent" }, + "to": { "$ref": "#/definitions/compoundOutcome" } + } + } + ] + }, + "adjacency": { + "enum": ["anywhere", "somewhereToLeft", "somewhereToRight", "adjacentToLeft", "adjacentToRight"] + }, + "adhocProhibition": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid", "disabled", "primary", "adjacency"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["allomorph"] }, + "guid": { "$ref": "#/definitions/guid" }, + "disabled": { "type": "boolean" }, + "primary": { "$ref": "#/definitions/guid" }, + "others": { "$ref": "#/definitions/guidArray" }, + "adjacency": { "$ref": "#/definitions/adjacency" } + } + }, + { + "type": "object", + "required": ["kind", "guid", "disabled", "primary", "adjacency"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["morpheme"] }, + "guid": { "$ref": "#/definitions/guid" }, + "disabled": { "type": "boolean" }, + "primary": { "$ref": "#/definitions/guid" }, + "others": { "$ref": "#/definitions/guidArray" }, + "adjacency": { "$ref": "#/definitions/adjacency" } + } + } + ] + }, + "exceptionFeature": { + "type": "object", + "required": ["guid", "name", "abbreviation"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" } + } + }, + "lexEntryInflType": { + "type": "object", + "required": ["guid", "name", "abbreviation", "glossPrepend", "glossAppend"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "name": { "type": "string" }, + "abbreviation": { "type": "string" }, + "glossPrepend": { "type": "string" }, + "glossAppend": { "type": "string" }, + "slots": { "$ref": "#/definitions/guidArray" }, + "inflectionFeatures": { "$ref": "#/definitions/featureStructure" } + } + }, + "parserParameters": { + "type": "object", + "required": ["notOnClitics", "acceptUnspecifiedGraphemes", "noDefaultCompounding"], + "additionalProperties": false, + "properties": { + "notOnClitics": { "type": "boolean" }, + "acceptUnspecifiedGraphemes": { "type": "boolean" }, + "noDefaultCompounding": { "type": "boolean" }, + "strata": { "type": "string", "minLength": 1 }, + "compoundRuleMaxApplications": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "required": ["compoundRule", "maxApplications"], + "additionalProperties": false, + "properties": { + "compoundRule": { "$ref": "#/definitions/guid" }, + "maxApplications": { "type": "integer" } + } + } + } + } + }, + "morphType": { + "enum": [ + "stem", "boundStem", "root", "boundRoot", "prefix", "suffix", "infix", "circumfix", + "proclitic", "enclitic", "clitic", "particle", "phrase", "discontigPhrase", + "prefixingInterfix", "infixingInterfix", "suffixingInterfix" + ] + }, + "lexEntry": { + "type": "object", + "required": ["guid", "lexemeMorphType", "allomorphs"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "citationForm": { "$ref": "#/definitions/wsFormArray" }, + "lexemeMorphType": { "$ref": "#/definitions/morphType" }, + "allomorphs": { "type": "array", "items": { "$ref": "#/definitions/allomorph" } }, + "msas": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/msa" } }, + "senses": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/sense" } }, + "entryRefs": { "type": "array", "minItems": 1, "items": { "$ref": "#/definitions/entryRef" } } + } + }, + "allomorph": { + "type": "object", + "required": ["guid", "morphType", "isAbstract"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "morphType": { "$ref": "#/definitions/morphType" }, + "isAbstract": { "type": "boolean" }, + "forms": { "$ref": "#/definitions/wsFormArray" }, + "environments": { "$ref": "#/definitions/guidArray" }, + "positions": { "$ref": "#/definitions/guidArray" }, + "stemName": { "$ref": "#/definitions/guid" }, + "inflectionClasses": { "$ref": "#/definitions/guidArray" }, + "msEnvFeatures": { "$ref": "#/definitions/featureStructure" }, + "msEnvPartOfSpeech": { "$ref": "#/definitions/guid" }, + "process": { "$ref": "#/definitions/affixProcess" } + } + }, + "affixProcess": { + "type": "object", + "additionalProperties": false, + "properties": { + "input": { "$ref": "#/definitions/phonContextArray" }, + "output": { + "type": "array", + "minItems": 1, + "items": { "$ref": "#/definitions/ruleMapping" } + } + } + }, + "ruleMapping": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "naturalClass"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["insertNaturalClass"] }, + "naturalClass": { "$ref": "#/definitions/guid" } + } + }, + { + "type": "object", + "required": ["kind", "part"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["copyFromInput"] }, + "part": { "type": "integer", "minimum": 1 } + } + }, + { + "type": "object", + "required": ["kind", "text"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["insertSegments"] }, + "text": { "type": "string", "minLength": 1 } + } + }, + { + "type": "object", + "required": ["kind", "part", "naturalClass"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["modifyFromInput"] }, + "part": { "type": "integer", "minimum": 1 }, + "naturalClass": { "$ref": "#/definitions/guid" } + } + } + ] + }, + "msa": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["stem"] }, + "guid": { "$ref": "#/definitions/guid" }, + "partOfSpeech": { "$ref": "#/definitions/guid" }, + "inflectionClass": { "$ref": "#/definitions/guid" }, + "features": { "$ref": "#/definitions/featureStructure" }, + "exceptionFeatures": { "$ref": "#/definitions/guidArray" }, + "fromPartsOfSpeech": { "$ref": "#/definitions/guidArray" }, + "slots": { "$ref": "#/definitions/guidArray" } + } + }, + { + "type": "object", + "required": ["kind", "guid"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["inflectional"] }, + "guid": { "$ref": "#/definitions/guid" }, + "partOfSpeech": { "$ref": "#/definitions/guid" }, + "slots": { "$ref": "#/definitions/guidArray" }, + "features": { "$ref": "#/definitions/featureStructure" }, + "exceptionFeatures": { "$ref": "#/definitions/guidArray" } + } + }, + { + "type": "object", + "required": ["kind", "guid"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["derivational"] }, + "guid": { "$ref": "#/definitions/guid" }, + "fromPartOfSpeech": { "$ref": "#/definitions/guid" }, + "toPartOfSpeech": { "$ref": "#/definitions/guid" }, + "fromFeatures": { "$ref": "#/definitions/featureStructure" }, + "toFeatures": { "$ref": "#/definitions/featureStructure" }, + "fromInflectionClass": { "$ref": "#/definitions/guid" }, + "toInflectionClass": { "$ref": "#/definitions/guid" }, + "fromExceptionFeatures": { "$ref": "#/definitions/guidArray" }, + "toExceptionFeatures": { "$ref": "#/definitions/guidArray" }, + "fromStemName": { "$ref": "#/definitions/guid" } + } + }, + { + "type": "object", + "required": ["kind", "guid"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["unclassified"] }, + "guid": { "$ref": "#/definitions/guid" }, + "partOfSpeech": { "$ref": "#/definitions/guid" } + } + } + ] + }, + "sense": { + "type": "object", + "required": ["guid"], + "additionalProperties": false, + "properties": { + "guid": { "$ref": "#/definitions/guid" }, + "gloss": { "$ref": "#/definitions/wsFormArray" }, + "definition": { "$ref": "#/definitions/wsFormArray" }, + "msa": { "$ref": "#/definitions/guid" } + } + }, + "entryRef": { + "oneOf": [ + { + "type": "object", + "required": ["kind", "guid"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["variant"] }, + "guid": { "$ref": "#/definitions/guid" }, + "componentLexemes": { "$ref": "#/definitions/guidArray" }, + "variantEntryTypes": { "$ref": "#/definitions/guidArray" } + } + }, + { + "type": "object", + "required": ["kind", "guid", "complexEntryTypes"], + "additionalProperties": false, + "properties": { + "kind": { "enum": ["complexForm"] }, + "guid": { "$ref": "#/definitions/guid" }, + "componentLexemes": { "$ref": "#/definitions/guidArray" }, + "complexEntryTypes": { "$ref": "#/definitions/guidArray" } + } + } + ] + } + } +} diff --git a/src/SIL.LCModel/DomainServices/GrammarJsonServices.cs b/src/SIL.LCModel/DomainServices/GrammarJsonServices.cs new file mode 100644 index 00000000..d121e2fb --- /dev/null +++ b/src/SIL.LCModel/DomainServices/GrammarJsonServices.cs @@ -0,0 +1,1576 @@ +// Copyright (c) 2026 SIL International +// This software is licensed under the LGPL, version 2.1 or later +// (http://www.gnu.org/licenses/lgpl-2.1.html) + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Text; +using System.Xml.Linq; +using Newtonsoft.Json; +using SIL.LCModel.Core.KernelInterfaces; +using SIL.LCModel.Core.WritingSystems; + +namespace SIL.LCModel.DomainServices +{ + /// + /// Exports the parser-relevant subset of a project — phonology, morphology, and lexicon — + /// as "LCM Grammar JSON": a deterministic, GUID-keyed JSON document with the envelope + /// {"format":"lcm-grammar","version":1,...}. This is the interchange format consumed + /// by external morphological-parser tooling (grammar verification, conformance fixtures, + /// field deployment); it is a read-only projection, not an editing or synchronization format. + /// + /// Determinism rules (so that two exports of the same data are byte-identical, and so that an + /// independent implementation reading the raw .fwdata XML can reproduce the same bytes): + /// - Owning/reference sequences (OS/RS properties) are written in model order, which is + /// semantically meaningful (rule order, slot order, allomorph disjunctive order, ...). + /// - Unordered collections (OC/RC properties, repository instances) are sorted by the ordinal + /// comparison of their lowercase-hyphenated GUID string. + /// - Multi-writing-system string values are written as arrays of {"ws":tag,"form":text}, + /// sorted ordinally by writing-system tag; empty alternatives are skipped. + /// - Optional values that are absent, and empty arrays, are omitted entirely (exception: + /// a lexical entry's "allomorphs" array is always present, even when empty). + /// + /// Known limitation: because this exporter walks resolved LCM objects, a dangling reference + /// (e.g. an ad hoc rule whose primary morpheme was deleted) surfaces as null and the original + /// GUID is unrecoverable, so such records are skipped with a warning rather than exported + /// with the stale reference preserved. + /// + public static class GrammarJsonServices + { + /// The value of the envelope "format" field. + public const string FormatName = "lcm-grammar"; + + /// The value of the envelope "version" field. + public const int FormatVersion = 1; + + /// + /// Exports the grammar of the given project and returns it as a JSON string. + /// + public static string ExportGrammar(LcmCache cache) + { + return ExportGrammar(cache, null); + } + + /// + /// Exports the grammar of the given project and returns it as a JSON string, adding a + /// message to (when non-null) for each piece of data that + /// could not be represented and was skipped. + /// + public static string ExportGrammar(LcmCache cache, ICollection warnings) + { + var sb = new StringBuilder(); + using (var writer = new StringWriter(sb)) + ExportGrammar(cache, writer, warnings); + return sb.ToString(); + } + + /// + /// Exports the grammar of the given project as JSON to the given writer. + /// + public static void ExportGrammar(LcmCache cache, TextWriter textWriter, ICollection warnings = null) + { + if (cache == null) + throw new ArgumentNullException(nameof(cache)); + if (textWriter == null) + throw new ArgumentNullException(nameof(textWriter)); + new Exporter(cache, textWriter, warnings).Export(); + } + + private sealed class Exporter + { + // FieldWorks stores a literal "***" to mean "no value" in some gloss fields; a missing + // multistring alternative is also rendered as "***" by the Best*Alternative accessors. + private const string MissingValueSentinel = "***"; + + private static readonly Dictionary MorphTypeNames = new Dictionary + { + { MoMorphTypeTags.kguidMorphStem, "stem" }, + { MoMorphTypeTags.kguidMorphBoundStem, "boundStem" }, + { MoMorphTypeTags.kguidMorphRoot, "root" }, + { MoMorphTypeTags.kguidMorphBoundRoot, "boundRoot" }, + { MoMorphTypeTags.kguidMorphPrefix, "prefix" }, + { MoMorphTypeTags.kguidMorphSuffix, "suffix" }, + { MoMorphTypeTags.kguidMorphInfix, "infix" }, + { MoMorphTypeTags.kguidMorphCircumfix, "circumfix" }, + { MoMorphTypeTags.kguidMorphProclitic, "proclitic" }, + { MoMorphTypeTags.kguidMorphEnclitic, "enclitic" }, + { MoMorphTypeTags.kguidMorphClitic, "clitic" }, + { MoMorphTypeTags.kguidMorphParticle, "particle" }, + { MoMorphTypeTags.kguidMorphPhrase, "phrase" }, + { MoMorphTypeTags.kguidMorphDiscontiguousPhrase, "discontigPhrase" }, + { MoMorphTypeTags.kguidMorphPrefixingInterfix, "prefixingInterfix" }, + { MoMorphTypeTags.kguidMorphInfixingInterfix, "infixingInterfix" }, + { MoMorphTypeTags.kguidMorphSuffixingInterfix, "suffixingInterfix" } + }; + + private readonly LcmCache m_cache; + private readonly ILangProject m_langProject; + private readonly JsonTextWriter m_json; + private readonly ICollection m_warnings; + + internal Exporter(LcmCache cache, TextWriter textWriter, ICollection warnings) + { + m_cache = cache; + m_langProject = cache.LanguageProject; + m_warnings = warnings; + m_json = new JsonTextWriter(textWriter) + { + Formatting = Formatting.Indented, + Indentation = 2, + IndentChar = ' ' + }; + } + + internal void Export() + { + m_json.WriteStartObject(); + WriteProp("format", FormatName); + m_json.WritePropertyName("version"); + m_json.WriteValue(FormatVersion); + WriteProject(); + WriteFeatureSystems(); + WritePhonology(); + WriteMorphology(); + WriteLexicon(); + m_json.WriteEndObject(); + m_json.Flush(); + } + + #region Shared helpers + + private void Warn(string message) + { + m_warnings?.Add(message); + } + + private static string GuidStr(ICmObject obj) + { + return obj.Guid.ToString(); + } + + private static IEnumerable ByGuid(IEnumerable objs) where T : ICmObject + { + return objs.OrderBy(o => o.Guid.ToString(), StringComparer.Ordinal); + } + + /// + /// The best text of a multistring name-like field — analysis writing systems first, + /// falling back to vernacular (phoneme names, for example, are usually authored only + /// in a vernacular writing system) — or "" when the field has no usable value. + /// + private static string Best(IMultiAccessorBase multiString) + { + if (multiString == null || multiString.StringCount == 0) + return string.Empty; + string text = multiString.BestAnalysisVernacularAlternative?.Text; + return text == null || text == MissingValueSentinel ? string.Empty : text; + } + + private string WsTag(int wsHandle) + { + // GetStrFromWs returns null (rather than throwing) for a handle that no longer + // resolves to a writing system; callers skip such alternatives. + return m_cache.ServiceLocator.WritingSystemManager.GetStrFromWs(wsHandle); + } + + private static string StripDottedCircles(string text) + { + return text?.Replace("◌", string.Empty); + } + + private void WriteProp(string name, string value) + { + m_json.WritePropertyName(name); + // LCM holds strings NFD in memory, but the .fwdata at-rest form is NFC; exports + // are NFC so that an independent reader of the raw XML reproduces the same bytes. + m_json.WriteValue((value ?? string.Empty).Normalize(NormalizationForm.FormC)); + } + + private void WriteProp(string name, bool value) + { + m_json.WritePropertyName(name); + m_json.WriteValue(value); + } + + private void WriteProp(string name, int value) + { + m_json.WritePropertyName(name); + m_json.WriteValue(value); + } + + private void WriteGuidProp(string name, ICmObject obj) + { + if (obj == null) + return; + WriteProp(name, GuidStr(obj)); + } + + /// Writes an array of GUID strings; omitted entirely when empty. + private void WriteGuidArray(string name, IEnumerable objs, bool ordered) where T : ICmObject + { + var list = (ordered ? objs : ByGuid(objs)).ToList(); + if (list.Count == 0) + return; + m_json.WritePropertyName(name); + m_json.WriteStartArray(); + foreach (T obj in list) + m_json.WriteValue(GuidStr(obj)); + m_json.WriteEndArray(); + } + + /// + /// Writes a multistring as an array of {"ws","form"}, sorted by writing-system tag; + /// empty alternatives skipped; omitted entirely when nothing remains (unless + /// ). + /// + private void WriteWsForms(string name, IMultiAccessorBase multiString, bool alwaysEmit = false) + { + var forms = GetWsForms(multiString); + if (forms.Count == 0 && !alwaysEmit) + return; + m_json.WritePropertyName(name); + WriteWsFormArray(forms); + } + + private List> GetWsForms(IMultiAccessorBase multiString, bool stripDottedCircles = false) + { + var forms = new List>(); + if (multiString == null) + return forms; + foreach (int ws in multiString.AvailableWritingSystemIds) + { + string text = multiString.get_String(ws)?.Text; + if (stripDottedCircles) + text = StripDottedCircles(text); + if (string.IsNullOrEmpty(text)) + continue; + string tag = WsTag(ws); + if (string.IsNullOrEmpty(tag)) + { + Warn($"writing system handle {ws} does not resolve to a writing system; alternative skipped"); + continue; + } + forms.Add(new KeyValuePair(tag, text)); + } + forms.Sort((a, b) => string.CompareOrdinal(a.Key, b.Key)); + return forms; + } + + private void WriteWsFormArray(List> forms) + { + m_json.WriteStartArray(); + foreach (var form in forms) + { + m_json.WriteStartObject(); + WriteProp("ws", form.Key); + WriteProp("form", form.Value); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + #endregion + + #region project + + private void WriteProject() + { + m_json.WritePropertyName("project"); + m_json.WriteStartObject(); + WriteProp("name", m_cache.ProjectId.Name); + IWritingSystemContainer wsContainer = m_cache.ServiceLocator.WritingSystems; + WriteWsList("vernacularWritingSystems", wsContainer.DefaultVernacularWritingSystem, + wsContainer.CurrentVernacularWritingSystems); + WriteWsList("analysisWritingSystems", wsContainer.DefaultAnalysisWritingSystem, + wsContainer.CurrentAnalysisWritingSystems); + m_json.WriteEndObject(); + } + + private void WriteWsList(string name, CoreWritingSystemDefinition defaultWs, + IEnumerable currentWss) + { + var tags = new List(); + if (defaultWs != null) + tags.Add(defaultWs.Id); + foreach (CoreWritingSystemDefinition ws in currentWss) + { + if (!tags.Contains(ws.Id)) + tags.Add(ws.Id); + } + if (tags.Count == 0) + return; + m_json.WritePropertyName(name); + m_json.WriteStartArray(); + foreach (string tag in tags) + m_json.WriteValue(tag); + m_json.WriteEndArray(); + } + + #endregion + + #region featureSystems + + private void WriteFeatureSystems() + { + m_json.WritePropertyName("featureSystems"); + m_json.WriteStartObject(); + WriteFeatureSystem("phonological", m_langProject.PhFeatureSystemOA); + WriteFeatureSystem("morphosyntactic", m_langProject.MsFeatureSystemOA); + m_json.WriteEndObject(); + } + + private void WriteFeatureSystem(string name, IFsFeatureSystem system) + { + m_json.WritePropertyName(name); + m_json.WriteStartObject(); + if (system != null) + { + var closed = ByGuid(system.FeaturesOC.OfType()).ToList(); + if (closed.Count > 0) + { + m_json.WritePropertyName("closedFeatures"); + m_json.WriteStartArray(); + foreach (IFsClosedFeature feature in closed) + WriteClosedFeature(feature); + m_json.WriteEndArray(); + } + var complex = ByGuid(system.FeaturesOC.OfType()).ToList(); + if (complex.Count > 0) + { + m_json.WritePropertyName("complexFeatures"); + m_json.WriteStartArray(); + foreach (IFsComplexFeature feature in complex) + WriteComplexFeature(feature); + m_json.WriteEndArray(); + } + } + m_json.WriteEndObject(); + } + + private void WriteClosedFeature(IFsClosedFeature feature) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(feature)); + WriteProp("name", Best(feature.Name)); + WriteProp("abbreviation", Best(feature.Abbreviation)); + var values = ByGuid(feature.ValuesOC).ToList(); + if (values.Count > 0) + { + m_json.WritePropertyName("values"); + m_json.WriteStartArray(); + foreach (IFsSymFeatVal value in values) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(value)); + WriteProp("name", Best(value.Name)); + WriteProp("abbreviation", Best(value.Abbreviation)); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private void WriteComplexFeature(IFsComplexFeature feature) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(feature)); + WriteProp("name", Best(feature.Name)); + WriteProp("abbreviation", Best(feature.Abbreviation)); + WriteGuidProp("featureType", feature.TypeRA); + m_json.WriteEndObject(); + } + + /// + /// Writes a feature structure property; omitted entirely when the structure is null or + /// empty and is false. + /// + private void WriteFeatureStructure(string name, IFsFeatStruc fs, bool alwaysEmit = false) + { + if ((fs == null || fs.FeatureSpecsOC.Count == 0) && !alwaysEmit) + return; + m_json.WritePropertyName(name); + WriteFeatureStructureValue(fs); + } + + private void WriteFeatureStructureValue(IFsFeatStruc fs) + { + m_json.WriteStartObject(); + var specs = fs == null + ? new List() + : ByGuid(fs.FeatureSpecsOC).Where(IsWritableFeatureSpec).ToList(); + if (specs.Count > 0) + { + m_json.WritePropertyName("values"); + m_json.WriteStartArray(); + foreach (IFsFeatureSpecification spec in specs) + { + m_json.WriteStartObject(); + WriteProp("feature", GuidStr(spec.FeatureRA)); + m_json.WritePropertyName("value"); + m_json.WriteStartObject(); + if (spec is IFsClosedValue closedValue) + { + WriteProp("kind", "closed"); + WriteProp("value", GuidStr(closedValue.ValueRA)); + } + else + { + var complexValue = (IFsComplexValue)spec; + WriteProp("kind", "complex"); + m_json.WritePropertyName("value"); + WriteFeatureStructureValue((IFsFeatStruc)complexValue.ValueOA); + } + m_json.WriteEndObject(); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private bool IsWritableFeatureSpec(IFsFeatureSpecification spec) + { + if (spec.FeatureRA == null) + { + Warn($"feature specification {GuidStr(spec)}: no feature; skipped"); + return false; + } + switch (spec) + { + case IFsClosedValue closedValue when closedValue.ValueRA != null: + return true; + case IFsComplexValue complexValue when complexValue.ValueOA is IFsFeatStruc: + return true; + default: + Warn($"feature specification {GuidStr(spec)}: unsupported or empty value; skipped"); + return false; + } + } + + #endregion + + #region phonology + + private void WritePhonology() + { + m_json.WritePropertyName("phonology"); + m_json.WriteStartObject(); + IPhPhonData phonData = m_langProject.PhonologicalDataOA; + if (phonData != null) + { + IPhPhonemeSet phonemeSet = phonData.PhonemeSetsOS.FirstOrDefault(); + if (phonData.PhonemeSetsOS.Count > 1) + Warn("project has multiple phoneme sets; only the first is exported"); + if (phonemeSet != null) + { + WritePhonemes(phonemeSet); + WriteBoundaryMarkers(phonemeSet); + } + WriteNaturalClasses(phonData); + WriteEnvironments(phonData); + WritePhonologicalRules(phonData); + WriteFeatureConstraints(phonData); + } + m_json.WriteEndObject(); + } + + private void WritePhonemes(IPhPhonemeSet phonemeSet) + { + var phonemes = ByGuid(phonemeSet.PhonemesOC).ToList(); + if (phonemes.Count == 0) + return; + m_json.WritePropertyName("phonemes"); + m_json.WriteStartArray(); + foreach (IPhPhoneme phoneme in phonemes) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(phoneme)); + WriteProp("name", Best(phoneme.Name)); + WriteRepresentations(phoneme); + WriteFeatureStructure("features", phoneme.FeaturesOA); + string ipa = phoneme.BasicIPASymbol?.Text; + if (!string.IsNullOrEmpty(ipa)) + WriteProp("basicIpaSymbol", ipa); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + /// + /// Writes a terminal unit's "representations": each code's per-writing-system forms + /// (dotted circles stripped), codes in model order, forms within a code sorted by tag. + /// + private void WriteRepresentations(IPhTerminalUnit unit) + { + var forms = new List>(); + foreach (IPhCode code in unit.CodesOS) + forms.AddRange(GetWsForms(code.Representation, stripDottedCircles: true)); + if (forms.Count == 0) + { + Warn($"phoneme or boundary marker {GuidStr(unit)} ({Best(unit.Name)}): no usable representations"); + return; + } + m_json.WritePropertyName("representations"); + WriteWsFormArray(forms); + } + + private void WriteBoundaryMarkers(IPhPhonemeSet phonemeSet) + { + var markers = ByGuid(phonemeSet.BoundaryMarkersOC + .Where(marker => marker.Guid != LangProjectTags.kguidPhRuleWordBdry)).ToList(); + if (markers.Count == 0) + return; + m_json.WritePropertyName("boundaryMarkers"); + m_json.WriteStartArray(); + foreach (IPhBdryMarker marker in markers) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(marker)); + WriteProp("name", Best(marker.Name)); + WriteRepresentations(marker); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private bool IsWritableNaturalClass(IPhNaturalClass naturalClass) + { + if (naturalClass is IPhNCSegments || naturalClass is IPhNCFeatures) + return true; + Warn($"natural class {GuidStr(naturalClass)}: unsupported class {naturalClass.ClassName}; skipped"); + return false; + } + + private void WriteNaturalClasses(IPhPhonData phonData) + { + var naturalClasses = phonData.NaturalClassesOS.Where(IsWritableNaturalClass).ToList(); + if (naturalClasses.Count == 0) + return; + m_json.WritePropertyName("naturalClasses"); + m_json.WriteStartArray(); + foreach (IPhNaturalClass naturalClass in naturalClasses) + { + switch (naturalClass) + { + case IPhNCSegments segments: + m_json.WriteStartObject(); + WriteProp("kind", "segments"); + WriteProp("guid", GuidStr(segments)); + WriteProp("name", Best(segments.Abbreviation)); + WriteGuidArray("phonemes", segments.SegmentsRC, ordered: false); + m_json.WriteEndObject(); + break; + case IPhNCFeatures features: + m_json.WriteStartObject(); + WriteProp("kind", "features"); + WriteProp("guid", GuidStr(features)); + WriteProp("name", Best(features.Abbreviation)); + WriteFeatureStructure("features", features.FeaturesOA, alwaysEmit: true); + m_json.WriteEndObject(); + break; + } + } + m_json.WriteEndArray(); + } + + private void WriteEnvironments(IPhPhonData phonData) + { + if (phonData.EnvironmentsOS.Count == 0) + return; + m_json.WritePropertyName("environments"); + m_json.WriteStartArray(); + foreach (IPhEnvironment environment in phonData.EnvironmentsOS) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(environment)); + WriteProp("name", Best(environment.Name)); + WriteProp("representation", environment.StringRepresentation?.Text); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private bool IsWritablePhonologicalRule(IPhSegmentRule rule) + { + if (rule is IPhRegularRule || rule is IPhMetathesisRule) + return true; + Warn($"phonological rule {GuidStr(rule)}: unsupported class {rule.ClassName}; skipped"); + return false; + } + + private void WritePhonologicalRules(IPhPhonData phonData) + { + var rules = phonData.PhonRulesOS.Where(rule => !rule.Disabled) + .OrderBy(rule => rule.OrderNumber).Where(IsWritablePhonologicalRule).ToList(); + if (rules.Count == 0) + return; + m_json.WritePropertyName("rules"); + m_json.WriteStartArray(); + foreach (IPhSegmentRule rule in rules) + { + switch (rule) + { + case IPhRegularRule regularRule: + WriteRewriteRule(regularRule); + break; + case IPhMetathesisRule metathesisRule: + WriteMetathesisRule(metathesisRule); + break; + } + } + m_json.WriteEndArray(); + } + + private static string DirectionName(int direction) + { + switch (direction) + { + case 1: + return "rightToLeft"; + case 2: + return "simultaneous"; + default: + return "leftToRight"; + } + } + + private void WriteRewriteRule(IPhRegularRule rule) + { + m_json.WriteStartObject(); + WriteProp("kind", "rewrite"); + WriteProp("guid", GuidStr(rule)); + WriteProp("name", Best(rule.Name)); + WriteProp("direction", DirectionName(rule.Direction)); + WritePhonContexts("structuralDescription", rule.StrucDescOS); + // The order of this enumeration is the order in which alpha variables are assigned + // Greek letters downstream, so it is preserved as-is. + WriteGuidArray("featureConstraintVariables", rule.FeatureConstraints, ordered: true); + if (rule.RightHandSidesOS.Count > 0) + { + m_json.WritePropertyName("rightHandSides"); + m_json.WriteStartArray(); + foreach (IPhSegRuleRHS rhs in rule.RightHandSidesOS) + { + m_json.WriteStartObject(); + WritePhonContexts("structuralChange", rhs.StrucChangeOS); + WritePhonContextProp("leftContext", rhs.LeftContextOA); + WritePhonContextProp("rightContext", rhs.RightContextOA); + WriteGuidArray("requiredPartsOfSpeech", rhs.InputPOSesRC, ordered: false); + WriteRuleFeatures("requiredRuleFeatures", rhs.ReqRuleFeatsRC); + WriteRuleFeatures("excludedRuleFeatures", rhs.ExclRuleFeatsRC); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private void WriteRuleFeatures(string name, IEnumerable ruleFeats) + { + // Each IPhPhonRuleFeat is a wrapper; the reference target is the wrapped item + // (an inflection class or an exception-feature possibility). + var items = new List(); + foreach (IPhPhonRuleFeat ruleFeat in ruleFeats) + { + if (ruleFeat.ItemRA == null) + Warn($"phonological rule feature {GuidStr(ruleFeat)}: no referenced item; skipped"); + else + items.Add(ruleFeat.ItemRA); + } + WriteGuidArray(name, items, ordered: false); + } + + private void WriteMetathesisRule(IPhMetathesisRule rule) + { + m_json.WriteStartObject(); + WriteProp("kind", "metathesis"); + WriteProp("guid", GuidStr(rule)); + WriteProp("name", Best(rule.Name)); + WriteProp("direction", DirectionName(rule.Direction)); + WritePhonContexts("structuralDescription", rule.StrucDescOS); + WriteProp("leftSwitchIndex", rule.LeftSwitchIndex); + WriteProp("rightSwitchIndex", rule.RightSwitchIndex); + m_json.WriteEndObject(); + } + + private void WriteFeatureConstraints(IPhPhonData phonData) + { + var constraints = phonData.FeatConstraintsOS.Where(constraint => + { + if (constraint.FeatureRA != null) + return true; + Warn($"feature constraint {GuidStr(constraint)}: no feature; skipped"); + return false; + }).ToList(); + if (constraints.Count == 0) + return; + m_json.WritePropertyName("featureConstraints"); + m_json.WriteStartArray(); + foreach (IPhFeatureConstraint constraint in constraints) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(constraint)); + WriteProp("feature", GuidStr(constraint.FeatureRA)); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + #endregion + + #region PhonContext + + /// Writes an array of pattern contexts; omitted when nothing is writable. + private void WritePhonContexts(string name, IEnumerable contexts) + { + var writable = contexts.Where(IsWritableContext).ToList(); + if (writable.Count == 0) + return; + m_json.WritePropertyName(name); + m_json.WriteStartArray(); + foreach (IPhContextOrVar context in writable) + WritePhonContextValue(context); + m_json.WriteEndArray(); + } + + private void WritePhonContextProp(string name, IPhPhonContext context) + { + if (context == null || !IsWritableContext(context)) + return; + m_json.WritePropertyName(name); + WritePhonContextValue(context); + } + + private bool IsWritableContext(IPhContextOrVar context) + { + switch (context) + { + case IPhSequenceContext sequence: + // A sequence is only faithful if every member can be written; dropping + // members would silently change what the pattern matches. + if (sequence.MembersRS.All(IsWritableContext)) + return true; + Warn($"sequence context {GuidStr(context)}: unrepresentable member; skipped"); + return false; + case IPhVariable _: + return true; + case IPhIterationContext iteration: + if (iteration.MemberRA != null && IsWritableContext(iteration.MemberRA)) + return true; + Warn($"iteration context {GuidStr(context)}: no member; skipped"); + return false; + case IPhSimpleContextSeg segment: + if (segment.FeatureStructureRA != null) + return true; + Warn($"segment context {GuidStr(context)}: no phoneme; skipped"); + return false; + case IPhSimpleContextNC naturalClass: + if (naturalClass.FeatureStructureRA != null) + return true; + Warn($"natural-class context {GuidStr(context)}: no natural class; skipped"); + return false; + case IPhSimpleContextBdry boundary: + if (boundary.FeatureStructureRA != null) + return true; + Warn($"boundary context {GuidStr(context)}: no marker; skipped"); + return false; + default: + Warn($"pattern context {GuidStr(context)}: unsupported class {context.ClassName}; skipped"); + return false; + } + } + + private void WritePhonContextValue(IPhContextOrVar context) + { + m_json.WriteStartObject(); + switch (context) + { + case IPhSequenceContext sequence: + WriteProp("kind", "sequence"); + m_json.WritePropertyName("members"); + m_json.WriteStartArray(); + foreach (IPhPhonContext member in sequence.MembersRS.Where(IsWritableContext)) + WritePhonContextValue(member); + m_json.WriteEndArray(); + break; + case IPhIterationContext iteration: + WriteProp("kind", "iteration"); + WriteProp("min", iteration.Minimum); + WriteProp("max", iteration.Maximum); + m_json.WritePropertyName("member"); + WritePhonContextValue(iteration.MemberRA); + break; + case IPhSimpleContextSeg segment: + WriteProp("kind", "segment"); + WriteProp("phoneme", GuidStr(segment.FeatureStructureRA)); + break; + case IPhSimpleContextNC naturalClass: + WriteProp("kind", "naturalClass"); + WriteProp("naturalClass", GuidStr(naturalClass.FeatureStructureRA)); + WriteGuidArray("plusVariables", naturalClass.PlusConstrRS, ordered: true); + WriteGuidArray("minusVariables", naturalClass.MinusConstrRS, ordered: true); + break; + case IPhSimpleContextBdry boundary: + if (boundary.FeatureStructureRA.Guid == LangProjectTags.kguidPhRuleWordBdry) + { + WriteProp("kind", "wordBoundary"); + } + else + { + WriteProp("kind", "boundary"); + WriteProp("marker", GuidStr(boundary.FeatureStructureRA)); + } + break; + case IPhVariable _: + WriteProp("kind", "variable"); + break; + } + m_json.WriteEndObject(); + } + + #endregion + + #region morphology + + private void WriteMorphology() + { + m_json.WritePropertyName("morphology"); + m_json.WriteStartObject(); + WritePartsOfSpeech(); + IMoMorphData morphData = m_langProject.MorphologicalDataOA; + if (morphData != null) + WriteCompoundRules(morphData); + WriteAdhocProhibitions(); + WriteExceptionFeatures(morphData); + WriteLexEntryInflTypes(); + WriteParserParameters(morphData); + m_json.WriteEndObject(); + } + + private void WritePartsOfSpeech() + { + var topLevel = m_langProject.PartsOfSpeechOA?.PossibilitiesOS.OfType().ToList(); + if (topLevel == null || topLevel.Count == 0) + return; + m_json.WritePropertyName("partsOfSpeech"); + m_json.WriteStartArray(); + foreach (IPartOfSpeech pos in topLevel) + WritePartOfSpeech(pos); + m_json.WriteEndArray(); + } + + private void WritePartOfSpeech(IPartOfSpeech pos) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(pos)); + WriteProp("name", Best(pos.Name)); + WriteProp("abbreviation", Best(pos.Abbreviation)); + var children = pos.SubPossibilitiesOS.OfType().ToList(); + if (children.Count > 0) + { + m_json.WritePropertyName("children"); + m_json.WriteStartArray(); + foreach (IPartOfSpeech child in children) + WritePartOfSpeech(child); + m_json.WriteEndArray(); + } + var inflectionClasses = ByGuid(pos.InflectionClassesOC).ToList(); + if (inflectionClasses.Count > 0) + { + m_json.WritePropertyName("inflectionClasses"); + m_json.WriteStartArray(); + foreach (IMoInflClass inflectionClass in inflectionClasses) + WriteInflectionClass(inflectionClass); + m_json.WriteEndArray(); + } + WriteGuidProp("defaultInflectionClass", pos.DefaultInflectionClassRA); + WriteGuidArray("inflectableFeatures", pos.InflectableFeatsRC, ordered: false); + WriteStemNames(pos); + WriteAffixSlots(pos); + WriteAffixTemplates(pos); + m_json.WriteEndObject(); + } + + private void WriteInflectionClass(IMoInflClass inflectionClass) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(inflectionClass)); + WriteProp("name", Best(inflectionClass.Name)); + WriteProp("abbreviation", Best(inflectionClass.Abbreviation)); + var children = ByGuid(inflectionClass.SubclassesOC).ToList(); + if (children.Count > 0) + { + m_json.WritePropertyName("children"); + m_json.WriteStartArray(); + foreach (IMoInflClass child in children) + WriteInflectionClass(child); + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private void WriteStemNames(IPartOfSpeech pos) + { + var stemNames = ByGuid(pos.StemNamesOC).ToList(); + if (stemNames.Count == 0) + return; + m_json.WritePropertyName("stemNames"); + m_json.WriteStartArray(); + foreach (IMoStemName stemName in stemNames) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(stemName)); + WriteProp("name", Best(stemName.Name)); + string abbreviation = Best(stemName.Abbreviation); + if (abbreviation.Length > 0) + WriteProp("abbreviation", abbreviation); + var regions = ByGuid(stemName.RegionsOC.Where(region => region.FeatureSpecsOC.Count > 0)).ToList(); + if (regions.Count > 0) + { + m_json.WritePropertyName("regions"); + m_json.WriteStartArray(); + foreach (IFsFeatStruc region in regions) + WriteFeatureStructureValue(region); + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private void WriteAffixSlots(IPartOfSpeech pos) + { + var slots = ByGuid(pos.AffixSlotsOC).ToList(); + if (slots.Count == 0) + return; + m_json.WritePropertyName("affixSlots"); + m_json.WriteStartArray(); + foreach (IMoInflAffixSlot slot in slots) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(slot)); + WriteProp("name", Best(slot.Name)); + WriteProp("optional", slot.Optional); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private void WriteAffixTemplates(IPartOfSpeech pos) + { + if (pos.AffixTemplatesOS.Count == 0) + return; + m_json.WritePropertyName("affixTemplates"); + m_json.WriteStartArray(); + foreach (IMoInflAffixTemplate template in pos.AffixTemplatesOS) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(template)); + WriteProp("name", Best(template.Name)); + WriteProp("disabled", template.Disabled); + WriteGuidArray("prefixSlots", template.PrefixSlotsRS, ordered: true); + WriteGuidArray("suffixSlots", template.SuffixSlotsRS, ordered: true); + WriteProp("isFinal", template.Final); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private bool IsWritableCompoundRule(IMoCompoundRule rule) + { + if (rule is IMoEndoCompound || rule is IMoExoCompound) + return true; + Warn($"compound rule {GuidStr(rule)}: unsupported class {rule.ClassName}; skipped"); + return false; + } + + private void WriteCompoundRules(IMoMorphData morphData) + { + var compoundRules = morphData.CompoundRulesOS.Where(IsWritableCompoundRule).ToList(); + if (compoundRules.Count == 0) + return; + m_json.WritePropertyName("compoundRules"); + m_json.WriteStartArray(); + foreach (IMoCompoundRule rule in compoundRules) + { + switch (rule) + { + case IMoEndoCompound endo: + m_json.WriteStartObject(); + WriteProp("kind", "endocentric"); + WriteProp("guid", GuidStr(endo)); + WriteProp("name", Best(endo.Name)); + WriteProp("disabled", endo.Disabled); + WriteProp("headLast", endo.HeadLast); + WriteCompoundConstituent("left", endo.LeftMsaOA); + WriteCompoundConstituent("right", endo.RightMsaOA); + WriteCompoundOutcome("overriding", endo.OverridingMsaOA); + m_json.WriteEndObject(); + break; + case IMoExoCompound exo: + m_json.WriteStartObject(); + WriteProp("kind", "exocentric"); + WriteProp("guid", GuidStr(exo)); + WriteProp("name", Best(exo.Name)); + WriteProp("disabled", exo.Disabled); + WriteCompoundConstituent("left", exo.LeftMsaOA); + WriteCompoundConstituent("right", exo.RightMsaOA); + WriteCompoundOutcome("to", exo.ToMsaOA); + m_json.WriteEndObject(); + break; + } + } + m_json.WriteEndArray(); + } + + private void WriteCompoundConstituent(string name, IMoStemMsa msa) + { + m_json.WritePropertyName(name); + m_json.WriteStartObject(); + if (msa != null) + { + WriteGuidProp("partOfSpeech", msa.PartOfSpeechRA); + WriteGuidArray("exceptionFeatures", msa.ProdRestrictRC, ordered: false); + } + m_json.WriteEndObject(); + } + + private void WriteCompoundOutcome(string name, IMoStemMsa msa) + { + m_json.WritePropertyName(name); + m_json.WriteStartObject(); + if (msa != null) + { + WriteGuidProp("partOfSpeech", msa.PartOfSpeechRA); + WriteGuidProp("inflectionClass", msa.InflectionClassRA); + } + m_json.WriteEndObject(); + } + + // Mirrors HCLoader.GetAdjacency (FieldWorks ParserCore): 0 anywhere, + // 1 somewhereToLeft, 2 somewhereToRight, 3 adjacentToLeft, 4 adjacentToRight. + private static string AdjacencyName(int adjacency) + { + switch (adjacency) + { + case 1: + return "somewhereToLeft"; + case 2: + return "somewhereToRight"; + case 3: + return "adjacentToLeft"; + case 4: + return "adjacentToRight"; + default: + return "anywhere"; + } + } + + private void WriteAdhocProhibitions() + { + var allo = ByGuid(m_cache.ServiceLocator.GetInstance().AllInstances()) + .Cast(); + var morpheme = ByGuid(m_cache.ServiceLocator.GetInstance().AllInstances()) + .Cast(); + var prohibitions = allo.Concat(morpheme).ToList(); + bool wroteAny = false; + foreach (IMoAdhocProhib prohibition in prohibitions) + { + switch (prohibition) + { + case IMoAlloAdhocProhib alloProhib: + if (alloProhib.FirstAllomorphRA == null) + { + Warn($"allomorph ad hoc rule {GuidStr(prohibition)}: no primary allomorph; skipped"); + continue; + } + EnsureAdhocArrayStarted(ref wroteAny); + m_json.WriteStartObject(); + WriteProp("kind", "allomorph"); + WriteProp("guid", GuidStr(prohibition)); + WriteProp("disabled", prohibition.Disabled); + WriteProp("primary", GuidStr(alloProhib.FirstAllomorphRA)); + WriteGuidArray("others", alloProhib.RestOfAllosRS, ordered: true); + WriteProp("adjacency", AdjacencyName(prohibition.Adjacency)); + m_json.WriteEndObject(); + break; + case IMoMorphAdhocProhib morphProhib: + if (morphProhib.FirstMorphemeRA == null) + { + Warn($"morpheme ad hoc rule {GuidStr(prohibition)}: no primary morpheme; skipped"); + continue; + } + EnsureAdhocArrayStarted(ref wroteAny); + m_json.WriteStartObject(); + WriteProp("kind", "morpheme"); + WriteProp("guid", GuidStr(prohibition)); + WriteProp("disabled", prohibition.Disabled); + WriteProp("primary", GuidStr(morphProhib.FirstMorphemeRA)); + WriteGuidArray("others", morphProhib.RestOfMorphsRS, ordered: true); + WriteProp("adjacency", AdjacencyName(prohibition.Adjacency)); + m_json.WriteEndObject(); + break; + } + } + if (wroteAny) + m_json.WriteEndArray(); + } + + private void EnsureAdhocArrayStarted(ref bool wroteAny) + { + if (wroteAny) + return; + m_json.WritePropertyName("adhocProhibitions"); + m_json.WriteStartArray(); + wroteAny = true; + } + + private void WriteExceptionFeatures(IMoMorphData morphData) + { + var features = new Dictionary(); + if (morphData?.ProdRestrictOA != null) + { + foreach (ICmPossibility possibility in morphData.ProdRestrictOA.ReallyReallyAllPossibilities) + features[possibility.Guid] = possibility; + } + IPhPhonData phonData = m_langProject.PhonologicalDataOA; + if (phonData?.PhonRuleFeatsOA != null) + { + foreach (IPhPhonRuleFeat ruleFeat in phonData.PhonRuleFeatsOA.PossibilitiesOS.OfType()) + { + // Inflection-class items are represented by morphology.partsOfSpeech's + // inflection-class hierarchy; only possibility items are exception features. + if (ruleFeat.ItemRA is ICmPossibility possibility && !(ruleFeat.ItemRA is IMoInflClass)) + features[possibility.Guid] = possibility; + } + } + if (features.Count == 0) + return; + m_json.WritePropertyName("exceptionFeatures"); + m_json.WriteStartArray(); + foreach (ICmPossibility possibility in ByGuid(features.Values)) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(possibility)); + WriteProp("name", Best(possibility.Name)); + WriteProp("abbreviation", Best(possibility.Abbreviation)); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private void WriteLexEntryInflTypes() + { + var inflTypes = ByGuid(m_cache.ServiceLocator.GetInstance().AllInstances()).ToList(); + if (inflTypes.Count == 0) + return; + m_json.WritePropertyName("lexEntryInflTypes"); + m_json.WriteStartArray(); + foreach (ILexEntryInflType inflType in inflTypes) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(inflType)); + WriteProp("name", Best(inflType.Name)); + WriteProp("abbreviation", Best(inflType.Abbreviation)); + WriteProp("glossPrepend", Best(inflType.GlossPrepend)); + WriteProp("glossAppend", Best(inflType.GlossAppend)); + WriteGuidArray("slots", inflType.SlotsRC, ordered: false); + WriteFeatureStructure("inflectionFeatures", inflType.InflFeatsOA); + m_json.WriteEndObject(); + } + m_json.WriteEndArray(); + } + + private void WriteParserParameters(IMoMorphData morphData) + { + XElement hcElem = null; + XElement compoundRulesElem = null; + string parserParams = morphData?.ParserParameters; + if (!string.IsNullOrWhiteSpace(parserParams)) + { + try + { + XElement root = XElement.Parse(parserParams); + hcElem = root.Element("HC"); + compoundRulesElem = root.Element("CompoundRules"); + } + catch (System.Xml.XmlException e) + { + Warn($"parser parameters could not be parsed as XML ({e.Message}); defaults used"); + } + } + m_json.WritePropertyName("parserParameters"); + m_json.WriteStartObject(); + // Note the default-true polarity of notOnClitics: absent means true. + WriteProp("notOnClitics", hcElem == null || ((bool?)hcElem.Element("NotOnClitics") ?? true)); + WriteProp("acceptUnspecifiedGraphemes", hcElem != null && ((bool?)hcElem.Element("AcceptUnspecifiedGraphemes") ?? false)); + WriteProp("noDefaultCompounding", hcElem != null && ((bool?)hcElem.Element("NoDefaultCompounding") ?? false)); + string strata = (string)hcElem?.Element("Strata"); + if (!string.IsNullOrEmpty(strata)) + WriteProp("strata", strata); + if (compoundRulesElem != null) + { + bool wroteAny = false; + foreach (XElement ruleElem in compoundRulesElem.Elements()) + { + string guidValue = (string)ruleElem.Attribute("guid"); + string maxAppsValue = (string)ruleElem.Attribute("maxApps"); + if (!Guid.TryParse(guidValue, out Guid ruleGuid) || !int.TryParse(maxAppsValue, out int maxApps)) + { + Warn($"parser parameters: malformed compound-rule max-applications entry ({ruleElem}); skipped"); + continue; + } + if (!wroteAny) + { + m_json.WritePropertyName("compoundRuleMaxApplications"); + m_json.WriteStartArray(); + wroteAny = true; + } + m_json.WriteStartObject(); + WriteProp("compoundRule", ruleGuid.ToString()); + WriteProp("maxApplications", maxApps); + m_json.WriteEndObject(); + } + if (wroteAny) + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + #endregion + + #region lexicon + + private void WriteLexicon() + { + m_json.WritePropertyName("lexicon"); + m_json.WriteStartObject(); + var entries = ByGuid(m_cache.ServiceLocator.GetInstance().AllInstances()).ToList(); + if (entries.Count > 0) + { + m_json.WritePropertyName("entries"); + m_json.WriteStartArray(); + foreach (ILexEntry entry in entries) + WriteLexEntry(entry); + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private void WriteLexEntry(ILexEntry entry) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(entry)); + WriteWsForms("citationForm", entry.CitationForm); + WriteProp("lexemeMorphType", GetLexemeMorphType(entry)); + // AlternateForms first, LexemeForm last: this order carries the disjunctive-order + // semantics of allomorph selection. + var allomorphs = entry.AlternateFormsOS.Concat( + entry.LexemeFormOA == null ? Enumerable.Empty() : new[] { entry.LexemeFormOA }); + m_json.WritePropertyName("allomorphs"); + m_json.WriteStartArray(); + foreach (IMoForm form in allomorphs) + WriteAllomorph(form); + m_json.WriteEndArray(); + var msas = ByGuid(entry.MorphoSyntaxAnalysesOC).Where(IsWritableMsa).ToList(); + if (msas.Count > 0) + { + m_json.WritePropertyName("msas"); + m_json.WriteStartArray(); + foreach (IMoMorphSynAnalysis msa in msas) + WriteMsa(msa); + m_json.WriteEndArray(); + } + // AllSenses deliberately flattens the subsense tree (pre-order, parent before its + // subsenses): the parser consumes senses as a flat list, and the reference + // implementation of this format does the same. + var senses = entry.AllSenses; + if (senses.Count > 0) + { + m_json.WritePropertyName("senses"); + m_json.WriteStartArray(); + foreach (ILexSense sense in senses) + WriteSense(sense); + m_json.WriteEndArray(); + } + if (entry.EntryRefsOS.Count > 0) + { + m_json.WritePropertyName("entryRefs"); + m_json.WriteStartArray(); + foreach (ILexEntryRef entryRef in entry.EntryRefsOS) + WriteEntryRef(entryRef); + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + private string GetLexemeMorphType(ILexEntry entry) + { + IMoMorphType morphType = entry.LexemeFormOA?.MorphTypeRA; + if (morphType != null && MorphTypeNames.TryGetValue(morphType.Guid, out string name)) + return name; + Warn($"entry {GuidStr(entry)}: no usable lexeme-form morph type; defaulting to stem"); + return "stem"; + } + + private void WriteAllomorph(IMoForm form) + { + if (form.MorphTypeRA == null || !MorphTypeNames.TryGetValue(form.MorphTypeRA.Guid, out string morphType)) + { + string typeName = form.MorphTypeRA == null ? "(none)" : Best(form.MorphTypeRA.Name); + Warn($"allomorph {GuidStr(form)}: unsupported morph type {typeName}; skipped"); + return; + } + // An affix process's output mappings reference input parts positionally + // (1-based position in InputOS), so the whole allomorph must be skipped if any + // input part cannot be written (dropping one would misalign the indices) or any + // output mapping cannot be written (dropping one would change the recipe). + if (form is IMoAffixProcess processForm && + (!processForm.InputOS.All(IsWritableContext) || + !processForm.OutputOS.All(mapping => IsWritableRuleMapping(mapping, processForm)))) + { + Warn($"affix process {GuidStr(form)}: unrepresentable input or output part; allomorph skipped"); + return; + } + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(form)); + WriteProp("morphType", morphType); + WriteProp("isAbstract", form.IsAbstract); + WriteWsForms("forms", form.Form); + switch (form) + { + case IMoStemAllomorph stem: + WriteGuidArray("environments", stem.PhoneEnvRC, ordered: false); + WriteGuidProp("stemName", stem.StemNameRA); + break; + case IMoAffixAllomorph affix: + WriteGuidArray("environments", affix.PhoneEnvRC, ordered: false); + WriteGuidArray("positions", affix.PositionRS, ordered: true); + WriteGuidArray("inflectionClasses", affix.InflectionClassesRC, ordered: false); + WriteFeatureStructure("msEnvFeatures", affix.MsEnvFeaturesOA); + WriteGuidProp("msEnvPartOfSpeech", affix.MsEnvPartOfSpeechRA); + break; + case IMoAffixProcess process: + WriteGuidArray("inflectionClasses", process.InflectionClassesRC, ordered: false); + WriteAffixProcess(process); + break; + } + m_json.WriteEndObject(); + } + + private void WriteAffixProcess(IMoAffixProcess process) + { + m_json.WritePropertyName("process"); + m_json.WriteStartObject(); + WritePhonContexts("input", process.InputOS); + if (process.OutputOS.Count > 0) + { + m_json.WritePropertyName("output"); + m_json.WriteStartArray(); + foreach (IMoRuleMapping mapping in process.OutputOS) + WriteRuleMapping(mapping); + m_json.WriteEndArray(); + } + m_json.WriteEndObject(); + } + + /// + /// Whether a rule mapping can be represented. Copy/modify mappings must reference a + /// direct member of the owning process's input sequence — "part" is that member's + /// 1-based position, so a reference to anything else (a nested context, another + /// process's input) would emit a silently wrong index. + /// + private bool IsWritableRuleMapping(IMoRuleMapping mapping, IMoAffixProcess process) + { + switch (mapping) + { + case IMoInsertNC insertNC: + if (insertNC.ContentRA != null) + return true; + Warn($"rule mapping {GuidStr(mapping)}: no natural class to insert; skipped"); + return false; + case IMoCopyFromInput copy: + if (copy.ContentRA != null && copy.ContentRA.Owner == process) + return true; + Warn($"rule mapping {GuidStr(mapping)}: input-part reference missing or not a top-level input part; skipped"); + return false; + case IMoInsertPhones insertPhones: + if (GetInsertPhonesText(insertPhones).Length > 0) + return true; + Warn($"rule mapping {GuidStr(mapping)}: no insertable segments; skipped"); + return false; + case IMoModifyFromInput modify: + if (modify.ContentRA != null && modify.ContentRA.Owner == process && modify.ModificationRA != null) + return true; + Warn($"rule mapping {GuidStr(mapping)}: incomplete modification or non-top-level input-part reference; skipped"); + return false; + default: + Warn($"rule mapping {GuidStr(mapping)}: unsupported class {mapping.ClassName}; skipped"); + return false; + } + } + + /// Writes one rule mapping, which IsWritableRuleMapping has already vetted. + private void WriteRuleMapping(IMoRuleMapping mapping) + { + switch (mapping) + { + case IMoInsertNC insertNC: + m_json.WriteStartObject(); + WriteProp("kind", "insertNaturalClass"); + WriteProp("naturalClass", GuidStr(insertNC.ContentRA)); + m_json.WriteEndObject(); + break; + case IMoCopyFromInput copy: + m_json.WriteStartObject(); + WriteProp("kind", "copyFromInput"); + WriteProp("part", copy.ContentRA.IndexInOwner + 1); + m_json.WriteEndObject(); + break; + case IMoInsertPhones insertPhones: + m_json.WriteStartObject(); + WriteProp("kind", "insertSegments"); + WriteProp("text", GetInsertPhonesText(insertPhones)); + m_json.WriteEndObject(); + break; + case IMoModifyFromInput modify: + m_json.WriteStartObject(); + WriteProp("kind", "modifyFromInput"); + WriteProp("part", modify.ContentRA.IndexInOwner + 1); + WriteProp("naturalClass", GuidStr(modify.ModificationRA)); + m_json.WriteEndObject(); + break; + } + } + + private string GetInsertPhonesText(IMoInsertPhones insertPhones) + { + var sb = new StringBuilder(); + foreach (IPhTerminalUnit unit in insertPhones.ContentRS) + { + IPhCode code = unit.CodesOS.FirstOrDefault(); + if (code == null) + continue; + // Boundary markers may only have a representation in a non-default vernacular + // writing system, so fall back across vernacular writing systems for them. + string text = unit is IPhBdryMarker + ? code.Representation.BestVernacularAlternative?.Text + : code.Representation.VernacularDefaultWritingSystem?.Text; + if (text == MissingValueSentinel) + text = null; + text = StripDottedCircles(text)?.Trim(); + if (!string.IsNullOrEmpty(text)) + sb.Append(text); + } + return sb.ToString(); + } + + private bool IsWritableMsa(IMoMorphSynAnalysis msa) + { + if (msa is IMoStemMsa || msa is IMoInflAffMsa || msa is IMoDerivAffMsa || msa is IMoUnclassifiedAffixMsa) + return true; + Warn($"MSA {GuidStr(msa)}: unsupported class {msa.ClassName}; skipped"); + return false; + } + + /// Writes one MSA, which IsWritableMsa has already vetted. + private void WriteMsa(IMoMorphSynAnalysis msa) + { + switch (msa) + { + case IMoStemMsa stem: + m_json.WriteStartObject(); + WriteProp("kind", "stem"); + WriteProp("guid", GuidStr(stem)); + WriteGuidProp("partOfSpeech", stem.PartOfSpeechRA); + WriteGuidProp("inflectionClass", stem.InflectionClassRA); + WriteFeatureStructure("features", stem.MsFeaturesOA); + WriteGuidArray("exceptionFeatures", stem.ProdRestrictRC, ordered: false); + WriteGuidArray("fromPartsOfSpeech", stem.FromPartsOfSpeechRC, ordered: false); + WriteGuidArray("slots", stem.SlotsRC, ordered: false); + m_json.WriteEndObject(); + break; + case IMoInflAffMsa inflectional: + m_json.WriteStartObject(); + WriteProp("kind", "inflectional"); + WriteProp("guid", GuidStr(inflectional)); + WriteGuidProp("partOfSpeech", inflectional.PartOfSpeechRA); + WriteGuidArray("slots", inflectional.SlotsRC, ordered: false); + WriteFeatureStructure("features", inflectional.InflFeatsOA); + WriteGuidArray("exceptionFeatures", inflectional.FromProdRestrictRC, ordered: false); + m_json.WriteEndObject(); + break; + case IMoDerivAffMsa derivational: + m_json.WriteStartObject(); + WriteProp("kind", "derivational"); + WriteProp("guid", GuidStr(derivational)); + WriteGuidProp("fromPartOfSpeech", derivational.FromPartOfSpeechRA); + WriteGuidProp("toPartOfSpeech", derivational.ToPartOfSpeechRA); + WriteFeatureStructure("fromFeatures", derivational.FromMsFeaturesOA); + WriteFeatureStructure("toFeatures", derivational.ToMsFeaturesOA); + WriteGuidProp("fromInflectionClass", derivational.FromInflectionClassRA); + WriteGuidProp("toInflectionClass", derivational.ToInflectionClassRA); + WriteGuidArray("fromExceptionFeatures", derivational.FromProdRestrictRC, ordered: false); + WriteGuidArray("toExceptionFeatures", derivational.ToProdRestrictRC, ordered: false); + WriteGuidProp("fromStemName", derivational.FromStemNameRA); + m_json.WriteEndObject(); + break; + case IMoUnclassifiedAffixMsa unclassified: + m_json.WriteStartObject(); + WriteProp("kind", "unclassified"); + WriteProp("guid", GuidStr(unclassified)); + WriteGuidProp("partOfSpeech", unclassified.PartOfSpeechRA); + m_json.WriteEndObject(); + break; + } + } + + private void WriteSense(ILexSense sense) + { + m_json.WriteStartObject(); + WriteProp("guid", GuidStr(sense)); + WriteWsForms("gloss", sense.Gloss); + WriteWsForms("definition", sense.Definition); + // Normally a sense's MSA belongs to its own entry, but stray references into + // another entry's msas exist in real projects; carry them, but never silently. + if (sense.MorphoSyntaxAnalysisRA != null && sense.MorphoSyntaxAnalysisRA.Owner != sense.Entry) + { + Warn($"sense {GuidStr(sense)}: its MSA {GuidStr(sense.MorphoSyntaxAnalysisRA)} " + + "belongs to a different entry; exported as-is (resolve msa references document-wide)"); + } + WriteGuidProp("msa", sense.MorphoSyntaxAnalysisRA); + m_json.WriteEndObject(); + } + + private void WriteEntryRef(ILexEntryRef entryRef) + { + m_json.WriteStartObject(); + // The FieldWorks UI treats variant and complex-form types as mutually exclusive; + // a ref is a complex form only when it has complex-form types and no variant types. + // Anything else — variant types only, both kinds, or neither — is a variant. + if (entryRef.ComplexEntryTypesRS.Count > 0 && entryRef.VariantEntryTypesRS.Count == 0) + { + WriteProp("kind", "complexForm"); + WriteProp("guid", GuidStr(entryRef)); + WriteGuidArray("componentLexemes", entryRef.ComponentLexemesRS, ordered: true); + WriteGuidArray("complexEntryTypes", entryRef.ComplexEntryTypesRS, ordered: true); + } + else + { + if (entryRef.ComplexEntryTypesRS.Count > 0) + Warn($"entry ref {GuidStr(entryRef)}: has both variant and complex-form types; exported as a variant, complex-form types dropped"); + WriteProp("kind", "variant"); + WriteProp("guid", GuidStr(entryRef)); + WriteGuidArray("componentLexemes", entryRef.ComponentLexemesRS, ordered: true); + WriteGuidArray("variantEntryTypes", entryRef.VariantEntryTypesRS, ordered: true); + } + m_json.WriteEndObject(); + } + + #endregion + } + } +} diff --git a/tests/SIL.LCModel.Tests/DomainServices/GrammarJsonServicesTests.cs b/tests/SIL.LCModel.Tests/DomainServices/GrammarJsonServicesTests.cs new file mode 100644 index 00000000..933f1f22 --- /dev/null +++ b/tests/SIL.LCModel.Tests/DomainServices/GrammarJsonServicesTests.cs @@ -0,0 +1,582 @@ +// Copyright (c) 2026 SIL International +// This software is licensed under the LGPL, version 2.1 or later +// (http://www.gnu.org/licenses/lgpl-2.1.html) + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using Newtonsoft.Json.Linq; +using NUnit.Framework; +using SIL.LCModel.Core.Text; + +namespace SIL.LCModel.DomainServices +{ + /// + /// Tests for (LCM Grammar JSON export). + /// + [TestFixture] + public class GrammarJsonServicesTests : MemoryOnlyBackendProviderRestoredForEachTestTestBase + { + private JObject Export() + { + return JObject.Parse(GrammarJsonServices.ExportGrammar(Cache)); + } + + private ILexEntry MakeStemEntry(string form, string gloss) + { + ILexEntry entry = Cache.ServiceLocator.GetInstance().Create(); + IMoStemAllomorph lexemeForm = Cache.ServiceLocator.GetInstance().Create(); + entry.LexemeFormOA = lexemeForm; + lexemeForm.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphStem); + lexemeForm.Form.SetVernacularDefaultWritingSystem(form); + entry.CitationForm.SetVernacularDefaultWritingSystem(form); + ILexSense sense = Cache.ServiceLocator.GetInstance().Create(); + entry.SensesOS.Add(sense); + sense.Gloss.SetAnalysisDefaultWritingSystem(gloss); + return entry; + } + + private IPartOfSpeech MakePartOfSpeech(string name) + { + IPartOfSpeech pos = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.PartsOfSpeechOA.PossibilitiesOS.Add(pos); + pos.Name.SetAnalysisDefaultWritingSystem(name); + pos.Abbreviation.SetAnalysisDefaultWritingSystem(name.Substring(0, 1)); + return pos; + } + + private IPhPhonemeSet EnsurePhonemeSet() + { + IPhPhonData phonData = Cache.LangProject.PhonologicalDataOA; + if (phonData.PhonemeSetsOS.Count == 0) + phonData.PhonemeSetsOS.Add(Cache.ServiceLocator.GetInstance().Create()); + return phonData.PhonemeSetsOS[0]; + } + + /// The envelope and all sections are present even for an empty project. + [Test] + public void ExportGrammar_EmptyProject_WritesEnvelopeAndSections() + { + JObject json = Export(); + + Assert.AreEqual(GrammarJsonServices.FormatName, (string)json["format"]); + Assert.AreEqual(GrammarJsonServices.FormatVersion, (int)json["version"]); + Assert.IsNotNull(json["project"], "project section"); + Assert.IsNotNull((string)json["project"]["name"], "project name"); + Assert.IsTrue(json["project"]["vernacularWritingSystems"].Any(), "vernacular writing systems"); + Assert.IsTrue(json["project"]["analysisWritingSystems"].Any(), "analysis writing systems"); + Assert.IsNotNull(json["featureSystems"]["phonological"], "phonological feature system"); + Assert.IsNotNull(json["featureSystems"]["morphosyntactic"], "morphosyntactic feature system"); + Assert.IsNotNull(json["phonology"], "phonology section"); + Assert.IsNotNull(json["morphology"], "morphology section"); + Assert.IsNotNull(json["lexicon"], "lexicon section"); + // Empty collections are omitted, not written as []. + Assert.IsNull(json["lexicon"]["entries"], "empty lexicon should omit entries"); + AssertValidatesAgainstSchema(json.ToString(), "empty project"); + } + + /// Two exports of the same project are byte-identical. + [Test] + public void ExportGrammar_IsDeterministic() + { + IPartOfSpeech pos = MakePartOfSpeech("verb"); + ILexEntry entry = MakeStemEntry("kick", "kick"); + IMoStemMsa msa = Cache.ServiceLocator.GetInstance().Create(); + entry.MorphoSyntaxAnalysesOC.Add(msa); + msa.PartOfSpeechRA = pos; + MakeStemEntry("sing", "sing"); + IPhPhonemeSet phonemeSet = EnsurePhonemeSet(); + IPhPhoneme phoneme = Cache.ServiceLocator.GetInstance().Create(); + phonemeSet.PhonemesOC.Add(phoneme); + IPhCode code = Cache.ServiceLocator.GetInstance().Create(); + phoneme.CodesOS.Add(code); + code.Representation.SetVernacularDefaultWritingSystem("a"); + + string first = GrammarJsonServices.ExportGrammar(Cache); + string second = GrammarJsonServices.ExportGrammar(Cache); + + Assert.AreEqual(first, second); + } + + /// A stem entry round-trips its core fields. + [Test] + public void ExportGrammar_WritesLexEntryCoreFields() + { + IPartOfSpeech pos = MakePartOfSpeech("verb"); + ILexEntry entry = MakeStemEntry("kick", "kick"); + ILexSense sense = entry.SensesOS[0]; + sense.Definition.SetAnalysisDefaultWritingSystem("to strike with the foot"); + IMoStemMsa msa = Cache.ServiceLocator.GetInstance().Create(); + entry.MorphoSyntaxAnalysesOC.Add(msa); + msa.PartOfSpeechRA = pos; + sense.MorphoSyntaxAnalysisRA = msa; + + JObject json = Export(); + + var entries = (JArray)json["lexicon"]["entries"]; + Assert.AreEqual(1, entries.Count); + JObject jsonEntry = (JObject)entries[0]; + Assert.AreEqual(entry.Guid.ToString(), (string)jsonEntry["guid"]); + Assert.AreEqual("stem", (string)jsonEntry["lexemeMorphType"]); + Assert.AreEqual("kick", (string)jsonEntry["citationForm"][0]["form"]); + + var allomorphs = (JArray)jsonEntry["allomorphs"]; + Assert.AreEqual(1, allomorphs.Count, "lexeme form should be the single allomorph"); + Assert.AreEqual("stem", (string)allomorphs[0]["morphType"]); + Assert.AreEqual("kick", (string)allomorphs[0]["forms"][0]["form"]); + + JObject jsonMsa = (JObject)jsonEntry["msas"][0]; + Assert.AreEqual("stem", (string)jsonMsa["kind"]); + Assert.AreEqual(msa.Guid.ToString(), (string)jsonMsa["guid"]); + Assert.AreEqual(pos.Guid.ToString(), (string)jsonMsa["partOfSpeech"]); + + JObject jsonSense = (JObject)jsonEntry["senses"][0]; + Assert.AreEqual(sense.Guid.ToString(), (string)jsonSense["guid"]); + Assert.AreEqual("kick", (string)jsonSense["gloss"][0]["form"]); + Assert.AreEqual("to strike with the foot", (string)jsonSense["definition"][0]["form"]); + Assert.AreEqual(msa.Guid.ToString(), (string)jsonSense["msa"]); + } + + /// Lexical entries (an unordered collection) are sorted by GUID string. + [Test] + public void ExportGrammar_SortsEntriesByGuid() + { + for (int i = 0; i < 5; i++) + MakeStemEntry("form" + i, "gloss" + i); + + JObject json = Export(); + + var guids = json["lexicon"]["entries"].Select(e => (string)e["guid"]).ToList(); + var sorted = guids.OrderBy(g => g, StringComparer.Ordinal).ToList(); + CollectionAssert.AreEqual(sorted, guids); + } + + /// Phonemes and environments are exported with their representations. + [Test] + public void ExportGrammar_WritesPhonology() + { + IPhPhonemeSet phonemeSet = EnsurePhonemeSet(); + IPhPhoneme phoneme = Cache.ServiceLocator.GetInstance().Create(); + phonemeSet.PhonemesOC.Add(phoneme); + phoneme.Name.SetVernacularDefaultWritingSystem("a"); + IPhCode code = Cache.ServiceLocator.GetInstance().Create(); + phoneme.CodesOS.Add(code); + code.Representation.SetVernacularDefaultWritingSystem("a"); + IPhEnvironment environment = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.PhonologicalDataOA.EnvironmentsOS.Add(environment); + environment.StringRepresentation = TsStringUtils.MakeString("/_[C]", Cache.DefaultVernWs); + + JObject json = Export(); + + JObject jsonPhoneme = (JObject)json["phonology"]["phonemes"] + .Single(p => (string)p["guid"] == phoneme.Guid.ToString()); + Assert.AreEqual("a", (string)jsonPhoneme["name"], + "a phoneme name authored only in a vernacular writing system must not be lost"); + Assert.AreEqual("a", (string)jsonPhoneme["representations"][0]["form"]); + JObject jsonEnvironment = (JObject)json["phonology"]["environments"] + .Single(e => (string)e["guid"] == environment.Guid.ToString()); + Assert.AreEqual("/_[C]", (string)jsonEnvironment["representation"]); + } + + /// Parser parameters fall back to the documented defaults. + [Test] + public void ExportGrammar_ParserParameterDefaults() + { + Cache.LangProject.MorphologicalDataOA.ParserParameters = string.Empty; + + JObject json = Export(); + + JObject parameters = (JObject)json["morphology"]["parserParameters"]; + Assert.IsTrue((bool)parameters["notOnClitics"], "notOnClitics defaults to true"); + Assert.IsFalse((bool)parameters["acceptUnspecifiedGraphemes"]); + Assert.IsFalse((bool)parameters["noDefaultCompounding"]); + Assert.IsNull(parameters["strata"]); + Assert.IsNull(parameters["compoundRuleMaxApplications"]); + } + + /// Explicit parser parameters are read from the stored XML. + [Test] + public void ExportGrammar_ParserParametersFromXml() + { + Guid ruleGuid = Guid.NewGuid(); + Cache.LangProject.MorphologicalDataOA.ParserParameters = + "true" + + "Morphology,Phonology" + + $""; + + JObject json = Export(); + + JObject parameters = (JObject)json["morphology"]["parserParameters"]; + Assert.IsTrue((bool)parameters["notOnClitics"], "absent notOnClitics still defaults to true"); + Assert.IsTrue((bool)parameters["noDefaultCompounding"]); + Assert.AreEqual("Morphology,Phonology", (string)parameters["strata"]); + JObject maxApps = (JObject)parameters["compoundRuleMaxApplications"][0]; + Assert.AreEqual(ruleGuid.ToString(), (string)maxApps["compoundRule"]); + Assert.AreEqual(3, (int)maxApps["maxApplications"]); + } + + /// + /// An entry ref is a complex form only when it has complex-form types and no variant + /// types; a ref carrying both kinds is a variant. + /// + [Test] + public void ExportGrammar_EntryRefWithBothTypeKinds_IsVariant() + { + ILexEntry main = MakeStemEntry("go", "go"); + ILexEntry variant = MakeStemEntry("went", "went"); + ILexEntryRef entryRef = Cache.ServiceLocator.GetInstance().Create(); + variant.EntryRefsOS.Add(entryRef); + entryRef.ComponentLexemesRS.Add(main); + ILexDb lexDb = Cache.LangProject.LexDbOA; + if (lexDb.VariantEntryTypesOA == null) + lexDb.VariantEntryTypesOA = Cache.ServiceLocator.GetInstance().Create(); + if (lexDb.ComplexEntryTypesOA == null) + lexDb.ComplexEntryTypesOA = Cache.ServiceLocator.GetInstance().Create(); + ILexEntryType variantType = Cache.ServiceLocator.GetInstance().Create(); + lexDb.VariantEntryTypesOA.PossibilitiesOS.Add(variantType); + entryRef.VariantEntryTypesRS.Add(variantType); + ILexEntryType complexType = Cache.ServiceLocator.GetInstance().Create(); + lexDb.ComplexEntryTypesOA.PossibilitiesOS.Add(complexType); + entryRef.ComplexEntryTypesRS.Add(complexType); + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + JObject jsonEntry = (JObject)json["lexicon"]["entries"] + .Single(e => (string)e["guid"] == variant.Guid.ToString()); + JObject jsonRef = (JObject)jsonEntry["entryRefs"][0]; + Assert.AreEqual("variant", (string)jsonRef["kind"]); + Assert.AreEqual(main.Guid.ToString(), (string)jsonRef["componentLexemes"][0]); + Assert.AreEqual(variantType.Guid.ToString(), (string)jsonRef["variantEntryTypes"][0]); + Assert.IsTrue(warnings.Any(w => w.Contains(entryRef.Guid.ToString())), + "dropping the complex-form types must be reported in warnings"); + } + + /// An affix process exports its input parts and output mappings. + [Test] + public void ExportGrammar_WritesAffixProcess() + { + ILexEntry entry = Cache.ServiceLocator.GetInstance().Create(); + IMoAffixProcess process = Cache.ServiceLocator.GetInstance().Create(); + entry.LexemeFormOA = process; + process.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphSuffix); + IPhVariable variable = Cache.ServiceLocator.GetInstance().Create(); + process.InputOS.Add(variable); + IMoCopyFromInput copy = Cache.ServiceLocator.GetInstance().Create(); + process.OutputOS.Add(copy); + copy.ContentRA = variable; + + JObject json = Export(); + + JObject jsonProcess = (JObject)json["lexicon"]["entries"][0]["allomorphs"][0]["process"]; + Assert.AreEqual("variable", (string)jsonProcess["input"][0]["kind"]); + Assert.AreEqual("copyFromInput", (string)jsonProcess["output"][0]["kind"]); + Assert.AreEqual(1, (int)jsonProcess["output"][0]["part"]); + } + + /// + /// An affix process with an unrepresentable input part is skipped entirely: output + /// mappings reference input parts by position, so dropping one part would silently + /// misalign every index after it. + /// + [Test] + public void ExportGrammar_SkipsAffixProcessWithUnrepresentableInput() + { + ILexEntry entry = Cache.ServiceLocator.GetInstance().Create(); + IMoAffixProcess process = Cache.ServiceLocator.GetInstance().Create(); + entry.LexemeFormOA = process; + process.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphSuffix); + // A segment context with no phoneme reference cannot be written. + IPhSimpleContextSeg broken = Cache.ServiceLocator.GetInstance().Create(); + process.InputOS.Add(broken); + process.InputOS.Add(Cache.ServiceLocator.GetInstance().Create()); + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + Assert.AreEqual(0, ((JArray)json["lexicon"]["entries"][0]["allomorphs"]).Count); + Assert.IsTrue(warnings.Any(w => w.Contains(process.Guid.ToString()) && w.Contains("allomorph skipped")), + "skipped process should be reported in warnings"); + } + + /// A morpheme ad hoc rule exports its members and adjacency. + [Test] + public void ExportGrammar_WritesMorphemeAdhocProhibition() + { + ILexEntry first = MakeStemEntry("kick", "kick"); + IMoStemMsa firstMsa = Cache.ServiceLocator.GetInstance().Create(); + first.MorphoSyntaxAnalysesOC.Add(firstMsa); + ILexEntry other = MakeStemEntry("sing", "sing"); + IMoStemMsa otherMsa = Cache.ServiceLocator.GetInstance().Create(); + other.MorphoSyntaxAnalysesOC.Add(otherMsa); + IMoMorphAdhocProhib prohibition = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.MorphologicalDataOA.AdhocCoProhibitionsOC.Add(prohibition); + prohibition.FirstMorphemeRA = firstMsa; + prohibition.RestOfMorphsRS.Add(otherMsa); + prohibition.Adjacency = 3; + + JObject json = Export(); + + JObject jsonProhibition = (JObject)json["morphology"]["adhocProhibitions"][0]; + Assert.AreEqual("morpheme", (string)jsonProhibition["kind"]); + Assert.AreEqual(firstMsa.Guid.ToString(), (string)jsonProhibition["primary"]); + Assert.AreEqual(otherMsa.Guid.ToString(), (string)jsonProhibition["others"][0]); + Assert.AreEqual("adjacentToLeft", (string)jsonProhibition["adjacency"]); + } + + /// + /// The export validates against the published contract (doc/lcm-grammar.schema.json) — + /// both for an empty project and for a project exercising lexicon, phonology, MSAs, + /// senses, entry refs, affix processes, ad hoc rules, and parser parameters. + /// + [Test] + public void ExportGrammar_ValidatesAgainstPublishedSchema() + { + // Lexicon: stem entry with POS, MSA, sense (gloss + definition), citation form. + IPartOfSpeech pos = MakePartOfSpeech("verb"); + ILexEntry entry = MakeStemEntry("kick", "kick"); + entry.SensesOS[0].Definition.SetAnalysisDefaultWritingSystem("to strike with the foot"); + IMoStemMsa msa = Cache.ServiceLocator.GetInstance().Create(); + entry.MorphoSyntaxAnalysesOC.Add(msa); + msa.PartOfSpeechRA = pos; + entry.SensesOS[0].MorphoSyntaxAnalysisRA = msa; + // Variant entry ref. + ILexEntry variant = MakeStemEntry("kicked", "kicked"); + ILexEntryRef entryRef = Cache.ServiceLocator.GetInstance().Create(); + variant.EntryRefsOS.Add(entryRef); + entryRef.ComponentLexemesRS.Add(entry); + // Affix process with a copy mapping. + ILexEntry affixEntry = Cache.ServiceLocator.GetInstance().Create(); + IMoAffixProcess process = Cache.ServiceLocator.GetInstance().Create(); + affixEntry.LexemeFormOA = process; + process.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphSuffix); + IPhVariable variable = Cache.ServiceLocator.GetInstance().Create(); + process.InputOS.Add(variable); + IMoCopyFromInput copy = Cache.ServiceLocator.GetInstance().Create(); + process.OutputOS.Add(copy); + copy.ContentRA = variable; + // Morpheme ad hoc rule. + IMoMorphAdhocProhib prohibition = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.MorphologicalDataOA.AdhocCoProhibitionsOC.Add(prohibition); + prohibition.FirstMorphemeRA = msa; + IMoStemMsa otherMsa = Cache.ServiceLocator.GetInstance().Create(); + variant.MorphoSyntaxAnalysesOC.Add(otherMsa); + prohibition.RestOfMorphsRS.Add(otherMsa); + prohibition.Adjacency = 2; + // Phonology: phoneme, environment, natural class, rewrite rule. + IPhPhonemeSet phonemeSet = EnsurePhonemeSet(); + IPhPhoneme phoneme = Cache.ServiceLocator.GetInstance().Create(); + phonemeSet.PhonemesOC.Add(phoneme); + IPhCode code = Cache.ServiceLocator.GetInstance().Create(); + phoneme.CodesOS.Add(code); + code.Representation.SetVernacularDefaultWritingSystem("a"); + IPhEnvironment environment = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.PhonologicalDataOA.EnvironmentsOS.Add(environment); + environment.StringRepresentation = TsStringUtils.MakeString("/_[C]", Cache.DefaultVernWs); + IPhNCSegments naturalClass = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.PhonologicalDataOA.NaturalClassesOS.Add(naturalClass); + naturalClass.Abbreviation.SetAnalysisDefaultWritingSystem("C"); + naturalClass.SegmentsRC.Add(phoneme); + IPhRegularRule rewriteRule = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.PhonologicalDataOA.PhonRulesOS.Add(rewriteRule); + IPhSimpleContextSeg ruleInput = Cache.ServiceLocator.GetInstance().Create(); + rewriteRule.StrucDescOS.Add(ruleInput); + ruleInput.FeatureStructureRA = phoneme; + IPhSegRuleRHS rhs = Cache.ServiceLocator.GetInstance().Create(); + rewriteRule.RightHandSidesOS.Add(rhs); + IPhSimpleContextNC ruleChange = Cache.ServiceLocator.GetInstance().Create(); + rhs.StrucChangeOS.Add(ruleChange); + ruleChange.FeatureStructureRA = naturalClass; + // More rule-mapping kinds on the affix process. + IMoInsertPhones insertPhones = Cache.ServiceLocator.GetInstance().Create(); + process.OutputOS.Add(insertPhones); + insertPhones.ContentRS.Add(phoneme); + // Morphology: compound rule, affix slot + template, exception feature, + // irregularly-inflected-form type, inflectional and derivational MSAs. + IMoEndoCompound compoundRule = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.MorphologicalDataOA.CompoundRulesOS.Add(compoundRule); + compoundRule.HeadLast = true; + if (compoundRule.LeftMsaOA == null) + compoundRule.LeftMsaOA = Cache.ServiceLocator.GetInstance().Create(); + compoundRule.LeftMsaOA.PartOfSpeechRA = pos; + IMoInflAffixSlot slot = Cache.ServiceLocator.GetInstance().Create(); + pos.AffixSlotsOC.Add(slot); + slot.Optional = true; + IMoInflAffixTemplate template = Cache.ServiceLocator.GetInstance().Create(); + pos.AffixTemplatesOS.Add(template); + template.SuffixSlotsRS.Add(slot); + if (Cache.LangProject.MorphologicalDataOA.ProdRestrictOA == null) + { + Cache.LangProject.MorphologicalDataOA.ProdRestrictOA = + Cache.ServiceLocator.GetInstance().Create(); + } + ICmPossibility restriction = Cache.ServiceLocator.GetInstance().Create(); + Cache.LangProject.MorphologicalDataOA.ProdRestrictOA.PossibilitiesOS.Add(restriction); + msa.ProdRestrictRC.Add(restriction); + ILexDb lexDb = Cache.LangProject.LexDbOA; + if (lexDb.VariantEntryTypesOA == null) + lexDb.VariantEntryTypesOA = Cache.ServiceLocator.GetInstance().Create(); + ILexEntryInflType inflType = Cache.ServiceLocator.GetInstance().Create(); + lexDb.VariantEntryTypesOA.PossibilitiesOS.Add(inflType); + inflType.SlotsRC.Add(slot); + IMoInflAffMsa inflMsa = Cache.ServiceLocator.GetInstance().Create(); + affixEntry.MorphoSyntaxAnalysesOC.Add(inflMsa); + inflMsa.PartOfSpeechRA = pos; + inflMsa.SlotsRC.Add(slot); + IMoDerivAffMsa derivMsa = Cache.ServiceLocator.GetInstance().Create(); + affixEntry.MorphoSyntaxAnalysesOC.Add(derivMsa); + derivMsa.FromPartOfSpeechRA = pos; + derivMsa.ToPartOfSpeechRA = pos; + // Complex-form entry ref. + if (lexDb.ComplexEntryTypesOA == null) + lexDb.ComplexEntryTypesOA = Cache.ServiceLocator.GetInstance().Create(); + ILexEntryType complexType = Cache.ServiceLocator.GetInstance().Create(); + lexDb.ComplexEntryTypesOA.PossibilitiesOS.Add(complexType); + ILexEntry compoundEntry = MakeStemEntry("kickball", "kickball"); + ILexEntryRef complexRef = Cache.ServiceLocator.GetInstance().Create(); + compoundEntry.EntryRefsOS.Add(complexRef); + complexRef.ComponentLexemesRS.Add(entry); + complexRef.ComplexEntryTypesRS.Add(complexType); + // Parser parameters. + Cache.LangProject.MorphologicalDataOA.ParserParameters = + "Morphology,Phonology"; + + AssertValidatesAgainstSchema(GrammarJsonServices.ExportGrammar(Cache), "populated project"); + } + + private static void AssertValidatesAgainstSchema(string json, string description) + { + string schemaPath = Path.Combine(TestContext.CurrentContext.TestDirectory, + "doc", "lcm-grammar.schema.json"); + var schema = NJsonSchema.JsonSchema.FromJsonAsync(File.ReadAllText(schemaPath)) + .GetAwaiter().GetResult(); + var errors = schema.Validate(json); + Assert.IsEmpty(errors, description + " export should validate against doc/lcm-grammar.schema.json: " + + string.Join("; ", errors.Select(e => e.ToString()))); + } + + /// + /// A sense whose MSA belongs to a different entry (stray but real-world data) is carried + /// as-is with a warning — msa references resolve document-wide, not per-entry. + /// + [Test] + public void ExportGrammar_WarnsOnCrossEntrySenseMsa() + { + ILexEntry owner = MakeStemEntry("kick", "kick"); + IMoStemMsa foreignMsa = Cache.ServiceLocator.GetInstance().Create(); + owner.MorphoSyntaxAnalysesOC.Add(foreignMsa); + ILexEntry stray = MakeStemEntry("sing", "sing"); + stray.SensesOS[0].MorphoSyntaxAnalysisRA = foreignMsa; + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + JObject strayEntry = (JObject)json["lexicon"]["entries"] + .Single(e => (string)e["guid"] == stray.Guid.ToString()); + Assert.AreEqual(foreignMsa.Guid.ToString(), (string)strayEntry["senses"][0]["msa"], + "the cross-entry reference must be carried as-is"); + Assert.IsTrue(warnings.Any(w => w.Contains(stray.SensesOS[0].Guid.ToString())), + "the cross-entry msa must be reported in warnings"); + } + + /// + /// Exported text is NFC even though LCM holds strings NFD in memory, so an independent + /// reader of the raw .fwdata XML (which is NFC at rest) reproduces the same bytes. + /// + [Test] + public void ExportGrammar_NormalizesTextToNfc() + { + ILexEntry entry = MakeStemEntry("kick", "kick"); + entry.SensesOS[0].Gloss.set_String(Cache.DefaultAnalWs, + TsStringUtils.MakeString("bambu\u0301", Cache.DefaultAnalWs)); + + string json = GrammarJsonServices.ExportGrammar(Cache); + + // Ordinal comparisons: culture-sensitive search treats NFC and NFD as equal. + Assert.IsTrue(json.IndexOf("bamb\u00FA", StringComparison.Ordinal) >= 0, + "output must be NFC"); + Assert.IsTrue(json.IndexOf("bambu\u0301", StringComparison.Ordinal) < 0, + "output must not carry NFD sequences"); + } + + /// + /// When every MSA on an entry is an unsupported class, the msas property is omitted + /// entirely (never written as an empty array, which the schema forbids). + /// + [Test] + public void ExportGrammar_OmitsMsasWhenAllUnsupported() + { + ILexEntry entry = MakeStemEntry("kick", "kick"); + IMoDerivStepMsa step = Cache.ServiceLocator.GetInstance().Create(); + entry.MorphoSyntaxAnalysesOC.Add(step); + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + JObject jsonEntry = (JObject)json["lexicon"]["entries"][0]; + Assert.IsNull(jsonEntry["msas"], "unsupported-only msas must be omitted, not []"); + Assert.IsTrue(warnings.Any(w => w.Contains(step.Guid.ToString())), + "the skipped MSA must be reported in warnings"); + AssertValidatesAgainstSchema(json.ToString(), "entry with only unsupported MSAs"); + } + + /// + /// A copy-from-input mapping that references something other than a top-level part of its + /// own process's input would emit a silently wrong positional index, so the whole + /// allomorph is skipped. + /// + [Test] + public void ExportGrammar_SkipsAffixProcessWithForeignInputReference() + { + ILexEntry entryA = Cache.ServiceLocator.GetInstance().Create(); + IMoAffixProcess processA = Cache.ServiceLocator.GetInstance().Create(); + entryA.LexemeFormOA = processA; + processA.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphSuffix); + ILexEntry entryB = Cache.ServiceLocator.GetInstance().Create(); + IMoAffixProcess processB = Cache.ServiceLocator.GetInstance().Create(); + entryB.LexemeFormOA = processB; + processB.MorphTypeRA = Cache.ServiceLocator.GetInstance() + .GetObject(MoMorphTypeTags.kguidMorphSuffix); + IPhVariable foreignPart = Cache.ServiceLocator.GetInstance().Create(); + processB.InputOS.Add(foreignPart); + IMoCopyFromInput copy = Cache.ServiceLocator.GetInstance().Create(); + processA.OutputOS.Add(copy); + copy.ContentRA = foreignPart; + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + JObject jsonEntryA = (JObject)json["lexicon"]["entries"] + .Single(e => (string)e["guid"] == entryA.Guid.ToString()); + Assert.AreEqual(0, ((JArray)jsonEntryA["allomorphs"]).Count, + "the allomorph with the foreign input reference must be skipped whole"); + Assert.IsTrue(warnings.Any(w => w.Contains(processA.Guid.ToString())), + "the skipped process must be reported in warnings"); + } + + /// Unrepresentable data is skipped with a warning, never silently. + [Test] + public void ExportGrammar_WarnsAndSkipsAllomorphWithoutMorphType() + { + ILexEntry entry = Cache.ServiceLocator.GetInstance().Create(); + IMoStemAllomorph lexemeForm = Cache.ServiceLocator.GetInstance().Create(); + entry.LexemeFormOA = lexemeForm; + lexemeForm.Form.SetVernacularDefaultWritingSystem("mystery"); + + var warnings = new List(); + JObject json = JObject.Parse(GrammarJsonServices.ExportGrammar(Cache, warnings)); + + JObject jsonEntry = (JObject)json["lexicon"]["entries"][0]; + Assert.AreEqual("stem", (string)jsonEntry["lexemeMorphType"], "defaults to stem"); + Assert.AreEqual(0, ((JArray)jsonEntry["allomorphs"]).Count, "allomorph skipped but array present"); + Assert.IsTrue(warnings.Any(w => w.Contains(lexemeForm.Guid.ToString())), + "skipped allomorph should be reported in warnings"); + } + } +} diff --git a/tests/SIL.LCModel.Tests/SIL.LCModel.Tests.csproj b/tests/SIL.LCModel.Tests/SIL.LCModel.Tests.csproj index a6afe657..afd3bbf3 100644 --- a/tests/SIL.LCModel.Tests/SIL.LCModel.Tests.csproj +++ b/tests/SIL.LCModel.Tests/SIL.LCModel.Tests.csproj @@ -14,6 +14,7 @@ This package provides unit tests for SIL.LCModel. + @@ -24,6 +25,10 @@ This package provides unit tests for SIL.LCModel. + + + +