diff --git a/src/__tests__/parsers/pt9/interlinearSetupXmlParser.test.ts b/src/__tests__/parsers/pt9/interlinearSetupXmlParser.test.ts
new file mode 100644
index 00000000..0b52bd80
--- /dev/null
+++ b/src/__tests__/parsers/pt9/interlinearSetupXmlParser.test.ts
@@ -0,0 +1,156 @@
+///
+
+import * as fs from 'node:fs';
+import * as path from 'node:path';
+
+import { InterlinearSetupXmlParser } from 'parsers/pt9/interlinearSetupXmlParser';
+
+describe('InterlinearSetupXmlParser', () => {
+ let parser: InterlinearSetupXmlParser;
+
+ beforeEach(() => {
+ parser = new InterlinearSetupXmlParser();
+ });
+
+ describe('parse() - valid XML', () => {
+ it('parses a setup with every field populated', () => {
+ const xml = `
+
+
+ French
+ Arial
+ 10
+ false
+ true
+ true
+ MDL
+ 1234567890abcdef
+ true
+ BT1
+ fedcba0987654321
+
+
+ `;
+
+ expect(parser.parse(xml)).toStrictEqual({
+ Setups: [
+ {
+ Type: 'BackTranslation',
+ LanguageId: 'fr',
+ LanguageName: 'French',
+ FontName: 'Arial',
+ FontSize: '10',
+ RightToLeft: false,
+ RelatedLanguages: true,
+ ExportOnApprove: true,
+ MdlScrTextName: 'MDL',
+ MdlScrTextId: '1234567890abcdef',
+ MdlIsResource: true,
+ ExportScrTextName: 'BT1',
+ ExportScrTextId: 'fedcba0987654321',
+ },
+ ],
+ });
+ });
+
+ it('parses an empty root element as no setups', () => {
+ expect(parser.parse('')).toStrictEqual({ Setups: [] });
+ });
+
+ it('parses a root with no InterlinearSetup children as no setups', () => {
+ expect(parser.parse('')).toStrictEqual({
+ Setups: [],
+ });
+ });
+
+ it('parses an empty InterlinearSetup element as a setup with no fields', () => {
+ const xml = `
+
+
+
+ `;
+ expect(parser.parse(xml)).toStrictEqual({ Setups: [{}] });
+ });
+
+ it('keeps absent fields absent on a setup with attributes only', () => {
+ const xml = `
+
+
+
+ `;
+ expect(parser.parse(xml)).toStrictEqual({
+ Setups: [{ Type: 'Glossing', LanguageId: 'en' }],
+ });
+ });
+
+ it('parses an unrecognized boolean element text as false', () => {
+ const xml = `
+
+
+ maybe
+
+
+ `;
+ expect(parser.parse(xml).Setups[0].RightToLeft).toBe(false);
+ });
+
+ it('parses an unknown interlinear type name as its raw string', () => {
+ const xml = `
+
+
+
+ `;
+ expect(parser.parse(xml).Setups[0].Type).toBe('FutureType');
+ });
+
+ it('parses the real test-data setup fixture', () => {
+ const xmlPath = path.join(
+ __dirname,
+ '..',
+ '..',
+ '..',
+ '..',
+ 'test-data',
+ 'InterlinearSetup.xml',
+ );
+ const result = parser.parse(fs.readFileSync(xmlPath, 'utf-8'));
+
+ expect(result.Setups).toStrictEqual([
+ {
+ Type: 'Glossing',
+ LanguageId: 'en',
+ LanguageName: 'English',
+ FontName: 'Charis SIL',
+ FontSize: '12',
+ RightToLeft: false,
+ RelatedLanguages: false,
+ ExportOnApprove: false,
+ },
+ {
+ Type: 'BackTranslation',
+ LanguageId: 'fr',
+ LanguageName: 'French',
+ MdlScrTextName: 'MDL',
+ MdlScrTextId: '1234567890abcdef',
+ MdlIsResource: true,
+ ExportOnApprove: true,
+ ExportScrTextName: 'BT1',
+ ExportScrTextId: 'fedcba0987654321',
+ },
+ ]);
+ });
+ });
+
+ describe('parse() - invalid XML / errors', () => {
+ it('throws when the InterlinearSetupList root element is absent', () => {
+ expect(() => parser.parse('')).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining(
+ 'Invalid XML: Missing InterlinearSetupList root element',
+ ),
+ }),
+ );
+ });
+ });
+});
diff --git a/src/__tests__/parsers/pt9/lexemeKey.test.ts b/src/__tests__/parsers/pt9/lexemeKey.test.ts
new file mode 100644
index 00000000..d5d548d1
--- /dev/null
+++ b/src/__tests__/parsers/pt9/lexemeKey.test.ts
@@ -0,0 +1,106 @@
+///
+
+import {
+ composeLexemeKeyId,
+ LexemeKeyData,
+ lexemeKeysEqual,
+ parseLexemeKeyId,
+} from 'parsers/pt9/lexemeKey';
+
+describe('parseLexemeKeyId', () => {
+ it('parses a plain id with no homograph suffix', () => {
+ expect(parseLexemeKeyId('Word:hello')).toStrictEqual({ Type: 'Word', Form: 'hello' });
+ });
+
+ it('parses a trailing :digits segment as the homograph', () => {
+ expect(parseLexemeKeyId('Word:a:2')).toStrictEqual({ Type: 'Word', Form: 'a', Homograph: 2 });
+ });
+
+ it('parses an explicit :1 suffix as homograph 1', () => {
+ expect(parseLexemeKeyId('Word:a:1')).toStrictEqual({ Type: 'Word', Form: 'a', Homograph: 1 });
+ });
+
+ it('keeps interior colons in the form and reads only the trailing digits as homograph', () => {
+ expect(parseLexemeKeyId('Stem:foo:bar:3')).toStrictEqual({
+ Type: 'Stem',
+ Form: 'foo:bar',
+ Homograph: 3,
+ });
+ });
+
+ it('keeps a non-digit trailing segment in the form', () => {
+ expect(parseLexemeKeyId('Word:a:b')).toStrictEqual({ Type: 'Word', Form: 'a:b' });
+ });
+
+ it('parses an empty form', () => {
+ expect(parseLexemeKeyId('Word:')).toStrictEqual({ Type: 'Word', Form: '' });
+ });
+
+ it('parses a form containing spaces (phrase lexemes)', () => {
+ expect(parseLexemeKeyId('Phrase:hello world')).toStrictEqual({
+ Type: 'Phrase',
+ Form: 'hello world',
+ });
+ });
+
+ it.each(['hello', '', ':x', 'Word-x'])('returns undefined for non-matching id "%s"', (id) => {
+ expect(parseLexemeKeyId(id)).toBeUndefined();
+ });
+});
+
+describe('composeLexemeKeyId', () => {
+ it('omits an absent homograph', () => {
+ expect(composeLexemeKeyId({ Type: 'Word', Form: 'hello' })).toBe('Word:hello');
+ });
+
+ it('omits homograph 1', () => {
+ expect(composeLexemeKeyId({ Type: 'Word', Form: 'hello', Homograph: 1 })).toBe('Word:hello');
+ });
+
+ it('appends a homograph greater than 1', () => {
+ expect(composeLexemeKeyId({ Type: 'Word', Form: 'a', Homograph: 2 })).toBe('Word:a:2');
+ });
+
+ it('produces an id that re-parses with a :digits form tail read as the homograph', () => {
+ const key: LexemeKeyData = { Type: 'Word', Form: 'a:1' };
+ expect(parseLexemeKeyId(composeLexemeKeyId(key))).toStrictEqual({
+ Type: 'Word',
+ Form: 'a',
+ Homograph: 1,
+ });
+ });
+});
+
+describe('lexemeKeysEqual', () => {
+ it('treats identical keys as equal', () => {
+ expect(
+ lexemeKeysEqual(
+ { Type: 'Word', Form: 'a', Homograph: 2 },
+ { Type: 'Word', Form: 'a', Homograph: 2 },
+ ),
+ ).toBe(true);
+ });
+
+ it('treats an absent homograph as homograph 1 on either side', () => {
+ expect(
+ lexemeKeysEqual({ Type: 'Word', Form: 'a' }, { Type: 'Word', Form: 'a', Homograph: 1 }),
+ ).toBe(true);
+ expect(
+ lexemeKeysEqual({ Type: 'Word', Form: 'a', Homograph: 1 }, { Type: 'Word', Form: 'a' }),
+ ).toBe(true);
+ });
+
+ it('distinguishes types', () => {
+ expect(lexemeKeysEqual({ Type: 'Word', Form: 'a' }, { Type: 'Stem', Form: 'a' })).toBe(false);
+ });
+
+ it('distinguishes forms', () => {
+ expect(lexemeKeysEqual({ Type: 'Word', Form: 'a' }, { Type: 'Word', Form: 'b' })).toBe(false);
+ });
+
+ it('distinguishes homographs', () => {
+ expect(
+ lexemeKeysEqual({ Type: 'Word', Form: 'a' }, { Type: 'Word', Form: 'a', Homograph: 2 }),
+ ).toBe(false);
+ });
+});
diff --git a/src/__tests__/parsers/pt9/lexiconXmlParser.test.ts b/src/__tests__/parsers/pt9/lexiconXmlParser.test.ts
new file mode 100644
index 00000000..72fa44c3
--- /dev/null
+++ b/src/__tests__/parsers/pt9/lexiconXmlParser.test.ts
@@ -0,0 +1,519 @@
+///
+
+import * as fs from 'node:fs';
+import * as path from 'node:path';
+
+import { LexiconXmlParser } from 'parsers/pt9/lexiconXmlParser';
+
+describe('LexiconXmlParser', () => {
+ let parser: LexiconXmlParser;
+
+ beforeEach(() => {
+ parser = new LexiconXmlParser();
+ });
+
+ describe('parse() - valid XML', () => {
+ it('parses a minimal lexicon with one entry', () => {
+ const xml = `
+
+ en
+ Arial
+ 10
+
+ -
+
+
+
+ is
+
+
+
+
+
+ `;
+
+ expect(parser.parse(xml)).toStrictEqual({
+ Language: 'en',
+ FontName: 'Arial',
+ FontSize: '10',
+ Entries: [
+ {
+ Key: { Type: 'Word', Form: 'voici', Homograph: 1 },
+ Senses: [{ Id: 'CKVPllxu', Glosses: [{ Language: 'en', Text: 'is' }] }],
+ },
+ ],
+ Analyses: {},
+ });
+ });
+
+ it('parses an empty root element as an empty lexicon', () => {
+ expect(parser.parse('')).toStrictEqual({ Entries: [], Analyses: {} });
+ });
+
+ it('parses a lexicon with no Entries or Analyses containers', () => {
+ expect(parser.parse('fr')).toStrictEqual({
+ Language: 'fr',
+ Entries: [],
+ Analyses: {},
+ });
+ });
+
+ it('parses empty Entries and Analyses containers as empty collections', () => {
+ expect(parser.parse('')).toStrictEqual({
+ Entries: [],
+ Analyses: {},
+ });
+ });
+
+ it('parses containers with no item children as empty collections', () => {
+ const xml = `
+
+
+
+
+ `;
+ expect(parser.parse(xml)).toStrictEqual({ Entries: [], Analyses: {} });
+ });
+
+ it('preserves an absent Homograph attribute as an absent field', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+ `;
+ const result = parser.parse(xml);
+
+ expect(result.Entries[0].Key).toStrictEqual({ Type: 'Stem', Form: 'exauc' });
+ });
+
+ it('parses an empty Entry element as an entry with no senses', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses).toStrictEqual([]);
+ });
+
+ it('parses an item with no Entry element as an entry with no senses', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses).toStrictEqual([]);
+ });
+
+ it('parses an empty Sense element as a sense with no id and no glosses', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses).toStrictEqual([{ Glosses: [] }]);
+ });
+
+ it('parses a Sense with an Id and no glosses', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses).toStrictEqual([{ Id: 'k2PH7X/I', Glosses: [] }]);
+ });
+
+ it('parses a Gloss with no Language attribute as text with an absent Language', () => {
+ const xml = `
+
+
+ -
+
+
+
+ bare
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses[0].Glosses).toStrictEqual([{ Text: 'bare' }]);
+ });
+
+ it('parses an empty Gloss element as an empty string text', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses[0].Glosses).toStrictEqual([
+ { Language: 'en', Text: '' },
+ ]);
+ });
+
+ it('parses an Entry containing no Sense elements as an entry with no senses', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses).toStrictEqual([]);
+ });
+
+ it('parses a Gloss carrying only a foreign attribute as text with an absent Language', () => {
+ const xml = `
+
+
+ -
+
+
+
+ bare
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses[0].Glosses).toStrictEqual([{ Text: 'bare' }]);
+ });
+
+ it('parses an ArrayOfLexeme containing no Lexeme elements as an analysis with no lexemes', () => {
+ const xml = `
+
+
+ -
+ word
+
+
+
+
+ `;
+ expect(parser.parse(xml).Analyses).toStrictEqual({ word: [] });
+ });
+
+ it('parses multiple glosses per sense in document order', () => {
+ const xml = `
+
+
+ -
+
+
+
+ one
+ un
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Senses[0].Glosses).toStrictEqual([
+ { Language: 'en', Text: 'one' },
+ { Language: 'fr', Text: 'un' },
+ ]);
+ });
+
+ it('parses legacy Analyses items with their lexeme keys', () => {
+ const xml = `
+
+
+ -
+ exaucera
+
+
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Analyses).toStrictEqual({
+ exaucera: [
+ { Type: 'Stem', Form: 'exauc', Homograph: 1 },
+ { Type: 'Suffix', Form: 'era', Homograph: 1 },
+ ],
+ });
+ });
+
+ it('parses an empty ArrayOfLexeme as an analysis with no lexemes', () => {
+ const xml = `
+
+
+ -
+ word
+
+
+
+
+ `;
+ expect(parser.parse(xml).Analyses).toStrictEqual({ word: [] });
+ });
+
+ it('parses an Analyses item with no ArrayOfLexeme as an analysis with no lexemes', () => {
+ const xml = `
+
+
+ -
+ word
+
+
+
+ `;
+ expect(parser.parse(xml).Analyses).toStrictEqual({ word: [] });
+ });
+
+ it('parses the real test-data lexicon fixture', () => {
+ const xmlPath = path.join(__dirname, '..', '..', '..', '..', 'test-data', 'Lexicon.xml');
+ const result = parser.parse(fs.readFileSync(xmlPath, 'utf-8'));
+
+ expect(result.Language).toBe('en');
+ expect(result.Entries).toHaveLength(7);
+
+ const hello = result.Entries.find((e) => e.Key.Type === 'Word' && e.Key.Form === 'hello');
+ expect(hello?.Senses[0].Id).toBe('WvbPwa9D');
+ expect(hello?.Senses[0].Glosses).toStrictEqual([
+ { Language: 'en', Text: 'greeting' },
+ { Language: 'fr', Text: 'salut' },
+ ]);
+
+ const homographs = result.Entries.filter((e) => e.Key.Type === 'Word' && e.Key.Form === 'a');
+ expect(homographs.map((e) => e.Key.Homograph)).toStrictEqual([1, 2]);
+
+ const senselessStem = result.Entries.find(
+ (e) => e.Key.Type === 'Stem' && e.Key.Form === 'ab',
+ );
+ expect(senselessStem?.Senses).toStrictEqual([]);
+
+ expect(result.Analyses).toStrictEqual({
+ aaaa: [{ Type: 'Stem', Form: 'aaaa', Homograph: 1 }],
+ });
+ });
+ });
+
+ describe('parse() - invalid XML / errors', () => {
+ it('throws when the Lexicon root element is absent', () => {
+ expect(() => parser.parse('')).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Missing Lexicon root element'),
+ }),
+ );
+ });
+
+ it('throws when an Entries item has no Lexeme key element', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining(
+ 'Invalid XML: Entries item missing its Lexeme key element',
+ ),
+ }),
+ );
+ });
+
+ it.each([
+ ['', 'missing Type'],
+ ['', 'missing Form'],
+ ['', 'empty Type'],
+ ])('throws when the key element is %s (%s)', (lexeme) => {
+ const xml = `
+
+
+ -
+ ${lexeme}
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining(
+ 'Invalid XML: Lexeme key missing Type or Form attribute',
+ ),
+ }),
+ );
+ });
+
+ it.each(['x', '-1', '1.5', ''])(
+ 'throws when a Homograph attribute is the non-numeric "%s"',
+ (homograph) => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('non-numeric Homograph attribute'),
+ }),
+ );
+ },
+ );
+
+ it('throws on duplicate entry keys', () => {
+ const xml = `
+
+
+ -
+
+
+
+ -
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Duplicate lexicon entry key "Word:a:2"'),
+ }),
+ );
+ });
+
+ it('throws on duplicate entry keys when one side writes Homograph="1" and the other omits it', () => {
+ const xml = `
+
+
+ -
+
+
+
+ -
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Duplicate lexicon entry key "Word:a"'),
+ }),
+ );
+ });
+
+ it('throws when an Analyses item has no wordform key', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Analyses item missing its wordform key'),
+ }),
+ );
+ });
+
+ it('throws when an Analyses wordform key is empty', () => {
+ const xml = `
+
+
+ -
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Analyses item missing its wordform key'),
+ }),
+ );
+ });
+
+ it('throws on duplicate analyses wordforms', () => {
+ const xml = `
+
+
+ -
+ word
+
+
+ -
+ word
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Duplicate analyses wordform "word"'),
+ }),
+ );
+ });
+ });
+});
diff --git a/src/__tests__/parsers/pt9/wordAnalysesXmlParser.test.ts b/src/__tests__/parsers/pt9/wordAnalysesXmlParser.test.ts
new file mode 100644
index 00000000..70b578a5
--- /dev/null
+++ b/src/__tests__/parsers/pt9/wordAnalysesXmlParser.test.ts
@@ -0,0 +1,170 @@
+///
+
+import * as fs from 'node:fs';
+import * as path from 'node:path';
+
+import { WordAnalysesXmlParser } from 'parsers/pt9/wordAnalysesXmlParser';
+
+describe('WordAnalysesXmlParser', () => {
+ let parser: WordAnalysesXmlParser;
+
+ beforeEach(() => {
+ parser = new WordAnalysesXmlParser();
+ });
+
+ describe('parse() - valid XML', () => {
+ it('parses an entry with one analysis of two lexemes', () => {
+ const xml = `
+
+
+
+ Stem:exauc
+ Suffix:era
+
+
+
+ `;
+
+ expect(parser.parse(xml)).toStrictEqual({
+ Entries: [{ Word: 'exaucera', Analyses: [{ LexemeIds: ['Stem:exauc', 'Suffix:era'] }] }],
+ });
+ });
+
+ it('parses an empty root element as an empty inventory', () => {
+ expect(parser.parse('')).toStrictEqual({ Entries: [] });
+ });
+
+ it('parses a root with no Entry children as an empty inventory', () => {
+ expect(parser.parse('')).toStrictEqual({
+ Entries: [],
+ });
+ });
+
+ it('parses multiple analyses for one wordform in document order', () => {
+ const xml = `
+
+
+
+ Stem:ab
+ Suffix:e
+
+
+ Stem:abe
+
+
+
+ `;
+
+ expect(parser.parse(xml).Entries[0].Analyses).toStrictEqual([
+ { LexemeIds: ['Stem:ab', 'Suffix:e'] },
+ { LexemeIds: ['Stem:abe'] },
+ ]);
+ });
+
+ it('parses an empty Analysis element as an analysis with no lexemes', () => {
+ const xml = `
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Analyses).toStrictEqual([{ LexemeIds: [] }]);
+ });
+
+ it('parses an Analysis containing no Lexeme elements as an analysis with no lexemes', () => {
+ const xml = `
+
+
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0].Analyses).toStrictEqual([{ LexemeIds: [] }]);
+ });
+
+ it('parses an Entry with no Analysis children as an entry with no analyses', () => {
+ const xml = `
+
+
+
+ `;
+ expect(parser.parse(xml).Entries[0]).toStrictEqual({ Word: 'word', Analyses: [] });
+ });
+
+ it('parses the real test-data word-analyses fixture', () => {
+ const xmlPath = path.join(__dirname, '..', '..', '..', '..', 'test-data', 'WordAnalyses.xml');
+ const result = parser.parse(fs.readFileSync(xmlPath, 'utf-8'));
+
+ expect(result.Entries).toStrictEqual([
+ { Word: 'helloing', Analyses: [{ LexemeIds: ['Stem:hello', 'Suffix:ing'] }] },
+ {
+ Word: 'abe',
+ Analyses: [{ LexemeIds: ['Stem:ab', 'Suffix:e'] }, { LexemeIds: ['Stem:abe'] }],
+ },
+ ]);
+ });
+ });
+
+ describe('parse() - invalid XML / errors', () => {
+ it('throws when the WordAnalyses root element is absent', () => {
+ expect(() => parser.parse('')).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Missing WordAnalyses root element'),
+ }),
+ );
+ });
+
+ it('throws when an Entry is missing its Word attribute', () => {
+ const xml = `
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Entry missing its Word attribute'),
+ }),
+ );
+ });
+
+ it('throws when an Entry Word attribute is empty', () => {
+ const xml = `
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Entry missing its Word attribute'),
+ }),
+ );
+ });
+
+ it('throws on duplicate wordform entries', () => {
+ const xml = `
+
+
+
+
+
+
+
+
+ `;
+ expect(() => parser.parse(xml)).toThrow(
+ expect.objectContaining({
+ name: 'SyntaxError',
+ message: expect.stringContaining('Invalid XML: Duplicate word analyses entry "word"'),
+ }),
+ );
+ });
+ });
+});
diff --git a/src/parsers/pt9/interlinearSetupXmlParser.ts b/src/parsers/pt9/interlinearSetupXmlParser.ts
new file mode 100644
index 00000000..a457268b
--- /dev/null
+++ b/src/parsers/pt9/interlinearSetupXmlParser.ts
@@ -0,0 +1,148 @@
+import { X2jOptions, XMLParser } from 'fast-xml-parser';
+
+/**
+ * One per-gloss-language interlinear configuration. Every field is optional and preserved as
+ * written; nothing here is validated against PT9's enums so files from future PT9 versions parse.
+ */
+export interface InterlinearSetupData {
+ /**
+ * Interlinear type name (XML attribute type) — e.g. `"BackTranslation"`, `"Glossing"`,
+ * `"Adaptation"`. Kept as the raw string; PT9's list of names may grow.
+ */
+ Type?: string;
+ /** Gloss language id (XML attribute language); keys the `Interlinear_{language}` directory. */
+ LanguageId?: string;
+ LanguageName?: string;
+ FontName?: string;
+ /** Kept as the raw element text rather than a number. */
+ FontSize?: string;
+ RightToLeft?: boolean;
+ /** Whether PT9's related-language gloss guessing is enabled for this setup. */
+ RelatedLanguages?: boolean;
+ /** Whether approving a verse also exports it to the export project. */
+ ExportOnApprove?: boolean;
+ /** Name of the model text this setup glosses against. */
+ MdlScrTextName?: string;
+ /** Hex id of the model text, kept as the raw string. */
+ MdlScrTextId?: string;
+ MdlIsResource?: boolean;
+ /** Name of the project the interlinearization exports into. */
+ ExportScrTextName?: string;
+ /** Hex id of the export project, kept as the raw string. */
+ ExportScrTextId?: string;
+}
+
+/** Root setups data: one entry per configured gloss language. */
+export interface InterlinearSetupsData {
+ /** Setups in document order. */
+ Setups: InterlinearSetupData[];
+}
+
+/** InterlinearSetup: type/language attributes plus text elements; empty parses as a bare string. */
+type ParsedSetup =
+ | string
+ | {
+ ['@_type']?: string;
+ ['@_language']?: string;
+ LanguageName?: string;
+ FontName?: string;
+ FontSize?: string;
+ RightToLeft?: string;
+ RelatedLanguages?: string;
+ ExportOnApprove?: string;
+ MdlScrTextName?: string;
+ MdlScrTextId?: string;
+ MdlIsResource?: string;
+ ExportScrTextName?: string;
+ ExportScrTextId?: string;
+ };
+
+/**
+ * Root InterlinearSetupList element; an empty element parses as a bare string. The string carries
+ * no data; it marks the root as present so a file with no configured setups parses as valid rather
+ * than erroring as a missing root.
+ */
+type ParsedSetupListRoot = string | { InterlinearSetup?: ParsedSetup[] };
+
+/** Root document: InterlinearSetupList. */
+interface ParsedSetupXml {
+ InterlinearSetupList?: ParsedSetupListRoot;
+}
+
+/**
+ * Parses a serialized boolean element's text, treating any value other than `"true"` as false. An
+ * absent element stays absent.
+ */
+function parseBool(raw: string | undefined): boolean | undefined {
+ if (raw === undefined) return undefined;
+ return raw === 'true';
+}
+
+/** Maps a parsed InterlinearSetup to {@link InterlinearSetupData}; a bare string is an empty setup. */
+function extractSetup(setup: ParsedSetup): InterlinearSetupData {
+ if (typeof setup === 'string') return {};
+ const rightToLeft = parseBool(setup.RightToLeft);
+ const relatedLanguages = parseBool(setup.RelatedLanguages);
+ const exportOnApprove = parseBool(setup.ExportOnApprove);
+ const mdlIsResource = parseBool(setup.MdlIsResource);
+ return {
+ ...(setup['@_type'] !== undefined && { Type: setup['@_type'] }),
+ ...(setup['@_language'] !== undefined && { LanguageId: setup['@_language'] }),
+ ...(setup.LanguageName !== undefined && { LanguageName: setup.LanguageName }),
+ ...(setup.FontName !== undefined && { FontName: setup.FontName }),
+ ...(setup.FontSize !== undefined && { FontSize: setup.FontSize }),
+ ...(rightToLeft !== undefined && { RightToLeft: rightToLeft }),
+ ...(relatedLanguages !== undefined && { RelatedLanguages: relatedLanguages }),
+ ...(exportOnApprove !== undefined && { ExportOnApprove: exportOnApprove }),
+ ...(setup.MdlScrTextName !== undefined && { MdlScrTextName: setup.MdlScrTextName }),
+ ...(setup.MdlScrTextId !== undefined && { MdlScrTextId: setup.MdlScrTextId }),
+ ...(mdlIsResource !== undefined && { MdlIsResource: mdlIsResource }),
+ ...(setup.ExportScrTextName !== undefined && { ExportScrTextName: setup.ExportScrTextName }),
+ ...(setup.ExportScrTextId !== undefined && { ExportScrTextId: setup.ExportScrTextId }),
+ };
+}
+
+/**
+ * Parses PT9 `InterlinearSetup.xml` strings into {@link InterlinearSetupsData}.
+ *
+ * Setups carry configuration only, so parsing is fully lenient: every field is optional and unknown
+ * enum names survive as raw strings. Expects the schema described in [pt9-xml.md](pt9-xml.md).
+ *
+ * Each instance holds a configured `XMLParser`; create one parser and reuse it across multiple
+ * `parse()` calls rather than constructing a new instance per file.
+ */
+export class InterlinearSetupXmlParser {
+ private readonly parser: XMLParser;
+
+ constructor() {
+ const arrayPaths = new Set(['InterlinearSetupList.InterlinearSetup']);
+
+ const options: Partial = {
+ ignoreAttributes: false,
+ attributeNamePrefix: '@_',
+ ignoreDeclaration: true,
+ ignorePiTags: true,
+ trimValues: false,
+ parseTagValue: false,
+ parseAttributeValue: false,
+ isArray: (_tagName, jPath) => arrayPaths.has(`${jPath}`),
+ };
+ this.parser = new XMLParser(options);
+ }
+
+ /**
+ * Parses an `InterlinearSetup.xml` string into {@link InterlinearSetupsData}.
+ *
+ * @throws {SyntaxError} If the `InterlinearSetupList` root element is absent.
+ */
+ parse(xml: string): InterlinearSetupsData {
+ const parsed: ParsedSetupXml = this.parser.parse(xml);
+ const root = parsed.InterlinearSetupList;
+ if (root === undefined) {
+ throw new SyntaxError('Invalid XML: Missing InterlinearSetupList root element');
+ }
+ if (typeof root === 'string') return { Setups: [] };
+
+ return { Setups: (root.InterlinearSetup ?? []).map(extractSetup) };
+ }
+}
diff --git a/src/parsers/pt9/lexemeKey.ts b/src/parsers/pt9/lexemeKey.ts
new file mode 100644
index 00000000..96cd8e82
--- /dev/null
+++ b/src/parsers/pt9/lexemeKey.ts
@@ -0,0 +1,71 @@
+/**
+ * A PT9 lexeme key: the identity of a lexicon entry. Appears in PT9's XML in two shapes — as a
+ * composed id string (e.g. `"Stem:exauc"`, `"Word:a:2"`) and as an attribute triple on `Lexeme`
+ * elements — both of which this type represents.
+ */
+export interface LexemeKeyData {
+ /** Lexeme type name (e.g. `"Word"`, `"Stem"`). PT9 may add names, so unknown values are legal. */
+ Type: string;
+ /** Lexical form as written in the file. */
+ Form: string;
+ /**
+ * Homograph number. Absent when the XML carries none (an id without a homograph suffix, or a
+ * `Lexeme` element without the attribute); PT9 treats absence as homograph 1.
+ */
+ Homograph?: number;
+}
+
+/**
+ * Lexeme type names PT9 defines. Ids in the wild are expected to use these, but parsing does not
+ * require it — PT9 treats its type list as append-only, so unknown names must survive.
+ */
+export const KNOWN_LEXEME_TYPES = [
+ 'Phrase',
+ 'Word',
+ 'Lemma',
+ 'Stem',
+ 'Prefix',
+ 'Suffix',
+ 'Infix',
+] as const;
+
+/**
+ * PT9's id grammar: `Type:Form` with an optional `:digits` homograph suffix. The lazy form group
+ * lets forms contain colons, while a trailing `:digits` always reads as the homograph — matching
+ * PT9's own parsing of ambiguous ids.
+ */
+const LEXEME_KEY_ID_RE = /^(\w+):(.*?)(?::([0-9]+))?$/;
+
+/**
+ * Parses a composed lexeme-key id string.
+ *
+ * @returns The parsed key, or `undefined` when the string does not match PT9's id grammar. A
+ * trailing `:digits` segment is returned as `Homograph`; without one, `Homograph` is absent.
+ */
+export function parseLexemeKeyId(id: string): LexemeKeyData | undefined {
+ const match = LEXEME_KEY_ID_RE.exec(id);
+ if (!match) return undefined;
+ const [, type, form, homograph] = match;
+ return {
+ Type: type,
+ Form: form,
+ ...(homograph !== undefined && { Homograph: Number.parseInt(homograph, 10) }),
+ };
+}
+
+/**
+ * Composes a lexeme key into its id string, omitting homograph 1 the way PT9 does.
+ *
+ * A form whose text ends in `:digits` produces an id that parses back with that tail read as the
+ * homograph — the ambiguity is inherent to PT9's id grammar, not avoidable here.
+ */
+export function composeLexemeKeyId(key: LexemeKeyData): string {
+ const homograph = key.Homograph ?? 1;
+ const base = `${key.Type}:${key.Form}`;
+ return homograph === 1 ? base : `${base}:${homograph}`;
+}
+
+/** Compares two keys by identity, treating an absent homograph as homograph 1. */
+export function lexemeKeysEqual(a: LexemeKeyData, b: LexemeKeyData): boolean {
+ return a.Type === b.Type && a.Form === b.Form && (a.Homograph ?? 1) === (b.Homograph ?? 1);
+}
diff --git a/src/parsers/pt9/lexiconXmlParser.ts b/src/parsers/pt9/lexiconXmlParser.ts
new file mode 100644
index 00000000..a28d5d6f
--- /dev/null
+++ b/src/parsers/pt9/lexiconXmlParser.ts
@@ -0,0 +1,278 @@
+import { X2jOptions, XMLParser } from 'fast-xml-parser';
+
+import { composeLexemeKeyId, LexemeKeyData } from './lexemeKey';
+
+/** One per-language gloss on a sense. */
+export interface LexiconGlossData {
+ /** BCP 47 tag or legacy language name (XML attribute Language). Absent when the file omits it. */
+ Language?: string;
+ /** Gloss text; an empty element yields an empty string. */
+ Text: string;
+}
+
+/** One sense of a lexicon entry. */
+export interface LexiconSenseData {
+ /**
+ * Sense id (XML attribute Id) — 8 chars of Base64 in PT9-written files, so `+` and `/` are legal.
+ * Absent when the file omits the attribute; such a sense cannot be referenced by interlinear
+ * data.
+ */
+ Id?: string;
+ /** Glosses in document order; empty when the sense has none. */
+ Glosses: LexiconGlossData[];
+}
+
+/** One lexicon entry: a lexeme key and its senses. */
+export interface LexiconEntryData {
+ /** Identity of the entry. */
+ Key: LexemeKeyData;
+ /**
+ * Senses in document order; empty for entries with an empty `Entry` element (common for
+ * morphemes).
+ */
+ Senses: LexiconSenseData[];
+}
+
+/**
+ * Root lexicon data. `Language`, `FontName`, and `FontSize` are preserved as written, but PT9
+ * overwrites them from project settings on every load — treat them as informational, not
+ * authoritative.
+ */
+export interface LexiconData {
+ /** Project language id or legacy name. */
+ Language?: string;
+ FontName?: string;
+ /** Kept as the raw attribute text rather than a number. */
+ FontSize?: string;
+ /** Lexicon entries in document order. */
+ Entries: LexiconEntryData[];
+ /**
+ * Legacy word analyses: each surface wordform's ordered morpheme keys, one record entry per
+ * wordform in document order. PT9 drains these into `WordAnalyses.xml` on read, but projects
+ * untouched since PT8 still carry them here.
+ */
+ Analyses: Record;
+}
+
+/** Lexeme key element: Type, Form, optional Homograph attributes. */
+interface ParsedLexemeKey {
+ ['@_Type']?: string;
+ ['@_Form']?: string;
+ ['@_Homograph']?: string;
+}
+
+/** Gloss: Language attribute plus text content; a text-only element parses as a bare string. */
+type ParsedGloss = string | { ['@_Language']?: string; ['#text']?: string };
+
+/** Sense: Id attribute and Gloss children; an empty element parses as a bare string. */
+type ParsedSense = string | { ['@_Id']?: string; Gloss?: ParsedGloss[] };
+
+/** Entry: Sense children; an empty element parses as a bare string. */
+type ParsedEntry = string | { Sense?: ParsedSense[] };
+
+/** Entries item: the Lexeme key element plus the Entry value. */
+interface ParsedEntriesItem {
+ Lexeme?: ParsedLexemeKey;
+ Entry?: ParsedEntry;
+}
+
+/** ArrayOfLexeme: Lexeme key children; an empty element parses as a bare string. */
+type ParsedArrayOfLexeme = string | { Lexeme?: ParsedLexemeKey[] };
+
+/** Analyses item: the wordform key plus its lexeme list. */
+interface ParsedAnalysesItem {
+ string?: string;
+ ArrayOfLexeme?: ParsedArrayOfLexeme;
+}
+
+/**
+ * Root Lexicon element; an empty element parses as a bare string. The string carries no data; it
+ * marks the root as present so an empty lexicon parses as valid rather than erroring as a missing
+ * root.
+ */
+type ParsedLexiconRoot =
+ | string
+ | {
+ Language?: string;
+ FontName?: string;
+ FontSize?: string;
+ Entries?: string | { item?: ParsedEntriesItem[] };
+ Analyses?: string | { item?: ParsedAnalysesItem[] };
+ };
+
+/** Root document: Lexicon. */
+interface ParsedLexiconXml {
+ Lexicon?: ParsedLexiconRoot;
+}
+
+/**
+ * Maps a parsed key element to {@link LexemeKeyData}, preserving an absent Homograph attribute as an
+ * absent field.
+ *
+ * @throws {SyntaxError} If the element is missing Type or Form, or Homograph is not a non-negative
+ * integer.
+ */
+function extractLexemeKey(element: ParsedLexemeKey): LexemeKeyData {
+ const type = element['@_Type'];
+ const form = element['@_Form'];
+ if (!type || form === undefined) {
+ throw new SyntaxError('Invalid XML: Lexeme key missing Type or Form attribute');
+ }
+ const homographRaw = element['@_Homograph'];
+ if (homographRaw === undefined) return { Type: type, Form: form };
+ if (!/^\d+$/.test(homographRaw)) {
+ throw new SyntaxError(
+ `Invalid XML: Lexeme key has non-numeric Homograph attribute "${homographRaw}"`,
+ );
+ }
+ return { Type: type, Form: form, Homograph: Number.parseInt(homographRaw, 10) };
+}
+
+/** Maps a parsed Gloss to {@link LexiconGlossData}; a bare string is text with no Language. */
+function extractGloss(gloss: ParsedGloss): LexiconGlossData {
+ if (typeof gloss === 'string') return { Text: gloss };
+ const language = gloss['@_Language'];
+ return {
+ ...(language !== undefined && { Language: language }),
+ Text: gloss['#text'] ?? '',
+ };
+}
+
+/**
+ * Maps a parsed Sense to {@link LexiconSenseData}. A bare string is a sense with no id or glosses;
+ * nothing can link to such a sense, but it is retained for completeness.
+ */
+function extractSense(sense: ParsedSense): LexiconSenseData {
+ if (typeof sense === 'string') return { Glosses: [] };
+ const id = sense['@_Id'];
+ return {
+ ...(id !== undefined && { Id: id }),
+ Glosses: (sense.Gloss ?? []).map(extractGloss),
+ };
+}
+
+/**
+ * Maps a parsed Entries item to {@link LexiconEntryData}. An empty or absent Entry element yields an
+ * entry with no senses, the normal state for morpheme lexemes (added to the lexicon when a parse is
+ * confirmed, often never glossed).
+ *
+ * @throws {SyntaxError} If the item has no Lexeme key element (propagated from key extraction for
+ * malformed keys).
+ */
+function extractEntry(item: ParsedEntriesItem): LexiconEntryData {
+ if (!item.Lexeme) {
+ throw new SyntaxError('Invalid XML: Entries item missing its Lexeme key element');
+ }
+ const key = extractLexemeKey(item.Lexeme);
+ const entry = item.Entry;
+ if (entry === undefined || typeof entry === 'string') return { Key: key, Senses: [] };
+ return { Key: key, Senses: (entry.Sense ?? []).map(extractSense) };
+}
+
+/**
+ * Maps a parsed Analyses item to its ordered morpheme keys. An absent or empty ArrayOfLexeme yields
+ * an empty list.
+ *
+ * @throws {SyntaxError} Propagated from key extraction for malformed lexeme keys.
+ */
+function extractAnalysisLexemes(item: ParsedAnalysesItem): LexemeKeyData[] {
+ const lexemes = item.ArrayOfLexeme;
+ if (lexemes === undefined || typeof lexemes === 'string') return [];
+ return (lexemes.Lexeme ?? []).map(extractLexemeKey);
+}
+
+/**
+ * Parses PT9 `Lexicon.xml` strings into {@link LexiconData}.
+ *
+ * Output is lossless with respect to optional data: absent attributes stay absent, senses without
+ * ids and entries without senses are preserved, and the legacy `Analyses` section is parsed
+ * alongside `Entries`. Expects the schema described in [pt9-xml.md](pt9-xml.md).
+ *
+ * Each instance holds a configured `XMLParser`; create one parser and reuse it across multiple
+ * `parse()` calls rather than constructing a new instance per file.
+ */
+export class LexiconXmlParser {
+ private readonly parser: XMLParser;
+
+ constructor() {
+ const arrayPaths = new Set([
+ 'Lexicon.Entries.item',
+ 'Lexicon.Entries.item.Entry.Sense',
+ 'Lexicon.Entries.item.Entry.Sense.Gloss',
+ 'Lexicon.Analyses.item',
+ 'Lexicon.Analyses.item.ArrayOfLexeme.Lexeme',
+ ]);
+
+ const options: Partial = {
+ ignoreAttributes: false,
+ attributeNamePrefix: '@_',
+ ignoreDeclaration: true,
+ ignorePiTags: true,
+ trimValues: false,
+ parseTagValue: false,
+ parseAttributeValue: false,
+ isArray: (_tagName, jPath) => arrayPaths.has(`${jPath}`),
+ };
+ this.parser = new XMLParser(options);
+ }
+
+ /**
+ * Parses a `Lexicon.xml` string into {@link LexiconData}.
+ *
+ * @throws {SyntaxError} If the `Lexicon` root element is absent.
+ * @throws {SyntaxError} If an `Entries` item has no `Lexeme` key element, or an `Analyses` item
+ * has no wordform key.
+ * @throws {SyntaxError} If a `Lexeme` key is missing `Type` or `Form`, or its `Homograph` is not
+ * a non-negative integer.
+ * @throws {SyntaxError} If two `Entries` items share a key (treating an absent homograph as
+ * homograph 1), or two `Analyses` items share a wordform.
+ */
+ parse(xml: string): LexiconData {
+ const parsed: ParsedLexiconXml = this.parser.parse(xml);
+ const root = parsed.Lexicon;
+ if (root === undefined) {
+ throw new SyntaxError('Invalid XML: Missing Lexicon root element');
+ }
+ if (typeof root === 'string') return { Entries: [], Analyses: {} };
+
+ const entriesContainer = root.Entries;
+ const entryItems =
+ entriesContainer === undefined || typeof entriesContainer === 'string'
+ ? []
+ : (entriesContainer.item ?? []);
+ const entries = entryItems.map(extractEntry);
+ const seenKeys = new Set();
+ entries.forEach((entry) => {
+ const id = composeLexemeKeyId(entry.Key);
+ if (seenKeys.has(id)) {
+ throw new SyntaxError(`Invalid XML: Duplicate lexicon entry key "${id}"`);
+ }
+ seenKeys.add(id);
+ });
+
+ const analysesContainer = root.Analyses;
+ const analysisItems =
+ analysesContainer === undefined || typeof analysesContainer === 'string'
+ ? []
+ : (analysesContainer.item ?? []);
+ const analyses = analysisItems.reduce>((acc, item) => {
+ const word = item.string;
+ if (!word) {
+ throw new SyntaxError('Invalid XML: Analyses item missing its wordform key');
+ }
+ if (Object.hasOwn(acc, word)) {
+ throw new SyntaxError(`Invalid XML: Duplicate analyses wordform "${word}"`);
+ }
+ acc[word] = extractAnalysisLexemes(item);
+ return acc;
+ }, {});
+
+ return {
+ ...(root.Language !== undefined && { Language: root.Language }),
+ ...(root.FontName !== undefined && { FontName: root.FontName }),
+ ...(root.FontSize !== undefined && { FontSize: root.FontSize }),
+ Entries: entries,
+ Analyses: analyses,
+ };
+ }
+}
diff --git a/src/parsers/pt9/pt9-xml.md b/src/parsers/pt9/pt9-xml.md
index 4a9ef3b3..98082846 100644
--- a/src/parsers/pt9/pt9-xml.md
+++ b/src/parsers/pt9/pt9-xml.md
@@ -1,8 +1,35 @@
# Paratext 9 XML schema
-The extension reads PT9 interlinear data from XML files (e.g. `Interlinear__.xml` in project data). The parser in `src/parsers/pt9/interlinearXmlParser.ts` expects the following structure. Sample files live in `test-data/` (e.g. `Interlinear_en_MAT.xml`).
+PT9 persists interlinear data in four project-local XML files, each read by its own parser in this
+directory. Sample files for all four live in `test-data/`.
-## Document structure
+| File | Contents | Parser |
+| ---------------------------------------------------------- | ------------------------------------------------------------- | ------------------------------ |
+| `Interlinear_{language}/Interlinear_{language}_{book}.xml` | Per-verse cluster selections for one gloss language and book | `interlinearXmlParser.ts` |
+| `Lexicon.xml` | Lexicon entries, senses, and gloss text; legacy word analyses | `lexiconXmlParser.ts` |
+| `WordAnalyses.xml` | Confirmed wordform-to-parse inventory | `wordAnalysesXmlParser.ts` |
+| `InterlinearSetup.xml` | Per-gloss-language configuration | `interlinearSetupXmlParser.ts` |
+
+## Shared conventions
+
+- **Dictionary serialization.** PT9 serializes dictionaries as repeated `item` elements, each wrapping
+ the serialized key followed by the serialized value. Verse dictionaries key with a bare
+ `` element; the lexicon's `Entries` keys with a `` element. Duplicate keys within
+ one dictionary cause a parse error (deliberately stricter than PT9's reader, which silently keeps
+ the last duplicate).
+- **Lexeme keys.** A lexeme's identity appears either as a composed id string —
+ `Type:Form[:Homograph]`, with homograph 1 omitted (e.g. `Word:voici`, `Word:a:2`) — or as a
+ `` attribute triple. Type names come from PT9's
+ append-only list, so parsers accept unknown names. `lexemeKey.ts` converts between the two
+ shapes.
+- **Absence is preserved.** Absent XML attributes and elements stay absent on parsed output — never
+ coalesced to empty strings or defaults. Parsers throw only on corrupt input: unparseable XML, a
+ missing root element, duplicate dictionary keys, and entries missing their identity (each file
+ section lists its own error conditions).
+
+## Interlinear_{language}_{book}.xml
+
+### Document structure
- **Root element:** `InterlinearData`
- **Attributes:**
@@ -36,11 +63,11 @@ The extension reads PT9 interlinear data from XML files (e.g. `Interlinear_
@@ -116,3 +143,134 @@ This example shows optional root attributes, verse `Hash`, multiple verses and c
```
+
+## Lexicon.xml
+
+### Document structure
+
+- **Root element:** `Lexicon`
+ - **Children (all optional):**
+ - **`Language`**, **`FontName`**, **`FontSize`** (element text): Informational only — PT9 overwrites all three from project settings on every load.
+ - **`Analyses`**: The legacy word-analysis store. PT9 drains it into `WordAnalyses.xml` on read, but projects untouched since PT8 still carry it.
+ - **`Entries`**: The lexicon proper.
+
+- **Analyses**
+ - **Children:** Zero or more `item` elements.
+ - **`string`** (element text): Surface wordform. Required and non-empty; a missing or empty key causes a parse error, and duplicate wordforms cause a parse error.
+ - **`ArrayOfLexeme`** (optional): `Lexeme` key elements in morpheme order; absent or empty means no lexemes.
+
+- **Entries**
+ - **Children:** Zero or more `item` elements.
+ - **`Lexeme`** (required): The entry's key as an attribute triple. A missing key element causes a parse error; duplicate keys (treating an absent `Homograph` as homograph 1) cause a parse error.
+ - **Attributes:** `Type` (required, non-empty), `Form` (required; may be empty), `Homograph` (optional; must be a non-negative integer when present, absent is preserved).
+ - **`Entry`** (optional): The entry's senses. Absent or empty means an entry with no senses (common for morphemes).
+
+- **Sense**
+ - **Attributes:** `Id` (optional): 8 chars of Base64 in PT9-written files, so `+` and `/` are legal. A sense without an id is preserved but cannot be referenced by interlinear data.
+ - **Children:** Zero or more `Gloss` elements.
+
+- **Gloss**
+ - **Attributes:** `Language` (optional): BCP 47 tag or legacy language name; absent is preserved.
+ - **Element text:** The gloss text; an empty element yields an empty string.
+
+### Parsed output (in-memory)
+
+Types exported from `src/parsers/pt9/lexiconXmlParser.ts`: **LexiconData** (`Language?`, `FontName?`, `FontSize?` as raw strings, `Entries`, `Analyses` as a record of wordform → `LexemeKeyData[]`, mirroring how string-keyed PT9 dictionaries parse elsewhere), **LexiconEntryData** (`Key` as a `LexemeKeyData`, `Senses`), **LexiconSenseData** (`Id?`, `Glosses`), **LexiconGlossData** (`Language?`, `Text`). `Entries` stays an array of key-carrying objects because its key is the non-string `LexemeKey`.
+
+### Example
+
+```xml
+
+
+ en
+ Charis SIL
+ 12
+
+ -
+ exaucera
+
+
+
+
+
+
+
+ -
+
+
+
+ is
+ voici
+
+
+
+ -
+
+
+
+
+
+```
+
+## WordAnalyses.xml
+
+### Document structure
+
+- **Root element:** `WordAnalyses`
+ - **Children:** Zero or more `Entry` elements.
+
+- **Entry**
+ - **Attributes:** `Word` (required, non-empty): Surface wordform. A missing or empty attribute causes a parse error; duplicate wordforms cause a parse error.
+ - **Children:** Zero or more `Analysis` elements — a wordform may carry more than one analysis.
+
+- **Analysis**
+ - **Children:** Zero or more `Lexeme` elements whose text is a composed lexeme-key id string (e.g. `Stem:exauc`), in morpheme order.
+
+### Parsed output (in-memory)
+
+Types exported from `src/parsers/pt9/wordAnalysesXmlParser.ts`: **WordAnalysesData** (`Entries`), **WordAnalysesEntryData** (`Word`, `Analyses`), **WordAnalysisData** (`LexemeIds` as the raw id strings).
+
+### Example
+
+```xml
+
+
+
+
+ Stem:exauc
+ Suffix:era
+
+
+
+```
+
+## InterlinearSetup.xml
+
+### Document structure
+
+- **Root element:** `InterlinearSetupList`
+ - **Children:** Zero or more `InterlinearSetup` elements, one per configured gloss language.
+
+- **InterlinearSetup** — every field optional; parsing never throws below the root.
+ - **Attributes:**
+ - `type`: Interlinear type name (e.g. `"BackTranslation"`, `"Glossing"`, `"Adaptation"`). Kept as the raw string; unknown names from future PT9 versions survive (PT9's own reader throws on them).
+ - `language`: Gloss language id; keys the `Interlinear_{language}` directory.
+ - **Children (element text):** `LanguageName`, `FontName`, `FontSize` (raw string), `RightToLeft`, `RelatedLanguages`, `ExportOnApprove`, `MdlIsResource` (booleans: `"true"` parses true, any other text false, absent stays absent), `MdlScrTextName`, `MdlScrTextId` (raw hex-id string), `ExportScrTextName`, `ExportScrTextId` (raw hex-id string).
+
+### Parsed output (in-memory)
+
+Types exported from `src/parsers/pt9/interlinearSetupXmlParser.ts`: **InterlinearSetupsData** (`Setups`), **InterlinearSetupData** (all fields optional, attribute `type` → `Type`, attribute `language` → `LanguageId`).
+
+### Example
+
+```xml
+
+
+
+ English
+ Charis SIL
+ 12
+ false
+
+
+```
diff --git a/src/parsers/pt9/wordAnalysesXmlParser.ts b/src/parsers/pt9/wordAnalysesXmlParser.ts
new file mode 100644
index 00000000..b9f1bb75
--- /dev/null
+++ b/src/parsers/pt9/wordAnalysesXmlParser.ts
@@ -0,0 +1,112 @@
+import { X2jOptions, XMLParser } from 'fast-xml-parser';
+
+/** One analysis of a wordform: its ordered morpheme lexeme ids. */
+export interface WordAnalysisData {
+ /** Composed lexeme-key id strings (e.g. `"Stem:exauc"`), in morpheme order. */
+ LexemeIds: string[];
+}
+
+/** All analyses recorded for one wordform. */
+export interface WordAnalysesEntryData {
+ /** Surface wordform the analyses apply to (XML attribute Word). */
+ Word: string;
+ /** Analyses in document order; a wordform may carry more than one. */
+ Analyses: WordAnalysisData[];
+}
+
+/** Root word-analyses data: the confirmed wordform-to-parse inventory. */
+export interface WordAnalysesData {
+ /** Entries in document order. */
+ Entries: WordAnalysesEntryData[];
+}
+
+/** Analysis: Lexeme id children; an empty element parses as a bare string. */
+type ParsedAnalysis = string | { Lexeme?: string[] };
+
+/** Entry: Word attribute plus Analysis children. */
+interface ParsedEntry {
+ ['@_Word']?: string;
+ Analysis?: ParsedAnalysis[];
+}
+
+/**
+ * Root WordAnalyses element; an empty element parses as a bare string. The string carries no data;
+ * it marks the root as present so an empty inventory (no parses confirmed yet) parses as valid
+ * rather than erroring as a missing root.
+ */
+type ParsedWordAnalysesRoot = string | { Entry?: ParsedEntry[] };
+
+/** Root document: WordAnalyses. */
+interface ParsedWordAnalysesXml {
+ WordAnalyses?: ParsedWordAnalysesRoot;
+}
+
+/** Maps a parsed Analysis to {@link WordAnalysisData}; a bare string is an analysis with no lexemes. */
+function extractAnalysis(analysis: ParsedAnalysis): WordAnalysisData {
+ if (typeof analysis === 'string') return { LexemeIds: [] };
+ return { LexemeIds: analysis.Lexeme ?? [] };
+}
+
+/**
+ * Parses PT9 `WordAnalyses.xml` strings into {@link WordAnalysesData}.
+ *
+ * Lexeme ids are kept as the raw composed strings from the file. Expects the schema described in
+ * [pt9-xml.md](pt9-xml.md).
+ *
+ * Each instance holds a configured `XMLParser`; create one parser and reuse it across multiple
+ * `parse()` calls rather than constructing a new instance per file.
+ */
+export class WordAnalysesXmlParser {
+ private readonly parser: XMLParser;
+
+ constructor() {
+ const arrayPaths = new Set([
+ 'WordAnalyses.Entry',
+ 'WordAnalyses.Entry.Analysis',
+ 'WordAnalyses.Entry.Analysis.Lexeme',
+ ]);
+
+ const options: Partial = {
+ ignoreAttributes: false,
+ attributeNamePrefix: '@_',
+ ignoreDeclaration: true,
+ ignorePiTags: true,
+ trimValues: false,
+ parseTagValue: false,
+ parseAttributeValue: false,
+ isArray: (_tagName, jPath) => arrayPaths.has(`${jPath}`),
+ };
+ this.parser = new XMLParser(options);
+ }
+
+ /**
+ * Parses a `WordAnalyses.xml` string into {@link WordAnalysesData}.
+ *
+ * @throws {SyntaxError} If the `WordAnalyses` root element is absent.
+ * @throws {SyntaxError} If an `Entry` is missing its `Word` attribute or the attribute is empty.
+ * @throws {SyntaxError} If two entries share a wordform.
+ */
+ parse(xml: string): WordAnalysesData {
+ const parsed: ParsedWordAnalysesXml = this.parser.parse(xml);
+ const root = parsed.WordAnalyses;
+ if (root === undefined) {
+ throw new SyntaxError('Invalid XML: Missing WordAnalyses root element');
+ }
+ if (typeof root === 'string') return { Entries: [] };
+
+ const seen = new Set();
+ const entries = (root.Entry ?? []).map((entry) => {
+ const word = entry['@_Word'];
+ if (!word) {
+ throw new SyntaxError('Invalid XML: Entry missing its Word attribute');
+ }
+ if (seen.has(word)) {
+ throw new SyntaxError(`Invalid XML: Duplicate word analyses entry "${word}"`);
+ }
+ seen.add(word);
+ return { Word: word, Analyses: (entry.Analysis ?? []).map(extractAnalysis) };
+ });
+
+ return { Entries: entries };
+ }
+}
diff --git a/test-data/InterlinearSetup.xml b/test-data/InterlinearSetup.xml
new file mode 100644
index 00000000..68075aa9
--- /dev/null
+++ b/test-data/InterlinearSetup.xml
@@ -0,0 +1,20 @@
+
+
+
+ English
+ Charis SIL
+ 12
+ false
+ false
+ false
+
+
+ French
+ MDL
+ 1234567890abcdef
+ true
+ true
+ BT1
+ fedcba0987654321
+
+
diff --git a/test-data/Lexicon.xml b/test-data/Lexicon.xml
new file mode 100644
index 00000000..4928109c
--- /dev/null
+++ b/test-data/Lexicon.xml
@@ -0,0 +1,72 @@
+
+
+ en
+ Charis SIL
+ 12
+
+ -
+ aaaa
+
+
+
+
+
+
+ -
+
+
+
+ greeting
+ salut
+
+
+
+ -
+
+
+
+ greet
+
+
+
+ -
+
+
+
+ PROG
+
+
+
+ -
+
+
+
+ one
+
+
+ per
+
+
+
+ -
+
+
+
+ ah
+
+
+
+ -
+
+
+
+ -
+
+
+
+ okay
+
+
+
+
+
diff --git a/test-data/WordAnalyses.xml b/test-data/WordAnalyses.xml
new file mode 100644
index 00000000..f49c9c24
--- /dev/null
+++ b/test-data/WordAnalyses.xml
@@ -0,0 +1,18 @@
+
+
+
+
+ Stem:hello
+ Suffix:ing
+
+
+
+
+ Stem:ab
+ Suffix:e
+
+
+ Stem:abe
+
+
+