From 7acd211ba6b3eddc810e0945b1201f04e2a3b394 Mon Sep 17 00:00:00 2001 From: Jason Naylor Date: Thu, 13 Aug 2026 11:28:06 -0700 Subject: [PATCH 1/2] Make the PT9 interlinear parser output contracts lossless Preserve what PT9 actually persists so approval state and legacy attributes survive a parse: - VerseData.Hash is optional: PT9 writes it only on verse approval, so absence is the not-approved state (previously coalesced to ''). - InterlinearData.ScrTextName is optional: legacy attribute, absent when the file lacks it. - LexemeData.SenseId is optional: absent when a Lexeme has no GlossId; an empty attribute value stays ''. - PunctuationData.TextRange is optional: Punctuation entries without a valid Range are kept rangeless instead of dropped. Update pt9-xml.md to match and point it at the parser module for the output types instead of interlinearizer.d.ts. Co-Authored-By: Claude Fable 5 --- .../parsers/pt9/interlinearXmlParser.test.ts | 109 ++++++++++++++---- src/parsers/pt9/interlinearXmlParser.ts | 81 +++++++------ src/parsers/pt9/pt9-xml.md | 18 +-- 3 files changed, 141 insertions(+), 67 deletions(-) diff --git a/src/__tests__/parsers/pt9/interlinearXmlParser.test.ts b/src/__tests__/parsers/pt9/interlinearXmlParser.test.ts index f2126ec7..1f4c6738 100644 --- a/src/__tests__/parsers/pt9/interlinearXmlParser.test.ts +++ b/src/__tests__/parsers/pt9/interlinearXmlParser.test.ts @@ -31,13 +31,11 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result).toEqual({ - ScrTextName: '', + expect(result).toStrictEqual({ GlossLanguage: 'en', BookId: 'MAT', Verses: { 'MAT 1:1': { - Hash: '', Clusters: [ { TextRange: { Index: 0, Length: 4 }, @@ -97,6 +95,40 @@ describe('InterlinearXmlParser', () => { expect(result.Verses['RUT 3:1'].Hash).toBe('123456'); }); + it('preserves the absent-vs-empty distinction for ScrTextName and Hash', () => { + const xml = ` + + + + RUT 3:1 + + + + + + + + + RUT 3:2 + + + + + + + + + + `; + const result = parser.parse(xml); + + // Empty attribute values are preserved as empty strings, not dropped. + expect(result.ScrTextName).toBe(''); + expect(result.Verses['RUT 3:1'].Hash).toBe(''); + // An absent Hash attribute is PT9's not-approved state and must stay absent. + expect(result.Verses['RUT 3:2']).not.toHaveProperty('Hash'); + }); + it('parses cluster with multiple lexemes and builds LexemesId and Id correctly', () => { const xml = ` @@ -200,7 +232,7 @@ describe('InterlinearXmlParser', () => { expect(cluster.Id).toBe('10-3'); }); - it('parses Lexeme without GlossId as empty SenseId', () => { + it('parses Lexeme without GlossId as absent SenseId', () => { const xml = ` @@ -218,7 +250,30 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toEqual({ + expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toStrictEqual({ + LexemeId: 'Word:a', + }); + }); + + it('preserves an empty GlossId attribute as an empty SenseId', () => { + const xml = ` + + + + MAT 1:1 + + + + + + + + + + `; + const result = parser.parse(xml); + + expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toStrictEqual({ LexemeId: 'Word:a', SenseId: '', }); @@ -284,7 +339,7 @@ describe('InterlinearXmlParser', () => { ]); }); - it('omits Punctuation entries without valid Range', () => { + it('preserves Punctuation entries without a Range element, with no TextRange', () => { const xml = ` @@ -311,15 +366,20 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result.Verses['MAT 1:1'].Punctuations).toHaveLength(1); - expect(result.Verses['MAT 1:1'].Punctuations[0]).toEqual({ - TextRange: { Index: 1, Length: 2 }, - BeforeText: 'c', - AfterText: 'd', - }); + expect(result.Verses['MAT 1:1'].Punctuations).toStrictEqual([ + { + BeforeText: 'a', + AfterText: 'b', + }, + { + TextRange: { Index: 1, Length: 2 }, + BeforeText: 'c', + AfterText: 'd', + }, + ]); }); - it('omits Punctuation entries when Range Index or Length is not finite (missing or non-numeric)', () => { + it('preserves Punctuation entries whose Range has missing or non-numeric attributes, with no TextRange', () => { const xml = ` @@ -357,12 +417,13 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result.Verses['MAT 1:1'].Punctuations).toHaveLength(1); - expect(result.Verses['MAT 1:1'].Punctuations[0]).toEqual({ - TextRange: { Index: 5, Length: 1 }, - BeforeText: 'valid', - AfterText: '', - }); + expect(result.Verses['MAT 1:1'].Punctuations).toStrictEqual([ + { TextRange: { Index: 5, Length: 1 }, BeforeText: 'valid', AfterText: '' }, + { BeforeText: 'no Index', AfterText: '' }, + { BeforeText: 'no Length', AfterText: '' }, + { BeforeText: 'non-numeric Index', AfterText: '' }, + { BeforeText: 'non-numeric Length', AfterText: '' }, + ]); }); it('parses Punctuation with valid Range but missing BeforeText/AfterText as empty strings', () => { @@ -426,7 +487,7 @@ describe('InterlinearXmlParser', () => { expect(result.Verses['MAT 1:2'].Clusters[0].Lexemes[0].LexemeId).toBe('b'); }); - it('parses item with missing VerseData as empty Hash, Clusters, Punctuations', () => { + it('parses item with missing VerseData as no Hash and empty Clusters, Punctuations', () => { const xml = ` @@ -438,8 +499,7 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result.Verses['MAT 1:1']).toEqual({ - Hash: '', + expect(result.Verses['MAT 1:1']).toStrictEqual({ Clusters: [], Punctuations: [], }); @@ -458,8 +518,7 @@ describe('InterlinearXmlParser', () => { `; const result = parser.parse(xml); - expect(result.Verses['MAT 1:11']).toEqual({ - Hash: '', + expect(result.Verses['MAT 1:11']).toStrictEqual({ Clusters: [], Punctuations: [], }); @@ -594,7 +653,7 @@ describe('InterlinearXmlParser', () => { expect(result.GlossLanguage).toBe('en'); expect(result.BookId).toBe('MAT'); - expect(result.ScrTextName).toBe(''); + expect(result.ScrTextName).toBeUndefined(); expect(Object.keys(result.Verses).length).toBeGreaterThan(0); const mat11 = result.Verses['MAT 1:1']; diff --git a/src/parsers/pt9/interlinearXmlParser.ts b/src/parsers/pt9/interlinearXmlParser.ts index 8dbca4d4..ab9a8b2b 100644 --- a/src/parsers/pt9/interlinearXmlParser.ts +++ b/src/parsers/pt9/interlinearXmlParser.ts @@ -12,8 +12,12 @@ export interface StringRange { export interface LexemeData { /** ID of the lexeme (e.g. from Lexicon; XML attribute Id). */ LexemeId: string; - /** ID of the sense/gloss used for this lexeme (XML attribute GlossId). */ - SenseId: string; + /** + * ID of the selected sense (XML attribute GlossId, a sense id despite the historical name). + * Absent when the lexeme carries no sense selection; an empty attribute value is preserved as an + * empty string. + */ + SenseId?: string; } /** Data on the interlinearization of a cluster. */ @@ -32,8 +36,11 @@ export interface ClusterData { /** Data on punctuation change. */ export interface PunctuationData { - /** Character range this punctuation occupies in the verse text. */ - TextRange: StringRange; + /** + * Character range this punctuation occupies in the verse text. Absent when the entry has no + * `Range` element with valid numeric attributes; the entry itself is still preserved. + */ + TextRange?: StringRange; /** Punctuation text before the change (or empty). */ BeforeText: string; /** Punctuation text after the change (or empty). */ @@ -42,8 +49,11 @@ export interface PunctuationData { /** Interlinear data for a single verse. */ export interface VerseData { - /** Hash of verse text when approved; empty string if not approved. */ - Hash: string; + /** + * Approval hash of the verse text. PT9 writes the attribute only when the verse is approved, so + * absence is the not-approved state; it is never coalesced to an empty string. + */ + Hash?: string; /** Lexeme clusters in this verse. */ Clusters: ClusterData[]; /** Punctuation changes in this verse. */ @@ -52,8 +62,11 @@ export interface VerseData { /** Root interlinear data: book + verses. */ export interface InterlinearData { - /** Source text / project name (e.g. from InterlinearData ScrTextName attribute). */ - ScrTextName: string; + /** + * Source text / project name (legacy ScrTextName attribute, no longer written by modern PT9). + * Absent when the file does not carry the attribute. + */ + ScrTextName?: string; /** Language code or name for the glosses. */ GlossLanguage: string; /** Book id (e.g. "RUT", "MAT"). */ @@ -138,8 +151,8 @@ interface ParsedInterlinearXml { } /** - * Maps a parsed Cluster's Lexeme children to {@link LexemeData}. A lexeme with no gloss id yields an - * empty sense id rather than being dropped. + * Maps a parsed Cluster's Lexeme children to {@link LexemeData}. A lexeme with no GlossId yields an + * absent SenseId; an empty attribute value is preserved as an empty string. * * @throws {SyntaxError} If any Lexeme element is missing the required Id attribute. */ @@ -151,7 +164,8 @@ function extractLexemesFromCluster(clusterElement: ParsedCluster): LexemeData[] if (!lexemeId) { throw new SyntaxError('Invalid XML: Lexeme missing required Id attribute'); } - return { LexemeId: lexemeId, SenseId: el['@_GlossId'] ?? '' }; + const senseId = el['@_GlossId']; + return { LexemeId: lexemeId, ...(senseId !== undefined && { SenseId: senseId }) }; }); } @@ -169,27 +183,24 @@ const parseStrictNumber = (raw: string | undefined): number | undefined => { /** * Maps a parsed VerseData's Punctuation array to {@link PunctuationData} array. * - * Uses a **lenient** parsing strategy: entries without a valid `Range` (or with missing / - * non-integer `Index`/`Length`) are silently filtered out rather than throwing. Punctuation data is - * non-critical to the interlinear display; clusters are validated strictly in + * Every entry is preserved. `TextRange` is set only when the entry has a `Range` element with valid + * (non-negative integer) `Index`/`Length`; otherwise the entry keeps its texts with no range. PT9 + * itself reads a missing `Range` as a `(0, 0)` default; absence is preserved here instead so no + * fabricated range enters the data. Clusters are validated strictly in * {@link extractClustersFromVerse} because they are required for alignment rendering. */ function extractPunctuationsFromVerse(verseDataElement: ParsedVerseData): PunctuationData[] { const elements = verseDataElement.Punctuation ?? []; - return elements.flatMap((el) => { - const rangeElement = el.Range; - if (!rangeElement) return []; - const index = parseStrictNumber(rangeElement['@_Index']); - const length = parseStrictNumber(rangeElement['@_Length']); - if (index === undefined || length === undefined) return []; - return [ - { - TextRange: { Index: index, Length: length }, - BeforeText: el.BeforeText ?? '', - AfterText: el.AfterText ?? '', - }, - ]; + return elements.map((el) => { + const index = parseStrictNumber(el.Range?.['@_Index']); + const length = parseStrictNumber(el.Range?.['@_Length']); + return { + ...(index !== undefined && + length !== undefined && { TextRange: { Index: index, Length: length } }), + BeforeText: el.BeforeText ?? '', + AfterText: el.AfterText ?? '', + }; }); } @@ -245,8 +256,11 @@ function extractClustersFromVerse(verseDataElement: ParsedVerseData): ClusterDat * Parses interlinear XML strings into {@link InterlinearData} using fast-xml-parser. * * Input is a raw XML string (caller is responsible for obtaining it, e.g. from file or network). - * Output is the {@link InterlinearData} shape declared in this module; no extra conversion is done. - * Expects the interlinear XML schema described in [pt9-xml.md](pt9-xml.md). + * Output is the {@link InterlinearData} shape declared in this module, lossless with respect to + * optional data: absent attributes (`ScrTextName`, `Hash`, `GlossId`) stay absent rather than being + * coalesced to empty strings (`Hash` absence is PT9's not-approved state), and Punctuation entries + * without a valid `Range` are kept rangeless rather than dropped. Expects the interlinear XML + * schema described in [pt9-xml.md](pt9-xml.md). * * Each instance holds a configured `XMLParser`; create one parser and reuse it across multiple * `parse()` calls rather than constructing a new instance per file. @@ -295,7 +309,7 @@ export class InterlinearXmlParser { throw new SyntaxError('Invalid XML: Missing InterlinearData root element'); } - const scrTextName = root['@_ScrTextName'] ?? ''; + const scrTextName = root['@_ScrTextName']; const glossLanguage = root['@_GlossLanguage'] ?? ''; const bookId = root['@_BookId'] ?? ''; if (!glossLanguage || !bookId) { @@ -323,12 +337,13 @@ export class InterlinearXmlParser { const verseDataElement = item.VerseData; if (!verseDataElement) { - acc[verseKey] = { Hash: '', Clusters: [], Punctuations: [] }; + acc[verseKey] = { Clusters: [], Punctuations: [] }; return acc; } + const hash = verseDataElement['@_Hash']; acc[verseKey] = { - Hash: verseDataElement['@_Hash'] ?? '', + ...(hash !== undefined && { Hash: hash }), Clusters: extractClustersFromVerse(verseDataElement), Punctuations: extractPunctuationsFromVerse(verseDataElement), }; @@ -336,7 +351,7 @@ export class InterlinearXmlParser { }, {}); return { - ScrTextName: scrTextName, + ...(scrTextName !== undefined && { ScrTextName: scrTextName }), GlossLanguage: glossLanguage, BookId: bookId, Verses: verses, diff --git a/src/parsers/pt9/pt9-xml.md b/src/parsers/pt9/pt9-xml.md index b570279e..4a9ef3b3 100644 --- a/src/parsers/pt9/pt9-xml.md +++ b/src/parsers/pt9/pt9-xml.md @@ -15,11 +15,11 @@ The extension reads PT9 interlinear data from XML files (e.g. `Interlinear_ Date: Wed, 19 Aug 2026 10:53:43 -0700 Subject: [PATCH 2/2] Say non-negative integer, not numeric, for Punctuation Range in pt9-xml.md Matches parseStrictNumber's actual guard and the PunctuationData docstring: negative or fractional Index/Length values yield no TextRange. Co-Authored-By: Claude Fable 5 --- src/parsers/pt9/pt9-xml.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/parsers/pt9/pt9-xml.md b/src/parsers/pt9/pt9-xml.md index 4a9ef3b3..d7896ebf 100644 --- a/src/parsers/pt9/pt9-xml.md +++ b/src/parsers/pt9/pt9-xml.md @@ -36,7 +36,7 @@ The extension reads PT9 interlinear data from XML files (e.g. `Interlinear_