Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
109 changes: 84 additions & 25 deletions src/__tests__/parsers/pt9/interlinearXmlParser.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -31,13 +31,11 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result).toEqual({
ScrTextName: '',
expect(result).toStrictEqual({
GlossLanguage: 'en',
BookId: 'MAT',
Verses: {
'MAT 1:1': {
Hash: '',
Clusters: [
{
TextRange: { Index: 0, Length: 4 },
Expand Down Expand Up @@ -97,6 +95,40 @@ describe('InterlinearXmlParser', () => {
expect(result.Verses['RUT 3:1'].Hash).toBe('123456');
});

it('preserves the absent-vs-empty distinction for ScrTextName and Hash', () => {
const xml = `
<InterlinearData ScrTextName="" GlossLanguage="en" BookId="RUT">
<Verses>
<item>
<string>RUT 3:1</string>
<VerseData Hash="">
<Cluster>
<Range Index="1" Length="2" />
<Lexeme Id="x" />
</Cluster>
</VerseData>
</item>
<item>
<string>RUT 3:2</string>
<VerseData>
<Cluster>
<Range Index="1" Length="2" />
<Lexeme Id="x" />
</Cluster>
</VerseData>
</item>
</Verses>
</InterlinearData>
`;
const result = parser.parse(xml);

// Empty attribute values are preserved as empty strings, not dropped.
expect(result.ScrTextName).toBe('');
expect(result.Verses['RUT 3:1'].Hash).toBe('');
// An absent Hash attribute is PT9's not-approved state and must stay absent.
expect(result.Verses['RUT 3:2']).not.toHaveProperty('Hash');
});

it('parses cluster with multiple lexemes and builds LexemesId and Id correctly', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
Expand Down Expand Up @@ -200,7 +232,7 @@ describe('InterlinearXmlParser', () => {
expect(cluster.Id).toBe('10-3');
});

it('parses Lexeme without GlossId as empty SenseId', () => {
it('parses Lexeme without GlossId as absent SenseId', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
<Verses>
Expand All @@ -218,7 +250,30 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toEqual({
expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toStrictEqual({
LexemeId: 'Word:a',
});
});

it('preserves an empty GlossId attribute as an empty SenseId', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
<Verses>
<item>
<string>MAT 1:1</string>
<VerseData>
<Cluster>
<Range Index="0" Length="1" />
<Lexeme Id="Word:a" GlossId="" />
</Cluster>
</VerseData>
</item>
</Verses>
</InterlinearData>
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:1'].Clusters[0].Lexemes[0]).toStrictEqual({
LexemeId: 'Word:a',
SenseId: '',
});
Expand Down Expand Up @@ -284,7 +339,7 @@ describe('InterlinearXmlParser', () => {
]);
});

it('omits Punctuation entries without valid Range', () => {
it('preserves Punctuation entries without a Range element, with no TextRange', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
<Verses>
Expand All @@ -311,15 +366,20 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:1'].Punctuations).toHaveLength(1);
expect(result.Verses['MAT 1:1'].Punctuations[0]).toEqual({
TextRange: { Index: 1, Length: 2 },
BeforeText: 'c',
AfterText: 'd',
});
expect(result.Verses['MAT 1:1'].Punctuations).toStrictEqual([
{
BeforeText: 'a',
AfterText: 'b',
},
{
TextRange: { Index: 1, Length: 2 },
BeforeText: 'c',
AfterText: 'd',
},
]);
});

it('omits Punctuation entries when Range Index or Length is not finite (missing or non-numeric)', () => {
it('preserves Punctuation entries whose Range has missing or non-numeric attributes, with no TextRange', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
<Verses>
Expand Down Expand Up @@ -357,12 +417,13 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:1'].Punctuations).toHaveLength(1);
expect(result.Verses['MAT 1:1'].Punctuations[0]).toEqual({
TextRange: { Index: 5, Length: 1 },
BeforeText: 'valid',
AfterText: '',
});
expect(result.Verses['MAT 1:1'].Punctuations).toStrictEqual([
{ TextRange: { Index: 5, Length: 1 }, BeforeText: 'valid', AfterText: '' },
{ BeforeText: 'no Index', AfterText: '' },
{ BeforeText: 'no Length', AfterText: '' },
{ BeforeText: 'non-numeric Index', AfterText: '' },
{ BeforeText: 'non-numeric Length', AfterText: '' },
]);
});

it('parses Punctuation with valid Range but missing BeforeText/AfterText as empty strings', () => {
Expand Down Expand Up @@ -426,7 +487,7 @@ describe('InterlinearXmlParser', () => {
expect(result.Verses['MAT 1:2'].Clusters[0].Lexemes[0].LexemeId).toBe('b');
});

it('parses item with missing VerseData as empty Hash, Clusters, Punctuations', () => {
it('parses item with missing VerseData as no Hash and empty Clusters, Punctuations', () => {
const xml = `
<InterlinearData GlossLanguage="en" BookId="MAT">
<Verses>
Expand All @@ -438,8 +499,7 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:1']).toEqual({
Hash: '',
expect(result.Verses['MAT 1:1']).toStrictEqual({
Clusters: [],
Punctuations: [],
});
Expand All @@ -458,8 +518,7 @@ describe('InterlinearXmlParser', () => {
`;
const result = parser.parse(xml);

expect(result.Verses['MAT 1:11']).toEqual({
Hash: '',
expect(result.Verses['MAT 1:11']).toStrictEqual({
Clusters: [],
Punctuations: [],
});
Expand Down Expand Up @@ -594,7 +653,7 @@ describe('InterlinearXmlParser', () => {

expect(result.GlossLanguage).toBe('en');
expect(result.BookId).toBe('MAT');
expect(result.ScrTextName).toBe('');
expect(result.ScrTextName).toBeUndefined();
expect(Object.keys(result.Verses).length).toBeGreaterThan(0);

const mat11 = result.Verses['MAT 1:1'];
Expand Down
81 changes: 48 additions & 33 deletions src/parsers/pt9/interlinearXmlParser.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,8 +12,12 @@ export interface StringRange {
export interface LexemeData {
/** ID of the lexeme (e.g. from Lexicon; XML attribute Id). */
LexemeId: string;
/** ID of the sense/gloss used for this lexeme (XML attribute GlossId). */
SenseId: string;
/**
* ID of the selected sense (XML attribute GlossId, a sense id despite the historical name).
* Absent when the lexeme carries no sense selection; an empty attribute value is preserved as an
* empty string.
*/
SenseId?: string;
}

/** Data on the interlinearization of a cluster. */
Expand All @@ -32,8 +36,11 @@ export interface ClusterData {

/** Data on punctuation change. */
export interface PunctuationData {
/** Character range this punctuation occupies in the verse text. */
TextRange: StringRange;
/**
* Character range this punctuation occupies in the verse text. Absent when the entry has no
* `Range` element with valid numeric attributes; the entry itself is still preserved.
*/
TextRange?: StringRange;
/** Punctuation text before the change (or empty). */
BeforeText: string;
/** Punctuation text after the change (or empty). */
Expand All @@ -42,8 +49,11 @@ export interface PunctuationData {

/** Interlinear data for a single verse. */
export interface VerseData {
/** Hash of verse text when approved; empty string if not approved. */
Hash: string;
/**
* Approval hash of the verse text. PT9 writes the attribute only when the verse is approved, so
* absence is the not-approved state; it is never coalesced to an empty string.
*/
Hash?: string;
/** Lexeme clusters in this verse. */
Clusters: ClusterData[];
/** Punctuation changes in this verse. */
Expand All @@ -52,8 +62,11 @@ export interface VerseData {

/** Root interlinear data: book + verses. */
export interface InterlinearData {
/** Source text / project name (e.g. from InterlinearData ScrTextName attribute). */
ScrTextName: string;
/**
* Source text / project name (legacy ScrTextName attribute, no longer written by modern PT9).
* Absent when the file does not carry the attribute.
*/
ScrTextName?: string;
/** Language code or name for the glosses. */
GlossLanguage: string;
/** Book id (e.g. "RUT", "MAT"). */
Expand Down Expand Up @@ -138,8 +151,8 @@ interface ParsedInterlinearXml {
}

/**
* Maps a parsed Cluster's Lexeme children to {@link LexemeData}. A lexeme with no gloss id yields an
* empty sense id rather than being dropped.
* Maps a parsed Cluster's Lexeme children to {@link LexemeData}. A lexeme with no GlossId yields an
* absent SenseId; an empty attribute value is preserved as an empty string.
*
* @throws {SyntaxError} If any Lexeme element is missing the required Id attribute.
*/
Expand All @@ -151,7 +164,8 @@ function extractLexemesFromCluster(clusterElement: ParsedCluster): LexemeData[]
if (!lexemeId) {
throw new SyntaxError('Invalid XML: Lexeme missing required Id attribute');
}
return { LexemeId: lexemeId, SenseId: el['@_GlossId'] ?? '' };
const senseId = el['@_GlossId'];
return { LexemeId: lexemeId, ...(senseId !== undefined && { SenseId: senseId }) };
});
}

Expand All @@ -169,27 +183,24 @@ const parseStrictNumber = (raw: string | undefined): number | undefined => {
/**
* Maps a parsed VerseData's Punctuation array to {@link PunctuationData} array.
*
* Uses a **lenient** parsing strategy: entries without a valid `Range` (or with missing /
* non-integer `Index`/`Length`) are silently filtered out rather than throwing. Punctuation data is
* non-critical to the interlinear display; clusters are validated strictly in
* Every entry is preserved. `TextRange` is set only when the entry has a `Range` element with valid
* (non-negative integer) `Index`/`Length`; otherwise the entry keeps its texts with no range. PT9
* itself reads a missing `Range` as a `(0, 0)` default; absence is preserved here instead so no
* fabricated range enters the data. Clusters are validated strictly in
* {@link extractClustersFromVerse} because they are required for alignment rendering.
*/
function extractPunctuationsFromVerse(verseDataElement: ParsedVerseData): PunctuationData[] {
const elements = verseDataElement.Punctuation ?? [];

return elements.flatMap((el) => {
const rangeElement = el.Range;
if (!rangeElement) return [];
const index = parseStrictNumber(rangeElement['@_Index']);
const length = parseStrictNumber(rangeElement['@_Length']);
if (index === undefined || length === undefined) return [];
return [
{
TextRange: { Index: index, Length: length },
BeforeText: el.BeforeText ?? '',
AfterText: el.AfterText ?? '',
},
];
return elements.map((el) => {
const index = parseStrictNumber(el.Range?.['@_Index']);
const length = parseStrictNumber(el.Range?.['@_Length']);
return {
...(index !== undefined &&
length !== undefined && { TextRange: { Index: index, Length: length } }),
BeforeText: el.BeforeText ?? '',
AfterText: el.AfterText ?? '',
};
});
}

Expand Down Expand Up @@ -245,8 +256,11 @@ function extractClustersFromVerse(verseDataElement: ParsedVerseData): ClusterDat
* Parses interlinear XML strings into {@link InterlinearData} using fast-xml-parser.
*
* Input is a raw XML string (caller is responsible for obtaining it, e.g. from file or network).
* Output is the {@link InterlinearData} shape declared in this module; no extra conversion is done.
* Expects the interlinear XML schema described in [pt9-xml.md](pt9-xml.md).
* Output is the {@link InterlinearData} shape declared in this module, lossless with respect to
* optional data: absent attributes (`ScrTextName`, `Hash`, `GlossId`) stay absent rather than being
* coalesced to empty strings (`Hash` absence is PT9's not-approved state), and Punctuation entries
* without a valid `Range` are kept rangeless rather than dropped. Expects the interlinear XML
* schema described in [pt9-xml.md](pt9-xml.md).
*
* Each instance holds a configured `XMLParser`; create one parser and reuse it across multiple
* `parse()` calls rather than constructing a new instance per file.
Expand Down Expand Up @@ -295,7 +309,7 @@ export class InterlinearXmlParser {
throw new SyntaxError('Invalid XML: Missing InterlinearData root element');
}

const scrTextName = root['@_ScrTextName'] ?? '';
const scrTextName = root['@_ScrTextName'];
const glossLanguage = root['@_GlossLanguage'] ?? '';
const bookId = root['@_BookId'] ?? '';
if (!glossLanguage || !bookId) {
Expand Down Expand Up @@ -323,20 +337,21 @@ export class InterlinearXmlParser {

const verseDataElement = item.VerseData;
if (!verseDataElement) {
acc[verseKey] = { Hash: '', Clusters: [], Punctuations: [] };
acc[verseKey] = { Clusters: [], Punctuations: [] };
return acc;
}

const hash = verseDataElement['@_Hash'];
acc[verseKey] = {
Hash: verseDataElement['@_Hash'] ?? '',
...(hash !== undefined && { Hash: hash }),
Clusters: extractClustersFromVerse(verseDataElement),
Punctuations: extractPunctuationsFromVerse(verseDataElement),
};
return acc;
}, {});

return {
ScrTextName: scrTextName,
...(scrTextName !== undefined && { ScrTextName: scrTextName }),
GlossLanguage: glossLanguage,
BookId: bookId,
Verses: verses,
Expand Down
Loading
Loading