diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 00000000..7e1a9705 --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,26 @@ +# Agent instructions for lameta + +## Issue tracker: none + +lameta uses **no card tracker**. There is no YouTrack project, and GitHub issues are not +used as work cards for branches. + +- Ticket ids: none. Branch names carry no ticket id. +- Tracker skill: none. + +Skills that want a tracker (`preflight`, `pr-ready-for-human`, `add-test-ideas`) must skip +every card step, note the skip once, and continue. Do not ask again, and do not search for a +tracker. A file name such as `playwright-screenshot-LAM-25.png` is a stray artifact, not +evidence of a tracker. + +## Toolchain + +Yarn, one TypeScript stack, no .NET. + +- Typecheck: `yarn tsc --noEmit` +- Lint: `yarn eslint src` +- Unit tests: `yarn vitest run` +- Build: `yarn build` +- End-to-end tests: `yarn build` first, then `yarn e2e`. The suite drives `dist/`, and + `playwright.config.ts` runs no build of its own, so a source edit without a build tests the + old bundle. diff --git a/CLAUDE.md b/CLAUDE.md new file mode 100644 index 00000000..43c994c2 --- /dev/null +++ b/CLAUDE.md @@ -0,0 +1 @@ +@AGENTS.md diff --git a/PAPERCUTS.md b/PAPERCUTS.md new file mode 100644 index 00000000..9b167ab0 --- /dev/null +++ b/PAPERCUTS.md @@ -0,0 +1,20 @@ +# Papercuts + +Short notes about friction and traps in this repo. Append; do not rewrite. + +## An e2e test that reads text cannot see a clipped dropdown (2026-08-26) + +`e2e/createLanguage.e2e.ts` asserted on the text of the react-select menu. The menu was in +the document, so the tests passed, but on the Session and Person forms an ancestor with +`overflow-y: hidden` cut off all 300 pixels of it, so the user saw nothing. Playwright's +`:visible` does not help: it tests the bounding box, not clipping by an ancestor. + +For a dropdown, a tooltip, or any element that leaves its own box, assert the geometry as +well as the text. `expectNothingClipsTheMenu` in that file walks up from the menu and fails +if an ancestor with overflow hidden ends above the menu's bottom. + +## The e2e suite runs `dist/`, not the source (2026-08-26) + +`playwright.config.ts` has no build step, so `npx playwright test` drives whatever +`yarn build` last wrote. A source edit has no effect on an e2e run until you build. Cost: two +test runs that measured the old bundle and reported the defect as fixed. diff --git a/archive-configurations/extractFromJSON.ts b/archive-configurations/extractFromJSON.ts index 3c5d55bb..90457b25 100644 --- a/archive-configurations/extractFromJSON.ts +++ b/archive-configurations/extractFromJSON.ts @@ -94,9 +94,16 @@ function createPoFilesFromOneSetOfJsons(set: string) { console.log( `extractFromJSON: Gathered ${messages.length} strings from ${set}.` ); + // A PO string ends at the first unescaped quote, so a description holding a quote, e.g. + // `New speakers of Edolo`, truncates the entry and leaves stray tokens after it. Escape the + // backslash first, then the quote, or the escape gets escaped a second time. + const escapeForPo = (text: string) => + text.replace(/\\/g, "\\\\").replace(/"/g, '\\"'); const content = messages .map((m) => { - return `msgctxt "${m.context}"\nmsgid "${m.message}"\nmsgstr ""\n`; + return `msgctxt "${escapeForPo(m.context)}"\nmsgid "${escapeForPo( + m.message + )}"\nmsgstr ""\n`; }) .join("\n"); //console.log(content); diff --git a/archive-configurations/poEscaping.spec.ts b/archive-configurations/poEscaping.spec.ts new file mode 100644 index 00000000..0bb5b141 --- /dev/null +++ b/archive-configurations/poEscaping.spec.ts @@ -0,0 +1,32 @@ +import { describe, it, expect } from "vitest"; +import * as fs from "fs"; +import * as path from "path"; + +// extractFromJSON.ts writes these catalogs from the json5 configuration files. A PO string +// ends at the first unescaped quote, so a field description holding a quote used to truncate +// the entry and leave stray tokens after it. See escapeForPo in extractFromJSON.ts. +const generatedCatalogs = ["fields.po", "vocabularies.po"]; + +// A well-formed PO string: an opening quote, then any run of characters in which every quote +// and every backslash is escaped, then the closing quote. +const wellFormedPoString = /^(msgid|msgctxt|msgstr) "(?:[^"\\]|\\.)*"$/; + +describe("the generated PO catalogs", () => { + generatedCatalogs.forEach((catalog) => { + it(`${catalog} escapes every quote`, () => { + const text = fs.readFileSync( + path.join("locale", "en", catalog), + "utf-8" + ); + const badLines = text + .split(/\r?\n/) + .map((line, index) => ({ line: line.trim(), number: index + 1 })) + .filter( + (l) => + /^(msgid|msgctxt|msgstr) "/.test(l.line) && + !wellFormedPoString.test(l.line) + ); + expect(badLines).toEqual([]); + }); + }); +}); diff --git a/e2e/createLanguage.e2e.ts b/e2e/createLanguage.e2e.ts index 1306b312..82db2fce 100644 --- a/e2e/createLanguage.e2e.ts +++ b/e2e/createLanguage.e2e.ts @@ -1,4 +1,4 @@ -import { test, expect, Page } from "@playwright/test"; +import { test, expect, Page, Locator } from "@playwright/test"; import { LametaE2ERunner } from "./lametaE2ERunner"; import { launchWithProject, E2eProject } from "./various-e2e-helpers"; @@ -6,7 +6,48 @@ let lameta: LametaE2ERunner; let page: Page; let project: E2eProject; +// Each language that ISO 639-3 does not list gets its own code from the qaa..qtz range, +// so an archive gets a distinct 3-letter code for each one. These tests run in order and +// share one project, so the first language gets qaa, the second qab, and so on. + +async function getSubjectLanguagesField(): Promise { + await project.goToProjectLanguages(); + const container = page + .locator('div.field:has(label:has-text("Subject Languages")):visible') + .first(); + await container.waitFor({ state: "visible", timeout: 10000 }); + return container; +} + +// The pill shows the name, and beside it the full tag in the .isoCode span. +// The session and person forms hold native setName(e.target.value)} + onKeyDown={(e) => { + if (e.key === "Enter") accept(); + }} + /> + + setCode(e.target.value)} + onKeyDown={(e) => { + if (e.key === "Enter") accept(); + }} + /> + +
+ {validationMessage} +
+ + + + + + + ); +}; diff --git a/src/containers/App.tsx b/src/containers/App.tsx index 40393182..7a5b22f6 100644 --- a/src/containers/App.tsx +++ b/src/containers/App.tsx @@ -4,6 +4,7 @@ import { ConfirmDeleteDialog } from "../components/ConfirmDeleteDialog/ConfirmDe import LanguagePickerDialog from "../components/LanguagePickerDialog/LanguagePickerDialog"; import * as ReactModal from "react-modal"; import { RenameFileDialog } from "../components/RenameFileDialog/RenameFileDialog"; +import { UnlistedLanguageDialog } from "../components/UnlistedLanguageDialog/UnlistedLanguageDialog"; import { I18nProvider } from "@lingui/react"; import { i18n } from "../other/localization"; import RegistrationDialog from "../components/registration/RegistrationDialog"; @@ -68,6 +69,7 @@ export const App: React.FunctionComponent = observer(() => { + diff --git a/src/export/ImdiGenerator.ts b/src/export/ImdiGenerator.ts index 835acb78..e688e79a 100644 --- a/src/export/ImdiGenerator.ts +++ b/src/export/ImdiGenerator.ts @@ -31,6 +31,14 @@ import { fieldElement } from "./Imdi-static-fns"; import { GetOtherConfigurationSettings } from "../model/Project/OtherConfigurationSettings"; import { ImdiVocabularyTranslator } from "./ImdiVocabularyTranslator"; import { staticLanguageFinder } from "../languageFinder/LanguageFinder"; +import { + parseLanguageCodeAndName, + splitLanguageFieldValue +} from "../languageFinder/languageTagFieldValue"; +import { + getPrimarySubtag, + resolvePrivateUseCodes +} from "../languageFinder/privateUseLanguages"; // IMDI Date_Value_Type pattern from IMDI_3.0.xsd // Valid: YYYY, YYYY-MM, YYYY-MM-DD, date ranges with /, "Unknown", "Unspecified", or empty @@ -102,32 +110,8 @@ export function emitTildeBirthYearWarningOnce(): boolean { return false; } -/** - * Parse a language string that may be in various formats: - * - "code : Name" (legacy format, e.g., "pta : Guarani") - * - "code:Name" (no spaces, e.g., "qaa-x-MyLang:My Language") - * - "code" (plain code, e.g., "fra") - * - * Returns { code, name } where name is undefined if not present in the string. - */ -export function parseLanguageCodeAndName(languageString: string): { - code: string; - name: string | undefined; -} { - const trimmed = languageString.trim(); - - // Check for "code : Name" or "code:Name" format - // Use first colon to split, in case the name contains colons - const colonIndex = trimmed.indexOf(":"); - if (colonIndex > 0) { - const code = trimmed.substring(0, colonIndex).trim(); - const name = trimmed.substring(colonIndex + 1).trim(); - return { code, name: name.length > 0 ? name : undefined }; - } - - // Plain code format - return { code: trimmed, name: undefined }; -} +// Re-exported here because callers and tests have long imported it from this module. +export { parseLanguageCodeAndName }; export enum IMDIMode { OPEX, // wrap in OPEX elements, name .opex @@ -150,6 +134,10 @@ export default class ImdiGenerator { private keysThatHaveBeenOutput = new Set(); private mode: IMDIMode; private vocabularyTranslator: ImdiVocabularyTranslator; + // Which 3-letter code each of the project's private-use language tags exports as. See + // resolvePrivateUseCodes: this only does work for projects written by lameta 3.0.x, where + // every invented language was minted as "qaa-x-...", so they would all export as "qaa". + private privateUseCodes: Map = new Map(); // note, folder wil equal project if we're generating at the project level // otherwise, folder will be a session or person public constructor(mode: IMDIMode, folder?: Folder, project?: Project) { @@ -158,6 +146,53 @@ export default class ImdiGenerator { if (folder) this.folderInFocus = folder; if (project) this.project = project; this.vocabularyTranslator = new ImdiVocabularyTranslator(project); + if (project) { + this.privateUseCodes = resolvePrivateUseCodes( + ["SubjectLanguages", "WorkingLanguages", "metadataLanguages"].flatMap( + (key) => + splitLanguageFieldValue( + project.properties.getTextStringOrEmpty(key) + ).map((entry) => entry.code) + ) + ); + } + } + + /** + * The ISO 639-3 code to write for one of the project's languages. A tag absent from the + * collision map, such as one hand-edited into a session file, falls back to its own primary + * subtag, so nothing ever reaches IMDI with subtags still attached. + */ + private iso639_3ForExport(tag: string): string { + // Lowercase, because a project written by 3.0.x holds "qaa-x-MyLanguage" while a session + // that used the picker holds the same tag in lower case. A hand-edited file can hold a + // language element with no tag at all, so guard against undefined. + const key = (tag || "").trim().toLowerCase(); + const resolved = this.privateUseCodes.get(key); + if (resolved) return resolved; + const finder = staticLanguageFinder || this.project?.languageFinder; + return finder ? finder.getIso639_3Code(key) : getPrimarySubtag(key); + } + + /** + * The name to write for a language. IMDI must never carry a tag such as "qaa-x-Tolo" as a + * name. The project's own fields are the only place that holds the name of a language that + * ISO 639-3 does not list, so look there first, ignoring case. + */ + private nameForExport(tag: string): string { + const key = (tag || "").trim().toLowerCase(); + if (key.length === 0) return ""; + const fromProject = this.project + ?.getAllProjectLanguages() + .find((l) => (l.iso639_3 || "").toLowerCase() === key); + if (fromProject?.englishName && fromProject.englishName !== key) + return fromProject.englishName; + const finder = this.project?.languageFinder || staticLanguageFinder; + const name = finder?.findOneLanguageNameFromCode_Or_ReturnCode(key) ?? key; + if (name.toLowerCase() !== key) return name; + // Nothing knows this language. Its own tag carries the name after "-x-". + const afterX = (tag || "").trim().split(/-x-/i)[1]; + return afterX || name; } public static generateCorpus( @@ -502,12 +537,10 @@ export default class ImdiGenerator { languages.forEach((langString) => { // Parse "code : Name" or "code:Name" format, or plain code const { code, name } = parseLanguageCodeAndName(langString); - // If name was provided in the string, use it; otherwise look it up - const langName = - name || - this.project.languageFinder.findOneLanguageNameFromCode_Or_ReturnCode( - code - ); + // If name was provided in the string, use it; otherwise look it up. Use + // nameForExport, not the finder alone: for a tag that nothing knows, the finder + // returns the tag itself, which would put a raw tag in a Name element. + const langName = name || this.nameForExport(code); this.addSessionLanguage(code, langName, "Subject Language"); }); } else { @@ -522,12 +555,10 @@ export default class ImdiGenerator { workingLanguages.forEach((langString) => { // Parse "code : Name" or "code:Name" format, or plain code const { code, name } = parseLanguageCodeAndName(langString); - // If name was provided in the string, use it; otherwise look it up - const langName = - name || - this.project.languageFinder.findOneLanguageNameFromCode_Or_ReturnCode( - code - ); + // If name was provided in the string, use it; otherwise look it up. Use + // nameForExport, not the finder alone: for a tag that nothing knows, the finder + // returns the tag itself, which would put a raw tag in a Name element. + const langName = name || this.nameForExport(code); this.addSessionLanguage(code, langName, "Working Language"); }); } else { @@ -565,17 +596,13 @@ export default class ImdiGenerator { languages.forEach((langString) => { const { code, name } = parseLanguageCodeAndName(langString); // If name was provided in the string, use it; otherwise look it up - const langName = - name || - this.project.languageFinder.findOneLanguageNameFromCode_Or_ReturnCode( - code - ); + const langName = name || this.nameForExport(code); this.addSessionLanguage(code, langName, description); }); } private addSessionLanguage(code: string, name: string, description: string) { this.group("Language", () => { - this.element("Id", "ISO639-3:" + code); + this.element("Id", "ISO639-3:" + this.iso639_3ForExport(code)); this.element( "Name", name, @@ -607,15 +634,10 @@ export default class ImdiGenerator { // } else { // this.element("Id", "ISO639-3:" + lang); // } - this.element("Id", "ISO639-3:" + lang.code); + this.element("Id", "ISO639-3:" + this.iso639_3ForExport(lang.code)); //this.fieldLiteral("Id", lang); - this.element( - "Name", - this.project.languageFinder.findOneLanguageNameFromCode_Or_ReturnCode( - lang.code - ) - ); + this.element("Name", this.nameForExport(lang.code)); this.attributeLiteral( "Link", "http://www.mpi.nl/IMDI/Schema/MPI-Languages.xml" @@ -696,9 +718,7 @@ export default class ImdiGenerator { this.group("Keys", () => { for (const slot of metadataSlots) { // Always use ISO639-3 (3-letter) codes - ELAR, at least, can't handle 2-letter ISO639-1 codes - const iso639_3 = staticLanguageFinder - ? staticLanguageFinder.getIso639_3Code(slot.tag) - : slot.tag; + const iso639_3 = this.iso639_3ForExport(slot.tag); const value = `ISO639-3:${iso639_3}: ${slot.name}`; this.keyElement("MetadataLanguage", value); } @@ -849,11 +869,7 @@ export default class ImdiGenerator { this.attributeLiteral("Name", capitalCase(key)); // Always use ISO639-3 (3-letter) codes - archives can't handle 2-letter ISO639-1 codes - const languageFinder = - staticLanguageFinder || this.project.languageFinder; - const iso639_3 = languageFinder - ? languageFinder.getIso639_3Code(lang) - : lang; + const iso639_3 = this.iso639_3ForExport(lang); this.attributeLiteral("LanguageId", "ISO639-3:" + iso639_3); // Add index (1-based) to tie together translations of the same concept @@ -1681,9 +1697,7 @@ export default class ImdiGenerator { normalizedTranslation ); // Always use ISO639-3 (3-letter) codes - archives can't handle 2-letter ISO639-1 codes - const iso639_3 = staticLanguageFinder - ? staticLanguageFinder.getIso639_3Code(slot.tag) - : slot.tag; + const iso639_3 = this.iso639_3ForExport(slot.tag); newElement.attribute("LanguageId", "ISO639-3:" + iso639_3); newElement.attribute("Link", vocabularyUrl); newElement.attribute("Type", "OpenVocabulary"); @@ -1752,9 +1766,7 @@ export default class ImdiGenerator { if (translation) { const newElement = this.tail.element(elementName, translation); // Always use ISO639-3 (3-letter) codes - archives can't handle 2-letter ISO639-1 codes - const iso639_3 = staticLanguageFinder - ? staticLanguageFinder.getIso639_3Code(slot.tag) - : slot.tag; + const iso639_3 = this.iso639_3ForExport(slot.tag); newElement.attribute("LanguageId", "ISO639-3:" + iso639_3); this.mostRecentElement = newElement; this.tail = newElement.up(); diff --git a/src/export/personImdi-edolo.spec.ts b/src/export/personImdi-edolo.spec.ts index bf12ab88..a72925d4 100644 --- a/src/export/personImdi-edolo.spec.ts +++ b/src/export/personImdi-edolo.spec.ts @@ -125,6 +125,62 @@ describe("actor imdi export", () => { expect("Actor/Languages/Language[1]/Name").toHaveText("Blah blah"); }); + it("should reduce a person's custom language tag to its 3-letter code", () => { + // An archive's ingest rejects "ISO639-3:qab-x-tolo". IMDI must get the code alone, + // with the name the user typed in the Name element. + project.setSubjectLanguages("qab-x-tolo:Tolo"); + person.languages.splice(0, 10); + person.languages.push({ code: "qab-x-tolo", mother: true, primary: true }); + const gen = new ImdiGenerator(IMDIMode.RAW_IMDI, person, project); + const xml = gen.actor(person, "pretend-role", pretendSessionDate) as string; + setResultXml(xml); + expect("Actor/Languages/Language[1]/Id").toHaveText("ISO639-3:qab"); + expect("Actor/Languages/Language[1]/Name").toHaveText("Tolo"); + expect(xml.indexOf("-x-")).toBe(-1); + }); + + it("should match a person's tag to the project's whatever the case", () => { + // A .person file written by 3.0.x holds the tag with the capital letter the user typed. + project.setSubjectLanguages("qab-x-tolo:Tolo"); + person.languages.splice(0, 10); + person.languages.push({ code: "qab-x-Tolo", mother: true, primary: true }); + const gen = new ImdiGenerator(IMDIMode.RAW_IMDI, person, project); + const xml = gen.actor(person, "pretend-role", pretendSessionDate) as string; + setResultXml(xml); + expect("Actor/Languages/Language[1]/Id").toHaveText("ISO639-3:qab"); + expect("Actor/Languages/Language[1]/Name").toHaveText("Tolo"); + expect(xml.indexOf("-x-")).toBe(-1); + }); + + it("should never write a tag as a language name", () => { + // The project does not know this language, so nothing can give its real name. The name + // after "-x-" is the only thing left, and a tag must not reach the file as a name. + project.setSubjectLanguages("eng:English"); + person.languages.splice(0, 10); + person.languages.push({ code: "qac-x-Wobbly", mother: true, primary: true }); + const gen = new ImdiGenerator(IMDIMode.RAW_IMDI, person, project); + const xml = gen.actor(person, "pretend-role", pretendSessionDate) as string; + setResultXml(xml); + expect("Actor/Languages/Language[1]/Id").toHaveText("ISO639-3:qac"); + expect("Actor/Languages/Language[1]/Name").toHaveText("Wobbly"); + expect(xml.indexOf("-x-")).toBe(-1); + }); + + it("should not throw when a language element has no tag", () => { + // A hand-edited .person file can hold . + project.setSubjectLanguages("eng:English"); + person.languages.splice(0, 10); + person.languages.push({ + code: undefined as unknown as string, + mother: false, + primary: true + }); + const gen = new ImdiGenerator(IMDIMode.RAW_IMDI, person, project); + expect(() => + gen.actor(person, "pretend-role", pretendSessionDate) + ).not.toThrow(); + }); + /* we now remove that field, so we cannot test it this way it("should not output migrated language fields", () => { person.languages.splice(0, 10); diff --git a/src/export/sessionImdi-custom.spec.ts b/src/export/sessionImdi-custom.spec.ts index 18c5d101..e586f3de 100644 --- a/src/export/sessionImdi-custom.spec.ts +++ b/src/export/sessionImdi-custom.spec.ts @@ -312,7 +312,10 @@ describe("session imdi export", () => { expect("//Languages/Language[3]/Name").toHaveText("English"); }); - it("should handle custom language codes (qaa-x-*) correctly", () => { + // "ISO639-3:" promises an ISO 639-3 code, so only the primary subtag may follow it. An + // archive rejected "ISO639-3:qaa-x-Kürbinian" for exactly this reason. The IMDI schema + // pattern is (ISO639(-1|-2|-3)?:.*)?, so xmllint will not catch a mistake here. + it("should reduce a custom language tag (qaa-x-*) to its 3-letter code", () => { // Custom language with private-use code session.properties.setText("languages", "qaa-x-MyLanguage:My Custom Language"); @@ -325,10 +328,96 @@ describe("session imdi export", () => { ) ); - expect("//Languages/Language[1]/Id").toHaveText("ISO639-3:qaa-x-MyLanguage"); + expect("//Languages/Language[1]/Id").toHaveText("ISO639-3:qaa"); expect("//Languages/Language[1]/Name").toHaveText("My Custom Language"); }); + it("should give each custom language of a 3.0.x project a different code", () => { + // lameta 3.0.x minted every invented language as "qaa-x-...", so a project from that + // version can hold several that all claim qaa. They are sorted alphabetically and given + // the next free codes, which needs no rewriting of the user's files. + project.properties.setText( + "SubjectLanguages", + "qaa-x-zebra:Zebra;qaa-x-tolo:Tolo" + ); + session.properties.setText( + "languages", + "qaa-x-zebra:Zebra;qaa-x-tolo:Tolo" + ); + + const xml = ImdiGenerator.generateSession( + IMDIMode.RAW_IMDI, + session, + project, + true /*omit namespace*/ + ); + setResultXml(xml); + + expect("//Languages/Language[1]/Id").toHaveText("ISO639-3:qab"); + expect("//Languages/Language[1]/Name").toHaveText("Zebra"); + expect("//Languages/Language[2]/Id").toHaveText("ISO639-3:qaa"); + expect("//Languages/Language[2]/Name").toHaveText("Tolo"); + + // No Id may carry subtags. The schema would let them through, so we check ourselves. + const ids = xml.match(/]*>[^<]*<\/Id>/g) || []; + expect(ids.length).toBeGreaterThan(0); + ids.forEach((id) => expect(id).not.toContain("-x-")); + + project.properties.setText("SubjectLanguages", ""); + }); + + it("should match the project's tag to the session's whatever the case", () => { + // The project file is from 3.0.x and holds a capital letter. The session was edited by + // this version, which writes the tag in lower case. They are one language, so they must + // export as one code. + project.properties.setText( + "SubjectLanguages", + "qaa-x-Zebra:Zebra;qaa-x-Tolo:Tolo" + ); + session.properties.setText("languages", "qaa-x-zebra:Zebra"); + + setResultXml( + ImdiGenerator.generateSession( + IMDIMode.RAW_IMDI, + session, + project, + true /*omit namespace*/ + ) + ); + + // Tolo sorts first, so it keeps qaa and Zebra gets qab. + expect("//Languages/Language[1]/Id").toHaveText("ISO639-3:qab"); + expect("//Languages/Language[1]/Name").toHaveText("Zebra"); + + project.properties.setText("SubjectLanguages", ""); + }); + + it("should give a working language and a metadata language their own codes", () => { + // The collision map has to read all three of the project's language fields. If it read + // only SubjectLanguages, these two would both export as qaa. + project.properties.setText("SubjectLanguages", ""); + project.properties.setText("WorkingLanguages", "qaa-x-alpha:Alpha"); + project.properties.setText("metadataLanguages", "qaa-x-beta:Beta"); + session.properties.setText("languages", "qaa-x-alpha:Alpha;qaa-x-beta:Beta"); + + setResultXml( + ImdiGenerator.generateSession( + IMDIMode.RAW_IMDI, + session, + project, + true /*omit namespace*/ + ) + ); + + expect("//Languages/Language[1]/Id").toHaveText("ISO639-3:qaa"); + expect("//Languages/Language[1]/Name").toHaveText("Alpha"); + expect("//Languages/Language[2]/Id").toHaveText("ISO639-3:qab"); + expect("//Languages/Language[2]/Name").toHaveText("Beta"); + + project.properties.setText("WorkingLanguages", ""); + project.properties.setText("metadataLanguages", ""); + }); + it("should fall back to project languages when session has no languages", () => { // Set up project-level languages in "code : Name" format project.properties.setText("SubjectLanguages", "ita : Italian;deu : German"); diff --git a/src/languageFinder/LanguageFinder.spec.ts b/src/languageFinder/LanguageFinder.spec.ts index e969e4f2..af35edf4 100644 --- a/src/languageFinder/LanguageFinder.spec.ts +++ b/src/languageFinder/LanguageFinder.spec.ts @@ -220,5 +220,137 @@ describe("LanguageFinder", () => { expect(languageFinder.getIso639_3Code("")).toBe(""); expect(languageFinder.getIso639_3Code(" ")).toBe(""); }); + + // "ISO639-3:" promises an ISO 639-3 code, so nothing but the primary subtag may follow it. + it("should drop the private-use subtag of a custom language", () => { + expect(languageFinder.getIso639_3Code("qaa-x-foo")).toBe("qaa"); + expect(languageFinder.getIso639_3Code("qab-x-tolo")).toBe("qab"); + expect(languageFinder.getIso639_3Code("QAB-x-Tolo")).toBe("qab"); + }); + + it("should drop region and script subtags", () => { + expect(languageFinder.getIso639_3Code("en-US")).toBe("eng"); + expect(languageFinder.getIso639_3Code("pt-BR")).toBe("por"); + expect(languageFinder.getIso639_3Code("tpi-PG")).toBe("tpi"); + // The subtag is dropped, but "zh" stays 2 letters: our index has no iso639_1 on the + // "zho" entry, so the 2-to-3-letter lookup finds nothing. That gap is older than this. + expect(languageFinder.getIso639_3Code("zh-Hans")).toBe("zh"); + }); + }); + + describe("private-use codes in the qaa..qtz range", () => { + // A code with no name is of no use to an archive, so no row offers one. The user adds a + // language that ISO 639-3 does not list through the dialog, which asks for the name. + it("does not offer the typed code by itself", () => { + const matches = languageFinder.makeMatchesAndLabelsForSelect("qab"); + expect(matches.some((m) => m.languageInfo.iso639_3 === "qab")).toBe(false); + // Qabiao, whose English name starts with those same three letters, is still there. + expect(matches.some((m) => m.languageInfo.iso639_3 === "laq")).toBe(true); + }); + + it("still finds a real language once enough letters are typed", () => { + const matches = languageFinder.makeMatchesAndLabelsForSelect("qabi"); + expect(matches[0].languageInfo.iso639_3).toBe("laq"); + }); + + it("does not invent a language for a code outside the range", () => { + // "u" is past "t", so qua is not private-use. It is Quapaw, a real ISO 639-3 code, + // and the finder must give the real language, not an "[Unlisted]" row. + const matches = languageFinder.makeMatchesAndLabelsForSelect("qua"); + const qua = matches.filter((m) => m.languageInfo.iso639_3 === "qua"); + // Exactly one: a synthesized row would make a second. + expect(qua.length).toBe(1); + expect(qua[0].languageInfo.englishName).toBe("Quapaw"); + }); + + it("gives the project's own language when its code is typed", () => { + // If the project holds qac-x-foobar, typing "qac" and typing "foobar" must both give + // that one entry. A second row holding the code alone would be a different language. + const finder = new LanguageFinder( + () => undefined, + () => [{ iso639_3: "qac-x-foobar", englishName: "Foobar" }] + ); + for (const typed of ["qac", "foobar"]) { + const matches = finder.makeMatchesAndLabelsForSelect(typed); + expect(matches[0].languageInfo.iso639_3).toBe("qac-x-foobar"); + expect(matches[0].languageInfo.englishName).toBe("Foobar"); + expect( + matches.filter((m) => m.languageInfo.iso639_3 === "qac").length + ).toBe(0); + } + }); + + it("drops the index's own row for a code the project has taken", () => { + // The index carries a "qaa" row named "Unlisted Language". Once the project has given + // qaa to a language of its own, that row is only a way to make a duplicate. + const finder = new LanguageFinder( + () => undefined, + () => [{ iso639_3: "qaa-x-alpha", englishName: "Alpha" }] + ); + const matches = finder.makeMatchesAndLabelsForSelect("qaa"); + expect(matches[0].languageInfo.iso639_3).toBe("qaa-x-alpha"); + expect(matches.some((m) => m.languageInfo.iso639_3 === "qaa")).toBe(false); + }); + + it("offers no nameless code however few letters are typed", () => { + // The index carries one row in the range, qaa, named "Unlisted Language". The trie finds + // it from "qa" and from "unlisted" as well as from "qaa". + for (const typed of ["qa", "qaa", "unlisted"]) { + const matches = languageFinder.makeMatchesAndLabelsForSelect(typed); + expect( + matches.filter((m) => m.languageInfo.iso639_3?.toLowerCase() === "qaa") + ).toEqual([]); + } + }); + + it("keeps a bare code that the project itself uses", () => { + // An older project holds "qaa:Foo Bar", with the name in the field. That language is + // real, so the field must still offer it. + const finder = new LanguageFinder( + () => undefined, + () => [{ iso639_3: "qaa", englishName: "Foo Bar" }] + ); + for (const typed of ["qaa", "Foo"]) { + const matches = finder.makeMatchesAndLabelsForSelect(typed); + const qaa = matches.filter((m) => m.languageInfo.iso639_3 === "qaa"); + expect(qaa.length).toBe(1); + expect(qaa[0].languageInfo.englishName).toBe("Foo Bar"); + } + }); + + it("offers nothing for a free private-use code", () => { + // The field itself offers to add one, and the dialog asks for the name. The finder has + // no language to give. + const finder = new LanguageFinder( + () => undefined, + () => [{ iso639_3: "qac-x-foobar", englishName: "Foobar" }] + ); + const matches = finder.makeMatchesAndLabelsForSelect("qad"); + expect(matches.some((m) => m.languageInfo.iso639_3 === "qad")).toBe(false); + }); + + it("finds any of the project's languages by name, not just the first", () => { + const finder = new LanguageFinder( + () => ({ iso639_3: "qaa-x-kurbinia", englishName: "Kürbinian" }), + () => [ + { iso639_3: "qaa-x-kurbinia", englishName: "Kürbinian" }, + { iso639_3: "qab-x-tolo", englishName: "Tolo" } + ] + ); + const matches = finder.makeMatchesAndLabelsForSelect("Tolo"); + expect(matches.some((m) => m.languageInfo.iso639_3 === "qab-x-tolo")).toBe( + true + ); + }); + + it("labels a typed code with the project's name for it", () => { + const finder = new LanguageFinder( + () => undefined, + () => [{ iso639_3: "qab", englishName: "Tolo" }] + ); + const matches = finder.makeMatchesAndLabelsForSelect("qab"); + expect(matches[0].languageInfo.iso639_3).toBe("qab"); + expect(matches[0].nameMatchingWhatTheyTyped).toBe("Tolo"); + }); }); }); diff --git a/src/languageFinder/LanguageFinder.ts b/src/languageFinder/LanguageFinder.ts index b9739420..4a762fbd 100644 --- a/src/languageFinder/LanguageFinder.ts +++ b/src/languageFinder/LanguageFinder.ts @@ -1,6 +1,7 @@ import TrieSearch from "trie-search"; // this was before vite, I don't know if it's still true: NOTE: this sometimes seems to give incomplete (or empty?) json during the GeneriCsvEporter.Spect.ts run... maybe some timing bug with the webpack loader? import langIndex from "./langindex.json"; +import { getPrimarySubtag, isPrivateUseTag } from "./privateUseLanguages"; export class Language { public englishName: string; @@ -51,9 +52,17 @@ export function setupLanguageFinderForTests() { export class LanguageFinder { private index: TrieSearch; private getDefaultLanguage: () => ILangIndexEntry | undefined; - - constructor(getDefaultLanguage: () => ILangIndexEntry | undefined) { + // All of the project's languages, not just the first one. Without this, a language the user + // invented (e.g. "qab-x-tolo") could not be found by name anywhere except the project field + // that created it. Optional because many unit tests construct a finder with no project. + private getProjectLanguages: () => ILangIndexEntry[]; + + constructor( + getDefaultLanguage: () => ILangIndexEntry | undefined, + getProjectLanguages?: () => ILangIndexEntry[] + ) { this.getDefaultLanguage = getDefaultLanguage; + this.getProjectLanguages = getProjectLanguages ?? (() => []); // currently this uses a trie, which is overkill for this number of items, // but I'm using it because I do need prefix matching and this library does that. @@ -156,20 +165,52 @@ export class LanguageFinder { public makeMatchesAndLabelsForSelect( prefix: string ): { languageInfo: Language; nameMatchingWhatTheyTyped: string }[] { - const projectContentLanguage = this.getDefaultLanguage(); const pfx = prefix.toLocaleLowerCase(); - const sortedListOfMatches = this.findMatchesForSelect(prefix); + const unfiltered = this.findMatchesForSelect(prefix); + // Never offer a code from the qaa..qtz range with no name of its own, whatever the user + // typed. The index carries one such row, qaa, named "Unlisted Language", which the trie + // also finds from "qa" and from "unlisted". A language with a code and no name is of no + // use to an archive. To add one, the user goes through the dialog, which asks for the + // name. See UnlistedLanguageDialog. + const tagsTheProjectHas = new Set( + this.getProjectLanguages().map((l) => (l.iso639_3 || "").toLowerCase()) + ); + const sortedListOfMatches = unfiltered.filter((l) => { + const code = (l.iso639_3 || "").toLowerCase(); + return !( + isPrivateUseTag(code) && + code.length === 3 && + !tagsTheProjectHas.has(code) + ); + }); // see https://tools.ietf.org/html/bcp47 note these are language tags, not subtags, so are qaa-qtz, not qaaa-qabx, which are script subtags - if (pfx >= "qaa" && pfx <= "qtz") { - const l = new Language({ - iso639_3: prefix, - englishName: - // if they have given us the name for this custom language in the Project settings, use it - projectContentLanguage?.iso639_3 === prefix - ? projectContentLanguage?.englishName - : `${prefix} [Unlisted]` - }); - sortedListOfMatches.push(l); + if (isPrivateUseTag(pfx) && pfx.length === 3) { + // The project may already have given this code to a language, as "qac-x-foobar". Then + // typing "qac" must give that same language. A second entry holding the code alone + // would be a different language to an archive. + const owner = this.getProjectLanguages().find( + (l) => getPrimarySubtag(l.iso639_3) === pfx + ); + const removeFirst = (test: (l: Language) => boolean) => { + const at = sortedListOfMatches.findIndex(test); + return at >= 0 ? sortedListOfMatches.splice(at, 1)[0] : undefined; + }; + if (owner) { + const fromIndex = removeFirst((m) => m.iso639_3 === owner.iso639_3); + // The project's own name wins. The index calls qaa "Unlisted Language", but a project + // holding "qaa:Foo Bar" has named that language itself. A project entry with no name + // of its own carries the code as its name, and then the index reads better. + const ownerHasItsOwnName = + !!owner.englishName && + owner.englishName.toLowerCase() !== + (owner.iso639_3 || "").toLowerCase(); + // unshift, not push: the sort above has already run, so a pushed entry lands at the + // bottom. Typing "qac" used to select laq (Qabiao), whose English name starts with + // those same three letters. The real language stays in the list, one row down. + sortedListOfMatches.unshift( + ownerHasItsOwnName || !fromIndex ? new Language(owner) : fromIndex + ); + } } return sortedListOfMatches.map((l) => ({ languageInfo: l, @@ -207,8 +248,9 @@ export class LanguageFinder { // (e.g., "qaa" -> "My Indigenous Language" instead of "Unlisted Language"). // For standard ISO codes like "por", we should NEVER override - the index // has the correct name ("Portuguese"). - const code = projectContentLanguage.iso639_3.toLowerCase(); - const isPrivateUseCode = code >= "qaa" && code <= "qtz"; + const isPrivateUseCode = isPrivateUseTag( + projectContentLanguage.iso639_3 + ); if ( isPrivateUseCode && projectContentLanguage.englishName && @@ -229,6 +271,21 @@ export class LanguageFinder { } } } + + // The languages the user invented in this project are not in the index, and only the first + // subject language arrives via getDefaultLanguage. Without this, typing "Tolo" on a Person + // form finds nothing, even though the project defined a language called Tolo. + const typed = s.toLowerCase().trim(); + if (typed.length > 0) { + this.getProjectLanguages().forEach((projectLanguage) => { + if (langs.some((l) => l.iso639_3 === projectLanguage.iso639_3)) return; + const name = (projectLanguage.englishName || "").toLowerCase().trim(); + const code = (projectLanguage.iso639_3 || "").toLowerCase().trim(); + if (name.startsWith(typed) || code.startsWith(typed)) { + langs.push(projectLanguage); + } + }); + } return langs; } @@ -351,7 +408,7 @@ export class LanguageFinder { // if we got something other than the code back, that means we did recognize it as a known code. if (c.length > 0 && c.toLowerCase() !== codeOrLanguageName.toLowerCase()) return codeOrLanguageName; - else if (c >= "qaa" && c <= "qtz") return c; + else if (isPrivateUseTag(c)) return c; else return this.convertNameToCode(codeOrLanguageName); } @@ -494,13 +551,15 @@ export class LanguageFinder { } /** - * Convert a language code to its ISO 639-3 (3-letter) form. - * If already 3 letters, returns as-is. If 2-letter ISO 639-1 code, - * looks up the corresponding 3-letter code. - * E.g., "en" -> "eng", "es" -> "spa", "etr" -> "etr" + * Convert a language tag to its ISO 639-3 (3-letter) form. + * + * Only the primary subtag survives, because "ISO639-3:" promises an ISO 639-3 code and + * nothing else. An archive rejected "ISO639-3:qaa-x-Kürbinian" for exactly this reason. + * E.g., "en" -> "eng", "etr" -> "etr", "qaa-x-Foo" -> "qaa", "en-US" -> "eng", + * "zh-Hans" -> "zh", because our index has no iso639_1 on the "zho" entry */ public getIso639_3Code(code: string): string { - const lowered = code.toLowerCase().trim(); + const lowered = getPrimarySubtag(code); if (lowered.length === 0) return lowered; // Already a 3-letter code? Return as-is diff --git a/src/languageFinder/languageTagFieldValue.spec.ts b/src/languageFinder/languageTagFieldValue.spec.ts new file mode 100644 index 00000000..cb47ca52 --- /dev/null +++ b/src/languageFinder/languageTagFieldValue.spec.ts @@ -0,0 +1,86 @@ +import { describe, it, expect } from "vitest"; +import { + parseLanguageCodeAndName, + splitLanguageFieldValue, + serializeLanguageFieldValue +} from "./languageTagFieldValue"; + +describe("parseLanguageCodeAndName", () => { + it("reads the legacy 'code : Name' form", () => { + expect(parseLanguageCodeAndName("pta : Guarani")).toEqual({ + code: "pta", + name: "Guarani" + }); + }); + + it("splits on the first colon, so a name may contain one", () => { + expect(parseLanguageCodeAndName("qab-x-tolo:Tolo: the north dialect")).toEqual( + { code: "qab-x-tolo", name: "Tolo: the north dialect" } + ); + }); + + it("gives no name for a plain code", () => { + expect(parseLanguageCodeAndName("fra")).toEqual({ + code: "fra", + name: undefined + }); + }); +}); + +describe("splitLanguageFieldValue", () => { + it("drops the empty entries of a field written by hand", () => { + expect(splitLanguageFieldValue(";;eng;")).toEqual([ + { code: "eng", name: undefined } + ]); + }); +}); + +describe("serializeLanguageFieldValue", () => { + it("keeps the name of a language that ISO 639-3 does not list", () => { + expect( + serializeLanguageFieldValue([{ value: "qab-x-tolo", label: "Tolo" }]) + ).toBe("qab-x-tolo:Tolo"); + }); + + it("keeps a name the file already carried, whatever the code", () => { + // Dragging the pills into a new order used to turn "pta : Guarani" into "pta". + expect( + serializeLanguageFieldValue([ + { value: "pta", label: "Guarani", hadName: true } + ]) + ).toBe("pta:Guarani"); + }); + + it("writes no name for a language the index knows", () => { + expect( + serializeLanguageFieldValue([{ value: "eng", label: "English" }]) + ).toBe("eng"); + }); + + it("keeps the order it is given, and joins with a semicolon", () => { + expect( + serializeLanguageFieldValue([ + { value: "qab-x-tolo", label: "Tolo" }, + { value: "eng", label: "English" } + ]) + ).toBe("qab-x-tolo:Tolo;eng"); + }); + + it("does not write the name twice", () => { + expect( + serializeLanguageFieldValue([ + { value: "qab-x-tolo:Tolo", label: "Tolo", hadName: true } + ]) + ).toBe("qab-x-tolo:Tolo"); + }); + + it("round-trips a field through parse and serialize", () => { + const text = "qab-x-tolo:Tolo;pta:Guarani;eng"; + const parsed = splitLanguageFieldValue(text).map((e) => ({ + value: e.code, + label: e.name ?? "", + hadName: e.name !== undefined + })); + expect(serializeLanguageFieldValue(parsed)).toBe(text); + }); +}); diff --git a/src/languageFinder/languageTagFieldValue.ts b/src/languageFinder/languageTagFieldValue.ts new file mode 100644 index 00000000..45f261c9 --- /dev/null +++ b/src/languageFinder/languageTagFieldValue.ts @@ -0,0 +1,72 @@ +// lameta stores a list of languages in one field as "code:Name" pairs joined by semicolons, +// e.g. "eng;qab-x-tolo:Tolo". The semicolon comes from the participants field, which has used +// it for years. The name is carried alongside the tag because a language the user invented is +// not in the language index, so there is nowhere else to look it up. + +import { isPrivateUseTag } from "./privateUseLanguages"; + +export interface ILanguageCodeAndName { + code: string; + name: string | undefined; +} + +/** + * Parse one language entry, which may be in any of these forms: + * - "code : Name" (legacy format, e.g., "pta : Guarani") + * - "code:Name" (no spaces, e.g., "qab-x-tolo:Tolo") + * - "code" (plain code, e.g., "fra") + * + * Returns { code, name } where name is undefined if not present in the string. + */ +export function parseLanguageCodeAndName( + languageString: string +): ILanguageCodeAndName { + const trimmed = languageString.trim(); + + // Check for "code : Name" or "code:Name" format + // Use first colon to split, in case the name contains colons + const colonIndex = trimmed.indexOf(":"); + if (colonIndex > 0) { + const code = trimmed.substring(0, colonIndex).trim(); + const name = trimmed.substring(colonIndex + 1).trim(); + return { code, name: name.length > 0 ? name : undefined }; + } + + // Plain code format + return { code: trimmed, name: undefined }; +} + +/** Split the whole text of a language field into its entries. */ +export function splitLanguageFieldValue( + fieldText: string +): ILanguageCodeAndName[] { + return (fieldText || "") + .split(";") + .map((s) => s.trim()) + .filter((s) => s.length > 0) + .map(parseLanguageCodeAndName); +} + +/** + * Write the entries of a language field back into the field's text. + * + * The name is kept when the file already carried one, whatever the code, and for a language + * in the qaa..qtz range, because nothing outside the file knows the name of a language that + * ISO 639-3 does not list. Dropping the name here used to lose it whenever the user dragged + * the pills into a new order. + */ +export function serializeLanguageFieldValue( + choices: Array<{ value: string; label: string; hadName?: boolean }> +): string { + return choices + .map((o) => { + const alreadyHasName = o.value.indexOf(":") > 0; + const needsName = + !alreadyHasName && + (o.hadName || isPrivateUseTag(o.value)) && + o.label && + o.label !== o.value; + return needsName ? `${o.value}:${o.label}` : o.value; + }) + .join(";"); +} diff --git a/src/languageFinder/privateUseLanguages.spec.ts b/src/languageFinder/privateUseLanguages.spec.ts new file mode 100644 index 00000000..f60dea1d --- /dev/null +++ b/src/languageFinder/privateUseLanguages.spec.ts @@ -0,0 +1,167 @@ +import { describe, it, expect } from "vitest"; +import { + allocateNextPrivateUseTag, + getPrimarySubtag, + isPrivateUseTag, + resolvePrivateUseCodes, + slugifyForPrivateUseSubtag +} from "./privateUseLanguages"; + +describe("getPrimarySubtag", () => { + it("takes the part before the first hyphen", () => { + expect(getPrimarySubtag("qaa-x-Foo")).toBe("qaa"); + expect(getPrimarySubtag("en-US")).toBe("en"); + expect(getPrimarySubtag("zh-Hans-CN")).toBe("zh"); + }); + it("lowercases and trims", () => { + expect(getPrimarySubtag(" QAB-x-Tolo ")).toBe("qab"); + }); + it("passes through a bare code", () => { + expect(getPrimarySubtag("eng")).toBe("eng"); + }); +}); + +describe("isPrivateUseTag", () => { + it("accepts the ends of the qaa..qtz range", () => { + expect(isPrivateUseTag("qaa")).toBe(true); + expect(isPrivateUseTag("qtz")).toBe(true); + expect(isPrivateUseTag("qkm")).toBe(true); + }); + it("accepts a tag whose primary subtag is in the range", () => { + expect(isPrivateUseTag("qab-x-tolo")).toBe(true); + }); + // A plain string comparison would wrongly accept this, because "qaa-x-foo" < "qtz". + it("rejects a full tag when the whole string is tested as one subtag", () => { + expect(isPrivateUseTag("qaax")).toBe(false); + }); + it("rejects codes just outside the range", () => { + expect(isPrivateUseTag("qua")).toBe(false); // "u" is past "t" + expect(isPrivateUseTag("qzz")).toBe(false); + }); + it("rejects real ISO 639-3 codes", () => { + expect(isPrivateUseTag("laq")).toBe(false); + expect(isPrivateUseTag("eng")).toBe(false); + expect(isPrivateUseTag("que")).toBe(false); + }); + it("rejects empty and short input", () => { + expect(isPrivateUseTag("")).toBe(false); + expect(isPrivateUseTag("qa")).toBe(false); + }); +}); + +describe("slugifyForPrivateUseSubtag", () => { + it("drops accents", () => { + expect(slugifyForPrivateUseSubtag("Kürbinian")).toBe("kurbinia"); + }); + it("truncates to the 8 characters BCP 47 allows in a subtag", () => { + expect(slugifyForPrivateUseSubtag("abcdefghijkl")).toBe("abcdefgh"); + }); + it("removes spaces and the field's own separators", () => { + expect(slugifyForPrivateUseSubtag("My Lang")).toBe("mylang"); + expect(slugifyForPrivateUseSubtag("a:b;c")).toBe("abc"); + }); + it("keeps digits", () => { + expect(slugifyForPrivateUseSubtag("Lang2")).toBe("lang2"); + }); + it("falls back when nothing survives, because a subtag cannot be empty", () => { + expect(slugifyForPrivateUseSubtag("!!!")).toBe("lang"); + expect(slugifyForPrivateUseSubtag("")).toBe("lang"); + }); +}); + +describe("allocateNextPrivateUseTag", () => { + it("gives the first language qaa", () => { + expect(allocateNextPrivateUseTag("Tolo", [])).toBe("qaa-x-tolo"); + }); + it("gives the second language qab", () => { + expect(allocateNextPrivateUseTag("Tolo", ["eng", "qaa-x-kurbinia"])).toBe( + "qab-x-tolo" + ); + }); + it("skips over a gap rather than reusing a taken code", () => { + expect( + allocateNextPrivateUseTag("Third", ["qaa-x-one", "qac-x-three"]) + ).toBe("qab-x-third"); + }); + it("ignores codes outside the range when deciding what is taken", () => { + expect(allocateNextPrivateUseTag("Tolo", ["laq", "que", "eng"])).toBe( + "qaa-x-tolo" + ); + }); + it("returns undefined once all 520 codes are taken", () => { + const all: string[] = []; + for (let second = 0; second < 20; second++) { + for (let third = 0; third < 26; third++) { + all.push( + "q" + + String.fromCharCode(97 + second) + + String.fromCharCode(97 + third) + ); + } + } + expect(all.length).toBe(520); + expect(allocateNextPrivateUseTag("Tolo", all)).toBeUndefined(); + // one freed code is enough + expect(allocateNextPrivateUseTag("Tolo", all.slice(1))).toBe("qaa-x-tolo"); + }); +}); + +describe("resolvePrivateUseCodes", () => { + it("leaves distinct tags on their own codes", () => { + const m = resolvePrivateUseCodes(["qaa-x-kurbinia", "qab-x-tolo"]); + expect(m.get("qaa-x-kurbinia")).toBe("qaa"); + expect(m.get("qab-x-tolo")).toBe("qab"); + }); + it("sorts a colliding group and hands out the next free codes", () => { + // what a project written by lameta 3.0.x looks like + const m = resolvePrivateUseCodes(["qaa-x-zebra", "qaa-x-tolo"]); + expect(m.get("qaa-x-tolo")).toBe("qaa"); // alphabetically first keeps the code + expect(m.get("qaa-x-zebra")).toBe("qab"); + }); + it("does not steal a code another group already holds", () => { + const m = resolvePrivateUseCodes([ + "qaa-x-zebra", + "qaa-x-tolo", + "qab-x-alpha" + ]); + expect(m.get("qaa-x-tolo")).toBe("qaa"); + expect(m.get("qab-x-alpha")).toBe("qab"); + expect(m.get("qaa-x-zebra")).toBe("qac"); + }); + it("ignores tags outside the range", () => { + const m = resolvePrivateUseCodes(["eng", "laq", "qaa-x-tolo"]); + expect(m.has("eng")).toBe(false); + expect(m.has("laq")).toBe(false); + expect(m.get("qaa-x-tolo")).toBe("qaa"); + }); + it("maps a bare code to itself", () => { + const m = resolvePrivateUseCodes(["qaa", "qab"]); + expect(m.get("qaa")).toBe("qaa"); + expect(m.get("qab")).toBe("qab"); + }); + it("treats the same tag in two cases as one language", () => { + // lameta 3.0.x built the tag from the name the user typed, so one file holds + // "qaa-x-Tolo" and another holds "qaa-x-tolo". They are one language. + const map = resolvePrivateUseCodes(["qaa-x-Tolo", "qaa-x-tolo"]); + expect(map.size).toBe(1); + expect(map.get("qaa-x-tolo")).toBe("qaa"); + }); + + it("keys the map in lower case", () => { + const map = resolvePrivateUseCodes(["qaa-x-Zebra", "qaa-x-Tolo"]); + expect(map.get("qaa-x-tolo")).toBe("qaa"); + expect(map.get("qaa-x-zebra")).toBe("qab"); + }); + + it("treats a repeated tag as one language", () => { + const m = resolvePrivateUseCodes(["qaa-x-tolo", "qaa-x-tolo"]); + expect(m.size).toBe(1); + expect(m.get("qaa-x-tolo")).toBe("qaa"); + }); + it("gives every tag a different code", () => { + const tags = ["qaa-x-one", "qaa-x-two", "qaa-x-three", "qaa-x-four"]; + const m = resolvePrivateUseCodes(tags); + const codes = tags.map((t) => m.get(t)); + expect(new Set(codes).size).toBe(4); + }); +}); diff --git a/src/languageFinder/privateUseLanguages.ts b/src/languageFinder/privateUseLanguages.ts new file mode 100644 index 00000000..ae8d8520 --- /dev/null +++ b/src/languageFinder/privateUseLanguages.ts @@ -0,0 +1,127 @@ +// BCP 47 and ISO 639-3 reserve the range qaa..qtz for local use. lameta uses it for languages +// that the ISO 639-3 index does not list, giving each one its own code so that an archive can +// tell them apart. The full tag we store looks like "qab-x-tolo", and the name the user typed +// is kept after a colon, e.g. "qab-x-tolo:Tolo". + +const kPrivateUsePattern = /^q[a-t][a-z]$/; +const kMaxSubtagLength = 8; // BCP 47 allows 1-8 alphanumerics per subtag + +/** The part of a language tag before the first hyphen, e.g. "qaa-x-Foo" --> "qaa" */ +export function getPrimarySubtag(tag: string): string { + return (tag || "").trim().toLowerCase().split("-")[0]; +} + +/** + * True if the tag's primary subtag is in the qaa..qtz range that ISO 639-3 reserves for + * local use. Note that a plain string comparison would wrongly accept "qaa-x-foo", because + * "qaa-x-foo" < "qtz". + */ +export function isPrivateUseTag(tag: string): boolean { + return kPrivateUsePattern.test(getPrimarySubtag(tag)); +} + +/** + * Turn a language name into something legal in the "-x-" part of a tag: lowercase ASCII + * alphanumerics, at most 8 of them. E.g. "Kürbinian" --> "kurbinia". Nothing reads this part; + * it is there so that a human reading the file can tell the tags apart. The full name is + * stored after the colon. + */ +export function slugifyForPrivateUseSubtag(name: string): string { + const slug = (name || "") + .normalize("NFD") + .replace(/[̀-ͯ]/g, "") // drop combining accents, so "ü" becomes "u" + .toLowerCase() + .replace(/[^a-z0-9]/g, "") + .slice(0, kMaxSubtagLength); + // A subtag must have at least one character. "私" and "!!!" both slugify to nothing. + return slug.length > 0 ? slug : "lang"; +} + +/** Every code in the qaa..qtz range, in order. 520 of them. */ +function* privateUseCodesInOrder(): Generator { + for (let second = "a".charCodeAt(0); second <= "t".charCodeAt(0); second++) { + for (let third = "a".charCodeAt(0); third <= "z".charCodeAt(0); third++) { + yield "q" + String.fromCharCode(second) + String.fromCharCode(third); + } + } +} + +/** + * Build a tag for a new language, using the first code in qaa..qtz that no tag in tagsInUse + * has already claimed. Returns undefined when all 520 are taken, so that the caller can say + * so rather than handing out a duplicate. + */ +export function allocateNextPrivateUseTag( + name: string, + tagsInUse: string[] +): string | undefined { + const taken = new Set(tagsInUse.map((t) => getPrimarySubtag(t))); + for (const code of privateUseCodesInOrder()) { + if (!taken.has(code)) { + return `${code}-x-${slugifyForPrivateUseSubtag(name)}`; + } + } + return undefined; +} + +/** + * Decide which 3-letter code each private-use tag should export as. + * + * A tag whose primary subtag is unique in the project keeps it, so a language created by this + * version always exports as the code in its own tag. Only tags that share a primary subtag + * need help, which happens in projects written by lameta 3.0.x, where every invented language + * was minted as "qaa-x-...". Those are sorted alphabetically and given the next free codes. + * + * qaa-x-tolo, qaa-x-zebra --> qaa-x-tolo: qaa, qaa-x-zebra: qab + * + * Tags outside the qaa..qtz range are ignored; they are not ours to renumber. + * + * Both the keys and the comparisons are lowercase. lameta 3.0.x minted "qaa-x-MyLanguage" + * from the name the user typed, so the same language appears in one file with a capital + * letter and in another without one. + */ +export function resolvePrivateUseCodes(tags: string[]): Map { + const result = new Map(); + + const groups = new Map(); + for (const rawTag of tags) { + const tag = (rawTag || "").trim().toLowerCase(); + if (!isPrivateUseTag(tag)) continue; + const code = getPrimarySubtag(tag); + if (!groups.has(code)) groups.set(code, []); + const members = groups.get(code)!; + if (!members.includes(tag)) members.push(tag); + } + + // Every group's own code is reserved, so renumbering one group never steals from another. + const claimed = new Set(groups.keys()); + + for (const [code, members] of groups) { + if (members.length === 1) { + result.set(members[0], code); + continue; + } + const sorted = [...members].sort((a, b) => a.localeCompare(b)); + // The alphabetically first member keeps the shared code; the rest get new ones. + result.set(sorted[0], code); + const next = privateUseCodesInOrder(); + for (const tag of sorted.slice(1)) { + let assigned: string | undefined; + for (let item = next.next(); !item.done; item = next.next()) { + if (!claimed.has(item.value)) { + assigned = item.value; + break; + } + } + if (assigned === undefined) { + // All 520 codes are in use. Leave the tag on its original code rather than dropping it. + result.set(tag, code); + continue; + } + claimed.add(assigned); + result.set(tag, assigned); + } + } + + return result; +} diff --git a/src/model/Project/Project.ts b/src/model/Project/Project.ts index 0f1de79e..20302c7f 100644 --- a/src/model/Project/Project.ts +++ b/src/model/Project/Project.ts @@ -23,9 +23,11 @@ import { t } from "@lingui/macro"; import { analyticsEvent } from "../../other/analytics"; import userSettings from "../../other/UserSettings"; import { + ILangIndexEntry, LanguageFinder, staticLanguageFinder } from "../../languageFinder/LanguageFinder"; +import { splitLanguageFieldValue } from "../../languageFinder/languageTagFieldValue"; import { LanguageSlot } from "../field/TextHolder"; // FIXED: Import safeCaptureException instead of using Sentry directly // CONTEXT: This prevents E2E test failures from Sentry RendererTransport errors @@ -477,7 +479,8 @@ export class Project extends Folder { this.knownFields = fieldDefinitionsOfCurrentConfig.project; // for csv export this.languageFinder = new LanguageFinder( - () => this.getFirstSubjectLanguageCodeAndName() // REVIEW now that we have multiple languages + () => this.getFirstSubjectLanguageCodeAndName(), // used for sort priority + () => this.getAllProjectLanguages() ); this.loadSettingsFromConfiguration(); @@ -1381,29 +1384,44 @@ export class Project extends Folder { this.properties.setText("SubjectLanguages", content); } + private getLanguagesOfField(fieldName: string): ILangIndexEntry[] { + return splitLanguageFieldValue( + this.properties.getTextStringOrEmpty(fieldName) + ).map((entry) => ({ + iso639_3: entry.code.toLowerCase(), + // In the degenerate case in which there was no ":" in the incoming xml element, + // use the code as the name. + englishName: entry.name ?? entry.code + })); + } + + /** + * Every language named anywhere in the project's own language fields. The language finder + * needs all of them, not just the first subject language, or a language the user invented + * cannot be found by name outside the field that created it. + */ + public getAllProjectLanguages(): ILangIndexEntry[] { + const seen = new Set(); + return [ + "SubjectLanguages", + "WorkingLanguages", + "metadataLanguages" + ].flatMap((key) => + this.getLanguagesOfField(key).filter((l) => { + if (l.iso639_3.length === 0 || seen.has(l.iso639_3)) return false; + seen.add(l.iso639_3); + return true; + }) + ); + } + private getFirstLanguageCodeAndName(fieldName: string): | { iso639_3: string; englishName: string; } | undefined { - const languages: string = this.properties.getTextStringOrEmpty(fieldName); - - if (languages.trim().length === 0) { - return undefined; // hasn't been defined yet, e.g. a new project - } - const firstCodeLangPair = languages.split(";")[0].trim(); - if (!firstCodeLangPair) { - return undefined; - } - const parts = firstCodeLangPair.split(":"); - // In the degenerate case in which there was no ":" in the incoming xml element, - // use the code as the name. - const langName = parts.length > 1 ? parts[1] : parts[0]; - return { - iso639_3: parts[0].trim().toLowerCase(), - englishName: langName.trim() - }; + return this.getLanguagesOfField(fieldName)[0]; // undefined for a new project } public getFirstSubjectLanguageCodeAndName(): | {