-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmorph.js
More file actions
407 lines (356 loc) · 13.7 KB
/
Copy pathmorph.js
File metadata and controls
407 lines (356 loc) · 13.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
// morph.js
// -----------------------------------------------------------------------------
// Parse a CATSS POS-morph code (e.g. "VAI-AAI3S") into a 'human-readable' version.
//
// Exported:
// parseMorphCode(str) → string
//
// Returns the gloss followed by " (original)" so callers can embed the
// result directly in the concordance panel. If parsing fails at any
// point, the original code is returned unchanged.
//
// https://ccat.sas.upenn.edu/gopher/text/religion/biblical/lxxmorph/*Morph-Coding
(function () {
// ── PoS tables ───────────────────────────────────────────────────────────
// Single-character codes with no morph following.
const POS_SIMPLE = {
C: "conj.",
X: "partic.",
I: "interj.",
M: "num.",
P: "prep.",
D: "adv.",
};
// Single-character codes that may take a second character and, for V, a
// third. The single-character form is consulted only when no longer
// match applies.
const POS_HEAD = {
N: "indecl. p. n.",
A: "adj.",
R: "art./pron.",
V: "v.",
};
// Two-character POS codes.
const POS_2 = {
N1: "1ª decl.",
N2: "2ª decl.",
N3: "3ª decl.",
A1: "adj.", // i.e. 'thematic adj.' ('-OS/-H/-ON pattern endings'), probably not worth specifying, it's the default case
A3: "3ª decl. adj.",
RA: "art.",
RD: "dem. pron.",
RI: "interr./indef. pron.",
RP: "pers./poss. pron.",
RR: "rel. pron.",
RX: "indef.rel./interr. pron.",
V1: "pres.",
V2: "-έω pres.",
V3: "-άω pres.",
V4: "-όω pres.",
V5: "-μι pres.",
V6: "-α-μι pres.",
V7: "-ε-μι pres.",
V8: "-ο-μι pres.",
V9: "εἰμί/εἶμι",
VA: "1st aor. act.",
VB: "2nd aor. act.",
VZ: "2nd aor. act. (irreg.)",
VH: "-η-μι aor.",
VE: "-ε-μι aor.",
VO: "-ο-μι aor.",
VC: "aor./fut. pass. (θ)",
VD: "aor./fut. pass. (s. θ)",
VV: "lab. aor./fut. pass.",
VS: "dent. aor./fut. pass.",
VQ: "gutt. aor./fut. pass.",
VX: "pf. act.",
VM: "pf. med.",
VP: "lab. pf. med.",
VT: "dent. pf. med.",
VK: "gutt. pf. med.",
VF: "fut.",
};
// Three-character POS codes. Only VF is given expanded 3rd-char forms;
// the actual POS codes have lots of nominal stem subclasses not represented here.
/* List for completeness:
N1A: "stem ending in -A (fem.)",
N1M: "masc. with nom. in -HS",
N1S: "stem in -H, nom. in -A (f.)",
N1T: "masc. with nom. in -AS",
N2N: "neuters (in -ON)",
N3D: "FOINIKI/S, -I/DOS",
N3E: "A)/NQOS, -OUS",
N3G: "MA/STIC, -IGOS",
N3H: "MNHSTH/R, -H=ROS",
N3I: "O)/NHSIS, -EWS",
N3K: "E(/LIC, -IKOS",
N3M: "A(/RMA, -ATOS",
N3N: "DAI/MWN, -ONOS",
N3P: "GRU/Y, GRUPO/S",
N3R: "A)LA/STWR, -OROS",
N3S: "SWKRA/THS, -OUS",
N3T: "A)KRO/THS, -HTOS",
N3U: "DRU=S, DRUO/S",
N3V: "BASILEU/S, -E/WS",
N3W: "A)GW/N, -W=NOS",
A1A: "-OS/-A/-ON",
A1B: "-OS/-OS/-ON",
A1C: "-OUS/-OUS/-OUN",
A1S: "nom. in -A, stem in -H",
A3E: "XARI/EIS, XARI/ENTOS",
A3H: "A)BLABH/S, A)BLABOU=S",
A3N: "A)/FRWN, A)/FRONOS",
A3U: "BAQU/S, BAQE/OS",
A3C: "irregular comparative"
*/
const POS_3 = {
VF2: "-ῶ fut.", // "liquid type future"
VF3: "-ῶ fut.", // "ἐλαύνω type future", rare, just 3 tokens: ἀναβῶ [ἀναβαίνω], ἀνελῶ [ἀναιρέω], ἀπελάσω [ἀπελαύνω]
VFX: "fut. pf.",
};
// ── Morph-field components ───────────────────────────────────────────────
const CASE = { N: "nom.", G: "gen.", D: "dat.", A: "acc.", V: "voc." };
const NUMBER = { S: "sg.", D: "du.", P: "pl." };
const GENDER = { M: "m.", F: "f.", N: "n." };
// Verb morph pieces
const V_TENSE = { P: "pres.", I: "ipf.", F: "fut.", A: "aor.", X: "pf.", Y: "plupf." }; // or "pqp.", "plqpf." for "plupf."?
const V_VOICE = { A: "act.", M: "med.", P: "pass." };
const V_MOOD = { I: "ind.", D: "ipv.", S: "subj.", O: "opt.", N: "inf.", P: "ptc." };
// Adjective degree
const A_DEGREE = { C: "comp.", S: "superl." };
// Person
const PERSON = { "1": "1ª", "2": "2ª", "3": "3ª" };
// ── Parsing ──────────────────────────────────────────────────────────────
function parseCaseNumberGender(s) {
// Three characters [NGDAV][SDP][MFN] → "nom. sg. m." etc.
// Two characters [NGDAV][SDP] → "nom. sg. m./n." (gender-collapsed).
if (s.length >= 3) {
const c = CASE[s[0]];
const n = NUMBER[s[1]];
const g = GENDER[s[2]];
if (c && n && g) return `${c} ${n} ${g}`;
}
if (s.length >= 2) {
const c = CASE[s[0]];
const n = NUMBER[s[1]];
if (c && n) return `${c} ${n} m./n.`;
}
return null;
}
// Structured version: returns {str, caseChar, numChar, genderStr} or null.
// genderStr is the *display* form: "m.", "f.", "n.", or "m./n.".
function parseCNG(s) {
if (s.length >= 3) {
const c = CASE[s[0]], n = NUMBER[s[1]], g = GENDER[s[2]];
if (c && n && g)
return { str: `${c} ${n} ${g}`, caseChar: s[0], numChar: s[1], genderStr: g };
}
if (s.length >= 2) {
const c = CASE[s[0]], n = NUMBER[s[1]];
if (c && n)
return { str: `${c} ${n} m./n.`, caseChar: s[0], numChar: s[1], genderStr: "m./n." };
}
return null;
}
// Parse a '+'-separated nominal morph like "NPN+APN", returning a single
// human-readable string with "; " between parts. After expanding each
// part individually, tries to merge nom.+acc. of the same number and
// (neuter-compatible) gender into "nom./acc.".
function expandNominalMorphParts(parts, p1) {
const parsed = [];
for (const part of parts) {
const p = parseCNG(part);
if (!p) return null;
// Adjective degree suffix (4th char) — only with 3-char CNG base
if (p1 === "A" && part.length >= 4 && part[3] in A_DEGREE) {
p.str += " " + A_DEGREE[part[3]];
}
parsed.push(p);
}
// Combine nom.+acc. when the number and gender match and the gender
// is neuter-compatible ("n." or "m./n.").
const used = new Set();
const result = [];
for (let i = 0; i < parsed.length; i++) {
if (used.has(i)) continue;
const pi = parsed[i];
if (pi.caseChar === "N" &&
(pi.genderStr === "n." || pi.genderStr === "m./n.")) {
// Look for a matching acc. later in the list
for (let j = i + 1; j < parsed.length; j++) {
if (used.has(j)) continue;
const pj = parsed[j];
if (pj.caseChar === "A" &&
pj.numChar === pi.numChar &&
pj.genderStr === pi.genderStr) {
result.push(pi.str.replace(/^nom\./, "nom./acc."));
used.add(i);
used.add(j);
break;
}
}
}
if (!used.has(i)) {
result.push(pi.str);
used.add(i);
}
}
return result.join("; ");
}
function parsePos(pos) {
// Returns the parsed PoS string, or null if unparseable.
// Applies the V.I (augmented) special rule.
if (!pos) return null;
const p1 = pos[0];
// Simple indeclinable POS with no morph
if (p1 in POS_SIMPLE) return POS_SIMPLE[p1];
// Head POS that may take further characters
if (!(p1 in POS_HEAD)) return null;
// Try 3-char match first (VF2, VF3, VFX)
if (pos.length >= 3 && pos in POS_3) {
return POS_3[pos];
}
// V.I special rule: first char V, third char I ⇒ augmented tense.
// Use the 2-char code (characters 1-2), prepend "augm. ", and rewrite
// any "pf." in the 2-char string to "plupf." (augmented perfect is
// the pluperfect). Does not fire for VFX (the 3-char lookup above
// wins) nor for codes whose third char isn't I.
if (p1 === "V" && pos.length >= 3 && pos[2] === "I") {
const base2 = pos.slice(0, 2);
if (base2 in POS_2) {
const s = POS_2[base2].split("pf.").join("plupf.");
return "augm. " + s;
}
// Fall through if base2 isn't a known 2-char code.
}
// Two-char match
if (pos.length >= 2 && pos.slice(0, 2) in POS_2) {
// If pos has extra characters beyond 2 that we don't recognise,
// we still return the 2-char match (the extra chars are a soft
// failure for POS parsing but not a fatal one).
return POS_2[pos.slice(0, 2)];
}
// Single-char fallback
return POS_HEAD[p1];
}
function parseMorph(morph, p1, posStr) {
// Returns the expanded morph string, or "" if nothing meaningful can
// be parsed. p1 is the first character of the POS code; posStr is
// the already-expanded gloss of the POS (used to suppress
// redundant tense/voice labels for verbs). Morph is only interpreted
// for p1 in {N, A, R, V}.
if (!morph) return "";
if (!"NARV".includes(p1)) return "";
if (p1 === "N" || p1 === "R") {
const cng = parseCaseNumberGender(morph);
return cng || "";
}
if (p1 === "A") {
// Case-number-gender in positions 1-3, optional degree in position 4.
const cng = parseCaseNumberGender(morph);
if (!cng) return "";
const parts = [cng];
if (morph.length >= 4 && morph[3] in A_DEGREE) {
parts.push(A_DEGREE[morph[3]]);
}
return parts.join(" ");
}
// p1 === "V"
if (morph.length < 3) return "";
let tense = V_TENSE[morph[0]];
let voice = V_VOICE[morph[1]];
const mood = V_MOOD [morph[2]];
if (!tense || !voice || !mood) return "";
// Redundancy elimination: if the expanded PoS already names the
// tense or voice, drop that label here. Two tripwires:
// • "aor./fut." in PoS (e.g. VC "aor./fut. pass. (θ-type)") means
// both aor. and fut. are candidates, so the morph's choice is
// informative — never drop.
// • "pf." matching inside "plupf." would be a false positive, but
// that can't happen for this data (an augmented pf. gets "plupf."
// in PoS and the morph will say "pf." — so the morph reinforces
// "pf." under the "plupf." parent).
const posForCheck = posStr || "";
const splitTenseAmbiguous = posForCheck.includes("aor./fut.");
if (!splitTenseAmbiguous && posForCheck.includes(tense)) {
tense = "";
}
if (posForCheck.includes(voice)) {
voice = "";
}
const out = [];
if (tense) out.push(tense);
if (voice) out.push(voice);
out.push(mood);
if (morph[2] === "P") {
// Participle: chars 4-6 are case-number-gender.
const cng = parseCaseNumberGender(morph.slice(3));
if (cng) out.push(cng);
} else {
// Finite: chars 4-5 are person + number.
if (morph.length >= 5) {
const per = PERSON[morph[3]];
const num = NUMBER[morph[4]];
if (per && num) out.push(per, num);
}
}
return out.join(" ");
}
// ── Public API ───────────────────────────────────────────────────────────
// Build the gloss for a single (non-combined) POS with its morph string,
// which may itself contain '+'.
function glossSinglePos(pos, morphRaw) {
const posStr = parsePos(pos);
if (!posStr) return null;
if (!morphRaw) return posStr;
const p1 = pos[0] || "";
let morphStr;
// '+'-separated morph for nominal types (N, A, R)
if (morphRaw.includes("+") && "NAR".includes(p1)) {
morphStr = expandNominalMorphParts(morphRaw.split("+"), p1);
} else {
morphStr = parseMorph(morphRaw, p1, posStr);
}
if (morphStr) return `${posStr}: ${morphStr}`;
return posStr;
}
function parseMorphCode(code, printcode=false) {
// Input format from the concordance: "VAI-AAI3S", "N1-DSF", "P",
// and now also combined forms:
// morph '+': "A1A-NPN+APN", "RR-GP+GPF"
// pos '+': "RP+X-NS", "C+X-", "C+D"
if (!code) return "";
const idx = code.indexOf("-");
const posRaw = idx < 0 ? code : code.slice(0, idx);
const morphRaw = idx < 0 ? "" : code.slice(idx + 1);
// ── Single POS (no '+' in pos) ──────────────────────────────────
if (!posRaw.includes("+")) {
const g = glossSinglePos(posRaw, morphRaw);
if (!g) return code;
return printcode? `${g} (${code})` : `${g}`;
}
// ── Combined POS (e.g. "RP+X", "C+D") ──────────────────────────
const posParts = posRaw.split("+");
const DECL = "NARV";
// Find which part (at most one) is declinable and gets the morph.
let declIdx = -1;
for (let i = 0; i < posParts.length; i++) {
if (posParts[i].length > 0 && DECL.includes(posParts[i][0])) {
declIdx = i;
break;
}
}
// Expand each part in order; morph attaches to the declinable one.
const out = [];
for (let i = 0; i < posParts.length; i++) {
const g = (i === declIdx)
? glossSinglePos(posParts[i], morphRaw)
: glossSinglePos(posParts[i], "");
if (!g) return code;
out.push(g);
}
return printcode? `${out.join("; ")} (${code})` : `${out.join("; ")}`;
}
// Expose on window
window.parseMorphCode = parseMorphCode;
})();