diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 9a270764..3d2789ed 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -27,6 +27,10 @@ jobs: run: | sudo dpkg -i gf-${GF_VERSION}-ubuntu-24.04.deb + - name: Czech regression tests + timeout-minutes: 5 + run: sh tests/czech/check.sh + - name: Build RGL run: | mkdir -p ${DEST} diff --git a/src/api/TryCze.gf b/src/api/TryCze.gf index 30a52365..9137de2c 100644 --- a/src/api/TryCze.gf +++ b/src/api/TryCze.gf @@ -1,6 +1,6 @@ --# -path=.:../czech:../common:../abstract:../prelude -resource TryCze = SyntaxCze, LexiconCze, ParadigmsCze -[mkAdv, mkDet,mkQuant]** +resource TryCze = ExtraCze, SyntaxCze, LexiconCze, ParadigmsCze -[mkAdv, mkDet,mkQuant]** open (P = ParadigmsCze) in { -- oper @@ -10,4 +10,3 @@ resource TryCze = SyntaxCze, LexiconCze, ParadigmsCze -[mkAdv, mkDet,mkQuant]** -- } ; } - diff --git a/src/czech/AdjectiveCze.gf b/src/czech/AdjectiveCze.gf index 38950f0b..1b8e391c 100644 --- a/src/czech/AdjectiveCze.gf +++ b/src/czech/AdjectiveCze.gf @@ -2,22 +2,32 @@ concrete AdjectiveCze of Adjective = CatCze ** open ResCze, Prelude in { lin - PositA a = adjFormsAdjective a ** {isPost = False} ; + PositA a = let ap = adjFormsAdjective a in ap ** {pred = longPredicate ap ; isPost = False} ; - AdAP ada ap = ap ** {s = \\g,n,c => ada.s ++ ap.s ! g ! n ! c} ; + AdAP ada ap = ap ** { + s = \\g,n,c => ada.s ++ ap.s ! g ! n ! c ; + pred = \\a => ada.s ++ ap.pred ! a + } ; ComplA2 a np = - let ap = adjFormsAdjective a + let ap = adjFormsAdjective a ; + compl = fullComplement a.c np.s np.prep in ap ** { - s = \\g,n,c => ap.s ! g ! n ! c ++ a.c.s ++ np.s ! a.c.c ; + s = \\g,n,c => ap.s ! g ! n ! c ++ compl ; + pred = \\agr => longPredicate ap ! agr ++ compl ; isPost = True ; } ; - UseA2 a = adjFormsAdjective a ** {isPost = True} ; + UseA2 a = let ap = adjFormsAdjective a in ap ** {pred = longPredicate ap ; isPost = True} ; - UseComparA a = adjFormsAdjective a ** {isPost = False} ; ---- TODO: this gives positive forms + UseComparA a = let ap = adjFormsAdjective a.compar in ap ** {pred = longPredicate ap ; isPost = False} ; + ComparA a np = AdvAP (UseComparA a) {s = "než" ++ np.s ! Nom} ; + AdjOrd ord = ord ** {pred = longPredicate ord ; isPost = False} ; - AdvAP ap adv = ap ** {s = \\g,n,c => ap.s ! g ! n ! c ++ adv.s; isPost = True} ; + AdvAP ap adv = ap ** { + s = \\g,n,c => ap.s ! g ! n ! c ++ adv.s ; + pred = \\a => ap.pred ! a ++ adv.s ; isPost = True + } ; } diff --git a/src/czech/AdverbCze.gf b/src/czech/AdverbCze.gf index b6b18631..0514dbdb 100644 --- a/src/czech/AdverbCze.gf +++ b/src/czech/AdverbCze.gf @@ -3,11 +3,11 @@ concrete AdverbCze of Adverb = CatCze ** lin PrepNP prep np = { - s = prep.s ++ np.prep ! prep.c + s = fullComplement prep np.s np.prep } ; SubjS subj s = { - s = subj.s ++ s.s + s = (frontSentence subj.s s).s } ; } diff --git a/src/czech/AllCze.gf b/src/czech/AllCze.gf index 6d25b824..69364718 100644 --- a/src/czech/AllCze.gf +++ b/src/czech/AllCze.gf @@ -2,6 +2,7 @@ concrete AllCze of AllCzeAbs = LangCze, - ExtendCze + ExtendCze, + ExtraCze ; diff --git a/src/czech/AllCzeAbs.gf b/src/czech/AllCzeAbs.gf index a18b6447..02387c34 100644 --- a/src/czech/AllCzeAbs.gf +++ b/src/czech/AllCzeAbs.gf @@ -2,6 +2,7 @@ abstract AllCzeAbs = Lang, - Extend + Extend, + ExtraCzeAbs ; diff --git a/src/czech/CatCze.gf b/src/czech/CatCze.gf index 20c560e3..331c22f4 100644 --- a/src/czech/CatCze.gf +++ b/src/czech/CatCze.gf @@ -8,37 +8,52 @@ concrete CatCze of Cat = Phr = {s : Str} ; Utt = {s : Str} ; - S = {s : Str} ; - Cl = {subj,clit,compl : Str ; verb : VerbForms ; a : Agr} ; + S = ResCze.Sentence ; + Cl = {subj,clit,compl : Str ; verb : VerbForms ; a : Agr ; isDrop,clitPresent : Bool} ; Comp = {s : Agr => Str} ; - QS = {s : Str} ; ---- TODO: indirect questions - QCl = {subj,clit,compl : Str ; verb : VerbForms ; a : Agr} ; -- = Cl ---- check if enough - IAdv = {s : Str} ; + QS = {s,ind : Str} ; + QCl = {q,subj,clit,compl : Str ; verb : VerbForms ; a : Agr ; yesNo : Bool} ; + IAdv, IComp = {s : Str} ; + IP = {s : Case => Str ; a : Agr} ; + IDet = Determiner ; + IQuant = Adjective ; + Imp = {s : Bool => Agr => Str} ; RS = {s : Agr => Str} ; RCl = {subj,clit,compl : Agr => Str ; verb : VerbForms} ; ---- RAgr with composite RP RP = AdjForms ; - VP = {verb : VerbForms ; clit,compl : Agr => Str} ; ---- more fields probably needed - VPSlash = {verb : VerbForms ; clit,compl : Agr => Str ; c : ComplementCase ; ind : Agr => Str} ; -- ind : incorporated indirect object, rendered after the object slot + -- clitPresent records an overt clitic in this domain, not NP eligibility. + VP = {verb : VerbForms ; clit,compl : Agr => Str ; clitPresent : Bool} ; ---- more fields probably needed + -- clit/clitAfter surround the open slot in the eventual clitic cluster. + -- Its presence flag covers both sides of the slot. + VPSlash = {verb : VerbForms ; clit,clitAfter,compl : Agr => Str ; clitPresent : Bool ; c : ComplementCase ; ind : Agr => Str} ; -- ind : incorporated second object, rendered after the object slot V = ResCze.VerbForms ; V2 = ResCze.VerbForms ** {c : ComplementCase} ; V3 = ResCze.VerbForms ** {c,c2 : ComplementCase} ; -- c : direct object, c2 : indirect object - VS,VQ,VV = ResCze.VerbForms ; + VS,VQ = ResCze.VerbForms ; + VV = ResCze.VerbForms ** {isAux : Bool} ; - A = ResCze.AdjForms ; - AP = ResCze.Adjective ** {isPost : Bool} ; -- {s : Gender => Number => Case => Str} - A2 = ResCze.AdjForms ** {c : ComplementCase} ; + A = ResCze.DegreeForms ; + AP = ResCze.Adjective ** {pred : Agr => Str ; isPost : Bool} ; + A2 = ResCze.DegreeForms ** {c : ComplementCase} ; AdA = {s : Str} ; N = ResCze.NounForms ; CN = ResCze.Noun ; -- {s : Number => Case => Str ; g : Gender} - NP = {s,clit,prep : Case => Str ; a : Agr ; hasClit : Bool} ; -- clit,prep differ for pronouns + -- Object-clitic eligibility and subject omission are independent. + -- Extend.ProDrop selects isDrop; clit ! Nom retains its empty constituent. + -- Modifiers restore full forms. s and prep are always available for strong use. + -- m controls NP modifiers; a controls the clause. Scale nouns can differ. + -- A modified pronoun can keep its pronominal head without a weak form. + NP = NPForms ** {clit : Case => Str ; a : Agr ; m : ModifierAgr ; hasClit,isDrop,isPron : Bool} ; PN = {s : Case => Str ; g : Gender} ; + Ord = Adjective ; Det = Determiner ; -- {s : Gender => Case => Str ; size : NumSize} ; -- can contain a numeral, therefore NumSize Quant = {s : Gender => Number => Case => Str} ; -- same as AP + Predet = Adjective ** {postPron : Bool} ; Num = Determiner ; Card = Determiner ; -- {s : Gender => Case => Str ; size : NumSize} ; Pron = PronForms ** {poss : DemPronForms} ; diff --git a/src/czech/ConjunctionCze.gf b/src/czech/ConjunctionCze.gf index 11a45f37..3c42fd71 100644 --- a/src/czech/ConjunctionCze.gf +++ b/src/czech/ConjunctionCze.gf @@ -3,9 +3,9 @@ concrete ConjunctionCze of Conjunction = CatCze ** lincat [Adv] = {s1,s2 : Str} ; - [AP] = {s1,s2 : Gender => Number => Case => Str ; isPost : Bool} ; - [NP] = {s1,s2,prep1,prep2 : Case => Str ; a : Agr} ; - [S] = {s1,s2 : Str} ; + [AP] = {s1,s2 : Gender => Number => Case => Str ; pred1,pred2 : Agr => Str ; isPost : Bool} ; + [NP] = {s1,s2,prep1,prep2 : Case => Str ; a : Agr ; m : ModifierAgr} ; + [S] = {s1 : Sentence ; s2 : Str} ; [RS] = {s1,s2 : Agr => Str} ; lin @@ -13,27 +13,28 @@ concrete ConjunctionCze of Conjunction = CatCze ** ConsAdv = consrSS comma ; BaseAP x y = twoTable3 Gender Number Case x y - ** {isPost = orB x.isPost y.isPost} ; ---- should be so in Pol too + ** {pred1 = x.pred ; pred2 = y.pred ; isPost = orB x.isPost y.isPost} ; ConsAP x xs = consrTable3 Gender Number Case comma x xs - ** {isPost = orB x.isPost xs.isPost} ; + ** {pred1 = \\a => x.pred ! a ++ comma ++ xs.pred1 ! a ; pred2 = xs.pred2 ; isPost = orB x.isPost xs.isPost} ; + -- A shared preposed modifier agrees with the first conjunct. BaseNP x y = { s1 = x.s ; s2 = y.s ; prep1 = x.prep ; prep2 = y.prep ; - a = y.a + a = y.a ; m = x.m } ; -- clitics disappear ---- Agr TODO ConsNP x xs = { s1 = \\c => x.s ! c ++ comma ++ xs.s1 ! c ; s2 = xs.s2 ; prep1 = \\c => x.prep ! c ++ comma ++ xs.prep1 ! c ; prep2 = xs.prep2 ; - a = xs.a ---- + a = xs.a ; m = x.m ---- } ; - BaseS = twoSS ; - ConsS = consrSS comma ; + BaseS x y = {s1 = x ; s2 = y.s} ; + ConsS x xs = {s1 = appendSentence x (comma ++ xs.s1.s) ; s2 = xs.s2} ; BaseRS = twoTable Agr ; ConsRS = consrTable Agr comma ; @@ -41,16 +42,19 @@ concrete ConjunctionCze of Conjunction = CatCze ** ConjAdv = conjunctDistrSS ; ConjAP conj xs = conjunctDistrTable3 Gender Number Case conj xs - ** {isPost = xs.isPost} ; + ** {pred = \\a => conj.s1 ++ xs.pred1 ! a ++ conj.s2 ++ xs.pred2 ! a ; isPost = xs.isPost} ; - ConjNP conj xs = { - s,clit = \\c => conj.s1 ++ xs.s1 ! c ++ conj.s2 ++ xs.s2 ! c ; - prep = \\c => conj.s1 ++ xs.prep1 ! c ++ conj.s2 ++ xs.prep2 ! c ; + ConjNP conj xs = + let s : Case => Str = \\c => conj.s1 ++ xs.s1 ! c ++ conj.s2 ++ xs.s2 ! c ; + prep : Case => Str = \\c => conj.s1 ++ xs.prep1 ! c ++ conj.s2 ++ xs.prep2 ! c + in npForms s prep ** { + clit = s ; a = xs.a ; ---- dep. on conj as well - hasClit = False ; + m = xs.m ; + hasClit = False ; isDrop = False ; isPron = False ; } ; - ConjS = conjunctDistrSS ; + ConjS conj xs = prefixSentence conj.s1 (appendSentence xs.s1 (conj.s2 ++ xs.s2)) ; ConjRS = conjunctDistrTable Agr ; } diff --git a/src/czech/ExtendCze.gf b/src/czech/ExtendCze.gf index 97b9db43..d406f5f8 100644 --- a/src/czech/ExtendCze.gf +++ b/src/czech/ExtendCze.gf @@ -1,8 +1,11 @@ concrete ExtendCze of Extend = CatCze ** ExtendFunctor - [ - ReflPossPron + RNP, RNPList, ReflRNP, ReflPron, ReflPoss, PredetRNP, + ConjRNP, Base_rr_RNP, Base_nr_RNP, Base_rn_RNP, Cons_rr_RNP, Cons_nr_RNP, + ReflPossPron, ProDrop, UttAccNP, UttDatNP, + iFem_Pron, youFem_Pron, weFem_Pron, youPlFem_Pron, + theyFem_Pron, theyNeutr_Pron, youPolFem_Pron, youPolPlFem_Pron ---- constant not found (yet) - ,youPolFem_Pron ,UttVPShort ,UttAccIP ,UttDatIP @@ -31,10 +34,113 @@ concrete ExtendCze of Extend = CatCze ** with (Grammar = GrammarCze) ** open - ResCze + ResCze, Prelude, (S = SyntaxCze), (P = ParadigmsCze) in { -lin ReflPossPron = justDemPronFormsAdjective reflPossessivePron ; +lincat + RNP = BoundNPForms ** {m : RNPHead ; isPron : Bool} ; + RNPList = {s1,s2,prep1,prep2 : Agr => Case => Str ; m : RNPHead} ; + +param + RNPHead = AntecedentHead | FixedHead ModifierAgr ; + +lin + -- Standalone oblique NPs use full forms, never clitics or prepositional forms. + UttAccNP np = {s = np.s ! Acc} ; + UttDatNP np = {s = np.s ! Dat} ; + + -- Retain full forms for objects, coordination and NP modifiers. + ProDrop pron = pron ** {isDrop = True} ; + + iFem_Pron = P.genderPron Fem S.i_Pron ; + youFem_Pron = P.genderPron Fem S.youSg_Pron ; + weFem_Pron = P.genderPron Fem S.we_Pron ; + youPlFem_Pron = P.genderPron Fem S.youPl_Pron ; + theyFem_Pron = P.genderPron Fem S.they_Pron ; + theyNeutr_Pron = P.genderPron Neutr S.they_Pron ; + youPolFem_Pron = P.genderPron Fem S.youPol_Pron ; + youPolPlFem_Pron = P.genderPron Fem S.youPl_Pron ; + + ReflPossPron = justDemPronFormsAdjective reflPossessivePron ; + + -- RNPs are full noun phrases. ReflPron consequently uses sebe/sobě, + -- including in coordination, rather than a lexical reflexive clitic. + ReflPron = + let s : Case => Str = table { + Nom | ResCze.Voc => nonExist ; Gen | Acc => "sebe" ; + Dat | Loc => "sobě" ; Ins => "sebou" + } + in boundNPForms (\\_ => npForms s s) ** { + m = AntecedentHead ; isPron = True + } ; + ReflPoss num cn = fullRNP (S.mkNP (lin Quant (justDemPronFormsAdjective reflPossessivePron)) + ) ; + ReflRNP vps rnp = vps ** { + clit = \\a => vps.clit ! a ++ vps.clitAfter ! a ; + compl = \\a => vps.compl ! a ++ + fullComplement vps.c (rnp.s ! a) (rnp.prep ! a) ++ vps.ind ! a + } ; + PredetRNP pred rnp = rnp ** boundNPForms (\\a => + predetNPForms (andB pred.postPron rnp.isPron) + (\\c => predetForm pred (rnpAgr rnp.m a) c) (rnpForms rnp a)) ; + AdvRNP np p rnp = boundNPForms (\\a => + appendNPForms np (fullComplement p (rnp.s ! a) (rnp.prep ! a))) ** { + m = FixedHead np.m ; isPron = np.isPron + } ; + AdvRVP vp p rnp = vp ** { + compl = \\a => vp.compl ! a ++ fullComplement p (rnp.s ! a) (rnp.prep ! a) + } ; + AdvRAP ap p rnp = ap ** { + s = \\g,n,c => ap.s ! g ! n ! c ++ fullComplement p (rnp.s ! Ag g n P3) (rnp.prep ! Ag g n P3) ; + pred = \\a => ap.pred ! a ++ fullComplement p (rnp.s ! a) (rnp.prep ! a) ; isPost = True + } ; + Base_rr_RNP x y = baseRNP (lin RNP x) (lin RNP y) ; + Base_nr_RNP x y = baseRNP (fullRNP (lin NP x)) (lin RNP y) ; + Base_rn_RNP x y = baseRNP (lin RNP x) (fullRNP (lin NP y)) ; + Cons_rr_RNP x xs = consRNP (lin RNP x) (lin RNPList xs) ; + Cons_nr_RNP x xs = consRNP (fullRNP (lin NP x)) (lin RNPList xs) ; + ConjRNP conj xs = boundNPForms (\\a => npForms + (\\c => conj.s1 ++ xs.s1 ! a ! c ++ conj.s2 ++ xs.s2 ! a ! c) + (\\c => conj.s1 ++ xs.prep1 ! a ! c ++ conj.s2 ++ xs.prep2 ! a ! c)) ** { + m = xs.m ; isPron = False + } ; + +oper + BoundNPForms : Type = { + s,prep,before,prepBefore : Agr => Case => Str ; after : Agr => Str + } ; + -- Placement uses the same NP operations at each antecedent agreement. + boundNPForms : (Agr => NPForms) -> BoundNPForms = \forms -> { + s = \\a => (forms ! a).s ; prep = \\a => (forms ! a).prep ; + before = \\a => (forms ! a).before ; prepBefore = \\a => (forms ! a).prepBefore ; + after = \\a => (forms ! a).after + } ; + rnpForms : RNP -> Agr -> NPForms = \rnp,a -> { + s = rnp.s ! a ; prep = rnp.prep ! a ; + before = rnp.before ! a ; prepBefore = rnp.prepBefore ! a ; + after = rnp.after ! a + } ; + -- Ordinary NPs and possessed heads have their own modifier agreement. + fullRNP : S.NP -> RNP = \np -> lin RNP (boundNPForms (\\_ => np) ** { + m = FixedHead np.m ; isPron = np.isPron + }) ; + -- A reflexive inherits gender and number, but its own complement position + -- selects case: pět dětí miluje sebe všechny, not sebe všech. + rnpAgr : RNPHead -> Agr -> ModifierAgr = \head,a -> case head of { + AntecedentHead => case a of { + AgQuant g => Mod g Pl ; _ => modifierAgr a + } ; + FixedHead m => m + } ; + -- As for ordinary NPs, preposed modifiers agree with the first conjunct. + baseRNP : RNP -> RNP -> RNPList = \x,y -> lin RNPList { + s1 = x.s ; s2 = y.s ; prep1 = x.prep ; prep2 = y.prep ; m = x.m + } ; + consRNP : RNP -> RNPList -> RNPList = \x,xs -> xs ** { + s1 = \\a,c => x.s ! a ! c ++ SOFT_BIND ++ "," ++ xs.s1 ! a ! c ; + prep1 = \\a,c => x.prep ! a ! c ++ SOFT_BIND ++ "," ++ xs.prep1 ! a ! c ; + m = x.m + } ; } diff --git a/src/czech/ExtraCze.gf b/src/czech/ExtraCze.gf new file mode 100644 index 00000000..e158d542 --- /dev/null +++ b/src/czech/ExtraCze.gf @@ -0,0 +1,23 @@ +concrete ExtraCze of ExtraCzeAbs = CatCze ** open ResCze, Prelude, (V = VerbCze) in { +lin + -- The adjective agrees with the subject; ordinary slash completion + -- supplies the object's case, prepositional form and clitic placement. + SlashV2AP v ap = (V.SlashV2a v) ** {compl = ap.pred} ; + + -- Dative experiencer, with agreement controlled by the predicative NP. + -- je mi pět let / jsou mi dva roky / mému synovi je pět let + DativeCopulaCl experiencer predicate = { + subj = case experiencer.hasClit of {True => [] ; False => experiencer.s ! Dat} ; + clit = case experiencer.hasClit of {True => experiencer.clit ! Dat ; False => []} ; + compl = predicate.s ! Nom ; verb = copulaVerbForms ; a = predicate.a ; + isDrop = experiencer.hasClit ; clitPresent = experiencer.hasClit + } ; + + DativeCopulaQCl experiencer predicate = { + q = predicate.s ! Nom ; + subj = case experiencer.hasClit of {True => [] ; False => experiencer.s ! Dat} ; + clit = case experiencer.hasClit of {True => experiencer.clit ! Dat ; False => []} ; + compl = [] ; verb = copulaVerbForms ; a = predicate.a ; + yesNo = False + } ; +} diff --git a/src/czech/ExtraCzeAbs.gf b/src/czech/ExtraCzeAbs.gf new file mode 100644 index 00000000..715d3edd --- /dev/null +++ b/src/czech/ExtraCzeAbs.gf @@ -0,0 +1,11 @@ +-- Czech constructions beyond the portable Syntax API. +abstract ExtraCzeAbs = Cat ** { +fun + -- Subject-oriented secondary adjective: mít rád (něco). + -- The object remains open for ComplSlash or Extend.ReflRNP. + SlashV2AP : V2 -> AP -> VPSlash ; + -- First argument: dative experiencer. Second: nominative expression, + -- which controls agreement: je mi pět let / jsou mi dva roky. + DativeCopulaCl : NP -> NP -> Cl ; + DativeCopulaQCl : NP -> IP -> QCl ; +} diff --git a/src/czech/IdiomCze.gf b/src/czech/IdiomCze.gf index b7d61c65..f9e6706f 100644 --- a/src/czech/IdiomCze.gf +++ b/src/czech/IdiomCze.gf @@ -1,8 +1,10 @@ concrete IdiomCze of Idiom = CatCze ** open Prelude, ResCze in { lin + ImpPl1 vp = {s = vp.verb.imppl1 ++ vp.clit ! Ag (Masc Anim) Pl P1 ++ vp.compl ! Ag (Masc Anim) Pl P1} ; ImpP3 np vp = { - s = "nechť" ++ np.s ! Nom ++ vp.clit ! np.a ++ + s = "nechť" ++ vp.clit ! np.a ++ + (case np.isDrop of {True => np.clit ! Nom ; False => np.s ! Nom}) ++ verbAgr vp.verb np.a True ++ vp.compl ! np.a } ; @@ -10,7 +12,8 @@ lin subj = [] ; clit = vp.clit ! agr ; compl = vp.compl ! agr ; - verb = vp.verb ; + verb = vp.verb ; clitPresent = vp.clitPresent ; + isDrop = True ; a = agr } ; @@ -18,14 +21,17 @@ lin subj = [] ; clit = vp.clit ! agr ; compl = vp.compl ! agr ; - verb = vp.verb ; + verb = vp.verb ; clitPresent = vp.clitPresent ; + isDrop = True ; a = agr } ; ExistNP np = { + clitPresent = False ; subj, clit = [] ; compl = np.s ! Nom ; verb = iii_kupovatVerbForms "existovat" ; + isDrop = True ; a = np.a } ; diff --git a/src/czech/LexiconCze.gf b/src/czech/LexiconCze.gf index cec51cbd..7139573a 100644 --- a/src/czech/LexiconCze.gf +++ b/src/czech/LexiconCze.gf @@ -6,6 +6,11 @@ concrete LexiconCze of Lexicon = in { lin + child_N = (kureN "dítě") ** { + sgen = "dítěte" ; sdat,sloc = "dítěti" ; sins = "dítětem" ; + pnom,pacc = "děti" ; pgen = "dětí" ; pdat = "dětem" ; ploc = "dětech" ; pins = "dětmi" ; gPl = Fem + } ; + year_N = (hradN "rok") ** {sgen,svoc = "roku" ; sloc = "roce" ; pgen = "let" ; pdat = "letům" ; ploc = "letech" ; pins = "lety"} ; boy_N = declPAN "kluk" ; man_N = declMUZ "muž" ; teacher_N = declMUZ "učitel" ; @@ -18,7 +23,7 @@ concrete LexiconCze of Lexicon = machine_N = declSTROJ "stroj" ; woman_N = declZENA "žena" ; - school_N = declZENA "škola" ; ---- + school_N = zenaN "škola" ; skirt_N = declRUZE "sukně"; street_N = declRUZE "ulice" ; rose_N = declRUZE "růže" ; @@ -28,32 +33,61 @@ concrete LexiconCze of Lexicon = bone_N = declKOST "kost" ; village_N = declKOST "ves" ; ---- - city_N = declMESTO "město" ; - apple_N = declMESTO "jablko" ; ---- + city_N = (mestoN "město") ** {sloc = "městě"} ; + apple_N = declMESTO "jablko" ** {pgen = "jablek" ; ploc = "jablkách"} ; sea_N = declMORE "moře" ; airport_N = declMORE "letiště" ; chicken_N = declKURE "kuře" ; house_N = declSTAVENI "stavení" ; --- building, house station_N = declSTAVENI "nádraží" ; - young_A = mkA "mladý" ; - old_A = mkA "starý" ; - good_A = mkA "dobrý" ; - bad_A = mkA "špatný" ; - beautiful_A = mkA "krásný" ; - clean_A = mkA "čistý" ; - dirty_A = mkA "špinavý" ; + young_A = mkA "mladý" "mladší" ; + old_A = mkA "starý" "starší" ; + good_A = mkA "dobrý" "lepší" ; + bad_A = mkA "špatný" "horší" ; + beautiful_A = mkA "krásný" "krásnější" ; + clean_A = mkA "čistý" "čistší" ; + dirty_A = mkA "špinavý" "špinavější" ; - white_A = mkA "bílý" ; - black_A = mkA "černý" ; - red_A = mkA "červený" ; - brown_A = mkA "hnědý" ; - blue_A = mkA "modrý" ; - green_A = mkA "zelený" ; - yellow_A = mkA "žlutý" ; + white_A = mkA "bílý" "bělejší" ; + black_A = mkA "černý" "černější" ; + red_A = mkA "červený" "červenější" ; + brown_A = mkA "hnědý" "hnědší" ; + blue_A = mkA "modrý" "modřejší" ; + green_A = mkA "zelený" "zelenější" ; + yellow_A = mkA "žlutý" "žlutější" ; - buy_V2 = mkV2 (iii_kupovatVerbForms "kupovat") ; - love_V2 = mkV2 (iii_kupovatVerbForms "milovat") ; + buy_V2 = mkV2 (kupovatV "kupovat") ; + love_V2 = mkV2 (kupovatV "milovat") ; + + drink_V2 = mkV2 (krytV "pít") ; + eat_V2 = mkV2 (mkV "jíst" "jím" "jíš" "jí" "jíme" "jíte" "jedí" "jedl" "jedli" "jez" "jezme" "jezte") ; + read_V2 = mkV2 (mkV "číst" "čtu" "čteš" "čte" "čteme" "čtete" "čtou" "četl" "četli" "čti" "čtěme" "čtěte") ; + write_V2 = mkV2 (mkV "psát" "píši" "píšeš" "píše" "píšeme" "píšete" "píší" "psal" "psali" "piš" "pišme" "pište") ; + wait_V2 = mkV2 (mkV "čekat" "čekám" "čekáš" "čeká" "čekáme" "čekáte" "čekají" "čekal" "čekali" "čekej" "čekejme" "čekejte") (mkPrep "na" accusative) ; + play_V = mkV "hrát" "hraji" "hraješ" "hraje" "hrajeme" "hrajete" "hrají" "hrál" "hráli" "hraj" "hrajme" "hrajte" ; + run_V = mkV "běžet" "běžím" "běžíš" "běží" "běžíme" "běžíte" "běží" "běžel" "běželi" "běž" "běžme" "běžte" ; + sit_V = mkV "sedět" "sedím" "sedíš" "sedí" "sedíme" "sedíte" "sedí" "seděl" "seděli" "seď" "seďme" "seďte" ; + sleep_V = mkV "spát" "spím" "spíš" "spí" "spíme" "spíte" "spí" "spal" "spali" "spi" "spěme" "spěte" ; + swim_V = mkV "plavat" "plavu" "plaveš" "plave" "plaveme" "plavete" "plavou" "plaval" "plavali" "plav" "plavme" "plavte" ; + walk_V = mkV "chodit" "chodím" "chodíš" "chodí" "chodíme" "chodíte" "chodí" "chodil" "chodili" "choď" "choďme" "choďte" ; + go_V = mkV "jít" "jdu" "jdeš" "jde" "jdeme" "jdete" "jdou" "šel" "šli" "jdi" "jděme" "jděte" ; + know_V2 = mkV2 (mkV "znát" "znám" "znáš" "zná" "známe" "znáte" "znají" "znal" "znali" "znej" "znejme" "znejte") ; + know_VS = mkVS knowV ; + know_VQ = mkVQ knowV ; + today_Adv = mkAdv "dnes" ; + + beer_N = mestoN "pivo" ; + bread_N = mkN "chléb" "chleba" mascInanimate ; + fish_N = zenaN "ryba" ; + milk_N = (mestoN "mléko") ** {sloc = "mléce" ; ploc = "mlékách"} ; + salt_N = (kostN "sol") ** {snom,sacc = "sůl" ; pdat = "solím" ; ploc = "solích" ; pins = "solemi"} ; + water_N = zenaN "voda" ; + wine_N = mestoN "víno" ; + cold_A = mkA "studený" "studenější" ; + warm_A = mkA "teplý" "teplejší" ; + +oper + knowV : V = mkV "vědět" "vím" "víš" "ví" "víme" "víte" "vědí" "věděl" "věděli" "věz" "vězme" "vězte" ; } - diff --git a/src/czech/MarkupCze.gf b/src/czech/MarkupCze.gf index 93843cd9..e2b0b492 100644 --- a/src/czech/MarkupCze.gf +++ b/src/czech/MarkupCze.gf @@ -1,12 +1,12 @@ --# -path=.:../abstract:../common -concrete MarkupCze of Markup = CatCze, MarkHTMLX ** open ResCze in { +concrete MarkupCze of Markup = CatCze, MarkHTMLX ** open ResCze, Prelude in { lin MarkupCN m cn = cn ** {s = \\n,c => appMark m (cn.s ! n ! c)} ; -- s, clit and prep are alternative surface forms, so each is marked; - -- but clit ! Nom is the pro-drop subject, empty for every pronoun, + -- but clit ! Nom is the form selected by ProDrop, empty for every pronoun, -- and marking it up would leave the tags around nothing MarkupNP m np = np ** { s = \\c => appMark m (np.s ! c) ; @@ -14,12 +14,27 @@ lin Nom => np.clit ! Nom ; _ => appMark m (np.clit ! c) } ; - prep = \\c => appMark m (np.prep ! c) + prep = \\c => appMark m (np.prep ! c) ; + -- Later predetermination can split the original marked constituent. + before = \\c => appMark m (np.before ! c) ; + prepBefore = \\c => appMark m (np.prepBefore ! c) ; + -- Accept empty wrappers to avoid an NP parameter used only by markup. + after = appMark m np.after } ; - MarkupAP m ap = ap ** {s = \\g,n,c => appMark m (ap.s ! g ! n ! c)} ; + MarkupAP m ap = ap ** { + s = \\g,n,c => appMark m (ap.s ! g ! n ! c) ; + pred = \\a => appMark m (ap.pred ! a) + } ; MarkupAdv m adv = {s = appMark m adv.s} ; - MarkupS m s = {s = appMark m s.s} ; + -- The complete s and fronted orders each use one wrapper. Later fronting + -- composes separately marked clitic/body pieces; their wrappers remain + -- separate even when the pieces become adjacent. + MarkupS m s = s ** { + s = appMark m s.s ; fronted = appMark m s.fronted ; + clit = case s.clitPresent of {True => appMark m s.clit ; False => s.clit} ; + body = appMark m s.body + } ; MarkupUtt m utt = {s = appMark m utt.s} ; MarkupPhr m phr = {s = appMark m phr.s} ; MarkupText m txt = {s = appMark m txt.s} ; diff --git a/src/czech/MissingCze.gf b/src/czech/MissingCze.gf index be4b9cb2..f6a91735 100644 --- a/src/czech/MissingCze.gf +++ b/src/czech/MissingCze.gf @@ -5,62 +5,32 @@ oper AAnter : Ant = notYet "AAnter" ; oper AdAdv : AdA -> Adv -> Adv = notYet "AdAdv" ; oper AdNum : AdN -> Card -> Card = notYet "AdNum" ; oper AdVVP : AdV -> VP -> VP = notYet "AdVVP" ; -oper AdjOrd : Ord -> AP = notYet "AdjOrd" ; oper AdnCAdv : CAdv -> AdN = notYet "AdnCAdv" ; -oper AdvIAdv : IAdv -> Adv -> IAdv = notYet "AdvIAdv" ; -oper AdvIP : IP -> Adv -> IP = notYet "AdvIP" ; oper AdvSlash : ClSlash -> Adv -> ClSlash = notYet "AdvSlash" ; oper CAdvAP : CAdv -> AP -> NP -> AP = notYet "CAdvAP" ; oper CleftAdv : Adv -> S -> Cl = notYet "CleftAdv" ; oper CleftNP : NP -> RS -> Cl = notYet "CleftNP" ; -oper CompIAdv : IAdv -> IComp = notYet "CompIAdv" ; -oper CompIP : IP -> IComp = notYet "CompIP" ; -oper ComparA : A -> NP -> AP = notYet "ComparA" ; oper ComparAdvAdj : CAdv -> A -> NP -> Adv = notYet "ComparAdvAdj" ; oper ComparAdvAdjS : CAdv -> A -> S -> Adv = notYet "ComparAdvAdjS" ; oper ComplN2 : N2 -> NP -> CN = notYet "ComplN2" ; oper ComplN3 : N3 -> NP -> N2 = notYet "ComplN3" ; oper ComplVA : VA -> AP -> VP = notYet "ComplVA" ; -oper ComplVQ : VQ -> QS -> VP = notYet "ComplVQ" ; -oper ComplVS : VS -> S -> VP = notYet "ComplVS" ; -oper ComplVV : VV -> VP -> VP = notYet "ComplVV" ; oper DetNP : Det -> NP = notYet "DetNP" ; -oper DetQuantOrd : Quant -> Num -> Ord -> Det = notYet "DetQuantOrd" ; -oper EmbedQS : QS -> SC = notYet "EmbedQS" ; -oper EmbedS : S -> SC = notYet "EmbedS" ; -oper EmbedVP : VP -> SC = notYet "EmbedVP" ; oper ExistIP : IP -> QCl = notYet "ExistIP" ; -oper ExistNP : NP -> Cl = notYet "ExistNP" ; oper FunRP : Prep -> NP -> RP -> RP = notYet "FunRP" ; -oper GenericCl : VP -> Cl = notYet "GenericCl" ; -oper IdetCN : IDet -> CN -> IP = notYet "IdetCN" ; -oper IdetIP : IDet -> IP = notYet "IdetIP" ; -oper IdetQuant : IQuant -> Num -> IDet = notYet "IdetQuant" ; -oper ImpPl1 : VP -> Utt = notYet "ImpPl1" ; -oper ImpVP : VP -> Imp = notYet "ImpVP" ; -oper ImpersCl : VP -> Cl = notYet "ImpersCl" ; oper OrdDigits : Digits -> Ord = notYet "OrdDigits" ; oper OrdNumeral : Numeral -> Ord = notYet "OrdNumeral" ; -oper OrdSuperl : A -> Ord = notYet "OrdSuperl" ; oper PPartNP : NP -> V2 -> NP = notYet "PPartNP" ; -oper PassV2 : V2 -> VP = notYet "PassV2" ; oper PositAdvAdj : A -> Adv = notYet "PositAdvAdj" ; -oper PossPron : Pron -> Quant = notYet "PossPron" ; oper PredSCVP : SC -> VP -> Cl = notYet "PredSCVP" ; -oper PredetNP : Predet -> NP -> NP = notYet "PredetNP" ; -oper PrepIP : Prep -> IP -> IAdv = notYet "PrepIP" ; oper ProgrVP : VP -> VP = notYet "ProgrVP" ; -oper QuestIAdv : IAdv -> Cl -> QCl = notYet "QuestIAdv" ; -oper QuestIComp : IComp -> NP -> QCl = notYet "QuestIComp" ; oper QuestSlash : IP -> ClSlash -> QCl = notYet "QuestSlash" ; -oper QuestVP : IP -> VP -> QCl = notYet "QuestVP" ; oper ReflA2 : A2 -> AP = notYet "ReflA2" ; oper ReflVP : VPSlash -> VP = notYet "ReflVP" ; oper RelCl : Cl -> RCl = notYet "RelCl" ; oper RelNP : NP -> RS -> NP = notYet "RelNP" ; oper RelSlash : RP -> ClSlash -> RCl = notYet "RelSlash" ; oper SentAP : AP -> SC -> AP = notYet "SentAP" ; -oper SentCN : CN -> SC -> CN = notYet "SentCN" ; oper SlashPrep : Cl -> Prep -> ClSlash = notYet "SlashPrep" ; oper SlashV2A : V2A -> AP -> VPSlash = notYet "SlashV2A" ; oper SlashV2Q : V2Q -> QS -> VPSlash = notYet "SlashV2Q" ; @@ -70,29 +40,11 @@ oper SlashV2VNP : V2V -> NP -> VPSlash -> VPSlash = notYet "SlashV2VNP" ; oper SlashVP : NP -> VPSlash -> ClSlash = notYet "SlashVP" ; oper SlashVS : NP -> VS -> SSlash -> ClSlash = notYet "SlashVS" ; oper SlashVV : VV -> VPSlash -> VPSlash = notYet "SlashVV" ; -oper SubjS : Subj -> S -> Adv = notYet "SubjS" ; oper TCond : Tense = notYet "TCond" ; oper TFut : Tense = notYet "TFut" ; oper TPast : Tense = notYet "TPast" ; oper Use2N3 : N3 -> N2 = notYet "Use2N3" ; oper UseN2 : N2 -> CN = notYet "UseN2" ; oper UseSlash : Temp -> Pol -> ClSlash -> SSlash = notYet "UseSlash" ; -oper UttCard : Card -> Utt = notYet "UttCard" ; -oper UttIAdv : IAdv -> Utt = notYet "UttIAdv" ; -oper UttIP : IP -> Utt = notYet "UttIP" ; -oper UttImpPl : Pol -> Imp -> Utt = notYet "UttImpPl" ; -oper UttImpPol : Pol -> Imp -> Utt = notYet "UttImpPol" ; -oper UttImpSg : Pol -> Imp -> Utt = notYet "UttImpSg" ; -oper UttQS : QS -> Utt = notYet "UttQS" ; -oper UttVP : VP -> Utt = notYet "UttVP" ; -oper by8agent_Prep : Prep = notYet "by8agent_Prep" ; -oper it_Pron : Pron = notYet "it_Pron" ; -oper they_Pron : Pron = notYet "they_Pron" ; -oper we_Pron : Pron = notYet "we_Pron" ; -oper whatSg_IP : IP = notYet "whatSg_IP" ; -oper which_IQuant : IQuant = notYet "which_IQuant" ; -oper whoSg_IP : IP = notYet "whoSg_IP" ; -oper youPl_Pron : Pron = notYet "youPl_Pron" ; -oper youPol_Pron : Pron = notYet "youPol_Pron" ; } diff --git a/src/czech/NounCze.gf b/src/czech/NounCze.gf index 59372f68..1351753f 100644 --- a/src/czech/NounCze.gf +++ b/src/czech/NounCze.gf @@ -5,106 +5,120 @@ concrete NounCze of Noun = open ResCze, Prelude in { lin - DetCN det cn = { - s,prep,clit = \\c => det.s ! cn.g ! c ++ numSizeForm cn.s det.size c ; - a = numSizeAgr cn.g det.size P3 ; - hasClit = False ; + DetCN det cn = + let s : Case => Str = \\c => det.s ! nounGender cn (numSizeNumber det.size) ! c ++ numSizeForm cn.s det.size c + in npForms s s ** { + clit = s ; + a = numeralAgr (nounGender cn (numSizeNumber det.size)) det P3 ; + m = numeralModAgr (nounGender cn (numSizeNumber det.size)) det ; + hasClit = False ; isDrop = False ; isPron = False ; } ; - MassNP cn = { - s,prep,clit = \\c => cn.s ! Sg ! c ; - a = Ag cn.g Sg P3 ; - hasClit = False ; + MassNP cn = + let s = cn.s ! Sg in npForms s s ** { + clit = s ; + a = Ag cn.g Sg P3 ; m = Mod cn.g Sg ; isPron = False ; + hasClit = False ; isDrop = False ; } ; - DetQuant quant num = { - s = \\g,c => num.s ! g ! c ++ quant.s ! g ! numSizeNumber num.size ! c ; - size = num.size + DetQuant = quantifyNumeral ; + + OrdSuperl a = adjFormsAdjective a.superl ; + -- With a scale head, the quantifier modifies the scale, but the following + -- ordinal modifies the counted noun: s tímto tisícem nejlepších korun. + DetQuantOrd quant num ord = + let det = quantifyNumeral quant num in det ** { + s = \\g,c => det.s ! g ! c ++ + ord.s ! g ! numSizeNumber num.size ! countCase num.size c } ; DefArt = {s = \\_,_,_ => []} ; IndefArt = {s = \\_,_,_ => []} ; - NumPl = {s = \\_,_ => [] ; size = Num2_4} ; ---- size - NumSg = {s = \\_,_ => [] ; size = Num1} ; + NumPl = invarDeterminer [] Num2_4 ; + NumSg = invarDeterminer [] Num1 ; - UsePron pron = { - s = table { - Nom | Voc => pron.nom ; + UsePron pron = + let s : Case => Str = table { + Nom | ResCze.Voc => pron.nom ; Gen => pron.gen ; Dat => pron.dat ; Acc => pron.acc ; Loc => pron.loc ; Ins => pron.ins } ; + prep : Case => Str = table { + Nom | ResCze.Voc => pron.nom ; + Gen => pron.pgen ; + Dat => pron.pdat ; + Acc => pron.pacc ; + Loc => pron.loc ; + Ins => pron.pins + } + in npForms s prep ** { clit = table { Nom => pron.cnom ; - Voc => pron.nom ; + ResCze.Voc => pron.nom ; Gen => pron.cgen ; Dat => pron.cdat ; Acc => pron.cacc ; Loc => pron.loc ; Ins => pron.ins } ; - prep = table { - Nom | Voc => pron.nom ; - Gen => pron.pgen ; - Dat => pron.pdat ; - Acc => pron.pacc ; - Loc => pron.loc ; - Ins => pron.pins - } ; - a = pron.a ; - hasClit = True ; + a = pron.a ; m = modifierAgr pron.a ; + hasClit = True ; isDrop = pron.isDrop ; isPron = True ; } ; PossPron pron = justDemPronFormsAdjective pron.poss ; - UsePN pn = { - s,clit,prep = \\c => pn.s ! c ; - a = Ag pn.g Sg P3 ; - hasClit = False ; + UsePN pn = npForms pn.s pn.s ** { + clit = pn.s ; + a = Ag pn.g Sg P3 ; m = Mod pn.g Sg ; isPron = False ; + hasClit = False ; isDrop = False ; } ; AdjCN ap cn = { - s = \\n,c => preOrPost (notB ap.isPost) (ap.s ! cn.g ! n ! c) (cn.s ! n ! c) ; - g = cn.g + s = \\n,c => preOrPost (notB ap.isPost) (ap.s ! nounGender cn n ! n ! c) (cn.s ! n ! c) ; + g = cn.g ; gPl = cn.gPl } ; RelCN cn rs = { - s = \\n,c => cn.s ! n ! c ++ rs.s ! Ag cn.g n P3 ; - g = cn.g + s = \\n,c => cn.s ! n ! c ++ rs.s ! Ag (nounGender cn n) n P3 ; + g = cn.g ; gPl = cn.gPl } ; AdvCN cn adv = { s = \\n,c => cn.s ! n ! c ++ adv.s ; - g = cn.g + g = cn.g ; gPl = cn.gPl } ; - AdvNP np adv = { - s,clit = \\c => np.s ! c ++ adv.s ; - prep = \\c => np.prep ! c ++ adv.s ; - a = np.a ; - hasClit = False ; + AdvNP np adv = + let forms = appendNPForms np adv.s in np ** forms ** { + clit = forms.s ; + hasClit = False ; isDrop = False ; } ; UseN n = nounFormsNoun n ; ApposCN cn np = { s = \\n,c => cn.s ! n ! c ++ np.s ! c ; ---- TODO check apposition order - g = cn.g + g = cn.g ; gPl = cn.gPl } ; NumCard c = c ; - NumDigits ds = ds ** {s = \\_,_ => ds.s} ; - NumDecimal ds = ds ** {s = \\_,_ => ds.s} ; + NumDigits ds = invarDeterminer ds.s ds.size ; + NumDecimal ds = invarDeterminer ds.s ds.size ; NumNumeral nu = nu ; SentCN cn sc = cn ** {s = \\n,c => cn.s ! n ! c ++ sc.s} ; - PredetNP pred np = np ** { - s = \\c => pred.s ++ np.s ! c ; - clit = \\c => pred.s ++ np.clit ! c ; - prep = \\c => pred.s ++ np.prep ! c + PredetNP pred np = + let forms = predetNPForms (andB pred.postPron np.isPron) + (\\c => predetForm pred np.m c) np + in np ** forms ** { + -- A predeterminer modifies a full NP: jen já, jen jeho. Its scope + -- cannot be preserved by an omitted subject or an object clitic. + clit = forms.s ; + hasClit = False ; isDrop = False } ; } diff --git a/src/czech/NumeralCze.gf b/src/czech/NumeralCze.gf index ece5fd5a..c6e9b126 100644 --- a/src/czech/NumeralCze.gf +++ b/src/czech/NumeralCze.gf @@ -1,102 +1,99 @@ -concrete NumeralCze of Numeral = +concrete NumeralCze of Numeral = CatCze [Numeral,Digits,Decimal] ** + open ResCze, Prelude in { - CatCze [Numeral,Digits,Decimal] ** - - open - ResCze, - Prelude - in { - --- from gf-contrib/numerals/czech.gf, added inflections --- AR 2020-03-20 ----- TODO ordinal forms - - -oper LinNumeral = Determiner ; -- {s : NumeralForms ; size : NumSize} ; -oper LinDigit = {unit : Gender => Case => Str ; teen, ten, hundred : Str ; size : NumSize} ; - -lincat Digit = LinDigit ; -lincat Sub10 = LinDigit ; - -lincat Sub100 = LinNumeral ; -lincat Sub1000 = LinNumeral ; -lincat Sub1000000 = LinNumeral ; - -oper mkNum : Determiner -> Str -> Str -> Str -> LinDigit = - \dva, dvanast, dvadsat, dveste -> { - unit = dva.s ; - teen = dvanast + "náct" ; - ten = dvadsat ; - hundred = dveste ; - size = dva.size ; - } ; - -oper mk2Num : Determiner -> Str -> Str -> Str -> LinDigit = - \unit, teenbase, tenbase, hundred -> - mkNum unit teenbase (tenbase + "cet") hundred ; - -oper mk5Num : Str -> Str -> Str -> Str -> LinDigit = - \unit,uniti, teenbase, tenbase -> - mkNum (regNumeral unit uniti) teenbase (tenbase + "desát") (unit ++ "set") ; - -oper bigNumeral : Str -> LinNumeral = \s -> invarNumeral s ; - -lin num x = x ; - -lin n2 = mk2Num twoNumeral "dva" "dva" ("dvě" ++ "stě") ; -lin n3 = mk2Num threeNumeral "tři" "tři" ("tři" ++ "sta") ; -lin n4 = mk2Num fourNumeral "čtr" "čtyři" ("čtyři" ++ "sta") ; -lin n5 = mk5Num "pět" "pěti" "pat" "pa" ; -lin n6 = mk5Num "šest" "šesti" "šest" "še" ; -lin n7 = mk5Num "sedm" "sedmi" "sedm" "sedm"; -lin n8 = mk5Num "osm" "osmi" "osm" "osm"; -lin n9 = mk5Num "devět" "devíti" "devate" "deva" ; - -lin pot01 = { - unit = oneNumeral.s ; hundred = "sto" ; ten = "deset" ; teen = "jedenáct" ; - size = Num1 - } ; -lin pot0 d = d ; - -lin pot110 = bigNumeral "deset" ; -lin pot111 = bigNumeral "jedenáct" ; -lin pot1to19 d = bigNumeral d.teen ; - -lin pot0as1 n = {s = n.unit ; size = n.size} ; -lin pot1 d = bigNumeral d.ten ; -lin pot1plus d e = { - s = (invarNumeral (d.ten ++ determinerStr (e ** {s = e.unit}))).s ; ---- TODO inflection? - size = tfSize e.size - } ; - ---- variants { d.s ! ten ++ e.s ! unit ; glue (glue (e.s ! unit) "a") (d.s ! ten)} ; size = tfSize e.size} ; - -lin pot1as2 n = n ; -lin pot2 d = bigNumeral d.hundred ; -lin pot2plus d e = { - s = (invarNumeral (d.hundred ++ determinerStr e)).s ; ---- TODO inflection? - size = tfSize e.size - } ; - -lin pot2as3 n = n ; -lin pot3 n = bigNumeral (mkTh (determinerStr n) n.size) ; - -lin pot3plus n m = { - s = (invarNumeral (mkTh (determinerStr n) n.size ++ determinerStr m)).s ; ---- TODO inflection? - size = tfSize m.size - } ; - -oper tfSize : NumSize -> NumSize = \sz -> - table {Num1 => Num5 ; other => other} ! sz ; - -oper mkTh : Str -> NumSize -> Str = \attr,size -> - case size of { - Num1 => "tisíc" ; - Num2_4 => attr ++ "tisíce" ; - Num5 => attr ++ "tisíc" +-- Keep inflection until the numeral receives its case. Compounds ending in +-- units or tens use quantified agreement: mých dvacet jedna stromů, dvacet dva dětí. +-- Agreement with the final unit does not compose with possessive modifiers. +-- See https://www.czechency.org/slovnik/ČÍSLOVKA. +oper + -- Only units vary in count agreement. Keeping four full Determiners here + -- would create a product of independent agreement states during compilation. + LinDigit : Type = {unit : Determiner ; teen,ten,hundred : Gender => Case => Str} ; + digit : Determiner -> Str -> Str -> Str -> LinDigit = \u,teen,ten,hundred -> { + unit = u ; teen = (regNumeral teen (teen + "i")).s ; + ten = (regNumeral ten (ten + "i")).s ; + hundred = \\_,c => case c of {Nom|Acc|ResCze.Voc => hundred ; _ => (scale QuantifiedScale u hundredN).s ! Neutr ! c} } ; + counted : (Gender => Case => Str) -> Determiner = \s -> { + s = s ; size = Num5 ; head = CountedHead + } ; + hundreds : LinDigit -> Determiner = \d -> { + s = d.hundred ; size = NumScale ; head = ScaleHead Neutr d.unit.size QuantifiedScale + } ; + plus : Determiner -> Determiner -> Determiner = \a,b -> { + -- Compound jedna stays fixed even in oblique cases; dva inflects but + -- does not vary with the counted noun's gender (CEG 6.1.5--6.1.6). + s = \\g,c => a.s ! g ! c ++ case b.size of { + Num1 => b.s ! Fem ! Nom ; _ => b.s ! Masc Inanim ! c + } ; + -- A final scale still governs genitive in every case; other compound + -- tails take ordinary quantified agreement, including final 1--4. + size = case b.size of {NumScale => NumScale ; _ => Num5} ; + -- Sums ending in a scale retain the leading scale's agreement head: + -- tyto dva tisíce dvě stě korun, not tato dva tisíce dvě stě korun. + head = case b.size of {NumScale => a.head ; _ => CountedHead} + } ; + scale : ScaleAgreement -> Determiner -> Noun -> Determiner = \agr,d,n -> { + s = \\_,c => d.s ! n.g ! c ++ numSizeForm n.s d.size c ; size = NumScale ; + head = case d.head of {CountedHead => ScaleHead n.g d.size agr ; h => h} + } ; + bareScale : ScaleAgreement -> Noun -> Determiner = \agr,n -> { + s = \\_,c => n.s ! Sg ! c ; size = NumScale ; head = ScaleHead n.g Num1 agr + } ; + decimalScale : ScaleAgreement -> {s : Str ; size : NumSize ; hasDot : Bool} -> Noun -> Determiner = \agr,d,n -> { + s = \\_,c => d.s ++ case d.hasDot of { + True => n.s ! Sg ! Gen ; False => numSizeForm n.s d.size c + } ; size = NumScale ; head = ScaleHead n.g d.size agr + } ; + hundredN : Noun = nounFormsNoun ((declMESTO "sto") ** {pgen = "set" ; pdat = "stům" ; ploc = "stech"}) ; + thousandN : Noun = nounFormsNoun ((declSTROJ "tisíc") ** {pgen = "tisíc"}) ; + millionN : Noun = nounFormsNoun (declHRAD "milion") ; + billionN : Noun = nounFormsNoun (declZENA "miliarda") ; -oper determinerStr : Determiner -> Str = \d -> d.s ! Masc Anim ! Nom ; - +lincat + Digit,Sub10 = LinDigit ; + Sub100,Sub1000,Sub1000000,Sub1000000000,Sub1000000000000 = Determiner ; +lin + num x = x ; + n2 = digit twoNumeral "dvanáct" "dvacet" "dvě stě" ; + n3 = digit threeNumeral "třináct" "třicet" "tři sta" ; + n4 = digit fourNumeral "čtrnáct" "čtyřicet" "čtyři sta" ; + n5 = digit (regNumeral "pět" "pěti") "patnáct" "padesát" "pět set" ; + n6 = digit (regNumeral "šest" "šesti") "šestnáct" "šedesát" "šest set" ; + n7 = digit (regNumeral "sedm" "sedmi") "sedmnáct" "sedmdesát" "sedm set" ; + n8 = digit (regNumeral "osm" "osmi") "osmnáct" "osmdesát" "osm set" ; + n9 = digit (regNumeral "devět" "devíti") "devatenáct" "devadesát" "devět set" ; + pot01 = {unit = oneNumeral ; teen = (regNumeral "jedenáct" "jedenácti").s ; + ten = (regNumeral "deset" "deseti").s ; hundred = (bareScale QuantifiedScale hundredN).s} ; + pot0 d = d ; + pot0as1 d = d.unit ; + pot110 = regNumeral "deset" "deseti" ; + pot111 = regNumeral "jedenáct" "jedenácti" ; + pot1to19 d = counted d.teen ; + pot1 d = counted d.ten ; + pot1plus d e = plus (counted d.ten) e.unit ; + pot1as2 n = n ; + pot21 = bareScale QuantifiedScale hundredN ; + pot2 d = hundreds d ; + pot2plus d e = plus (hundreds d) e ; + pot2as3 n = n ; + -- Generation default: hundreds/thousands take quantified agreement. + -- Tisíc also admits nominal agreement; the noun's declension is independent. + pot31 = bareScale QuantifiedScale thousandN ; + pot3 n = scale QuantifiedScale n thousandN ; + pot3plus n m = plus (scale QuantifiedScale n thousandN) m ; + pot3as4 n = n ; + pot3decimal d = decimalScale QuantifiedScale d thousandN ; + -- Millions/billions instead agree with their nominal head (dva miliony jsou). + pot41 = bareScale NominalScale millionN ; + pot4 n = scale NominalScale n millionN ; + pot4plus n m = plus (scale NominalScale n millionN) m ; + pot4as5 n = n ; + pot4decimal d = decimalScale NominalScale d millionN ; + pot51 = bareScale NominalScale billionN ; + pot5 n = scale NominalScale n billionN ; + pot5plus n m = plus (scale NominalScale n billionN) m ; + pot5decimal d = decimalScale NominalScale d billionN ; -- -- Numerals as sequences of digits have a separate, simpler grammar lincat Dig = {s:Str ; size : NumSize} ; @@ -106,7 +103,7 @@ oper determinerStr : Determiner -> Str = \d -> d.s ! Masc Anim ! Nom ; IIDig d dd = {s = d.s ++ Predef.BIND ++ dd.s ; size = Num5} ; ---- leading zeros ?? - D_0 = { s = "0" ; size = Num1} ; ---- ?? + D_0 = { s = "0" ; size = Num5} ; D_1 = { s = "1" ; size = Num1} ; D_2 = { s = "2" ; size = Num2_4} ; D_3 = { s = "3" ; size = Num2_4} ; @@ -120,12 +117,12 @@ oper determinerStr : Determiner -> Str = \d -> d.s ! Masc Anim ! Nom ; PosDecimal d = d ** {hasDot=False} ; NegDecimal d = { s = "-" ++ Predef.BIND ++ d.s ; - size = Num5 ; + size = d.size ; hasDot=False } ; IFrac d i = { s = d.s ++ - if_then_Str d.hasDot BIND (BIND++"."++BIND) ++ + if_then_Str d.hasDot BIND (BIND++","++BIND) ++ i.s ; size = Num5 ; hasDot=True diff --git a/src/czech/ParadigmsCze.gf b/src/czech/ParadigmsCze.gf index afd5e634..473179b4 100644 --- a/src/czech/ParadigmsCze.gf +++ b/src/czech/ParadigmsCze.gf @@ -40,10 +40,19 @@ oper mkN = overload { mkN : (nom : Str) -> N = \nom -> lin N (guessNounForms nom) ; + -- Select a default paradigm; mixed endings and stem alternations may + -- still need lexical overrides on the result. mkN : (nom,gen : Str) -> Gender -> N = \nom,gen,g -> lin N (declensionNounForms nom gen g) ; } ; + mkPN = overload { + -- Indeclinable name: every case uses the supplied string. + mkPN : Str -> Gender -> PN = \s,g -> lin PN {s = \\_ => s ; g = g} ; + -- Inflected name: use the noun paradigm's singular cases and gender. + mkPN : N -> PN = \n -> lin PN {s = (nounFormsNoun n).s ! Sg ; g = n.g} ; + } ; + -- The following standard declensions can be used with good accuracy. -- However, they have some defaults that may have to be overwritten. -- This can be done easily by overriding those formes with record extension (**). @@ -81,31 +90,47 @@ oper -- The full definition of the noun record is -- { -- snom,sgen,sdat,sacc,svoc,sloc,sins, pnom,pgen,pdat,pacc,ploc,pins : Str ; --- g : Gender +-- g,gPl : Gender -- } --------------------- -- Adjectives --- Only positive forms so far ---- +-- Guess regular comparison; supply a principal part for exceptions, or +-- nonExist as the comparative for a positive-only adjective. mkA = overload { mkA : Str -> A - = \s -> lin A (guessAdjForms s) ; + = \s -> lin A (degreeAdjForms s (guessComparative s)) ; + mkA : (positive,comparative : Str) -> A + = \p,c -> lin A (degreeAdjForms p c) ; } ; + -- Declension constructors supply positive forms only. mladyA : Str -> A - = \s -> lin A (mladyAdjForms s) ; + = \s -> lin A (positiveAdj (mladyAdjForms s)) ; jarniA : Str -> A - = \s -> lin A (jarniAdjForms s) ; + = \s -> lin A (positiveAdj (jarniAdjForms s)) ; otcuvA : Str -> A - = \s -> lin A (otcuvAdjForms s) ; + = \s -> lin A (positiveAdj (otcuvAdjForms s)) ; matcinA : Str -> A - = \s -> lin A (matcinAdjForms s) ; + = \s -> lin A (positiveAdj (matcinAdjForms s)) ; invarA : Str -> A - = \s -> lin A (invarAdjForms s) ; + = \s -> lin A (positiveAdj (invarAdjForms s)) ; + + -- Short adjectives supply predicates, not attributive AP forms. + shortAP : (m,f,n,mp,fp,np : Str) -> AP = \m,f,n,mp,fp,np -> + let ap : Adjective = { + s = \\g,num,c => case of { + => m ; => f ; => n ; + => mp ; => np ; + => fp ; _ => nonExist + } + } in lin AP { + s = \\_,_,_ => nonExist ; pred = shortPredicate ap ; isPost = True + } ; mkA2 : A -> Prep -> A2 = \a,p -> lin A2 (a ** {c = p}) ; @@ -113,21 +138,72 @@ oper ------------------------- -- Verbs + -- Class constructors, not guesses from an arbitrary infinitive. + -- kupovat: -ovat, -uji, -oval, -uj; kryt: -ýt/-ít, -yji/-iji, -yl/-il. + kupovatV : Str -> V = \s -> lin V (iii_kupovatVerbForms s) ; + krytV : Str -> V = \s -> lin V (iii_krýtVerbForms s) ; + + -- Full present and imperative forms, with masculine singular/plural past + -- participles. Keep this twelve-field input compatible as VerbForms grows. + -- Storing past participles does not yet implement past-tense clauses. + VerbPrincipalParts : Type = PositiveVerbForms ; + + mkV = overload { + mkV : VerbPrincipalParts -> V = \v -> lin V (withNeg v) ; + mkV : (inf,p1sg,p2sg,p3sg,p1pl,p2pl,p3pl,pastsg,pastpl,imp2sg,imp1pl,imp2pl : Str) -> V = + \inf,p1sg,p2sg,p3sg,p1pl,p2pl,p3pl,pastsg,pastpl,imp2sg,imp1pl,imp2pl -> lin V (withNeg { + inf = inf ; pressg1 = p1sg ; pressg2 = p2sg ; pressg3 = p3sg ; + prespl1 = p1pl ; prespl2 = p2pl ; prespl3 = p3pl ; + pastpartsg = pastsg ; pastpartpl = pastpl ; + impsg2 = imp2sg ; imppl1 = imp1pl ; imppl2 = imp2pl + }) ; + } ; + + -- Lexical reflexive clitics. The case-based interface accepts only Acc/Dat. + seV : V -> V = \v -> reflV v Acc ; + siV : V -> V = \v -> reflV v Dat ; + reflV : V -> Case -> V = \v,c -> v ** { + isRefl = True ; refl = case c of {Acc => "se" ; Dat => "si" ; _ => nonExist} + } ; + + mkVS : V -> VS = \v -> lin VS v ; + mkVQ : V -> VQ = \v -> lin VQ v ; + -- Ordinary VV: the infinitive retains its own clitic domain. + mkVV : V -> VV = \v -> lin VV (v ** {isAux = False}) ; + -- Modals allow their infinitive's clitics in the finite clause. A lexical + -- reflexive on the matrix verb blocks this climbing. + mkModalVV : V -> VV = \v -> lin VV (v ** {isAux = notB v.isRefl}) ; + -- Third-person singular gender selects personal and possessive forms. + -- A no-op keeps lexical overrides; conversion uses the standard target forms. + genderPron : Gender -> Pron -> Pron = \g,p -> case p.a of { + Ag old Sg P3 => case of { + | | + | => p ; + _ => lin Pron ((mkPron (Ag g Sg P3)) ** {isDrop = p.isDrop}) + } ; + _ => p ** { + a = case p.a of {Ag _ n person => Ag g n person ; AgPol _ => AgPol g ; AgQuant _ => AgQuant g} ; + nom = case p.a of { + Ag _ Pl P3 => (personalPron (Ag g Pl P3)).nom ; _ => p.nom + } + } + } ; + mkV2 = overload { - mkV2 : VerbForms -> VerbForms ** {c : ComplementCase} - = \vf -> vf ** {c = {s = [] ; c = Acc ; hasPrep = False}} ; - mkV2 : VerbForms -> Case -> VerbForms ** {c : ComplementCase} - = \vf,c -> vf ** {c = {s = [] ; c = c ; hasPrep = False}} ; - mkV2 : VerbForms -> ComplementCase -> VerbForms ** {c : ComplementCase} - = \vf,c -> vf ** {c = c} ; + mkV2 : V -> V2 + = \v -> lin V2 (v ** {c = {s = [] ; c = Acc ; hasPrep = False}}) ; + mkV2 : V -> Case -> V2 + = \v,c -> lin V2 (v ** {c = {s = [] ; c = c ; hasPrep = False}}) ; + mkV2 : V -> Prep -> V2 + = \v,p -> lin V2 (v ** {c = p}) ; } ; mkV3 = overload { - mkV3 : VerbForms -> VerbForms ** {c,c2 : ComplementCase} - = \vf -> vf ** {c = {s = [] ; c = Acc ; hasPrep = False} ; - c2 = {s = [] ; c = Dat ; hasPrep = False}} ; - mkV3 : VerbForms -> ComplementCase -> ComplementCase -> VerbForms ** {c,c2 : ComplementCase} - = \vf,c,c2 -> vf ** {c = c ; c2 = c2} ; + mkV3 : V -> V3 + = \v -> lin V3 (v ** {c = {s = [] ; c = Acc ; hasPrep = False} ; + c2 = {s = [] ; c = Dat ; hasPrep = False}}) ; + mkV3 : V -> Prep -> Prep -> V3 + = \v,p,p2 -> lin V3 (v ** {c = p ; c2 = p2}) ; } ; ------------------------ @@ -139,8 +215,17 @@ oper mkAdv : Str -> Adv = \s -> lin Adv {s = s} ; - mkPrep : Str -> Case -> Prep - = \s,c -> lin Prep {s = s ; c = c ; hasPrep = True} ; ---- True if s /= "" + mkPrep = overload { + -- Bare case government: use this instead of mkPrep "" c. + mkPrep : Case -> Prep + = \c -> lin Prep {s = [] ; c = c ; hasPrep = False} ; + -- Overt preposition, possibly with token-dependent allomorphs. + mkPrep : Str -> Case -> Prep + = \s,c -> lin Prep {s = s ; c = c ; hasPrep = True} ; + } ; + + -- The same vocalization applies to locative and accusative v. + v_Prep : Case -> Prep = \c -> mkPrep vPreposition c ; mkConj : Str -> Conj = \s -> lin Conj {s1 = [] ; s2 = s} ; diff --git a/src/czech/PhraseCze.gf b/src/czech/PhraseCze.gf index 50bd9dc9..d733b5ad 100644 --- a/src/czech/PhraseCze.gf +++ b/src/czech/PhraseCze.gf @@ -1,16 +1,24 @@ concrete PhraseCze of Phrase = CatCze ** open Prelude, ResCze in { lin - UttS s = s ; + UttS s = {s = s.s} ; + UttQS q = {s = q.s} ; + UttIAdv a = a ; + UttIP ip = {s = ip.s ! Nom} ; + -- Choose feminine forms for simple units (jedna, dvě). Compounds use + -- their own gender-independent forms (dvacet jedna, dvacet dva). + UttCard c = {s = c.s ! Fem ! Nom} ; UttAdv adv = adv ; UttCN cn = {s = cn.s ! Sg ! Nom} ; - UttAP ap = {s = ap.s ! Masc Anim ! Sg ! Nom} ; + UttAP ap = {s = ap.pred ! Ag (Masc Anim) Sg P3} ; UttNP np = {s = np.s ! Nom} ; - UttVP vp = let agr = Ag Neutr Sg P3 in {s = vp.clit ! agr ++ vp.verb.inf ++ vp.compl ! agr} ; + UttVP vp = let agr = Ag Neutr Sg P3 in {s = vp.verb.inf ++ vp.clit ! agr ++ vp.compl ! agr} ; - UttImpSg pol imp = {s = pol.s ++ imp.s} ; - UttImpPl pol imp = {s = pol.s ++ imp.s} ; - UttImpPol pol imp = {s = pol.s ++ imp.s} ; + -- pol.p selects the verb form; empty pol.s retains the Pol constituent. + -- Without it, parsing "nečti ji" recovers UttImpSg ?1 instead of PNeg. + UttImpSg pol imp = {s = pol.s ++ imp.s ! pol.p ! Ag (Masc Anim) Sg P2} ; + UttImpPl pol imp = {s = pol.s ++ imp.s ! pol.p ! Ag (Masc Anim) Pl P2} ; + UttImpPol pol imp = {s = pol.s ++ imp.s ! pol.p ! AgPol (Masc Anim)} ; PhrUtt pconj utt voc = {s = pconj.s ++ utt.s ++ voc.s} ; @@ -19,6 +27,6 @@ lin PConjConj conj = {s = conj.s2} ; NoVoc = {s = []} ; - VocNP np = {s = np.s ! Voc} ; + VocNP np = {s = np.s ! ResCze.Voc} ; } diff --git a/src/czech/QuestionCze.gf b/src/czech/QuestionCze.gf index d0bea0ee..e87f276b 100644 --- a/src/czech/QuestionCze.gf +++ b/src/czech/QuestionCze.gf @@ -1,9 +1,24 @@ -concrete QuestionCze of Question = CatCze ** - open ResCze, Prelude in { - +concrete QuestionCze of Question = CatCze ** open ResCze, Prelude in { lin - QuestCl cl = cl ; ---- - - QuestIAdv iadv cl = cl ** {clit = iadv.s ++ cl.clit} ; - + QuestCl cl = cl ** {q = [] ; yesNo = True} ; + QuestIAdv adv cl = cl ** {q = adv.s ; yesNo = False} ; + QuestIComp comp np = { + q = comp.s ; subj = case np.isDrop of {True => np.clit ! Nom ; False => np.s ! Nom} ; + clit,compl = [] ; verb = copulaVerbForms ; a = np.a ; yesNo = False + } ; + QuestVP ip vp = { + q = ip.s ! Nom ; subj = [] ; clit = vp.clit ! ip.a ; + compl = vp.compl ! ip.a ; verb = vp.verb ; a = ip.a ; yesNo = False + } ; + CompIAdv adv = adv ; + CompIP ip = {s = ip.s ! Nom} ; + IdetCN det cn = { + s = \\c => det.s ! nounGender cn (numSizeNumber det.size) ! c ++ numSizeForm cn.s det.size c ; + a = numeralAgr (nounGender cn (numSizeNumber det.size)) det P3 + } ; + IdetIP det = {s = \\c => det.s ! Neutr ! c ; a = numeralAgr Neutr det P3} ; + IdetQuant = quantifyNumeral ; + PrepIP p ip = {s = p.s ++ ip.s ! p.c} ; + AdvIP ip adv = ip ** {s = \\c => ip.s ! c ++ adv.s} ; + AdvIAdv a b = {s = a.s ++ b.s} ; } diff --git a/src/czech/RelativeCze.gf b/src/czech/RelativeCze.gf index 8726b851..fc9b67c7 100644 --- a/src/czech/RelativeCze.gf +++ b/src/czech/RelativeCze.gf @@ -8,11 +8,11 @@ lin subj = let rel = (adjFormsAdjective rp).s in \\a => case a of { - Ag g n _ => rel ! g ! n ! Nom - } + Ag g n _ => rel ! g ! n ! Nom ; AgPol g => rel ! g ! Sg ! Nom ; AgQuant g => rel ! g ! Pl ! Nom + } } ; - - IdRP = mkA "který" ; + + IdRP = guessAdjForms "který" ; } diff --git a/src/czech/ResCze.gf b/src/czech/ResCze.gf index cb733689..193bc389 100644 --- a/src/czech/ResCze.gf +++ b/src/czech/ResCze.gf @@ -17,7 +17,7 @@ param Person = P1 | P2 | P3 ; - Agr = Ag Gender Number Person ; + Agr = Ag Gender Number Person | AgPol Gender | AgQuant Gender ; -- polite singular: plural verb, singular predicate CTense = CTPres | CTPast ; ----- TODO complete the tense system to match Czech verb morphology @@ -71,12 +71,28 @@ oper _ => init (addI s) + "í" } ; + -- Before i/í/ě the vowel letter carries the dental's palatalization. + dentalStem : Str -> Str = \s -> case s of { + stem + "ň" => stem + "n" ; stem + "ť" => stem + "t" ; + stem + "ď" => stem + "d" ; _ => s + } ; + + -- The žena ending is spelled i after a soft consonant, otherwise y. + addY : Str -> Str = \s -> case s of { + _ + #softConsonant => dentalStem s + "i" ; _ => s + "y" + } ; + -- 3.4.10, in particular when also final 'a' is dropped addE : Str -> Str = \s -> case s of { re + "k" => re + "ce" ; pra + ("g"|"h") => pra + "ze" ; stre + "ch" => stre + "še" ; sest + "r" => sest + "ře" ; + stem + "l" => stem + "le" ; + stem + "z" => stem + "ze" ; + stem + "s" => stem + "se" ; + _ + ("ň"|"ť"|"ď") => dentalStem s + "ě" ; + _ + #softConsonant => s + "e" ; pan => pan + "ě" } ; @@ -106,12 +122,12 @@ oper -- so this is the lincat of N - NounForms : Type = {snom,sgen,sdat,sacc,svoc,sloc,sins, pnom,pgen,pdat,pacc,ploc,pins : Str ; g : Gender} ; + NounForms : Type = {snom,sgen,sdat,sacc,svoc,sloc,sins, pnom,pgen,pdat,pacc,ploc,pins : Str ; g,gPl : Gender} ; -- But traditional tables make agreement easier to handle in syntax -- so this is the lincat of CN - Noun : Type = {s : Number => Case => Str ; g : Gender} ; + Noun : Type = {s : Number => Case => Str ; g,gPl : Gender} ; -- this is used in UseN @@ -136,7 +152,7 @@ oper Ins => forms.pins } } ; - g = forms.g + g = forms.g ; gPl = forms.gPl } ; -- terminology of CEG @@ -209,7 +225,7 @@ oper pdat = pan + "ům" ; pacc,pins = pan + "y" ; ploc = addEch pan ; - g = Masc Anim + g,gPl = Masc Anim } ; declPREDSEDA : DeclensionType = \predseda -> --- 3.5.4: sgen y/i @@ -231,7 +247,7 @@ oper pdat = predsed + "ům" ; pacc,pins = predsed + "y" ; ploc = addEch predsed ; - g = Masc Anim + g,gPl = Masc Anim } ; -- the oblique stem is a separate argument, because it cannot always be @@ -248,7 +264,7 @@ oper pgen = hrd + "ů" ; pdat = hrd + "ům" ; ploc = addEch hrd ; - g = Masc Inanim + g,gPl = Masc Inanim } ; declHRAD : DeclensionType = \hrad -> --- 3.5.2: sloc u/ě/e extra arg, sport-u, hrad-ě ; sgen u/a @@ -259,18 +275,18 @@ oper in { snom = zena ; - sgen = zen + "y" ; --- i after soft cons sometimes - sdat,sloc = zen + "ě" ; --- i after soft cons sometimes ; skol+e + sgen = addY zen ; + sdat,sloc = addE zen ; sacc = zen + "u" ; svoc = shortenVowel zen + "o" ; ---- shorten ? sins = zen + "ou" ; - pnom,pacc = zen + "y" ; --- also sgen + pnom,pacc = addY zen ; pgen = zen ; --- sometimes with vowel shortening pdat = zen + "ám" ; ploc = zen + "ách" ; pins = zen + "ami" ; - g = Fem + g,gPl = Fem } ; declMESTO : DeclensionType = \mesto -> --- 3.7.1 sloc u/e ; pgen vowel shortening sometimes ; ploc variations @@ -288,7 +304,7 @@ oper pdat = mest + "ům" ; ploc = mest + "ech" ; --- with variations pins = mest + "y" ; - g = Neutr + g,gPl = Neutr } ; -- Latin masculines in -us: the ending is dropped outside the nominative @@ -301,7 +317,7 @@ oper } ; declLATINUSA : DeclensionType = \genius -> - declLATINUS genius ** {g = Masc Anim} ; + declLATINUS genius ** {g,gPl = Masc Anim} ; -- Latin neuters in -um: the ending is dropped outside the nominative -- (kontinuum - kontinua), otherwise they follow město @@ -324,7 +340,7 @@ oper pdat = schemat + "ům" ; ploc = schemat + "ech" ; pins = schemat + "y" ; - g = Neutr + g,gPl = Neutr } ; -- the hrad type with genitive -a instead of -u (les - lesa, zákon - zákona) @@ -347,7 +363,7 @@ oper pgen,ploc = a.pgen ; pdat = a.msins ; pins = a.pins ; - g = Fem + g,gPl = Fem } ; declADJM : DeclensionType = \nulty -> @@ -362,14 +378,14 @@ oper pgen,ploc = a.pgen ; pdat = a.msins ; pins = a.pins ; - g = Masc Inanim + g,gPl = Masc Inanim } ; -- indeclinable loans: bombé, tamari, software declINVAR : Gender -> DeclensionType = \g,s -> { snom,sgen,sdat,sacc,svoc,sloc,sins = s ; pnom,pgen,pdat,pacc,ploc,pins = s ; - g = g + g,gPl = g } ; declMUZ : DeclensionType = \muz_ -> --- 3.5.3 : sdat,sloc ; pnom @@ -395,7 +411,7 @@ oper pdat = muz + "ům" ; ploc = muz + "ích" ; pins = muz + "i" ; - g = Masc Anim + g,gPl = Masc Anim } ; declSOUDCE : DeclensionType = \soudce -> --- 3.5.3: sdat/sloc i,ovi ; pnom i/ové @@ -412,7 +428,7 @@ oper pacc = soudce ; ploc = soudc + "ích" ; pins = soudc + "i" ; - g = Masc Anim + g,gPl = Masc Anim } ; declSTROJ : DeclensionType = \stroj -> @@ -427,7 +443,7 @@ oper pdat = stroj + "ům" ; ploc = stroj + "ích" ; pins = stroj + "i" ; - g = Masc Inanim + g,gPl = Masc Inanim } ; declRUZE : DeclensionType = \ruze -> --- 3.6.2: pgen ulice-ulic, chvile-cvil @@ -443,11 +459,11 @@ oper pdat = ruz + "ím" ; ploc = ruz + "ích" ; pins = ruz + "emi" ; - g = Fem + g,gPl = Fem } ; declPISEN : DeclensionType = \pisen -> - let pisn = dropFleetingE pisen + let pisn = dentalStem (dropFleetingE pisen) in { snom,sacc = pisen ; @@ -460,21 +476,23 @@ oper pdat = pisn + "ím" ; ploc = pisn + "ích" ; pins = pisn + "ěmi" ; - g = Fem + g,gPl = Fem } ; declKOST : DeclensionType = \kost -> + let stem = dentalStem kost + in { snom,sacc = kost ; - sgen,sdat,svoc,sloc = kost + "i" ; --- pnom,pacc - sins = kost + "í" ; --- pgen + sgen,sdat,svoc,sloc = stem + "i" ; --- pnom,pacc + sins = stem + "í" ; --- pgen - pnom,pacc = kost + "i" ; - pgen = kost + "í" ; - pdat = kost + "em" ; - ploc = kost + "ech" ; + pnom,pacc = stem + "i" ; + pgen = stem + "í" ; + pdat = stem + "em" ; + ploc = stem + "ech" ; pins = kost + "mi" ; - g = Fem + g,gPl = Fem } ; declKURE : DeclensionType = \kure -> @@ -491,7 +509,7 @@ oper pdat = kur + "atům" ; ploc = kur + "atech" ; pins = kur + "aty" ; - g = Neutr + g,gPl = Neutr } ; declMORE : DeclensionType = \more -> --- 3.7.2 pgen zero sometimes @@ -507,7 +525,7 @@ oper pdat = mor + "ím" ; ploc = mor + "ích" ; pins = mor + "i" ; - g = Neutr + g,gPl = Neutr } ; declSTAVENI : DeclensionType = \staveni -> @@ -519,7 +537,7 @@ oper pdat = staveni + "m" ; ploc = staveni + "ch" ; pins = staveni + "mi" ; - g = Neutr + g,gPl = Neutr } ; --------------------------- @@ -528,8 +546,24 @@ oper -- to be used for AP: 56 forms for each degree Adjective : Type = {s : Gender => Number => Case => Str} ; + -- Long predicates agree with the counted noun; short predicates instead + -- use neuter singular with a quantified subject. + longPredicate : Adjective -> Agr => Str = \ap -> \\a => case a of { + Ag g n _ => ap.s ! g ! n ! Nom ; + AgPol g => ap.s ! g ! Sg ! Nom ; + AgQuant g => ap.s ! g ! Pl ! Gen + } ; + shortPredicate : Adjective -> Agr => Str = \ap -> \\a => case a of { + AgQuant _ => ap.s ! Neutr ! Sg ! Nom ; + _ => longPredicate ap ! a + } ; + -- to be used for A, in three degrees: 15 forms in each ----- TODO other degrees than positive + DegreeForms : Type = AdjForms ** {compar,superl : AdjForms} ; + + positiveAdj : AdjForms -> DegreeForms = \a -> a ** { + compar,superl = invarAdjForms nonExist + } ; AdjForms : Type = { msnom, fsnom, nsnom : Str ; -- svoc = snom @@ -585,6 +619,36 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { } ; + -- Regular comparison plus common lexical exceptions. Spelling cannot + -- determine semantic gradability or every stem alternation: callers can + -- supply a comparative or nonExist explicitly. + guessComparative : Str -> Str = \s -> case s of { + "dobrý" => "lepší" ; "špatný" | "zlý" => "horší" ; + "malý" => "menší" ; "velký" => "větší" ; "dlouhý" => "delší" ; + "mladý" => "mladší" ; "starý" => "starší" ; + "chudý" => "chudší" ; "tvrdý" => "tvrdší" ; "bledý" => "bledší" ; + "bílý" => "bělejší" ; "hnědý" => "hnědší" ; "hezký" => "hezčí" ; + stem + "cký" => stem + "čtější" ; + stem + "ský" => stem + "štější" ; + stem + ("ný" | "ní") => stem + "nější" ; + stem + "lý" => stem + "lejší" ; + stem + "rý" => stem + "řejší" ; + stem + "vý" => stem + "vější" ; + stem + "dý" => stem + "dější" ; + stem + "tý" => stem + "tější" ; + stem + "pý" => stem + "pější" ; + stem + "bý" => stem + "bější" ; + stem + "mý" => stem + "mější" ; + stem + "zí" => stem + "zejší" ; + stem + "ží" => stem + "žejší" ; + _ => nonExist + } ; + + degreeAdjForms : Str -> Str -> DegreeForms = \p,c -> + (guessAdjForms p) ** { + compar = guessAdjForms c ; superl = guessAdjForms ("nej" + c) + } ; + guessAdjForms : Str -> AdjForms = \s -> case s of { _ + "ý" => mladyAdjForms s ; _ + "í" => jarniAdjForms s ; @@ -652,32 +716,133 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { --------------------- -- Verbs - VerbForms : Type = { ---- TODO more forms to add + -- Public input schema of ParadigmsCze.VerbPrincipalParts. Keep existing + -- record literals valid: derived forms belong in VerbForms; additional + -- principal parts need a new constructor or overload. + PositiveVerbForms : Type = { inf, + impsg2, imppl1, imppl2, pressg1, pressg2, pressg3, prespl1, prespl2, prespl3, - pastpartsg, pastpartpl, ----- passpart, - negpressg3 : Str -- matters only for copula + pastpartsg, pastpartpl : Str } ; + VerbForms : Type = PositiveVerbForms ** { + negpressg1,negpressg2,negpressg3,negprespl1,negprespl2,negprespl3, + negimpsg2,negimppl1,negimppl2,refl : Str ; isRefl : Bool + } ; + + -- Prefix at lexical construction time, so ordinary spelling also parses. + withNeg : PositiveVerbForms -> VerbForms = \v -> v ** { + refl = [] ; isRefl = False ; + negpressg1 = "ne" + v.pressg1 ; negpressg2 = "ne" + v.pressg2 ; + negpressg3 = "ne" + v.pressg3 ; + negprespl1 = "ne" + v.prespl1 ; negprespl2 = "ne" + v.prespl2 ; + negprespl3 = "ne" + v.prespl3 ; + negimpsg2 = "ne" + v.impsg2 ; negimppl1 = "ne" + v.imppl1 ; negimppl2 = "ne" + v.imppl2 + } ; + + -- Vocalization depends on the next realized token, not on the noun head. + -- These environments choose a neutral standard form; other clusters can + -- admit stylistic variants (https://prirucka.ujc.cas.cz/?id=770). + vPreposition : Str = + let continuation : Str -> Strs = \p -> strs { + p+"a"; p+"á"; p+"b"; p+"c"; p+"č"; p+"d"; p+"ď"; p+"e"; p+"é"; p+"ě"; + p+"f"; p+"g"; p+"h"; p+"i"; p+"í"; p+"j"; p+"k"; p+"l"; p+"m"; p+"n"; + p+"ň"; p+"o"; p+"ó"; p+"p"; p+"q"; p+"r"; p+"ř"; p+"s"; p+"š"; p+"t"; + p+"ť"; p+"u"; p+"ú"; p+"ů"; p+"v"; p+"w"; p+"x"; p+"y"; p+"ý"; p+"z"; p+"ž" + } ; + longerMe : Strs = continuation "mě" ; + longerMeCapital : Strs = continuation "Mě" + in pre { + -- pre matches prefixes: apply the měst- default to derived words too, + -- then distinguish pronoun mě from longer words such as měna and měřítko. + "měst" | "Měst" => "ve" ; + longerMe => "v" ; + longerMeCapital => "v" ; + "mlýn" | "Mlýn" => "ve" ; + "v" | "V" | "f" | "F" | "mě" | "mně" | "mne" | "mz" | "dv" | "čt" | "tř" | "hř" => "ve" ; + "sb" | "sc" | "sd" | "sf" | "sh" | "sk" | "sl" | "sm" | "sn" | "sp" | "st" | "sv" | + "zb" | "zd" | "zh" | "zk" | "zl" | "zm" | "zn" | "zv" | + "šk" | "šp" | "št" | "šv" | "Šk" | "Šp" | "Št" | "Šv" | "žd" | "žl" | "žr" => "ve" ; + "Mě" | "Mně" | "Mne" | "Mz" | "Dv" | "Čt" | "Tř" | "Hř" | "Sb" | "Sc" | "Sd" | "Sf" | "Sh" | "Sk" | "Sl" | "Sm" | "Sn" | "Sp" | "St" | "Sv" | "Zb" | "Zd" | "Zh" | "Zk" | "Zl" | "Zm" | "Zn" | "Zv" | "Žd" | "Žl" | "Žr" => "ve" ; + _ => "v" + } ; + + -- s/z share these common environments; v has a different profile. + -- These defaults do not enumerate every lexical/style variant. + szPreposition : Str -> Str -> Str = \bare,vocalized -> pre { + "s" | "z" | "š" | "ž" | "mn" | "mz" | "vš" | "vs" | "vz" | "vč" | "dv" | "čt" | "tř" | "ps" | + "S" | "Z" | "Š" | "Ž" | "Mn" | "Mz" | "Vš" | "Vs" | "Vz" | "Vč" | "Dv" | "Čt" | "Tř" | "Ps" => vocalized ; + _ => bare + } ; + + sPreposition : Str = szPreposition "s" "se" ; + zPreposition : Str = szPreposition "z" "ze" ; + ComplementCase : Type = {s : Str ; c : Case ; hasPrep : Bool} ; - verbAgr : VerbForms -> Agr -> Bool -> Str ---- TODO tenses - = \vf,a,b -> case a of { - Ag _ Sg P1 => vf.pressg1 ; - Ag _ Sg P2 => vf.pressg2 ; - Ag _ Sg P3 => case b of { - True => vf.pressg3 ; - False => vf.negpressg3 -- matters only for copula - } ; - Ag _ Pl P1 => vf.prespl1 ; - Ag _ Pl P2 => vf.prespl2 ; - Ag _ Pl P3 => vf.prespl3 + hasCliticComplement : ComplementCase -> Bool -> Bool = \p,hasClit -> + case of { + => True ; + _ => False } ; - copulaVerbForms : VerbForms = { + -- Dative precedes genitive/accusative; otherwise retain argument order. + cliticBefore : Case -> Case -> Bool = \first,second -> case of { + => False ; + _ => True + } ; + + -- Full complements: prepositions select n-forms, bare cases select j-forms. + -- Clitic eligibility and placement are handled separately by the caller. + fullComplement : ComplementCase -> (Case => Str) -> (Case => Str) -> Str = + \p,bare,prep -> p.s ++ case p.hasPrep of { + True => prep ! p.c ; False => bare ! p.c + } ; + + verbAgr : VerbForms -> Agr -> Bool -> Str + = \vf,a,b -> case of { + => vf.pressg1 ; => vf.negpressg1 ; + => vf.pressg2 ; => vf.negpressg2 ; + => vf.pressg3 ; => vf.negpressg3 ; + => vf.prespl1 ; => vf.negprespl1 ; + => vf.prespl2 ; => vf.negprespl2 ; + => vf.prespl3 ; => vf.negprespl3 + } ; + + imperativeAgr : VerbForms -> Agr -> Bool -> Str = \v,a,pos -> case of { + => v.impsg2 ; => v.negimpsg2 ; + => v.imppl1 ; => v.negimppl1 ; + <_,True> => v.imppl2 ; <_,False> => v.negimppl2 + } ; + + + -- s is the ordinary order; fronted places the clitics first, ready for + -- an external host. clit/body retain the pieces needed by further fronting. + -- Keep complete orders as well as pieces so markup can enclose a sentence + -- in either order without leaking the moved clitics outside its scope. + Sentence : Type = {s,fronted,clit,body : Str ; clitPresent : Bool} ; + sentence : Bool -> Str -> Str -> Str -> Sentence = \present,first,clit,rest -> { + s = first ++ clit ++ rest ; fronted = clit ++ first ++ rest ; + clit = clit ; body = first ++ rest ; clitPresent = present + } ; + frontSentence : Str -> Sentence -> Sentence = \first,s -> s ** { + s = first ++ s.fronted ; fronted = s.clit ++ first ++ s.body ; + body = first ++ s.body + } ; + prefixSentence : Str -> Sentence -> Sentence = \first,s -> s ** { + s = first ++ s.s ; fronted = s.clit ++ first ++ s.body ; + body = first ++ s.body + } ; + -- Appended clauses retain their own domains; only s's clitics can move. + appendSentence : Sentence -> Str -> Sentence = \s,last -> s ** { + s = s.s ++ last ; fronted = s.fronted ++ last ; body = s.body ++ last + } ; + + copulaVerbForms : VerbForms = (withNeg { inf = "být" ; + impsg2 = "buď" ; imppl1 = "buďme" ; imppl2 = "buďte" ; pressg1 = "jsem" ; pressg2 = "jsi" ; pressg3 = "je" ; @@ -686,14 +851,14 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { prespl3 = "jsou" ; pastpartsg = "byl" ; pastpartpl = "byli" ; - negpressg3 = "ní" ; -- ne is added to this - } ; + }) ** {negpressg3 = "není"} ; - haveVerbForms : VerbForms = { + haveVerbForms : VerbForms = withNeg { inf = "mít" ; + impsg2 = "měj" ; imppl1 = "mějme" ; imppl2 = "mějte" ; pressg1 = "mám" ; pressg2 = "máš" ; - pressg3, negpressg3 = "má" ; + pressg3 = "má" ; prespl1 = "máme" ; prespl2 = "máte" ; prespl3 = "mají" ; @@ -709,11 +874,12 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { kupo = Predef.tk 3 kupovat ; kupu = Predef.tk 1 kupo + "u" in - { + withNeg { inf = kupovat ; + impsg2 = kupu + "j" ; imppl1 = kupu + "jme" ; imppl2 = kupu + "jte" ; pressg1 = kupu + "ji" ; --- kupuju pressg2 = kupu + "ješ" ; - pressg3, negpressg3 = kupu + "je" ; + pressg3 = kupu + "je" ; prespl1 = kupu + "jeme" ; prespl2 = kupu + "jete" ; prespl3 = kupu + "jí" ; --- kupujou @@ -725,11 +891,12 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { let kry = shortenVowel (Predef.tk 1 krýt) ; in - { + withNeg { inf = krýt ; + impsg2 = kry + "j" ; imppl1 = kry + "jme" ; imppl2 = kry + "jte" ; pressg1 = kry + "ji" ; pressg2 = kry + "ješ" ; - pressg3, negpressg3 = kry + "je" ; + pressg3 = kry + "je" ; prespl1 = kry + "jeme" ; prespl2 = kry + "jete" ; prespl3 = kry + "jí" ; @@ -747,12 +914,13 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { dat, cdat,pdat, loc, ins,pins : Str ; - a : Agr + a : Agr ; isDrop : Bool } ; personalPron : Agr -> PronForms = \a -> - {a = a ; cnom = []} ** + {a = a ; cnom = [] ; isDrop = False} ** case a of { + AgQuant _ => {nom,gen,cgen,pgen,acc,cacc,pacc,dat,cdat,pdat,loc,ins,pins = nonExist} ; Ag _ Sg P1 => { nom = "já" ; gen,acc,pgen,pacc = "mne" ; @@ -783,9 +951,10 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { } ; Ag Fem Sg P3 => { nom = "ona" ; - gen = "její" ; - dat,acc,cgen,cacc,cdat,ins = "ji" ; - pgen,pdat,pacc,loc,pins = "ní" ; + gen,dat,cgen,cdat,ins = "jí" ; + acc,cacc = "ji" ; + pacc = "ni" ; + pgen,pdat,loc,pins = "ní" ; } ; Ag Neutr Sg P3 => { nom = "ono" ; @@ -810,7 +979,7 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { dat,cdat,pdat = "nám" ; ins,pins = "námi" ; } ; - Ag _ Pl P2 => { + Ag _ Pl P2 | AgPol _ => { nom = "vy" ; gen,acc, cgen,cacc, @@ -821,7 +990,8 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { } ; Ag g Pl P3 => { nom = case g of { - Masc _ => "oni" ; + Masc Anim => "oni" ; + Masc Inanim => "ony" ; Fem => "ony" ; Neutr => "ona" } ; @@ -842,31 +1012,33 @@ adjFormsAdjective : AdjForms -> Adjective = \afs -> { Ag _ Sg P1 => mladyAdjForms "my" ** {msnom = "můj" ; pdat = "mým"} ; --- alts: moje, moji,... Ag _ Sg P2 => mladyAdjForms "tvy" ** {msnom = "tvůj" ; pdat = "tvým"} ; - Ag _ Pl P1 => jarniAdjForms "naše" ** { - msnom = "náš" ; - fsgen,mpnom = "naši" ; - fsins = "naší" ; - pdat, msins = "našim" ; - pgen = "našich" ; - pins = "našimi" ; - } ; - Ag _ Pl P2 => jarniAdjForms "vaše" ** { - msnom = "váš" ; - fsgen,mpnom = "vaši" ; - fsins = "vaší" ; - pdat, msins = "vašim" ; - pgen = "vašich" ; - pins = "vašimi" ; - } ; + Ag _ Pl P1 => nasPossessiveForms "náš" "naš" ; + Ag _ Pl P2 | AgPol _ => nasPossessiveForms "váš" "vaš" ; Ag Fem Sg P3 => jarniAdjForms "její" ** {pdat = "jejím"} ; Ag (Masc _ | Neutr) Sg P3 => invarDemPronForms "jeho" ** {pdat = "jeho"} ; - Ag _ Pl P3 => invarDemPronForms "jejich" ** {pdat = "jejich"} + Ag _ Pl P3 | AgQuant _ => invarDemPronForms "jejich" ** {pdat = "jejich"} } ; + -- Náš/váš distinguish feminine accusative naši from oblique naší, + -- and singular instrumental naším from plural dative našim. + nasPossessiveForms : Str -> Str -> DemPronForms = \nas,nasStem -> { + msnom = nas ; + fsnom,nsnom,fpnom = nasStem + "e" ; + msgen = nasStem + "eho" ; + fsgen,fsins = nasStem + "í" ; + msdat = nasStem + "emu" ; + fsacc,mpnom = nasStem + "i" ; + msloc = nasStem + "em" ; + msins = nasStem + "ím" ; + pgen = nasStem + "ich" ; + pdat = nasStem + "im" ; + pins = nasStem + "imi" + } ; + reflPossessivePron : DemPronForms = mladyAdjForms "svy" ** {msnom = "svůj" ; pdat = "svým"} ; mkPron : Agr -> PronForms ** {poss : DemPronForms} = \a -> @@ -918,7 +1090,8 @@ oper Determiner : Type = { s : Gender => Case => Str ; - size : NumSize + size : NumSize ; -- number and case of the counted noun + head : NumHead -- agreement of a quantifier preceding the numeral } ; mkDemPronForms : Str -> DemPronForms = \t -> { @@ -984,50 +1157,38 @@ oper adjAdj = adjFormsAdjective demAdj in { s = \\g,c => adjAdj.s ! g ! Sg ! c ; - size = size + size = size ; head = CountedHead } ; -- example: number 1 oneNumeral : Determiner = numeralFormsDeterminer ((mkDemPronForms "jedn") ** {msnom = "jeden"}) Num1 ; - -- numbers 2,3,4 ---- to check if everything comes out right with the determiner type - twoNumeral : Determiner = - let forms = { - msnom = "dva" ; fsnom, nsnom, fsacc = "dvě" ; - msgen, fsgen, msloc = "dvou" ; - msdat, msins, fsins = "dvěma" - } - in numeralFormsDeterminer forms Num2_4 ; - - threeNumeral : Determiner = - let forms = { - msnom, fsnom, nsnom, fsacc, msgen, fsgen = "tři" ; - msdat = "třem" ; - msloc = "třech" ; - msins,fsins = "třemi" ; - } - in numeralFormsDeterminer forms Num2_4 ; - - fourNumeral : Determiner = - let forms = { - msnom, fsnom, nsnom, fsacc = "čtyři" ; - msgen, fsgen = "čtyř" ; - msdat = "čtyřem" ; - msloc = "čtyřech" ; - msins,fsins = "čtyřmi" ; - } - in numeralFormsDeterminer forms Num2_4 ; + -- Unlike adjectives, 2--4 do not use the genitive for animate accusatives. + twoNumeral : Determiner = { + s = \\g,c => case c of { + Nom|Acc|Voc => case g of {Masc _ => "dva" ; _ => "dvě"} ; + Gen|Loc => "dvou" ; Dat|Ins => "dvěma" + } ; size = Num2_4 ; head = CountedHead + } ; + threeNumeral : Determiner = { + s = \\_,c => case c of { + Nom|Acc|Voc => "tři" ; Gen => "tří" ; Dat => "třem" ; Loc => "třech" ; Ins => "třemi" + } ; size = Num2_4 ; head = CountedHead + } ; + fourNumeral : Determiner = { + s = \\_,c => case c of { + Nom|Acc|Voc => "čtyři" ; Gen => "čtyř" ; Dat => "čtyřem" ; Loc => "čtyřech" ; Ins => "čtyřmi" + } ; size = Num2_4 ; head = CountedHead + } ; -- for the numbers 5 upwards - regNumeral : Str -> Str -> Determiner = \pet,peti -> - let forms = { - msnom,fsnom,nsnom = pet ; - msgen, fsgen, msdat, fsacc, msloc, msins, fsins = peti - } - in numeralFormsDeterminer forms Num5 ; + regNumeral : Str -> Str -> Determiner = \pet,peti -> { + s = \\_,c => case c of {Nom | Acc | Voc => pet ; _ => peti} ; + size = Num5 ; head = CountedHead + } ; invarDeterminer : Str -> NumSize -> Determiner = \sto,size -> - regNumeral sto sto ; + (regNumeral sto sto) ** {size = size} ; invarNumeral : Str -> Determiner = \s -> invarDeterminer s Num5 ; @@ -1035,22 +1196,105 @@ oper -- combining nouns with numerals param - NumSize = Num1 | Num2_4 | Num5 ; -- CEG 6.1 + NumSize = Num1 | Num2_4 | Num5 | NumScale ; -- CEG 6.1 + -- A scale noun has its own agreement: tímto tisícem vs. těchto pěti tisíců. + -- Nested scales retain the outer head: tato dvě stě tisíc korun. + ScaleAgreement = QuantifiedScale | NominalScale ; + NumHead = CountedHead | ScaleHead Gender NumSize ScaleAgreement ; + -- NP modifiers have gender/number/case agreement, never verbal person. + ModifierAgr = Mod Gender Number | ModQuant Gender ; oper + quantifierForm : Adjective -> Determiner -> Gender -> Case -> Str = \q,num,g,c -> + case num.head of { + CountedHead => q.s ! g ! numSizeNumber num.size ! countCase num.size c ; + ScaleHead sg size _ => q.s ! sg ! numSizeNumber size ! countCase size c + } ; + + quantifyNumeral : Adjective -> Determiner -> Determiner = \q,num -> num ** { + s = \\g,c => quantifierForm q num g c ++ num.s ! g ! c + } ; + + -- Predeterminers agree with the NP head, including a quantified head in + -- the genitive. This differs from clause agreement for e.g. tisíc korun. + numeralModAgr : Gender -> Determiner -> ModifierAgr = \g,num -> case num.head of { + CountedHead => modifierAgr (numSizeAgr g num.size P3) ; + ScaleHead sg size _ => modifierAgr (numSizeAgr sg size P3) + } ; + + modifierAgr : Agr -> ModifierAgr = \a -> case a of { + Ag g n _ => Mod g n ; AgPol g => Mod g Sg ; AgQuant g => ModQuant g + } ; + + predetForm : Adjective -> ModifierAgr -> Case -> Str = \pred,a,c -> case a of { + Mod g n => pred.s ! g ! n ! c ; + ModQuant g => pred.s ! g ! Pl ! countCase Num5 c + } ; + + -- Keep the boundary for my všichni doma, also after AdvNP. Complete forms + -- retain a single markup wrapper; insertion can use separately marked pieces. + NPForms : Type = { + s,prep,before,prepBefore : Case => Str ; + after : Str + } ; + + npForms : (Case => Str) -> (Case => Str) -> NPForms = \s,prep -> { + s,before = s ; prep,prepBefore = prep ; after = [] + } ; + + appendNPForms : NPForms -> Str -> NPForms = \np,adv -> np ** { + s = \\c => np.s ! c ++ adv ; prep = \\c => np.prep ! c ++ adv ; + after = np.after ++ adv + } ; + + predetNPForms : Bool -> (Case => Str) -> NPForms -> NPForms = \post,pred,np -> + case post of { + True => np ** { + s = \\c => np.before ! c ++ pred ! c ++ np.after ; + prep = \\c => np.prepBefore ! c ++ pred ! c ++ np.after ; + before = \\c => np.before ! c ++ pred ! c ; + prepBefore = \\c => np.prepBefore ! c ++ pred ! c + } ; + False => np ** { + s = \\c => pred ! c ++ np.s ! c ; + prep = \\c => pred ! c ++ np.prep ! c ; + before = \\c => pred ! c ++ np.before ! c ; + prepBefore = \\c => pred ! c ++ np.prepBefore ! c + } + } ; + + nounGender : Noun -> Number -> Gender = \cn,n -> case n of { + Sg => cn.g ; Pl => cn.gPl + } ; + + countCase : NumSize -> Case -> Case = \n,c -> case of { + | => Gen ; _ => c + } ; + numSizeForm : (Number => Case => Str) -> NumSize -> Case -> Str = \cns,n,c -> case n of { Num1 => cns ! Sg ! c ; + NumScale => cns ! Pl ! Gen ; Num2_4 => cns ! Pl ! c ; Num5 => case c of { - Nom | Acc => cns ! Pl ! Gen ; + Nom | Acc | Voc => cns ! Pl ! Gen ; _ => cns ! Pl ! c } } ; + -- Clause agreement is independent of the counted noun's genitive case. + -- Millions/billions normally agree with their scale head; hundreds and + -- thousands use quantified agreement by default. Five million still has + -- a quantified head, while two million has a plural nominal head. + numeralAgr : Gender -> Determiner -> Person -> Agr = \g,num,p -> + case num.head of { + ScaleHead sg size NominalScale => numSizeAgr sg size p ; + _ => numSizeAgr g num.size p + } ; + numSizeAgr : Gender -> NumSize -> Person -> Agr = \g,ns,p -> case ns of { - Num5 => Ag Neutr Sg p ; -- essential grammar 6.1.4 + Num5 | NumScale => AgQuant g ; -- essential grammar 6.1.4 Num2_4 => Ag g Pl p ; Num1 => Ag g Sg p } ; diff --git a/src/czech/SentenceCze.gf b/src/czech/SentenceCze.gf index 661df8e6..29e4dcfb 100644 --- a/src/czech/SentenceCze.gf +++ b/src/czech/SentenceCze.gf @@ -1,51 +1,42 @@ -concrete SentenceCze of Sentence = CatCze ** - open Prelude, ResCze in { - +concrete SentenceCze of Sentence = CatCze ** open Prelude, ResCze in { lin PredVP np vp = { - subj = case np.hasClit of { - True => np.clit ! Nom ; -- pro-drop - False => np.s ! Nom - } ; - verb = vp.verb ; - clit = vp.clit ! np.a ; - compl = vp.compl ! np.a ; - a = np.a ; + -- A dropped subject still contributes an empty constituent, so PGF + -- retains the pronoun and constrains its person through agreement. + subj = case np.isDrop of {True => np.clit ! Nom ; False => np.s ! Nom} ; + verb = vp.verb ; clit = vp.clit ! np.a ; compl = vp.compl ! np.a ; + a = np.a ; isDrop = np.isDrop ; clitPresent = vp.clitPresent } ; - UseCl temp pol cl = { - s = temp.s ++ cl.subj ++ cl.clit ++ pol.s ++ verbAgr cl.verb cl.a pol.p ++ cl.compl ; - } ; + UseCl temp pol cl = let v = pol.s ++ verbAgr cl.verb cl.a pol.p in + case cl.isDrop of { + True => sentence cl.clitPresent (temp.s ++ cl.subj ++ v) cl.clit cl.compl ; + False => sentence cl.clitPresent (temp.s ++ cl.subj) cl.clit (v ++ cl.compl) + } ; - --- TODO is inversion the standard? ; add indirect questions - UseQCl temp pol cl = { - s = temp.s ++ cl.clit ++ pol.s ++ verbAgr cl.verb cl.a pol.p ++ cl.subj ++ cl.compl ; + UseQCl temp pol cl = let v = pol.s ++ verbAgr cl.verb cl.a pol.p in { + s = temp.s ++ case cl.yesNo of { + True => v ++ cl.clit ++ cl.subj ++ cl.compl ; + False => cl.q ++ cl.clit ++ v ++ cl.subj ++ cl.compl + } ; + ind = temp.s ++ case cl.yesNo of { + True => "jestli" ++ cl.clit ++ cl.subj ++ v ++ cl.compl ; + False => cl.q ++ cl.clit ++ v ++ cl.subj ++ cl.compl + } } ; UseRCl temp pol rcl = { - s = \\a => temp.s ++ - rcl.subj ! a ++ rcl.clit ! a ++ - pol.s ++ verbAgr rcl.verb a pol.p ++ - rcl.compl ! a ; + s = \\a => temp.s ++ rcl.subj ! a ++ rcl.clit ! a ++ + pol.s ++ verbAgr rcl.verb a pol.p ++ rcl.compl ! a } ; - --- no imperative in VerbForms yet; the 1st person plural present is used --- instead, which is the normal register in mathematical Czech --- ("předpokládáme, že ..." = "we assume that ...") - ImpVP vp = let agr = Ag (Masc Anim) Pl P1 in - {s = vp.clit ! agr ++ verbAgr vp.verb agr True ++ vp.compl ! agr} ; - - EmbedS s = {s = "že" ++ s.s} ; - - EmbedQS qs = {s = qs.s} ; - + ImpVP vp = {s = \\pos,a => + imperativeAgr vp.verb a pos ++ vp.clit ! a ++ vp.compl ! a + } ; + EmbedS s = {s = (frontSentence "že" s).s} ; + EmbedQS qs = {s = qs.ind} ; EmbedVP vp = let agr = Ag Neutr Sg P3 in - {s = vp.clit ! agr ++ vp.verb.inf ++ vp.compl ! agr} ; - - AdvS a s = {s = a.s ++ s.s} ; - - ExtAdvS a s = {s = a.s ++ SOFT_BIND ++ "," ++ s.s} ; - - SSubjS a subj b = {s = a.s ++ SOFT_BIND ++ "," ++ subj.s ++ b.s} ; - + {s = vp.verb.inf ++ vp.clit ! agr ++ vp.compl ! agr} ; + AdvS a s = frontSentence a.s s ; + ExtAdvS a s = prefixSentence (a.s ++ SOFT_BIND ++ ",") s ; + SSubjS a subj b = appendSentence a (SOFT_BIND ++ "," ++ (frontSentence subj.s b).s) ; } diff --git a/src/czech/StructuralCze.gf b/src/czech/StructuralCze.gf index b56c7ffa..cbfdcef9 100644 --- a/src/czech/StructuralCze.gf +++ b/src/czech/StructuralCze.gf @@ -5,27 +5,38 @@ concrete StructuralCze of Structural = CatCze ** oper adjDet : AdjForms -> Determiner = \afs -> { s = \\g,c => (adjFormsAdjective afs).s ! g ! Sg ! c ; - size = Num1 + size = Num1 ; head = CountedHead } ; lin - all_Predet = {s = "všechny"} ; + all_Predet = {s = \\g,n,c => case of { + => "všech" ; => "všem" ; => "všemi" ; + => "všichni" ; + => "všechna" ; => "všechny" ; + => "všechna" ; => "všechnu" ; + => "vší" ; + | => "všeho" ; + => "všemu" ; => "všem" ; => "vším" ; + => "všechno" ; => "všechen" + } ; postPron = True} ; + only_Predet = {s = \\_,_,_ => "jen" ; postPron = False} ; and_Conj = mkConj "a" ; both7and_DConj = {s1 = "jak" ; s2 = "tak"} ; between_Prep = mkPrep "mezi" Ins ; by8agent_Prep = mkPrep "od" Gen ; ---- TODO this means "from", there might be no good translation by8means_Prep = mkPrep "pomocí" Gen ; - can_VV = { + can_VV = mkModalVV (lin V (withNeg { inf = "moci" ; + impsg2,imppl1,imppl2 = nonExist ; pressg1 = "mohu" ; pressg2 = "můžeš" ; - pressg3, negpressg3 = "může" ; + pressg3 = "může" ; prespl1 = "můžeme" ; prespl2 = "můžete" ; prespl3 = "mohou" ; pastpartsg = "mohl" ; pastpartpl = "mohli" ; - } ; + })) ; either7or_DConj = {s1 = "buď" ; s2 = "nebo"} ; every_Det = adjDet (mladyAdjForms "každý") ; few_Det = invarNumeral "málo" ; -- CEG 6.8 --- TODO genitive mála @@ -37,20 +48,24 @@ lin that_Subj = {s = "že"} ; under_Prep = mkPrep "pod" Ins ; where_IAdv = {s = "kde"} ; - from_Prep = mkPrep (pre {"s"|"z" => "ze" ; _ => "z"}) Gen ; ---- consonant clusters - have_V2 = mkV2 haveVerbForms ; - in_Prep = mkPrep (pre {"v"|"m" => "ve" ; _ => "v"}) Loc ; ---- + from_Prep = mkPrep zPreposition Gen ; + have_V2 = mkV2 ; + in_Prep = v_Prep Loc ; many_Det = regNumeral "mnoho" "mnoha" ; -- CEG 6.8 ---- or_Conj = mkConj "nebo" ; somePl_Det = regNumeral "několik" "několika" ; -- CEG 6.8 ---- - something_NP = {s,clit,prep = \\c => "ně" + coForms ! c ; a = Ag Neutr Sg P3 ; hasClit = False} ; -- CEG 5.6.3 - possess_Prep = mkPrep "" Gen ; + something_NP = + let s : Case => Str = \\c => "ně" + coForms ! c in npForms s s ** { + clit = s ; a = Ag Neutr Sg P3 ; m = Mod Neutr Sg ; + hasClit = False ; isDrop = False ; isPron = False + } ; -- CEG 5.6.3 + possess_Prep = mkPrep Gen ; that_Quant = demPronFormsAdjective (mkDemPronForms "tamt") "" ; this_Quant = demPronFormsAdjective (mkDemPronForms "t") "to" ; to_Prep = mkPrep "do" Gen ; - with_Prep = mkPrep (pre {"s"|"z" => "se" ; _ => "s"}) Ins ; ---- + with_Prep = mkPrep sPreposition Ins ; - i_Pron = mkPron (Ag (Masc Anim) Sg P1) ; --- to add Fem pronouns in Extend + i_Pron = mkPron (Ag (Masc Anim) Sg P1) ; youSg_Pron = mkPron (Ag (Masc Anim) Sg P2) ; he_Pron = mkPron (Ag (Masc Anim) Sg P3) ; she_Pron = mkPron (Ag Fem Sg P3) ; @@ -58,5 +73,19 @@ lin we_Pron = mkPron (Ag (Masc Anim) Pl P1) ; youPl_Pron = mkPron (Ag (Masc Anim) Pl P2) ; they_Pron = mkPron (Ag (Masc Anim) Pl P3) ; - + youPol_Pron = mkPron (AgPol (Masc Anim)) ; + want_VV = mkModalVV (mkV "chtít" "chci" "chceš" "chce" "chceme" "chcete" "chtějí" "chtěl" "chtěli" "chtěj" "chtějme" "chtějte") ; + must_VV = mkModalVV (mkV "muset" "musím" "musíš" "musí" "musíme" "musíte" "musí" "musel" "museli" nonExist nonExist nonExist) ; + can8know_VV = mkModalVV (mkV "umět" "umím" "umíš" "umí" "umíme" "umíte" "umějí" "uměl" "uměli" "uměj" "umějme" "umějte") ; + yes_Utt = {s = "ano"} ; + no_Utt = {s = "ne"} ; + please_Voc = {s = "prosím"} ; + very_AdA = mkAdA "velmi" ; + too_AdA = mkAdA "příliš" ; + how_IAdv = {s = "jak"} ; + how8much_IAdv = {s = "kolik"} ; + whatSg_IP = {s = coForms ; a = Ag Neutr Sg P3} ; + whoSg_IP = {s = kdoForms ; a = Ag (Masc Anim) Sg P3} ; + how8many_IDet = regNumeral "kolik" "kolika" ; + which_IQuant = adjFormsAdjective (guessAdjForms "který") ; } diff --git a/src/czech/SymbolCze.gf b/src/czech/SymbolCze.gf index a64c3a76..bf36266e 100644 --- a/src/czech/SymbolCze.gf +++ b/src/czech/SymbolCze.gf @@ -19,29 +19,36 @@ lin NumPN card = lin PN {s = \\c => card.s ! Neutr ! c ; g = Neutr} ; -- the numeral is an invariable label: "úroveň pět", "na úrovni pět" - CNNumNP cn card = { - s,clit,prep = \\c => cn.s ! Sg ! c ++ card.s ! cn.g ! Nom ; - a = Ag cn.g Sg P3 ; - hasClit = False ; + CNNumNP cn card = + let s : Case => Str = \\c => cn.s ! Sg ! c ++ card.s ! cn.g ! Nom in npForms s s ** { + clit = s ; + a = Ag cn.g Sg P3 ; m = Mod cn.g Sg ; + hasClit = False ; isDrop = False ; isPron = False ; } ; - CNIntNP cn i = { - s,clit,prep = \\c => cn.s ! Sg ! c ++ i.s ; - a = Ag cn.g Sg P3 ; - hasClit = False ; + CNIntNP cn i = + let s : Case => Str = \\c => cn.s ! Sg ! c ++ i.s in npForms s s ** { + clit = s ; + a = Ag cn.g Sg P3 ; m = Mod cn.g Sg ; + hasClit = False ; isDrop = False ; isPron = False ; } ; -- as DetCN in NounCze, with the symbols in apposition - CNSymbNP det cn xs = { - s,clit,prep = \\c => det.s ! cn.g ! c ++ numSizeForm cn.s det.size c ++ xs.s ; - a = numSizeAgr cn.g det.size P3 ; - hasClit = False ; + CNSymbNP det cn xs = + let s : Case => Str = \\c => det.s ! nounGender cn (numSizeNumber det.size) ! c ++ numSizeForm cn.s det.size c ++ xs.s + in npForms s s ** { + clit = s ; + a = numeralAgr (nounGender cn (numSizeNumber det.size)) det P3 ; + m = numeralModAgr (nounGender cn (numSizeNumber det.size)) det ; + hasClit = False ; isDrop = False ; isPron = False ; } ; - SymbS sy = sy ; + SymbS sy = sentence False sy.s [] [] ; - SymbNum sy = {s = \\_,_ => sy.s ; size = Num5} ; -- "n čísel", like numerals from 5 up - SymbOrd sy = {s = glue sy.s "-tý"} ; ---- Ord is still an uninflected string + SymbNum sy = invarNumeral sy.s ; -- "n čísel", like numerals from 5 up + SymbOrd sy = {s = \\g,n,c => + glue sy.s ((adjFormsAdjective (mladyAdjForms "-tý")).s ! g ! n ! c) + } ; oper symbolPN : Str -> PN diff --git a/src/czech/TenseCze.gf b/src/czech/TenseCze.gf index 667cc1f2..c5b225e1 100644 --- a/src/czech/TenseCze.gf +++ b/src/czech/TenseCze.gf @@ -6,7 +6,7 @@ concrete TenseCze of Tense = in { lin PNeg = { - s = "ne" ++ Predef.BIND ; + s = [] ; p = False } ; PPos = { diff --git a/src/czech/VerbCze.gf b/src/czech/VerbCze.gf index cc98d4a3..b549cae0 100644 --- a/src/czech/VerbCze.gf +++ b/src/czech/VerbCze.gf @@ -2,62 +2,73 @@ concrete VerbCze of Verb = CatCze ** open ResCze, Prelude in { lin UseV v = { - verb = v ; - clit,compl = \\_ => [] + verb = v ; clitPresent = v.isRefl ; + clit = \\_ => v.refl ; compl = \\_ => [] } ; - ComplSlash vps np = case of { - => vps ** { - clit = \\a => vps.clit ! a ++ np.clit ! vps.c.c ; + ComplSlash vps np = case hasCliticComplement vps.c np.hasClit of { + True => vps ** { + clitPresent = True ; + clit = \\a => vps.clit ! a ++ np.clit ! vps.c.c ++ vps.clitAfter ! a ; compl = \\a => vps.compl ! a ++ vps.ind ! a } ; - _ => vps ** { - compl = \\a => vps.compl ! a ++ vps.c.s ++ np.s ! vps.c.c ++ vps.ind ! a + False => vps ** { + clit = \\a => vps.clit ! a ++ vps.clitAfter ! a ; + compl = \\a => vps.compl ! a ++ fullComplement vps.c np.s np.prep ++ vps.ind ! a } } ; SlashV2a v = { - verb = v ; - clit,compl = \\_ => [] ; + verb = v ; clitPresent = v.isRefl ; + clit = \\_ => v.refl ; compl = \\_ => [] ; c = v.c ; - ind = \\_ => [] + ind = \\_ => [] ; clitAfter = \\_ => [] } ; - -- three-place verbs: c = direct object case, c2 = indirect object case - Slash2V3 v np = { -- fill the direct object, leave the indirect open - verb = v ; - clit = \\_ => [] ; - compl = \\_ => v.c.s ++ np.s ! v.c.c ; + -- Full objects retain c-before-c2 order; weak objects follow case order. + Slash2V3 v np = let + isClit = hasCliticComplement v.c np.hasClit ; + weak = case isClit of {True => np.clit ! v.c.c ; False => []} ; + before = cliticBefore v.c.c v.c2.c + in { + verb = v ; clitPresent = orB v.isRefl isClit ; + clit = \\_ => v.refl ++ case before of {True => weak ; False => []} ; + clitAfter = \\_ => case before of {True => [] ; False => weak} ; + compl = \\_ => case isClit of {True => [] ; False => fullComplement v.c np.s np.prep} ; c = v.c2 ; ind = \\_ => [] } ; - Slash3V3 v np = { -- fill the indirect object (rendered after the object slot) - verb = v ; - clit = \\_ => [] ; + Slash3V3 v np = let + isClit = hasCliticComplement v.c2 np.hasClit ; + weak = case isClit of {True => np.clit ! v.c2.c ; False => []} ; + before = cliticBefore v.c.c v.c2.c + in { + verb = v ; clitPresent = orB v.isRefl isClit ; + clit = \\_ => v.refl ++ case before of {True => [] ; False => weak} ; + clitAfter = \\_ => case before of {True => weak ; False => []} ; compl = \\_ => [] ; c = v.c ; - ind = \\_ => v.c2.s ++ np.s ! v.c2.c + ind = \\_ => case isClit of {True => [] ; False => fullComplement v.c2 np.s np.prep} } ; UseComp comp = { - verb = copulaVerbForms ; + verb = copulaVerbForms ; clitPresent = False ; clit = \\_ => [] ; compl = comp.s } ; - CompAP ap = { - s = \\a => case a of { - Ag g n p_ => ap.s ! g ! n ! Nom - } - } ; + CompAP ap = {s = ap.pred} ; CompNP np = { - s = \\a_ => np.s ! Nom ; ---- InstrC in Pol + -- An identifying NP retains its own number; only its case is selected. + s = \\a_ => np.s ! Nom ; } ; CompCN cn = { s = \\a => case a of { - Ag _ n _ => cn.s ! n ! Nom ---- InstrC also possible + Ag _ n _ => cn.s ! n ! Nom ; + AgQuant _ => cn.s ! Pl ! Ins ; -- selected formal predicative instrumental + AgPol _ => cn.s ! Sg ! Nom } } ; @@ -72,21 +83,25 @@ lin -- VerbForms has no passive participle yet, so the reflexive passive is used: -- "číslo se dělí" = "the number is divided" PassV2 v = { - verb = v ; + verb = v ; clitPresent = True ; clit = \\_ => "se" ; compl = \\_ => [] } ; ComplVV vv vp = { - verb = vv ; - clit = vp.clit ; - compl = \\a => vp.verb.inf ++ vp.compl ! a + verb = vv ; clitPresent = orB vv.isRefl (andB vv.isAux vp.clitPresent) ; + clit = \\a => vv.refl ++ case vv.isAux of {True => vp.clit ! a ; False => []} ; + compl = \\a => vp.verb.inf ++ case vv.isAux of {True => [] ; False => vp.clit ! a} ++ vp.compl ! a } ; ComplVS vs s = { - verb = vs ; - clit = \\_ => [] ; - compl = \\_ => SOFT_BIND ++ "," ++ "že" ++ s.s + verb = vs ; clitPresent = vs.isRefl ; + clit = \\_ => vs.refl ; + compl = \\_ => SOFT_BIND ++ "," ++ (frontSentence "že" s).s } ; + ComplVQ v q = { + verb = v ; clitPresent = v.isRefl ; clit = \\_ => v.refl ; + compl = \\_ => SOFT_BIND ++ "," ++ q.ind + } ; } diff --git a/tests/czech/CzeExtensionRoundTrip.gf b/tests/czech/CzeExtensionRoundTrip.gf new file mode 100644 index 00000000..cf8fb983 --- /dev/null +++ b/tests/czech/CzeExtensionRoundTrip.gf @@ -0,0 +1,39 @@ +-- A finite fragment of the Czech extensions. Subject and BareRNP limit +-- ProDrop and predeterminers to one application without recursive aliases. +abstract CzeExtensionRoundTrip = AllCzeAbs [ + Utt, S, Cl, NP, VP, VPSlash, Pron, V, V2, AP, QCl, QS, IP, + N, CN, Det, Quant, Num, IDet, Predet, Adv, Prep, Subj, Temp, Tense, Ant, Pol, + PrepNP, possess_Prep, by8agent_Prep, with_Prep, he_Pron, we_Pron, + IAdv, UttIAdv, PrepIP, QuestVP, whoSg_IP, + A, Comp, UseComp, CompAP, young_A, + UttS, UttQS, UttAdv, UseCl, UseQCl, UsePron, UseV, SlashV2a, ComplSlash, + SlashV2AP, DativeCopulaCl, DativeCopulaQCl, SubjS, + UseN, DetCN, DetQuant, DefArt, PossPron, this_Quant, NumSg, NumPl, IdetCN, UttNP, + TTAnt, TPres, ASimul, PPos, PNeg, + i_Pron, she_Pron, youPol_Pron, have_V2, love_V2, wait_V2, + year_N, child_N, woman_N, apple_N, how8many_IDet, only_Predet, all_Predet, if_Subj +] ** { + flags startcat = Utt ; + cat Subject ; BareRNP ; RNP ; ComparisonNP ; ModifiedNP ; NPModifier ; + fun + -- NPModifier excludes PrepNP, avoiding recursion through NP and Adv. + AdvPron : Pron -> NPModifier -> ModifiedNP ; + PredetNP : Predet -> ModifiedNP -> NP ; + home_Adv : NPModifier ; + ComparisonPron : Pron -> ComparisonNP ; + Compare : A -> ComparisonNP -> AP ; + ModifiedN : AP -> N -> CN ; + FullSubject : NP -> Subject ; + DroppedSubject : Pron -> Subject ; + PredSubject : Subject -> VP -> Cl ; + ReflPron : BareRNP ; + ReflPoss : Num -> CN -> BareRNP ; + UseRNP : BareRNP -> RNP ; + PredetRNP : Predet -> BareRNP -> RNP ; + ReflRNP : VPSlash -> RNP -> VP ; + like_AP : AP ; son_N : N ; wash_V : V ; + fast_A : A ; + two_Num, five_Num, twentyOne_Num, twentyTwo_Num : Num ; + twelveHundred_Num, twentyTwoHundred_Num : Num ; + writeAbout_V2 : V2 ; +} diff --git a/tests/czech/CzeExtensionRoundTripCze.gf b/tests/czech/CzeExtensionRoundTripCze.gf new file mode 100644 index 00000000..c72c87cd --- /dev/null +++ b/tests/czech/CzeExtensionRoundTripCze.gf @@ -0,0 +1,44 @@ +concrete CzeExtensionRoundTripCze of CzeExtensionRoundTrip = AllCze [ + Utt, S, Cl, NP, VP, VPSlash, Pron, V, V2, AP, QCl, QS, IP, + N, CN, Det, Quant, Num, IDet, Predet, Adv, Prep, Subj, Temp, Tense, Ant, Pol, + PrepNP, possess_Prep, by8agent_Prep, with_Prep, he_Pron, we_Pron, + IAdv, UttIAdv, PrepIP, QuestVP, whoSg_IP, + A, Comp, UseComp, CompAP, young_A, + UttS, UttQS, UttAdv, UseCl, UseQCl, UsePron, UseV, SlashV2a, ComplSlash, + SlashV2AP, DativeCopulaCl, DativeCopulaQCl, SubjS, + UseN, DetCN, DetQuant, DefArt, PossPron, this_Quant, NumSg, NumPl, IdetCN, UttNP, + TTAnt, TPres, ASimul, PPos, PNeg, + i_Pron, she_Pron, youPol_Pron, have_V2, love_V2, wait_V2, + year_N, child_N, woman_N, apple_N, how8many_IDet, only_Predet, all_Predet, if_Subj +] ** open SyntaxCze, ParadigmsCze, (E = ExtendCze), (N = NumeralCze) in { + lincat Subject, ComparisonNP, ModifiedNP = NP ; BareRNP, RNP = E.RNP ; + NPModifier = Adv ; + lin + AdvPron p adv = SyntaxCze.mkNP (SyntaxCze.mkNP p) adv ; + PredetNP pred np = SyntaxCze.mkNP pred np ; + home_Adv = ParadigmsCze.mkAdv "doma" ; + ComparisonPron = UsePron ; + Compare a np = mkAP a np ; + ModifiedN ap n = mkCN ap n ; + FullSubject np = np ; + DroppedSubject p = UsePron (E.ProDrop p) ; + PredSubject = PredVP ; + ReflPron = E.ReflPron ; + ReflPoss = E.ReflPoss ; + UseRNP rnp = rnp ; + PredetRNP = E.PredetRNP ; + ReflRNP = E.ReflRNP ; + like_AP = shortAP "rád" "ráda" "rádo" "rádi" "rády" "ráda" ; + fast_A = mkA "rychlý" ; + son_N = panN "syn" ; + wash_V = seV (krytV "mýt") ; + two_Num = mkNum "2" ; + five_Num = mkNum "5" ; + twentyOne_Num = mkNum "21" ; + twentyTwo_Num = mkNum "22" ; + twelveHundred_Num = mkNum (N.num + (N.pot3plus (N.pot1as2 (N.pot0as1 N.pot01)) (N.pot2 (N.pot0 N.n2)))) ; + twentyTwoHundred_Num = mkNum (N.num + (N.pot3plus (N.pot1as2 (N.pot0as1 (N.pot0 N.n2))) (N.pot2 (N.pot0 N.n2)))) ; + writeAbout_V2 = LexiconCze.write_V2 ** {c = mkPrep "o" locative} ; +} diff --git a/tests/czech/CzeMarkup.gf b/tests/czech/CzeMarkup.gf new file mode 100644 index 00000000..c6b01c74 --- /dev/null +++ b/tests/czech/CzeMarkup.gf @@ -0,0 +1,9 @@ +resource CzeMarkup = CzeTests ** open Prelude, SyntaxCze, ParadigmsCze, + (L = LexiconCze), (E = ExtendCze), (G = GrammarCze), (M = MarkupCze) in { +oper + washing : S = mkS (mkCl (mkNP (E.ProDrop i_Pron)) wash_V) ; + markedWashing : S = M.MarkupS M.b_Mark washing ; + todayMarked : S = G.AdvS L.today_Adv markedWashing ; + swimming : S = mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.swim_V) ; + loving : S = mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.love_V2 (mkNP she_Pron)) ; +} diff --git a/tests/czech/CzeRoundTrip.gf b/tests/czech/CzeRoundTrip.gf new file mode 100644 index 00000000..52b84c66 --- /dev/null +++ b/tests/czech/CzeRoundTrip.gf @@ -0,0 +1,16 @@ +-- A finite RGL fragment: all parses can be checked without truncation. +abstract CzeRoundTrip = Lang [ + Utt, S, Cl, NP, VP, VPSlash, Pron, V2, V3, A, AP, Comp, Imp, + Adv, Prep, N, CN, Det, Quant, Num, + PrepNP, UttAdv, UseN, DetCN, DetQuant, DefArt, NumSg, in_Prep, city_N, + Temp, Tense, Ant, Pol, + UttS, UseCl, PredVP, UsePron, ComplSlash, SlashV2a, Slash2V3, Slash3V3, + UseComp, CompAP, PositA, ImpVP, UttImpSg, UttImpPl, UttImpPol, + TTAnt, TPres, ASimul, PPos, PNeg, + i_Pron, youSg_Pron, he_Pron, she_Pron, youPl_Pron, youPol_Pron, + love_V2, read_V2, young_A +] ** { + flags startcat = Utt ; + fun buy_V3 : V3 ; + fun currency_N : N ; vAcc_Prep : Prep ; +} diff --git a/tests/czech/CzeRoundTripCze.gf b/tests/czech/CzeRoundTripCze.gf new file mode 100644 index 00000000..72e25784 --- /dev/null +++ b/tests/czech/CzeRoundTripCze.gf @@ -0,0 +1,18 @@ +concrete CzeRoundTripCze of CzeRoundTrip = LangCze [ + Utt, S, Cl, NP, VP, VPSlash, Pron, V2, V3, A, AP, Comp, Imp, + Adv, Prep, N, CN, Det, Quant, Num, + PrepNP, UttAdv, UseN, DetCN, DetQuant, DefArt, NumSg, in_Prep, city_N, + Temp, Tense, Ant, Pol, + UttS, UseCl, PredVP, UsePron, ComplSlash, SlashV2a, Slash2V3, Slash3V3, + UseComp, CompAP, PositA, ImpVP, UttImpSg, UttImpPl, UttImpPol, + TTAnt, TPres, ASimul, PPos, PNeg, + youSg_Pron, he_Pron, she_Pron, youPl_Pron, youPol_Pron, + love_V2, read_V2, young_A +] ** open ParadigmsCze in { + lin + buy_V3 = mkV3 (kupovatV "kupovat") ; + currency_N = zenaN "měna" ; + vAcc_Prep = v_Prep accusative ; + -- Select the short prepositional accusative without a duplicate Pron. + i_Pron = StructuralCze.i_Pron ** {pacc = "mě"} ; +} diff --git a/tests/czech/CzeTests.gf b/tests/czech/CzeTests.gf new file mode 100644 index 00000000..c903f1fe --- /dev/null +++ b/tests/czech/CzeTests.gf @@ -0,0 +1,25 @@ +resource CzeTests = open Prelude, SyntaxCze, SymbolicCze, ExtraCze, ParadigmsCze, (L = LexiconCze), (N = NumeralCze), (I = IdiomCze), (E = ExtendCze) in { +oper + vocalized_Prep : Prep = v_Prep locative ; + -- Typed consumers exercise every public verb-valency overload. + drink_V2 : V2 = mkV2 (krytV "pít") ; + drinkAcc_V2 : V2 = mkV2 (krytV "pít") accusative ; + drinkFrom_V2 : V2 = mkV2 (krytV "pít") (ParadigmsCze.mkPrep "z" genitive) ; + buy_V3 : V3 = mkV3 (kupovatV "kupovat") ; + -- Buying from a source for a recipient: both complements are prepositional. + buyFor_V3 : V3 = mkV3 (kupovatV "kupovat") + (ParadigmsCze.mkPrep "od" genitive) (ParadigmsCze.mkPrep "pro" accusative) ; + himSelf_RNP : E.RNP = E.ConjRNP and_Conj (E.Base_nr_RNP (mkNP he_Pron) E.ReflPron) ; + about_Prep : Prep = ParadigmsCze.mkPrep "o" locative ; + like_AP : AP = shortAP "rád" "ráda" "rádo" "rádi" "rády" "ráda" ; + ready_AP : AP = shortAP "připraven" "připravena" "připraveno" "připraveni" "připraveny" "připravena" ; + wash_V : V = seV (krytV "mýt") ; + learn_V : V = seV (mkV { + inf = "učit" ; + pressg1 = "učím" ; pressg2 = "učíš" ; pressg3 = "učí" ; + prespl1 = "učíme" ; prespl2 = "učíte" ; prespl3 = "učí" ; + pastpartsg = "učil" ; pastpartpl = "učili" ; + impsg2 = "uč" ; imppl1 = "učme" ; imppl2 = "učte" + }) ; + learn_VV : VV = mkVV learn_V ; +} diff --git a/tests/czech/README.md b/tests/czech/README.md new file mode 100644 index 00000000..964f349c --- /dev/null +++ b/tests/czech/README.md @@ -0,0 +1,57 @@ +# Czech RGL regression tests + +From the RGL checkout, run: + + sh tests/czech/check.sh + +The script uses `gf` from `PATH` when `GF` is unset or empty. To select another +executable, including a path containing spaces, use: + + GF='/path/to/gf' sh tests/czech/check.sh + +`GF` names one executable, not a command with flags. Generated files go into +a temporary directory removed on exit. No application grammar is required. + +`regressions.gfs` checks morphology and grammatical composition through the +source API, grouped by feature. Its expected output is in `regressions.out`. +`markup.gfs` separately checks clitic movement through fronting, embedding +and coordination, including discontinuous marked constituents. It also checks +that NP predetermination preserves markup scope when inserting before a modifier. +Source computation exposes `Predef.BIND` and `Predef.SOFT_BIND` markers; +PGF linearization handles them as token joining and punctuation. + +The runner also builds `AllCze`, `TryCze` and `SymbolicCze` into a fresh +directory, then compiles `CzeTests.gf` using only that distribution. +Typed declarations check every public V2/V3 paradigm overload. The runner +requires the consumer's `.gfo`, since GF can report errors and still exit +successfully. + +`CzeRoundTrip` and `roundtrip.tsv` check PGF generation and parsing, including +joined negation, polarity recovery and polite/plural ambiguity. +`CzeExtensionRoundTrip` and `extension-roundtrip.tsv` cover secondary and +dative predicates, bound objects, subject omission, complement forms and +predeterminer placement with a modified pronoun. Compound cardinals are checked +with possessives, oblique case and predicate agreement, including agreement +with the leading scale in sums such as 2,200. +The abstract fragments are acyclic: parsing must return the complete +expected set of trees, without truncation or reliance on enumeration order. +Repeated strings record distinct intended analyses. These small fragments +do not establish parsing coverage for the full Czech RGL. + +Preserve the empty `Pol.s` constituent: selecting a verb form with `pol.p` +alone does not recover polarity. For example, parsing `nečti ji` should give: + + UttImpSg PNeg (ImpVP (ComplSlash (SlashV2a read_V2) (UsePron she_Pron))) + +Removing the constituent leaves `?1` in place of `PNeg`, even though the +negative spelling is recognized. + +Prefer treebank coverage where practical. Keep source checks for paradigms +and distinctions outside those fragments; do not repeat treebank examples +unless the source API adds a separate contract. These tests cover present +clauses, infinitives and imperatives, not the unimplemented RGL tenses or +anteriority. + +Coordinated-subject person/number agreement and subordinate-clause punctuation +also retain inherited gaps. The finite fragments do not establish coverage +of those constructions. diff --git a/tests/czech/check.sh b/tests/czech/check.sh new file mode 100644 index 00000000..07800cc7 --- /dev/null +++ b/tests/czech/check.sh @@ -0,0 +1,54 @@ +#!/bin/sh +set -eu +: "${GF:=gf}" +command -v "$GF" >/dev/null 2>&1 || { + printf 'GF executable not found: %s\n' "$GF" >&2 + exit 1 +} +cd "$(dirname "$0")/../.." +work=$(mktemp -d "${TMPDIR:-/tmp}/czech-rgl-tests.XXXXXX") +trap 'rm -rf "$work"' EXIT HUP INT TERM +src=src/api:src/czech:src/common:src/abstract:src/prelude +mkdir -p "$work/source" "$work/api" "$work/consumer" "$work/pgf" +for suite in regressions markup; do + "$GF" -run -path="$src" -gfo-dir="$work/source" \ + < "tests/czech/$suite.gfs" > "$work/$suite.out" + diff -u "tests/czech/$suite.out" "$work/$suite.out" +done +# These are the normal Setup.hs language and API roots. AllCze includes ExtendCze. +"$GF" -c -path="$src" -gfo-dir="$work/api" \ + src/czech/AllCze.gf src/api/TryCze.gf src/api/SymbolicCze.gf \ + "$work/api.log" 2>&1 || { cat "$work/api.log"; exit 1; } +test -f "$work/api/ExtraCze.gfo" +test -f "$work/api/ExtendCze.gfo" +# Compile a consumer with only the resulting distribution on its search path. +GF_LIB_PATH="$work/api" "$GF" -c -path="$work/api" -gfo-dir="$work/consumer" \ + tests/czech/CzeTests.gf \ + "$work/consumer.log" 2>&1 || { cat "$work/consumer.log"; exit 1; } +test -s "$work/consumer/CzeTests.gfo" || { cat "$work/consumer.log"; exit 1; } +# Acyclic abstract grammars make the complete parse sets finite. Compare +# whole trees without relying on parser enumeration order; repeated strings in +# the treebank record genuine ambiguity, including polite/plural address. +for grammar in CzeRoundTrip CzeExtensionRoundTrip; do +case "$grammar" in + CzeRoundTrip) bank=tests/czech/roundtrip.tsv ;; + CzeExtensionRoundTrip) bank=tests/czech/extension-roundtrip.tsv ;; +esac +"$GF" -make -path="$src":tests/czech -gfo-dir="$work/api" -output-dir="$work/pgf" \ + "tests/czech/${grammar}Cze.gf" \ + "$work/pgf.log" 2>&1 || { cat "$work/pgf.log"; exit 1; } +while IFS="$(printf '\t')" read -r tree surface; do + printf 'i %s\nl -lang=%sCze %s\np -lang=%sCze -cat=Utt "%s"\nq\n' \ + "$work/pgf/$grammar.pgf" "$grammar" "$tree" "$grammar" "$surface" \ + | "$GF" -run > "$work/roundtrip.out" + sed '/^$/d' "$work/roundtrip.out" > "$work/roundtrip.lines" + sed -n '1p' "$work/roundtrip.lines" > "$work/linearization.out" + printf '%s\n' "$surface" > "$work/linearization.expected" + diff -u "$work/linearization.expected" "$work/linearization.out" + sed '1d' "$work/roundtrip.lines" | LC_ALL=C sort > "$work/parses.out" + awk -F '\t' -v surface="$surface" '$2 == surface {print $1}' "$bank" \ + | LC_ALL=C sort > "$work/parses.expected" + diff -u "$work/parses.expected" "$work/parses.out" +done < "$bank" +done +printf 'Czech source regressions, AllCze, installed API imports and PGF round trips passed.\n' diff --git a/tests/czech/extension-roundtrip.tsv b/tests/czech/extension-roundtrip.tsv new file mode 100644 index 00000000..1d8fee44 --- /dev/null +++ b/tests/czech/extension-roundtrip.tsv @@ -0,0 +1,44 @@ +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (UsePron i_Pron)) (ComplSlash (SlashV2AP have_V2 like_AP) (UsePron she_Pron)))) já ji mám rád +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ComplSlash (SlashV2AP have_V2 like_AP) (UsePron she_Pron)))) mám ji rád +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject she_Pron) (ReflRNP (SlashV2AP have_V2 like_AP) (UseRNP (ReflPoss NumSg (UseN son_N)))))) má ráda svého syna +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject she_Pron) (ReflRNP (SlashV2AP have_V2 like_AP) (PredetRNP only_Predet ReflPron)))) má ráda jen sebe +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ReflRNP (SlashV2a wait_V2) (PredetRNP all_Predet (ReflPoss NumPl (UseN child_N)))))) čekám na všechny své děti +UttS (UseCl (TTAnt TPres ASimul) PPos (DativeCopulaCl (UsePron i_Pron) (DetCN (DetQuant DefArt two_Num) (UseN year_N)))) jsou mi dva roky +UttS (UseCl (TTAnt TPres ASimul) PNeg (DativeCopulaCl (UsePron i_Pron) (DetCN (DetQuant DefArt five_Num) (UseN year_N)))) není mi pět let +UttQS (UseQCl (TTAnt TPres ASimul) PPos (DativeCopulaQCl (UsePron youPol_Pron) (IdetCN how8many_IDet (UseN year_N)))) kolik let vám je +UttS (UseCl (TTAnt TPres ASimul) PPos (DativeCopulaCl (DetCN (DetQuant DefArt NumSg) (UseN son_N)) (DetCN (DetQuant DefArt five_Num) (UseN year_N)))) synovi je pět let +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (UsePron i_Pron)) (UseV wash_V))) já se myji +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (UseV wash_V))) myji se +UttAdv (SubjS if_Subj (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (UseV wash_V)))) jestliže se myji +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ReflRNP (SlashV2a writeAbout_V2) (PredetRNP all_Predet (ReflPoss NumPl (UseN child_N)))))) píši o všech svých dětech +UttAdv (PrepNP possess_Prep (UsePron he_Pron)) jeho +UttAdv (PrepNP by8agent_Prep (UsePron he_Pron)) od něho +UttQS (UseQCl (TTAnt TPres ASimul) PPos (QuestVP whoSg_IP (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) kdo ji miluje +UttIAdv (PrepIP by8agent_Prep whoSg_IP) od koho +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (UseComp (CompAP (Compare young_A (ComparisonPron she_Pron)))))) jsem mladší než ona +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (UseComp (CompAP (Compare fast_A (ComparisonPron she_Pron)))))) jsem rychlejší než ona +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (DetCN (DetQuant DefArt five_Num) (UseN child_N))) (UseComp (CompAP (Compare young_A (ComparisonPron she_Pron)))))) pět dětí je mladších než ona +UttAdv (PrepNP by8agent_Prep (DetCN (DetQuant DefArt NumSg) (ModifiedN (Compare young_A (ComparisonPron she_Pron)) child_N))) od dítěte mladšího než ona +UttAdv (PrepNP with_Prep (PredetNP all_Predet (AdvPron we_Pron home_Adv))) s námi všemi doma +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (DetCN (DetQuant DefArt two_Num) (UseN child_N))) (ReflRNP (SlashV2a love_V2) (PredetRNP all_Predet ReflPron)))) dvě děti milují sebe všechny +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (DetCN (DetQuant DefArt five_Num) (UseN child_N))) (ReflRNP (SlashV2a love_V2) (PredetRNP all_Predet ReflPron)))) pět dětí miluje sebe všechny +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (DetCN (DetQuant DefArt five_Num) (UseN child_N))) (ReflRNP (SlashV2a wait_V2) (PredetRNP all_Predet ReflPron)))) pět dětí čeká na sebe všechny +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ReflRNP (SlashV2a wait_V2) (PredetRNP all_Predet (ReflPoss five_Num (UseN child_N)))))) čekám na všech svých pět dětí +UttNP (DetCN (DetQuant (PossPron i_Pron) twentyOne_Num) (UseN child_N)) mých dvacet jedna dětí +UttNP (DetCN (DetQuant (PossPron i_Pron) twentyTwo_Num) (UseN child_N)) mých dvacet dva dětí +UttAdv (PrepNP with_Prep (DetCN (DetQuant (PossPron i_Pron) twentyOne_Num) (UseN child_N))) s mými dvaceti jedna dětmi +UttAdv (PrepNP with_Prep (DetCN (DetQuant (PossPron i_Pron) twentyTwo_Num) (UseN child_N))) s mými dvaceti dvěma dětmi +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (FullSubject (DetCN (DetQuant (PossPron i_Pron) twentyTwo_Num) (UseN child_N))) (UseComp (CompAP (Compare young_A (ComparisonPron she_Pron)))))) mých dvacet dva dětí je mladších než ona +UttNP (DetCN (DetQuant this_Quant twentyTwoHundred_Num) (UseN child_N)) tyto dva tisíce dvě stě dětí +UttNP (DetCN (DetQuant (PossPron i_Pron) twentyTwoHundred_Num) (UseN child_N)) mé dva tisíce dvě stě dětí +UttAdv (PrepNP with_Prep (DetCN (DetQuant this_Quant twelveHundred_Num) (UseN child_N))) s tímto jedním tisícem dvěma sty dětí +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ComplSlash (SlashV2a love_V2) (DetCN (DetQuant (PossPron youPol_Pron) NumSg) (UseN woman_N))))) miluji vaši ženu +UttS (UseCl (TTAnt TPres ASimul) PPos (PredSubject (DroppedSubject i_Pron) (ComplSlash (SlashV2a love_V2) (DetCN (DetQuant (PossPron we_Pron) NumSg) (UseN woman_N))))) miluji naši ženu +UttS (UseCl (TTAnt TPres ASimul) PPos (DativeCopulaCl (DetCN (DetQuant (PossPron we_Pron) NumSg) (UseN woman_N)) (DetCN (DetQuant DefArt five_Num) (UseN year_N)))) naší ženě je pět let +UttS (UseCl (TTAnt TPres ASimul) PPos (DativeCopulaCl (DetCN (DetQuant (PossPron youPol_Pron) NumSg) (UseN woman_N)) (DetCN (DetQuant DefArt five_Num) (UseN year_N)))) vaší ženě je pět let +UttAdv (PrepNP possess_Prep (DetCN (DetQuant (PossPron youPol_Pron) NumSg) (UseN woman_N))) vaší ženy +UttAdv (PrepNP with_Prep (DetCN (DetQuant (PossPron we_Pron) NumSg) (UseN son_N))) s naším synem +UttAdv (PrepNP with_Prep (DetCN (DetQuant (PossPron youPol_Pron) NumSg) (UseN child_N))) s vaším dítětem +UttAdv (PrepNP with_Prep (DetCN (DetQuant (PossPron youPol_Pron) NumPl) (UseN child_N))) s vašimi dětmi +UttNP (DetCN (DetQuant DefArt two_Num) (UseN apple_N)) dvě jablka +UttNP (DetCN (DetQuant DefArt five_Num) (UseN apple_N)) pět jablek diff --git a/tests/czech/markup.gfs b/tests/czech/markup.gfs new file mode 100644 index 00000000..5ee110f3 --- /dev/null +++ b/tests/czech/markup.gfs @@ -0,0 +1,24 @@ +i -retain tests/czech/CzeMarkup.gf +cc -one (mkUtt markedWashing) +cc -one (mkUtt todayMarked) +cc -one (G.SubjS if_Subj markedWashing) +-- Repeated fronting splits the marked constituent; both pieces retain nested markup. +cc -one (mkUtt (G.AdvS (ParadigmsCze.mkAdv "doma") (G.AdvS L.today_Adv (M.MarkupS M.i_Mark markedWashing)))) +cc -one (G.SubjS if_Subj todayMarked) +-- Coordination keeps the second clause's clitics in their own domain. +cc -one (mkUtt (mkS and_Conj markedWashing washing)) +cc -one (G.SubjS if_Subj (mkS and_Conj markedWashing washing)) +cc -one (G.SubjS if_Subj (M.MarkupS M.b_Mark (mkS and_Conj washing washing))) +cc -one (mkUtt (G.ExtAdvS L.today_Adv markedWashing)) +cc -one (G.SubjS if_Subj (G.ExtAdvS L.today_Adv markedWashing)) +cc -one (G.SubjS if_Subj (G.SSubjS markedWashing if_Subj washing)) +-- Empty movable components must not gain tags, even across coordination. +cc -one (mkUtt (G.AdvS (ParadigmsCze.mkAdv "doma") (G.AdvS L.today_Adv (M.MarkupS M.b_Mark swimming)))) +cc -one (mkUtt (G.AdvS (ParadigmsCze.mkAdv "doma") (G.AdvS L.today_Adv (M.MarkupS M.b_Mark loving)))) +cc -one (mkUtt (G.AdvS (ParadigmsCze.mkAdv "doma") (G.AdvS L.today_Adv (M.MarkupS M.b_Mark (mkS and_Conj swimming washing))))) +-- Predetermination respects the scope of an already marked NP. +cc -one (mkUtt (mkNP all_Predet (M.MarkupNP M.b_Mark (mkNP (mkNP we_Pron) (ParadigmsCze.mkAdv "doma"))))) +cc -one (mkUtt (mkNP only_Predet (M.MarkupNP M.b_Mark (mkNP (mkNP we_Pron) (ParadigmsCze.mkAdv "doma"))))) +-- An empty following piece may retain a wrapper after insertion. +cc -one (mkUtt (mkNP all_Predet (M.MarkupNP M.b_Mark (mkNP we_Pron)))) +q diff --git a/tests/czech/markup.out b/tests/czech/markup.out new file mode 100644 index 00000000..31ab0744 --- /dev/null +++ b/tests/czech/markup.out @@ -0,0 +1,17 @@ + myji se +dnes se myji +jestliže se myji +doma se dnes myji +jestliže se dnes myji + myji se a myji se +jestliže se myji a myji se +jestliže se myji a myji se +dnes Predef.SOFT_BIND , myji se +jestliže se dnes Predef.SOFT_BIND , myji +jestliže se myji Predef.SOFT_BIND , jestliže se myji +doma dnes plavu +doma ji dnes miluji +doma dnes plavu a myji se + my všichni doma +jen my doma + my všichni diff --git a/tests/czech/regressions.gfs b/tests/czech/regressions.gfs new file mode 100644 index 00000000..ea385434 --- /dev/null +++ b/tests/czech/regressions.gfs @@ -0,0 +1,267 @@ +i -retain tests/czech/CzeTests.gf + +-- Noun inflection, possession and proper names. +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (zenaN "banka"))) +cc -one (SyntaxCze.mkAdv to_Prep (mkNP (zenaN "banka"))) +cc -one (SyntaxCze.mkAdv possess_Prep (mkNP i_Pron (zenaN "manželka"))) +cc -one (mkUtt (mkS (mkCl (mkNP this_Quant plNum L.child_N) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP this_Quant L.child_N) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkPN "NN" mascAnimate)) (mkA "unavený")))) +cc -one (SyntaxCze.mkAdv to_Prep (mkNP (mkPN (zenaN "Praha")))) +cc -one ((kostN "paměť").snom ++ (kostN "paměť").sacc ++ (kostN "paměť").sgen ++ (kostN "paměť").sins ++ (kostN "paměť").pdat ++ (kostN "paměť").ploc ++ (kostN "paměť").pins) +cc -one ((mkN "loď" "lodi" feminine).sgen ++ (mkN "loď" "lodi" feminine).sins ++ (mkN "loď" "lodi" feminine).pins) +cc -one ((pisenN "dlaň").snom ++ (pisenN "dlaň").sacc ++ (pisenN "dlaň").sgen ++ (pisenN "dlaň").sdat ++ (pisenN "dlaň").sins ++ (pisenN "dlaň").pins) +cc -one ((pisenN "píseň").sgen ++ (pisenN "píseň").sdat ++ (pisenN "píseň").pins) +cc -one ((zenaN "Nataša").sgen ++ (zenaN "Nataša").sdat ++ (zenaN "Nataša").sloc ++ (zenaN "Nataša").pnom ++ (zenaN "Nataša").pacc) +cc -one ((zenaN "Máňa").sgen ++ (zenaN "Máňa").sdat ++ (zenaN "Máňa").sloc ++ (zenaN "Máňa").pnom ++ (zenaN "Máňa").pacc) +cc -one ((zenaN "Naďa").sgen ++ (zenaN "Naďa").sdat ++ (zenaN "Káťa").sgen ++ (zenaN "Káťa").sdat) +cc -one ((zenaN "Zoja").sgen ++ (zenaN "Zoja").sdat) +cc -one (L.school_N.sdat ++ (zenaN "váza").sdat ++ (zenaN "mísa").sdat ++ (zenaN "banka").sdat) +cc -one (L.salt_N.snom ++ L.salt_N.sacc ++ L.salt_N.svoc) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP plNum L.milk_N)) +cc -one (mkVoc (mkNP L.year_N)) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP plNum L.apple_N)) + +-- Náš/váš inflect for the possessed noun, independently of addressee number. +cc -one (SyntaxCze.mkAdv possess_Prep (mkNP we_Pron L.woman_N)) +cc -one (SyntaxCze.mkAdv about_Prep (mkNP we_Pron L.woman_N)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP youPol_Pron L.woman_N)) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP youPol_Pron L.city_N)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP youPl_Pron L.man_N)) +cc -one (SyntaxCze.mkAdv (ParadigmsCze.mkPrep dative) (mkNP (mkDet we_Pron plNum) L.man_N)) +cc -one (SyntaxCze.mkAdv (ParadigmsCze.mkPrep dative) (mkNP (mkDet youPl_Pron plNum) L.man_N)) +cc -one (mkUtt (mkNP (mkDet we_Pron plNum) L.man_N)) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop she_Pron)) L.love_V2 (mkNP (mkDet we_Pron plNum) L.man_N)))) +cc -one (mkUtt (mkNP (mkDet youPol_Pron plNum) L.city_N)) + +-- Adjective degrees retain case and agreement. +cc -one (SyntaxCze.mkAdv to_Prep (mkNP (mkDet the_Quant (mkOrd L.good_A)) (hradN "hotel"))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkDet the_Quant (mkOrd L.blue_A)) (zenaN "barva"))) +cc -one ((mkA "mladý").compar.msnom) +cc -one ((mkA "hrdý").compar.msnom ++ (mkA "rudý").superl.msgen ++ (mkA "chudý").compar.msnom) +cc -one ((mkA "rychlý").compar.msnom ++ (mkA "rychlý").superl.msnom) +cc -one ((mkA "krásný").compar.msnom ++ (mkA "moderní").compar.msnom ++ (mkA "chytrý").compar.msnom) +cc -one ((mkA "praktický").compar.msnom ++ (mkA "lidský").compar.msnom ++ (mkA "cizí").compar.msnom) +cc -one ((mkA "čistý" "čistší").compar.msnom) +cc -one ((mkA "jarní" nonExist).msgen) +cc -one ((mkA "jarní" nonExist).compar.msnom) +cc -one ((mkA "jarní" nonExist).superl.msnom) +cc -one ((mkA "lehký").compar.msnom) +cc -one ((mkA "lehký" "lehčí").compar.msnom) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (mkCN (comparAP L.young_A) (panN "student")))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (comparAP L.young_A)))) + +-- Verb principal parts and public valency constructors. +cc -one (mkUtt (mkCl (mkNP i_Pron) (mkVP drink_V2 (mkNP (zenaN "voda"))))) +cc -one (mkUtt (mkCl (mkNP i_Pron) (mkVP buy_V3 (mkNP (zenaN "kniha")) (mkNP (panN "syn"))))) +cc -one ((kupovatV "kupovat").pressg1 ++ (kupovatV "kupovat").negprespl3 ++ (kupovatV "kupovat").pastpartsg ++ (kupovatV "kupovat").impsg2 ++ (kupovatV "kupovat").negimppl2) +cc -one ((krytV "krýt").pressg1 ++ (krytV "krýt").negprespl3 ++ (krytV "krýt").pastpartsg ++ (krytV "krýt").impsg2 ++ (krytV "krýt").negimppl2) +cc -one ((krytV "pít").pressg1 ++ (krytV "pít").pastpartsg ++ (krytV "pít").impsg2) + +-- Pronoun gender and polite singular versus plural predicate agreement. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop youPol_Pron)) (mkA "unavený")))) +cc -one (mkUtt (mkS negativePol (mkCl (mkNP (E.ProDrop (genderPron feminine youPol_Pron))) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop (genderPron feminine youPol_Pron))) (mkVP (mkCN (zenaN "Angličanka")))))) +cc -one (mkUtt (mkS (mkCl (mkNP E.theyFem_Pron) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP E.theyNeutr_Pron) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP E.youPolFem_Pron) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP E.youPolPlFem_Pron) (mkA "unavený")))) + +-- CN predicates take predicative case; NP predicates retain their own number. +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (mkVP (mkCN (panN "student")))))) +cc -one (mkUtt (mkS (mkCl (mkNP he_Pron) (mkNP (mkPN (panN "Petr")))))) +cc -one (mkUtt (mkS (mkCl (mkNP we_Pron) (mkNP (mkDet the_Quant plNum) (panN "student"))))) +cc -one (mkUtt (mkS (mkCl (mkNP we_Pron) (mkNP (mkCard "2") L.child_N)))) +cc -one (mkUtt (mkS (mkCl (mkNP we_Pron) (mkNP (zenaN "rodina"))))) + +-- Clitic domains in clauses, questions, infinitives and imperatives. +cc -one (mkUtt (mkCl (mkNP i_Pron) (mkVP buy_V3 (mkNP (zenaN "kniha")) (mkNP he_Pron)))) +cc -one (mkUtt (mkCl (mkNP i_Pron) (mkVP (mkVPSlash buy_V3 (mkNP she_Pron)) (mkNP only_Predet (mkNP he_Pron))))) +cc -one (mkUtt (mkQS (mkQCl where_IAdv (mkNP (hradN "hotel"))))) +cc -one (mkUtt (mkS negativePol (mkCl (mkNP (E.ProDrop i_Pron)) want_VV (mkVP L.swim_V)))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkVP L.wait_V2 (mkNP she_Pron))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.know_VQ (mkQS (mkQCl (mkCl (mkNP (E.ProDrop youSg_Pron)) (mkA "unavený"))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.know_VQ (mkQS (mkQCl where_IAdv (mkNP (hradN "hotel"))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.love_V2 (mkNP she_Pron)))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) want_VV (mkVP wash_V)))) +cc -one (mkUtt politeImpForm negativePol (mkImp (mkVP wash_V))) +cc -one (mkUtt (mkVP wash_V)) +-- Ordinary VV keeps the embedded reflexive in its infinitive domain. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) learn_VV (mkVP wash_V)))) +cc -one (mkUtt (mkS L.today_Adv (mkS (mkCl (mkNP (E.ProDrop i_Pron)) wash_V)))) +cc -one (mkUtt (mkS L.today_Adv (mkS (ParadigmsCze.mkAdv "doma") (mkS (mkCl (mkNP (E.ProDrop i_Pron)) wash_V))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.know_VS (mkS (mkCl (mkNP (E.ProDrop she_Pron)) wash_V))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.know_VQ (mkQS (mkQCl (mkCl (mkNP (E.ProDrop youSg_Pron)) wash_V)))))) +cc -one (I.ImpP3 (mkNP (E.ProDrop he_Pron)) (mkVP wash_V)) +cc -one (I.ImpP3 (mkNP (mkCard "2") L.child_N) (mkVP wash_V)) +cc -one (I.ImpP3 (mkNP (E.ProDrop she_Pron)) (mkVP L.love_V2 (mkNP he_Pron))) +cc -one (SyntaxCze.mkAdv if_Subj (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.love_V2 (mkNP she_Pron)))) +cc -one (SyntaxCze.mkAdv if_Subj (mkS L.today_Adv (mkS (mkCl (mkNP (E.ProDrop i_Pron)) wash_V)))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (siV (kupovatV "kupovat"))))) +cc -one (mkUtt negativePol (mkImp (siV (kupovatV "kupovat")))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) can_VV (mkVP (siV (kupovatV "kupovat")))))) +-- A reflexive matrix verb blocks climbing even through mkModalVV. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkModalVV learn_V) (mkVP wash_V)))) + +-- Cardinal agreement and case government, including oblique compound numerals. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) have_V2 (mkNP (mkCard "5") (zenaN "pizza"))))) +cc -one (mkUtt (mkNP (mkCard "21") (hradN "rok"))) +cc -one (mkUtt (mkNP (mkCard "22") L.child_N)) +cc -one (mkUtt (mkNP (mkCard "101") L.child_N)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (mkCard (N.num (N.pot3plus (N.pot1as2 (N.pot0as1 N.n2)) (N.pot1as2 (N.pot0as1 N.pot01))))) L.child_N)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP this_Quant (mkNum (N.num (N.pot3 (N.pot1as2 (N.pot1plus N.n2 N.pot01))))) (zenaN "koruna"))) +cc -one (SyntaxCze.mkAdv possess_Prep (mkNP (mkCard "3") L.child_N)) +cc -one (SyntaxCze.mkAdv (ParadigmsCze.mkPrep "s" instrumental) (mkNP (mkCard "23") L.child_N)) +cc -one (SyntaxCze.mkAdv (ParadigmsCze.mkPrep "s" instrumental) (mkNP (mkCard "200") L.child_N)) +cc -one (SyntaxCze.mkAdv (ParadigmsCze.mkPrep "s" instrumental) (mkNP (mkCard (N.num N.pot31)) (zenaN "koruna"))) +cc -one (mkUtt (mkNP (mkCard (N.IDig N.D_0)) (zenaN "koruna"))) +cc -one (SyntaxCze.mkAdv to_Prep (mkNP (mkDet the_Quant ) (hradN "hotel"))) +cc -one (N.pot5 (N.pot1as2 (N.pot0as1 (N.pot0 N.n2)))) +cc -one (N.pot4decimal (N.IFrac (N.PosDecimal (N.IDig N.D_3)) N.D_5)) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (mkA "unavený")))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard (N.num (N.pot3 (N.pot1as2 (N.pot0as1 (N.pot0 N.n2)))))) L.child_N) (mkA "unavený")))) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (mkDet this_Quant (mkNum (mkCard (N.num N.pot31)))) (zenaN "koruna"))) +cc -one (mkUtt (mkNP (mkDet this_Quant (mkNum (mkCard (N.num (N.pot3 (N.pot1as2 (N.pot0as1 (N.pot0 N.n2)))))))) L.child_N)) +cc -one (SyntaxCze.mkAdv possess_Prep (mkNP (mkDet this_Quant (mkNum (mkCard (N.num (N.pot3 (N.pot1as2 (N.pot0as1 (N.pot0 N.n2)))))))) L.child_N)) +cc -one (mkUtt (mkNP (mkDet this_Quant (mkNum (mkCard "200"))) (zenaN "koruna"))) +cc -one (mkUtt (mkNP (mkDet this_Quant (mkNum (mkCard "500"))) (zenaN "koruna"))) +cc -one (mkUtt (mkNP (mkDet this_Quant (mkNum (mkCard (N.num (N.pot3 (N.pot2 N.n2)))))) L.child_N)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (mkDet this_Quant (mkNum (mkCard (N.num N.pot31))) (mkOrd (mkA "dobrý" "lepší"))) (zenaN "koruna"))) +cc -one (mkUtt (mkIP which_IQuant (mkNum (mkCard "5")) (mkCN L.child_N))) +cc -one (mkUtt (mkIP which_IQuant (mkNum (mkCard "2")) (mkCN L.child_N))) +cc -one (mkUtt (mkIP which_IQuant (mkNum (mkCard (N.num N.pot31))) (mkCN L.child_N))) +cc -one (mkUtt (mkCard "1")) +cc -one (mkUtt (mkCard "21")) +cc -one (mkUtt (mkCard "2")) +cc -one (mkUtt (mkNP (mkCard "1") L.child_N)) + +-- Scale heads control agreement independently of their genitive complements. +-- The shared num stops below a million; larger subcategories are used directly. +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) + +-- Sums ending in a scale retain their leading head for modifiers and predicates. +cc -one (mkUtt (mkNP all_Predet (mkNP (mkCard (N.num (N.pot3plus (N.pot1as2 (N.pot0as1 (N.pot0 N.n2))) (N.pot2 (N.pot0 N.n2))))) L.child_N))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard ) (zenaN "koruna")) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP this_Quant (mkNum (mkCard )) (zenaN "koruna")) ready_AP))) + +-- Signed integers retain agreement; fractions govern genitive. +cc -one (mkUtt (mkCard )) +cc -one (mkUtt (mkCard )) +cc -one (mkUtt (mkCard )) +cc -one (mkUtt (mkNP (mkCard (N.num (N.pot3decimal (N.NegDecimal (N.IDig N.D_2))))) (zenaN "koruna"))) +cc -one (mkUtt (mkCard )) + +-- Preposition vocalization follows the next word, including modifiers and capitals. +cc -one (SyntaxCze.mkAdv in_Prep (mkNP L.school_N)) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkDet the_Quant (mkOrd L.good_A)) L.school_N)) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkCN (mkA "moderní") L.school_N))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkN "muzeum" "muzea" neuter))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkPN (zenaN "Stromovka")))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (zenaN "Měna"))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (L.city_N ** {sloc = "Městě"}))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkCN (mkA "mělký") L.city_N))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP ((hradN "mlýn") ** {sloc = "mlýně"}))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (mkCN (mkA "nový") ((hradN "mlýn") ** {sloc = "mlýně"})))) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP i_Pron)) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (zenaN "žena"))) +cc -one (SyntaxCze.mkAdv from_Prep (mkNP L.school_N)) +cc -one (SyntaxCze.mkAdv from_Prep (mkNP all_Predet (mkNP plNum L.child_N))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (zenaN "mzda"))) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP i_Pron)) +cc -one (SyntaxCze.mkAdv (v_Prep accusative) (mkNP i_Pron)) +cc -one (SyntaxCze.mkAdv in_Prep (mkNP (zenaN "hra"))) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (zenaN "hra"))) +cc -one (SyntaxCze.mkAdv from_Prep (mkNP (mkPN (zenaN "Škola")))) + +-- Predeterminers retain full pronouns and inflect with the NP head. +cc -one (mkUtt (mkS (mkCl (mkNP only_Predet (mkNP i_Pron)) L.swim_V))) +cc -one (mkUtt (mkS (mkCl (mkNP only_Predet (mkNP i_Pron)) wash_V))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) L.love_V2 (mkNP only_Predet (mkNP he_Pron))))) +cc -one (mkUtt (mkS (mkCl (mkNP all_Predet (mkNP (genderPron feminine they_Pron))) wash_V))) +-- NP focus restores the full form even when given a ProDrop pronoun. +cc -one (mkUtt (mkS (mkCl (mkNP only_Predet (mkNP (E.ProDrop i_Pron))) wash_V))) +cc -one (mkUtt (mkNP all_Predet (mkNP plNum (panN "student")))) +cc -one (mkUtt (mkNP all_Predet (mkNP (mkCard "5") L.child_N))) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP all_Predet (mkNP plNum L.child_N))) +cc -one (mkUtt (mkNP all_Predet (mkNP (mkCN (zenaN "voda"))))) +cc -one (mkUtt (mkNP all_Predet (mkNP (mkCard (N.num N.pot31)) (zenaN "koruna")))) +-- Shared preposed modifiers agree with the first conjunct, also after ConsNP. +cc -one (mkUtt (mkNP all_Predet (mkNP and_Conj (mkNP plNum L.man_N) (mkNP plNum L.woman_N)))) +cc -one (mkUtt (mkNP all_Predet (mkNP and_Conj (mkListNP (mkNP plNum L.woman_N) (mkListNP (mkNP plNum L.man_N) (mkNP plNum L.child_N)))))) + +-- Complement forms distinguish bare case government from prepositions. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkAP (mkA2 (mkA "hrdý") (ParadigmsCze.mkPrep "na" accusative)) (mkNP he_Pron))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkAP (mkA2 (mkA "hrdý") (ParadigmsCze.mkPrep "na" accusative)) (mkNP she_Pron))))) +cc -one (mkUtt (mkNP (mkCN (mkAP (mkA2 (mkA "hrdý") (ParadigmsCze.mkPrep "na" accusative)) (mkNP she_Pron)) L.child_N))) +cc -one (mkUtt (mkNP (mkCN (mkAP (mkA2 (mkA "věrný") (lin Prep {s = [] ; c = dative ; hasPrep = False})) (mkNP he_Pron)) L.child_N))) +cc -one (SyntaxCze.mkAdv vocalized_Prep (mkNP he_Pron)) +-- mkVPSlash V3 NP completes Slash2V3. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkVP (mkVPSlash buyFor_V3 (mkNP he_Pron)) (mkNP she_Pron))))) +-- mkVP V3 NP NP instead completes Slash3V3; both must preserve slot order. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkVP buyFor_V3 (mkNP he_Pron) (mkNP she_Pron))))) + +-- Subject-bound RNPs compose with possession, coordination and modifiers. +cc -one (mkUtt (mkCl (mkNP i_Pron) (E.ReflRNP (mkVPSlash buy_V3 (mkNP she_Pron)) E.ReflPron))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop he_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) E.ReflPron)))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop she_Pron)) (E.ReflRNP (mkVPSlash L.wait_V2) E.ReflPron)))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop he_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.ReflPoss sgNum (mkCN (zenaN "manželka"))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop he_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.AdvRNP (mkNP (zenaN "manželka")) possess_Prep (E.ReflPoss sgNum (mkCN (panN "syn")))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop she_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.ConjRNP and_Conj (E.Base_rr_RNP E.ReflPron (E.ReflPoss plNum (mkCN L.child_N)))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.AdvRVP (mkVP ) (ParadigmsCze.mkPrep "o" locative) E.ReflPron)))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (E.AdvRAP (mkAP (mkA "hrdý")) (ParadigmsCze.mkPrep "na" accusative) E.ReflPron)))) +-- Reflexive modifiers take the complement's case, also with quantified subjects. +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (E.AdvRAP (mkAP (mkA "spokojený")) with_Prep (E.PredetRNP all_Predet E.ReflPron))))) +cc -one (mkUtt (mkNP (mkCN (E.AdvRAP (mkAP (mkA "hrdý")) (ParadigmsCze.mkPrep "na" accusative) E.ReflPron) L.child_N))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.ReflRNP (mkVPSlash L.wait_V2) (E.ConjRNP and_Conj (E.Base_nr_RNP (mkNP she_Pron) E.ReflPron)))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop she_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.PredetRNP only_Predet E.ReflPron))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.AdvRVP (mkVP L.write_V2 (mkNP (zenaN "kniha"))) about_Prep (E.PredetRNP all_Predet (E.ReflPoss plNum (mkCN L.child_N))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.ReflRNP (mkVPSlash have_V2) (E.PredetRNP all_Predet (E.ReflPoss (mkNum (mkCard (N.num N.pot31))) (mkCN (zenaN "koruna")))))))) +-- RNP coordination keeps the same first-conjunct rule through list extension. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.PredetRNP all_Predet (E.ConjRNP and_Conj (E.Base_rr_RNP (E.ReflPoss plNum (mkCN (panN "syn"))) (E.ReflPoss sgNum (mkCN (zenaN "dcera")))))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.ReflRNP (mkVPSlash L.love_V2) (E.PredetRNP all_Predet (E.ConjRNP and_Conj (E.Cons_rr_RNP (E.ReflPoss plNum (mkCN (panN "syn"))) (E.Base_rr_RNP (E.ReflPoss sgNum (mkCN (zenaN "dcera"))) E.ReflPron)))))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.ReflRNP (mkVPSlash have_V2) (E.AdvRNP (mkNP L.child_N) possess_Prep himSelf_RNP))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.AdvRVP (mkVP L.write_V2 (mkNP (zenaN "kniha"))) (ParadigmsCze.mkPrep dative) himSelf_RNP)))) +cc -one (mkUtt (mkNP (mkCN (E.AdvRAP (mkAP (mkA "plný")) possess_Prep himSelf_RNP) (zenaN "kniha")))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (E.AdvRAP (mkAP (mkA "plný")) possess_Prep himSelf_RNP)))) + +-- Secondary and short predicates retain agreement through modification and coordination. +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (mkVP (SlashV2AP have_V2 like_AP) (mkNP (zenaN "pizza")))))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) ready_AP))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (mkAP very_AdA ready_AP)))) +cc -one (mkUtt (mkS (mkCl (mkNP (mkCard "5") L.child_N) (mkAP and_Conj ready_AP (mkAP (mkA "unavený")))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkVP (SlashV2AP have_V2 like_AP) (mkNP he_Pron))))) + +-- Kdo has masculine singular agreement and inflects for case. +cc -one (mkUtt (mkQS (mkQCl whoSg_IP (mkAP L.young_A)))) +cc -one (whoSg_IP.s ! genitive ++ whoSg_IP.s ! dative ++ whoSg_IP.s ! instrumental) + +-- Comparison terms remain nominative while the adjective inflects. +cc -one (mkUtt (mkNP (mkCN (mkAP L.young_A (mkNP he_Pron)) (panN "student")))) +cc -one (SyntaxCze.mkAdv with_Prep (mkNP (mkCN (mkAP L.young_A (mkNP he_Pron)) (panN "student")))) +-- A ProDrop comparison pronoun must still be overt. +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop i_Pron)) (mkAP L.young_A (mkNP (E.ProDrop she_Pron)))))) +-- Addressing a quantified group retains genitive government. +cc -one (mkVoc (mkNP (mkCard "5") L.child_N)) +cc -one (mkVoc (mkNP all_Predet (mkNP (mkCard "5") L.child_N))) +-- Pronominal heads retain their order after modification and binding. +cc -one (SyntaxCze.mkAdv with_Prep (mkNP all_Predet (mkNP youPl_Pron))) +-- The outer pronoun controls all, independently of the reflexive's antecedent. +cc -one (mkUtt (mkS (mkCl (mkNP she_Pron) (E.AdvRVP (mkVP L.swim_V) with_Prep (E.PredetRNP all_Predet (E.AdvRNP (mkNP we_Pron) (ParadigmsCze.mkPrep "u" genitive) E.ReflPron)))))) +cc -one (mkUtt (mkS (mkCl (mkNP (E.ProDrop they_Pron)) (E.AdvRVP (mkVP L.write_V2 (mkNP (zenaN "kniha"))) about_Prep (E.PredetRNP all_Predet E.ReflPron))))) +-- Short forms allow predicate responses but cannot modify a noun. +cc -one (mkUtt (mkNP (mkCN like_AP (panN "student")))) +cc -one (mkUtt ready_AP) +-- Third-person gender changes the whole paradigm and preserves omission. +cc -one ((genderPron feminine he_Pron).nom ++ (genderPron feminine he_Pron).cacc ++ (genderPron feminine he_Pron).pacc ++ (genderPron feminine he_Pron).poss.msnom) +cc -one (mkUtt (mkS (mkCl (mkNP (genderPron feminine (E.ProDrop he_Pron))) ready_AP))) +cc -one ((genderPron mascAnimate (he_Pron ** {pacc = "něj"})).pacc) +-- Standalone case utterances preserve NP inflection and strong pronoun forms. +cc -one (E.UttAccNP (mkNP this_Quant (mkCN (mkA "studený") (zenaN "káva")))) +cc -one (E.UttDatNP (mkNP this_Quant (mkCN (mkA "studený") (zenaN "káva")))) +cc -one (E.UttAccNP (mkNP he_Pron)) +cc -one (E.UttDatNP (mkNP (E.ProDrop he_Pron))) +q diff --git a/tests/czech/regressions.out b/tests/czech/regressions.out new file mode 100644 index 00000000..c75a2bc1 --- /dev/null +++ b/tests/czech/regressions.out @@ -0,0 +1,213 @@ +v bance +do banky +mé manželky +tyto děti jsou unavené +toto dítě je unavené +NN je unavený +do Prahy +paměť paměť paměti pamětí pamětem pamětech paměťmi +lodi lodí loďmi +dlaň dlaň dlaně dlani dlaní dlaněmi +písně písni písněmi +Nataši Nataše Nataše Nataši Nataši +Máni Máně Máně Máni Máni +Nadi Nadě Káti Kátě +Zoji Zoje +škole váze míse bance +sůl sůl soli +v mlékách +roku +v jablkách +naší ženy +o naší ženě +s vaší ženou +ve vašem městě +s vaším mužem +našim mužům +vašim mužům +naši muži +miluje naše muže +vaše města +do nejlepšího hotelu +v nejmodřejší barvě +mladší +hrdější nejrudějšího chudší +rychlejší nejrychlejší +krásnější modernější chytřejší +praktičtější lidštější cizejší +čistší +jarního +Predef.nonExist +Predef.nonExist +Predef.nonExist +lehčí +s mladším studentem +pět dětí je mladších +já piji vodu +já kupuji knihu synovi +kupuji nekupují kupoval kupuj nekupujte +kryji nekryjí kryl kryj nekryjte +piji pil pij +jste unavený +nejste unavená +jste Angličanka +ony jsou unavené +ona jsou unavená +vy jste unavená +vy jste unavené +pět dětí je studenty +on je Petr +my jsme studenti +my jsme dvě děti +my jsme rodina +já mu kupuji knihu +já ji kupuji jen jemu +kde je hotel +nechci plavat +čekám na ni +vím Predef.SOFT_BIND , jestli jsi unavený +vím Predef.SOFT_BIND , kde je hotel +miluji ji +chci se mýt +nemyjte se +mýt se +učím se mýt se +dnes se myji +dnes se doma myji +vím Predef.SOFT_BIND , že se myje +vím Predef.SOFT_BIND , jestli se myješ +nechť se myje +nechť se dvě děti myjí +nechť ho miluje +jestliže ji miluji +jestliže se dnes myji +kupuji si +nekupuj si +mohu si kupovat +učím se mýt se +mám pět pizz +dvacet jedna roků +dvacet dva dětí +sto jedna dětí +se dvěma tisíci jedna dětmi +s těmito dvaceti jedna tisíci korun +tří dětí +s dvaceti třemi dětmi +s dvěma sty dětí +s tisícem korun +0 korun +do n Predef.BIND -tého hotelu +dvě miliardy +3 Predef.BIND , Predef.BIND 5 milionu +pět dětí je unavených +dva tisíce dětí je unavených +s tímto tisícem korun +tyto dva tisíce dětí +těchto dvou tisíc dětí +tato dvě stě korun +těchto pět set korun +tato dvě stě tisíc dětí +s tímto tisícem nejlepších korun +kterých pět dětí +které dvě děti +který tisíc dětí +jedna +dvacet jedna +dvě +jedno dítě +dva miliony korun jsou připraveny +miliarda korun je připravena +pět milionů korun je připraveno +dvě stě korun je připraveno +dvě stě milionů korun je připraveno +všechny dva tisíce dvě stě dětí +dva miliony dvě stě korun jsou připraveny +tato jedna miliarda dva miliony dvě stě korun je připravena +1 milion +- Predef.BIND 1 milion +- Predef.BIND 2 miliony +- Predef.BIND 2 tisíce korun +- Predef.BIND 1 Predef.BIND , Predef.BIND 5 milionu +ve škole +v nejlepší škole +v moderní škole +v muzeu +ve Stromovce +v Měně +ve Městě +v mělkém městě +ve mlýně +v novém mlýně +se mnou +se ženou +ze školy +ze všech dětí +ve mzdě +ve mně +ve mne +ve hře +s hrou +ze Školy +jen já plavu +jen já se myji +miluji jen jeho +ony všechny se myjí +jen já se myji +všichni studenti +všech pět dětí +se všemi dětmi +všechna voda +všechen tisíc korun +všichni muži a ženy +všechny ženy Predef.SOFT_BIND , muži a děti +jsem hrdý na něho +jsem hrdý na ni +dítě hrdé na ni +dítě věrné jemu +v něm +kupuji od něho pro ni +kupuji od něho pro ni +já ji kupuji sobě +miluje sebe +čeká na sebe +miluje svou manželku +miluje manželku svého syna +miluje sebe a své děti +učím se o sobě +pět dětí je hrdých na sebe +pět dětí je spokojených se sebou všemi +dítě hrdé na sebe +čekám na ni a sebe +miluje jen sebe +píši knihu o všech svých dětech +mám všechen svůj tisíc korun +miluji všechny své syny a svou dceru +miluji všechny své syny Predef.SOFT_BIND , svou dceru a sebe +mám dítě jeho a sebe +píši knihu jemu a sobě +kniha plná jeho a sebe +jsem plný jeho a sebe +pět dětí má rádo pizzu +pět dětí je připraveno +pět dětí je velmi připraveno +pět dětí je připraveno a unavených +mám ho rád +kdo je mladý +koho komu kým +student mladší než on +se studentem mladším než on +jsem mladší než ona +pět dětí +všech pět dětí +s vámi všemi +ona plave s námi všemi u sebe +píší knihu o sobě všech +Predef.nonExist +připraven +ona ji ni její +je připravena +něj +tuto studenou kávu +této studené kávě +jeho +jemu diff --git a/tests/czech/roundtrip.tsv b/tests/czech/roundtrip.tsv new file mode 100644 index 00000000..fe2499f9 --- /dev/null +++ b/tests/czech/roundtrip.tsv @@ -0,0 +1,18 @@ +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron i_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) já ji miluji +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron youSg_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) ty ji miluješ +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron he_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) on ji miluje +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron she_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) ona ji miluje +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron youPl_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) vy ji milujete +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron youPol_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) vy ji milujete +UttS (UseCl (TTAnt TPres ASimul) PNeg (PredVP (UsePron i_Pron) (ComplSlash (SlashV2a love_V2) (UsePron she_Pron)))) já ji nemiluji +UttS (UseCl (TTAnt TPres ASimul) PNeg (PredVP (UsePron youPol_Pron) (UseComp (CompAP (PositA young_A))))) vy nejste mladý +UttS (UseCl (TTAnt TPres ASimul) PNeg (PredVP (UsePron youPl_Pron) (UseComp (CompAP (PositA young_A))))) vy nejste mladí +UttImpPl PNeg (ImpVP (ComplSlash (SlashV2a read_V2) (UsePron she_Pron))) nečtěte ji +UttImpPol PNeg (ImpVP (ComplSlash (SlashV2a read_V2) (UsePron she_Pron))) nečtěte ji +UttImpSg PNeg (ImpVP (ComplSlash (SlashV2a read_V2) (UsePron she_Pron))) nečti ji +UttImpSg PPos (ImpVP (ComplSlash (SlashV2a read_V2) (UsePron she_Pron))) čti ji +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron i_Pron) (ComplSlash (Slash3V3 buy_V3 (UsePron he_Pron)) (UsePron she_Pron)))) já mu ji kupuji +UttS (UseCl (TTAnt TPres ASimul) PPos (PredVP (UsePron i_Pron) (ComplSlash (Slash2V3 buy_V3 (UsePron she_Pron)) (UsePron he_Pron)))) já mu ji kupuji +UttAdv (PrepNP vAcc_Prep (UsePron i_Pron)) ve mě +UttAdv (PrepNP in_Prep (DetCN (DetQuant DefArt NumSg) (UseN currency_N))) v měně +UttAdv (PrepNP in_Prep (DetCN (DetQuant DefArt NumSg) (UseN city_N))) ve městě diff --git a/tests/readme.rst b/tests/readme.rst index 02ce1536..8973b2bd 100644 --- a/tests/readme.rst +++ b/tests/readme.rst @@ -17,3 +17,11 @@ For a test to pass For instance, if there is a test named ``my-test``, the gf commands should be in ``my-test.gfs`` and the expected output in ``my-test.out`` + +Czech tests +----------- + +From the checkout root, run ``sh tests/czech/check.sh``. See the +`Czech test README `_ for executable selection and coverage. +The runner uses the same ``.gfs``/``.out`` convention and additionally checks +API imports and generation/parsing through finite PGF fragments.