ankira / prompts.py
nofater's picture
mvp
2a0825b
Raw
History Blame
2.27 kB
"""Dictation generation prompt + light text cleanup for the Gradio Space.
The LLM turns a learner's word list + CEFR level into ONE short German
dictation paragraph. This is distinct from the Anki flashcard prompts in
sys.md / sys1.md, which emit JSON — here we want plain prose to be read aloud.
"""
import re
DICTATION_SYSTEM_PROMPT = (
"You are a German teacher writing a short dictation (Diktat) for a language "
"learner.\n"
"\n"
"Write ONE coherent German paragraph of 2-4 natural, everyday sentences that "
"uses EVERY target word the user provides at least once (inflect it naturally; "
"the exact base form need not appear). The text must suit the given CEFR level:\n"
"- A1/A2: short main clauses, common vocabulary, present tense, simple word order.\n"
"- B1/B2: allow subordinate clauses, connectors and varied tenses, still everyday.\n"
"\n"
"Use correct standard German spelling, capitalization, umlauts (ä/ö/ü) and ß.\n"
"Output ONLY the German dictation text. No title, no label, no quotation marks, "
"no translation, no commentary, no markdown."
)
def build_user_prompt(words: list[str], level: str) -> str:
"""The user turn: the target words and the requested level."""
joined = ", ".join(words)
return f"Level: {level}\nTarget words: {joined}"
def parse_word_list(raw: str) -> list[str]:
"""Split learner input on commas and/or newlines into trimmed words,
dropping blanks while preserving order (spec §4)."""
parts = re.split(r"[,\n]+", raw or "")
return [w.strip() for w in parts if w.strip()]
_LABEL_RE = re.compile(r"^\s*(diktat|dictation|text)\s*:\s*", re.IGNORECASE)
def clean_dictation(text: str) -> str:
"""Strip a leading label line (e.g. 'Diktat:') and wrapping quotes the LLM
may add despite instructions (spec §4)."""
text = (text or "").strip()
text = _LABEL_RE.sub("", text, count=1).strip()
# Strip one layer of matching surrounding quotes (straight or typographic).
pairs = [('"', '"'), ("'", "'"), ("“", "”"), ("„", "“")]
for open_q, close_q in pairs:
if len(text) >= 2 and text[0] == open_q and text[-1] == close_q:
text = text[1:-1].strip()
break
return text