- EN
- UA
Building Blocks
| Block | What it does |
|---|---|
| Chat API | OpenAI-compatible LLM (Lapa 32K, Mamay 64K, Рада) |
| Rukopys OCR | Ukrainian handwriting → structured JSON regions |
| Web search | Tavily→Exa→DDG, Russian content filtered |
| MCP connectors | Prozorro tenders + Держстат 119 datasets live in chat |
1. OCR + LLM Post-processing
Digitise handwritten documents and extract structured data (dates, names, scores, sums).import httpx, base64, json
def ocr_then_extract(image_path: str, prompt: str, api_key: str) -> dict:
"""Two-step: OCR via /chat/stream, then LLM extraction."""
with open(image_path, "rb") as f:
b64 = base64.b64encode(f.read()).decode()
headers = {"Authorization": f"Bearer {api_key}"}
# Step 1 — OCR: send image to rukopys-ocr
ocr_resp = httpx.post(
"https://app.lapathoniia.top/chat/stream",
headers=headers,
json={
"message": "Розпізнай текст",
"model": "rukopys-ocr",
"session_id": "ocr-pipeline",
"user_id": "agent",
"web_search": False,
"connectors": False,
"chat_history": [],
"images": [{"base64": b64, "mime_type": "image/jpeg"}],
},
)
ocr_text = "".join(
json.loads(l[6:])["text"]
for l in ocr_resp.iter_lines()
if l.startswith("data: ") and "text" in json.loads(l[6:])
)
# Step 2 — Extract: send OCR text to MamayLM with structured prompt
llm_resp = httpx.post(
"https://app.lapathoniia.top/chat/stream",
headers=headers,
json={
"message": f"{prompt}\n\nText:\n{ocr_text}",
"model": "MamayLM-Gemma-3-27B-IT-v2.0",
"session_id": "ocr-pipeline",
"user_id": "agent",
"web_search": False,
"connectors": False,
"chat_history": [],
},
)
result_text = "".join(
json.loads(l[6:])["text"]
for l in llm_resp.iter_lines()
if l.startswith("data: ") and "text" in json.loads(l[6:])
)
return json.loads(result_text)
# Example: grade extraction from handwritten tests
result = ocr_then_extract(
"test.jpg",
"Extract: student name, date, subject, score (0-100). Return JSON only.",
api_key="sk-...",
)
# → {"student": "Ivan Petrenko", "date": "2024-05-12", "subject": "Math", "score": 87}
2. RAG (Retrieval-Augmented Generation)
Q&A over document corpora with Lapathoniia as LLM backend.from openai import OpenAI
import chromadb
# base_url must point to api.lapathoniia.top/v1
client = OpenAI(api_key="sk-...", base_url="https://api.lapathoniia.top/v1")
collection = chromadb.Client().get_or_create_collection("docs")
def rag_query(question: str) -> str:
results = collection.query(query_texts=[question], n_results=5)
context = "\n\n".join(results["documents"][0])
response = client.chat.completions.create(
model="MamayLM-Gemma-3-27B-IT-v2.0",
messages=[
{"role": "system", "content": f"Answer based on:\n\n{context}"},
{"role": "user", "content": question},
],
)
return response.choices[0].message.content
3. MCP Connectors
Query live Ukrainian government data in natural language.import httpx, json
def query_prozorro(question: str, api_key: str) -> str:
resp = httpx.post(
"https://app.lapathoniia.top/chat/stream",
headers={"Authorization": f"Bearer {api_key}"},
json={
"message": question,
"model": "hybrid",
"session_id": "procurement",
"user_id": "analyst",
"web_search": False,
"connectors": {"prozorro": True, "stats": False},
"chat_history": [],
},
)
return "".join(
json.loads(l[6:])["text"]
for l in resp.iter_lines()
if l.startswith("data: ") and "text" in json.loads(l[6:])
)
result = query_prozorro("Find Ukrzaliznytsia tenders sorted by amount")
# → markdown table: name, amount, date, winner, link
4. Web Search in Pipelines
Enrich answers with current web data. All Russian sources automatically filtered.def search_and_answer(query: str, api_key: str) -> tuple[str, list]:
resp = httpx.post(
"https://app.lapathoniia.top/chat/stream",
headers={"Authorization": f"Bearer {api_key}"},
json={
"message": query,
"model": "MamayLM-Gemma-3-27B-IT-v2.0",
"session_id": "research",
"user_id": "agent",
"web_search": True,
"connectors": False,
"chat_history": [],
},
)
text, sources = [], []
for line in resp.iter_lines():
if line.startswith("data: "):
d = json.loads(line[6:])
if "text" in d: text.append(d["text"])
if "sources" in d: sources = d["sources"]
return "".join(text), sources
5. Batch Processing Pipeline
Process entire folders of documents, combine OCR + LLM + analytics.import httpx, json, time
def batch_ocr_pipeline(drive_folder_id: str, api_key: str) -> list[dict]:
"""Submit all files in a Drive folder through OCR → LLM pipeline."""
resp = httpx.post(
"https://app.lapathoniia.top/batch/jobs",
headers={"Authorization": f"Bearer {api_key}"},
json={
"job_type": "auto", # auto-detects ocr vs chat per file
"drive": {"folder_id": drive_folder_id},
"pipeline": "ocr_then_chat",
"pipeline_instruction": "Extract: date, signatories, amounts. Return JSON.",
"pipeline_model": "MamayLM-Gemma-3-27B-IT-v2.0",
},
)
# Returns array of job objects — one per file
return resp.json()
def poll_job(job_id: str, api_key: str) -> dict:
"""Poll until a single job is done."""
while True:
status = httpx.get(
f"https://app.lapathoniia.top/batch/jobs/{job_id}",
headers={"Authorization": f"Bearer {api_key}"},
).json()
if status["status"] == "done":
return status
time.sleep(5)
job_type values: ocr (force OCR), chat (LLM only), auto (detect per file from extension/MIME).
Input: either inline (messages array) or drive (file_id or folder_id).Building Blocks
| Блок | Що робить |
|---|---|
| Chat API | OpenAI-сумісний LLM (Lapa 32K, Mamay 64K, Рада) |
| Rukopys OCR | Рукопис → структуровані JSON регіони |
| Веб-пошук | Tavily→Exa→DDG, фільтрація російського контенту |
| MCP конектори | Prozorro тендери + Держстат 119 датасетів живі в чаті |
1. OCR + LLM постобробка
Оцифруйте рукописні документи та витягніть структуровані дані (дати, імена, оцінки, суми).import httpx, base64, json
def ocr_та_витяг(шлях: str, інструкція: str, api_key: str) -> dict:
with open(шлях, "rb") as f:
b64 = base64.b64encode(f.read()).decode()
headers = {"Authorization": f"Bearer {api_key}"}
# Крок 1 — OCR
ocr_resp = httpx.post(
"https://app.lapathoniia.top/chat/stream", headers=headers,
json={"message": "Розпізнай текст", "model": "rukopys-ocr",
"session_id": "ocr-pipeline", "user_id": "agent",
"web_search": False, "connectors": False, "chat_history": [],
"images": [{"base64": b64, "mime_type": "image/jpeg"}]},
)
ocr_text = "".join(
json.loads(l[6:])["text"] for l in ocr_resp.iter_lines()
if l.startswith("data: ") and "text" in json.loads(l[6:])
)
# Крок 2 — Витяг
llm_resp = httpx.post(
"https://app.lapathoniia.top/chat/stream", headers=headers,
json={"message": f"{інструкція}
Текст:
{ocr_text}",
"model": "MamayLM-Gemma-3-27B-IT-v2.0",
"session_id": "ocr-pipeline", "user_id": "agent",
"web_search": False, "connectors": False, "chat_history": []},
)
return json.loads("".join(
json.loads(l[6:])["text"] for l in llm_resp.iter_lines()
if l.startswith("data: ") and "text" in json.loads(l[6:])
))
result = ocr_та_витяг(
"test.jpg",
"Витягни: ім’я учня, дату, предмет, оцінку (0-100). Тільки JSON.",
api_key="sk-...",
)
# → {"student": "Іван Петренко", "date": "2024-05-12", "subject": "Математика", "score": 87}
Застосування: оцінювання тестів · аналіз договорів · медичні записи · архівні дослідження
2. RAG (пошук + генерація)
Q&A система над корпусом документів з Lapathoniia як LLM бекенд.# base_url має вказувати на api.lapathoniia.top/v1
client = OpenAI(api_key="sk-...", base_url="https://api.lapathoniia.top/v1")
def rag_query(question: str) -> str:
results = collection.query(query_texts=[question], n_results=5)
context = "\n\n".join(results["documents"][0])
response = client.chat.completions.create(
model="MamayLM-Gemma-3-27B-IT-v2.0",
messages=[
{"role": "system", "content": f"Відповідай на основі:\n\n{context}"},
{"role": "user", "content": question},
],
)
return response.choices[0].message.content
3. MCP конектори
Запити до живих держданих природною мовою.result = query_prozorro("Знайди тендери Укрзалізниці відсортовані за сумою")
# → markdown таблиця: назва, сума, дата, переможець, посилання
4. Веб-пошук у пайплайнах
Збагатіть відповіді актуальними веб-даними. Всі російські джерела автоматично відфільтровані.answer, sources = search_and_answer(
"Які останні рішення РНБО щодо санкцій проти Росії?",
api_key="sk-...",
)
5. Пакетна обробка
Обробляйте цілі папки документів, комбінуйте OCR + LLM + аналітику.# Відправте папку Google Drive — по одному завданню на файл
resp = httpx.post(
"https://app.lapathoniia.top/batch/jobs",
headers={"Authorization": f"Bearer {api_key}"},
json={
"job_type": "auto", # auto визначає ocr/chat по типу файлу
"drive": {"folder_id": "YOUR_FOLDER_ID"},
"pipeline": "ocr_then_chat",
"pipeline_instruction": "Витягни: дату, підписантів, суми. JSON.",
"pipeline_model": "MamayLM-Gemma-3-27B-IT-v2.0",
},
)
# Повертає масив job об'єктів — по одному на файл
jobs = resp.json()
Допустимі значення
job_type: ocr, chat, auto (визначає тип по розширенню файлу).
Вхідні дані: або inline (масив повідомлень), або drive (file_id або folder_id).