# extractor_slm.py import json import logging import os import re import requests logger = logging.getLogger(__name__) OLLAMA_URL = os.getenv("OLLAMA_URL", "http://localhost:11434") MODEL = os.getenv("MODEL_NAME", "qwen2.5:7b-instruct-q4_K_M") # Tested empirically: 7B q4 on a T4 takes ~8s on cold first token, # then ~1.5s per doc chunk. 45s covers bad days. _TIMEOUT = 45 _SYSTEM_PROMPT = """\ You are a metadata extractor for research documents. Return ONLY a JSON object — no explanation, no markdown, no surrounding text. Fields to extract: - methodology_type (required): one of experimental | observational | review | simulation | mixed - dataset_source (required): where the data came from - year (required): integer - primary_metric: main eval metric if present - confidence_score: your confidence 0.0–1.0 Output example: {"methodology_type": "experimental", "dataset_source": "ImageNet", "year": 2022, "primary_metric": "top-1 accuracy", "confidence_score": 0.95} """ def call_ollama(doc_text: str) -> dict: payload = { "model": MODEL, "messages": [ {"role": "system", "content": _SYSTEM_PROMPT}, {"role": "user", "content": doc_text[:6000]}, ], "stream": False, "options": { "temperature": 0, "seed": 42, # determinism — this is the whole point }, } try: resp = requests.post(f"{OLLAMA_URL}/api/chat", json=payload, timeout=_TIMEOUT) resp.raise_for_status() except requests.exceptions.Timeout: raise RuntimeError(f"Ollama timed out after {_TIMEOUT}s — is the model loaded?") except requests.exceptions.ConnectionError: raise RuntimeError(f"Can't reach Ollama at {OLLAMA_URL} — is the container running?") raw = resp.json()["message"]["content"].strip() cleaned = re.sub(r"^