getting a bit further with the frontend and demoing the analysis to matti

This commit is contained in:
itsamejms
2025-09-19 14:57:03 +02:00
parent a44833d054
commit b2e412b3c5
8 changed files with 1354 additions and 233 deletions
+31 -1
View File
@@ -21,4 +21,34 @@ def get_ollama_embedding(text, model="nomic-embed-text"):
}
response = requests.post(url, json=payload)
response.raise_for_status()
return response.json()["embedding"]
return response.json()["embedding"]
def parse_combined_text_to_dict(s: str) -> dict:
"""Parse a string of the form 'key1: value1 | key2: value2' into a dict.
Rules:
- Split on ' | ' to get key:value segments.
- For each segment, split on the first ':' to separate key and value.
- Strip whitespace. If a value is 'nan' (case-insensitive) or empty, use None.
- Return a dict mapping keys to values or None.
"""
if not s:
return {}
result = {}
parts = [p.strip() for p in s.split("|")]
for part in parts:
if not part:
continue
# split on the first colon
if ':' in part:
k, v = part.split(':', 1)
k = k.strip()
v = v.strip()
if v.lower() == 'nan' or v == '':
result[k] = None
else:
result[k] = v
else:
# fallback: store whole segment under a numeric key
result.setdefault('_extra', []).append(part)
return result