forked from theAfish/mat_know_base
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathbasic_usage.py
More file actions
177 lines (146 loc) · 7.51 KB
/
Copy pathbasic_usage.py
File metadata and controls
177 lines (146 loc) · 7.51 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
"""
Basic usage of the Materials Knowledge Base Python API.
Prerequisites:
- Docker services running: `make up`
- Package installed: `pip install -e ".[dev,materials,server]"`
- .env configured with LLM credentials (see README.md)
This example walks through the full pipeline:
1. Setup database
2. Ingest files
3. Process into LLM-readable formats
4. Extract knowledge frames via LLM (with multi-pass review)
5. Query the knowledge base
6. Create a space and run projections
7. Work with feedback
"""
from mkb import api
# ── 1. Setup ─────────────────────────────────────────────────────
# Create database tables (idempotent — safe to call multiple times)
api.setup()
# ── 2. Ingest ────────────────────────────────────────────────────
# Ingest a directory containing a paper and its supplementary files.
# All files in the directory become one "project" (research package).
result = api.ingest(
"./data/papers/smith2024_catalysis",
label="Smith 2024 - Catalysis",
)
print(f"Ingested: {result}")
# → {'total': 4, 'ingested': 4, 'duplicates': 0, 'errors': 0, 'project_id': '...'}
# Or sync all project subfolders at once:
# api.sync("./data/papers")
# ── 3. Process ───────────────────────────────────────────────────
# Convert raw files to LLM-readable formats:
# PDF → Markdown + images + tables
# CSV/XLSX → Parquet
# Images → JSON metadata
result = api.process()
print(f"Processed: {result}")
# ── 4. Extract ───────────────────────────────────────────────────
# Run LLM extraction on all projects without a completed frame.
# The agent reads processed data and produces one knowledge frame per project.
#
# max_passes=1: initial extraction only
# max_passes=2+: initial extraction + review passes for refinement
result = api.extract(max_passes=2, verbose=True)
print(f"Extraction: {result}")
# Or extract one specific project:
# api.extract(project_id="<project-uuid>", max_passes=2)
# ── 5. Query Knowledge Frames ───────────────────────────────────
# List all frames
frames = api.list_frames()
for f in frames:
print(f" Project {f['project_id']} status={f['status']} v{f.get('extraction_version', 0)}")
print(f" Summary: {f['extraction_summary']}")
# Get full frame for a specific project
projects = api.list_projects()
if projects:
project_id = projects[0]["project_id"]
frame = api.get_frame(project_id)
if frame:
content = frame["content"]
# Paper metadata (always present)
paper = content.get("paper", {})
print(f"\nPaper: {paper.get('title', 'N/A')}")
print(f"Domain: {content.get('domain', 'N/A')}")
# The rest of the keys are agent-decided — iterate dynamically
for key, items in content.items():
if key in ("paper", "domain"):
continue
if isinstance(items, list):
print(f"\n{key} ({len(items)} items):")
for item in items[:3]: # show first 3
if isinstance(item, dict):
# Show evidence level and main descriptive field
ev = item.get("evidence_level", "?")
desc = next(
(item[k] for k in ("name", "claim", "property", "description", "method")
if k in item),
str(item)[:80]
)
print(f" [L{ev}] {desc}")
# View extraction history
history = api.get_extraction_history(project_id)
print(f"\nExtraction passes: {len(history)}")
for h in history:
print(f" Pass {h['pass_number']} ({h['pass_type']})")
# ── 6. Spaces & Projections ─────────────────────────────────────
# Create a space (domain-specific extraction schema)
space_result = api.create_space(
name="catalysis",
domain="heterogeneous catalysis",
extraction_schema={
"catalysts": {
"type": "list",
"description": "All catalyst materials studied, including composition and performance metrics.",
"item_schema": {
"name": {"type": "string", "required": True},
"composition": {"type": "string", "required": True},
"support": {"type": "string", "required": False},
"surface_area_m2_g": {"type": "number", "required": False},
"selectivity_percent": {"type": "number", "required": False},
"conversion_percent": {"type": "number", "required": False},
}
},
"reactions": {
"type": "list",
"description": "All chemical reactions described, with reactants, products, and conditions.",
"item_schema": {
"name": {"type": "string", "required": True},
"reactants": {"type": "list", "required": True},
"products": {"type": "list", "required": True},
"temperature_C": {"type": "number", "required": False},
"pressure_atm": {"type": "number", "required": False},
}
},
},
system_prompt="Extract catalyst materials and reactions from this paper.",
description="Heterogeneous catalysis data extraction",
)
print(f"Space created: {space_result}")
# Run projection on a project's frame
# api.project(space_id=space_result["space_id"], project_id=project_id)
# Or project all completed frames:
# api.project_all(space_id=space_result["space_id"])
# List projections
# projections = api.list_projections(space_id=space_result["space_id"])
# ── 7. Feedback ──────────────────────────────────────────────────
# During projection, agents may flag unclear data.
# List open feedback:
# feedback = api.list_feedback(project_id=project_id, status="OPEN")
# Run feedback review (KB agent reviews and resolves feedback):
# api.review_feedback(project_id=project_id)
# Manually resolve feedback:
# api.resolve_feedback(feedback_id="...", status="RESOLVED", notes="Fixed")
# ── 8. React UI ──────────────────────────────────────────────────
# Launch the API and current React interface, then open http://127.0.0.1:5173:
# make dev
# ── 9. Other queries ─────────────────────────────────────────────
# List all projects
for p in api.list_projects():
print(f" {p['project_id']} {p['label']} {p['asset_count']} files frame: {p['frame_status']}")
# List assets in a project
# for a in api.list_assets(project_id=project_id):
# print(f" {a['asset_id']} {a['filename']} ({a['mime_type']})")
# ── 10. Reset (destructive development operation) ───────────────────────
# The CLI requires an exact interactive confirmation before dropping all tables:
# .venv/bin/python -m mkb.cli reset-db