-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathdomain_profile.py
More file actions
743 lines (677 loc) · 29.5 KB
/
Copy pathdomain_profile.py
File metadata and controls
743 lines (677 loc) · 29.5 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
"""DomainProfile — domain/locale configuration injection point.
A ``DomainProfile`` bundles the stopwords, metadata patterns, ontology
hints, and tuning parameters needed to apply the generic extraction
pipeline (``PhraseExtractor``, ``EntityLinker``, ``DocumentIngester``) to
a specific corpus.
Library code must NEVER hardcode domain-specific values. Instead accept
a ``DomainProfile`` as a constructor argument. Call sites in ``eval/`` or
user applications build the profile (either in Python or by loading a
TOML file) and inject it.
This is how ``synaptic`` keeps ``src/synaptic/`` domain-agnostic while
still supporting Korean corpora, legal corpora, biomed corpora, etc.
Quick use::
# Python construction:
profile = DomainProfile(
name="myproject",
locale="ko",
stopwords_extra=frozenset({"분류번호", "진단항목"}),
ontology_hints={"규정": NodeKind.RULE},
)
# TOML file (profiles/myproject.toml):
# name = "myproject"
# locale = "ko"
# stopwords_extra = ["분류번호", "진단항목"]
# [ontology_hints]
# "규정" = "RULE"
profile = DomainProfile.load("profiles/myproject.toml")
# Built-in generic profiles (locale only, no domain):
profile = DomainProfile.generic_korean()
profile = DomainProfile.generic_english()
"""
from __future__ import annotations
import re
import tomllib
from dataclasses import dataclass, field
from pathlib import Path
from synaptic.models import NodeKind
# --- Locale-default stopwords ---
#
# These are LANGUAGE-level stopwords — particles, pronouns, common
# function words. Domain stopwords (e.g. metadata schema terms like
# "분류번호") should go into ``DomainProfile.stopwords_extra`` instead.
_STOPWORDS_KO_DEFAULT: frozenset[str] = frozenset(
{
# particle-suffixed forms that leak through extraction
"조직의",
"있는지",
"되는지",
"것이다",
"것이며",
"것이고",
"것인지",
"것으로",
"하기로",
"하기에",
# generic high-frequency terms
"경우",
"내용",
"결과",
"부문",
"해당",
"다음",
"관련",
"포함",
"제공",
"수행",
"실시",
"사항",
"항목",
"있다",
"없다",
"되다",
"하다",
"이다",
"통해",
"대한",
"따라",
"위한",
"관한",
"대해",
# temporal fragments
"년도",
"반기",
"분기",
}
)
_STOPWORDS_EN_DEFAULT: frozenset[str] = frozenset(
{
"the",
"a",
"an",
"is",
"are",
"was",
"were",
"be",
"been",
"have",
"has",
"had",
"do",
"does",
"did",
"will",
"would",
"should",
"could",
"may",
"might",
"must",
"shall",
"can",
"this",
"that",
"these",
"those",
"it",
"its",
"and",
"or",
"but",
"if",
"then",
"else",
"so",
"of",
"in",
"on",
"at",
"to",
"from",
"by",
"with",
"as",
"into",
"through",
"before",
"after",
"during",
"between",
"also",
"just",
"very",
"more",
"most",
"some",
"any",
"all",
}
)
def locale_default_stopwords(locale: str) -> frozenset[str]:
"""Return the built-in stopword set for a locale.
Unknown locales return an empty set rather than raising — callers
can always supplement via ``DomainProfile.stopwords_extra``.
"""
match locale:
case "ko":
return _STOPWORDS_KO_DEFAULT
case "en":
return _STOPWORDS_EN_DEFAULT
case _:
return frozenset()
@dataclass(slots=True)
class DomainProfile:
"""Dependency-injectable domain configuration.
A profile is a plain data bundle. It holds no references to a graph
backend or running state — the same profile instance may be shared
across ingestion, extraction, and query pipelines.
Attributes:
name: Short identifier, e.g. ``"biomed"``, ``"legal"``. Used
in logs and result filenames only.
locale: Primary language code. Drives ``PhraseExtractor``
dispatch. Accepted: ``"ko"``, ``"en"``, ``"ja"``, ``"multi"``.
stopwords_extra: Additional stopwords layered on top of the
locale defaults. Must contain DOMAIN terms (schema labels,
boilerplate phrases) — general-language stopwords belong in
the locale default list.
metadata_strip_patterns: Compiled regexes. Matches are stripped
from chunk content BEFORE phrase extraction. Used for
parser-generated metadata blocks, template headers, boiler-
plate footers.
ontology_hints: Map from free-form category label (folder name,
doc tag, category string) to the ``NodeKind`` that should
classify documents with that label. Example:
``{"규정 및 지침": NodeKind.RULE}``.
reference_patterns: Regexes that capture reference phrases such
as "~에 따라", "~에 의거", "see also". Used by the relation
detector to infer CITES/REFERENCES edges.
entity_hint_patterns: Extra regexes to run on top of the generic
noun-phrase detector, e.g. ``(주)플래티어``, organization
abbreviations in parentheses.
min_df: Minimum number of distinct chunks a phrase must occur
in to be retained as a hub entity.
max_df_ratio: Upper bound on ``df / total_chunks`` — prevents
ubiquitous terms (metadata headers, etc.) from becoming
entities.
openie_min_candidate_entities: Minimum number of DF-retained
candidate entities a chunk must contain before the optional
OpenIE pass spends an LLM call on it.
openie_sample_rate: Deterministic sampling fraction for the
optional OpenIE pass. ``1.0`` means no sampling.
openie_max_chunks: Maximum selected chunks for one OpenIE pass.
openie_max_concurrency: Maximum concurrent LLM extraction calls
for the optional OpenIE post-pass. Graph writes are still
applied in deterministic chunk order.
openie_model_profile: Optional model-specific defaults. Known
values: ``qwen36_local``, ``deepseek_v4_flash``,
``generic_openai_compatible``.
openie_max_output_tokens: Max LLM completion tokens for one
OpenIE chunk extraction.
openie_max_triples_per_chunk: Cap on triples materialized per
chunk so a noisy model cannot flood the graph.
min_phrase_len: Minimum character length per phrase.
max_phrase_len: Maximum character length per phrase.
"""
name: str
locale: str = "multi"
stopwords_extra: frozenset[str] = frozenset()
metadata_strip_patterns: tuple[re.Pattern[str], ...] = ()
ontology_hints: dict[str, NodeKind] = field(default_factory=dict)
reference_patterns: tuple[re.Pattern[str], ...] = ()
entity_hint_patterns: tuple[re.Pattern[str], ...] = ()
min_df: int = 3
max_df_ratio: float = 0.3
min_phrase_len: int = 3
max_phrase_len: int = 20
# Authority ranking — maps NodeKind to trust level (0-10). Higher
# means "more authoritative" at conflict resolution time: a RULE
# outranks a DECISION which outranks an OBSERVATION. The agent
# reads this via ``node_metadata.authority_of()`` when sorting
# evidence across conflicting sources. Default empty dict means
# "unknown authority" — treat all kinds equally.
authority_by_kind: dict[NodeKind, int] = field(default_factory=dict)
# Kind-query hints: keywords that signal a query is looking for a
# specific NodeKind. Used by search.py to boost matching kinds.
# When empty, the built-in defaults in search.py are used.
# Example: {"RULE": ["규칙", "정책", "policy"], "LESSON": ["실패", "error"]}
kind_query_hints: dict[str, list[str]] = field(default_factory=dict)
# Table-query hints: keywords that signal a query is looking for a
# row from a specific structured table (matches ``_table_name``
# property emitted by db_ingester / table_ingester). When a hint
# fires, EvidenceSearch boosts FTS seeds from that table and
# augments the seed pool with a targeted secondary FTS call. Used
# only on corpora ingested from relational sources; invisible to
# document-only corpora. Example for an e-commerce table graph:
# {"sizes": ["사이즈", "size"],
# "sales_partners": ["파트너", "판매처", "판매 파트너"],
# "reviews": ["리뷰", "후기"]}
table_query_hints: dict[str, list[str]] = field(default_factory=dict)
# Document content enrichment — when True, DocumentIngester joins
# the title with the first few chunks' text so Document nodes
# become meaningfully searchable via FTS. Without this Document
# content is just the title (or empty), which is why KRRA top-k
# misses when the query doesn't match the title verbatim.
enrich_document_content: bool = True
document_preview_chars: int = 600
# --- Structural reference linking (WS-A) ---
# When a corpus has a *clean target inventory* — every document
# carries a canonical, low-collision identifier (a statute article
# number, a standard clause code, a manual section id) — explicit
# cross-references in document text can be turned into REFERENCES
# edges without an LLM. ``StructuralReferenceLinker`` consumes these
# three fields; all empty/None means "no structural reference
# linking" (the default — safe for any corpus).
#
# ``reference_token_pattern`` — regex matching a reference token in
# text; its *full match* must equal a target node's key value
# (e.g. ``제\d+조(?:의\d+)?`` matches "제30조", and a node's
# ``article_no`` property is stored as "제30조").
# ``reference_key_property`` — node property holding that canonical
# key. Empty disables linking.
# ``reference_scope_property`` — optional; references resolve only
# among nodes sharing this property value (e.g. "law", so a
# citation resolves within the same statute). Empty = global.
reference_token_pattern: re.Pattern[str] | None = None
reference_key_property: str = ""
reference_scope_property: str = ""
# ``reference_crossscope_pattern`` — optional regex with named groups
# ``scope`` and ``key`` for citations that name their *own* target
# scope, e.g. a statute article citing a different statute
# (「은행법」 제5조). The captured ``scope`` is matched against the
# same ``reference_scope_property`` index, so cross-document
# references resolve too. Spans matched here are excluded from the
# intra-scope matcher so a "제5조" inside such a citation is not
# mis-resolved to the citing document's own scope.
reference_crossscope_pattern: re.Pattern[str] | None = None
# --- Optional LLM OpenIE semantic layer (v0.30 P0) ---
# Off by default. Callers must opt in explicitly and inject an LLM
# extractor; the deterministic structural ingest path never runs
# OpenIE merely because a profile exists.
openie_enabled: bool = False
openie_alias_map: dict[str, str] = field(default_factory=dict)
openie_relation_whitelist: tuple[str, ...] = ()
openie_min_candidate_entities: int = 2
openie_max_candidate_df_ratio: float = 0.3
openie_sample_rate: float = 1.0
openie_max_chunks: int = 1_000_000
openie_max_concurrency: int = 4
openie_model_profile: str = ""
openie_max_output_tokens: int = 1024
openie_max_triples_per_chunk: int = 24
def stopwords(self) -> frozenset[str]:
"""Effective stopword set = locale default ∪ extra."""
return locale_default_stopwords(self.locale) | self.stopwords_extra
# --- Factory constructors ---
@classmethod
def generic_korean(cls, *, name: str = "generic_ko") -> DomainProfile:
"""Locale-only Korean profile. No domain stopwords, no ontology
hints. Safe default for any Korean corpus."""
return cls(name=name, locale="ko")
@classmethod
def generic_english(cls, *, name: str = "generic_en") -> DomainProfile:
"""Locale-only English profile."""
return cls(name=name, locale="en")
@classmethod
def with_references(
cls,
*,
key_property: str,
scope_property: str = "",
crossscope_pattern: str | None = None,
name: str = "references",
locale: str = "multi",
) -> DomainProfile:
"""One-call profile that enables structural cross-reference linking.
This is the minimal profile for building a multi-hop relation
ontology — no TOML file, no hand-written regex. Declare which
node property holds each document's canonical identifier and
:class:`StructuralReferenceLinker` turns in-text citations into
``REFERENCES`` edges automatically during ingest.
Args:
key_property: Node property holding the canonical identifier
each document is cited by (e.g. ``"article_no"``,
``"clause_id"``). The corpus's documents must carry it.
scope_property: Optional property to scope resolution within
(e.g. ``"law"`` — a "제5조" citation resolves inside the
same statute). Empty means global resolution.
crossscope_pattern: Optional regex with named groups
``scope`` and ``key`` for citations that name their own
target scope (e.g. ``「(?P<scope>[^」]+)」\\s*(?P<key>...)``).
name: Profile name.
locale: ``"ko"`` / ``"en"`` / ``"multi"``.
Example::
graph = await SynapticGraph.from_data(
"./statutes/",
profile=DomainProfile.with_references(
key_property="article_no", scope_property="law"
),
)
"""
return cls(
name=name,
locale=locale,
reference_key_property=key_property,
reference_scope_property=scope_property,
reference_crossscope_pattern=(
re.compile(crossscope_pattern) if crossscope_pattern else None
),
)
# --- Serialization ---
def to_dict(self) -> dict[str, object]:
"""Return a JSON/TOML-friendly dict representation.
Compiled regex tuples are serialized via each ``Pattern``'s
``.pattern`` attribute — ``re.compile(s).pattern == s``, so the
round-trip through ``from_dict`` / ``load`` preserves the source
string exactly. ``stopwords_extra`` is sorted so the output is
stable across runs.
"""
out: dict[str, object] = {
"name": self.name,
"locale": self.locale,
"stopwords_extra": sorted(self.stopwords_extra),
"metadata_strip_patterns": [p.pattern for p in self.metadata_strip_patterns],
"reference_patterns": [p.pattern for p in self.reference_patterns],
"entity_hint_patterns": [p.pattern for p in self.entity_hint_patterns],
"min_df": self.min_df,
"max_df_ratio": self.max_df_ratio,
"min_phrase_len": self.min_phrase_len,
"max_phrase_len": self.max_phrase_len,
"enrich_document_content": self.enrich_document_content,
"document_preview_chars": self.document_preview_chars,
"ontology_hints": {k: v.value.upper() for k, v in self.ontology_hints.items()},
"authority_by_kind": {k.value.upper(): v for k, v in self.authority_by_kind.items()},
"reference_key_property": self.reference_key_property,
"reference_scope_property": self.reference_scope_property,
"reference_token_pattern": (
self.reference_token_pattern.pattern
if self.reference_token_pattern is not None
else ""
),
"reference_crossscope_pattern": (
self.reference_crossscope_pattern.pattern
if self.reference_crossscope_pattern is not None
else ""
),
"openie_enabled": self.openie_enabled,
"openie_alias_map": dict(sorted(self.openie_alias_map.items())),
"openie_relation_whitelist": list(self.openie_relation_whitelist),
"openie_min_candidate_entities": self.openie_min_candidate_entities,
"openie_max_candidate_df_ratio": self.openie_max_candidate_df_ratio,
"openie_sample_rate": self.openie_sample_rate,
"openie_max_chunks": self.openie_max_chunks,
"openie_max_concurrency": self.openie_max_concurrency,
"openie_model_profile": self.openie_model_profile,
"openie_max_output_tokens": self.openie_max_output_tokens,
"openie_max_triples_per_chunk": self.openie_max_triples_per_chunk,
}
return out
def save(self, path: Path | str) -> None:
"""Write the profile to a TOML file.
Produces a human-readable TOML that round-trips through
:meth:`load`. Uses a hand-rolled writer because ``tomllib`` is
read-only in the stdlib and we don't want to pull in ``tomli_w``
as a dependency just for this.
The generated file layout mirrors the schema documented in
:meth:`load`: top-level scalars first, arrays second, then the
``[ontology_hints]`` table.
"""
data = self.to_dict()
path = Path(path)
path.parent.mkdir(parents=True, exist_ok=True)
lines: list[str] = []
lines.append(f'name = "{_toml_escape(str(data["name"]))}"')
lines.append(f'locale = "{_toml_escape(str(data["locale"]))}"')
lines.append(f"min_df = {data['min_df']}")
lines.append(f"max_df_ratio = {data['max_df_ratio']}")
lines.append(f"min_phrase_len = {data['min_phrase_len']}")
lines.append(f"max_phrase_len = {data['max_phrase_len']}")
lines.append(f"openie_min_candidate_entities = {data['openie_min_candidate_entities']}")
lines.append(f"openie_max_candidate_df_ratio = {data['openie_max_candidate_df_ratio']}")
lines.append(f"openie_sample_rate = {data['openie_sample_rate']}")
lines.append(f"openie_max_chunks = {data['openie_max_chunks']}")
lines.append(f"openie_max_concurrency = {data['openie_max_concurrency']}")
lines.append(f"openie_max_output_tokens = {data['openie_max_output_tokens']}")
lines.append(f"openie_max_triples_per_chunk = {data['openie_max_triples_per_chunk']}")
if data.get("openie_model_profile"):
lines.append(
f'openie_model_profile = "{_toml_escape(str(data["openie_model_profile"]))}"'
)
if data.get("openie_enabled"):
lines.append("openie_enabled = true")
for ref_key in (
"reference_key_property",
"reference_scope_property",
"reference_token_pattern",
"reference_crossscope_pattern",
):
ref_val = data.get(ref_key, "")
if ref_val:
lines.append(f'{ref_key} = "{_toml_escape(str(ref_val))}"')
lines.append("")
for key in (
"stopwords_extra",
"metadata_strip_patterns",
"reference_patterns",
"entity_hint_patterns",
"openie_relation_whitelist",
):
items = data[key]
if not isinstance(items, list) or not items:
lines.append(f"{key} = []")
continue
lines.append(f"{key} = [")
for item in items:
lines.append(f' "{_toml_escape(str(item))}",')
lines.append("]")
lines.append("")
hints = data["ontology_hints"]
if isinstance(hints, dict) and hints:
lines.append("[ontology_hints]")
for k, v in hints.items():
lines.append(f'"{_toml_escape(str(k))}" = "{_toml_escape(str(v))}"')
lines.append("")
auth = data.get("authority_by_kind", {})
if isinstance(auth, dict) and auth:
lines.append("[authority_by_kind]")
for k, v in auth.items():
lines.append(f'"{_toml_escape(str(k))}" = {v}')
lines.append("")
aliases = data.get("openie_alias_map", {})
if isinstance(aliases, dict) and aliases:
lines.append("[openie_alias_map]")
for k, v in aliases.items():
lines.append(f'"{_toml_escape(str(k))}" = "{_toml_escape(str(v))}"')
lines.append("")
path.write_text("\n".join(lines), encoding="utf-8")
# --- TOML loader ---
@classmethod
def load(cls, path: Path | str) -> DomainProfile:
"""Load a profile from a TOML file.
TOML schema::
name = "myproject"
locale = "ko"
stopwords_extra = ["분류번호", "진단항목"]
metadata_strip_patterns = ["<Document-Metadata>.*?</Document-Metadata>"]
reference_patterns = ["(.+?)에 따라", "(.+?)에 의거"]
entity_hint_patterns = ["\\(([주사재])\\)([\\w]+)"]
min_df = 3
max_df_ratio = 0.3
min_phrase_len = 3
max_phrase_len = 20
[ontology_hints]
"규정 및 지침" = "RULE"
"운영계획" = "DECISION"
"조사 및 평가" = "OBSERVATION"
Unknown keys are ignored. Missing keys fall back to dataclass
defaults.
"""
path = Path(path)
with path.open("rb") as f:
data = tomllib.load(f)
name = data.get("name")
if not isinstance(name, str) or not name:
msg = f"Profile {path}: 'name' is required and must be a non-empty string"
raise ValueError(msg)
locale = str(data.get("locale", "multi"))
stopwords_raw = data.get("stopwords_extra", [])
stopwords_extra = (
frozenset(str(x) for x in stopwords_raw)
if isinstance(stopwords_raw, list)
else frozenset()
)
metadata_strip = _compile_patterns(
data.get("metadata_strip_patterns", []),
re.DOTALL,
source=f"{path}:metadata_strip_patterns",
)
reference_patterns = _compile_patterns(
data.get("reference_patterns", []),
0,
source=f"{path}:reference_patterns",
)
entity_hint_patterns = _compile_patterns(
data.get("entity_hint_patterns", []),
0,
source=f"{path}:entity_hint_patterns",
)
ontology_hints: dict[str, NodeKind] = {}
hints_raw = data.get("ontology_hints", {})
if isinstance(hints_raw, dict):
import warnings
for key, value in hints_raw.items():
if not isinstance(key, str) or not isinstance(value, str):
continue
try:
ontology_hints[key] = NodeKind(value.lower())
except ValueError:
# Try by name for convenience (RULE / rule both work)
try:
ontology_hints[key] = NodeKind[value.upper()]
except KeyError:
# Unknown NodeKind — warn and skip rather than
# fail the whole profile load. Profiles often
# outlive individual NodeKind renames and we
# want the rest of the config (table_query_hints,
# stopwords, etc.) to stay usable.
warnings.warn(
f"Profile {path}: unknown NodeKind "
f"'{value}' for ontology_hints['{key}'] — "
f"skipping. Valid kinds: "
f"{[k.value for k in NodeKind]}",
stacklevel=2,
)
continue
authority_by_kind: dict[NodeKind, int] = {}
auth_raw = data.get("authority_by_kind", {})
if isinstance(auth_raw, dict):
for key, value in auth_raw.items():
if not isinstance(key, str):
continue
kind = None
try:
kind = NodeKind(key.lower())
except ValueError:
try:
kind = NodeKind[key.upper()]
except KeyError:
pass
if kind is not None:
try:
authority_by_kind[kind] = int(value)
except (ValueError, TypeError):
pass
table_query_hints: dict[str, list[str]] = {}
table_hints_raw = data.get("table_query_hints", {})
if isinstance(table_hints_raw, dict):
for table_name, hints in table_hints_raw.items():
if not isinstance(table_name, str) or not isinstance(hints, list):
continue
table_query_hints[table_name] = [str(h) for h in hints if isinstance(h, str) and h]
openie_alias_map: dict[str, str] = {}
aliases_raw = data.get("openie_alias_map", {})
if isinstance(aliases_raw, dict):
for key, value in aliases_raw.items():
if isinstance(key, str) and isinstance(value, str):
openie_alias_map[key] = value
whitelist_raw = data.get("openie_relation_whitelist", [])
openie_relation_whitelist = (
tuple(str(x) for x in whitelist_raw if isinstance(x, str))
if isinstance(whitelist_raw, list)
else ()
)
def _opt_pattern(field_name: str) -> re.Pattern[str] | None:
raw = data.get(field_name, "")
if not isinstance(raw, str) or not raw:
return None
try:
return re.compile(raw)
except re.error as exc:
msg = f"Profile {path}: invalid {field_name} {raw!r} — {exc}"
raise ValueError(msg) from exc
reference_token_pattern = _opt_pattern("reference_token_pattern")
reference_crossscope_pattern = _opt_pattern("reference_crossscope_pattern")
return cls(
name=name,
locale=locale,
stopwords_extra=stopwords_extra,
metadata_strip_patterns=metadata_strip,
ontology_hints=ontology_hints,
reference_patterns=reference_patterns,
entity_hint_patterns=entity_hint_patterns,
min_df=int(data.get("min_df", 3)),
max_df_ratio=float(data.get("max_df_ratio", 0.3)),
min_phrase_len=int(data.get("min_phrase_len", 3)),
max_phrase_len=int(data.get("max_phrase_len", 20)),
authority_by_kind=authority_by_kind,
table_query_hints=table_query_hints,
enrich_document_content=bool(data.get("enrich_document_content", True)),
document_preview_chars=int(data.get("document_preview_chars", 600)),
reference_token_pattern=reference_token_pattern,
reference_crossscope_pattern=reference_crossscope_pattern,
reference_key_property=str(data.get("reference_key_property", "")),
reference_scope_property=str(data.get("reference_scope_property", "")),
openie_enabled=bool(data.get("openie_enabled", False)),
openie_alias_map=openie_alias_map,
openie_relation_whitelist=openie_relation_whitelist,
openie_min_candidate_entities=int(data.get("openie_min_candidate_entities", 2)),
openie_max_candidate_df_ratio=float(data.get("openie_max_candidate_df_ratio", 0.3)),
openie_sample_rate=float(data.get("openie_sample_rate", 1.0)),
openie_max_chunks=int(data.get("openie_max_chunks", 1_000_000)),
openie_max_concurrency=int(data.get("openie_max_concurrency", 4)),
openie_model_profile=str(data.get("openie_model_profile", "")),
openie_max_output_tokens=int(data.get("openie_max_output_tokens", 1024)),
openie_max_triples_per_chunk=int(data.get("openie_max_triples_per_chunk", 24)),
)
def _toml_escape(value: str) -> str:
"""Minimal TOML basic-string escape.
Handles the characters that would corrupt a double-quoted TOML
string: backslash, double-quote, and control characters. Full TOML
escape rules are more permissive, but this subset is enough for
profile round-tripping and keeps the writer dependency-free.
"""
return (
value.replace("\\", "\\\\")
.replace('"', '\\"')
.replace("\n", "\\n")
.replace("\r", "\\r")
.replace("\t", "\\t")
)
def _compile_patterns(
raw: object,
flags: int,
*,
source: str,
) -> tuple[re.Pattern[str], ...]:
"""Compile a list of regex strings into a tuple of Pattern objects."""
if not isinstance(raw, list):
return ()
compiled: list[re.Pattern[str]] = []
for i, item in enumerate(raw):
if not isinstance(item, str):
continue
try:
compiled.append(re.compile(item, flags))
except re.error as exc:
msg = f"{source}[{i}]: invalid regex '{item}' — {exc}"
raise ValueError(msg) from exc
return tuple(compiled)