Add Presidio detector service returning typed findings (+ custom ROC-ID recognizer)
This commit is contained in:
59
presidio/app.py
Normal file
59
presidio/app.py
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
"""
|
||||||
|
PII detector service (step 2) -- Presidio Analyzer behind a tiny HTTP API.
|
||||||
|
|
||||||
|
CONTRACT: it returns typed FINDINGS (entity type + offsets + the original text),
|
||||||
|
it does NOT redact. Fusion owns the mint/splice/vault-write. This is the whole
|
||||||
|
"detection != redaction" point from the design discussion.
|
||||||
|
|
||||||
|
Demo scope: tuned to catch the scripted query's entities (a Chinese name + a
|
||||||
|
Taiwan ROC national ID). NOT production-grade zh-TW recall -- see README caveats.
|
||||||
|
"""
|
||||||
|
from fastapi import FastAPI
|
||||||
|
from pydantic import BaseModel
|
||||||
|
from presidio_analyzer import AnalyzerEngine, PatternRecognizer, Pattern
|
||||||
|
from presidio_analyzer.nlp_engine import NlpEngineProvider
|
||||||
|
|
||||||
|
app = FastAPI(title="hncb-deid-detector")
|
||||||
|
|
||||||
|
|
||||||
|
# spaCy zh model gives us Chinese PERSON detection. See Dockerfile.
|
||||||
|
_provider = NlpEngineProvider(nlp_configuration={
|
||||||
|
"nlp_engine_name": "spacy",
|
||||||
|
"models": [{"lang_code": "zh", "model_name": "zh_core_web_lg"}],
|
||||||
|
})
|
||||||
|
analyzer = AnalyzerEngine(nlp_engine=_provider.create_engine(), supported_languages=["zh"])
|
||||||
|
|
||||||
|
# Custom recognizer: Taiwan ROC national ID (1 letter + [1|2] + 8 digits).
|
||||||
|
roc_id = PatternRecognizer(
|
||||||
|
supported_entity="TW_ROC_ID",
|
||||||
|
supported_language="zh",
|
||||||
|
patterns=[Pattern(name="roc_id", regex=r"\b[A-Z][12]\d{8}\b", score=0.85)],
|
||||||
|
context=["身分證", "身份證", "統一編號", "ID"],
|
||||||
|
)
|
||||||
|
analyzer.registry.add_recognizer(roc_id)
|
||||||
|
|
||||||
|
|
||||||
|
class AnalyzeRequest(BaseModel):
|
||||||
|
text: str
|
||||||
|
language: str = "zh"
|
||||||
|
|
||||||
|
|
||||||
|
@app.get("/health")
|
||||||
|
def health():
|
||||||
|
return {"status": "ok"}
|
||||||
|
|
||||||
|
|
||||||
|
@app.post("/analyze")
|
||||||
|
def analyze(req: AnalyzeRequest):
|
||||||
|
results = analyzer.analyze(text=req.text, language=req.language)
|
||||||
|
# Return typed findings incl. the original substring. Fusion tokenizes from this.
|
||||||
|
return [
|
||||||
|
{
|
||||||
|
"entity_type": r.entity_type,
|
||||||
|
"start": r.start,
|
||||||
|
"end": r.end,
|
||||||
|
"score": round(r.score, 3),
|
||||||
|
"text": req.text[r.start:r.end],
|
||||||
|
}
|
||||||
|
for r in results
|
||||||
|
]
|
||||||
Reference in New Issue
Block a user