Release v4928.1.4.2 stable
This commit is contained in:
86
scripts/apply_migrations.py
Executable file
86
scripts/apply_migrations.py
Executable file
@@ -0,0 +1,86 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Apply SQL migrations stored in ./migrations.
|
||||
|
||||
Usage:
|
||||
python scripts/apply_migrations.py
|
||||
python scripts/apply_migrations.py --dry-run
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
import os
|
||||
import sys
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
MIGRATIONS_DIR = ROOT / "migrations"
|
||||
|
||||
|
||||
def ensure_ledger(conn) -> None:
|
||||
conn.execute(text("""
|
||||
CREATE TABLE IF NOT EXISTS schema_migrations (
|
||||
version TEXT PRIMARY KEY,
|
||||
name TEXT NOT NULL,
|
||||
applied_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
"""))
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--dry-run", action="store_true", help="List pending migrations without applying them.")
|
||||
args = parser.parse_args()
|
||||
|
||||
files = sorted(MIGRATIONS_DIR.glob("*.sql"))
|
||||
if not files:
|
||||
print("No migrations found.")
|
||||
return 0
|
||||
|
||||
with engine.begin() as conn:
|
||||
ensure_ledger(conn)
|
||||
applied = {
|
||||
row[0]
|
||||
for row in conn.execute(text("SELECT version FROM schema_migrations"))
|
||||
}
|
||||
|
||||
pending = []
|
||||
for path in files:
|
||||
version = path.stem.split("_", 1)[0]
|
||||
if version not in applied:
|
||||
pending.append((version, path))
|
||||
|
||||
if not pending:
|
||||
print("No pending migrations.")
|
||||
return 0
|
||||
|
||||
print("Pending migrations:")
|
||||
for version, path in pending:
|
||||
print(f"- {version}: {path.name}")
|
||||
|
||||
if args.dry_run:
|
||||
return 0
|
||||
|
||||
for version, path in pending:
|
||||
sql = path.read_text()
|
||||
print(f"Applying {path.name}...")
|
||||
conn.execute(text(sql))
|
||||
conn.execute(text("""
|
||||
INSERT INTO schema_migrations(version, name)
|
||||
VALUES (:version, :name)
|
||||
ON CONFLICT (version) DO NOTHING
|
||||
"""), {"version": version, "name": path.name})
|
||||
|
||||
print("Migrations applied.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
338
scripts/backfill_chatwoot_inbox_to_clientflow.py
Executable file
338
scripts/backfill_chatwoot_inbox_to_clientflow.py
Executable file
@@ -0,0 +1,338 @@
|
||||
#!/usr/bin/env python3
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
|
||||
def env(name: str, default: str = "") -> str:
|
||||
return os.getenv(name, default).strip()
|
||||
|
||||
|
||||
CHATWOOT_BASE_URL = env("CHATWOOT_BASE_URL").rstrip("/")
|
||||
CHATWOOT_ACCOUNT_ID = env("CHATWOOT_ACCOUNT_ID")
|
||||
CHATWOOT_API_TOKEN = env("CHATWOOT_API_TOKEN")
|
||||
CLIENTFLOW_WEBHOOK_SECRET = env("CLIENTFLOW_WEBHOOK_SECRET")
|
||||
CLIENTFLOW_WEBHOOK_URL = env("CLIENTFLOW_WEBHOOK_URL", "http://127.0.0.1:8020/webhooks/chatwoot")
|
||||
|
||||
BACKFILL_DAYS = int(env("BACKFILL_DAYS", "5"))
|
||||
BACKFILL_TO_CLIENTFLOW = env("BACKFILL_TO_CLIENTFLOW", "false").lower() == "true"
|
||||
BACKFILL_STATUSES = [s.strip() for s in env("BACKFILL_STATUSES", "open,pending").split(",") if s.strip()]
|
||||
BACKFILL_MAX_PAGES = int(env("BACKFILL_MAX_PAGES", "20"))
|
||||
|
||||
|
||||
def request_json(method: str, url: str, body: Optional[Dict[str, Any]] = None, headers: Optional[Dict[str, str]] = None) -> Dict[str, Any]:
|
||||
data = None
|
||||
final_headers = headers.copy() if headers else {}
|
||||
|
||||
if body is not None:
|
||||
data = json.dumps(body, ensure_ascii=False).encode("utf-8")
|
||||
final_headers["Content-Type"] = "application/json"
|
||||
|
||||
req = urllib.request.Request(url, data=data, headers=final_headers, method=method)
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=45) as resp:
|
||||
raw = resp.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": 200 <= resp.status < 300,
|
||||
"status": resp.status,
|
||||
"json": json.loads(raw) if raw else {},
|
||||
"raw": raw,
|
||||
}
|
||||
except urllib.error.HTTPError as e:
|
||||
raw = e.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": False,
|
||||
"status": e.code,
|
||||
"json": None,
|
||||
"raw": raw,
|
||||
}
|
||||
|
||||
|
||||
def chatwoot_headers() -> Dict[str, str]:
|
||||
return {
|
||||
"api_access_token": CHATWOOT_API_TOKEN,
|
||||
"Accept": "application/json",
|
||||
}
|
||||
|
||||
|
||||
def payload_list(data: Any) -> List[Dict[str, Any]]:
|
||||
if isinstance(data, list):
|
||||
return data
|
||||
|
||||
if not isinstance(data, dict):
|
||||
return []
|
||||
|
||||
candidates = [
|
||||
data.get("payload"),
|
||||
data.get("data", {}).get("payload") if isinstance(data.get("data"), dict) else None,
|
||||
data.get("data"),
|
||||
data.get("messages"),
|
||||
]
|
||||
|
||||
for item in candidates:
|
||||
if isinstance(item, list):
|
||||
return item
|
||||
|
||||
return []
|
||||
|
||||
|
||||
def ts_to_datetime(value: Any) -> Optional[datetime]:
|
||||
if value is None:
|
||||
return None
|
||||
|
||||
try:
|
||||
if isinstance(value, (int, float)):
|
||||
return datetime.fromtimestamp(float(value), tz=timezone.utc)
|
||||
|
||||
s = str(value).replace("Z", "+00:00")
|
||||
return datetime.fromisoformat(s).astimezone(timezone.utc)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def conversation_last_activity(conversation: Dict[str, Any]) -> Optional[datetime]:
|
||||
for key in ["last_activity_at", "updated_at", "created_at"]:
|
||||
dt = ts_to_datetime(conversation.get(key))
|
||||
if dt:
|
||||
return dt
|
||||
return None
|
||||
|
||||
|
||||
def fetch_conversations(status: str) -> List[Dict[str, Any]]:
|
||||
all_items: List[Dict[str, Any]] = []
|
||||
|
||||
for page in range(1, BACKFILL_MAX_PAGES + 1):
|
||||
query = urllib.parse.urlencode({
|
||||
"status": status,
|
||||
"page": page,
|
||||
})
|
||||
|
||||
url = f"{CHATWOOT_BASE_URL}/api/v1/accounts/{CHATWOOT_ACCOUNT_ID}/conversations?{query}"
|
||||
result = request_json("GET", url, headers=chatwoot_headers())
|
||||
|
||||
if not result["ok"]:
|
||||
print(f"ERRO Chatwoot conversations status={status} page={page}: {result['status']} {result['raw'][:300]}")
|
||||
break
|
||||
|
||||
items = payload_list(result["json"])
|
||||
|
||||
if not items:
|
||||
break
|
||||
|
||||
all_items.extend(items)
|
||||
|
||||
if len(items) < 10:
|
||||
break
|
||||
|
||||
return all_items
|
||||
|
||||
|
||||
def fetch_messages(conversation_id: str) -> List[Dict[str, Any]]:
|
||||
url = f"{CHATWOOT_BASE_URL}/api/v1/accounts/{CHATWOOT_ACCOUNT_ID}/conversations/{conversation_id}/messages"
|
||||
result = request_json("GET", url, headers=chatwoot_headers())
|
||||
|
||||
if not result["ok"]:
|
||||
print(f"ERRO Chatwoot messages conversation={conversation_id}: {result['status']} {result['raw'][:300]}")
|
||||
return []
|
||||
|
||||
return payload_list(result["json"])
|
||||
|
||||
|
||||
def is_incoming(message: Dict[str, Any]) -> bool:
|
||||
mt = message.get("message_type")
|
||||
return mt == "incoming" or mt == 0 or str(mt).lower() == "incoming"
|
||||
|
||||
|
||||
def message_created_at(message: Dict[str, Any]) -> datetime:
|
||||
return ts_to_datetime(message.get("created_at")) or datetime.fromtimestamp(0, tz=timezone.utc)
|
||||
|
||||
|
||||
def sender_from_conversation(conversation: Dict[str, Any], message: Dict[str, Any]) -> Dict[str, Any]:
|
||||
sender = {}
|
||||
|
||||
meta = conversation.get("meta") or {}
|
||||
if isinstance(meta, dict) and isinstance(meta.get("sender"), dict):
|
||||
sender.update(meta.get("sender") or {})
|
||||
|
||||
if isinstance(conversation.get("contact"), dict):
|
||||
sender.update({k: v for k, v in conversation["contact"].items() if v is not None})
|
||||
|
||||
if isinstance(message.get("sender"), dict):
|
||||
sender.update({k: v for k, v in message["sender"].items() if v is not None})
|
||||
|
||||
return sender
|
||||
|
||||
|
||||
def sign_body(body_raw: str) -> Dict[str, str]:
|
||||
ts = str(int(time.time()))
|
||||
msg = ts.encode("utf-8") + b"." + body_raw.encode("utf-8")
|
||||
sig = "sha256=" + hmac.new(
|
||||
CLIENTFLOW_WEBHOOK_SECRET.encode("utf-8"),
|
||||
msg,
|
||||
hashlib.sha256,
|
||||
).hexdigest()
|
||||
|
||||
return {
|
||||
"Content-Type": "application/json",
|
||||
"X-Chatwoot-Timestamp": ts,
|
||||
"X-Chatwoot-Signature": sig,
|
||||
}
|
||||
|
||||
|
||||
def post_to_clientflow(payload: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# Importante: assinar e enviar exatamente o mesmo body_raw.
|
||||
# Se o JSON for reformatado depois da assinatura, o webhook rejeita com 401.
|
||||
body_raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":"))
|
||||
headers = sign_body(body_raw)
|
||||
|
||||
req = urllib.request.Request(
|
||||
CLIENTFLOW_WEBHOOK_URL,
|
||||
data=body_raw.encode("utf-8"),
|
||||
headers=headers,
|
||||
method="POST",
|
||||
)
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=45) as resp:
|
||||
raw = resp.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": 200 <= resp.status < 300,
|
||||
"status": resp.status,
|
||||
"json": json.loads(raw) if raw else {},
|
||||
"raw": raw,
|
||||
}
|
||||
except urllib.error.HTTPError as e:
|
||||
raw = e.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": False,
|
||||
"status": e.code,
|
||||
"json": None,
|
||||
"raw": raw,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
required = {
|
||||
"CHATWOOT_BASE_URL": CHATWOOT_BASE_URL,
|
||||
"CHATWOOT_ACCOUNT_ID": CHATWOOT_ACCOUNT_ID,
|
||||
"CHATWOOT_API_TOKEN": CHATWOOT_API_TOKEN,
|
||||
"CLIENTFLOW_WEBHOOK_SECRET": CLIENTFLOW_WEBHOOK_SECRET,
|
||||
}
|
||||
|
||||
missing = [k for k, v in required.items() if not v]
|
||||
if missing:
|
||||
raise SystemExit(f"Faltam variáveis: {', '.join(missing)}")
|
||||
|
||||
cutoff = datetime.now(timezone.utc) - timedelta(days=BACKFILL_DAYS)
|
||||
|
||||
print(f"BACKFILL_DAYS={BACKFILL_DAYS}")
|
||||
print(f"BACKFILL_STATUSES={BACKFILL_STATUSES}")
|
||||
print(f"BACKFILL_TO_CLIENTFLOW={BACKFILL_TO_CLIENTFLOW}")
|
||||
print(f"CUTOFF={cutoff.isoformat()}")
|
||||
|
||||
seen_conversations = set()
|
||||
selected = 0
|
||||
posted = 0
|
||||
failed = 0
|
||||
skipped = 0
|
||||
|
||||
for status in BACKFILL_STATUSES:
|
||||
conversations = fetch_conversations(status)
|
||||
print(f"--- status={status} conversations={len(conversations)}")
|
||||
|
||||
for conv in conversations:
|
||||
conv_id = str(conv.get("id") or "")
|
||||
if not conv_id or conv_id in seen_conversations:
|
||||
continue
|
||||
|
||||
seen_conversations.add(conv_id)
|
||||
|
||||
last_activity = conversation_last_activity(conv)
|
||||
if last_activity and last_activity < cutoff:
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
messages = fetch_messages(conv_id)
|
||||
incoming_messages = [
|
||||
m for m in messages
|
||||
if is_incoming(m)
|
||||
and not m.get("private")
|
||||
and str(m.get("content") or "").strip()
|
||||
and message_created_at(m) >= cutoff
|
||||
]
|
||||
|
||||
if not incoming_messages:
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
incoming_messages.sort(key=message_created_at)
|
||||
last_msg = incoming_messages[-1]
|
||||
|
||||
content = str(last_msg.get("content") or "").strip()
|
||||
sender = sender_from_conversation(conv, last_msg)
|
||||
contact_id = str(sender.get("id") or conv.get("contact_id") or conv.get("contact", {}).get("id") or "")
|
||||
|
||||
selected += 1
|
||||
|
||||
payload = {
|
||||
"event": "message_created",
|
||||
"message": {
|
||||
"id": str(last_msg.get("id") or f"backfill-{conv_id}"),
|
||||
"content": content,
|
||||
"message_type": "incoming",
|
||||
"conversation_id": conv_id,
|
||||
"sender": sender,
|
||||
"created_at": last_msg.get("created_at"),
|
||||
},
|
||||
"conversation": {
|
||||
"id": conv_id,
|
||||
"status": conv.get("status") or status,
|
||||
"contact": sender,
|
||||
"meta": {
|
||||
"sender": sender,
|
||||
},
|
||||
},
|
||||
"backfill": {
|
||||
"source": "chatwoot_inbox_last_days",
|
||||
"days": BACKFILL_DAYS,
|
||||
"status": status,
|
||||
"last_activity_at": conv.get("last_activity_at"),
|
||||
},
|
||||
}
|
||||
|
||||
print("---")
|
||||
print(f"conversation={conv_id} contact={contact_id} msg={payload['message']['id']}")
|
||||
print(f"content={content[:160].replace(chr(10), ' ')}")
|
||||
|
||||
if not BACKFILL_TO_CLIENTFLOW:
|
||||
print("DRY_RUN")
|
||||
continue
|
||||
|
||||
result = post_to_clientflow(payload)
|
||||
|
||||
if result["ok"]:
|
||||
print(f"POSTED status={result['status']}")
|
||||
posted += 1
|
||||
else:
|
||||
print(f"FAILED status={result['status']} body={result['raw'][:500]}")
|
||||
failed += 1
|
||||
|
||||
print("---")
|
||||
print(f"selected={selected}")
|
||||
print(f"posted={posted}")
|
||||
print(f"failed={failed}")
|
||||
print(f"skipped={skipped}")
|
||||
|
||||
return 0 if failed == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
312
scripts/backfill_jasmin_opportunity_details.py
Executable file
312
scripts/backfill_jasmin_opportunity_details.py
Executable file
@@ -0,0 +1,312 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Backfill Jasmin document details into existing opportunities.
|
||||
|
||||
Use when an opportunity was created from a Jasmin reconciliation item before
|
||||
v4926.6, so it still has no commercial_documents/opportunity_items/value.
|
||||
|
||||
Examples:
|
||||
PYTHONPATH=. python scripts/backfill_jasmin_opportunity_details.py \
|
||||
--opportunity-id a4e210f5-b870-48b1-882d-b55c71fbfd38
|
||||
|
||||
PYTHONPATH=. python scripts/backfill_jasmin_opportunity_details.py \
|
||||
--document-number ORC.ORC2026.156
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
import sys
|
||||
from decimal import Decimal, InvalidOperation
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.reconciliation_service import (
|
||||
_apply_jasmin_documents_to_opportunity, # noqa: PLC2701 - deliberate operator backfill script
|
||||
_jasmin_document_lines_from_item, # noqa: PLC2701
|
||||
_jasmin_document_totals, # noqa: PLC2701
|
||||
_payload_record, # noqa: PLC2701
|
||||
)
|
||||
|
||||
|
||||
def _as_text(value: Any) -> str:
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def _money_value(value: Any) -> Any:
|
||||
if isinstance(value, dict):
|
||||
for key in ("amount", "baseAmount", "reportingAmount", "value"):
|
||||
if value.get(key) not in (None, ""):
|
||||
return value.get(key)
|
||||
return None
|
||||
return value
|
||||
|
||||
|
||||
def _decimal_or_none(value: Any) -> Optional[str]:
|
||||
value = _money_value(value)
|
||||
if value in (None, ""):
|
||||
return None
|
||||
try:
|
||||
return str(Decimal(str(value).replace(",", ".")).quantize(Decimal("0.01")))
|
||||
except (InvalidOperation, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _json(value: Any) -> str:
|
||||
return json.dumps(value, ensure_ascii=False, default=str)
|
||||
|
||||
|
||||
def _ids_from_metadata(metadata: Any) -> List[str]:
|
||||
if not isinstance(metadata, dict):
|
||||
return []
|
||||
ids: List[str] = []
|
||||
for key in ("created_from_reconciliation_item_id", "reconciliation_item_id"):
|
||||
value = metadata.get(key)
|
||||
if value:
|
||||
ids.append(str(value))
|
||||
for key in ("item_ids", "reconciliation_item_ids"):
|
||||
value = metadata.get(key)
|
||||
if isinstance(value, list):
|
||||
ids.extend(str(v) for v in value if v)
|
||||
return list(dict.fromkeys(ids))
|
||||
|
||||
|
||||
def _load_opportunity(opportunity_id: Optional[str], document_number: Optional[str]) -> Optional[Dict[str, Any]]:
|
||||
with engine.begin() as conn:
|
||||
if opportunity_id:
|
||||
row = conn.execute(text("""
|
||||
SELECT id::text, title, value_amount, product_interest, local_customer_id::text,
|
||||
customer_name, customer_email, metadata
|
||||
FROM opportunities
|
||||
WHERE id = CAST(:id AS UUID)
|
||||
LIMIT 1
|
||||
"""), {"id": opportunity_id}).mappings().first()
|
||||
return dict(row) if row else None
|
||||
if document_number:
|
||||
row = conn.execute(text("""
|
||||
SELECT id::text, title, value_amount, product_interest, local_customer_id::text,
|
||||
customer_name, customer_email, metadata
|
||||
FROM opportunities
|
||||
WHERE metadata->>'document_number' = :document_number
|
||||
OR title ILIKE '%' || :document_number || '%'
|
||||
ORDER BY updated_at DESC
|
||||
LIMIT 1
|
||||
"""), {"document_number": document_number}).mappings().first()
|
||||
return dict(row) if row else None
|
||||
return None
|
||||
|
||||
|
||||
def _load_candidate_items(opportunity: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||
metadata = opportunity.get("metadata") if isinstance(opportunity.get("metadata"), dict) else {}
|
||||
ids = _ids_from_metadata(metadata)
|
||||
external_id = _as_text(metadata.get("external_id"))
|
||||
document_number = _as_text(metadata.get("document_number"))
|
||||
opportunity_id = _as_text(opportunity.get("id"))
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text("""
|
||||
SELECT id::text, source_system, external_type, external_id, title, description,
|
||||
status, priority, suggested_action, confidence, opportunity_id::text,
|
||||
customer_id::text, customer_name, customer_email, customer_tax_id,
|
||||
document_number, document_date, amount, currency, payload,
|
||||
resolution_note, created_at, updated_at, resolved_at
|
||||
FROM reconciliation_items
|
||||
WHERE source_system = 'jasmin'
|
||||
AND (
|
||||
opportunity_id = CAST(:opportunity_id AS UUID)
|
||||
OR (CAST(:ids AS TEXT[]) IS NOT NULL AND id::text = ANY(CAST(:ids AS TEXT[])))
|
||||
OR (CAST(:external_id AS TEXT) <> '' AND external_id = CAST(:external_id AS TEXT))
|
||||
OR (CAST(:document_number AS TEXT) <> '' AND document_number = CAST(:document_number AS TEXT))
|
||||
OR (CAST(:document_number AS TEXT) <> '' AND payload::text ILIKE '%' || CAST(:document_number AS TEXT) || '%')
|
||||
)
|
||||
ORDER BY updated_at DESC, created_at DESC
|
||||
"""), {
|
||||
"opportunity_id": opportunity_id,
|
||||
"ids": ids or [],
|
||||
"external_id": external_id,
|
||||
"document_number": document_number,
|
||||
}).mappings().all()
|
||||
|
||||
# De-duplicate while keeping recency order.
|
||||
seen = set()
|
||||
result = []
|
||||
for row in rows:
|
||||
item = dict(row)
|
||||
item_id = item.get("id")
|
||||
if item_id in seen:
|
||||
continue
|
||||
seen.add(item_id)
|
||||
result.append(item)
|
||||
return result
|
||||
|
||||
|
||||
async def _fetch_jasmin_detail_async(item: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
external_type = _as_text(item.get("external_type"))
|
||||
external_id = _as_text(item.get("external_id"))
|
||||
if not external_id:
|
||||
return None
|
||||
from app.jasmin_client import JasminClient
|
||||
|
||||
client = JasminClient()
|
||||
if external_type == "jasmin_quotation":
|
||||
return await client.get_quotation(external_id)
|
||||
if external_type == "jasmin_invoice":
|
||||
return await client.get_invoice(external_id)
|
||||
# Some tenants represent pro-forma as a quotation. Try quotation detail as a
|
||||
# conservative fallback when the external id is present.
|
||||
if external_type == "jasmin_proforma":
|
||||
try:
|
||||
return await client.get_quotation(external_id)
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _with_jasmin_detail(item: Dict[str, Any], *, fetch_detail: bool) -> Dict[str, Any]:
|
||||
if not fetch_detail:
|
||||
return item
|
||||
existing_lines = _jasmin_document_lines_from_item(item)
|
||||
if existing_lines:
|
||||
return item
|
||||
try:
|
||||
detail = asyncio.run(_fetch_jasmin_detail_async(item))
|
||||
except Exception as exc:
|
||||
item = dict(item)
|
||||
payload = item.get("payload") if isinstance(item.get("payload"), dict) else {}
|
||||
item["payload"] = {
|
||||
**payload,
|
||||
"detail_fetch_error": f"{type(exc).__name__}: {exc}",
|
||||
}
|
||||
return item
|
||||
if not isinstance(detail, dict):
|
||||
return item
|
||||
|
||||
payload = item.get("payload") if isinstance(item.get("payload"), dict) else {}
|
||||
enriched = dict(item)
|
||||
enriched["payload"] = {
|
||||
**payload,
|
||||
"record": detail,
|
||||
"detail_source": "jasmin_api",
|
||||
"previous_record": payload.get("record"),
|
||||
}
|
||||
|
||||
# Fill top-level fields if the detailed document exposes them only there.
|
||||
record_number = detail.get("documentNumber") or detail.get("naturalKey") or detail.get("number")
|
||||
if record_number and not enriched.get("document_number"):
|
||||
enriched["document_number"] = record_number
|
||||
total = (
|
||||
detail.get("payableAmount")
|
||||
or detail.get("totalAmount")
|
||||
or detail.get("grossAmount")
|
||||
or detail.get("amount")
|
||||
)
|
||||
if total and not enriched.get("amount"):
|
||||
enriched["amount"] = _decimal_or_none(total) or total
|
||||
return enriched
|
||||
|
||||
|
||||
def _summary_for_items(items: Iterable[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
result = []
|
||||
for item in items:
|
||||
totals = _jasmin_document_totals(item)
|
||||
lines = _jasmin_document_lines_from_item(item)
|
||||
result.append({
|
||||
"id": item.get("id"),
|
||||
"external_type": item.get("external_type"),
|
||||
"external_id": item.get("external_id"),
|
||||
"document_number": item.get("document_number"),
|
||||
"amount": item.get("amount"),
|
||||
"totals": totals,
|
||||
"lines": len(lines),
|
||||
"payload_keys": list((item.get("payload") or {}).keys()) if isinstance(item.get("payload"), dict) else [],
|
||||
})
|
||||
return result
|
||||
|
||||
|
||||
def _post_import_summary(opportunity_id: str) -> Dict[str, Any]:
|
||||
with engine.begin() as conn:
|
||||
opportunity = conn.execute(text("""
|
||||
SELECT id::text, title, value_amount, product_interest, metadata
|
||||
FROM opportunities
|
||||
WHERE id = CAST(:id AS UUID)
|
||||
"""), {"id": opportunity_id}).mappings().first()
|
||||
docs = conn.execute(text("""
|
||||
SELECT id::text, document_kind, document_number, amount, total_amount, currency, document_date
|
||||
FROM commercial_documents
|
||||
WHERE opportunity_id = CAST(:id AS UUID)
|
||||
ORDER BY created_at DESC
|
||||
"""), {"id": opportunity_id}).mappings().all()
|
||||
items = conn.execute(text("""
|
||||
SELECT product_name, quantity, unit_price, total_price, jasmin_sales_item
|
||||
FROM opportunity_items
|
||||
WHERE opportunity_id = CAST(:id AS UUID)
|
||||
ORDER BY created_at
|
||||
"""), {"id": opportunity_id}).mappings().all()
|
||||
lines = conn.execute(text("""
|
||||
SELECT cdl.description, cdl.quantity, cdl.unit_price, cdl.total_amount, cdl.jasmin_sales_item
|
||||
FROM commercial_document_lines cdl
|
||||
JOIN commercial_documents cd ON cd.id = cdl.document_id
|
||||
WHERE cd.opportunity_id = CAST(:id AS UUID)
|
||||
ORDER BY cdl.line_index
|
||||
"""), {"id": opportunity_id}).mappings().all()
|
||||
return {
|
||||
"opportunity": dict(opportunity or {}),
|
||||
"documents": [dict(r) for r in docs],
|
||||
"opportunity_items": [dict(r) for r in items],
|
||||
"document_lines": [dict(r) for r in lines],
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id")
|
||||
parser.add_argument("--document-number")
|
||||
parser.add_argument("--no-fetch-jasmin-detail", action="store_true")
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
parser.add_argument("--actor", default="operator_backfill")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.opportunity_id and not args.document_number:
|
||||
parser.error("usa --opportunity-id ou --document-number")
|
||||
|
||||
opportunity = _load_opportunity(args.opportunity_id, args.document_number)
|
||||
if not opportunity:
|
||||
print(json.dumps({"ok": False, "error": "opportunity_not_found"}, ensure_ascii=False, indent=2))
|
||||
return 2
|
||||
|
||||
items = _load_candidate_items(opportunity)
|
||||
enriched_items = [
|
||||
_with_jasmin_detail(item, fetch_detail=not args.no_fetch_jasmin_detail)
|
||||
for item in items
|
||||
]
|
||||
print(json.dumps({
|
||||
"opportunity_id": opportunity.get("id"),
|
||||
"title": opportunity.get("title"),
|
||||
"candidate_items": _summary_for_items(enriched_items),
|
||||
"dry_run": args.dry_run,
|
||||
}, ensure_ascii=False, indent=2, default=str))
|
||||
|
||||
if not enriched_items:
|
||||
print(json.dumps({"ok": False, "error": "no_jasmin_reconciliation_items_found"}, ensure_ascii=False, indent=2))
|
||||
return 3
|
||||
|
||||
if args.dry_run:
|
||||
return 0
|
||||
|
||||
with engine.begin() as conn:
|
||||
result = _apply_jasmin_documents_to_opportunity(
|
||||
conn,
|
||||
enriched_items,
|
||||
str(opportunity["id"]),
|
||||
actor=args.actor,
|
||||
)
|
||||
print(json.dumps({"ok": True, "import_result": result}, ensure_ascii=False, indent=2, default=str))
|
||||
print(json.dumps(_post_import_summary(str(opportunity["id"])), ensure_ascii=False, indent=2, default=str))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
89
scripts/backfill_opportunity_product_mappings.py
Executable file
89
scripts/backfill_opportunity_product_mappings.py
Executable file
@@ -0,0 +1,89 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Backfill Jasmin product mappings for Odoo-imported opportunity lines.
|
||||
|
||||
Maps Odoo line metadata product ids to the ClientFlow catalogue convention:
|
||||
``product_id=3`` -> ``products.sku='ODOO-3'`` -> ``jasmin_sales_item``.
|
||||
|
||||
Usage:
|
||||
PYTHONPATH=. python scripts/backfill_opportunity_product_mappings.py
|
||||
PYTHONPATH=. python scripts/backfill_opportunity_product_mappings.py --opportunity-id <uuid> --apply
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.product_service import ensure_product_schema
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id", default="", help="Optional opportunity UUID to restrict the backfill")
|
||||
parser.add_argument("--apply", action="store_true", help="Apply changes. Without this flag it runs as dry-run.")
|
||||
args = parser.parse_args()
|
||||
|
||||
ensure_product_schema()
|
||||
where_opp = "AND oi.opportunity_id = CAST(:opportunity_id AS UUID)" if args.opportunity_id else ""
|
||||
params = {"opportunity_id": args.opportunity_id} if args.opportunity_id else {}
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text(f"""
|
||||
SELECT
|
||||
oi.id::text AS item_id,
|
||||
oi.opportunity_id::text AS opportunity_id,
|
||||
oi.product_name,
|
||||
oi.sku AS old_sku,
|
||||
oi.jasmin_sales_item AS old_jasmin_sales_item,
|
||||
oi.metadata->>'product_id' AS odoo_product_id,
|
||||
p.id::text AS product_id,
|
||||
p.sku AS new_sku,
|
||||
p.jasmin_sales_item AS new_jasmin_sales_item,
|
||||
p.name AS catalog_name
|
||||
FROM opportunity_items oi
|
||||
JOIN products p
|
||||
ON p.sku = ('ODOO-' || (oi.metadata->>'product_id'))
|
||||
WHERE oi.metadata->>'source_system' = 'odoo'
|
||||
AND COALESCE(oi.metadata->>'product_id','') <> ''
|
||||
AND (
|
||||
oi.product_id IS DISTINCT FROM p.id
|
||||
OR COALESCE(oi.sku,'') IS DISTINCT FROM COALESCE(p.sku,'')
|
||||
OR COALESCE(oi.jasmin_sales_item,'') IS DISTINCT FROM COALESCE(p.jasmin_sales_item,'')
|
||||
)
|
||||
{where_opp}
|
||||
ORDER BY oi.created_at DESC
|
||||
"""), params).mappings().all()
|
||||
|
||||
print(f"Candidatos a atualizar: {len(rows)}")
|
||||
for row in rows:
|
||||
print(dict(row))
|
||||
|
||||
if args.apply and rows:
|
||||
result = conn.execute(text(f"""
|
||||
UPDATE opportunity_items oi
|
||||
SET product_id = p.id,
|
||||
sku = p.sku,
|
||||
jasmin_sales_item = p.jasmin_sales_item,
|
||||
metadata = COALESCE(oi.metadata, '{{}}'::jsonb) || jsonb_build_object(
|
||||
'resolved_sku', p.sku,
|
||||
'resolved_jasmin_sales_item', p.jasmin_sales_item,
|
||||
'product_mapping_status', CASE WHEN COALESCE(p.jasmin_sales_item,'') <> '' THEN 'mapped' ELSE 'missing_jasmin' END,
|
||||
'catalog_name', p.name,
|
||||
'backfilled_at', now()::text
|
||||
),
|
||||
updated_at = now()
|
||||
FROM products p
|
||||
WHERE p.sku = ('ODOO-' || (oi.metadata->>'product_id'))
|
||||
AND oi.metadata->>'source_system' = 'odoo'
|
||||
AND COALESCE(oi.metadata->>'product_id','') <> ''
|
||||
{where_opp}
|
||||
"""), params)
|
||||
print(f"Aplicado: {result.rowcount or 0} linha(s) atualizada(s)")
|
||||
elif not args.apply:
|
||||
print("Dry-run. Usa --apply para aplicar.")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
347
scripts/batch_validate_email_identity.py
Executable file
347
scripts/batch_validate_email_identity.py
Executable file
@@ -0,0 +1,347 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Batch validation for email identity extraction.
|
||||
|
||||
Safe for LLM runs: prints progress, truncates long bodies, applies a per-message
|
||||
process timeout, and writes JSONL incrementally so partial results are kept even
|
||||
if a provider call stalls.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import json
|
||||
import multiprocessing as mp
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.commercial_service import normalize_fiscal_name
|
||||
from app.db import engine
|
||||
|
||||
|
||||
LEGAL_SUFFIX_TOKENS = {
|
||||
"lda", "limitada", "unipessoal", "sa", "s", "a", "sociedade",
|
||||
"mediação", "mediacao", "seguro", "seguros", "importação", "importacao",
|
||||
"exportação", "exportacao", "fabricação", "fabricacao", "representação",
|
||||
"representacao", "soluções", "solucoes", "metálicas", "metalicas",
|
||||
}
|
||||
|
||||
|
||||
def clean(value: Any) -> str:
|
||||
return re.sub(r"\s+", " ", str(value or "")).strip()
|
||||
|
||||
|
||||
def norm(value: Any) -> str:
|
||||
return normalize_fiscal_name(value or "") or clean(value).casefold()
|
||||
|
||||
|
||||
def meaningful_tokens(value: Any) -> set[str]:
|
||||
n = norm(value)
|
||||
tokens = {t for t in re.split(r"[^a-z0-9áàâãéèêíìîóòôõúùûç]+", n) if len(t) >= 3}
|
||||
return {t for t in tokens if t not in LEGAL_SUFFIX_TOKENS}
|
||||
|
||||
|
||||
def mentions_match_fiscal(mentions: List[str], fiscal_name: str | None) -> bool:
|
||||
if not mentions or not fiscal_name:
|
||||
return False
|
||||
|
||||
nf = norm(fiscal_name)
|
||||
fiscal_tokens = meaningful_tokens(fiscal_name)
|
||||
|
||||
for mention in mentions:
|
||||
nm = norm(mention)
|
||||
if not nm:
|
||||
continue
|
||||
if nm == nf:
|
||||
return True
|
||||
if len(nm) >= 4 and (nm in nf or nf in nm):
|
||||
return True
|
||||
mention_tokens = meaningful_tokens(mention)
|
||||
if not mention_tokens or not fiscal_tokens:
|
||||
continue
|
||||
overlap = mention_tokens & fiscal_tokens
|
||||
if len(overlap) >= 2:
|
||||
return True
|
||||
if len(mention_tokens) <= 2 and overlap:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def classify(identity: Dict[str, Any], fiscal_customer: str | None) -> str:
|
||||
if identity.get("_error"):
|
||||
return "EXTRACTION_ERROR"
|
||||
if identity.get("_timeout"):
|
||||
return "EXTRACTION_TIMEOUT"
|
||||
|
||||
mentions = identity.get("company_mentions") or []
|
||||
domain = identity.get("domain") or ""
|
||||
person = identity.get("person_name") or ""
|
||||
|
||||
if mentions and fiscal_customer:
|
||||
if mentions_match_fiscal(mentions, fiscal_customer):
|
||||
return "OK_MENTION_COMPATIBLE_WITH_FISCAL"
|
||||
return "CONFLICT_MENTION_DIFFERS_FROM_FISCAL"
|
||||
if mentions and not fiscal_customer:
|
||||
return "OK_MENTION_AVAILABLE_NO_FISCAL"
|
||||
if not mentions and fiscal_customer:
|
||||
return "WEAK_NO_COMPANY_MENTION_HAS_FISCAL"
|
||||
if domain and person:
|
||||
return "WEAK_PERSON_AND_DOMAIN_ONLY"
|
||||
if domain:
|
||||
return "WEAK_DOMAIN_ONLY"
|
||||
return "NO_USEFUL_IDENTITY"
|
||||
|
||||
|
||||
def fetch_cases(limit: int, offset: int) -> List[Dict[str, Any]]:
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text("""
|
||||
WITH ranked AS (
|
||||
SELECT
|
||||
o.id::text AS opportunity_id,
|
||||
o.title AS opportunity_title,
|
||||
o.customer_name,
|
||||
o.customer_email,
|
||||
c.name AS fiscal_customer,
|
||||
c.tax_id AS fiscal_tax_id,
|
||||
c.email AS fiscal_email,
|
||||
t.id::text AS task_id,
|
||||
t.action_code,
|
||||
t.route,
|
||||
t.created_at AS task_created_at,
|
||||
m.id::text AS message_id,
|
||||
COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) AS body,
|
||||
COALESCE(
|
||||
NULLIF(m.metadata->>'subject', ''),
|
||||
NULLIF(re.payload->'conversation'->'additional_attributes'->>'mail_subject', ''),
|
||||
NULLIF(re.payload->'content_attributes'->'email'->>'subject', ''),
|
||||
NULLIF(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '')
|
||||
) AS subject,
|
||||
COALESCE(
|
||||
NULLIF(re.payload->'sender'->>'email', ''),
|
||||
NULLIF(re.payload->'conversation'->'meta'->'sender'->>'email', ''),
|
||||
NULLIF(re.payload->'conversation'->'contact_inbox'->>'source_id', ''),
|
||||
NULLIF(o.customer_email, '')
|
||||
) AS sender_email,
|
||||
ROW_NUMBER() OVER (
|
||||
PARTITION BY o.id
|
||||
ORDER BY t.created_at DESC
|
||||
) AS rn
|
||||
FROM opportunities o
|
||||
JOIN tasks t ON t.opportunity_id = o.id
|
||||
LEFT JOIN messages m ON m.id = t.message_id
|
||||
LEFT JOIN raw_events re ON re.id = t.raw_event_id
|
||||
LEFT JOIN customers c ON c.id = o.local_customer_id
|
||||
WHERE COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) IS NOT NULL
|
||||
)
|
||||
SELECT *
|
||||
FROM ranked
|
||||
WHERE rn = 1
|
||||
ORDER BY task_created_at DESC
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""), {"limit": limit, "offset": offset}).mappings().all()
|
||||
return [dict(r) for r in rows]
|
||||
|
||||
|
||||
def _domain_from_email(email: str) -> str:
|
||||
if "@" not in (email or ""):
|
||||
return ""
|
||||
return email.rsplit("@", 1)[1].lower().strip()
|
||||
|
||||
|
||||
def worker_extract(queue: mp.Queue, body: str, email: str, subject: str, use_llm: bool) -> None:
|
||||
try:
|
||||
from app.email_identity_extraction_service import extract_email_identity
|
||||
|
||||
identity = extract_email_identity(
|
||||
body or "",
|
||||
email=email or "",
|
||||
subject=subject or "",
|
||||
use_llm=use_llm,
|
||||
)
|
||||
queue.put({"ok": True, "identity": identity})
|
||||
except Exception as exc: # noqa: BLE001 validation tool should keep going
|
||||
queue.put({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
|
||||
|
||||
|
||||
def extract_with_timeout(body: str, email: str, subject: str, use_llm: bool, timeout_seconds: int) -> Dict[str, Any]:
|
||||
if not use_llm:
|
||||
from app.email_identity_extraction_service import extract_email_identity
|
||||
return extract_email_identity(body or "", email=email or "", subject=subject or "", use_llm=False)
|
||||
|
||||
queue: mp.Queue = mp.Queue()
|
||||
proc = mp.Process(target=worker_extract, args=(queue, body, email, subject, use_llm))
|
||||
proc.start()
|
||||
proc.join(timeout_seconds)
|
||||
|
||||
if proc.is_alive():
|
||||
proc.terminate()
|
||||
proc.join(5)
|
||||
return {
|
||||
"_timeout": True,
|
||||
"method": "timeout",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": [f"timeout após {timeout_seconds}s"],
|
||||
}
|
||||
|
||||
if queue.empty():
|
||||
return {
|
||||
"_error": True,
|
||||
"method": "error",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": ["processo terminou sem resultado"],
|
||||
}
|
||||
|
||||
result = queue.get()
|
||||
if result.get("ok"):
|
||||
return result["identity"]
|
||||
|
||||
return {
|
||||
"_error": True,
|
||||
"method": "error",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": [result.get("error")],
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--limit", type=int, default=20)
|
||||
parser.add_argument("--offset", type=int, default=0)
|
||||
parser.add_argument("--use-llm", action="store_true")
|
||||
parser.add_argument("--model", default="", help="Override EMAIL_IDENTITY_LLM_MODEL for this validation run")
|
||||
parser.add_argument("--fallback-model", default="", help="Override EMAIL_IDENTITY_LLM_FALLBACK_MODEL for this validation run")
|
||||
parser.add_argument("--timeout-seconds", type=int, default=30)
|
||||
parser.add_argument("--max-body-chars", type=int, default=3500)
|
||||
parser.add_argument("--out-dir", default="reports")
|
||||
args = parser.parse_args()
|
||||
if args.model:
|
||||
os.environ["EMAIL_IDENTITY_LLM_MODEL"] = args.model
|
||||
if args.fallback_model:
|
||||
os.environ["EMAIL_IDENTITY_LLM_FALLBACK_MODEL"] = args.fallback_model
|
||||
|
||||
cases = fetch_cases(args.limit, args.offset)
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
mode = "llm" if args.use_llm else "regex"
|
||||
jsonl_path = out_dir / f"email_identity_validation_{mode}_{stamp}.jsonl"
|
||||
csv_path = out_dir / f"email_identity_validation_{mode}_{stamp}.csv"
|
||||
|
||||
results: List[Dict[str, Any]] = []
|
||||
counts: Dict[str, int] = {}
|
||||
|
||||
model_label = os.getenv("EMAIL_IDENTITY_LLM_MODEL") or os.getenv("OPENROUTER_MODEL", "")
|
||||
fallback_label = os.getenv("EMAIL_IDENTITY_LLM_FALLBACK_MODEL", "")
|
||||
print(
|
||||
f"Casos: {len(cases)} | mode={mode} | offset={args.offset} | "
|
||||
f"timeout={args.timeout_seconds}s | max_body_chars={args.max_body_chars} | "
|
||||
f"model={model_label if args.use_llm else '-'} | fallback={fallback_label if args.use_llm and fallback_label else '-'}",
|
||||
flush=True,
|
||||
)
|
||||
print(f"JSONL incremental: {jsonl_path}", flush=True)
|
||||
|
||||
with jsonl_path.open("w", encoding="utf-8") as jf:
|
||||
for idx, row in enumerate(cases, start=1):
|
||||
email = row.get("sender_email") or row.get("customer_email") or ""
|
||||
body = (row.get("body") or "")[: args.max_body_chars]
|
||||
subject = row.get("subject") or ""
|
||||
|
||||
print(f"\n[{idx}/{len(cases)}] {row.get('opportunity_title')} | {email}", flush=True)
|
||||
|
||||
identity = extract_with_timeout(
|
||||
body=body,
|
||||
email=email,
|
||||
subject=subject,
|
||||
use_llm=args.use_llm,
|
||||
timeout_seconds=args.timeout_seconds,
|
||||
)
|
||||
status = classify(identity, row.get("fiscal_customer"))
|
||||
|
||||
result = {
|
||||
"status": status,
|
||||
"opportunity_id": row.get("opportunity_id"),
|
||||
"opportunity_title": row.get("opportunity_title"),
|
||||
"task_id": row.get("task_id"),
|
||||
"action_code": row.get("action_code"),
|
||||
"sender_email": email,
|
||||
"customer_name": row.get("customer_name"),
|
||||
"customer_email": row.get("customer_email"),
|
||||
"fiscal_customer": row.get("fiscal_customer"),
|
||||
"fiscal_tax_id": row.get("fiscal_tax_id"),
|
||||
"fiscal_email": row.get("fiscal_email"),
|
||||
"person_name": identity.get("person_name"),
|
||||
"company_mentions": identity.get("company_mentions") or [],
|
||||
"domain": identity.get("domain"),
|
||||
"address": identity.get("address"),
|
||||
"phones": identity.get("phones") or [],
|
||||
"confidence": identity.get("confidence"),
|
||||
"method": identity.get("method"),
|
||||
"llm_model": identity.get("llm_model"),
|
||||
"fallback_used": identity.get("fallback_used"),
|
||||
"evidence": identity.get("evidence") or [],
|
||||
}
|
||||
results.append(result)
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
jf.write(json.dumps(result, ensure_ascii=False, default=str) + "\n")
|
||||
jf.flush()
|
||||
|
||||
print(
|
||||
"status:", status,
|
||||
"| person:", result["person_name"],
|
||||
"| companies:", result["company_mentions"],
|
||||
flush=True,
|
||||
)
|
||||
|
||||
csv_fields = [
|
||||
"status", "opportunity_id", "opportunity_title", "task_id", "action_code",
|
||||
"sender_email", "customer_name", "customer_email", "fiscal_customer",
|
||||
"fiscal_tax_id", "fiscal_email", "person_name", "company_mentions",
|
||||
"domain", "address", "phones", "confidence", "method", "llm_model", "fallback_used", "evidence",
|
||||
]
|
||||
with csv_path.open("w", encoding="utf-8", newline="") as f:
|
||||
writer = csv.DictWriter(f, fieldnames=csv_fields)
|
||||
writer.writeheader()
|
||||
for r in results:
|
||||
csv_row = dict(r)
|
||||
for key in ["company_mentions", "phones", "evidence"]:
|
||||
csv_row[key] = " | ".join(str(x) for x in csv_row.get(key) or [])
|
||||
writer.writerow(csv_row)
|
||||
|
||||
print("\nResumo:")
|
||||
for status, count in sorted(counts.items()):
|
||||
print(f"{status}: {count}")
|
||||
|
||||
print("\nFicheiros gerados:")
|
||||
print(jsonl_path)
|
||||
print(csv_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
347
scripts/batch_validate_email_identity_safe.py
Executable file
347
scripts/batch_validate_email_identity_safe.py
Executable file
@@ -0,0 +1,347 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Batch validation for email identity extraction.
|
||||
|
||||
Safe for LLM runs: prints progress, truncates long bodies, applies a per-message
|
||||
process timeout, and writes JSONL incrementally so partial results are kept even
|
||||
if a provider call stalls.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import json
|
||||
import multiprocessing as mp
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.commercial_service import normalize_fiscal_name
|
||||
from app.db import engine
|
||||
|
||||
|
||||
LEGAL_SUFFIX_TOKENS = {
|
||||
"lda", "limitada", "unipessoal", "sa", "s", "a", "sociedade",
|
||||
"mediação", "mediacao", "seguro", "seguros", "importação", "importacao",
|
||||
"exportação", "exportacao", "fabricação", "fabricacao", "representação",
|
||||
"representacao", "soluções", "solucoes", "metálicas", "metalicas",
|
||||
}
|
||||
|
||||
|
||||
def clean(value: Any) -> str:
|
||||
return re.sub(r"\s+", " ", str(value or "")).strip()
|
||||
|
||||
|
||||
def norm(value: Any) -> str:
|
||||
return normalize_fiscal_name(value or "") or clean(value).casefold()
|
||||
|
||||
|
||||
def meaningful_tokens(value: Any) -> set[str]:
|
||||
n = norm(value)
|
||||
tokens = {t for t in re.split(r"[^a-z0-9áàâãéèêíìîóòôõúùûç]+", n) if len(t) >= 3}
|
||||
return {t for t in tokens if t not in LEGAL_SUFFIX_TOKENS}
|
||||
|
||||
|
||||
def mentions_match_fiscal(mentions: List[str], fiscal_name: str | None) -> bool:
|
||||
if not mentions or not fiscal_name:
|
||||
return False
|
||||
|
||||
nf = norm(fiscal_name)
|
||||
fiscal_tokens = meaningful_tokens(fiscal_name)
|
||||
|
||||
for mention in mentions:
|
||||
nm = norm(mention)
|
||||
if not nm:
|
||||
continue
|
||||
if nm == nf:
|
||||
return True
|
||||
if len(nm) >= 4 and (nm in nf or nf in nm):
|
||||
return True
|
||||
mention_tokens = meaningful_tokens(mention)
|
||||
if not mention_tokens or not fiscal_tokens:
|
||||
continue
|
||||
overlap = mention_tokens & fiscal_tokens
|
||||
if len(overlap) >= 2:
|
||||
return True
|
||||
if len(mention_tokens) <= 2 and overlap:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def classify(identity: Dict[str, Any], fiscal_customer: str | None) -> str:
|
||||
if identity.get("_error"):
|
||||
return "EXTRACTION_ERROR"
|
||||
if identity.get("_timeout"):
|
||||
return "EXTRACTION_TIMEOUT"
|
||||
|
||||
mentions = identity.get("company_mentions") or []
|
||||
domain = identity.get("domain") or ""
|
||||
person = identity.get("person_name") or ""
|
||||
|
||||
if mentions and fiscal_customer:
|
||||
if mentions_match_fiscal(mentions, fiscal_customer):
|
||||
return "OK_MENTION_COMPATIBLE_WITH_FISCAL"
|
||||
return "CONFLICT_MENTION_DIFFERS_FROM_FISCAL"
|
||||
if mentions and not fiscal_customer:
|
||||
return "OK_MENTION_AVAILABLE_NO_FISCAL"
|
||||
if not mentions and fiscal_customer:
|
||||
return "WEAK_NO_COMPANY_MENTION_HAS_FISCAL"
|
||||
if domain and person:
|
||||
return "WEAK_PERSON_AND_DOMAIN_ONLY"
|
||||
if domain:
|
||||
return "WEAK_DOMAIN_ONLY"
|
||||
return "NO_USEFUL_IDENTITY"
|
||||
|
||||
|
||||
def fetch_cases(limit: int, offset: int) -> List[Dict[str, Any]]:
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text("""
|
||||
WITH ranked AS (
|
||||
SELECT
|
||||
o.id::text AS opportunity_id,
|
||||
o.title AS opportunity_title,
|
||||
o.customer_name,
|
||||
o.customer_email,
|
||||
c.name AS fiscal_customer,
|
||||
c.tax_id AS fiscal_tax_id,
|
||||
c.email AS fiscal_email,
|
||||
t.id::text AS task_id,
|
||||
t.action_code,
|
||||
t.route,
|
||||
t.created_at AS task_created_at,
|
||||
m.id::text AS message_id,
|
||||
COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) AS body,
|
||||
COALESCE(
|
||||
NULLIF(m.metadata->>'subject', ''),
|
||||
NULLIF(re.payload->'conversation'->'additional_attributes'->>'mail_subject', ''),
|
||||
NULLIF(re.payload->'content_attributes'->'email'->>'subject', ''),
|
||||
NULLIF(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '')
|
||||
) AS subject,
|
||||
COALESCE(
|
||||
NULLIF(re.payload->'sender'->>'email', ''),
|
||||
NULLIF(re.payload->'conversation'->'meta'->'sender'->>'email', ''),
|
||||
NULLIF(re.payload->'conversation'->'contact_inbox'->>'source_id', ''),
|
||||
NULLIF(o.customer_email, '')
|
||||
) AS sender_email,
|
||||
ROW_NUMBER() OVER (
|
||||
PARTITION BY o.id
|
||||
ORDER BY t.created_at DESC
|
||||
) AS rn
|
||||
FROM opportunities o
|
||||
JOIN tasks t ON t.opportunity_id = o.id
|
||||
LEFT JOIN messages m ON m.id = t.message_id
|
||||
LEFT JOIN raw_events re ON re.id = t.raw_event_id
|
||||
LEFT JOIN customers c ON c.id = o.local_customer_id
|
||||
WHERE COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) IS NOT NULL
|
||||
)
|
||||
SELECT *
|
||||
FROM ranked
|
||||
WHERE rn = 1
|
||||
ORDER BY task_created_at DESC
|
||||
LIMIT :limit
|
||||
OFFSET :offset
|
||||
"""), {"limit": limit, "offset": offset}).mappings().all()
|
||||
return [dict(r) for r in rows]
|
||||
|
||||
|
||||
def _domain_from_email(email: str) -> str:
|
||||
if "@" not in (email or ""):
|
||||
return ""
|
||||
return email.rsplit("@", 1)[1].lower().strip()
|
||||
|
||||
|
||||
def worker_extract(queue: mp.Queue, body: str, email: str, subject: str, use_llm: bool) -> None:
|
||||
try:
|
||||
from app.email_identity_extraction_service import extract_email_identity
|
||||
|
||||
identity = extract_email_identity(
|
||||
body or "",
|
||||
email=email or "",
|
||||
subject=subject or "",
|
||||
use_llm=use_llm,
|
||||
)
|
||||
queue.put({"ok": True, "identity": identity})
|
||||
except Exception as exc: # noqa: BLE001 validation tool should keep going
|
||||
queue.put({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
|
||||
|
||||
|
||||
def extract_with_timeout(body: str, email: str, subject: str, use_llm: bool, timeout_seconds: int) -> Dict[str, Any]:
|
||||
if not use_llm:
|
||||
from app.email_identity_extraction_service import extract_email_identity
|
||||
return extract_email_identity(body or "", email=email or "", subject=subject or "", use_llm=False)
|
||||
|
||||
queue: mp.Queue = mp.Queue()
|
||||
proc = mp.Process(target=worker_extract, args=(queue, body, email, subject, use_llm))
|
||||
proc.start()
|
||||
proc.join(timeout_seconds)
|
||||
|
||||
if proc.is_alive():
|
||||
proc.terminate()
|
||||
proc.join(5)
|
||||
return {
|
||||
"_timeout": True,
|
||||
"method": "timeout",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": [f"timeout após {timeout_seconds}s"],
|
||||
}
|
||||
|
||||
if queue.empty():
|
||||
return {
|
||||
"_error": True,
|
||||
"method": "error",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": ["processo terminou sem resultado"],
|
||||
}
|
||||
|
||||
result = queue.get()
|
||||
if result.get("ok"):
|
||||
return result["identity"]
|
||||
|
||||
return {
|
||||
"_error": True,
|
||||
"method": "error",
|
||||
"confidence": 0,
|
||||
"email": email,
|
||||
"domain": _domain_from_email(email),
|
||||
"person_name": "",
|
||||
"company_mentions": [],
|
||||
"address": "",
|
||||
"phones": [],
|
||||
"websites": [],
|
||||
"evidence": [result.get("error")],
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--limit", type=int, default=20)
|
||||
parser.add_argument("--offset", type=int, default=0)
|
||||
parser.add_argument("--use-llm", action="store_true")
|
||||
parser.add_argument("--model", default="", help="Override EMAIL_IDENTITY_LLM_MODEL for this validation run")
|
||||
parser.add_argument("--fallback-model", default="", help="Override EMAIL_IDENTITY_LLM_FALLBACK_MODEL for this validation run")
|
||||
parser.add_argument("--timeout-seconds", type=int, default=30)
|
||||
parser.add_argument("--max-body-chars", type=int, default=3500)
|
||||
parser.add_argument("--out-dir", default="reports")
|
||||
args = parser.parse_args()
|
||||
if args.model:
|
||||
os.environ["EMAIL_IDENTITY_LLM_MODEL"] = args.model
|
||||
if args.fallback_model:
|
||||
os.environ["EMAIL_IDENTITY_LLM_FALLBACK_MODEL"] = args.fallback_model
|
||||
|
||||
cases = fetch_cases(args.limit, args.offset)
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||||
mode = "llm" if args.use_llm else "regex"
|
||||
jsonl_path = out_dir / f"email_identity_validation_{mode}_{stamp}.jsonl"
|
||||
csv_path = out_dir / f"email_identity_validation_{mode}_{stamp}.csv"
|
||||
|
||||
results: List[Dict[str, Any]] = []
|
||||
counts: Dict[str, int] = {}
|
||||
|
||||
model_label = os.getenv("EMAIL_IDENTITY_LLM_MODEL") or os.getenv("OPENROUTER_MODEL", "")
|
||||
fallback_label = os.getenv("EMAIL_IDENTITY_LLM_FALLBACK_MODEL", "")
|
||||
print(
|
||||
f"Casos: {len(cases)} | mode={mode} | offset={args.offset} | "
|
||||
f"timeout={args.timeout_seconds}s | max_body_chars={args.max_body_chars} | "
|
||||
f"model={model_label if args.use_llm else '-'} | fallback={fallback_label if args.use_llm and fallback_label else '-'}",
|
||||
flush=True,
|
||||
)
|
||||
print(f"JSONL incremental: {jsonl_path}", flush=True)
|
||||
|
||||
with jsonl_path.open("w", encoding="utf-8") as jf:
|
||||
for idx, row in enumerate(cases, start=1):
|
||||
email = row.get("sender_email") or row.get("customer_email") or ""
|
||||
body = (row.get("body") or "")[: args.max_body_chars]
|
||||
subject = row.get("subject") or ""
|
||||
|
||||
print(f"\n[{idx}/{len(cases)}] {row.get('opportunity_title')} | {email}", flush=True)
|
||||
|
||||
identity = extract_with_timeout(
|
||||
body=body,
|
||||
email=email,
|
||||
subject=subject,
|
||||
use_llm=args.use_llm,
|
||||
timeout_seconds=args.timeout_seconds,
|
||||
)
|
||||
status = classify(identity, row.get("fiscal_customer"))
|
||||
|
||||
result = {
|
||||
"status": status,
|
||||
"opportunity_id": row.get("opportunity_id"),
|
||||
"opportunity_title": row.get("opportunity_title"),
|
||||
"task_id": row.get("task_id"),
|
||||
"action_code": row.get("action_code"),
|
||||
"sender_email": email,
|
||||
"customer_name": row.get("customer_name"),
|
||||
"customer_email": row.get("customer_email"),
|
||||
"fiscal_customer": row.get("fiscal_customer"),
|
||||
"fiscal_tax_id": row.get("fiscal_tax_id"),
|
||||
"fiscal_email": row.get("fiscal_email"),
|
||||
"person_name": identity.get("person_name"),
|
||||
"company_mentions": identity.get("company_mentions") or [],
|
||||
"domain": identity.get("domain"),
|
||||
"address": identity.get("address"),
|
||||
"phones": identity.get("phones") or [],
|
||||
"confidence": identity.get("confidence"),
|
||||
"method": identity.get("method"),
|
||||
"llm_model": identity.get("llm_model"),
|
||||
"fallback_used": identity.get("fallback_used"),
|
||||
"evidence": identity.get("evidence") or [],
|
||||
}
|
||||
results.append(result)
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
jf.write(json.dumps(result, ensure_ascii=False, default=str) + "\n")
|
||||
jf.flush()
|
||||
|
||||
print(
|
||||
"status:", status,
|
||||
"| person:", result["person_name"],
|
||||
"| companies:", result["company_mentions"],
|
||||
flush=True,
|
||||
)
|
||||
|
||||
csv_fields = [
|
||||
"status", "opportunity_id", "opportunity_title", "task_id", "action_code",
|
||||
"sender_email", "customer_name", "customer_email", "fiscal_customer",
|
||||
"fiscal_tax_id", "fiscal_email", "person_name", "company_mentions",
|
||||
"domain", "address", "phones", "confidence", "method", "llm_model", "fallback_used", "evidence",
|
||||
]
|
||||
with csv_path.open("w", encoding="utf-8", newline="") as f:
|
||||
writer = csv.DictWriter(f, fieldnames=csv_fields)
|
||||
writer.writeheader()
|
||||
for r in results:
|
||||
csv_row = dict(r)
|
||||
for key in ["company_mentions", "phones", "evidence"]:
|
||||
csv_row[key] = " | ".join(str(x) for x in csv_row.get(key) or [])
|
||||
writer.writerow(csv_row)
|
||||
|
||||
print("\nResumo:")
|
||||
for status, count in sorted(counts.items()):
|
||||
print(f"{status}: {count}")
|
||||
|
||||
print("\nFicheiros gerados:")
|
||||
print(jsonl_path)
|
||||
print(csv_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
82
scripts/check_clientflow_health.py
Executable file
82
scripts/check_clientflow_health.py
Executable file
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Operational health check for ClientFlow deployments."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.config import settings
|
||||
from app.db import engine
|
||||
|
||||
|
||||
def scalar(sql: str, **params):
|
||||
with engine.begin() as conn:
|
||||
return conn.execute(text(sql), params).scalar()
|
||||
|
||||
|
||||
def rows(sql: str, **params):
|
||||
with engine.begin() as conn:
|
||||
return [dict(r) for r in conn.execute(text(sql), params).mappings().all()]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
result = {
|
||||
"app": settings.app_name,
|
||||
"env": settings.env,
|
||||
"database_ok": False,
|
||||
"jasmin_enabled": bool(settings.jasmin_enabled),
|
||||
"packlink_enabled": bool(settings.packlink_enabled),
|
||||
"outbox": {},
|
||||
"documents": {},
|
||||
"warnings": [],
|
||||
}
|
||||
|
||||
try:
|
||||
result["database_ok"] = bool(scalar("SELECT 1"))
|
||||
except Exception as exc:
|
||||
result["warnings"].append(f"database_error: {exc}")
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
|
||||
return 2
|
||||
|
||||
for row in rows("""
|
||||
SELECT target_system, status, count(*)::int AS total
|
||||
FROM integration_outbox
|
||||
GROUP BY target_system, status
|
||||
ORDER BY target_system, status
|
||||
"""):
|
||||
result["outbox"].setdefault(row["target_system"], {})[row["status"]] = row["total"]
|
||||
|
||||
for row in rows("""
|
||||
SELECT document_kind, status, count(*)::int AS total
|
||||
FROM commercial_documents
|
||||
GROUP BY document_kind, status
|
||||
ORDER BY document_kind, status
|
||||
"""):
|
||||
result["documents"].setdefault(row["document_kind"], {})[row["status"]] = row["total"]
|
||||
|
||||
missing_jasmin = scalar("""
|
||||
SELECT count(*)
|
||||
FROM products
|
||||
WHERE active = TRUE
|
||||
AND (jasmin_sales_item IS NULL OR jasmin_sales_item = '')
|
||||
""")
|
||||
if missing_jasmin:
|
||||
result["warnings"].append(f"active_products_without_jasmin_sales_item={missing_jasmin}")
|
||||
|
||||
failed_outbox = scalar("SELECT count(*) FROM integration_outbox WHERE status = 'failed'")
|
||||
if failed_outbox:
|
||||
result["warnings"].append(f"failed_outbox_items={failed_outbox}")
|
||||
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
|
||||
return 1 if result["warnings"] else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
131
scripts/cleanup_invalid_email_identity_suggestions.py
Executable file
131
scripts/cleanup_invalid_email_identity_suggestions.py
Executable file
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Clean stale invalid email-identity suggestions/extractions.
|
||||
|
||||
Use after strengthening company-mention filters. It targets suggestions and
|
||||
stored extractions created from invalid fragments such as ``pt`` or ``com``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.email_identity_extraction_service import is_plausible_company_mention
|
||||
|
||||
INVALID = {"pt", "com", "net", "org", "www", "http", "https", "mail", "email"}
|
||||
|
||||
|
||||
def _clean(value: Any) -> str:
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def _valid_companies(values: Any) -> list[str]:
|
||||
out: list[str] = []
|
||||
if not isinstance(values, list):
|
||||
return out
|
||||
for value in values:
|
||||
v = _clean(value).strip(" ,.;:-")
|
||||
if v and is_plausible_company_mention(v):
|
||||
out.append(v)
|
||||
return out
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id", default="")
|
||||
parser.add_argument("--apply", action="store_true")
|
||||
parser.add_argument("--include-accepted", action="store_true")
|
||||
parser.add_argument("--fix-extractions", action="store_true", help="Also rewrite stored email_identity_extractions company_mentions after current filters")
|
||||
args = parser.parse_args()
|
||||
|
||||
where = ["lookup_type LIKE 'email_identity%'"]
|
||||
params: dict[str, Any] = {}
|
||||
if args.opportunity_id:
|
||||
where.append("opportunity_id = CAST(:opportunity_id AS UUID)")
|
||||
params["opportunity_id"] = args.opportunity_id
|
||||
if args.include_accepted:
|
||||
where.append("status IN ('pending', 'accepted', 'rejected')")
|
||||
else:
|
||||
where.append("status = 'pending'")
|
||||
invalid_sql = ", ".join("'" + v.replace("'", "") + "'" for v in sorted(INVALID))
|
||||
where.append(f"lower(trim(COALESCE(lookup_value, ''))) IN ({invalid_sql})")
|
||||
|
||||
sql_where = " AND ".join(where)
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text(f"""
|
||||
SELECT id::text, opportunity_id::text, suggested_name, suggested_nif,
|
||||
lookup_type, lookup_value, confidence, status, reason, created_at
|
||||
FROM fiscal_customer_suggestions
|
||||
WHERE {sql_where}
|
||||
ORDER BY created_at DESC
|
||||
"""), params).mappings().all()
|
||||
|
||||
print(json.dumps({"invalid_suggestions": len(rows), "apply": args.apply}, ensure_ascii=False, indent=2, default=str))
|
||||
for row in rows:
|
||||
print(json.dumps(dict(row), ensure_ascii=False, indent=2, default=str))
|
||||
|
||||
if args.apply and rows:
|
||||
ids = [r["id"] for r in rows]
|
||||
conn.execute(text("""
|
||||
UPDATE fiscal_customer_suggestions
|
||||
SET status = 'rejected',
|
||||
reason = COALESCE(reason, '') || ' | rejected_invalid_email_identity_token',
|
||||
resolved_by = 'cleanup_invalid_email_identity_suggestions',
|
||||
resolved_at = now(),
|
||||
updated_at = now()
|
||||
WHERE id = ANY(CAST(:ids AS UUID[]))
|
||||
"""), {"ids": ids})
|
||||
print(json.dumps({"rejected": len(ids)}, ensure_ascii=False, indent=2))
|
||||
|
||||
if args.fix_extractions:
|
||||
extraction_where = []
|
||||
extraction_params: dict[str, Any] = {}
|
||||
if args.opportunity_id:
|
||||
extraction_where.append("opportunity_id = CAST(:opportunity_id AS UUID)")
|
||||
extraction_params["opportunity_id"] = args.opportunity_id
|
||||
extraction_sql = "WHERE " + " AND ".join(extraction_where) if extraction_where else ""
|
||||
ex_rows = conn.execute(text(f"""
|
||||
SELECT id::text, opportunity_id::text, company_mentions, confidence, raw_payload
|
||||
FROM email_identity_extractions
|
||||
{extraction_sql}
|
||||
ORDER BY updated_at DESC
|
||||
"""), extraction_params).mappings().all()
|
||||
changed = []
|
||||
for row in ex_rows:
|
||||
original = row.get("company_mentions") or []
|
||||
valid = _valid_companies(original)
|
||||
if list(original or []) == valid:
|
||||
continue
|
||||
changed.append({"id": row["id"], "opportunity_id": row["opportunity_id"], "before": original, "after": valid})
|
||||
if args.apply:
|
||||
try:
|
||||
confidence = float(row.get("confidence") or 0)
|
||||
except Exception:
|
||||
confidence = 0.0
|
||||
if not valid:
|
||||
confidence = min(confidence, 0.45)
|
||||
raw_payload = dict(row.get("raw_payload") or {}) if isinstance(row.get("raw_payload"), dict) else {}
|
||||
raw_payload["filtered_invalid_company_mentions"] = list(original or [])
|
||||
conn.execute(text("""
|
||||
UPDATE email_identity_extractions
|
||||
SET company_mentions = CAST(:company_mentions AS JSONB),
|
||||
confidence = :confidence,
|
||||
raw_payload = COALESCE(raw_payload, '{}'::jsonb) || CAST(:raw_payload AS JSONB),
|
||||
updated_at = now()
|
||||
WHERE id = CAST(:id AS UUID)
|
||||
"""), {
|
||||
"id": row["id"],
|
||||
"company_mentions": json.dumps(valid, ensure_ascii=False),
|
||||
"confidence": confidence,
|
||||
"raw_payload": json.dumps(raw_payload, ensure_ascii=False, default=str),
|
||||
})
|
||||
print(json.dumps({"extractions_to_fix": len(changed), "items": changed}, ensure_ascii=False, indent=2, default=str))
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
105
scripts/cleanup_non_commercial_opportunities.py
Executable file
105
scripts/cleanup_non_commercial_opportunities.py
Executable file
@@ -0,0 +1,105 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Identify or close opportunities that were probably created from system messages.
|
||||
|
||||
Conservative by default: use --dry-run to list candidates. Use --apply to mark
|
||||
safe candidates as LOST with a metadata reason. It only targets opportunities
|
||||
with value 0, no commercial documents and no shipments.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import uuid
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.opportunity_service import ensure_opportunity_schema
|
||||
|
||||
SYSTEM_TERMS = (
|
||||
"mail delivery subsystem",
|
||||
"mailer-daemon",
|
||||
"postmaster",
|
||||
"returned mail",
|
||||
"undelivered mail",
|
||||
"delivery status notification",
|
||||
"failure notice",
|
||||
"mail delivery failed",
|
||||
)
|
||||
|
||||
|
||||
def _json(value: object) -> str:
|
||||
return json.dumps(value or {}, ensure_ascii=False, default=str)
|
||||
|
||||
|
||||
def find_candidates(limit: int = 100) -> list[dict]:
|
||||
ensure_opportunity_schema()
|
||||
like_sql = " OR ".join(
|
||||
["lower(coalesce(o.title,'') || ' ' || coalesce(o.customer_name,'') || ' ' || coalesce(o.customer_email,'') || ' ' || coalesce(o.metadata::text,'')) LIKE :term_{}".format(i) for i, _ in enumerate(SYSTEM_TERMS)]
|
||||
)
|
||||
params = {f"term_{i}": f"%{term}%" for i, term in enumerate(SYSTEM_TERMS)}
|
||||
params["limit"] = int(limit)
|
||||
sql = text(f"""
|
||||
SELECT o.id::text, o.title, o.customer_name, o.customer_email, o.value_amount, o.stage, o.status, o.created_at
|
||||
FROM opportunities o
|
||||
WHERE o.status = 'open'
|
||||
AND COALESCE(o.value_amount, 0) = 0
|
||||
AND NOT EXISTS (SELECT 1 FROM commercial_documents cd WHERE cd.opportunity_id = o.id)
|
||||
AND NOT EXISTS (SELECT 1 FROM shipments s WHERE s.opportunity_id = o.id)
|
||||
AND ({like_sql})
|
||||
ORDER BY o.created_at DESC
|
||||
LIMIT :limit
|
||||
""")
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(sql, params).mappings().all()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
|
||||
def close_candidates(candidates: list[dict]) -> int:
|
||||
if not candidates:
|
||||
return 0
|
||||
ids = [row["id"] for row in candidates]
|
||||
with engine.begin() as conn:
|
||||
for opportunity_id in ids:
|
||||
conn.execute(text("""
|
||||
UPDATE opportunities
|
||||
SET status = 'closed',
|
||||
stage = 'LOST',
|
||||
closed_at = COALESCE(closed_at, now()),
|
||||
updated_at = now(),
|
||||
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:metadata AS JSONB)
|
||||
WHERE id = CAST(:opportunity_id AS UUID)
|
||||
"""), {
|
||||
"opportunity_id": opportunity_id,
|
||||
"metadata": _json({"closed_reason": "system_or_bounce_created_by_mistake", "closed_by": "cleanup_non_commercial_opportunities"}),
|
||||
})
|
||||
conn.execute(text("""
|
||||
INSERT INTO opportunity_events (id, opportunity_id, event_type, from_stage, to_stage, note, payload, created_by)
|
||||
VALUES (CAST(:id AS UUID), CAST(:opportunity_id AS UUID), 'cleanup_closed_non_commercial', NULL, 'LOST', :note, CAST(:payload AS JSONB), 'system')
|
||||
"""), {
|
||||
"id": str(uuid.uuid4()),
|
||||
"opportunity_id": opportunity_id,
|
||||
"note": "Oportunidade fechada por parecer mensagem automática/bounce sem atividade comercial.",
|
||||
"payload": _json({"reason": "system_or_bounce_created_by_mistake"}),
|
||||
})
|
||||
return len(ids)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--limit", type=int, default=100)
|
||||
parser.add_argument("--apply", action="store_true", help="Apply changes. Without this, only prints candidates.")
|
||||
args = parser.parse_args()
|
||||
candidates = find_candidates(limit=args.limit)
|
||||
print(f"Found {len(candidates)} candidate(s).")
|
||||
for row in candidates:
|
||||
print(f"- {row.get('id')} | {row.get('title')} | {row.get('customer_name')} | {row.get('created_at')}")
|
||||
if args.apply:
|
||||
total = close_candidates(candidates)
|
||||
print(f"Closed {total} candidate(s).")
|
||||
else:
|
||||
print("Dry-run only. Re-run with --apply to close candidates.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
158
scripts/cleanup_operations_noise.py
Executable file
158
scripts/cleanup_operations_noise.py
Executable file
@@ -0,0 +1,158 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Clean noisy pending Operations items created from mailbox/system messages.
|
||||
|
||||
Dry-run by default. With --apply, marks obvious bounce/NDR/system tasks as
|
||||
``skipped`` and stores a cleanup marker in task metadata. This does not delete
|
||||
messages, raw events, or Chatwoot conversations.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
SQL_TERMS = [
|
||||
"postmaster",
|
||||
"mailer-daemon",
|
||||
"mail delivery subsystem",
|
||||
"mail delivery system",
|
||||
"microsoft exchange",
|
||||
"office 365",
|
||||
"undeliverable",
|
||||
"returned mail",
|
||||
"delivery status notification",
|
||||
"non-delivery report",
|
||||
"non delivery report",
|
||||
"your message couldn't be delivered",
|
||||
"your message couldnt be delivered",
|
||||
"recipient wasn't found",
|
||||
"recipient was not found",
|
||||
"unknown to address",
|
||||
"delivery has failed",
|
||||
"mail delivery failed",
|
||||
"remote server returned",
|
||||
"550 5.1.1",
|
||||
"5.1.10",
|
||||
"wasn't found at",
|
||||
]
|
||||
|
||||
NOISE_ACTION_CODES = [
|
||||
"IGNORE_BOUNCE",
|
||||
"IGNORE_SPAM",
|
||||
"NO_ACTION",
|
||||
]
|
||||
|
||||
|
||||
def _json(value: object) -> str:
|
||||
return json.dumps(value or {}, ensure_ascii=False, default=str)
|
||||
|
||||
|
||||
def find_candidates(limit: int = 200) -> list[dict]:
|
||||
term_params = {f"term_{idx}": f"%{term}%" for idx, term in enumerate(SQL_TERMS)}
|
||||
term_sql = " OR ".join([f"noise_text ILIKE :term_{idx}" for idx, _ in enumerate(SQL_TERMS)])
|
||||
sql = text(f"""
|
||||
WITH task_context AS (
|
||||
SELECT
|
||||
t.id::text,
|
||||
t.created_at,
|
||||
t.action_code,
|
||||
t.route,
|
||||
t.priority,
|
||||
t.status,
|
||||
t.conversation_id,
|
||||
t.contact_id,
|
||||
t.opportunity_id::text,
|
||||
COALESCE(t.note, '') AS note,
|
||||
COALESCE(t.action, '') AS action,
|
||||
COALESCE(re.payload->'sender'->>'name', '') AS sender_name,
|
||||
COALESCE(re.payload->'sender'->>'email', '') AS sender_email,
|
||||
COALESCE(
|
||||
re.payload->'conversation'->'additional_attributes'->>'mail_subject',
|
||||
re.payload->'content_attributes'->'email'->>'subject',
|
||||
re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject',
|
||||
''
|
||||
) AS subject,
|
||||
COALESCE(m.clean_body, m.raw_body, re.payload->>'content', '') AS body,
|
||||
lower(
|
||||
COALESCE(t.action_code, '') || ' ' || COALESCE(t.route, '') || ' ' ||
|
||||
COALESCE(t.action, '') || ' ' || COALESCE(t.note, '') || ' ' ||
|
||||
COALESCE(re.payload->'sender'->>'name', '') || ' ' ||
|
||||
COALESCE(re.payload->'sender'->>'email', '') || ' ' ||
|
||||
COALESCE(re.payload->'conversation'->'additional_attributes'->>'mail_subject', '') || ' ' ||
|
||||
COALESCE(re.payload->'content_attributes'->'email'->>'subject', '') || ' ' ||
|
||||
COALESCE(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '') || ' ' ||
|
||||
COALESCE(m.clean_body, '') || ' ' || COALESCE(m.raw_body, '') || ' ' || COALESCE(re.payload->>'content', '')
|
||||
) AS noise_text
|
||||
FROM tasks t
|
||||
LEFT JOIN messages m ON m.id = t.message_id
|
||||
LEFT JOIN raw_events re ON re.id = t.raw_event_id
|
||||
WHERE t.status = 'pending'
|
||||
)
|
||||
SELECT id, created_at, action_code, route, priority, status, conversation_id, contact_id,
|
||||
opportunity_id, sender_name, sender_email, subject, action, note
|
||||
FROM task_context
|
||||
WHERE ({term_sql} OR upper(coalesce(action_code,'')) = ANY(:noise_action_codes))
|
||||
ORDER BY created_at DESC
|
||||
LIMIT :limit
|
||||
""")
|
||||
params = dict(term_params)
|
||||
params["noise_action_codes"] = NOISE_ACTION_CODES
|
||||
params["limit"] = int(limit)
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(sql, params).mappings().all()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
|
||||
def apply_cleanup(candidates: list[dict]) -> int:
|
||||
if not candidates:
|
||||
return 0
|
||||
ids = [row["id"] for row in candidates]
|
||||
payload = _json({
|
||||
"cleanup_reason": "operations_noise_bounce_or_system_message",
|
||||
"cleanup_by": "cleanup_operations_noise",
|
||||
"clientflow_version": "v4.9.0",
|
||||
})
|
||||
with engine.begin() as conn:
|
||||
for task_id in ids:
|
||||
conn.execute(text("""
|
||||
UPDATE tasks
|
||||
SET status = 'skipped',
|
||||
updated_at = now(),
|
||||
done_at = COALESCE(done_at, now()),
|
||||
done_by = 'cleanup_operations_noise',
|
||||
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:payload AS JSONB)
|
||||
WHERE id = CAST(:task_id AS UUID)
|
||||
AND status = 'pending'
|
||||
"""), {"task_id": task_id, "payload": payload})
|
||||
conn.execute(text("""
|
||||
INSERT INTO task_events (task_id, event_type, payload, created_by)
|
||||
VALUES (CAST(:task_id AS UUID), 'task_skipped_noise_cleanup', CAST(:payload AS JSONB), 'system')
|
||||
"""), {"task_id": task_id, "payload": payload})
|
||||
return len(ids)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Clean pending Operations noise created from bounces/NDRs/system messages.")
|
||||
parser.add_argument("--limit", type=int, default=200)
|
||||
parser.add_argument("--apply", action="store_true", help="Apply cleanup. Without this, only prints candidates.")
|
||||
args = parser.parse_args()
|
||||
|
||||
candidates = find_candidates(limit=args.limit)
|
||||
print(f"Found {len(candidates)} candidate task(s).")
|
||||
for row in candidates:
|
||||
subject = (row.get("subject") or row.get("note") or "")[:90]
|
||||
sender = row.get("sender_email") or row.get("sender_name") or row.get("contact_id") or "sem remetente"
|
||||
print(f"- {row.get('id')} | {row.get('action_code')} | {sender} | {subject}")
|
||||
|
||||
if args.apply:
|
||||
total = apply_cleanup(candidates)
|
||||
print(f"Marked {total} task(s) as skipped.")
|
||||
else:
|
||||
print("Dry-run only. Re-run with --apply to mark candidates as skipped.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
79
scripts/cleanup_stale_reconciliation_items.py
Executable file
79
scripts/cleanup_stale_reconciliation_items.py
Executable file
@@ -0,0 +1,79 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Ignore reconciliation items outside the intended working window.
|
||||
|
||||
Dry-run by default. Use this after an external sync staged too much history.
|
||||
|
||||
Examples:
|
||||
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --days 3
|
||||
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --days 3 --apply
|
||||
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --source odoo --days 3 --apply
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.reconciliation_service import (
|
||||
cleanup_reconciliation_outside_window,
|
||||
ensure_reconciliation_schema,
|
||||
)
|
||||
|
||||
|
||||
def _preview(*, days: int, source_system: str | None, limit: int) -> dict:
|
||||
ensure_reconciliation_schema()
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
days = max(int(days or 3), 1)
|
||||
cutoff = (datetime.now(timezone.utc).date() - timedelta(days=days)).isoformat()
|
||||
params = {"cutoff": cutoff, "limit": int(limit)}
|
||||
source_sql = ""
|
||||
if source_system:
|
||||
source_sql = "AND source_system = :source_system"
|
||||
params["source_system"] = source_system
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text(f"""
|
||||
SELECT id::text, source_system, external_type, document_number,
|
||||
customer_name, document_date, title, status
|
||||
FROM reconciliation_items
|
||||
WHERE status IN ('open', 'needs_review')
|
||||
{source_sql}
|
||||
AND document_date IS NOT NULL
|
||||
AND document_date < CAST(:cutoff AS DATE)
|
||||
ORDER BY document_date ASC, updated_at DESC
|
||||
LIMIT :limit
|
||||
"""), params).mappings().all()
|
||||
return {"days": days, "cutoff": cutoff, "source_system": source_system or "all", "matched": len(rows), "items": [dict(r) for r in rows]}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Ignore stale reconciliation candidates outside a recent working window.")
|
||||
parser.add_argument("--jasmin", action="store_true", help="same as --source jasmin")
|
||||
parser.add_argument("--source", choices=["jasmin", "odoo", "packlink", "manual"], help="only clean one source system")
|
||||
parser.add_argument("--days", type=int, default=3, help="keep open items from the last N days")
|
||||
parser.add_argument("--limit", type=int, default=1000, help="maximum rows to inspect/update")
|
||||
parser.add_argument("--apply", action="store_true", help="apply changes; otherwise dry-run")
|
||||
args = parser.parse_args()
|
||||
|
||||
source = "jasmin" if args.jasmin else args.source
|
||||
if args.apply:
|
||||
result = cleanup_reconciliation_outside_window(days=args.days, source_system=source, limit=args.limit, actor="cleanup_stale_reconciliation_items")
|
||||
else:
|
||||
result = _preview(days=args.days, source_system=source, limit=args.limit)
|
||||
|
||||
print(f"Janela operacional: manter itens >= {result['cutoff']} ({result['days']} dias)")
|
||||
print(f"Fonte: {result['source_system']}")
|
||||
print(f"Encontrados para ignorar: {result['matched']}")
|
||||
for row in result.get("items", [])[:50]:
|
||||
print(f"- {row.get('document_date')} · {row.get('source_system')} · {row.get('external_type')} · {row.get('document_number')} · {row.get('customer_name') or ''}")
|
||||
|
||||
if args.apply:
|
||||
print(f"Aplicado: {result['ignored']} itens marcados como ignored")
|
||||
else:
|
||||
print("Dry-run. Para aplicar, repetir com --apply")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
15
scripts/create_automatic_followups.py
Executable file
15
scripts/create_automatic_followups.py
Executable file
@@ -0,0 +1,15 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Automatic follow-up task creation is disabled in the clean architecture.
|
||||
|
||||
The current design keeps LLM triage action_codes short and uses
|
||||
opportunity/business events for pipeline follow-up state.
|
||||
"""
|
||||
|
||||
|
||||
def main() -> int:
|
||||
print("Automatic legacy follow-up task creation is disabled.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
42
scripts/enrich_fiscal_customers.py
Executable file
42
scripts/enrich_fiscal_customers.py
Executable file
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the ClientFlow fiscal enrichment worker.
|
||||
|
||||
Typical production use:
|
||||
|
||||
python scripts/enrich_fiscal_customers.py --incremental --limit 100
|
||||
|
||||
The worker is idempotent: it enriches/suggests fiscal customers for open
|
||||
opportunities missing local_customer_id and never creates opportunities.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from app.fiscal_enrichment_service import enrich_open_opportunities, ensure_fiscal_enrichment_schema
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Enriquecer oportunidades sem cliente fiscal")
|
||||
parser.add_argument("--limit", type=int, default=100, help="Máximo de oportunidades abertas a analisar")
|
||||
parser.add_argument("--no-auto-apply", action="store_true", help="Criar apenas sugestões, sem auto-associação forte")
|
||||
parser.add_argument("--daily", action="store_true", help="Marcar execução como diária/batch")
|
||||
parser.add_argument("--incremental", action="store_true", help="Marcar execução como incremental")
|
||||
args = parser.parse_args()
|
||||
|
||||
ensure_fiscal_enrichment_schema()
|
||||
mode = "daily" if args.daily else "incremental"
|
||||
result = enrich_open_opportunities(limit=args.limit, apply_safe=not args.no_auto_apply, mode=mode)
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
217
scripts/evaluate_action_decider.py
Executable file
217
scripts/evaluate_action_decider.py
Executable file
@@ -0,0 +1,217 @@
|
||||
#!/usr/bin/env python3
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List
|
||||
|
||||
try:
|
||||
import psycopg
|
||||
except Exception:
|
||||
psycopg = None
|
||||
|
||||
|
||||
def env(name: str, default: str = "") -> str:
|
||||
return os.getenv(name, default).strip()
|
||||
|
||||
|
||||
API_URL = env("EVAL_API_URL", "http://127.0.0.1:8020/analyze")
|
||||
DATASET = Path(env("EVAL_DATASET", "data/action_eval_cases.jsonl"))
|
||||
PSQL_DATABASE_URL = env("PSQL_DATABASE_URL")
|
||||
EVAL_CLEANUP = env("EVAL_CLEANUP", "true").lower() == "true"
|
||||
EVAL_LIMIT = int(env("EVAL_LIMIT", "0") or "0")
|
||||
EVAL_LLM_ONLY = env("EVAL_LLM_ONLY", "false").lower() == "true"
|
||||
|
||||
|
||||
def load_cases() -> List[Dict[str, Any]]:
|
||||
cases = []
|
||||
with DATASET.open("r", encoding="utf-8") as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if line:
|
||||
cases.append(json.loads(line))
|
||||
|
||||
if EVAL_LIMIT > 0:
|
||||
return cases[:EVAL_LIMIT]
|
||||
|
||||
return cases
|
||||
|
||||
|
||||
def post_json(url: str, payload: Dict[str, Any]) -> Dict[str, Any]:
|
||||
raw = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
|
||||
req = urllib.request.Request(
|
||||
url,
|
||||
data=raw,
|
||||
headers={"Content-Type": "application/json"},
|
||||
method="POST",
|
||||
)
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=90) as resp:
|
||||
body = resp.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": 200 <= resp.status < 300,
|
||||
"status": resp.status,
|
||||
"json": json.loads(body) if body else {},
|
||||
"raw": body,
|
||||
}
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
return {
|
||||
"ok": False,
|
||||
"status": e.code,
|
||||
"json": None,
|
||||
"raw": body,
|
||||
}
|
||||
|
||||
|
||||
def cleanup_eval_data() -> None:
|
||||
if not EVAL_CLEANUP or not PSQL_DATABASE_URL or psycopg is None:
|
||||
return
|
||||
|
||||
# Limpeza defensiva: algumas tabelas não têm source_system/source_event_id.
|
||||
# Por isso verificamos as colunas reais antes de construir o DELETE.
|
||||
tables = [
|
||||
"integration_outbox",
|
||||
"tasks",
|
||||
"action_runs",
|
||||
"messages",
|
||||
"raw_events",
|
||||
]
|
||||
|
||||
def table_columns(cur, table: str) -> set[str]:
|
||||
cur.execute(
|
||||
"""
|
||||
select column_name
|
||||
from information_schema.columns
|
||||
where table_name = %s
|
||||
""",
|
||||
(table,),
|
||||
)
|
||||
return {row[0] for row in cur.fetchall()}
|
||||
|
||||
with psycopg.connect(PSQL_DATABASE_URL) as conn:
|
||||
with conn.cursor() as cur:
|
||||
for table in tables:
|
||||
cols = table_columns(cur, table)
|
||||
conditions = []
|
||||
|
||||
if "source_system" in cols:
|
||||
conditions.append("source_system like 'action_eval%'")
|
||||
|
||||
if "conversation_id" in cols:
|
||||
conditions.append("conversation_id like 'eval-%'")
|
||||
|
||||
if "contact_id" in cols:
|
||||
conditions.append("contact_id like 'eval-contact-%'")
|
||||
|
||||
if "source_event_id" in cols:
|
||||
conditions.append("source_event_id like 'eval-%'")
|
||||
|
||||
if "idempotency_key" in cols:
|
||||
conditions.append("idempotency_key like '%eval-%'")
|
||||
|
||||
if "metadata" in cols:
|
||||
conditions.append("metadata::text like '%action_eval%'")
|
||||
|
||||
if not conditions:
|
||||
continue
|
||||
|
||||
sql = f"delete from {table} where " + " or ".join(conditions)
|
||||
cur.execute(sql)
|
||||
|
||||
conn.commit()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
cases = load_cases()
|
||||
|
||||
cleanup_eval_data()
|
||||
|
||||
results = []
|
||||
ok_count = 0
|
||||
|
||||
print(f"Dataset: {DATASET}")
|
||||
print(f"Cases: {len(cases)}")
|
||||
print(f"API: {API_URL}")
|
||||
print(f"EVAL_LLM_ONLY={EVAL_LLM_ONLY}")
|
||||
print("---")
|
||||
|
||||
for i, case in enumerate(cases, start=1):
|
||||
conv_id = f"eval-{case['id']}"
|
||||
payload = {
|
||||
"last_customer_message": case["message"],
|
||||
"previous_context": case.get("context", ""),
|
||||
"source": "action_eval_llm_only" if EVAL_LLM_ONLY else "action_eval",
|
||||
"conversation_id": conv_id,
|
||||
"contact_id": f"eval-contact-{i}",
|
||||
}
|
||||
|
||||
result = post_json(API_URL, payload)
|
||||
|
||||
if not result["ok"]:
|
||||
got = "HTTP_ERROR"
|
||||
confidence = None
|
||||
provider = None
|
||||
note = result["raw"][:300]
|
||||
else:
|
||||
data = result["json"] or {}
|
||||
decision = data.get("action_decision") or {}
|
||||
usage = data.get("usage") or {}
|
||||
got = decision.get("action_code") or data.get("action_result", {}).get("action_code") or "UNKNOWN"
|
||||
confidence = decision.get("confidence")
|
||||
provider = usage.get("provider")
|
||||
note = decision.get("note")
|
||||
|
||||
expected = case["expected"]
|
||||
passed = got == expected
|
||||
ok_count += 1 if passed else 0
|
||||
|
||||
results.append({
|
||||
"id": case["id"],
|
||||
"expected": expected,
|
||||
"got": got,
|
||||
"ok": passed,
|
||||
"confidence": confidence,
|
||||
"provider": provider,
|
||||
"note": note,
|
||||
})
|
||||
|
||||
status = "OK" if passed else "FAIL"
|
||||
print(f"{status:4} {case['id']}")
|
||||
print(f" expected={expected}")
|
||||
print(f" got ={got}")
|
||||
print(f" conf ={confidence} provider={provider}")
|
||||
if not passed:
|
||||
print(f" note ={note}")
|
||||
print("---")
|
||||
|
||||
accuracy = ok_count / len(cases) if cases else 0
|
||||
|
||||
print(f"accuracy={accuracy:.1%} ({ok_count}/{len(cases)})")
|
||||
|
||||
by_expected = {}
|
||||
for r in results:
|
||||
item = by_expected.setdefault(r["expected"], {"ok": 0, "total": 0})
|
||||
item["total"] += 1
|
||||
item["ok"] += 1 if r["ok"] else 0
|
||||
|
||||
print("--- by expected")
|
||||
for code, stats in sorted(by_expected.items()):
|
||||
print(f"{code}: {stats['ok']}/{stats['total']}")
|
||||
|
||||
Path("data/action_eval_last_results.json").write_text(
|
||||
json.dumps(results, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
cleanup_eval_data()
|
||||
|
||||
return 0 if accuracy >= 0.90 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
43
scripts/extract_email_identity.py
Executable file
43
scripts/extract_email_identity.py
Executable file
@@ -0,0 +1,43 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Extract identity signals from an opportunity email body/signature.
|
||||
|
||||
Usage:
|
||||
PYTHONPATH=. python scripts/extract_email_identity.py --opportunity-id <uuid>
|
||||
PYTHONPATH=. python scripts/extract_email_identity.py --text-file /tmp/email.txt --email user@example.pt
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from app.email_identity_extraction_service import extract_email_identity, extract_identity_for_opportunity
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id")
|
||||
parser.add_argument("--text-file")
|
||||
parser.add_argument("--email", default="")
|
||||
parser.add_argument("--subject", default="")
|
||||
parser.add_argument("--no-llm", action="store_true")
|
||||
parser.add_argument("--refresh", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.opportunity_id:
|
||||
result = extract_identity_for_opportunity(args.opportunity_id, refresh=args.refresh, use_llm=not args.no_llm)
|
||||
elif args.text_file:
|
||||
result = extract_email_identity(
|
||||
Path(args.text_file).read_text(encoding="utf-8"),
|
||||
email=args.email,
|
||||
subject=args.subject,
|
||||
use_llm=not args.no_llm,
|
||||
)
|
||||
else:
|
||||
parser.error("use --opportunity-id or --text-file")
|
||||
|
||||
print(json.dumps(result or {}, ensure_ascii=False, indent=2, default=str))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
60
scripts/ignore_manually_cleaned_outbox.py
Executable file
60
scripts/ignore_manually_cleaned_outbox.py
Executable file
@@ -0,0 +1,60 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Mark manually cleaned outbox failures as ignored.
|
||||
|
||||
Use when old Jasmin/Packlink test or duplicate failures were already handled
|
||||
outside the worker and should no longer pollute /operations.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
|
||||
MATCH_SQL = """
|
||||
status = 'failed'
|
||||
AND (
|
||||
COALESCE(last_error, '') ILIKE '%limpo manualmente%'
|
||||
OR COALESCE(last_error, '') ILIKE '%resolvido manualmente%'
|
||||
)
|
||||
"""
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Ignore outbox failures already resolved manually.")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Only show matching rows; do not update.")
|
||||
parser.add_argument("--limit", type=int, default=200, help="Maximum rows to inspect/update.")
|
||||
args = parser.parse_args()
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text(f"""
|
||||
SELECT id::text, target_system, action_type, status, last_error, created_at
|
||||
FROM integration_outbox
|
||||
WHERE {MATCH_SQL}
|
||||
ORDER BY created_at DESC
|
||||
LIMIT :limit
|
||||
"""), {"limit": args.limit}).mappings().all()
|
||||
|
||||
print(f"Matched {len(rows)} manually cleaned failed outbox item(s).")
|
||||
for row in rows:
|
||||
print(f"- {row['id']} {row['target_system']}.{row['action_type']} :: {row['last_error']}")
|
||||
|
||||
if args.dry_run or not rows:
|
||||
print("Dry-run/no-op; no rows updated.")
|
||||
return 0
|
||||
|
||||
ids = [row["id"] for row in rows]
|
||||
conn.execute(text("""
|
||||
UPDATE integration_outbox
|
||||
SET status = 'ignored',
|
||||
last_error = COALESCE(NULLIF(last_error, ''), 'Ignorado por limpeza operacional manual.'),
|
||||
updated_at = now()
|
||||
WHERE id::text = ANY(:ids)
|
||||
"""), {"ids": ids})
|
||||
print(f"Updated {len(ids)} item(s) to ignored.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
33
scripts/inspect_customer_tax_id_conflict.py
Normal file
33
scripts/inspect_customer_tax_id_conflict.py
Normal file
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect customers/opportunities involved in a duplicate NIF conflict."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
from app.commercial_service import normalize_tax_id
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--tax-id", required=True)
|
||||
args = parser.parse_args()
|
||||
tax_id = normalize_tax_id(args.tax_id)
|
||||
with engine.begin() as conn:
|
||||
customers = conn.execute(text("""
|
||||
SELECT c.id::text, c.name, c.tax_id, c.email, c.phone, c.street_name, c.postal_zone, c.city_name,
|
||||
c.country, c.created_at, c.updated_at,
|
||||
(SELECT count(*) FROM opportunities o WHERE o.local_customer_id = c.id)::int AS opportunities,
|
||||
(SELECT count(*) FROM commercial_documents cd WHERE cd.customer_id = c.id)::int AS documents
|
||||
FROM customers c
|
||||
WHERE c.tax_id = :tax_id OR c.name ILIKE '%' || :tax_id || '%'
|
||||
ORDER BY c.tax_id NULLS LAST, c.updated_at DESC
|
||||
"""), {"tax_id": tax_id}).mappings().all()
|
||||
print(json.dumps({"tax_id": tax_id, "customers": [dict(c) for c in customers]}, ensure_ascii=False, indent=2, default=str))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
42
scripts/inspect_jasmin_candidates_for_opportunity.py
Executable file
42
scripts/inspect_jasmin_candidates_for_opportunity.py
Executable file
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect Jasmin document candidates for an opportunity.
|
||||
|
||||
Optionally sync recent Jasmin documents first, then prints open/valid candidates first
|
||||
and ignored closed/completed/cancelled documents afterwards.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
from decimal import Decimal
|
||||
from typing import Any
|
||||
|
||||
|
||||
def _json_default(value: Any) -> str:
|
||||
if isinstance(value, Decimal):
|
||||
return str(value)
|
||||
return str(value)
|
||||
|
||||
|
||||
async def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id", required=True)
|
||||
parser.add_argument("--limit", type=int, default=20)
|
||||
parser.add_argument("--sync", action="store_true", help="Sync recent Jasmin reconciliation candidates before inspection")
|
||||
parser.add_argument("--days", type=int, default=30)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.sync:
|
||||
from app.external_reconciliation_sync import sync_jasmin_reconciliation_candidates
|
||||
sync_result = await sync_jasmin_reconciliation_candidates(limit=100, days=args.days)
|
||||
print(json.dumps({"sync_jasmin": sync_result}, ensure_ascii=False, indent=2, default=_json_default))
|
||||
|
||||
from app.jasmin_backfill_service import find_jasmin_document_candidates_for_opportunity
|
||||
candidates = find_jasmin_document_candidates_for_opportunity(args.opportunity_id, limit=args.limit)
|
||||
print(json.dumps({"opportunity_id": args.opportunity_id, "candidates": candidates}, ensure_ascii=False, indent=2, default=_json_default))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(asyncio.run(main()))
|
||||
38
scripts/install_systemd_timers.sh
Executable file
38
scripts/install_systemd_timers.sh
Executable file
@@ -0,0 +1,38 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
ROOT_DIR="${1:-/mnt/ssd/home/plx/clientflow_backend}"
|
||||
SYSTEMD_DIR="/etc/systemd/system"
|
||||
|
||||
if [[ ! -d "$ROOT_DIR" ]]; then
|
||||
echo "Project directory not found: $ROOT_DIR" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
cd "$ROOT_DIR"
|
||||
|
||||
sudo cp deploy/systemd/clientflow-outbox-jasmin.service "$SYSTEMD_DIR/"
|
||||
sudo cp deploy/systemd/clientflow-outbox-jasmin.timer "$SYSTEMD_DIR/"
|
||||
|
||||
if [[ "${ENABLE_FISCAL_ENRICHMENT_TIMER:-false}" == "true" ]]; then
|
||||
sudo cp deploy/systemd/clientflow-fiscal-enrichment.service "$SYSTEMD_DIR/"
|
||||
sudo cp deploy/systemd/clientflow-fiscal-enrichment.timer "$SYSTEMD_DIR/"
|
||||
fi
|
||||
|
||||
if [[ "${ENABLE_PACKLINK_TIMER:-false}" == "true" ]]; then
|
||||
sudo cp deploy/systemd/clientflow-outbox-packlink.service "$SYSTEMD_DIR/"
|
||||
sudo cp deploy/systemd/clientflow-outbox-packlink.timer "$SYSTEMD_DIR/"
|
||||
fi
|
||||
|
||||
sudo systemctl daemon-reload
|
||||
sudo systemctl enable --now clientflow-outbox-jasmin.timer
|
||||
|
||||
if [[ "${ENABLE_FISCAL_ENRICHMENT_TIMER:-false}" == "true" ]]; then
|
||||
sudo systemctl enable --now clientflow-fiscal-enrichment.timer
|
||||
fi
|
||||
|
||||
if [[ "${ENABLE_PACKLINK_TIMER:-false}" == "true" ]]; then
|
||||
sudo systemctl enable --now clientflow-outbox-packlink.timer
|
||||
fi
|
||||
|
||||
systemctl list-timers | grep clientflow || true
|
||||
512
scripts/prepare_task.py
Executable file
512
scripts/prepare_task.py
Executable file
@@ -0,0 +1,512 @@
|
||||
#!/usr/bin/env python3
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import psycopg
|
||||
|
||||
|
||||
def env(name: str, default: str = "") -> str:
|
||||
return os.getenv(name, default).strip()
|
||||
|
||||
|
||||
PSQL_DATABASE_URL = env("PSQL_DATABASE_URL")
|
||||
OPENROUTER_API_KEY = env("OPENROUTER_API_KEY")
|
||||
OPENROUTER_URL = env("OPENROUTER_URL", "https://openrouter.ai/api/v1/chat/completions")
|
||||
OPENROUTER_MODEL = env("EXTRACTION_MODEL", env("OPENROUTER_MODEL", "qwen/qwen3-30b-a3b"))
|
||||
|
||||
|
||||
def extract_first_json_object(raw: str) -> str:
|
||||
s = str(raw or "").strip()
|
||||
|
||||
if s.startswith("```"):
|
||||
lines = s.splitlines()
|
||||
if lines and lines[0].strip().startswith("```"):
|
||||
lines = lines[1:]
|
||||
if lines and lines[-1].strip().startswith("```"):
|
||||
lines = lines[:-1]
|
||||
s = "\n".join(lines).strip()
|
||||
|
||||
start = s.find("{")
|
||||
if start == -1:
|
||||
raise ValueError(f"no JSON object found: {s[:300]}")
|
||||
|
||||
in_string = False
|
||||
escaped = False
|
||||
depth = 0
|
||||
|
||||
for i in range(start, len(s)):
|
||||
ch = s[i]
|
||||
|
||||
if escaped:
|
||||
escaped = False
|
||||
continue
|
||||
|
||||
if ch == "\\":
|
||||
escaped = True
|
||||
continue
|
||||
|
||||
if ch == '"':
|
||||
in_string = not in_string
|
||||
continue
|
||||
|
||||
if in_string:
|
||||
continue
|
||||
|
||||
if ch == "{":
|
||||
depth += 1
|
||||
elif ch == "}":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return s[start:i + 1]
|
||||
|
||||
raise ValueError(f"incomplete JSON object: {s[:500]}")
|
||||
|
||||
|
||||
def parse_json(raw: str) -> Dict[str, Any]:
|
||||
try:
|
||||
return json.loads(raw)
|
||||
except Exception:
|
||||
return json.loads(extract_first_json_object(raw))
|
||||
|
||||
|
||||
def strip_html(text: str) -> str:
|
||||
text = re.sub(r"<br\s*/?>", "\n", text or "", flags=re.I)
|
||||
text = re.sub(r"</p\s*>", "\n", text, flags=re.I)
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
text = text.replace(" ", " ")
|
||||
text = text.replace("&", "&")
|
||||
text = text.replace(""", '"')
|
||||
text = text.replace("'", "'")
|
||||
return re.sub(r"[ \t]+", " ", text).strip()
|
||||
|
||||
|
||||
def remove_quoted_text(text: str) -> str:
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
markers = [
|
||||
"\nÀs ",
|
||||
"\nEm ",
|
||||
"\nOn ",
|
||||
"\n-----Original Message-----",
|
||||
"\nDe:",
|
||||
"\nFrom:",
|
||||
]
|
||||
|
||||
cut = len(text)
|
||||
for marker in markers:
|
||||
idx = text.find(marker)
|
||||
if idx != -1:
|
||||
cut = min(cut, idx)
|
||||
|
||||
lines = []
|
||||
for line in text[:cut].splitlines():
|
||||
if line.strip().startswith(">"):
|
||||
continue
|
||||
lines.append(line)
|
||||
|
||||
return "\n".join(lines).strip()
|
||||
|
||||
|
||||
def extract_message_content(payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
|
||||
msg = payload.get("message") or payload.get("messages") or {}
|
||||
|
||||
if isinstance(msg, list):
|
||||
msg = msg[0] if msg else {}
|
||||
|
||||
if not isinstance(msg, dict):
|
||||
return None
|
||||
|
||||
content = msg.get("content") or payload.get("content") or ""
|
||||
message_type = str(msg.get("message_type") or payload.get("message_type") or "").lower()
|
||||
private = bool(msg.get("private") or payload.get("private"))
|
||||
|
||||
if not content:
|
||||
return None
|
||||
|
||||
return {
|
||||
"content": remove_quoted_text(strip_html(str(content))),
|
||||
"message_type": message_type,
|
||||
"private": private,
|
||||
"id": str(msg.get("id") or payload.get("id") or ""),
|
||||
}
|
||||
|
||||
|
||||
def fetch_task(conn, task_id: Optional[str], conversation_id: Optional[str]) -> Dict[str, Any]:
|
||||
if task_id:
|
||||
sql = """
|
||||
select id::text, conversation_id, contact_id, action_code, route, status, action, note
|
||||
from tasks
|
||||
where id = %s
|
||||
limit 1
|
||||
"""
|
||||
params = (task_id,)
|
||||
else:
|
||||
sql = """
|
||||
select id::text, conversation_id, contact_id, action_code, route, status, action, note
|
||||
from tasks
|
||||
where conversation_id = %s
|
||||
order by created_at desc
|
||||
limit 1
|
||||
"""
|
||||
params = (conversation_id,)
|
||||
|
||||
with conn.cursor(row_factory=psycopg.rows.dict_row) as cur:
|
||||
cur.execute(sql, params)
|
||||
row = cur.fetchone()
|
||||
|
||||
if not row:
|
||||
raise SystemExit("ERRO: task não encontrada.")
|
||||
|
||||
return dict(row)
|
||||
|
||||
|
||||
def fetch_conversation_messages(conn, conversation_id: str, limit: int = 20) -> List[Dict[str, Any]]:
|
||||
with conn.cursor(row_factory=psycopg.rows.dict_row) as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
select created_at, payload
|
||||
from raw_events
|
||||
where source_system = 'chatwoot'
|
||||
and conversation_id = %s
|
||||
order by created_at asc
|
||||
limit %s
|
||||
""",
|
||||
(conversation_id, limit),
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
|
||||
messages = []
|
||||
for row in rows:
|
||||
payload = row.get("payload") or {}
|
||||
if isinstance(payload, str):
|
||||
try:
|
||||
payload = json.loads(payload)
|
||||
except Exception:
|
||||
payload = {}
|
||||
|
||||
msg = extract_message_content(payload)
|
||||
if not msg:
|
||||
continue
|
||||
|
||||
# Para extração, manter incoming e outgoing públicos, ignorar notas privadas.
|
||||
if msg["private"]:
|
||||
continue
|
||||
|
||||
messages.append({
|
||||
"created_at": str(row["created_at"]),
|
||||
**msg,
|
||||
})
|
||||
|
||||
return messages
|
||||
|
||||
|
||||
def build_prompt(task: Dict[str, Any], messages: List[Dict[str, Any]], prep_type: str) -> List[Dict[str, str]]:
|
||||
conversation_text = "\n\n".join(
|
||||
f"[{m['created_at']}] {m.get('message_type') or 'message'}:\n{m['content']}"
|
||||
for m in messages
|
||||
if m.get("content")
|
||||
)
|
||||
|
||||
if prep_type == "proforma":
|
||||
objective = """
|
||||
Objetivo: preparar dados para emitir fatura pró-forma.
|
||||
Extrai apenas dados relevantes para faturação/proforma:
|
||||
cliente, empresa, email, telefone, NIF, morada fiscal, produto, quantidade, preço, condições comerciais e dados em falta para emitir a pró-forma.
|
||||
|
||||
Para prep_type=proforma:
|
||||
- NÃO peças morada de entrega, destinatário ou telefone para transportadora, exceto se a conversa indicar que são necessários para a pró-forma.
|
||||
- Morada de entrega/recolha pertence à fase de envio/recolha, não à fase de pró-forma.
|
||||
- Se já houver NIF, morada fiscal, nome/empresa de faturação, email, produto e preço, missing_fields deve ser [].
|
||||
"""
|
||||
elif prep_type == "shipment":
|
||||
objective = """
|
||||
Objetivo: preparar envio.
|
||||
Extrai apenas dados relevantes para logística de entrega:
|
||||
morada de entrega, contacto no local, telefone, produto/equipamento, quantidade, instruções, estado do pagamento e dados em falta.
|
||||
|
||||
Para prep_type=shipment:
|
||||
- Usa shipment.delivery_address para a morada de entrega.
|
||||
- Usa shipment.recipient_name e shipment.recipient_phone para contacto da entrega.
|
||||
- Os campos obrigatórios são morada de entrega, contacto, telefone, produto/equipamento, quantidade e estado do pagamento.
|
||||
- NÃO coloques customer.tax_id, billing.tax_id, billing_address ou customer.email em missing_fields, exceto se forem explicitamente necessários para a transportadora.
|
||||
- Se já existir morada de entrega, contacto e telefone, não peças esses dados novamente.
|
||||
- Se faltar produto, quantidade ou comprovativo/estado de pagamento, pede apenas esses dados.
|
||||
- A suggested_reply deve falar em envio/entrega, nunca em recolha.
|
||||
"""
|
||||
elif prep_type == "pickup":
|
||||
objective = """
|
||||
Objetivo: preparar recolha.
|
||||
Extrai apenas dados relevantes para recolha ou assistência logística:
|
||||
morada de recolha, contacto no local, telefone, produto/equipamento a recolher, motivo/instruções, data preferida e dados em falta.
|
||||
|
||||
Para prep_type=pickup:
|
||||
- Usa shipment.pickup_address para a morada de recolha.
|
||||
- Usa shipment.recipient_name e shipment.recipient_phone para a pessoa de contacto da recolha.
|
||||
- Se a conversa só tiver uma morada e o objetivo é recolha, coloca essa morada em shipment.pickup_address, não em shipment.delivery_address.
|
||||
- NÃO coloques sale.total_estimate, customer.tax_id, billing.tax_id, billing_address ou customer.email em missing_fields.
|
||||
- Os campos importantes são pickup_address, recipient_name, recipient_phone, produto/equipamento, motivo/instruções da recolha.
|
||||
- payment.status só é obrigatório se a tarefa for claramente sobre pagamento ou envio após pagamento.
|
||||
- A suggested_reply deve falar em recolha, nunca em envio.
|
||||
"""
|
||||
else:
|
||||
objective = """
|
||||
Objetivo: preparar execução operacional da tarefa.
|
||||
Extrai dados úteis para a ação, dados em falta e resposta sugerida.
|
||||
"""
|
||||
|
||||
system = f"""
|
||||
És o ClientFlow Sales Assistant.
|
||||
A tua função é extrair dados operacionais de conversas B2B para ajudar a preparar pró-forma, fatura, envio ou recolha.
|
||||
|
||||
Regras:
|
||||
- Devolve apenas JSON puro.
|
||||
- Não inventes dados.
|
||||
- Se um dado não existir, usa null.
|
||||
- Ignora texto citado antigo, dados da BLIF, IBANs e assinaturas da BLIF.
|
||||
- Não associes IBAN ao cliente.
|
||||
- Distingue morada fiscal de morada de entrega/recolha.
|
||||
- Usa evidências curtas.
|
||||
- Se faltar dado necessário, coloca em missing_fields.
|
||||
- suggested_reply deve ser uma resposta curta em português para pedir dados em falta ou indicar o próximo passo.
|
||||
- Não digas que a pró-forma/fatura/envio já foi emitida, enviada, agendada ou concluída.
|
||||
- O assistente apenas prepara dados para revisão; usa linguagem como "vamos preparar", "podemos avançar", "dados suficientes para preparar".
|
||||
"""
|
||||
|
||||
user = f"""
|
||||
{objective}
|
||||
|
||||
Tarefa:
|
||||
action_code: {task.get('action_code')}
|
||||
route: {task.get('route')}
|
||||
action: {task.get('action')}
|
||||
note: {task.get('note')}
|
||||
|
||||
Conversa:
|
||||
{conversation_text}
|
||||
|
||||
Schema obrigatório:
|
||||
{{
|
||||
"customer": {{
|
||||
"name": null,
|
||||
"company": null,
|
||||
"email": null,
|
||||
"phone": null,
|
||||
"tax_id": null
|
||||
}},
|
||||
"billing": {{
|
||||
"billing_name": null,
|
||||
"tax_id": null,
|
||||
"billing_address": null,
|
||||
"billing_email": null
|
||||
}},
|
||||
"sale": {{
|
||||
"products": [
|
||||
{{
|
||||
"name": null,
|
||||
"description": null,
|
||||
"quantity": null,
|
||||
"unit_price": null,
|
||||
"currency": "EUR"
|
||||
}}
|
||||
],
|
||||
"total_estimate": null,
|
||||
"commercial_terms": null
|
||||
}},
|
||||
"payment": {{
|
||||
"status": null,
|
||||
"proof_mentioned": false
|
||||
}},
|
||||
"shipment": {{
|
||||
"delivery_address": null,
|
||||
"pickup_address": null,
|
||||
"recipient_name": null,
|
||||
"recipient_phone": null,
|
||||
"instructions": null
|
||||
}},
|
||||
"missing_fields": [],
|
||||
"suggested_reply": "",
|
||||
"confidence": 0.0,
|
||||
"evidence": []
|
||||
}}
|
||||
"""
|
||||
|
||||
return [
|
||||
{"role": "system", "content": system},
|
||||
{"role": "user", "content": user},
|
||||
]
|
||||
|
||||
|
||||
def call_openrouter(messages: List[Dict[str, str]]) -> Tuple[Dict[str, Any], Dict[str, Any], str]:
|
||||
body = {
|
||||
"model": OPENROUTER_MODEL,
|
||||
"messages": messages,
|
||||
"temperature": 0,
|
||||
}
|
||||
|
||||
req = urllib.request.Request(
|
||||
OPENROUTER_URL,
|
||||
data=json.dumps(body, ensure_ascii=False).encode("utf-8"),
|
||||
headers={
|
||||
"Authorization": f"Bearer {OPENROUTER_API_KEY}",
|
||||
"Content-Type": "application/json",
|
||||
"HTTP-Referer": "https://clientflow.blif.pt",
|
||||
"X-Title": "ClientFlow",
|
||||
},
|
||||
method="POST",
|
||||
)
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=120) as resp:
|
||||
raw = resp.read().decode("utf-8", errors="replace")
|
||||
parsed = json.loads(raw)
|
||||
except urllib.error.HTTPError as e:
|
||||
raw = e.read().decode("utf-8", errors="replace")
|
||||
raise RuntimeError(f"OpenRouter HTTP {e.code}: {raw[:1000]}")
|
||||
|
||||
content = parsed["choices"][0]["message"].get("content") or "{}"
|
||||
extracted = parse_json(content)
|
||||
return extracted, parsed, content
|
||||
|
||||
|
||||
def save_preparation(conn, task: Dict[str, Any], prep_type: str, extracted: Dict[str, Any], raw_response: Dict[str, Any]) -> str:
|
||||
usage = raw_response.get("usage") or {}
|
||||
model = raw_response.get("model") or OPENROUTER_MODEL
|
||||
provider = raw_response.get("provider") or raw_response.get("provider_name")
|
||||
|
||||
missing_fields = extracted.get("missing_fields") or []
|
||||
suggested_reply = extracted.get("suggested_reply") or ""
|
||||
confidence = extracted.get("confidence")
|
||||
|
||||
with conn.cursor() as cur:
|
||||
cur.execute(
|
||||
"""
|
||||
insert into task_preparations (
|
||||
task_id,
|
||||
conversation_id,
|
||||
contact_id,
|
||||
prep_type,
|
||||
status,
|
||||
extracted_data,
|
||||
missing_fields,
|
||||
suggested_reply,
|
||||
confidence,
|
||||
model,
|
||||
provider,
|
||||
total_tokens,
|
||||
cost,
|
||||
raw_response
|
||||
)
|
||||
values (
|
||||
%s, %s, %s, %s, 'draft',
|
||||
%s::jsonb,
|
||||
%s::jsonb,
|
||||
%s,
|
||||
%s,
|
||||
%s,
|
||||
%s,
|
||||
%s,
|
||||
%s,
|
||||
%s::jsonb
|
||||
)
|
||||
returning id::text
|
||||
""",
|
||||
(
|
||||
task["id"],
|
||||
task["conversation_id"],
|
||||
task.get("contact_id"),
|
||||
prep_type,
|
||||
json.dumps(extracted, ensure_ascii=False),
|
||||
json.dumps(missing_fields, ensure_ascii=False),
|
||||
suggested_reply,
|
||||
confidence,
|
||||
model,
|
||||
provider,
|
||||
int(usage.get("total_tokens") or 0),
|
||||
usage.get("cost") or 0,
|
||||
json.dumps(raw_response, ensure_ascii=False),
|
||||
),
|
||||
)
|
||||
prep_id = cur.fetchone()[0]
|
||||
|
||||
conn.commit()
|
||||
return prep_id
|
||||
|
||||
|
||||
def run_preparation(
|
||||
*,
|
||||
task_id: Optional[str] = None,
|
||||
conversation_id: Optional[str] = None,
|
||||
prep_type: str,
|
||||
database_url: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
db_url = database_url or PSQL_DATABASE_URL
|
||||
if not task_id and not conversation_id:
|
||||
raise ValueError("Usa task_id ou conversation_id.")
|
||||
if not db_url:
|
||||
raise RuntimeError("PSQL_DATABASE_URL/DATABASE_URL não definida.")
|
||||
if not OPENROUTER_API_KEY:
|
||||
raise RuntimeError("OPENROUTER_API_KEY não definida.")
|
||||
|
||||
with psycopg.connect(db_url) as conn:
|
||||
task = fetch_task(conn, task_id, conversation_id)
|
||||
messages = fetch_conversation_messages(conn, task["conversation_id"])
|
||||
if not messages:
|
||||
raise RuntimeError("Não encontrei mensagens públicas da conversa.")
|
||||
prompt = build_prompt(task, messages, prep_type)
|
||||
extracted, raw_response, raw_content = call_openrouter(prompt)
|
||||
prep_id = save_preparation(conn, task, prep_type, extracted, raw_response)
|
||||
|
||||
return {
|
||||
"preparation_id": prep_id,
|
||||
"task_id": task["id"],
|
||||
"conversation_id": task["conversation_id"],
|
||||
"prep_type": prep_type,
|
||||
"extracted": extracted,
|
||||
}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--task-id")
|
||||
parser.add_argument("--conversation-id")
|
||||
parser.add_argument("--type", choices=["proforma", "shipment", "pickup", "generic"], required=True)
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.task_id and not args.conversation_id:
|
||||
raise SystemExit("Usa --task-id ou --conversation-id.")
|
||||
|
||||
if not PSQL_DATABASE_URL:
|
||||
raise SystemExit("PSQL_DATABASE_URL não definida.")
|
||||
|
||||
if not OPENROUTER_API_KEY:
|
||||
raise SystemExit("OPENROUTER_API_KEY não definida.")
|
||||
|
||||
with psycopg.connect(PSQL_DATABASE_URL) as conn:
|
||||
task = fetch_task(conn, args.task_id, args.conversation_id)
|
||||
messages = fetch_conversation_messages(conn, task["conversation_id"])
|
||||
|
||||
if not messages:
|
||||
raise SystemExit("ERRO: não encontrei mensagens públicas da conversa.")
|
||||
|
||||
prompt = build_prompt(task, messages, args.type)
|
||||
extracted, raw_response, raw_content = call_openrouter(prompt)
|
||||
prep_id = save_preparation(conn, task, args.type, extracted, raw_response)
|
||||
|
||||
print(f"OK: preparation_id={prep_id}")
|
||||
print(f"task_id={task['id']}")
|
||||
print(f"conversation_id={task['conversation_id']}")
|
||||
print(f"prep_type={args.type}")
|
||||
print("--- extracted")
|
||||
print(json.dumps(extracted, ensure_ascii=False, indent=2))
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
318
scripts/process_outbox.py
Normal file
318
scripts/process_outbox.py
Normal file
@@ -0,0 +1,318 @@
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, Literal
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
|
||||
from app.config import settings
|
||||
from app.integration_outbox_service import (
|
||||
claim_pending_outbox,
|
||||
recover_stale_processing_outbox,
|
||||
mark_outbox_blocked,
|
||||
mark_outbox_dry_run,
|
||||
mark_outbox_failed,
|
||||
mark_outbox_sent,
|
||||
)
|
||||
|
||||
|
||||
Outcome = Literal["processed", "skipped"]
|
||||
|
||||
|
||||
def env_bool(name: str, default: bool = False) -> bool:
|
||||
fallback = "true" if default else "false"
|
||||
value = os.getenv(name, fallback).strip().lower()
|
||||
return value in {"true", "1", "yes", "on"}
|
||||
|
||||
|
||||
def is_dry_run() -> bool:
|
||||
return env_bool("OUTBOX_DRY_RUN", True)
|
||||
|
||||
|
||||
def integration_enabled(target_system: str) -> bool:
|
||||
key = f"{str(target_system or '').upper()}_OUTBOX_ENABLED"
|
||||
return env_bool(key, False)
|
||||
|
||||
|
||||
def build_chatwoot_note(payload: Dict[str, Any]) -> str:
|
||||
event_type = payload.get("event_type", "")
|
||||
action = payload.get("action", "")
|
||||
note = payload.get("note", "")
|
||||
conversation_id = payload.get("conversation_id", "")
|
||||
|
||||
return f"""🤖 ClientFlow
|
||||
|
||||
Evento:
|
||||
{event_type}
|
||||
|
||||
Ação:
|
||||
{action}
|
||||
|
||||
Nota:
|
||||
{note}
|
||||
|
||||
Conversa:
|
||||
{conversation_id}
|
||||
"""
|
||||
|
||||
|
||||
async def process_chatwoot_add_private_note(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
conversation_id = payload.get("conversation_id")
|
||||
|
||||
if not conversation_id:
|
||||
raise RuntimeError("conversation_id em falta no payload")
|
||||
|
||||
if is_dry_run():
|
||||
message = f"DRY-RUN chatwoot.add_private_note conversation_id={conversation_id}"
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
if not settings.chatwoot_write_enabled:
|
||||
raise RuntimeError("CHATWOOT_WRITE_ENABLED=false")
|
||||
|
||||
if not settings.chatwoot_base_url or not settings.chatwoot_account_id or not settings.chatwoot_api_token:
|
||||
raise RuntimeError("Configuração Chatwoot incompleta")
|
||||
|
||||
url = (
|
||||
settings.chatwoot_base_url.rstrip("/")
|
||||
+ f"/api/v1/accounts/{settings.chatwoot_account_id}"
|
||||
+ f"/conversations/{conversation_id}/messages"
|
||||
)
|
||||
|
||||
body = {
|
||||
"content": build_chatwoot_note(payload),
|
||||
"message_type": "outgoing",
|
||||
"private": True,
|
||||
"content_type": "text",
|
||||
"content_attributes": {},
|
||||
}
|
||||
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"api_access_token": settings.chatwoot_api_token,
|
||||
}
|
||||
|
||||
async with httpx.AsyncClient(timeout=30) as client:
|
||||
response = await client.post(url, headers=headers, json=body)
|
||||
|
||||
if response.status_code >= 400:
|
||||
raise RuntimeError(f"Chatwoot error {response.status_code}: {response.text}")
|
||||
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
|
||||
async def process_mautic_add_tag(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
|
||||
if is_dry_run():
|
||||
message = (
|
||||
"DRY-RUN mautic.add_tag "
|
||||
f"conversation_id={payload.get('conversation_id')} "
|
||||
f"tag={payload.get('tag')}"
|
||||
)
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
from app.mautic_client import add_tag_from_outbox_payload
|
||||
|
||||
add_tag_from_outbox_payload(payload)
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
|
||||
async def process_mautic_remove_tag(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
|
||||
if is_dry_run():
|
||||
message = (
|
||||
"DRY-RUN mautic.remove_tag "
|
||||
f"conversation_id={payload.get('conversation_id')} "
|
||||
f"tag={payload.get('tag')}"
|
||||
)
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
from app.mautic_client import remove_tag_from_outbox_payload
|
||||
|
||||
remove_tag_from_outbox_payload(payload)
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
|
||||
async def process_packlink_create_shipment(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
opportunity_id = payload.get("opportunity_id")
|
||||
|
||||
if is_dry_run():
|
||||
message = f"DRY-RUN packlink.create_shipment opportunity_id={opportunity_id}"
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
if not settings.packlink_enabled:
|
||||
raise RuntimeError("PACKLINK_ENABLED=false")
|
||||
|
||||
from app.packlink_service import create_shipment_from_outbox_payload
|
||||
|
||||
result = await create_shipment_from_outbox_payload(payload)
|
||||
print(f"Packlink shipment created reference={result.get('reference')}")
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
|
||||
|
||||
|
||||
async def process_jasmin_create_quotation(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
opportunity_id = payload.get("opportunity_id")
|
||||
|
||||
if is_dry_run():
|
||||
message = f"DRY-RUN jasmin.create_quotation opportunity_id={opportunity_id}"
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
if not settings.jasmin_enabled:
|
||||
raise RuntimeError("JASMIN_ENABLED=false")
|
||||
|
||||
from app.jasmin_service import process_create_quotation_outbox
|
||||
|
||||
result = await process_create_quotation_outbox(payload)
|
||||
print(f"Jasmin quotation created id={result.get('quotation_id')}")
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
|
||||
async def process_jasmin_convert_invoice(item: Dict[str, Any]) -> Outcome:
|
||||
payload = item.get("payload") or {}
|
||||
opportunity_id = payload.get("opportunity_id")
|
||||
|
||||
if is_dry_run():
|
||||
message = f"DRY-RUN jasmin.convert_quotation_to_invoice opportunity_id={opportunity_id}"
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
if not settings.jasmin_enabled:
|
||||
raise RuntimeError("JASMIN_ENABLED=false")
|
||||
|
||||
from app.jasmin_service import process_convert_invoice_outbox
|
||||
|
||||
result = await process_convert_invoice_outbox(payload)
|
||||
print(f"Jasmin invoice created id={result.get('invoice_id')}")
|
||||
mark_outbox_sent(item["id"])
|
||||
return "processed"
|
||||
|
||||
async def process_item(item: Dict[str, Any]) -> Outcome:
|
||||
target_system = item.get("target_system")
|
||||
action_type = item.get("action_type")
|
||||
|
||||
print(f"Processing {item['id']} {target_system}.{action_type}")
|
||||
|
||||
if not integration_enabled(target_system):
|
||||
message = f"Integração desativada: {target_system}.{action_type}. Ative {str(target_system or '').upper()}_OUTBOX_ENABLED=true para processar."
|
||||
print(f"BLOCKED {message}")
|
||||
mark_outbox_blocked(item["id"], message)
|
||||
return "skipped"
|
||||
|
||||
if target_system == "chatwoot" and action_type == "add_private_note":
|
||||
return await process_chatwoot_add_private_note(item)
|
||||
|
||||
if target_system == ("t" + "wenty"):
|
||||
print(f"SKIP legacy external CRM outbox item {item.get('id')}: integração removida")
|
||||
return "skipped"
|
||||
|
||||
if target_system == "mautic" and action_type == "add_tag":
|
||||
return await process_mautic_add_tag(item)
|
||||
|
||||
if target_system == "mautic" and action_type == "remove_tag":
|
||||
return await process_mautic_remove_tag(item)
|
||||
|
||||
if target_system == "packlink" and action_type == "create_shipment":
|
||||
return await process_packlink_create_shipment(item)
|
||||
|
||||
if target_system == "jasmin" and action_type == "create_quotation":
|
||||
return await process_jasmin_create_quotation(item)
|
||||
|
||||
if target_system == "jasmin" and action_type == "convert_quotation_to_invoice":
|
||||
return await process_jasmin_convert_invoice(item)
|
||||
|
||||
if is_dry_run():
|
||||
message = f"DRY-RUN unsupported-now {target_system}.{action_type}"
|
||||
print(message)
|
||||
mark_outbox_dry_run(item["id"], message)
|
||||
return "processed"
|
||||
|
||||
raise RuntimeError(f"Handler não implementado: {target_system}.{action_type}")
|
||||
|
||||
|
||||
async def main() -> int:
|
||||
|
||||
# Nota: as notas privadas do Chatwoot podem estar desligadas sem bloquear
|
||||
# outras integrações como Packlink, Mautic ou Jasmin. A decisão de processar
|
||||
# cada target_system fica em integration_enabled() e no handler específico.
|
||||
if os.getenv("CLIENTFLOW_DISABLE_CHATWOOT_PRIVATE_NOTES", "true").lower() in {"1", "true", "yes", "sim"}:
|
||||
print("ClientFlow Chatwoot private notes disabled by env; non-Chatwoot outbox will still run.")
|
||||
|
||||
limit = int(os.getenv("OUTBOX_LIMIT", "50"))
|
||||
target_system = os.getenv("OUTBOX_TARGET_SYSTEM", "").strip() or None
|
||||
|
||||
worker_id = os.getenv("OUTBOX_WORKER_ID", f"process_outbox:{os.getpid()}")
|
||||
|
||||
if env_bool("OUTBOX_RECOVER_STALE_BEFORE_PROCESS", True):
|
||||
recovered = recover_stale_processing_outbox(
|
||||
mode=os.getenv("OUTBOX_STALE_RECOVERY_MODE", "manual_only"),
|
||||
actor=worker_id,
|
||||
)
|
||||
if recovered:
|
||||
print(f"Recovered stale processing outbox items: {len(recovered)}")
|
||||
|
||||
items = claim_pending_outbox(
|
||||
limit=limit,
|
||||
target_system=target_system,
|
||||
lock_owner=worker_id,
|
||||
)
|
||||
|
||||
print(f"Claimed outbox items: {len(items)}")
|
||||
print(f"OUTBOX_WORKER_ID={worker_id}")
|
||||
print(f"OUTBOX_DRY_RUN={is_dry_run()}")
|
||||
print(f"OUTBOX_TARGET_SYSTEM={target_system or 'all'}")
|
||||
|
||||
processed = 0
|
||||
skipped = 0
|
||||
failed = 0
|
||||
|
||||
for item in items:
|
||||
try:
|
||||
outcome = await process_item(item)
|
||||
|
||||
if outcome == "processed":
|
||||
processed += 1
|
||||
else:
|
||||
skipped += 1
|
||||
|
||||
except Exception as exc:
|
||||
failed += 1
|
||||
print(f"FAILED {item.get('id')}: {exc}")
|
||||
mark_outbox_failed(item["id"], str(exc))
|
||||
|
||||
print(f"Processed: {processed}")
|
||||
print(f"Skipped: {skipped}")
|
||||
print(f"Failed: {failed}")
|
||||
|
||||
return 0 if failed == 0 else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(asyncio.run(main()))
|
||||
41
scripts/recover_stale_outbox.py
Executable file
41
scripts/recover_stale_outbox.py
Executable file
@@ -0,0 +1,41 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Recover or expose stale outbox processing rows.
|
||||
|
||||
Usage examples:
|
||||
OUTBOX_STALE_RECOVERY_MODE=manual_only python scripts/recover_stale_outbox.py
|
||||
OUTBOX_STALE_RECOVERY_MODE=retry_pending python scripts/recover_stale_outbox.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
|
||||
from app.integration_outbox_service import recover_stale_processing_outbox, outbox_stale_minutes
|
||||
|
||||
|
||||
def main() -> int:
|
||||
mode = os.getenv("OUTBOX_STALE_RECOVERY_MODE", "manual_only")
|
||||
actor = os.getenv("OUTBOX_WORKER_ID", f"recover_stale_outbox:{os.getpid()}")
|
||||
limit = int(os.getenv("OUTBOX_STALE_RECOVERY_LIMIT", "100"))
|
||||
stale_minutes = outbox_stale_minutes()
|
||||
recovered = recover_stale_processing_outbox(
|
||||
mode=mode,
|
||||
actor=actor,
|
||||
limit=limit,
|
||||
stale_minutes=stale_minutes,
|
||||
)
|
||||
print(f"Mode: {mode}")
|
||||
print(f"Stale threshold minutes: {stale_minutes}")
|
||||
print(f"Recovered/exposed stale items: {len(recovered)}")
|
||||
for item in recovered:
|
||||
print(f"- {item.get('id')} {item.get('target_system')}.{item.get('action_type')} -> {item.get('status')}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
71
scripts/reopen_chatwoot_review_tasks.py
Executable file
71
scripts/reopen_chatwoot_review_tasks.py
Executable file
@@ -0,0 +1,71 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Reabre tasks Chatwoot que foram classificadas como revisão/remoção mas ficaram skipped.
|
||||
|
||||
Uso seguro:
|
||||
PYTHONPATH=. python scripts/reopen_chatwoot_review_tasks.py --dry-run
|
||||
PYTHONPATH=. python scripts/reopen_chatwoot_review_tasks.py --days 7
|
||||
|
||||
Por defeito só olha para os últimos 7 dias e não toca em spam/NO_ACTION.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--days", type=int, default=7, help="Janela de dias a corrigir")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Mostra o que faria sem alterar")
|
||||
args = parser.parse_args()
|
||||
|
||||
params = {"days": int(args.days)}
|
||||
select_sql = text("""
|
||||
SELECT id::text, created_at, action_code, route, action, status, conversation_id, contact_id
|
||||
FROM tasks
|
||||
WHERE source_system = 'chatwoot'
|
||||
AND status = 'skipped'
|
||||
AND action_code IN ('REVIEW_MANUALLY', 'REMOVE_FROM_LIST')
|
||||
AND created_at >= now() - (:days * interval '1 day')
|
||||
ORDER BY created_at DESC
|
||||
""")
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(select_sql, params).mappings().all()
|
||||
|
||||
print(f"Encontradas {len(rows)} task(s) Chatwoot a reabrir.")
|
||||
for row in rows[:50]:
|
||||
print(f"- {row['created_at']} {row['action_code']} conversa={row['conversation_id']} task={row['id']}")
|
||||
|
||||
if args.dry_run or not rows:
|
||||
print("Dry-run: nenhuma alteração aplicada." if args.dry_run else "Nada para alterar.")
|
||||
return 0
|
||||
|
||||
ids = [row["id"] for row in rows]
|
||||
update_sql = text("""
|
||||
UPDATE tasks
|
||||
SET status = 'pending',
|
||||
priority = CASE WHEN action_code = 'REVIEW_MANUALLY' THEN 'alta' ELSE COALESCE(priority, 'normal') END,
|
||||
route = CASE WHEN action_code = 'REMOVE_FROM_LIST' THEN 'marketing' ELSE route END,
|
||||
updated_at = now(),
|
||||
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:patch AS JSONB)
|
||||
WHERE id = ANY(CAST(:ids AS uuid[]))
|
||||
""")
|
||||
|
||||
patch = json.dumps({
|
||||
"v46_reopened": True,
|
||||
"v46_reason": "REVIEW_MANUALLY/REMOVE_FROM_LIST devem gerar trabalho humano pendente",
|
||||
}, ensure_ascii=False)
|
||||
|
||||
with engine.begin() as conn:
|
||||
conn.execute(update_sql, {"ids": ids, "patch": patch})
|
||||
|
||||
print(f"Reabertas {len(ids)} task(s) como pending.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
89
scripts/repair_odoo_sale_order_customer_names.py
Normal file
89
scripts/repair_odoo_sale_order_customer_names.py
Normal file
@@ -0,0 +1,89 @@
|
||||
"""Repair Odoo sale-order references accidentally stored as customer names.
|
||||
|
||||
v4.9.25.3 fixes the source of the issue. This script repairs rows already
|
||||
created by older v4.9.25.x builds where customers.name became S00xxx although
|
||||
metadata.raw_customer_payload.partner_name contains the real fiscal customer.
|
||||
|
||||
Dry-run by default. Use --apply to update rows.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
|
||||
def _partner_name_from_metadata(metadata: Any) -> str:
|
||||
if not isinstance(metadata, dict):
|
||||
return ""
|
||||
raw = metadata.get("raw_customer_payload")
|
||||
if not isinstance(raw, dict):
|
||||
return ""
|
||||
partner_name = str(raw.get("partner_name") or "").strip()
|
||||
if partner_name:
|
||||
return partner_name
|
||||
partner_id = raw.get("partner_id")
|
||||
if isinstance(partner_id, (list, tuple)) and len(partner_id) > 1:
|
||||
return str(partner_id[1] or "").strip()
|
||||
return ""
|
||||
|
||||
|
||||
def find_rows() -> List[Dict[str, Any]]:
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text("""
|
||||
SELECT id::text, name, tax_id, email, metadata
|
||||
FROM customers
|
||||
WHERE name ~ '^S[0-9]{4,}'
|
||||
AND metadata->>'source_system' = 'odoo'
|
||||
ORDER BY updated_at DESC
|
||||
""")).mappings().all()
|
||||
result: List[Dict[str, Any]] = []
|
||||
for row in rows:
|
||||
partner_name = _partner_name_from_metadata(row.get("metadata"))
|
||||
if partner_name and not partner_name.upper().startswith("S00"):
|
||||
data = dict(row)
|
||||
data["new_name"] = partner_name
|
||||
result.append(data)
|
||||
return result
|
||||
|
||||
|
||||
def apply(rows: List[Dict[str, Any]]) -> int:
|
||||
updated = 0
|
||||
with engine.begin() as conn:
|
||||
for row in rows:
|
||||
conn.execute(text("""
|
||||
UPDATE customers
|
||||
SET name = :new_name,
|
||||
metadata = COALESCE(metadata, '{}'::jsonb) || jsonb_build_object(
|
||||
'odoo_sale_order_name_repaired', true,
|
||||
'previous_customer_name', CAST(:old_name AS TEXT)
|
||||
),
|
||||
updated_at = now()
|
||||
WHERE id = CAST(:id AS UUID)
|
||||
"""), {"id": row["id"], "old_name": row["name"], "new_name": row["new_name"]})
|
||||
updated += 1
|
||||
return updated
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Repair Odoo S00xxx customer names.")
|
||||
parser.add_argument("--apply", action="store_true", help="Apply updates. Default is dry-run.")
|
||||
args = parser.parse_args()
|
||||
|
||||
rows = find_rows()
|
||||
print(f"Clientes Odoo com nome S00xxx reparáveis: {len(rows)}")
|
||||
for row in rows[:50]:
|
||||
print(f"- {row['name']} -> {row['new_name']} | NIF {row.get('tax_id') or '-'} | email {row.get('email') or '-'}")
|
||||
if len(rows) > 50:
|
||||
print(f"... mais {len(rows)-50}")
|
||||
if not args.apply:
|
||||
print("Dry-run. Para aplicar: repetir com --apply")
|
||||
return
|
||||
print(f"Atualizados: {apply(rows)}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
44
scripts/replace_jasmin_document_for_opportunity.py
Executable file
44
scripts/replace_jasmin_document_for_opportunity.py
Executable file
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Replace imported Jasmin quotation/proforma details for an opportunity.
|
||||
|
||||
Use this when an old/closed quotation was linked by mistake and a newer open
|
||||
reconciliation item should become the active document for the opportunity.
|
||||
This deletes only ClientFlow imported Jasmin quotation/proforma artifacts; it does
|
||||
not delete documents in Jasmin.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
from decimal import Decimal
|
||||
from typing import Any
|
||||
|
||||
|
||||
def _json_default(value: Any) -> str:
|
||||
if isinstance(value, Decimal):
|
||||
return str(value)
|
||||
return str(value)
|
||||
|
||||
|
||||
async def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--opportunity-id", required=True)
|
||||
parser.add_argument("--item-id", required=True, help="reconciliation_items.id for the open/valid Jasmin candidate")
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
from app.jasmin_backfill_service import replace_jasmin_document_for_opportunity_async
|
||||
|
||||
result = await replace_jasmin_document_for_opportunity_async(
|
||||
opportunity_id=args.opportunity_id,
|
||||
item_id=args.item_id,
|
||||
actor="operator_cli_replace_jasmin_document",
|
||||
dry_run=args.dry_run,
|
||||
)
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=_json_default))
|
||||
return 0 if result.get("ok") else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(asyncio.run(main()))
|
||||
133
scripts/reset_reconciliation_generated.py
Executable file
133
scripts/reset_reconciliation_generated.py
Executable file
@@ -0,0 +1,133 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Reset generated reconciliation staging items so they can be rebuilt.
|
||||
|
||||
Safe by default: dry-run only and only affects generated external candidates
|
||||
from Jasmin/Odoo/Packlink in open/needs_review/ignored states. It does not
|
||||
remove opportunities, commercial documents, payment proofs, operation links or
|
||||
external system records.
|
||||
|
||||
Examples:
|
||||
PYTHONPATH=. python scripts/reset_reconciliation_generated.py
|
||||
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --apply
|
||||
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --source jasmin --source odoo --apply
|
||||
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --days 30 --apply
|
||||
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --include-manual --apply
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine, init_db
|
||||
from app.reconciliation_service import ensure_reconciliation_schema
|
||||
|
||||
DEFAULT_SOURCES = ["jasmin", "odoo", "packlink"]
|
||||
DEFAULT_STATUSES = ["open", "needs_review", "ignored"]
|
||||
|
||||
|
||||
def _backup_table_name() -> str:
|
||||
return "reconciliation_items_reset_backup_" + datetime.now(timezone.utc).strftime("%Y%m%d_%H%M%S")
|
||||
|
||||
|
||||
def _build_where(args: argparse.Namespace) -> tuple[str, Dict[str, Any]]:
|
||||
sources = list(args.source or DEFAULT_SOURCES)
|
||||
if args.include_manual and "manual" not in sources:
|
||||
sources.append("manual")
|
||||
statuses = list(args.status or DEFAULT_STATUSES)
|
||||
params: Dict[str, Any] = {"sources": sources, "statuses": statuses}
|
||||
clauses = [
|
||||
"source_system = ANY(CAST(:sources AS TEXT[]))",
|
||||
"status = ANY(CAST(:statuses AS TEXT[]))",
|
||||
]
|
||||
# Extra guard: never touch items already linked to an opportunity unless
|
||||
# the operator explicitly changes the status list and unlinks manually.
|
||||
clauses.append("opportunity_id IS NULL")
|
||||
|
||||
if args.days is not None:
|
||||
days = max(int(args.days), 1)
|
||||
cutoff = (datetime.now(timezone.utc).date() - timedelta(days=days - 1)).isoformat()
|
||||
params["cutoff"] = cutoff
|
||||
clauses.append("(document_date IS NULL OR document_date >= CAST(:cutoff AS DATE))")
|
||||
return " AND ".join(clauses), params
|
||||
|
||||
|
||||
def _summarize(where_sql: str, params: Dict[str, Any], limit: int) -> Dict[str, Any]:
|
||||
with engine.begin() as conn:
|
||||
counts = conn.execute(text(f"""
|
||||
SELECT source_system, external_type, status, COUNT(*) AS total
|
||||
FROM reconciliation_items
|
||||
WHERE {where_sql}
|
||||
GROUP BY source_system, external_type, status
|
||||
ORDER BY source_system, external_type, status
|
||||
"""), params).mappings().all()
|
||||
rows = conn.execute(text(f"""
|
||||
SELECT id::text, source_system, external_type, status, document_number,
|
||||
customer_name, customer_tax_id, document_date, amount, title
|
||||
FROM reconciliation_items
|
||||
WHERE {where_sql}
|
||||
ORDER BY updated_at DESC, created_at DESC
|
||||
LIMIT :limit
|
||||
"""), {**params, "limit": int(limit)}).mappings().all()
|
||||
return {"counts": [dict(r) for r in counts], "items": [dict(r) for r in rows]}
|
||||
|
||||
|
||||
def _apply_reset(where_sql: str, params: Dict[str, Any]) -> Dict[str, Any]:
|
||||
backup_table = _backup_table_name()
|
||||
# backup_table is generated internally from digits/underscore only.
|
||||
with engine.begin() as conn:
|
||||
total = conn.execute(text(f"SELECT COUNT(*) FROM reconciliation_items WHERE {where_sql}"), params).scalar() or 0
|
||||
conn.execute(text(f"CREATE TABLE {backup_table} AS SELECT * FROM reconciliation_items WHERE {where_sql}"), params)
|
||||
deleted = conn.execute(text(f"DELETE FROM reconciliation_items WHERE {where_sql}"), params).rowcount or 0
|
||||
return {"matched": int(total), "deleted": int(deleted), "backup_table": backup_table}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Reset generated reconciliation candidates and keep a DB backup table.")
|
||||
parser.add_argument("--source", action="append", choices=["jasmin", "odoo", "packlink", "manual"], help="source_system to reset; repeatable. Default: jasmin, odoo, packlink")
|
||||
parser.add_argument("--status", action="append", choices=["open", "needs_review", "ignored"], help="status to reset; repeatable. Default: open, needs_review, ignored")
|
||||
parser.add_argument("--days", type=int, help="only reset candidates inside the last N days; default is all dates")
|
||||
parser.add_argument("--include-manual", action="store_true", help="also include source_system=manual; use with care")
|
||||
parser.add_argument("--limit", type=int, default=50, help="preview sample size")
|
||||
parser.add_argument("--apply", action="store_true", help="delete matched staging rows after creating a backup table")
|
||||
args = parser.parse_args()
|
||||
|
||||
init_db()
|
||||
ensure_reconciliation_schema()
|
||||
where_sql, params = _build_where(args)
|
||||
summary = _summarize(where_sql, params, args.limit)
|
||||
|
||||
print("Alvo do reset:")
|
||||
print(f" fontes: {', '.join(params['sources'])}")
|
||||
print(f" estados: {', '.join(params['statuses'])}")
|
||||
print(" proteção: opportunity_id IS NULL")
|
||||
if args.days is not None:
|
||||
print(f" janela: >= {params['cutoff']} ({args.days} dias)")
|
||||
print("\nContagens:")
|
||||
if not summary["counts"]:
|
||||
print(" 0 itens encontrados")
|
||||
for row in summary["counts"]:
|
||||
print(f" {row['source_system']} · {row['external_type']} · {row['status']}: {row['total']}")
|
||||
|
||||
print("\nAmostra:")
|
||||
for row in summary["items"]:
|
||||
print(
|
||||
f" {row.get('document_date') or '-'} · {row.get('source_system')} · {row.get('external_type')} · "
|
||||
f"{row.get('status')} · {row.get('document_number') or '-'} · {row.get('customer_name') or '-'}"
|
||||
)
|
||||
|
||||
if not args.apply:
|
||||
print("\nDry-run. Para aplicar: repetir com --apply")
|
||||
return
|
||||
|
||||
result = _apply_reset(where_sql, params)
|
||||
print("\nReset aplicado:")
|
||||
print(f" encontrados: {result['matched']}")
|
||||
print(f" apagados: {result['deleted']}")
|
||||
print(f" backup: {result['backup_table']}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
89
scripts/review_domain_auto_applied_suggestions.py
Executable file
89
scripts/review_domain_auto_applied_suggestions.py
Executable file
@@ -0,0 +1,89 @@
|
||||
#!/usr/bin/env python3
|
||||
"""List/revert risky fiscal suggestions auto-applied from domain-only matches."""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine
|
||||
|
||||
DOMAIN_MATCHES = (
|
||||
"email_principal_dominio",
|
||||
"email_dominio_empresa_associada",
|
||||
"contacto_email_dominio",
|
||||
"dominio",
|
||||
"email_dominio",
|
||||
"website_dominio",
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--apply", action="store_true", help="revert accepted domain-only suggestions and unlink matching opportunity customer")
|
||||
args = parser.parse_args()
|
||||
|
||||
with engine.begin() as conn:
|
||||
rows = conn.execute(text("""
|
||||
SELECT
|
||||
s.id::text,
|
||||
s.opportunity_id::text,
|
||||
s.suggested_customer_id::text,
|
||||
s.suggested_name,
|
||||
s.suggested_nif,
|
||||
s.match_type,
|
||||
s.confidence,
|
||||
o.customer_name,
|
||||
o.customer_email,
|
||||
o.local_customer_id::text AS current_customer_id
|
||||
FROM fiscal_customer_suggestions s
|
||||
JOIN opportunities o ON o.id = s.opportunity_id
|
||||
WHERE s.status = 'accepted'
|
||||
AND s.auto_applied = TRUE
|
||||
AND s.match_type = ANY(:matches)
|
||||
ORDER BY s.updated_at DESC
|
||||
"""), {"matches": list(DOMAIN_MATCHES)}).mappings().all()
|
||||
|
||||
print(f"Sugestões por domínio auto-aplicadas: {len(rows)}")
|
||||
for r in rows:
|
||||
print(f"- {r['customer_name']} <{r['customer_email']}> -> {r['suggested_name']} / {r['suggested_nif']} | {r['match_type']} | {r['confidence']} | opp={r['opportunity_id']}")
|
||||
|
||||
if not args.apply:
|
||||
print("Dry-run. Para reverter: repetir com --apply")
|
||||
return
|
||||
|
||||
updated = 0
|
||||
with engine.begin() as conn:
|
||||
for r in rows:
|
||||
# Only unlink if the opportunity is still linked to the same customer suggested by this risky suggestion.
|
||||
if r["suggested_customer_id"] and r["current_customer_id"] == r["suggested_customer_id"]:
|
||||
conn.execute(text("""
|
||||
UPDATE opportunities
|
||||
SET local_customer_id = NULL,
|
||||
metadata = COALESCE(metadata, '{}'::jsonb) || jsonb_build_object(
|
||||
'domain_match_auto_apply_reverted', true,
|
||||
'domain_match_reverted_suggestion_id', :suggestion_id,
|
||||
'domain_match_reverted_customer_name', :customer_name,
|
||||
'domain_match_reverted_at', now()
|
||||
),
|
||||
updated_at = now()
|
||||
WHERE id = CAST(:opportunity_id AS UUID)
|
||||
"""), {
|
||||
"opportunity_id": r["opportunity_id"],
|
||||
"suggestion_id": r["id"],
|
||||
"customer_name": r["suggested_name"],
|
||||
})
|
||||
conn.execute(text("""
|
||||
UPDATE fiscal_customer_suggestions
|
||||
SET status = 'rejected', auto_applied = FALSE, resolved_by = 'domain_match_safety_review',
|
||||
resolved_at = now(), reason = COALESCE(reason, '') || ' | reverted: domain-only auto-apply is unsafe',
|
||||
updated_at = now()
|
||||
WHERE id = CAST(:id AS UUID)
|
||||
"""), {"id": r["id"]})
|
||||
updated += 1
|
||||
print(f"Revertidas: {updated}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
4
scripts/run_dev.sh
Executable file
4
scripts/run_dev.sh
Executable file
@@ -0,0 +1,4 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
python -m uvicorn app.main:app --reload --host 127.0.0.1 --port 8000
|
||||
48
scripts/run_reconciliation_pipeline.py
Executable file
48
scripts/run_reconciliation_pipeline.py
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Run the recommended ClientFlow pipeline: enrichment -> external sync.
|
||||
|
||||
This orchestrator is safe to run periodically. It does not apply low-confidence
|
||||
reconciliation decisions; it prepares fiscal identities first and then rebuilds
|
||||
external candidates for the operator.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
if str(ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(ROOT))
|
||||
|
||||
from app.fiscal_enrichment_service import enrich_open_opportunities, ensure_fiscal_enrichment_schema
|
||||
from app.external_reconciliation_sync import sync_all_external_reconciliation_candidates
|
||||
|
||||
|
||||
async def _run(args: argparse.Namespace) -> dict:
|
||||
ensure_fiscal_enrichment_schema()
|
||||
enrichment = enrich_open_opportunities(
|
||||
limit=args.enrichment_limit,
|
||||
apply_safe=not args.no_auto_apply,
|
||||
mode="pipeline",
|
||||
)
|
||||
reconciliation = await sync_all_external_reconciliation_candidates(limit=args.limit, days=args.days)
|
||||
return {"enrichment": enrichment, "reconciliation": reconciliation}
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Correr pipeline ClientFlow: enriquecimento fiscal + reconciliação")
|
||||
parser.add_argument("--days", type=int, default=7, help="Janela de reconciliação externa")
|
||||
parser.add_argument("--limit", type=int, default=100, help="Limite por fonte externa")
|
||||
parser.add_argument("--enrichment-limit", type=int, default=100, help="Limite de oportunidades a enriquecer antes da reconciliação")
|
||||
parser.add_argument("--no-auto-apply", action="store_true", help="Não auto-associar sugestões fiscais fortes")
|
||||
args = parser.parse_args()
|
||||
result = asyncio.run(_run(args))
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
77
scripts/sync_external_reconciliation.py
Executable file
77
scripts/sync_external_reconciliation.py
Executable file
@@ -0,0 +1,77 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sync external systems into reconciliation candidates.
|
||||
|
||||
Examples:
|
||||
PYTHONPATH=. python scripts/sync_external_reconciliation.py --all
|
||||
PYTHONPATH=. python scripts/sync_external_reconciliation.py --jasmin --limit 50 --days 3
|
||||
PYTHONPATH=. python scripts/sync_external_reconciliation.py --odoo --days 3
|
||||
PYTHONPATH=. python scripts/sync_external_reconciliation.py --customers --limit 200
|
||||
|
||||
The full --all pipeline first creates/updates fiscal customers from Jasmin/Odoo
|
||||
and then stages reconciliation_items. It never creates opportunities or confirms
|
||||
payments, so the operator can link/create/ignore in the UI.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import asyncio
|
||||
import json
|
||||
from typing import Any, Dict, List
|
||||
|
||||
from app.external_reconciliation_sync import (
|
||||
sync_all_external_reconciliation_candidates,
|
||||
sync_external_fiscal_customers_for_reconciliation,
|
||||
sync_jasmin_reconciliation_candidates,
|
||||
sync_odoo_reconciliation_candidates,
|
||||
sync_packlink_reconciliation_candidates,
|
||||
)
|
||||
|
||||
|
||||
def _print(result: Dict[str, Any]) -> None:
|
||||
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
|
||||
|
||||
|
||||
async def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Sync external APIs into ClientFlow reconciliation candidates.")
|
||||
parser.add_argument("--all", action="store_true", help="sync all enabled external systems")
|
||||
parser.add_argument("--jasmin", action="store_true", help="sync Jasmin quotations/invoices")
|
||||
parser.add_argument("--odoo", action="store_true", help="sync Odoo sale orders")
|
||||
parser.add_argument("--packlink", action="store_true", help="sync Packlink shipments")
|
||||
parser.add_argument("--customers", action="store_true", help="seed fiscal customers from Jasmin/Odoo before document reconciliation")
|
||||
parser.add_argument("--limit", type=int, default=50, help="maximum records per source")
|
||||
parser.add_argument("--days", type=int, default=3, help="lookback window for Jasmin/Odoo/Packlink syncs")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.all or not (args.jasmin or args.odoo or args.packlink or args.customers):
|
||||
_print(await sync_all_external_reconciliation_candidates(limit=args.limit, days=args.days))
|
||||
return
|
||||
|
||||
if args.customers and not (args.jasmin or args.odoo or args.packlink):
|
||||
_print(await sync_external_fiscal_customers_for_reconciliation(limit=max(args.limit, 200)))
|
||||
return
|
||||
|
||||
results: List[Dict[str, Any]] = []
|
||||
customer_result = None
|
||||
if args.customers:
|
||||
customer_result = await sync_external_fiscal_customers_for_reconciliation(limit=max(args.limit, 200))
|
||||
if args.jasmin:
|
||||
results.append(await sync_jasmin_reconciliation_candidates(limit=args.limit, days=args.days))
|
||||
if args.odoo:
|
||||
# Odoo XML-RPC client is sync; the service itself stays sync for easier reuse.
|
||||
results.append(sync_odoo_reconciliation_candidates(limit=args.limit, days=args.days))
|
||||
if args.packlink:
|
||||
results.append(await sync_packlink_reconciliation_candidates(limit=args.limit, days=args.days))
|
||||
output = {
|
||||
"seen": sum(int(r.get("seen") or 0) for r in results),
|
||||
"created_or_updated": sum(int(r.get("created_or_updated") or 0) for r in results),
|
||||
"results": results,
|
||||
}
|
||||
if customer_result is not None:
|
||||
output["customer_seen"] = customer_result.get("seen", 0)
|
||||
output["customers_created_or_updated"] = customer_result.get("created_or_updated", 0)
|
||||
output["customer_results"] = customer_result.get("results", [])
|
||||
_print(output)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
25
scripts/sync_reconciliation_candidates.py
Executable file
25
scripts/sync_reconciliation_candidates.py
Executable file
@@ -0,0 +1,25 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Create reconciliation candidates from local external records.
|
||||
|
||||
This script is safe to run from cron/systemd timer. It is idempotent and only
|
||||
creates/updates reconciliation_items for information already known locally, such
|
||||
as Jasmin commercial documents without opportunity_id.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from app.db import init_db
|
||||
from app.reconciliation_service import sync_local_documents_without_opportunity
|
||||
|
||||
|
||||
def main() -> None:
|
||||
init_db()
|
||||
result = sync_local_documents_without_opportunity(limit=500)
|
||||
print(
|
||||
"Reconciliation sync finished: "
|
||||
f"seen={result.get('seen', 0)} "
|
||||
f"created_or_updated={result.get('created_or_updated', 0)}"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
95
scripts/test_action_core_http.py
Normal file
95
scripts/test_action_core_http.py
Normal file
@@ -0,0 +1,95 @@
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
BASE_URL = os.getenv("CLIENTFLOW_BASE_URL", "http://127.0.0.1:8000")
|
||||
|
||||
|
||||
CASES = [
|
||||
{
|
||||
"conversation_id": "case-send-invoice",
|
||||
"message": "Recebemos o equipamento. Agradecemos o envio da factura.",
|
||||
},
|
||||
{
|
||||
"conversation_id": "case-payment",
|
||||
"message": "Segue comprovativo de pagamento em anexo.",
|
||||
},
|
||||
{
|
||||
"conversation_id": "case-info",
|
||||
"message": "Bom dia, podem enviar mais informações sobre carregadores monofásicos?",
|
||||
},
|
||||
{
|
||||
"conversation_id": "case-shipment",
|
||||
"message": "Boa tarde, gostava de saber se já enviaram o carregador.",
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
out_dir = Path("resultados-action-core")
|
||||
out_dir.mkdir(exist_ok=True)
|
||||
|
||||
rows = []
|
||||
|
||||
for case in CASES:
|
||||
payload = {
|
||||
"last_customer_message": case["message"],
|
||||
"previous_context": "Teste Action Core.",
|
||||
"source": "manual_test",
|
||||
"conversation_id": case["conversation_id"],
|
||||
"contact_id": "test-contact",
|
||||
}
|
||||
|
||||
response = requests.post(
|
||||
f"{BASE_URL}/analyze",
|
||||
headers={"Content-Type": "application/json"},
|
||||
json=payload,
|
||||
timeout=90,
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
action_decision = data.get("action_decision") or {}
|
||||
action_result = data.get("action_result") or {}
|
||||
|
||||
row = {
|
||||
"conversation_id": case["conversation_id"],
|
||||
"action_code": action_decision.get("action_code"),
|
||||
"route": action_result.get("route"),
|
||||
"action": action_result.get("action"),
|
||||
"safe_to_post": action_result.get("safe_to_post"),
|
||||
"task_id": data.get("task_id"),
|
||||
"needs_review": data.get("needs_review"),
|
||||
}
|
||||
|
||||
rows.append(row)
|
||||
|
||||
print("=" * 80)
|
||||
print(case["conversation_id"])
|
||||
print("action_code:", row["action_code"])
|
||||
print("route:", row["route"])
|
||||
print("action:", row["action"])
|
||||
print("safe_to_post:", row["safe_to_post"])
|
||||
print("task_id:", row["task_id"])
|
||||
|
||||
(out_dir / f"{case['conversation_id']}.json").write_text(
|
||||
json.dumps(data, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
(out_dir / "summary.json").write_text(
|
||||
json.dumps(rows, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
print("\nResumo:", out_dir / "summary.json")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
51
scripts/test_jasmin_connection.py
Executable file
51
scripts/test_jasmin_connection.py
Executable file
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Teste não destrutivo da integração Jasmin.
|
||||
|
||||
Não cria clientes/documentos. Valida OAuth, versão, OData de clientes,
|
||||
produtos, orçamentos e faturas.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
|
||||
__test__ = False
|
||||
|
||||
|
||||
def show(title: str, value, limit: int = 2500) -> None:
|
||||
print(f"\n=== {title} ===")
|
||||
try:
|
||||
text = json.dumps(value, indent=2, ensure_ascii=False, default=str)
|
||||
except Exception:
|
||||
text = str(value)
|
||||
print(text[:limit])
|
||||
|
||||
|
||||
async def main() -> int:
|
||||
from app.jasmin_client import JasminClient, JasminError
|
||||
|
||||
client = JasminClient()
|
||||
try:
|
||||
token = await client.get_token()
|
||||
print(f"OAuth OK: token_length={len(token)}")
|
||||
show("Versões", await client.get_versions())
|
||||
show("Clientes OData", await client.list_customers_odata(top=5))
|
||||
show("Produtos OData", await client.list_sales_items(top=10))
|
||||
show("Últimos orçamentos", await client.list_quotations(top=5))
|
||||
show("Últimas faturas", await client.list_invoices(top=5))
|
||||
except JasminError as exc:
|
||||
print(f"ERRO Jasmin: {exc}")
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(asyncio.run(main()))
|
||||
15
scripts/test_operational_flow.py
Normal file
15
scripts/test_operational_flow.py
Normal file
@@ -0,0 +1,15 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Smoke placeholder for the clean ClientFlow operational architecture.
|
||||
|
||||
Operational progression is represented by business_events and operation_links,
|
||||
not by extra triage action_codes.
|
||||
"""
|
||||
|
||||
|
||||
def main() -> int:
|
||||
print("Use API/UI smoke tests against current action codes and operation_links.")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
101
scripts/test_packlink_connection.py
Executable file
101
scripts/test_packlink_connection.py
Executable file
@@ -0,0 +1,101 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Smoke test Packlink PRO.
|
||||
|
||||
Uso:
|
||||
export PACKLINK_API_KEY=...
|
||||
export PACKLINK_BASE_URL=https://api.packlink.com/v1
|
||||
python scripts/test_packlink_connection.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(PROJECT_ROOT))
|
||||
os.chdir(PROJECT_ROOT)
|
||||
|
||||
# Permite que o ficheiro seja importado por pytest sem exigir .env real.
|
||||
os.environ.setdefault("OPENROUTER_API_KEY", "dummy")
|
||||
os.environ.setdefault("DATABASE_URL", "postgresql+psycopg://clientflow:password@127.0.0.1:5432/clientflow")
|
||||
|
||||
from app.packlink_client import PacklinkClient
|
||||
|
||||
|
||||
def normalize_packlink_zip(country: str, zip_code: str, *, for_quote: bool = True) -> str:
|
||||
country = str(country or "").upper().strip()
|
||||
zip_code = str(zip_code or "").strip()
|
||||
if country == "PT" and for_quote:
|
||||
import re
|
||||
match = re.search(r"\d{4}", zip_code)
|
||||
if match:
|
||||
return match.group(0)
|
||||
return zip_code
|
||||
|
||||
|
||||
def default_package() -> dict:
|
||||
return {
|
||||
"height": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_HEIGHT", "10"))),
|
||||
"width": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_WIDTH", "20"))),
|
||||
"length": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_LENGTH", "30"))),
|
||||
"weight": float(os.getenv("PACKLINK_DEFAULT_PACKAGE_WEIGHT", "2")),
|
||||
}
|
||||
|
||||
|
||||
def dump(title: str, value) -> None:
|
||||
print(f"\n=== {title} ===")
|
||||
print(json.dumps(value, ensure_ascii=False, indent=2, default=str)[:5000])
|
||||
|
||||
|
||||
async def main() -> int:
|
||||
if not os.getenv("PACKLINK_API_KEY"):
|
||||
print("ERRO: PACKLINK_API_KEY em falta")
|
||||
return 2
|
||||
|
||||
client = PacklinkClient()
|
||||
account = await client.get_client()
|
||||
dump("Conta", account)
|
||||
|
||||
warehouses = await client.get_warehouses()
|
||||
dump("Armazéns", warehouses[:3])
|
||||
|
||||
parcels = await client.get_parcels()
|
||||
dump("Volumes", parcels[:3])
|
||||
|
||||
from_zip = normalize_packlink_zip("PT", os.getenv("PACKLINK_TEST_FROM_ZIP", "3650-219"), for_quote=True)
|
||||
to_zip = normalize_packlink_zip("PT", os.getenv("PACKLINK_TEST_TO_ZIP", "4000-001"), for_quote=True)
|
||||
services = await client.quote_services(
|
||||
from_country="PT",
|
||||
from_zip=from_zip,
|
||||
to_country="PT",
|
||||
to_zip=to_zip,
|
||||
source=os.getenv("PACKLINK_SOURCE", "PRO"),
|
||||
packages=[default_package()],
|
||||
)
|
||||
|
||||
simple = [
|
||||
{
|
||||
"id": s.get("id"),
|
||||
"carrier": s.get("carrier_name"),
|
||||
"service": s.get("name"),
|
||||
"price": (s.get("price") or {}).get("total_price") or s.get("base_price"),
|
||||
"currency": s.get("currency") or (s.get("price") or {}).get("currency"),
|
||||
"dropoff": s.get("dropoff"),
|
||||
"parcelshop": s.get("delivery_to_parcelshop"),
|
||||
}
|
||||
for s in services
|
||||
]
|
||||
dump("Serviços", simple)
|
||||
|
||||
service_id = os.getenv("PACKLINK_DEFAULT_SERVICE_ID", "20571")
|
||||
details = await client.get_service_details(service_id)
|
||||
dump(f"Detalhes serviço {service_id}", details)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(asyncio.run(main()))
|
||||
5
scripts/test_sample.sh
Executable file
5
scripts/test_sample.sh
Executable file
@@ -0,0 +1,5 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
curl -sS -X POST http://127.0.0.1:8000/analyze \
|
||||
-H "Content-Type: application/json" \
|
||||
-d @tests/sample_cases/sample_analyze.json | jq .
|
||||
48
scripts/validate_v45_operational_core.py
Executable file
48
scripts/validate_v45_operational_core.py
Executable file
@@ -0,0 +1,48 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Lightweight v4.5 validation checks for a deployed ClientFlow backend."""
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from app.db import engine, init_db
|
||||
|
||||
|
||||
REQUIRED_TABLES = ["communications", "timeline_events", "tasks", "integration_outbox", "opportunities"]
|
||||
REQUIRED_TASK_COLUMNS = ["communication_id", "document_id", "shipment_id", "outbox_id", "priority", "assigned_to"]
|
||||
|
||||
|
||||
def main() -> int:
|
||||
init_db()
|
||||
with engine.begin() as conn:
|
||||
tables = {r[0] for r in conn.execute(text("""
|
||||
SELECT table_name
|
||||
FROM information_schema.tables
|
||||
WHERE table_schema = 'public'
|
||||
"""))}
|
||||
missing_tables = [t for t in REQUIRED_TABLES if t not in tables]
|
||||
|
||||
cols = {r[0] for r in conn.execute(text("""
|
||||
SELECT column_name
|
||||
FROM information_schema.columns
|
||||
WHERE table_schema = 'public' AND table_name = 'tasks'
|
||||
"""))}
|
||||
missing_cols = [c for c in REQUIRED_TASK_COLUMNS if c not in cols]
|
||||
|
||||
comm_count = conn.execute(text("SELECT COUNT(*) FROM communications")).scalar()
|
||||
timeline_count = conn.execute(text("SELECT COUNT(*) FROM timeline_events")).scalar()
|
||||
|
||||
if missing_tables or missing_cols:
|
||||
print("ClientFlow v4.5 validation failed")
|
||||
print("Missing tables:", ", ".join(missing_tables) or "none")
|
||||
print("Missing task columns:", ", ".join(missing_cols) or "none")
|
||||
return 1
|
||||
|
||||
print("ClientFlow v4.5 validation OK")
|
||||
print(f"communications={comm_count} timeline_events={timeline_count}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
Reference in New Issue
Block a user