Release v4928.1.4.2 stable

This commit is contained in:
2026-06-09 22:55:58 +01:00
commit 6445044ac6
280 changed files with 41775 additions and 0 deletions

86
scripts/apply_migrations.py Executable file
View File

@@ -0,0 +1,86 @@
#!/usr/bin/env python3
"""Apply SQL migrations stored in ./migrations.
Usage:
python scripts/apply_migrations.py
python scripts/apply_migrations.py --dry-run
"""
from __future__ import annotations
import argparse
from pathlib import Path
import os
import sys
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
from sqlalchemy import text
from app.db import engine
ROOT = Path(__file__).resolve().parents[1]
MIGRATIONS_DIR = ROOT / "migrations"
def ensure_ledger(conn) -> None:
conn.execute(text("""
CREATE TABLE IF NOT EXISTS schema_migrations (
version TEXT PRIMARY KEY,
name TEXT NOT NULL,
applied_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
"""))
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--dry-run", action="store_true", help="List pending migrations without applying them.")
args = parser.parse_args()
files = sorted(MIGRATIONS_DIR.glob("*.sql"))
if not files:
print("No migrations found.")
return 0
with engine.begin() as conn:
ensure_ledger(conn)
applied = {
row[0]
for row in conn.execute(text("SELECT version FROM schema_migrations"))
}
pending = []
for path in files:
version = path.stem.split("_", 1)[0]
if version not in applied:
pending.append((version, path))
if not pending:
print("No pending migrations.")
return 0
print("Pending migrations:")
for version, path in pending:
print(f"- {version}: {path.name}")
if args.dry_run:
return 0
for version, path in pending:
sql = path.read_text()
print(f"Applying {path.name}...")
conn.execute(text(sql))
conn.execute(text("""
INSERT INTO schema_migrations(version, name)
VALUES (:version, :name)
ON CONFLICT (version) DO NOTHING
"""), {"version": version, "name": path.name})
print("Migrations applied.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,338 @@
#!/usr/bin/env python3
import hashlib
import hmac
import json
import os
import time
import urllib.error
import urllib.parse
import urllib.request
from datetime import datetime, timedelta, timezone
from typing import Any, Dict, List, Optional
def env(name: str, default: str = "") -> str:
return os.getenv(name, default).strip()
CHATWOOT_BASE_URL = env("CHATWOOT_BASE_URL").rstrip("/")
CHATWOOT_ACCOUNT_ID = env("CHATWOOT_ACCOUNT_ID")
CHATWOOT_API_TOKEN = env("CHATWOOT_API_TOKEN")
CLIENTFLOW_WEBHOOK_SECRET = env("CLIENTFLOW_WEBHOOK_SECRET")
CLIENTFLOW_WEBHOOK_URL = env("CLIENTFLOW_WEBHOOK_URL", "http://127.0.0.1:8020/webhooks/chatwoot")
BACKFILL_DAYS = int(env("BACKFILL_DAYS", "5"))
BACKFILL_TO_CLIENTFLOW = env("BACKFILL_TO_CLIENTFLOW", "false").lower() == "true"
BACKFILL_STATUSES = [s.strip() for s in env("BACKFILL_STATUSES", "open,pending").split(",") if s.strip()]
BACKFILL_MAX_PAGES = int(env("BACKFILL_MAX_PAGES", "20"))
def request_json(method: str, url: str, body: Optional[Dict[str, Any]] = None, headers: Optional[Dict[str, str]] = None) -> Dict[str, Any]:
data = None
final_headers = headers.copy() if headers else {}
if body is not None:
data = json.dumps(body, ensure_ascii=False).encode("utf-8")
final_headers["Content-Type"] = "application/json"
req = urllib.request.Request(url, data=data, headers=final_headers, method=method)
try:
with urllib.request.urlopen(req, timeout=45) as resp:
raw = resp.read().decode("utf-8", errors="replace")
return {
"ok": 200 <= resp.status < 300,
"status": resp.status,
"json": json.loads(raw) if raw else {},
"raw": raw,
}
except urllib.error.HTTPError as e:
raw = e.read().decode("utf-8", errors="replace")
return {
"ok": False,
"status": e.code,
"json": None,
"raw": raw,
}
def chatwoot_headers() -> Dict[str, str]:
return {
"api_access_token": CHATWOOT_API_TOKEN,
"Accept": "application/json",
}
def payload_list(data: Any) -> List[Dict[str, Any]]:
if isinstance(data, list):
return data
if not isinstance(data, dict):
return []
candidates = [
data.get("payload"),
data.get("data", {}).get("payload") if isinstance(data.get("data"), dict) else None,
data.get("data"),
data.get("messages"),
]
for item in candidates:
if isinstance(item, list):
return item
return []
def ts_to_datetime(value: Any) -> Optional[datetime]:
if value is None:
return None
try:
if isinstance(value, (int, float)):
return datetime.fromtimestamp(float(value), tz=timezone.utc)
s = str(value).replace("Z", "+00:00")
return datetime.fromisoformat(s).astimezone(timezone.utc)
except Exception:
return None
def conversation_last_activity(conversation: Dict[str, Any]) -> Optional[datetime]:
for key in ["last_activity_at", "updated_at", "created_at"]:
dt = ts_to_datetime(conversation.get(key))
if dt:
return dt
return None
def fetch_conversations(status: str) -> List[Dict[str, Any]]:
all_items: List[Dict[str, Any]] = []
for page in range(1, BACKFILL_MAX_PAGES + 1):
query = urllib.parse.urlencode({
"status": status,
"page": page,
})
url = f"{CHATWOOT_BASE_URL}/api/v1/accounts/{CHATWOOT_ACCOUNT_ID}/conversations?{query}"
result = request_json("GET", url, headers=chatwoot_headers())
if not result["ok"]:
print(f"ERRO Chatwoot conversations status={status} page={page}: {result['status']} {result['raw'][:300]}")
break
items = payload_list(result["json"])
if not items:
break
all_items.extend(items)
if len(items) < 10:
break
return all_items
def fetch_messages(conversation_id: str) -> List[Dict[str, Any]]:
url = f"{CHATWOOT_BASE_URL}/api/v1/accounts/{CHATWOOT_ACCOUNT_ID}/conversations/{conversation_id}/messages"
result = request_json("GET", url, headers=chatwoot_headers())
if not result["ok"]:
print(f"ERRO Chatwoot messages conversation={conversation_id}: {result['status']} {result['raw'][:300]}")
return []
return payload_list(result["json"])
def is_incoming(message: Dict[str, Any]) -> bool:
mt = message.get("message_type")
return mt == "incoming" or mt == 0 or str(mt).lower() == "incoming"
def message_created_at(message: Dict[str, Any]) -> datetime:
return ts_to_datetime(message.get("created_at")) or datetime.fromtimestamp(0, tz=timezone.utc)
def sender_from_conversation(conversation: Dict[str, Any], message: Dict[str, Any]) -> Dict[str, Any]:
sender = {}
meta = conversation.get("meta") or {}
if isinstance(meta, dict) and isinstance(meta.get("sender"), dict):
sender.update(meta.get("sender") or {})
if isinstance(conversation.get("contact"), dict):
sender.update({k: v for k, v in conversation["contact"].items() if v is not None})
if isinstance(message.get("sender"), dict):
sender.update({k: v for k, v in message["sender"].items() if v is not None})
return sender
def sign_body(body_raw: str) -> Dict[str, str]:
ts = str(int(time.time()))
msg = ts.encode("utf-8") + b"." + body_raw.encode("utf-8")
sig = "sha256=" + hmac.new(
CLIENTFLOW_WEBHOOK_SECRET.encode("utf-8"),
msg,
hashlib.sha256,
).hexdigest()
return {
"Content-Type": "application/json",
"X-Chatwoot-Timestamp": ts,
"X-Chatwoot-Signature": sig,
}
def post_to_clientflow(payload: Dict[str, Any]) -> Dict[str, Any]:
# Importante: assinar e enviar exatamente o mesmo body_raw.
# Se o JSON for reformatado depois da assinatura, o webhook rejeita com 401.
body_raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":"))
headers = sign_body(body_raw)
req = urllib.request.Request(
CLIENTFLOW_WEBHOOK_URL,
data=body_raw.encode("utf-8"),
headers=headers,
method="POST",
)
try:
with urllib.request.urlopen(req, timeout=45) as resp:
raw = resp.read().decode("utf-8", errors="replace")
return {
"ok": 200 <= resp.status < 300,
"status": resp.status,
"json": json.loads(raw) if raw else {},
"raw": raw,
}
except urllib.error.HTTPError as e:
raw = e.read().decode("utf-8", errors="replace")
return {
"ok": False,
"status": e.code,
"json": None,
"raw": raw,
}
def main() -> int:
required = {
"CHATWOOT_BASE_URL": CHATWOOT_BASE_URL,
"CHATWOOT_ACCOUNT_ID": CHATWOOT_ACCOUNT_ID,
"CHATWOOT_API_TOKEN": CHATWOOT_API_TOKEN,
"CLIENTFLOW_WEBHOOK_SECRET": CLIENTFLOW_WEBHOOK_SECRET,
}
missing = [k for k, v in required.items() if not v]
if missing:
raise SystemExit(f"Faltam variáveis: {', '.join(missing)}")
cutoff = datetime.now(timezone.utc) - timedelta(days=BACKFILL_DAYS)
print(f"BACKFILL_DAYS={BACKFILL_DAYS}")
print(f"BACKFILL_STATUSES={BACKFILL_STATUSES}")
print(f"BACKFILL_TO_CLIENTFLOW={BACKFILL_TO_CLIENTFLOW}")
print(f"CUTOFF={cutoff.isoformat()}")
seen_conversations = set()
selected = 0
posted = 0
failed = 0
skipped = 0
for status in BACKFILL_STATUSES:
conversations = fetch_conversations(status)
print(f"--- status={status} conversations={len(conversations)}")
for conv in conversations:
conv_id = str(conv.get("id") or "")
if not conv_id or conv_id in seen_conversations:
continue
seen_conversations.add(conv_id)
last_activity = conversation_last_activity(conv)
if last_activity and last_activity < cutoff:
skipped += 1
continue
messages = fetch_messages(conv_id)
incoming_messages = [
m for m in messages
if is_incoming(m)
and not m.get("private")
and str(m.get("content") or "").strip()
and message_created_at(m) >= cutoff
]
if not incoming_messages:
skipped += 1
continue
incoming_messages.sort(key=message_created_at)
last_msg = incoming_messages[-1]
content = str(last_msg.get("content") or "").strip()
sender = sender_from_conversation(conv, last_msg)
contact_id = str(sender.get("id") or conv.get("contact_id") or conv.get("contact", {}).get("id") or "")
selected += 1
payload = {
"event": "message_created",
"message": {
"id": str(last_msg.get("id") or f"backfill-{conv_id}"),
"content": content,
"message_type": "incoming",
"conversation_id": conv_id,
"sender": sender,
"created_at": last_msg.get("created_at"),
},
"conversation": {
"id": conv_id,
"status": conv.get("status") or status,
"contact": sender,
"meta": {
"sender": sender,
},
},
"backfill": {
"source": "chatwoot_inbox_last_days",
"days": BACKFILL_DAYS,
"status": status,
"last_activity_at": conv.get("last_activity_at"),
},
}
print("---")
print(f"conversation={conv_id} contact={contact_id} msg={payload['message']['id']}")
print(f"content={content[:160].replace(chr(10), ' ')}")
if not BACKFILL_TO_CLIENTFLOW:
print("DRY_RUN")
continue
result = post_to_clientflow(payload)
if result["ok"]:
print(f"POSTED status={result['status']}")
posted += 1
else:
print(f"FAILED status={result['status']} body={result['raw'][:500]}")
failed += 1
print("---")
print(f"selected={selected}")
print(f"posted={posted}")
print(f"failed={failed}")
print(f"skipped={skipped}")
return 0 if failed == 0 else 1
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,312 @@
#!/usr/bin/env python3
"""Backfill Jasmin document details into existing opportunities.
Use when an opportunity was created from a Jasmin reconciliation item before
v4926.6, so it still has no commercial_documents/opportunity_items/value.
Examples:
PYTHONPATH=. python scripts/backfill_jasmin_opportunity_details.py \
--opportunity-id a4e210f5-b870-48b1-882d-b55c71fbfd38
PYTHONPATH=. python scripts/backfill_jasmin_opportunity_details.py \
--document-number ORC.ORC2026.156
"""
from __future__ import annotations
import argparse
import asyncio
import json
import sys
from decimal import Decimal, InvalidOperation
from typing import Any, Dict, Iterable, List, Optional
from sqlalchemy import text
from app.db import engine
from app.reconciliation_service import (
_apply_jasmin_documents_to_opportunity, # noqa: PLC2701 - deliberate operator backfill script
_jasmin_document_lines_from_item, # noqa: PLC2701
_jasmin_document_totals, # noqa: PLC2701
_payload_record, # noqa: PLC2701
)
def _as_text(value: Any) -> str:
return str(value or "").strip()
def _money_value(value: Any) -> Any:
if isinstance(value, dict):
for key in ("amount", "baseAmount", "reportingAmount", "value"):
if value.get(key) not in (None, ""):
return value.get(key)
return None
return value
def _decimal_or_none(value: Any) -> Optional[str]:
value = _money_value(value)
if value in (None, ""):
return None
try:
return str(Decimal(str(value).replace(",", ".")).quantize(Decimal("0.01")))
except (InvalidOperation, ValueError):
return None
def _json(value: Any) -> str:
return json.dumps(value, ensure_ascii=False, default=str)
def _ids_from_metadata(metadata: Any) -> List[str]:
if not isinstance(metadata, dict):
return []
ids: List[str] = []
for key in ("created_from_reconciliation_item_id", "reconciliation_item_id"):
value = metadata.get(key)
if value:
ids.append(str(value))
for key in ("item_ids", "reconciliation_item_ids"):
value = metadata.get(key)
if isinstance(value, list):
ids.extend(str(v) for v in value if v)
return list(dict.fromkeys(ids))
def _load_opportunity(opportunity_id: Optional[str], document_number: Optional[str]) -> Optional[Dict[str, Any]]:
with engine.begin() as conn:
if opportunity_id:
row = conn.execute(text("""
SELECT id::text, title, value_amount, product_interest, local_customer_id::text,
customer_name, customer_email, metadata
FROM opportunities
WHERE id = CAST(:id AS UUID)
LIMIT 1
"""), {"id": opportunity_id}).mappings().first()
return dict(row) if row else None
if document_number:
row = conn.execute(text("""
SELECT id::text, title, value_amount, product_interest, local_customer_id::text,
customer_name, customer_email, metadata
FROM opportunities
WHERE metadata->>'document_number' = :document_number
OR title ILIKE '%' || :document_number || '%'
ORDER BY updated_at DESC
LIMIT 1
"""), {"document_number": document_number}).mappings().first()
return dict(row) if row else None
return None
def _load_candidate_items(opportunity: Dict[str, Any]) -> List[Dict[str, Any]]:
metadata = opportunity.get("metadata") if isinstance(opportunity.get("metadata"), dict) else {}
ids = _ids_from_metadata(metadata)
external_id = _as_text(metadata.get("external_id"))
document_number = _as_text(metadata.get("document_number"))
opportunity_id = _as_text(opportunity.get("id"))
with engine.begin() as conn:
rows = conn.execute(text("""
SELECT id::text, source_system, external_type, external_id, title, description,
status, priority, suggested_action, confidence, opportunity_id::text,
customer_id::text, customer_name, customer_email, customer_tax_id,
document_number, document_date, amount, currency, payload,
resolution_note, created_at, updated_at, resolved_at
FROM reconciliation_items
WHERE source_system = 'jasmin'
AND (
opportunity_id = CAST(:opportunity_id AS UUID)
OR (CAST(:ids AS TEXT[]) IS NOT NULL AND id::text = ANY(CAST(:ids AS TEXT[])))
OR (CAST(:external_id AS TEXT) <> '' AND external_id = CAST(:external_id AS TEXT))
OR (CAST(:document_number AS TEXT) <> '' AND document_number = CAST(:document_number AS TEXT))
OR (CAST(:document_number AS TEXT) <> '' AND payload::text ILIKE '%' || CAST(:document_number AS TEXT) || '%')
)
ORDER BY updated_at DESC, created_at DESC
"""), {
"opportunity_id": opportunity_id,
"ids": ids or [],
"external_id": external_id,
"document_number": document_number,
}).mappings().all()
# De-duplicate while keeping recency order.
seen = set()
result = []
for row in rows:
item = dict(row)
item_id = item.get("id")
if item_id in seen:
continue
seen.add(item_id)
result.append(item)
return result
async def _fetch_jasmin_detail_async(item: Dict[str, Any]) -> Optional[Dict[str, Any]]:
external_type = _as_text(item.get("external_type"))
external_id = _as_text(item.get("external_id"))
if not external_id:
return None
from app.jasmin_client import JasminClient
client = JasminClient()
if external_type == "jasmin_quotation":
return await client.get_quotation(external_id)
if external_type == "jasmin_invoice":
return await client.get_invoice(external_id)
# Some tenants represent pro-forma as a quotation. Try quotation detail as a
# conservative fallback when the external id is present.
if external_type == "jasmin_proforma":
try:
return await client.get_quotation(external_id)
except Exception:
return None
return None
def _with_jasmin_detail(item: Dict[str, Any], *, fetch_detail: bool) -> Dict[str, Any]:
if not fetch_detail:
return item
existing_lines = _jasmin_document_lines_from_item(item)
if existing_lines:
return item
try:
detail = asyncio.run(_fetch_jasmin_detail_async(item))
except Exception as exc:
item = dict(item)
payload = item.get("payload") if isinstance(item.get("payload"), dict) else {}
item["payload"] = {
**payload,
"detail_fetch_error": f"{type(exc).__name__}: {exc}",
}
return item
if not isinstance(detail, dict):
return item
payload = item.get("payload") if isinstance(item.get("payload"), dict) else {}
enriched = dict(item)
enriched["payload"] = {
**payload,
"record": detail,
"detail_source": "jasmin_api",
"previous_record": payload.get("record"),
}
# Fill top-level fields if the detailed document exposes them only there.
record_number = detail.get("documentNumber") or detail.get("naturalKey") or detail.get("number")
if record_number and not enriched.get("document_number"):
enriched["document_number"] = record_number
total = (
detail.get("payableAmount")
or detail.get("totalAmount")
or detail.get("grossAmount")
or detail.get("amount")
)
if total and not enriched.get("amount"):
enriched["amount"] = _decimal_or_none(total) or total
return enriched
def _summary_for_items(items: Iterable[Dict[str, Any]]) -> List[Dict[str, Any]]:
result = []
for item in items:
totals = _jasmin_document_totals(item)
lines = _jasmin_document_lines_from_item(item)
result.append({
"id": item.get("id"),
"external_type": item.get("external_type"),
"external_id": item.get("external_id"),
"document_number": item.get("document_number"),
"amount": item.get("amount"),
"totals": totals,
"lines": len(lines),
"payload_keys": list((item.get("payload") or {}).keys()) if isinstance(item.get("payload"), dict) else [],
})
return result
def _post_import_summary(opportunity_id: str) -> Dict[str, Any]:
with engine.begin() as conn:
opportunity = conn.execute(text("""
SELECT id::text, title, value_amount, product_interest, metadata
FROM opportunities
WHERE id = CAST(:id AS UUID)
"""), {"id": opportunity_id}).mappings().first()
docs = conn.execute(text("""
SELECT id::text, document_kind, document_number, amount, total_amount, currency, document_date
FROM commercial_documents
WHERE opportunity_id = CAST(:id AS UUID)
ORDER BY created_at DESC
"""), {"id": opportunity_id}).mappings().all()
items = conn.execute(text("""
SELECT product_name, quantity, unit_price, total_price, jasmin_sales_item
FROM opportunity_items
WHERE opportunity_id = CAST(:id AS UUID)
ORDER BY created_at
"""), {"id": opportunity_id}).mappings().all()
lines = conn.execute(text("""
SELECT cdl.description, cdl.quantity, cdl.unit_price, cdl.total_amount, cdl.jasmin_sales_item
FROM commercial_document_lines cdl
JOIN commercial_documents cd ON cd.id = cdl.document_id
WHERE cd.opportunity_id = CAST(:id AS UUID)
ORDER BY cdl.line_index
"""), {"id": opportunity_id}).mappings().all()
return {
"opportunity": dict(opportunity or {}),
"documents": [dict(r) for r in docs],
"opportunity_items": [dict(r) for r in items],
"document_lines": [dict(r) for r in lines],
}
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id")
parser.add_argument("--document-number")
parser.add_argument("--no-fetch-jasmin-detail", action="store_true")
parser.add_argument("--dry-run", action="store_true")
parser.add_argument("--actor", default="operator_backfill")
args = parser.parse_args()
if not args.opportunity_id and not args.document_number:
parser.error("usa --opportunity-id ou --document-number")
opportunity = _load_opportunity(args.opportunity_id, args.document_number)
if not opportunity:
print(json.dumps({"ok": False, "error": "opportunity_not_found"}, ensure_ascii=False, indent=2))
return 2
items = _load_candidate_items(opportunity)
enriched_items = [
_with_jasmin_detail(item, fetch_detail=not args.no_fetch_jasmin_detail)
for item in items
]
print(json.dumps({
"opportunity_id": opportunity.get("id"),
"title": opportunity.get("title"),
"candidate_items": _summary_for_items(enriched_items),
"dry_run": args.dry_run,
}, ensure_ascii=False, indent=2, default=str))
if not enriched_items:
print(json.dumps({"ok": False, "error": "no_jasmin_reconciliation_items_found"}, ensure_ascii=False, indent=2))
return 3
if args.dry_run:
return 0
with engine.begin() as conn:
result = _apply_jasmin_documents_to_opportunity(
conn,
enriched_items,
str(opportunity["id"]),
actor=args.actor,
)
print(json.dumps({"ok": True, "import_result": result}, ensure_ascii=False, indent=2, default=str))
print(json.dumps(_post_import_summary(str(opportunity["id"])), ensure_ascii=False, indent=2, default=str))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,89 @@
#!/usr/bin/env python3
"""Backfill Jasmin product mappings for Odoo-imported opportunity lines.
Maps Odoo line metadata product ids to the ClientFlow catalogue convention:
``product_id=3`` -> ``products.sku='ODOO-3'`` -> ``jasmin_sales_item``.
Usage:
PYTHONPATH=. python scripts/backfill_opportunity_product_mappings.py
PYTHONPATH=. python scripts/backfill_opportunity_product_mappings.py --opportunity-id <uuid> --apply
"""
from __future__ import annotations
import argparse
from sqlalchemy import text
from app.db import engine
from app.product_service import ensure_product_schema
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id", default="", help="Optional opportunity UUID to restrict the backfill")
parser.add_argument("--apply", action="store_true", help="Apply changes. Without this flag it runs as dry-run.")
args = parser.parse_args()
ensure_product_schema()
where_opp = "AND oi.opportunity_id = CAST(:opportunity_id AS UUID)" if args.opportunity_id else ""
params = {"opportunity_id": args.opportunity_id} if args.opportunity_id else {}
with engine.begin() as conn:
rows = conn.execute(text(f"""
SELECT
oi.id::text AS item_id,
oi.opportunity_id::text AS opportunity_id,
oi.product_name,
oi.sku AS old_sku,
oi.jasmin_sales_item AS old_jasmin_sales_item,
oi.metadata->>'product_id' AS odoo_product_id,
p.id::text AS product_id,
p.sku AS new_sku,
p.jasmin_sales_item AS new_jasmin_sales_item,
p.name AS catalog_name
FROM opportunity_items oi
JOIN products p
ON p.sku = ('ODOO-' || (oi.metadata->>'product_id'))
WHERE oi.metadata->>'source_system' = 'odoo'
AND COALESCE(oi.metadata->>'product_id','') <> ''
AND (
oi.product_id IS DISTINCT FROM p.id
OR COALESCE(oi.sku,'') IS DISTINCT FROM COALESCE(p.sku,'')
OR COALESCE(oi.jasmin_sales_item,'') IS DISTINCT FROM COALESCE(p.jasmin_sales_item,'')
)
{where_opp}
ORDER BY oi.created_at DESC
"""), params).mappings().all()
print(f"Candidatos a atualizar: {len(rows)}")
for row in rows:
print(dict(row))
if args.apply and rows:
result = conn.execute(text(f"""
UPDATE opportunity_items oi
SET product_id = p.id,
sku = p.sku,
jasmin_sales_item = p.jasmin_sales_item,
metadata = COALESCE(oi.metadata, '{{}}'::jsonb) || jsonb_build_object(
'resolved_sku', p.sku,
'resolved_jasmin_sales_item', p.jasmin_sales_item,
'product_mapping_status', CASE WHEN COALESCE(p.jasmin_sales_item,'') <> '' THEN 'mapped' ELSE 'missing_jasmin' END,
'catalog_name', p.name,
'backfilled_at', now()::text
),
updated_at = now()
FROM products p
WHERE p.sku = ('ODOO-' || (oi.metadata->>'product_id'))
AND oi.metadata->>'source_system' = 'odoo'
AND COALESCE(oi.metadata->>'product_id','') <> ''
{where_opp}
"""), params)
print(f"Aplicado: {result.rowcount or 0} linha(s) atualizada(s)")
elif not args.apply:
print("Dry-run. Usa --apply para aplicar.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,347 @@
#!/usr/bin/env python3
"""Batch validation for email identity extraction.
Safe for LLM runs: prints progress, truncates long bodies, applies a per-message
process timeout, and writes JSONL incrementally so partial results are kept even
if a provider call stalls.
"""
from __future__ import annotations
import argparse
import csv
import json
import multiprocessing as mp
import os
import re
from datetime import datetime
from pathlib import Path
from typing import Any, Dict, List
from sqlalchemy import text
from app.commercial_service import normalize_fiscal_name
from app.db import engine
LEGAL_SUFFIX_TOKENS = {
"lda", "limitada", "unipessoal", "sa", "s", "a", "sociedade",
"mediação", "mediacao", "seguro", "seguros", "importação", "importacao",
"exportação", "exportacao", "fabricação", "fabricacao", "representação",
"representacao", "soluções", "solucoes", "metálicas", "metalicas",
}
def clean(value: Any) -> str:
return re.sub(r"\s+", " ", str(value or "")).strip()
def norm(value: Any) -> str:
return normalize_fiscal_name(value or "") or clean(value).casefold()
def meaningful_tokens(value: Any) -> set[str]:
n = norm(value)
tokens = {t for t in re.split(r"[^a-z0-9áàâãéèêíìîóòôõúùûç]+", n) if len(t) >= 3}
return {t for t in tokens if t not in LEGAL_SUFFIX_TOKENS}
def mentions_match_fiscal(mentions: List[str], fiscal_name: str | None) -> bool:
if not mentions or not fiscal_name:
return False
nf = norm(fiscal_name)
fiscal_tokens = meaningful_tokens(fiscal_name)
for mention in mentions:
nm = norm(mention)
if not nm:
continue
if nm == nf:
return True
if len(nm) >= 4 and (nm in nf or nf in nm):
return True
mention_tokens = meaningful_tokens(mention)
if not mention_tokens or not fiscal_tokens:
continue
overlap = mention_tokens & fiscal_tokens
if len(overlap) >= 2:
return True
if len(mention_tokens) <= 2 and overlap:
return True
return False
def classify(identity: Dict[str, Any], fiscal_customer: str | None) -> str:
if identity.get("_error"):
return "EXTRACTION_ERROR"
if identity.get("_timeout"):
return "EXTRACTION_TIMEOUT"
mentions = identity.get("company_mentions") or []
domain = identity.get("domain") or ""
person = identity.get("person_name") or ""
if mentions and fiscal_customer:
if mentions_match_fiscal(mentions, fiscal_customer):
return "OK_MENTION_COMPATIBLE_WITH_FISCAL"
return "CONFLICT_MENTION_DIFFERS_FROM_FISCAL"
if mentions and not fiscal_customer:
return "OK_MENTION_AVAILABLE_NO_FISCAL"
if not mentions and fiscal_customer:
return "WEAK_NO_COMPANY_MENTION_HAS_FISCAL"
if domain and person:
return "WEAK_PERSON_AND_DOMAIN_ONLY"
if domain:
return "WEAK_DOMAIN_ONLY"
return "NO_USEFUL_IDENTITY"
def fetch_cases(limit: int, offset: int) -> List[Dict[str, Any]]:
with engine.begin() as conn:
rows = conn.execute(text("""
WITH ranked AS (
SELECT
o.id::text AS opportunity_id,
o.title AS opportunity_title,
o.customer_name,
o.customer_email,
c.name AS fiscal_customer,
c.tax_id AS fiscal_tax_id,
c.email AS fiscal_email,
t.id::text AS task_id,
t.action_code,
t.route,
t.created_at AS task_created_at,
m.id::text AS message_id,
COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) AS body,
COALESCE(
NULLIF(m.metadata->>'subject', ''),
NULLIF(re.payload->'conversation'->'additional_attributes'->>'mail_subject', ''),
NULLIF(re.payload->'content_attributes'->'email'->>'subject', ''),
NULLIF(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '')
) AS subject,
COALESCE(
NULLIF(re.payload->'sender'->>'email', ''),
NULLIF(re.payload->'conversation'->'meta'->'sender'->>'email', ''),
NULLIF(re.payload->'conversation'->'contact_inbox'->>'source_id', ''),
NULLIF(o.customer_email, '')
) AS sender_email,
ROW_NUMBER() OVER (
PARTITION BY o.id
ORDER BY t.created_at DESC
) AS rn
FROM opportunities o
JOIN tasks t ON t.opportunity_id = o.id
LEFT JOIN messages m ON m.id = t.message_id
LEFT JOIN raw_events re ON re.id = t.raw_event_id
LEFT JOIN customers c ON c.id = o.local_customer_id
WHERE COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) IS NOT NULL
)
SELECT *
FROM ranked
WHERE rn = 1
ORDER BY task_created_at DESC
LIMIT :limit
OFFSET :offset
"""), {"limit": limit, "offset": offset}).mappings().all()
return [dict(r) for r in rows]
def _domain_from_email(email: str) -> str:
if "@" not in (email or ""):
return ""
return email.rsplit("@", 1)[1].lower().strip()
def worker_extract(queue: mp.Queue, body: str, email: str, subject: str, use_llm: bool) -> None:
try:
from app.email_identity_extraction_service import extract_email_identity
identity = extract_email_identity(
body or "",
email=email or "",
subject=subject or "",
use_llm=use_llm,
)
queue.put({"ok": True, "identity": identity})
except Exception as exc: # noqa: BLE001 validation tool should keep going
queue.put({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
def extract_with_timeout(body: str, email: str, subject: str, use_llm: bool, timeout_seconds: int) -> Dict[str, Any]:
if not use_llm:
from app.email_identity_extraction_service import extract_email_identity
return extract_email_identity(body or "", email=email or "", subject=subject or "", use_llm=False)
queue: mp.Queue = mp.Queue()
proc = mp.Process(target=worker_extract, args=(queue, body, email, subject, use_llm))
proc.start()
proc.join(timeout_seconds)
if proc.is_alive():
proc.terminate()
proc.join(5)
return {
"_timeout": True,
"method": "timeout",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": [f"timeout após {timeout_seconds}s"],
}
if queue.empty():
return {
"_error": True,
"method": "error",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": ["processo terminou sem resultado"],
}
result = queue.get()
if result.get("ok"):
return result["identity"]
return {
"_error": True,
"method": "error",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": [result.get("error")],
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=20)
parser.add_argument("--offset", type=int, default=0)
parser.add_argument("--use-llm", action="store_true")
parser.add_argument("--model", default="", help="Override EMAIL_IDENTITY_LLM_MODEL for this validation run")
parser.add_argument("--fallback-model", default="", help="Override EMAIL_IDENTITY_LLM_FALLBACK_MODEL for this validation run")
parser.add_argument("--timeout-seconds", type=int, default=30)
parser.add_argument("--max-body-chars", type=int, default=3500)
parser.add_argument("--out-dir", default="reports")
args = parser.parse_args()
if args.model:
os.environ["EMAIL_IDENTITY_LLM_MODEL"] = args.model
if args.fallback_model:
os.environ["EMAIL_IDENTITY_LLM_FALLBACK_MODEL"] = args.fallback_model
cases = fetch_cases(args.limit, args.offset)
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
mode = "llm" if args.use_llm else "regex"
jsonl_path = out_dir / f"email_identity_validation_{mode}_{stamp}.jsonl"
csv_path = out_dir / f"email_identity_validation_{mode}_{stamp}.csv"
results: List[Dict[str, Any]] = []
counts: Dict[str, int] = {}
model_label = os.getenv("EMAIL_IDENTITY_LLM_MODEL") or os.getenv("OPENROUTER_MODEL", "")
fallback_label = os.getenv("EMAIL_IDENTITY_LLM_FALLBACK_MODEL", "")
print(
f"Casos: {len(cases)} | mode={mode} | offset={args.offset} | "
f"timeout={args.timeout_seconds}s | max_body_chars={args.max_body_chars} | "
f"model={model_label if args.use_llm else '-'} | fallback={fallback_label if args.use_llm and fallback_label else '-'}",
flush=True,
)
print(f"JSONL incremental: {jsonl_path}", flush=True)
with jsonl_path.open("w", encoding="utf-8") as jf:
for idx, row in enumerate(cases, start=1):
email = row.get("sender_email") or row.get("customer_email") or ""
body = (row.get("body") or "")[: args.max_body_chars]
subject = row.get("subject") or ""
print(f"\n[{idx}/{len(cases)}] {row.get('opportunity_title')} | {email}", flush=True)
identity = extract_with_timeout(
body=body,
email=email,
subject=subject,
use_llm=args.use_llm,
timeout_seconds=args.timeout_seconds,
)
status = classify(identity, row.get("fiscal_customer"))
result = {
"status": status,
"opportunity_id": row.get("opportunity_id"),
"opportunity_title": row.get("opportunity_title"),
"task_id": row.get("task_id"),
"action_code": row.get("action_code"),
"sender_email": email,
"customer_name": row.get("customer_name"),
"customer_email": row.get("customer_email"),
"fiscal_customer": row.get("fiscal_customer"),
"fiscal_tax_id": row.get("fiscal_tax_id"),
"fiscal_email": row.get("fiscal_email"),
"person_name": identity.get("person_name"),
"company_mentions": identity.get("company_mentions") or [],
"domain": identity.get("domain"),
"address": identity.get("address"),
"phones": identity.get("phones") or [],
"confidence": identity.get("confidence"),
"method": identity.get("method"),
"llm_model": identity.get("llm_model"),
"fallback_used": identity.get("fallback_used"),
"evidence": identity.get("evidence") or [],
}
results.append(result)
counts[status] = counts.get(status, 0) + 1
jf.write(json.dumps(result, ensure_ascii=False, default=str) + "\n")
jf.flush()
print(
"status:", status,
"| person:", result["person_name"],
"| companies:", result["company_mentions"],
flush=True,
)
csv_fields = [
"status", "opportunity_id", "opportunity_title", "task_id", "action_code",
"sender_email", "customer_name", "customer_email", "fiscal_customer",
"fiscal_tax_id", "fiscal_email", "person_name", "company_mentions",
"domain", "address", "phones", "confidence", "method", "llm_model", "fallback_used", "evidence",
]
with csv_path.open("w", encoding="utf-8", newline="") as f:
writer = csv.DictWriter(f, fieldnames=csv_fields)
writer.writeheader()
for r in results:
csv_row = dict(r)
for key in ["company_mentions", "phones", "evidence"]:
csv_row[key] = " | ".join(str(x) for x in csv_row.get(key) or [])
writer.writerow(csv_row)
print("\nResumo:")
for status, count in sorted(counts.items()):
print(f"{status}: {count}")
print("\nFicheiros gerados:")
print(jsonl_path)
print(csv_path)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,347 @@
#!/usr/bin/env python3
"""Batch validation for email identity extraction.
Safe for LLM runs: prints progress, truncates long bodies, applies a per-message
process timeout, and writes JSONL incrementally so partial results are kept even
if a provider call stalls.
"""
from __future__ import annotations
import argparse
import csv
import json
import multiprocessing as mp
import os
import re
from datetime import datetime
from pathlib import Path
from typing import Any, Dict, List
from sqlalchemy import text
from app.commercial_service import normalize_fiscal_name
from app.db import engine
LEGAL_SUFFIX_TOKENS = {
"lda", "limitada", "unipessoal", "sa", "s", "a", "sociedade",
"mediação", "mediacao", "seguro", "seguros", "importação", "importacao",
"exportação", "exportacao", "fabricação", "fabricacao", "representação",
"representacao", "soluções", "solucoes", "metálicas", "metalicas",
}
def clean(value: Any) -> str:
return re.sub(r"\s+", " ", str(value or "")).strip()
def norm(value: Any) -> str:
return normalize_fiscal_name(value or "") or clean(value).casefold()
def meaningful_tokens(value: Any) -> set[str]:
n = norm(value)
tokens = {t for t in re.split(r"[^a-z0-9áàâãéèêíìîóòôõúùûç]+", n) if len(t) >= 3}
return {t for t in tokens if t not in LEGAL_SUFFIX_TOKENS}
def mentions_match_fiscal(mentions: List[str], fiscal_name: str | None) -> bool:
if not mentions or not fiscal_name:
return False
nf = norm(fiscal_name)
fiscal_tokens = meaningful_tokens(fiscal_name)
for mention in mentions:
nm = norm(mention)
if not nm:
continue
if nm == nf:
return True
if len(nm) >= 4 and (nm in nf or nf in nm):
return True
mention_tokens = meaningful_tokens(mention)
if not mention_tokens or not fiscal_tokens:
continue
overlap = mention_tokens & fiscal_tokens
if len(overlap) >= 2:
return True
if len(mention_tokens) <= 2 and overlap:
return True
return False
def classify(identity: Dict[str, Any], fiscal_customer: str | None) -> str:
if identity.get("_error"):
return "EXTRACTION_ERROR"
if identity.get("_timeout"):
return "EXTRACTION_TIMEOUT"
mentions = identity.get("company_mentions") or []
domain = identity.get("domain") or ""
person = identity.get("person_name") or ""
if mentions and fiscal_customer:
if mentions_match_fiscal(mentions, fiscal_customer):
return "OK_MENTION_COMPATIBLE_WITH_FISCAL"
return "CONFLICT_MENTION_DIFFERS_FROM_FISCAL"
if mentions and not fiscal_customer:
return "OK_MENTION_AVAILABLE_NO_FISCAL"
if not mentions and fiscal_customer:
return "WEAK_NO_COMPANY_MENTION_HAS_FISCAL"
if domain and person:
return "WEAK_PERSON_AND_DOMAIN_ONLY"
if domain:
return "WEAK_DOMAIN_ONLY"
return "NO_USEFUL_IDENTITY"
def fetch_cases(limit: int, offset: int) -> List[Dict[str, Any]]:
with engine.begin() as conn:
rows = conn.execute(text("""
WITH ranked AS (
SELECT
o.id::text AS opportunity_id,
o.title AS opportunity_title,
o.customer_name,
o.customer_email,
c.name AS fiscal_customer,
c.tax_id AS fiscal_tax_id,
c.email AS fiscal_email,
t.id::text AS task_id,
t.action_code,
t.route,
t.created_at AS task_created_at,
m.id::text AS message_id,
COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) AS body,
COALESCE(
NULLIF(m.metadata->>'subject', ''),
NULLIF(re.payload->'conversation'->'additional_attributes'->>'mail_subject', ''),
NULLIF(re.payload->'content_attributes'->'email'->>'subject', ''),
NULLIF(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '')
) AS subject,
COALESCE(
NULLIF(re.payload->'sender'->>'email', ''),
NULLIF(re.payload->'conversation'->'meta'->'sender'->>'email', ''),
NULLIF(re.payload->'conversation'->'contact_inbox'->>'source_id', ''),
NULLIF(o.customer_email, '')
) AS sender_email,
ROW_NUMBER() OVER (
PARTITION BY o.id
ORDER BY t.created_at DESC
) AS rn
FROM opportunities o
JOIN tasks t ON t.opportunity_id = o.id
LEFT JOIN messages m ON m.id = t.message_id
LEFT JOIN raw_events re ON re.id = t.raw_event_id
LEFT JOIN customers c ON c.id = o.local_customer_id
WHERE COALESCE(NULLIF(m.clean_body, ''), NULLIF(m.raw_body, '')) IS NOT NULL
)
SELECT *
FROM ranked
WHERE rn = 1
ORDER BY task_created_at DESC
LIMIT :limit
OFFSET :offset
"""), {"limit": limit, "offset": offset}).mappings().all()
return [dict(r) for r in rows]
def _domain_from_email(email: str) -> str:
if "@" not in (email or ""):
return ""
return email.rsplit("@", 1)[1].lower().strip()
def worker_extract(queue: mp.Queue, body: str, email: str, subject: str, use_llm: bool) -> None:
try:
from app.email_identity_extraction_service import extract_email_identity
identity = extract_email_identity(
body or "",
email=email or "",
subject=subject or "",
use_llm=use_llm,
)
queue.put({"ok": True, "identity": identity})
except Exception as exc: # noqa: BLE001 validation tool should keep going
queue.put({"ok": False, "error": f"{type(exc).__name__}: {exc}"})
def extract_with_timeout(body: str, email: str, subject: str, use_llm: bool, timeout_seconds: int) -> Dict[str, Any]:
if not use_llm:
from app.email_identity_extraction_service import extract_email_identity
return extract_email_identity(body or "", email=email or "", subject=subject or "", use_llm=False)
queue: mp.Queue = mp.Queue()
proc = mp.Process(target=worker_extract, args=(queue, body, email, subject, use_llm))
proc.start()
proc.join(timeout_seconds)
if proc.is_alive():
proc.terminate()
proc.join(5)
return {
"_timeout": True,
"method": "timeout",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": [f"timeout após {timeout_seconds}s"],
}
if queue.empty():
return {
"_error": True,
"method": "error",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": ["processo terminou sem resultado"],
}
result = queue.get()
if result.get("ok"):
return result["identity"]
return {
"_error": True,
"method": "error",
"confidence": 0,
"email": email,
"domain": _domain_from_email(email),
"person_name": "",
"company_mentions": [],
"address": "",
"phones": [],
"websites": [],
"evidence": [result.get("error")],
}
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=20)
parser.add_argument("--offset", type=int, default=0)
parser.add_argument("--use-llm", action="store_true")
parser.add_argument("--model", default="", help="Override EMAIL_IDENTITY_LLM_MODEL for this validation run")
parser.add_argument("--fallback-model", default="", help="Override EMAIL_IDENTITY_LLM_FALLBACK_MODEL for this validation run")
parser.add_argument("--timeout-seconds", type=int, default=30)
parser.add_argument("--max-body-chars", type=int, default=3500)
parser.add_argument("--out-dir", default="reports")
args = parser.parse_args()
if args.model:
os.environ["EMAIL_IDENTITY_LLM_MODEL"] = args.model
if args.fallback_model:
os.environ["EMAIL_IDENTITY_LLM_FALLBACK_MODEL"] = args.fallback_model
cases = fetch_cases(args.limit, args.offset)
out_dir = Path(args.out_dir)
out_dir.mkdir(parents=True, exist_ok=True)
stamp = datetime.now().strftime("%Y%m%d_%H%M%S")
mode = "llm" if args.use_llm else "regex"
jsonl_path = out_dir / f"email_identity_validation_{mode}_{stamp}.jsonl"
csv_path = out_dir / f"email_identity_validation_{mode}_{stamp}.csv"
results: List[Dict[str, Any]] = []
counts: Dict[str, int] = {}
model_label = os.getenv("EMAIL_IDENTITY_LLM_MODEL") or os.getenv("OPENROUTER_MODEL", "")
fallback_label = os.getenv("EMAIL_IDENTITY_LLM_FALLBACK_MODEL", "")
print(
f"Casos: {len(cases)} | mode={mode} | offset={args.offset} | "
f"timeout={args.timeout_seconds}s | max_body_chars={args.max_body_chars} | "
f"model={model_label if args.use_llm else '-'} | fallback={fallback_label if args.use_llm and fallback_label else '-'}",
flush=True,
)
print(f"JSONL incremental: {jsonl_path}", flush=True)
with jsonl_path.open("w", encoding="utf-8") as jf:
for idx, row in enumerate(cases, start=1):
email = row.get("sender_email") or row.get("customer_email") or ""
body = (row.get("body") or "")[: args.max_body_chars]
subject = row.get("subject") or ""
print(f"\n[{idx}/{len(cases)}] {row.get('opportunity_title')} | {email}", flush=True)
identity = extract_with_timeout(
body=body,
email=email,
subject=subject,
use_llm=args.use_llm,
timeout_seconds=args.timeout_seconds,
)
status = classify(identity, row.get("fiscal_customer"))
result = {
"status": status,
"opportunity_id": row.get("opportunity_id"),
"opportunity_title": row.get("opportunity_title"),
"task_id": row.get("task_id"),
"action_code": row.get("action_code"),
"sender_email": email,
"customer_name": row.get("customer_name"),
"customer_email": row.get("customer_email"),
"fiscal_customer": row.get("fiscal_customer"),
"fiscal_tax_id": row.get("fiscal_tax_id"),
"fiscal_email": row.get("fiscal_email"),
"person_name": identity.get("person_name"),
"company_mentions": identity.get("company_mentions") or [],
"domain": identity.get("domain"),
"address": identity.get("address"),
"phones": identity.get("phones") or [],
"confidence": identity.get("confidence"),
"method": identity.get("method"),
"llm_model": identity.get("llm_model"),
"fallback_used": identity.get("fallback_used"),
"evidence": identity.get("evidence") or [],
}
results.append(result)
counts[status] = counts.get(status, 0) + 1
jf.write(json.dumps(result, ensure_ascii=False, default=str) + "\n")
jf.flush()
print(
"status:", status,
"| person:", result["person_name"],
"| companies:", result["company_mentions"],
flush=True,
)
csv_fields = [
"status", "opportunity_id", "opportunity_title", "task_id", "action_code",
"sender_email", "customer_name", "customer_email", "fiscal_customer",
"fiscal_tax_id", "fiscal_email", "person_name", "company_mentions",
"domain", "address", "phones", "confidence", "method", "llm_model", "fallback_used", "evidence",
]
with csv_path.open("w", encoding="utf-8", newline="") as f:
writer = csv.DictWriter(f, fieldnames=csv_fields)
writer.writeheader()
for r in results:
csv_row = dict(r)
for key in ["company_mentions", "phones", "evidence"]:
csv_row[key] = " | ".join(str(x) for x in csv_row.get(key) or [])
writer.writerow(csv_row)
print("\nResumo:")
for status, count in sorted(counts.items()):
print(f"{status}: {count}")
print("\nFicheiros gerados:")
print(jsonl_path)
print(csv_path)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,82 @@
#!/usr/bin/env python3
"""Operational health check for ClientFlow deployments."""
from __future__ import annotations
import json
import os
import sys
from pathlib import Path
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
from sqlalchemy import text
from app.config import settings
from app.db import engine
def scalar(sql: str, **params):
with engine.begin() as conn:
return conn.execute(text(sql), params).scalar()
def rows(sql: str, **params):
with engine.begin() as conn:
return [dict(r) for r in conn.execute(text(sql), params).mappings().all()]
def main() -> int:
result = {
"app": settings.app_name,
"env": settings.env,
"database_ok": False,
"jasmin_enabled": bool(settings.jasmin_enabled),
"packlink_enabled": bool(settings.packlink_enabled),
"outbox": {},
"documents": {},
"warnings": [],
}
try:
result["database_ok"] = bool(scalar("SELECT 1"))
except Exception as exc:
result["warnings"].append(f"database_error: {exc}")
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
return 2
for row in rows("""
SELECT target_system, status, count(*)::int AS total
FROM integration_outbox
GROUP BY target_system, status
ORDER BY target_system, status
"""):
result["outbox"].setdefault(row["target_system"], {})[row["status"]] = row["total"]
for row in rows("""
SELECT document_kind, status, count(*)::int AS total
FROM commercial_documents
GROUP BY document_kind, status
ORDER BY document_kind, status
"""):
result["documents"].setdefault(row["document_kind"], {})[row["status"]] = row["total"]
missing_jasmin = scalar("""
SELECT count(*)
FROM products
WHERE active = TRUE
AND (jasmin_sales_item IS NULL OR jasmin_sales_item = '')
""")
if missing_jasmin:
result["warnings"].append(f"active_products_without_jasmin_sales_item={missing_jasmin}")
failed_outbox = scalar("SELECT count(*) FROM integration_outbox WHERE status = 'failed'")
if failed_outbox:
result["warnings"].append(f"failed_outbox_items={failed_outbox}")
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
return 1 if result["warnings"] else 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,131 @@
#!/usr/bin/env python3
"""Clean stale invalid email-identity suggestions/extractions.
Use after strengthening company-mention filters. It targets suggestions and
stored extractions created from invalid fragments such as ``pt`` or ``com``.
"""
from __future__ import annotations
import argparse
import json
from typing import Any
from sqlalchemy import text
from app.db import engine
from app.email_identity_extraction_service import is_plausible_company_mention
INVALID = {"pt", "com", "net", "org", "www", "http", "https", "mail", "email"}
def _clean(value: Any) -> str:
return str(value or "").strip()
def _valid_companies(values: Any) -> list[str]:
out: list[str] = []
if not isinstance(values, list):
return out
for value in values:
v = _clean(value).strip(" ,.;:-")
if v and is_plausible_company_mention(v):
out.append(v)
return out
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id", default="")
parser.add_argument("--apply", action="store_true")
parser.add_argument("--include-accepted", action="store_true")
parser.add_argument("--fix-extractions", action="store_true", help="Also rewrite stored email_identity_extractions company_mentions after current filters")
args = parser.parse_args()
where = ["lookup_type LIKE 'email_identity%'"]
params: dict[str, Any] = {}
if args.opportunity_id:
where.append("opportunity_id = CAST(:opportunity_id AS UUID)")
params["opportunity_id"] = args.opportunity_id
if args.include_accepted:
where.append("status IN ('pending', 'accepted', 'rejected')")
else:
where.append("status = 'pending'")
invalid_sql = ", ".join("'" + v.replace("'", "") + "'" for v in sorted(INVALID))
where.append(f"lower(trim(COALESCE(lookup_value, ''))) IN ({invalid_sql})")
sql_where = " AND ".join(where)
with engine.begin() as conn:
rows = conn.execute(text(f"""
SELECT id::text, opportunity_id::text, suggested_name, suggested_nif,
lookup_type, lookup_value, confidence, status, reason, created_at
FROM fiscal_customer_suggestions
WHERE {sql_where}
ORDER BY created_at DESC
"""), params).mappings().all()
print(json.dumps({"invalid_suggestions": len(rows), "apply": args.apply}, ensure_ascii=False, indent=2, default=str))
for row in rows:
print(json.dumps(dict(row), ensure_ascii=False, indent=2, default=str))
if args.apply and rows:
ids = [r["id"] for r in rows]
conn.execute(text("""
UPDATE fiscal_customer_suggestions
SET status = 'rejected',
reason = COALESCE(reason, '') || ' | rejected_invalid_email_identity_token',
resolved_by = 'cleanup_invalid_email_identity_suggestions',
resolved_at = now(),
updated_at = now()
WHERE id = ANY(CAST(:ids AS UUID[]))
"""), {"ids": ids})
print(json.dumps({"rejected": len(ids)}, ensure_ascii=False, indent=2))
if args.fix_extractions:
extraction_where = []
extraction_params: dict[str, Any] = {}
if args.opportunity_id:
extraction_where.append("opportunity_id = CAST(:opportunity_id AS UUID)")
extraction_params["opportunity_id"] = args.opportunity_id
extraction_sql = "WHERE " + " AND ".join(extraction_where) if extraction_where else ""
ex_rows = conn.execute(text(f"""
SELECT id::text, opportunity_id::text, company_mentions, confidence, raw_payload
FROM email_identity_extractions
{extraction_sql}
ORDER BY updated_at DESC
"""), extraction_params).mappings().all()
changed = []
for row in ex_rows:
original = row.get("company_mentions") or []
valid = _valid_companies(original)
if list(original or []) == valid:
continue
changed.append({"id": row["id"], "opportunity_id": row["opportunity_id"], "before": original, "after": valid})
if args.apply:
try:
confidence = float(row.get("confidence") or 0)
except Exception:
confidence = 0.0
if not valid:
confidence = min(confidence, 0.45)
raw_payload = dict(row.get("raw_payload") or {}) if isinstance(row.get("raw_payload"), dict) else {}
raw_payload["filtered_invalid_company_mentions"] = list(original or [])
conn.execute(text("""
UPDATE email_identity_extractions
SET company_mentions = CAST(:company_mentions AS JSONB),
confidence = :confidence,
raw_payload = COALESCE(raw_payload, '{}'::jsonb) || CAST(:raw_payload AS JSONB),
updated_at = now()
WHERE id = CAST(:id AS UUID)
"""), {
"id": row["id"],
"company_mentions": json.dumps(valid, ensure_ascii=False),
"confidence": confidence,
"raw_payload": json.dumps(raw_payload, ensure_ascii=False, default=str),
})
print(json.dumps({"extractions_to_fix": len(changed), "items": changed}, ensure_ascii=False, indent=2, default=str))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,105 @@
#!/usr/bin/env python3
"""Identify or close opportunities that were probably created from system messages.
Conservative by default: use --dry-run to list candidates. Use --apply to mark
safe candidates as LOST with a metadata reason. It only targets opportunities
with value 0, no commercial documents and no shipments.
"""
from __future__ import annotations
import argparse
import json
import uuid
from sqlalchemy import text
from app.db import engine
from app.opportunity_service import ensure_opportunity_schema
SYSTEM_TERMS = (
"mail delivery subsystem",
"mailer-daemon",
"postmaster",
"returned mail",
"undelivered mail",
"delivery status notification",
"failure notice",
"mail delivery failed",
)
def _json(value: object) -> str:
return json.dumps(value or {}, ensure_ascii=False, default=str)
def find_candidates(limit: int = 100) -> list[dict]:
ensure_opportunity_schema()
like_sql = " OR ".join(
["lower(coalesce(o.title,'') || ' ' || coalesce(o.customer_name,'') || ' ' || coalesce(o.customer_email,'') || ' ' || coalesce(o.metadata::text,'')) LIKE :term_{}".format(i) for i, _ in enumerate(SYSTEM_TERMS)]
)
params = {f"term_{i}": f"%{term}%" for i, term in enumerate(SYSTEM_TERMS)}
params["limit"] = int(limit)
sql = text(f"""
SELECT o.id::text, o.title, o.customer_name, o.customer_email, o.value_amount, o.stage, o.status, o.created_at
FROM opportunities o
WHERE o.status = 'open'
AND COALESCE(o.value_amount, 0) = 0
AND NOT EXISTS (SELECT 1 FROM commercial_documents cd WHERE cd.opportunity_id = o.id)
AND NOT EXISTS (SELECT 1 FROM shipments s WHERE s.opportunity_id = o.id)
AND ({like_sql})
ORDER BY o.created_at DESC
LIMIT :limit
""")
with engine.begin() as conn:
rows = conn.execute(sql, params).mappings().all()
return [dict(row) for row in rows]
def close_candidates(candidates: list[dict]) -> int:
if not candidates:
return 0
ids = [row["id"] for row in candidates]
with engine.begin() as conn:
for opportunity_id in ids:
conn.execute(text("""
UPDATE opportunities
SET status = 'closed',
stage = 'LOST',
closed_at = COALESCE(closed_at, now()),
updated_at = now(),
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:metadata AS JSONB)
WHERE id = CAST(:opportunity_id AS UUID)
"""), {
"opportunity_id": opportunity_id,
"metadata": _json({"closed_reason": "system_or_bounce_created_by_mistake", "closed_by": "cleanup_non_commercial_opportunities"}),
})
conn.execute(text("""
INSERT INTO opportunity_events (id, opportunity_id, event_type, from_stage, to_stage, note, payload, created_by)
VALUES (CAST(:id AS UUID), CAST(:opportunity_id AS UUID), 'cleanup_closed_non_commercial', NULL, 'LOST', :note, CAST(:payload AS JSONB), 'system')
"""), {
"id": str(uuid.uuid4()),
"opportunity_id": opportunity_id,
"note": "Oportunidade fechada por parecer mensagem automática/bounce sem atividade comercial.",
"payload": _json({"reason": "system_or_bounce_created_by_mistake"}),
})
return len(ids)
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--limit", type=int, default=100)
parser.add_argument("--apply", action="store_true", help="Apply changes. Without this, only prints candidates.")
args = parser.parse_args()
candidates = find_candidates(limit=args.limit)
print(f"Found {len(candidates)} candidate(s).")
for row in candidates:
print(f"- {row.get('id')} | {row.get('title')} | {row.get('customer_name')} | {row.get('created_at')}")
if args.apply:
total = close_candidates(candidates)
print(f"Closed {total} candidate(s).")
else:
print("Dry-run only. Re-run with --apply to close candidates.")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,158 @@
#!/usr/bin/env python3
"""Clean noisy pending Operations items created from mailbox/system messages.
Dry-run by default. With --apply, marks obvious bounce/NDR/system tasks as
``skipped`` and stores a cleanup marker in task metadata. This does not delete
messages, raw events, or Chatwoot conversations.
"""
from __future__ import annotations
import argparse
import json
from sqlalchemy import text
from app.db import engine
SQL_TERMS = [
"postmaster",
"mailer-daemon",
"mail delivery subsystem",
"mail delivery system",
"microsoft exchange",
"office 365",
"undeliverable",
"returned mail",
"delivery status notification",
"non-delivery report",
"non delivery report",
"your message couldn't be delivered",
"your message couldnt be delivered",
"recipient wasn't found",
"recipient was not found",
"unknown to address",
"delivery has failed",
"mail delivery failed",
"remote server returned",
"550 5.1.1",
"5.1.10",
"wasn't found at",
]
NOISE_ACTION_CODES = [
"IGNORE_BOUNCE",
"IGNORE_SPAM",
"NO_ACTION",
]
def _json(value: object) -> str:
return json.dumps(value or {}, ensure_ascii=False, default=str)
def find_candidates(limit: int = 200) -> list[dict]:
term_params = {f"term_{idx}": f"%{term}%" for idx, term in enumerate(SQL_TERMS)}
term_sql = " OR ".join([f"noise_text ILIKE :term_{idx}" for idx, _ in enumerate(SQL_TERMS)])
sql = text(f"""
WITH task_context AS (
SELECT
t.id::text,
t.created_at,
t.action_code,
t.route,
t.priority,
t.status,
t.conversation_id,
t.contact_id,
t.opportunity_id::text,
COALESCE(t.note, '') AS note,
COALESCE(t.action, '') AS action,
COALESCE(re.payload->'sender'->>'name', '') AS sender_name,
COALESCE(re.payload->'sender'->>'email', '') AS sender_email,
COALESCE(
re.payload->'conversation'->'additional_attributes'->>'mail_subject',
re.payload->'content_attributes'->'email'->>'subject',
re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject',
''
) AS subject,
COALESCE(m.clean_body, m.raw_body, re.payload->>'content', '') AS body,
lower(
COALESCE(t.action_code, '') || ' ' || COALESCE(t.route, '') || ' ' ||
COALESCE(t.action, '') || ' ' || COALESCE(t.note, '') || ' ' ||
COALESCE(re.payload->'sender'->>'name', '') || ' ' ||
COALESCE(re.payload->'sender'->>'email', '') || ' ' ||
COALESCE(re.payload->'conversation'->'additional_attributes'->>'mail_subject', '') || ' ' ||
COALESCE(re.payload->'content_attributes'->'email'->>'subject', '') || ' ' ||
COALESCE(re.payload->'conversation'->'messages'->0->'content_attributes'->'email'->>'subject', '') || ' ' ||
COALESCE(m.clean_body, '') || ' ' || COALESCE(m.raw_body, '') || ' ' || COALESCE(re.payload->>'content', '')
) AS noise_text
FROM tasks t
LEFT JOIN messages m ON m.id = t.message_id
LEFT JOIN raw_events re ON re.id = t.raw_event_id
WHERE t.status = 'pending'
)
SELECT id, created_at, action_code, route, priority, status, conversation_id, contact_id,
opportunity_id, sender_name, sender_email, subject, action, note
FROM task_context
WHERE ({term_sql} OR upper(coalesce(action_code,'')) = ANY(:noise_action_codes))
ORDER BY created_at DESC
LIMIT :limit
""")
params = dict(term_params)
params["noise_action_codes"] = NOISE_ACTION_CODES
params["limit"] = int(limit)
with engine.begin() as conn:
rows = conn.execute(sql, params).mappings().all()
return [dict(row) for row in rows]
def apply_cleanup(candidates: list[dict]) -> int:
if not candidates:
return 0
ids = [row["id"] for row in candidates]
payload = _json({
"cleanup_reason": "operations_noise_bounce_or_system_message",
"cleanup_by": "cleanup_operations_noise",
"clientflow_version": "v4.9.0",
})
with engine.begin() as conn:
for task_id in ids:
conn.execute(text("""
UPDATE tasks
SET status = 'skipped',
updated_at = now(),
done_at = COALESCE(done_at, now()),
done_by = 'cleanup_operations_noise',
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:payload AS JSONB)
WHERE id = CAST(:task_id AS UUID)
AND status = 'pending'
"""), {"task_id": task_id, "payload": payload})
conn.execute(text("""
INSERT INTO task_events (task_id, event_type, payload, created_by)
VALUES (CAST(:task_id AS UUID), 'task_skipped_noise_cleanup', CAST(:payload AS JSONB), 'system')
"""), {"task_id": task_id, "payload": payload})
return len(ids)
def main() -> None:
parser = argparse.ArgumentParser(description="Clean pending Operations noise created from bounces/NDRs/system messages.")
parser.add_argument("--limit", type=int, default=200)
parser.add_argument("--apply", action="store_true", help="Apply cleanup. Without this, only prints candidates.")
args = parser.parse_args()
candidates = find_candidates(limit=args.limit)
print(f"Found {len(candidates)} candidate task(s).")
for row in candidates:
subject = (row.get("subject") or row.get("note") or "")[:90]
sender = row.get("sender_email") or row.get("sender_name") or row.get("contact_id") or "sem remetente"
print(f"- {row.get('id')} | {row.get('action_code')} | {sender} | {subject}")
if args.apply:
total = apply_cleanup(candidates)
print(f"Marked {total} task(s) as skipped.")
else:
print("Dry-run only. Re-run with --apply to mark candidates as skipped.")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,79 @@
#!/usr/bin/env python3
"""Ignore reconciliation items outside the intended working window.
Dry-run by default. Use this after an external sync staged too much history.
Examples:
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --days 3
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --days 3 --apply
PYTHONPATH=. python scripts/cleanup_stale_reconciliation_items.py --source odoo --days 3 --apply
"""
from __future__ import annotations
import argparse
from sqlalchemy import text
from app.db import engine
from app.reconciliation_service import (
cleanup_reconciliation_outside_window,
ensure_reconciliation_schema,
)
def _preview(*, days: int, source_system: str | None, limit: int) -> dict:
ensure_reconciliation_schema()
from datetime import datetime, timedelta, timezone
days = max(int(days or 3), 1)
cutoff = (datetime.now(timezone.utc).date() - timedelta(days=days)).isoformat()
params = {"cutoff": cutoff, "limit": int(limit)}
source_sql = ""
if source_system:
source_sql = "AND source_system = :source_system"
params["source_system"] = source_system
with engine.begin() as conn:
rows = conn.execute(text(f"""
SELECT id::text, source_system, external_type, document_number,
customer_name, document_date, title, status
FROM reconciliation_items
WHERE status IN ('open', 'needs_review')
{source_sql}
AND document_date IS NOT NULL
AND document_date < CAST(:cutoff AS DATE)
ORDER BY document_date ASC, updated_at DESC
LIMIT :limit
"""), params).mappings().all()
return {"days": days, "cutoff": cutoff, "source_system": source_system or "all", "matched": len(rows), "items": [dict(r) for r in rows]}
def main() -> None:
parser = argparse.ArgumentParser(description="Ignore stale reconciliation candidates outside a recent working window.")
parser.add_argument("--jasmin", action="store_true", help="same as --source jasmin")
parser.add_argument("--source", choices=["jasmin", "odoo", "packlink", "manual"], help="only clean one source system")
parser.add_argument("--days", type=int, default=3, help="keep open items from the last N days")
parser.add_argument("--limit", type=int, default=1000, help="maximum rows to inspect/update")
parser.add_argument("--apply", action="store_true", help="apply changes; otherwise dry-run")
args = parser.parse_args()
source = "jasmin" if args.jasmin else args.source
if args.apply:
result = cleanup_reconciliation_outside_window(days=args.days, source_system=source, limit=args.limit, actor="cleanup_stale_reconciliation_items")
else:
result = _preview(days=args.days, source_system=source, limit=args.limit)
print(f"Janela operacional: manter itens >= {result['cutoff']} ({result['days']} dias)")
print(f"Fonte: {result['source_system']}")
print(f"Encontrados para ignorar: {result['matched']}")
for row in result.get("items", [])[:50]:
print(f"- {row.get('document_date')} · {row.get('source_system')} · {row.get('external_type')} · {row.get('document_number')} · {row.get('customer_name') or ''}")
if args.apply:
print(f"Aplicado: {result['ignored']} itens marcados como ignored")
else:
print("Dry-run. Para aplicar, repetir com --apply")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,15 @@
#!/usr/bin/env python3
"""Automatic follow-up task creation is disabled in the clean architecture.
The current design keeps LLM triage action_codes short and uses
opportunity/business events for pipeline follow-up state.
"""
def main() -> int:
print("Automatic legacy follow-up task creation is disabled.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,42 @@
#!/usr/bin/env python3
"""Run the ClientFlow fiscal enrichment worker.
Typical production use:
python scripts/enrich_fiscal_customers.py --incremental --limit 100
The worker is idempotent: it enriches/suggests fiscal customers for open
opportunities missing local_customer_id and never creates opportunities.
"""
from __future__ import annotations
import argparse
import json
import os
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from app.fiscal_enrichment_service import enrich_open_opportunities, ensure_fiscal_enrichment_schema
def main() -> int:
parser = argparse.ArgumentParser(description="Enriquecer oportunidades sem cliente fiscal")
parser.add_argument("--limit", type=int, default=100, help="Máximo de oportunidades abertas a analisar")
parser.add_argument("--no-auto-apply", action="store_true", help="Criar apenas sugestões, sem auto-associação forte")
parser.add_argument("--daily", action="store_true", help="Marcar execução como diária/batch")
parser.add_argument("--incremental", action="store_true", help="Marcar execução como incremental")
args = parser.parse_args()
ensure_fiscal_enrichment_schema()
mode = "daily" if args.daily else "incremental"
result = enrich_open_opportunities(limit=args.limit, apply_safe=not args.no_auto_apply, mode=mode)
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,217 @@
#!/usr/bin/env python3
import json
import os
import sys
import urllib.request
import urllib.error
from pathlib import Path
from typing import Any, Dict, List
try:
import psycopg
except Exception:
psycopg = None
def env(name: str, default: str = "") -> str:
return os.getenv(name, default).strip()
API_URL = env("EVAL_API_URL", "http://127.0.0.1:8020/analyze")
DATASET = Path(env("EVAL_DATASET", "data/action_eval_cases.jsonl"))
PSQL_DATABASE_URL = env("PSQL_DATABASE_URL")
EVAL_CLEANUP = env("EVAL_CLEANUP", "true").lower() == "true"
EVAL_LIMIT = int(env("EVAL_LIMIT", "0") or "0")
EVAL_LLM_ONLY = env("EVAL_LLM_ONLY", "false").lower() == "true"
def load_cases() -> List[Dict[str, Any]]:
cases = []
with DATASET.open("r", encoding="utf-8") as f:
for line in f:
line = line.strip()
if line:
cases.append(json.loads(line))
if EVAL_LIMIT > 0:
return cases[:EVAL_LIMIT]
return cases
def post_json(url: str, payload: Dict[str, Any]) -> Dict[str, Any]:
raw = json.dumps(payload, ensure_ascii=False).encode("utf-8")
req = urllib.request.Request(
url,
data=raw,
headers={"Content-Type": "application/json"},
method="POST",
)
try:
with urllib.request.urlopen(req, timeout=90) as resp:
body = resp.read().decode("utf-8", errors="replace")
return {
"ok": 200 <= resp.status < 300,
"status": resp.status,
"json": json.loads(body) if body else {},
"raw": body,
}
except urllib.error.HTTPError as e:
body = e.read().decode("utf-8", errors="replace")
return {
"ok": False,
"status": e.code,
"json": None,
"raw": body,
}
def cleanup_eval_data() -> None:
if not EVAL_CLEANUP or not PSQL_DATABASE_URL or psycopg is None:
return
# Limpeza defensiva: algumas tabelas não têm source_system/source_event_id.
# Por isso verificamos as colunas reais antes de construir o DELETE.
tables = [
"integration_outbox",
"tasks",
"action_runs",
"messages",
"raw_events",
]
def table_columns(cur, table: str) -> set[str]:
cur.execute(
"""
select column_name
from information_schema.columns
where table_name = %s
""",
(table,),
)
return {row[0] for row in cur.fetchall()}
with psycopg.connect(PSQL_DATABASE_URL) as conn:
with conn.cursor() as cur:
for table in tables:
cols = table_columns(cur, table)
conditions = []
if "source_system" in cols:
conditions.append("source_system like 'action_eval%'")
if "conversation_id" in cols:
conditions.append("conversation_id like 'eval-%'")
if "contact_id" in cols:
conditions.append("contact_id like 'eval-contact-%'")
if "source_event_id" in cols:
conditions.append("source_event_id like 'eval-%'")
if "idempotency_key" in cols:
conditions.append("idempotency_key like '%eval-%'")
if "metadata" in cols:
conditions.append("metadata::text like '%action_eval%'")
if not conditions:
continue
sql = f"delete from {table} where " + " or ".join(conditions)
cur.execute(sql)
conn.commit()
def main() -> int:
cases = load_cases()
cleanup_eval_data()
results = []
ok_count = 0
print(f"Dataset: {DATASET}")
print(f"Cases: {len(cases)}")
print(f"API: {API_URL}")
print(f"EVAL_LLM_ONLY={EVAL_LLM_ONLY}")
print("---")
for i, case in enumerate(cases, start=1):
conv_id = f"eval-{case['id']}"
payload = {
"last_customer_message": case["message"],
"previous_context": case.get("context", ""),
"source": "action_eval_llm_only" if EVAL_LLM_ONLY else "action_eval",
"conversation_id": conv_id,
"contact_id": f"eval-contact-{i}",
}
result = post_json(API_URL, payload)
if not result["ok"]:
got = "HTTP_ERROR"
confidence = None
provider = None
note = result["raw"][:300]
else:
data = result["json"] or {}
decision = data.get("action_decision") or {}
usage = data.get("usage") or {}
got = decision.get("action_code") or data.get("action_result", {}).get("action_code") or "UNKNOWN"
confidence = decision.get("confidence")
provider = usage.get("provider")
note = decision.get("note")
expected = case["expected"]
passed = got == expected
ok_count += 1 if passed else 0
results.append({
"id": case["id"],
"expected": expected,
"got": got,
"ok": passed,
"confidence": confidence,
"provider": provider,
"note": note,
})
status = "OK" if passed else "FAIL"
print(f"{status:4} {case['id']}")
print(f" expected={expected}")
print(f" got ={got}")
print(f" conf ={confidence} provider={provider}")
if not passed:
print(f" note ={note}")
print("---")
accuracy = ok_count / len(cases) if cases else 0
print(f"accuracy={accuracy:.1%} ({ok_count}/{len(cases)})")
by_expected = {}
for r in results:
item = by_expected.setdefault(r["expected"], {"ok": 0, "total": 0})
item["total"] += 1
item["ok"] += 1 if r["ok"] else 0
print("--- by expected")
for code, stats in sorted(by_expected.items()):
print(f"{code}: {stats['ok']}/{stats['total']}")
Path("data/action_eval_last_results.json").write_text(
json.dumps(results, ensure_ascii=False, indent=2),
encoding="utf-8",
)
cleanup_eval_data()
return 0 if accuracy >= 0.90 else 1
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,43 @@
#!/usr/bin/env python3
"""Extract identity signals from an opportunity email body/signature.
Usage:
PYTHONPATH=. python scripts/extract_email_identity.py --opportunity-id <uuid>
PYTHONPATH=. python scripts/extract_email_identity.py --text-file /tmp/email.txt --email user@example.pt
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from app.email_identity_extraction_service import extract_email_identity, extract_identity_for_opportunity
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id")
parser.add_argument("--text-file")
parser.add_argument("--email", default="")
parser.add_argument("--subject", default="")
parser.add_argument("--no-llm", action="store_true")
parser.add_argument("--refresh", action="store_true")
args = parser.parse_args()
if args.opportunity_id:
result = extract_identity_for_opportunity(args.opportunity_id, refresh=args.refresh, use_llm=not args.no_llm)
elif args.text_file:
result = extract_email_identity(
Path(args.text_file).read_text(encoding="utf-8"),
email=args.email,
subject=args.subject,
use_llm=not args.no_llm,
)
else:
parser.error("use --opportunity-id or --text-file")
print(json.dumps(result or {}, ensure_ascii=False, indent=2, default=str))
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,60 @@
#!/usr/bin/env python3
"""Mark manually cleaned outbox failures as ignored.
Use when old Jasmin/Packlink test or duplicate failures were already handled
outside the worker and should no longer pollute /operations.
"""
from __future__ import annotations
import argparse
from sqlalchemy import text
from app.db import engine
MATCH_SQL = """
status = 'failed'
AND (
COALESCE(last_error, '') ILIKE '%limpo manualmente%'
OR COALESCE(last_error, '') ILIKE '%resolvido manualmente%'
)
"""
def main() -> int:
parser = argparse.ArgumentParser(description="Ignore outbox failures already resolved manually.")
parser.add_argument("--dry-run", action="store_true", help="Only show matching rows; do not update.")
parser.add_argument("--limit", type=int, default=200, help="Maximum rows to inspect/update.")
args = parser.parse_args()
with engine.begin() as conn:
rows = conn.execute(text(f"""
SELECT id::text, target_system, action_type, status, last_error, created_at
FROM integration_outbox
WHERE {MATCH_SQL}
ORDER BY created_at DESC
LIMIT :limit
"""), {"limit": args.limit}).mappings().all()
print(f"Matched {len(rows)} manually cleaned failed outbox item(s).")
for row in rows:
print(f"- {row['id']} {row['target_system']}.{row['action_type']} :: {row['last_error']}")
if args.dry_run or not rows:
print("Dry-run/no-op; no rows updated.")
return 0
ids = [row["id"] for row in rows]
conn.execute(text("""
UPDATE integration_outbox
SET status = 'ignored',
last_error = COALESCE(NULLIF(last_error, ''), 'Ignorado por limpeza operacional manual.'),
updated_at = now()
WHERE id::text = ANY(:ids)
"""), {"ids": ids})
print(f"Updated {len(ids)} item(s) to ignored.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,33 @@
#!/usr/bin/env python3
"""Inspect customers/opportunities involved in a duplicate NIF conflict."""
from __future__ import annotations
import argparse
import json
from sqlalchemy import text
from app.db import engine
from app.commercial_service import normalize_tax_id
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--tax-id", required=True)
args = parser.parse_args()
tax_id = normalize_tax_id(args.tax_id)
with engine.begin() as conn:
customers = conn.execute(text("""
SELECT c.id::text, c.name, c.tax_id, c.email, c.phone, c.street_name, c.postal_zone, c.city_name,
c.country, c.created_at, c.updated_at,
(SELECT count(*) FROM opportunities o WHERE o.local_customer_id = c.id)::int AS opportunities,
(SELECT count(*) FROM commercial_documents cd WHERE cd.customer_id = c.id)::int AS documents
FROM customers c
WHERE c.tax_id = :tax_id OR c.name ILIKE '%' || :tax_id || '%'
ORDER BY c.tax_id NULLS LAST, c.updated_at DESC
"""), {"tax_id": tax_id}).mappings().all()
print(json.dumps({"tax_id": tax_id, "customers": [dict(c) for c in customers]}, ensure_ascii=False, indent=2, default=str))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,42 @@
#!/usr/bin/env python3
"""Inspect Jasmin document candidates for an opportunity.
Optionally sync recent Jasmin documents first, then prints open/valid candidates first
and ignored closed/completed/cancelled documents afterwards.
"""
from __future__ import annotations
import argparse
import asyncio
import json
from decimal import Decimal
from typing import Any
def _json_default(value: Any) -> str:
if isinstance(value, Decimal):
return str(value)
return str(value)
async def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id", required=True)
parser.add_argument("--limit", type=int, default=20)
parser.add_argument("--sync", action="store_true", help="Sync recent Jasmin reconciliation candidates before inspection")
parser.add_argument("--days", type=int, default=30)
args = parser.parse_args()
if args.sync:
from app.external_reconciliation_sync import sync_jasmin_reconciliation_candidates
sync_result = await sync_jasmin_reconciliation_candidates(limit=100, days=args.days)
print(json.dumps({"sync_jasmin": sync_result}, ensure_ascii=False, indent=2, default=_json_default))
from app.jasmin_backfill_service import find_jasmin_document_candidates_for_opportunity
candidates = find_jasmin_document_candidates_for_opportunity(args.opportunity_id, limit=args.limit)
print(json.dumps({"opportunity_id": args.opportunity_id, "candidates": candidates}, ensure_ascii=False, indent=2, default=_json_default))
return 0
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))

View File

@@ -0,0 +1,38 @@
#!/usr/bin/env bash
set -euo pipefail
ROOT_DIR="${1:-/mnt/ssd/home/plx/clientflow_backend}"
SYSTEMD_DIR="/etc/systemd/system"
if [[ ! -d "$ROOT_DIR" ]]; then
echo "Project directory not found: $ROOT_DIR" >&2
exit 1
fi
cd "$ROOT_DIR"
sudo cp deploy/systemd/clientflow-outbox-jasmin.service "$SYSTEMD_DIR/"
sudo cp deploy/systemd/clientflow-outbox-jasmin.timer "$SYSTEMD_DIR/"
if [[ "${ENABLE_FISCAL_ENRICHMENT_TIMER:-false}" == "true" ]]; then
sudo cp deploy/systemd/clientflow-fiscal-enrichment.service "$SYSTEMD_DIR/"
sudo cp deploy/systemd/clientflow-fiscal-enrichment.timer "$SYSTEMD_DIR/"
fi
if [[ "${ENABLE_PACKLINK_TIMER:-false}" == "true" ]]; then
sudo cp deploy/systemd/clientflow-outbox-packlink.service "$SYSTEMD_DIR/"
sudo cp deploy/systemd/clientflow-outbox-packlink.timer "$SYSTEMD_DIR/"
fi
sudo systemctl daemon-reload
sudo systemctl enable --now clientflow-outbox-jasmin.timer
if [[ "${ENABLE_FISCAL_ENRICHMENT_TIMER:-false}" == "true" ]]; then
sudo systemctl enable --now clientflow-fiscal-enrichment.timer
fi
if [[ "${ENABLE_PACKLINK_TIMER:-false}" == "true" ]]; then
sudo systemctl enable --now clientflow-outbox-packlink.timer
fi
systemctl list-timers | grep clientflow || true

512
scripts/prepare_task.py Executable file
View File

@@ -0,0 +1,512 @@
#!/usr/bin/env python3
import argparse
import json
import os
import re
import urllib.request
import urllib.error
from typing import Any, Dict, List, Optional, Tuple
import psycopg
def env(name: str, default: str = "") -> str:
return os.getenv(name, default).strip()
PSQL_DATABASE_URL = env("PSQL_DATABASE_URL")
OPENROUTER_API_KEY = env("OPENROUTER_API_KEY")
OPENROUTER_URL = env("OPENROUTER_URL", "https://openrouter.ai/api/v1/chat/completions")
OPENROUTER_MODEL = env("EXTRACTION_MODEL", env("OPENROUTER_MODEL", "qwen/qwen3-30b-a3b"))
def extract_first_json_object(raw: str) -> str:
s = str(raw or "").strip()
if s.startswith("```"):
lines = s.splitlines()
if lines and lines[0].strip().startswith("```"):
lines = lines[1:]
if lines and lines[-1].strip().startswith("```"):
lines = lines[:-1]
s = "\n".join(lines).strip()
start = s.find("{")
if start == -1:
raise ValueError(f"no JSON object found: {s[:300]}")
in_string = False
escaped = False
depth = 0
for i in range(start, len(s)):
ch = s[i]
if escaped:
escaped = False
continue
if ch == "\\":
escaped = True
continue
if ch == '"':
in_string = not in_string
continue
if in_string:
continue
if ch == "{":
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
return s[start:i + 1]
raise ValueError(f"incomplete JSON object: {s[:500]}")
def parse_json(raw: str) -> Dict[str, Any]:
try:
return json.loads(raw)
except Exception:
return json.loads(extract_first_json_object(raw))
def strip_html(text: str) -> str:
text = re.sub(r"<br\s*/?>", "\n", text or "", flags=re.I)
text = re.sub(r"</p\s*>", "\n", text, flags=re.I)
text = re.sub(r"<[^>]+>", " ", text)
text = text.replace("&nbsp;", " ")
text = text.replace("&amp;", "&")
text = text.replace("&quot;", '"')
text = text.replace("&#39;", "'")
return re.sub(r"[ \t]+", " ", text).strip()
def remove_quoted_text(text: str) -> str:
if not text:
return ""
markers = [
"\nÀs ",
"\nEm ",
"\nOn ",
"\n-----Original Message-----",
"\nDe:",
"\nFrom:",
]
cut = len(text)
for marker in markers:
idx = text.find(marker)
if idx != -1:
cut = min(cut, idx)
lines = []
for line in text[:cut].splitlines():
if line.strip().startswith(">"):
continue
lines.append(line)
return "\n".join(lines).strip()
def extract_message_content(payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
msg = payload.get("message") or payload.get("messages") or {}
if isinstance(msg, list):
msg = msg[0] if msg else {}
if not isinstance(msg, dict):
return None
content = msg.get("content") or payload.get("content") or ""
message_type = str(msg.get("message_type") or payload.get("message_type") or "").lower()
private = bool(msg.get("private") or payload.get("private"))
if not content:
return None
return {
"content": remove_quoted_text(strip_html(str(content))),
"message_type": message_type,
"private": private,
"id": str(msg.get("id") or payload.get("id") or ""),
}
def fetch_task(conn, task_id: Optional[str], conversation_id: Optional[str]) -> Dict[str, Any]:
if task_id:
sql = """
select id::text, conversation_id, contact_id, action_code, route, status, action, note
from tasks
where id = %s
limit 1
"""
params = (task_id,)
else:
sql = """
select id::text, conversation_id, contact_id, action_code, route, status, action, note
from tasks
where conversation_id = %s
order by created_at desc
limit 1
"""
params = (conversation_id,)
with conn.cursor(row_factory=psycopg.rows.dict_row) as cur:
cur.execute(sql, params)
row = cur.fetchone()
if not row:
raise SystemExit("ERRO: task não encontrada.")
return dict(row)
def fetch_conversation_messages(conn, conversation_id: str, limit: int = 20) -> List[Dict[str, Any]]:
with conn.cursor(row_factory=psycopg.rows.dict_row) as cur:
cur.execute(
"""
select created_at, payload
from raw_events
where source_system = 'chatwoot'
and conversation_id = %s
order by created_at asc
limit %s
""",
(conversation_id, limit),
)
rows = cur.fetchall()
messages = []
for row in rows:
payload = row.get("payload") or {}
if isinstance(payload, str):
try:
payload = json.loads(payload)
except Exception:
payload = {}
msg = extract_message_content(payload)
if not msg:
continue
# Para extração, manter incoming e outgoing públicos, ignorar notas privadas.
if msg["private"]:
continue
messages.append({
"created_at": str(row["created_at"]),
**msg,
})
return messages
def build_prompt(task: Dict[str, Any], messages: List[Dict[str, Any]], prep_type: str) -> List[Dict[str, str]]:
conversation_text = "\n\n".join(
f"[{m['created_at']}] {m.get('message_type') or 'message'}:\n{m['content']}"
for m in messages
if m.get("content")
)
if prep_type == "proforma":
objective = """
Objetivo: preparar dados para emitir fatura pró-forma.
Extrai apenas dados relevantes para faturação/proforma:
cliente, empresa, email, telefone, NIF, morada fiscal, produto, quantidade, preço, condições comerciais e dados em falta para emitir a pró-forma.
Para prep_type=proforma:
- NÃO peças morada de entrega, destinatário ou telefone para transportadora, exceto se a conversa indicar que são necessários para a pró-forma.
- Morada de entrega/recolha pertence à fase de envio/recolha, não à fase de pró-forma.
- Se já houver NIF, morada fiscal, nome/empresa de faturação, email, produto e preço, missing_fields deve ser [].
"""
elif prep_type == "shipment":
objective = """
Objetivo: preparar envio.
Extrai apenas dados relevantes para logística de entrega:
morada de entrega, contacto no local, telefone, produto/equipamento, quantidade, instruções, estado do pagamento e dados em falta.
Para prep_type=shipment:
- Usa shipment.delivery_address para a morada de entrega.
- Usa shipment.recipient_name e shipment.recipient_phone para contacto da entrega.
- Os campos obrigatórios são morada de entrega, contacto, telefone, produto/equipamento, quantidade e estado do pagamento.
- NÃO coloques customer.tax_id, billing.tax_id, billing_address ou customer.email em missing_fields, exceto se forem explicitamente necessários para a transportadora.
- Se já existir morada de entrega, contacto e telefone, não peças esses dados novamente.
- Se faltar produto, quantidade ou comprovativo/estado de pagamento, pede apenas esses dados.
- A suggested_reply deve falar em envio/entrega, nunca em recolha.
"""
elif prep_type == "pickup":
objective = """
Objetivo: preparar recolha.
Extrai apenas dados relevantes para recolha ou assistência logística:
morada de recolha, contacto no local, telefone, produto/equipamento a recolher, motivo/instruções, data preferida e dados em falta.
Para prep_type=pickup:
- Usa shipment.pickup_address para a morada de recolha.
- Usa shipment.recipient_name e shipment.recipient_phone para a pessoa de contacto da recolha.
- Se a conversa só tiver uma morada e o objetivo é recolha, coloca essa morada em shipment.pickup_address, não em shipment.delivery_address.
- NÃO coloques sale.total_estimate, customer.tax_id, billing.tax_id, billing_address ou customer.email em missing_fields.
- Os campos importantes são pickup_address, recipient_name, recipient_phone, produto/equipamento, motivo/instruções da recolha.
- payment.status só é obrigatório se a tarefa for claramente sobre pagamento ou envio após pagamento.
- A suggested_reply deve falar em recolha, nunca em envio.
"""
else:
objective = """
Objetivo: preparar execução operacional da tarefa.
Extrai dados úteis para a ação, dados em falta e resposta sugerida.
"""
system = f"""
És o ClientFlow Sales Assistant.
A tua função é extrair dados operacionais de conversas B2B para ajudar a preparar pró-forma, fatura, envio ou recolha.
Regras:
- Devolve apenas JSON puro.
- Não inventes dados.
- Se um dado não existir, usa null.
- Ignora texto citado antigo, dados da BLIF, IBANs e assinaturas da BLIF.
- Não associes IBAN ao cliente.
- Distingue morada fiscal de morada de entrega/recolha.
- Usa evidências curtas.
- Se faltar dado necessário, coloca em missing_fields.
- suggested_reply deve ser uma resposta curta em português para pedir dados em falta ou indicar o próximo passo.
- Não digas que a pró-forma/fatura/envio já foi emitida, enviada, agendada ou concluída.
- O assistente apenas prepara dados para revisão; usa linguagem como "vamos preparar", "podemos avançar", "dados suficientes para preparar".
"""
user = f"""
{objective}
Tarefa:
action_code: {task.get('action_code')}
route: {task.get('route')}
action: {task.get('action')}
note: {task.get('note')}
Conversa:
{conversation_text}
Schema obrigatório:
{{
"customer": {{
"name": null,
"company": null,
"email": null,
"phone": null,
"tax_id": null
}},
"billing": {{
"billing_name": null,
"tax_id": null,
"billing_address": null,
"billing_email": null
}},
"sale": {{
"products": [
{{
"name": null,
"description": null,
"quantity": null,
"unit_price": null,
"currency": "EUR"
}}
],
"total_estimate": null,
"commercial_terms": null
}},
"payment": {{
"status": null,
"proof_mentioned": false
}},
"shipment": {{
"delivery_address": null,
"pickup_address": null,
"recipient_name": null,
"recipient_phone": null,
"instructions": null
}},
"missing_fields": [],
"suggested_reply": "",
"confidence": 0.0,
"evidence": []
}}
"""
return [
{"role": "system", "content": system},
{"role": "user", "content": user},
]
def call_openrouter(messages: List[Dict[str, str]]) -> Tuple[Dict[str, Any], Dict[str, Any], str]:
body = {
"model": OPENROUTER_MODEL,
"messages": messages,
"temperature": 0,
}
req = urllib.request.Request(
OPENROUTER_URL,
data=json.dumps(body, ensure_ascii=False).encode("utf-8"),
headers={
"Authorization": f"Bearer {OPENROUTER_API_KEY}",
"Content-Type": "application/json",
"HTTP-Referer": "https://clientflow.blif.pt",
"X-Title": "ClientFlow",
},
method="POST",
)
try:
with urllib.request.urlopen(req, timeout=120) as resp:
raw = resp.read().decode("utf-8", errors="replace")
parsed = json.loads(raw)
except urllib.error.HTTPError as e:
raw = e.read().decode("utf-8", errors="replace")
raise RuntimeError(f"OpenRouter HTTP {e.code}: {raw[:1000]}")
content = parsed["choices"][0]["message"].get("content") or "{}"
extracted = parse_json(content)
return extracted, parsed, content
def save_preparation(conn, task: Dict[str, Any], prep_type: str, extracted: Dict[str, Any], raw_response: Dict[str, Any]) -> str:
usage = raw_response.get("usage") or {}
model = raw_response.get("model") or OPENROUTER_MODEL
provider = raw_response.get("provider") or raw_response.get("provider_name")
missing_fields = extracted.get("missing_fields") or []
suggested_reply = extracted.get("suggested_reply") or ""
confidence = extracted.get("confidence")
with conn.cursor() as cur:
cur.execute(
"""
insert into task_preparations (
task_id,
conversation_id,
contact_id,
prep_type,
status,
extracted_data,
missing_fields,
suggested_reply,
confidence,
model,
provider,
total_tokens,
cost,
raw_response
)
values (
%s, %s, %s, %s, 'draft',
%s::jsonb,
%s::jsonb,
%s,
%s,
%s,
%s,
%s,
%s,
%s::jsonb
)
returning id::text
""",
(
task["id"],
task["conversation_id"],
task.get("contact_id"),
prep_type,
json.dumps(extracted, ensure_ascii=False),
json.dumps(missing_fields, ensure_ascii=False),
suggested_reply,
confidence,
model,
provider,
int(usage.get("total_tokens") or 0),
usage.get("cost") or 0,
json.dumps(raw_response, ensure_ascii=False),
),
)
prep_id = cur.fetchone()[0]
conn.commit()
return prep_id
def run_preparation(
*,
task_id: Optional[str] = None,
conversation_id: Optional[str] = None,
prep_type: str,
database_url: Optional[str] = None,
) -> Dict[str, Any]:
db_url = database_url or PSQL_DATABASE_URL
if not task_id and not conversation_id:
raise ValueError("Usa task_id ou conversation_id.")
if not db_url:
raise RuntimeError("PSQL_DATABASE_URL/DATABASE_URL não definida.")
if not OPENROUTER_API_KEY:
raise RuntimeError("OPENROUTER_API_KEY não definida.")
with psycopg.connect(db_url) as conn:
task = fetch_task(conn, task_id, conversation_id)
messages = fetch_conversation_messages(conn, task["conversation_id"])
if not messages:
raise RuntimeError("Não encontrei mensagens públicas da conversa.")
prompt = build_prompt(task, messages, prep_type)
extracted, raw_response, raw_content = call_openrouter(prompt)
prep_id = save_preparation(conn, task, prep_type, extracted, raw_response)
return {
"preparation_id": prep_id,
"task_id": task["id"],
"conversation_id": task["conversation_id"],
"prep_type": prep_type,
"extracted": extracted,
}
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--task-id")
parser.add_argument("--conversation-id")
parser.add_argument("--type", choices=["proforma", "shipment", "pickup", "generic"], required=True)
args = parser.parse_args()
if not args.task_id and not args.conversation_id:
raise SystemExit("Usa --task-id ou --conversation-id.")
if not PSQL_DATABASE_URL:
raise SystemExit("PSQL_DATABASE_URL não definida.")
if not OPENROUTER_API_KEY:
raise SystemExit("OPENROUTER_API_KEY não definida.")
with psycopg.connect(PSQL_DATABASE_URL) as conn:
task = fetch_task(conn, args.task_id, args.conversation_id)
messages = fetch_conversation_messages(conn, task["conversation_id"])
if not messages:
raise SystemExit("ERRO: não encontrei mensagens públicas da conversa.")
prompt = build_prompt(task, messages, args.type)
extracted, raw_response, raw_content = call_openrouter(prompt)
prep_id = save_preparation(conn, task, args.type, extracted, raw_response)
print(f"OK: preparation_id={prep_id}")
print(f"task_id={task['id']}")
print(f"conversation_id={task['conversation_id']}")
print(f"prep_type={args.type}")
print("--- extracted")
print(json.dumps(extracted, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())

318
scripts/process_outbox.py Normal file
View File

@@ -0,0 +1,318 @@
import asyncio
import os
import sys
from pathlib import Path
from typing import Any, Dict, Literal
import httpx
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
from app.config import settings
from app.integration_outbox_service import (
claim_pending_outbox,
recover_stale_processing_outbox,
mark_outbox_blocked,
mark_outbox_dry_run,
mark_outbox_failed,
mark_outbox_sent,
)
Outcome = Literal["processed", "skipped"]
def env_bool(name: str, default: bool = False) -> bool:
fallback = "true" if default else "false"
value = os.getenv(name, fallback).strip().lower()
return value in {"true", "1", "yes", "on"}
def is_dry_run() -> bool:
return env_bool("OUTBOX_DRY_RUN", True)
def integration_enabled(target_system: str) -> bool:
key = f"{str(target_system or '').upper()}_OUTBOX_ENABLED"
return env_bool(key, False)
def build_chatwoot_note(payload: Dict[str, Any]) -> str:
event_type = payload.get("event_type", "")
action = payload.get("action", "")
note = payload.get("note", "")
conversation_id = payload.get("conversation_id", "")
return f"""🤖 ClientFlow
Evento:
{event_type}
Ação:
{action}
Nota:
{note}
Conversa:
{conversation_id}
"""
async def process_chatwoot_add_private_note(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
conversation_id = payload.get("conversation_id")
if not conversation_id:
raise RuntimeError("conversation_id em falta no payload")
if is_dry_run():
message = f"DRY-RUN chatwoot.add_private_note conversation_id={conversation_id}"
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
if not settings.chatwoot_write_enabled:
raise RuntimeError("CHATWOOT_WRITE_ENABLED=false")
if not settings.chatwoot_base_url or not settings.chatwoot_account_id or not settings.chatwoot_api_token:
raise RuntimeError("Configuração Chatwoot incompleta")
url = (
settings.chatwoot_base_url.rstrip("/")
+ f"/api/v1/accounts/{settings.chatwoot_account_id}"
+ f"/conversations/{conversation_id}/messages"
)
body = {
"content": build_chatwoot_note(payload),
"message_type": "outgoing",
"private": True,
"content_type": "text",
"content_attributes": {},
}
headers = {
"Content-Type": "application/json",
"api_access_token": settings.chatwoot_api_token,
}
async with httpx.AsyncClient(timeout=30) as client:
response = await client.post(url, headers=headers, json=body)
if response.status_code >= 400:
raise RuntimeError(f"Chatwoot error {response.status_code}: {response.text}")
mark_outbox_sent(item["id"])
return "processed"
async def process_mautic_add_tag(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
if is_dry_run():
message = (
"DRY-RUN mautic.add_tag "
f"conversation_id={payload.get('conversation_id')} "
f"tag={payload.get('tag')}"
)
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
from app.mautic_client import add_tag_from_outbox_payload
add_tag_from_outbox_payload(payload)
mark_outbox_sent(item["id"])
return "processed"
async def process_mautic_remove_tag(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
if is_dry_run():
message = (
"DRY-RUN mautic.remove_tag "
f"conversation_id={payload.get('conversation_id')} "
f"tag={payload.get('tag')}"
)
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
from app.mautic_client import remove_tag_from_outbox_payload
remove_tag_from_outbox_payload(payload)
mark_outbox_sent(item["id"])
return "processed"
async def process_packlink_create_shipment(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
opportunity_id = payload.get("opportunity_id")
if is_dry_run():
message = f"DRY-RUN packlink.create_shipment opportunity_id={opportunity_id}"
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
if not settings.packlink_enabled:
raise RuntimeError("PACKLINK_ENABLED=false")
from app.packlink_service import create_shipment_from_outbox_payload
result = await create_shipment_from_outbox_payload(payload)
print(f"Packlink shipment created reference={result.get('reference')}")
mark_outbox_sent(item["id"])
return "processed"
async def process_jasmin_create_quotation(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
opportunity_id = payload.get("opportunity_id")
if is_dry_run():
message = f"DRY-RUN jasmin.create_quotation opportunity_id={opportunity_id}"
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
if not settings.jasmin_enabled:
raise RuntimeError("JASMIN_ENABLED=false")
from app.jasmin_service import process_create_quotation_outbox
result = await process_create_quotation_outbox(payload)
print(f"Jasmin quotation created id={result.get('quotation_id')}")
mark_outbox_sent(item["id"])
return "processed"
async def process_jasmin_convert_invoice(item: Dict[str, Any]) -> Outcome:
payload = item.get("payload") or {}
opportunity_id = payload.get("opportunity_id")
if is_dry_run():
message = f"DRY-RUN jasmin.convert_quotation_to_invoice opportunity_id={opportunity_id}"
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
if not settings.jasmin_enabled:
raise RuntimeError("JASMIN_ENABLED=false")
from app.jasmin_service import process_convert_invoice_outbox
result = await process_convert_invoice_outbox(payload)
print(f"Jasmin invoice created id={result.get('invoice_id')}")
mark_outbox_sent(item["id"])
return "processed"
async def process_item(item: Dict[str, Any]) -> Outcome:
target_system = item.get("target_system")
action_type = item.get("action_type")
print(f"Processing {item['id']} {target_system}.{action_type}")
if not integration_enabled(target_system):
message = f"Integração desativada: {target_system}.{action_type}. Ative {str(target_system or '').upper()}_OUTBOX_ENABLED=true para processar."
print(f"BLOCKED {message}")
mark_outbox_blocked(item["id"], message)
return "skipped"
if target_system == "chatwoot" and action_type == "add_private_note":
return await process_chatwoot_add_private_note(item)
if target_system == ("t" + "wenty"):
print(f"SKIP legacy external CRM outbox item {item.get('id')}: integração removida")
return "skipped"
if target_system == "mautic" and action_type == "add_tag":
return await process_mautic_add_tag(item)
if target_system == "mautic" and action_type == "remove_tag":
return await process_mautic_remove_tag(item)
if target_system == "packlink" and action_type == "create_shipment":
return await process_packlink_create_shipment(item)
if target_system == "jasmin" and action_type == "create_quotation":
return await process_jasmin_create_quotation(item)
if target_system == "jasmin" and action_type == "convert_quotation_to_invoice":
return await process_jasmin_convert_invoice(item)
if is_dry_run():
message = f"DRY-RUN unsupported-now {target_system}.{action_type}"
print(message)
mark_outbox_dry_run(item["id"], message)
return "processed"
raise RuntimeError(f"Handler não implementado: {target_system}.{action_type}")
async def main() -> int:
# Nota: as notas privadas do Chatwoot podem estar desligadas sem bloquear
# outras integrações como Packlink, Mautic ou Jasmin. A decisão de processar
# cada target_system fica em integration_enabled() e no handler específico.
if os.getenv("CLIENTFLOW_DISABLE_CHATWOOT_PRIVATE_NOTES", "true").lower() in {"1", "true", "yes", "sim"}:
print("ClientFlow Chatwoot private notes disabled by env; non-Chatwoot outbox will still run.")
limit = int(os.getenv("OUTBOX_LIMIT", "50"))
target_system = os.getenv("OUTBOX_TARGET_SYSTEM", "").strip() or None
worker_id = os.getenv("OUTBOX_WORKER_ID", f"process_outbox:{os.getpid()}")
if env_bool("OUTBOX_RECOVER_STALE_BEFORE_PROCESS", True):
recovered = recover_stale_processing_outbox(
mode=os.getenv("OUTBOX_STALE_RECOVERY_MODE", "manual_only"),
actor=worker_id,
)
if recovered:
print(f"Recovered stale processing outbox items: {len(recovered)}")
items = claim_pending_outbox(
limit=limit,
target_system=target_system,
lock_owner=worker_id,
)
print(f"Claimed outbox items: {len(items)}")
print(f"OUTBOX_WORKER_ID={worker_id}")
print(f"OUTBOX_DRY_RUN={is_dry_run()}")
print(f"OUTBOX_TARGET_SYSTEM={target_system or 'all'}")
processed = 0
skipped = 0
failed = 0
for item in items:
try:
outcome = await process_item(item)
if outcome == "processed":
processed += 1
else:
skipped += 1
except Exception as exc:
failed += 1
print(f"FAILED {item.get('id')}: {exc}")
mark_outbox_failed(item["id"], str(exc))
print(f"Processed: {processed}")
print(f"Skipped: {skipped}")
print(f"Failed: {failed}")
return 0 if failed == 0 else 1
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))

41
scripts/recover_stale_outbox.py Executable file
View File

@@ -0,0 +1,41 @@
#!/usr/bin/env python3
"""Recover or expose stale outbox processing rows.
Usage examples:
OUTBOX_STALE_RECOVERY_MODE=manual_only python scripts/recover_stale_outbox.py
OUTBOX_STALE_RECOVERY_MODE=retry_pending python scripts/recover_stale_outbox.py
"""
from __future__ import annotations
import os
import sys
from pathlib import Path
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
from app.integration_outbox_service import recover_stale_processing_outbox, outbox_stale_minutes
def main() -> int:
mode = os.getenv("OUTBOX_STALE_RECOVERY_MODE", "manual_only")
actor = os.getenv("OUTBOX_WORKER_ID", f"recover_stale_outbox:{os.getpid()}")
limit = int(os.getenv("OUTBOX_STALE_RECOVERY_LIMIT", "100"))
stale_minutes = outbox_stale_minutes()
recovered = recover_stale_processing_outbox(
mode=mode,
actor=actor,
limit=limit,
stale_minutes=stale_minutes,
)
print(f"Mode: {mode}")
print(f"Stale threshold minutes: {stale_minutes}")
print(f"Recovered/exposed stale items: {len(recovered)}")
for item in recovered:
print(f"- {item.get('id')} {item.get('target_system')}.{item.get('action_type')} -> {item.get('status')}")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,71 @@
#!/usr/bin/env python3
"""Reabre tasks Chatwoot que foram classificadas como revisão/remoção mas ficaram skipped.
Uso seguro:
PYTHONPATH=. python scripts/reopen_chatwoot_review_tasks.py --dry-run
PYTHONPATH=. python scripts/reopen_chatwoot_review_tasks.py --days 7
Por defeito só olha para os últimos 7 dias e não toca em spam/NO_ACTION.
"""
from __future__ import annotations
import argparse
import json
from sqlalchemy import text
from app.db import engine
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--days", type=int, default=7, help="Janela de dias a corrigir")
parser.add_argument("--dry-run", action="store_true", help="Mostra o que faria sem alterar")
args = parser.parse_args()
params = {"days": int(args.days)}
select_sql = text("""
SELECT id::text, created_at, action_code, route, action, status, conversation_id, contact_id
FROM tasks
WHERE source_system = 'chatwoot'
AND status = 'skipped'
AND action_code IN ('REVIEW_MANUALLY', 'REMOVE_FROM_LIST')
AND created_at >= now() - (:days * interval '1 day')
ORDER BY created_at DESC
""")
with engine.begin() as conn:
rows = conn.execute(select_sql, params).mappings().all()
print(f"Encontradas {len(rows)} task(s) Chatwoot a reabrir.")
for row in rows[:50]:
print(f"- {row['created_at']} {row['action_code']} conversa={row['conversation_id']} task={row['id']}")
if args.dry_run or not rows:
print("Dry-run: nenhuma alteração aplicada." if args.dry_run else "Nada para alterar.")
return 0
ids = [row["id"] for row in rows]
update_sql = text("""
UPDATE tasks
SET status = 'pending',
priority = CASE WHEN action_code = 'REVIEW_MANUALLY' THEN 'alta' ELSE COALESCE(priority, 'normal') END,
route = CASE WHEN action_code = 'REMOVE_FROM_LIST' THEN 'marketing' ELSE route END,
updated_at = now(),
metadata = COALESCE(metadata, '{}'::jsonb) || CAST(:patch AS JSONB)
WHERE id = ANY(CAST(:ids AS uuid[]))
""")
patch = json.dumps({
"v46_reopened": True,
"v46_reason": "REVIEW_MANUALLY/REMOVE_FROM_LIST devem gerar trabalho humano pendente",
}, ensure_ascii=False)
with engine.begin() as conn:
conn.execute(update_sql, {"ids": ids, "patch": patch})
print(f"Reabertas {len(ids)} task(s) como pending.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,89 @@
"""Repair Odoo sale-order references accidentally stored as customer names.
v4.9.25.3 fixes the source of the issue. This script repairs rows already
created by older v4.9.25.x builds where customers.name became S00xxx although
metadata.raw_customer_payload.partner_name contains the real fiscal customer.
Dry-run by default. Use --apply to update rows.
"""
from __future__ import annotations
import argparse
from typing import Any, Dict, List
from sqlalchemy import text
from app.db import engine
def _partner_name_from_metadata(metadata: Any) -> str:
if not isinstance(metadata, dict):
return ""
raw = metadata.get("raw_customer_payload")
if not isinstance(raw, dict):
return ""
partner_name = str(raw.get("partner_name") or "").strip()
if partner_name:
return partner_name
partner_id = raw.get("partner_id")
if isinstance(partner_id, (list, tuple)) and len(partner_id) > 1:
return str(partner_id[1] or "").strip()
return ""
def find_rows() -> List[Dict[str, Any]]:
with engine.begin() as conn:
rows = conn.execute(text("""
SELECT id::text, name, tax_id, email, metadata
FROM customers
WHERE name ~ '^S[0-9]{4,}'
AND metadata->>'source_system' = 'odoo'
ORDER BY updated_at DESC
""")).mappings().all()
result: List[Dict[str, Any]] = []
for row in rows:
partner_name = _partner_name_from_metadata(row.get("metadata"))
if partner_name and not partner_name.upper().startswith("S00"):
data = dict(row)
data["new_name"] = partner_name
result.append(data)
return result
def apply(rows: List[Dict[str, Any]]) -> int:
updated = 0
with engine.begin() as conn:
for row in rows:
conn.execute(text("""
UPDATE customers
SET name = :new_name,
metadata = COALESCE(metadata, '{}'::jsonb) || jsonb_build_object(
'odoo_sale_order_name_repaired', true,
'previous_customer_name', CAST(:old_name AS TEXT)
),
updated_at = now()
WHERE id = CAST(:id AS UUID)
"""), {"id": row["id"], "old_name": row["name"], "new_name": row["new_name"]})
updated += 1
return updated
def main() -> None:
parser = argparse.ArgumentParser(description="Repair Odoo S00xxx customer names.")
parser.add_argument("--apply", action="store_true", help="Apply updates. Default is dry-run.")
args = parser.parse_args()
rows = find_rows()
print(f"Clientes Odoo com nome S00xxx reparáveis: {len(rows)}")
for row in rows[:50]:
print(f"- {row['name']} -> {row['new_name']} | NIF {row.get('tax_id') or '-'} | email {row.get('email') or '-'}")
if len(rows) > 50:
print(f"... mais {len(rows)-50}")
if not args.apply:
print("Dry-run. Para aplicar: repetir com --apply")
return
print(f"Atualizados: {apply(rows)}")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,44 @@
#!/usr/bin/env python3
"""Replace imported Jasmin quotation/proforma details for an opportunity.
Use this when an old/closed quotation was linked by mistake and a newer open
reconciliation item should become the active document for the opportunity.
This deletes only ClientFlow imported Jasmin quotation/proforma artifacts; it does
not delete documents in Jasmin.
"""
from __future__ import annotations
import argparse
import asyncio
import json
from decimal import Decimal
from typing import Any
def _json_default(value: Any) -> str:
if isinstance(value, Decimal):
return str(value)
return str(value)
async def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--opportunity-id", required=True)
parser.add_argument("--item-id", required=True, help="reconciliation_items.id for the open/valid Jasmin candidate")
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
from app.jasmin_backfill_service import replace_jasmin_document_for_opportunity_async
result = await replace_jasmin_document_for_opportunity_async(
opportunity_id=args.opportunity_id,
item_id=args.item_id,
actor="operator_cli_replace_jasmin_document",
dry_run=args.dry_run,
)
print(json.dumps(result, ensure_ascii=False, indent=2, default=_json_default))
return 0 if result.get("ok") else 2
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))

View File

@@ -0,0 +1,133 @@
#!/usr/bin/env python3
"""Reset generated reconciliation staging items so they can be rebuilt.
Safe by default: dry-run only and only affects generated external candidates
from Jasmin/Odoo/Packlink in open/needs_review/ignored states. It does not
remove opportunities, commercial documents, payment proofs, operation links or
external system records.
Examples:
PYTHONPATH=. python scripts/reset_reconciliation_generated.py
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --apply
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --source jasmin --source odoo --apply
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --days 30 --apply
PYTHONPATH=. python scripts/reset_reconciliation_generated.py --include-manual --apply
"""
from __future__ import annotations
import argparse
from datetime import datetime, timedelta, timezone
from typing import Any, Dict, List
from sqlalchemy import text
from app.db import engine, init_db
from app.reconciliation_service import ensure_reconciliation_schema
DEFAULT_SOURCES = ["jasmin", "odoo", "packlink"]
DEFAULT_STATUSES = ["open", "needs_review", "ignored"]
def _backup_table_name() -> str:
return "reconciliation_items_reset_backup_" + datetime.now(timezone.utc).strftime("%Y%m%d_%H%M%S")
def _build_where(args: argparse.Namespace) -> tuple[str, Dict[str, Any]]:
sources = list(args.source or DEFAULT_SOURCES)
if args.include_manual and "manual" not in sources:
sources.append("manual")
statuses = list(args.status or DEFAULT_STATUSES)
params: Dict[str, Any] = {"sources": sources, "statuses": statuses}
clauses = [
"source_system = ANY(CAST(:sources AS TEXT[]))",
"status = ANY(CAST(:statuses AS TEXT[]))",
]
# Extra guard: never touch items already linked to an opportunity unless
# the operator explicitly changes the status list and unlinks manually.
clauses.append("opportunity_id IS NULL")
if args.days is not None:
days = max(int(args.days), 1)
cutoff = (datetime.now(timezone.utc).date() - timedelta(days=days - 1)).isoformat()
params["cutoff"] = cutoff
clauses.append("(document_date IS NULL OR document_date >= CAST(:cutoff AS DATE))")
return " AND ".join(clauses), params
def _summarize(where_sql: str, params: Dict[str, Any], limit: int) -> Dict[str, Any]:
with engine.begin() as conn:
counts = conn.execute(text(f"""
SELECT source_system, external_type, status, COUNT(*) AS total
FROM reconciliation_items
WHERE {where_sql}
GROUP BY source_system, external_type, status
ORDER BY source_system, external_type, status
"""), params).mappings().all()
rows = conn.execute(text(f"""
SELECT id::text, source_system, external_type, status, document_number,
customer_name, customer_tax_id, document_date, amount, title
FROM reconciliation_items
WHERE {where_sql}
ORDER BY updated_at DESC, created_at DESC
LIMIT :limit
"""), {**params, "limit": int(limit)}).mappings().all()
return {"counts": [dict(r) for r in counts], "items": [dict(r) for r in rows]}
def _apply_reset(where_sql: str, params: Dict[str, Any]) -> Dict[str, Any]:
backup_table = _backup_table_name()
# backup_table is generated internally from digits/underscore only.
with engine.begin() as conn:
total = conn.execute(text(f"SELECT COUNT(*) FROM reconciliation_items WHERE {where_sql}"), params).scalar() or 0
conn.execute(text(f"CREATE TABLE {backup_table} AS SELECT * FROM reconciliation_items WHERE {where_sql}"), params)
deleted = conn.execute(text(f"DELETE FROM reconciliation_items WHERE {where_sql}"), params).rowcount or 0
return {"matched": int(total), "deleted": int(deleted), "backup_table": backup_table}
def main() -> None:
parser = argparse.ArgumentParser(description="Reset generated reconciliation candidates and keep a DB backup table.")
parser.add_argument("--source", action="append", choices=["jasmin", "odoo", "packlink", "manual"], help="source_system to reset; repeatable. Default: jasmin, odoo, packlink")
parser.add_argument("--status", action="append", choices=["open", "needs_review", "ignored"], help="status to reset; repeatable. Default: open, needs_review, ignored")
parser.add_argument("--days", type=int, help="only reset candidates inside the last N days; default is all dates")
parser.add_argument("--include-manual", action="store_true", help="also include source_system=manual; use with care")
parser.add_argument("--limit", type=int, default=50, help="preview sample size")
parser.add_argument("--apply", action="store_true", help="delete matched staging rows after creating a backup table")
args = parser.parse_args()
init_db()
ensure_reconciliation_schema()
where_sql, params = _build_where(args)
summary = _summarize(where_sql, params, args.limit)
print("Alvo do reset:")
print(f" fontes: {', '.join(params['sources'])}")
print(f" estados: {', '.join(params['statuses'])}")
print(" proteção: opportunity_id IS NULL")
if args.days is not None:
print(f" janela: >= {params['cutoff']} ({args.days} dias)")
print("\nContagens:")
if not summary["counts"]:
print(" 0 itens encontrados")
for row in summary["counts"]:
print(f" {row['source_system']} · {row['external_type']} · {row['status']}: {row['total']}")
print("\nAmostra:")
for row in summary["items"]:
print(
f" {row.get('document_date') or '-'} · {row.get('source_system')} · {row.get('external_type')} · "
f"{row.get('status')} · {row.get('document_number') or '-'} · {row.get('customer_name') or '-'}"
)
if not args.apply:
print("\nDry-run. Para aplicar: repetir com --apply")
return
result = _apply_reset(where_sql, params)
print("\nReset aplicado:")
print(f" encontrados: {result['matched']}")
print(f" apagados: {result['deleted']}")
print(f" backup: {result['backup_table']}")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,89 @@
#!/usr/bin/env python3
"""List/revert risky fiscal suggestions auto-applied from domain-only matches."""
from __future__ import annotations
import argparse
import json
from sqlalchemy import text
from app.db import engine
DOMAIN_MATCHES = (
"email_principal_dominio",
"email_dominio_empresa_associada",
"contacto_email_dominio",
"dominio",
"email_dominio",
"website_dominio",
)
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--apply", action="store_true", help="revert accepted domain-only suggestions and unlink matching opportunity customer")
args = parser.parse_args()
with engine.begin() as conn:
rows = conn.execute(text("""
SELECT
s.id::text,
s.opportunity_id::text,
s.suggested_customer_id::text,
s.suggested_name,
s.suggested_nif,
s.match_type,
s.confidence,
o.customer_name,
o.customer_email,
o.local_customer_id::text AS current_customer_id
FROM fiscal_customer_suggestions s
JOIN opportunities o ON o.id = s.opportunity_id
WHERE s.status = 'accepted'
AND s.auto_applied = TRUE
AND s.match_type = ANY(:matches)
ORDER BY s.updated_at DESC
"""), {"matches": list(DOMAIN_MATCHES)}).mappings().all()
print(f"Sugestões por domínio auto-aplicadas: {len(rows)}")
for r in rows:
print(f"- {r['customer_name']} <{r['customer_email']}> -> {r['suggested_name']} / {r['suggested_nif']} | {r['match_type']} | {r['confidence']} | opp={r['opportunity_id']}")
if not args.apply:
print("Dry-run. Para reverter: repetir com --apply")
return
updated = 0
with engine.begin() as conn:
for r in rows:
# Only unlink if the opportunity is still linked to the same customer suggested by this risky suggestion.
if r["suggested_customer_id"] and r["current_customer_id"] == r["suggested_customer_id"]:
conn.execute(text("""
UPDATE opportunities
SET local_customer_id = NULL,
metadata = COALESCE(metadata, '{}'::jsonb) || jsonb_build_object(
'domain_match_auto_apply_reverted', true,
'domain_match_reverted_suggestion_id', :suggestion_id,
'domain_match_reverted_customer_name', :customer_name,
'domain_match_reverted_at', now()
),
updated_at = now()
WHERE id = CAST(:opportunity_id AS UUID)
"""), {
"opportunity_id": r["opportunity_id"],
"suggestion_id": r["id"],
"customer_name": r["suggested_name"],
})
conn.execute(text("""
UPDATE fiscal_customer_suggestions
SET status = 'rejected', auto_applied = FALSE, resolved_by = 'domain_match_safety_review',
resolved_at = now(), reason = COALESCE(reason, '') || ' | reverted: domain-only auto-apply is unsafe',
updated_at = now()
WHERE id = CAST(:id AS UUID)
"""), {"id": r["id"]})
updated += 1
print(f"Revertidas: {updated}")
if __name__ == "__main__":
main()

4
scripts/run_dev.sh Executable file
View File

@@ -0,0 +1,4 @@
#!/usr/bin/env bash
set -euo pipefail
python -m uvicorn app.main:app --reload --host 127.0.0.1 --port 8000

View File

@@ -0,0 +1,48 @@
#!/usr/bin/env python3
"""Run the recommended ClientFlow pipeline: enrichment -> external sync.
This orchestrator is safe to run periodically. It does not apply low-confidence
reconciliation decisions; it prepares fiscal identities first and then rebuilds
external candidates for the operator.
"""
from __future__ import annotations
import argparse
import asyncio
import json
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parents[1]
if str(ROOT) not in sys.path:
sys.path.insert(0, str(ROOT))
from app.fiscal_enrichment_service import enrich_open_opportunities, ensure_fiscal_enrichment_schema
from app.external_reconciliation_sync import sync_all_external_reconciliation_candidates
async def _run(args: argparse.Namespace) -> dict:
ensure_fiscal_enrichment_schema()
enrichment = enrich_open_opportunities(
limit=args.enrichment_limit,
apply_safe=not args.no_auto_apply,
mode="pipeline",
)
reconciliation = await sync_all_external_reconciliation_candidates(limit=args.limit, days=args.days)
return {"enrichment": enrichment, "reconciliation": reconciliation}
def main() -> int:
parser = argparse.ArgumentParser(description="Correr pipeline ClientFlow: enriquecimento fiscal + reconciliação")
parser.add_argument("--days", type=int, default=7, help="Janela de reconciliação externa")
parser.add_argument("--limit", type=int, default=100, help="Limite por fonte externa")
parser.add_argument("--enrichment-limit", type=int, default=100, help="Limite de oportunidades a enriquecer antes da reconciliação")
parser.add_argument("--no-auto-apply", action="store_true", help="Não auto-associar sugestões fiscais fortes")
args = parser.parse_args()
result = asyncio.run(_run(args))
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,77 @@
#!/usr/bin/env python3
"""Sync external systems into reconciliation candidates.
Examples:
PYTHONPATH=. python scripts/sync_external_reconciliation.py --all
PYTHONPATH=. python scripts/sync_external_reconciliation.py --jasmin --limit 50 --days 3
PYTHONPATH=. python scripts/sync_external_reconciliation.py --odoo --days 3
PYTHONPATH=. python scripts/sync_external_reconciliation.py --customers --limit 200
The full --all pipeline first creates/updates fiscal customers from Jasmin/Odoo
and then stages reconciliation_items. It never creates opportunities or confirms
payments, so the operator can link/create/ignore in the UI.
"""
from __future__ import annotations
import argparse
import asyncio
import json
from typing import Any, Dict, List
from app.external_reconciliation_sync import (
sync_all_external_reconciliation_candidates,
sync_external_fiscal_customers_for_reconciliation,
sync_jasmin_reconciliation_candidates,
sync_odoo_reconciliation_candidates,
sync_packlink_reconciliation_candidates,
)
def _print(result: Dict[str, Any]) -> None:
print(json.dumps(result, ensure_ascii=False, indent=2, default=str))
async def main() -> None:
parser = argparse.ArgumentParser(description="Sync external APIs into ClientFlow reconciliation candidates.")
parser.add_argument("--all", action="store_true", help="sync all enabled external systems")
parser.add_argument("--jasmin", action="store_true", help="sync Jasmin quotations/invoices")
parser.add_argument("--odoo", action="store_true", help="sync Odoo sale orders")
parser.add_argument("--packlink", action="store_true", help="sync Packlink shipments")
parser.add_argument("--customers", action="store_true", help="seed fiscal customers from Jasmin/Odoo before document reconciliation")
parser.add_argument("--limit", type=int, default=50, help="maximum records per source")
parser.add_argument("--days", type=int, default=3, help="lookback window for Jasmin/Odoo/Packlink syncs")
args = parser.parse_args()
if args.all or not (args.jasmin or args.odoo or args.packlink or args.customers):
_print(await sync_all_external_reconciliation_candidates(limit=args.limit, days=args.days))
return
if args.customers and not (args.jasmin or args.odoo or args.packlink):
_print(await sync_external_fiscal_customers_for_reconciliation(limit=max(args.limit, 200)))
return
results: List[Dict[str, Any]] = []
customer_result = None
if args.customers:
customer_result = await sync_external_fiscal_customers_for_reconciliation(limit=max(args.limit, 200))
if args.jasmin:
results.append(await sync_jasmin_reconciliation_candidates(limit=args.limit, days=args.days))
if args.odoo:
# Odoo XML-RPC client is sync; the service itself stays sync for easier reuse.
results.append(sync_odoo_reconciliation_candidates(limit=args.limit, days=args.days))
if args.packlink:
results.append(await sync_packlink_reconciliation_candidates(limit=args.limit, days=args.days))
output = {
"seen": sum(int(r.get("seen") or 0) for r in results),
"created_or_updated": sum(int(r.get("created_or_updated") or 0) for r in results),
"results": results,
}
if customer_result is not None:
output["customer_seen"] = customer_result.get("seen", 0)
output["customers_created_or_updated"] = customer_result.get("created_or_updated", 0)
output["customer_results"] = customer_result.get("results", [])
_print(output)
if __name__ == "__main__":
asyncio.run(main())

View File

@@ -0,0 +1,25 @@
#!/usr/bin/env python3
"""Create reconciliation candidates from local external records.
This script is safe to run from cron/systemd timer. It is idempotent and only
creates/updates reconciliation_items for information already known locally, such
as Jasmin commercial documents without opportunity_id.
"""
from __future__ import annotations
from app.db import init_db
from app.reconciliation_service import sync_local_documents_without_opportunity
def main() -> None:
init_db()
result = sync_local_documents_without_opportunity(limit=500)
print(
"Reconciliation sync finished: "
f"seen={result.get('seen', 0)} "
f"created_or_updated={result.get('created_or_updated', 0)}"
)
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,95 @@
import json
import os
import sys
from pathlib import Path
import requests
BASE_URL = os.getenv("CLIENTFLOW_BASE_URL", "http://127.0.0.1:8000")
CASES = [
{
"conversation_id": "case-send-invoice",
"message": "Recebemos o equipamento. Agradecemos o envio da factura.",
},
{
"conversation_id": "case-payment",
"message": "Segue comprovativo de pagamento em anexo.",
},
{
"conversation_id": "case-info",
"message": "Bom dia, podem enviar mais informações sobre carregadores monofásicos?",
},
{
"conversation_id": "case-shipment",
"message": "Boa tarde, gostava de saber se já enviaram o carregador.",
},
]
def main() -> int:
out_dir = Path("resultados-action-core")
out_dir.mkdir(exist_ok=True)
rows = []
for case in CASES:
payload = {
"last_customer_message": case["message"],
"previous_context": "Teste Action Core.",
"source": "manual_test",
"conversation_id": case["conversation_id"],
"contact_id": "test-contact",
}
response = requests.post(
f"{BASE_URL}/analyze",
headers={"Content-Type": "application/json"},
json=payload,
timeout=90,
)
response.raise_for_status()
data = response.json()
action_decision = data.get("action_decision") or {}
action_result = data.get("action_result") or {}
row = {
"conversation_id": case["conversation_id"],
"action_code": action_decision.get("action_code"),
"route": action_result.get("route"),
"action": action_result.get("action"),
"safe_to_post": action_result.get("safe_to_post"),
"task_id": data.get("task_id"),
"needs_review": data.get("needs_review"),
}
rows.append(row)
print("=" * 80)
print(case["conversation_id"])
print("action_code:", row["action_code"])
print("route:", row["route"])
print("action:", row["action"])
print("safe_to_post:", row["safe_to_post"])
print("task_id:", row["task_id"])
(out_dir / f"{case['conversation_id']}.json").write_text(
json.dumps(data, ensure_ascii=False, indent=2),
encoding="utf-8",
)
(out_dir / "summary.json").write_text(
json.dumps(rows, ensure_ascii=False, indent=2),
encoding="utf-8",
)
print("\nResumo:", out_dir / "summary.json")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,51 @@
#!/usr/bin/env python3
"""Teste não destrutivo da integração Jasmin.
Não cria clientes/documentos. Valida OAuth, versão, OData de clientes,
produtos, orçamentos e faturas.
"""
from __future__ import annotations
import asyncio
import json
import os
import sys
from pathlib import Path
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
__test__ = False
def show(title: str, value, limit: int = 2500) -> None:
print(f"\n=== {title} ===")
try:
text = json.dumps(value, indent=2, ensure_ascii=False, default=str)
except Exception:
text = str(value)
print(text[:limit])
async def main() -> int:
from app.jasmin_client import JasminClient, JasminError
client = JasminClient()
try:
token = await client.get_token()
print(f"OAuth OK: token_length={len(token)}")
show("Versões", await client.get_versions())
show("Clientes OData", await client.list_customers_odata(top=5))
show("Produtos OData", await client.list_sales_items(top=10))
show("Últimos orçamentos", await client.list_quotations(top=5))
show("Últimas faturas", await client.list_invoices(top=5))
except JasminError as exc:
print(f"ERRO Jasmin: {exc}")
return 1
return 0
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))

View File

@@ -0,0 +1,15 @@
#!/usr/bin/env python3
"""Smoke placeholder for the clean ClientFlow operational architecture.
Operational progression is represented by business_events and operation_links,
not by extra triage action_codes.
"""
def main() -> int:
print("Use API/UI smoke tests against current action codes and operation_links.")
return 0
if __name__ == "__main__":
raise SystemExit(main())

View File

@@ -0,0 +1,101 @@
#!/usr/bin/env python3
"""Smoke test Packlink PRO.
Uso:
export PACKLINK_API_KEY=...
export PACKLINK_BASE_URL=https://api.packlink.com/v1
python scripts/test_packlink_connection.py
"""
from __future__ import annotations
import asyncio
import json
import os
import sys
from pathlib import Path
PROJECT_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(PROJECT_ROOT))
os.chdir(PROJECT_ROOT)
# Permite que o ficheiro seja importado por pytest sem exigir .env real.
os.environ.setdefault("OPENROUTER_API_KEY", "dummy")
os.environ.setdefault("DATABASE_URL", "postgresql+psycopg://clientflow:password@127.0.0.1:5432/clientflow")
from app.packlink_client import PacklinkClient
def normalize_packlink_zip(country: str, zip_code: str, *, for_quote: bool = True) -> str:
country = str(country or "").upper().strip()
zip_code = str(zip_code or "").strip()
if country == "PT" and for_quote:
import re
match = re.search(r"\d{4}", zip_code)
if match:
return match.group(0)
return zip_code
def default_package() -> dict:
return {
"height": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_HEIGHT", "10"))),
"width": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_WIDTH", "20"))),
"length": int(float(os.getenv("PACKLINK_DEFAULT_PACKAGE_LENGTH", "30"))),
"weight": float(os.getenv("PACKLINK_DEFAULT_PACKAGE_WEIGHT", "2")),
}
def dump(title: str, value) -> None:
print(f"\n=== {title} ===")
print(json.dumps(value, ensure_ascii=False, indent=2, default=str)[:5000])
async def main() -> int:
if not os.getenv("PACKLINK_API_KEY"):
print("ERRO: PACKLINK_API_KEY em falta")
return 2
client = PacklinkClient()
account = await client.get_client()
dump("Conta", account)
warehouses = await client.get_warehouses()
dump("Armazéns", warehouses[:3])
parcels = await client.get_parcels()
dump("Volumes", parcels[:3])
from_zip = normalize_packlink_zip("PT", os.getenv("PACKLINK_TEST_FROM_ZIP", "3650-219"), for_quote=True)
to_zip = normalize_packlink_zip("PT", os.getenv("PACKLINK_TEST_TO_ZIP", "4000-001"), for_quote=True)
services = await client.quote_services(
from_country="PT",
from_zip=from_zip,
to_country="PT",
to_zip=to_zip,
source=os.getenv("PACKLINK_SOURCE", "PRO"),
packages=[default_package()],
)
simple = [
{
"id": s.get("id"),
"carrier": s.get("carrier_name"),
"service": s.get("name"),
"price": (s.get("price") or {}).get("total_price") or s.get("base_price"),
"currency": s.get("currency") or (s.get("price") or {}).get("currency"),
"dropoff": s.get("dropoff"),
"parcelshop": s.get("delivery_to_parcelshop"),
}
for s in services
]
dump("Serviços", simple)
service_id = os.getenv("PACKLINK_DEFAULT_SERVICE_ID", "20571")
details = await client.get_service_details(service_id)
dump(f"Detalhes serviço {service_id}", details)
return 0
if __name__ == "__main__":
raise SystemExit(asyncio.run(main()))

5
scripts/test_sample.sh Executable file
View File

@@ -0,0 +1,5 @@
#!/usr/bin/env bash
set -euo pipefail
curl -sS -X POST http://127.0.0.1:8000/analyze \
-H "Content-Type: application/json" \
-d @tests/sample_cases/sample_analyze.json | jq .

View File

@@ -0,0 +1,48 @@
#!/usr/bin/env python3
"""Lightweight v4.5 validation checks for a deployed ClientFlow backend."""
from __future__ import annotations
import sys
from sqlalchemy import text
from app.db import engine, init_db
REQUIRED_TABLES = ["communications", "timeline_events", "tasks", "integration_outbox", "opportunities"]
REQUIRED_TASK_COLUMNS = ["communication_id", "document_id", "shipment_id", "outbox_id", "priority", "assigned_to"]
def main() -> int:
init_db()
with engine.begin() as conn:
tables = {r[0] for r in conn.execute(text("""
SELECT table_name
FROM information_schema.tables
WHERE table_schema = 'public'
"""))}
missing_tables = [t for t in REQUIRED_TABLES if t not in tables]
cols = {r[0] for r in conn.execute(text("""
SELECT column_name
FROM information_schema.columns
WHERE table_schema = 'public' AND table_name = 'tasks'
"""))}
missing_cols = [c for c in REQUIRED_TASK_COLUMNS if c not in cols]
comm_count = conn.execute(text("SELECT COUNT(*) FROM communications")).scalar()
timeline_count = conn.execute(text("SELECT COUNT(*) FROM timeline_events")).scalar()
if missing_tables or missing_cols:
print("ClientFlow v4.5 validation failed")
print("Missing tables:", ", ".join(missing_tables) or "none")
print("Missing task columns:", ", ".join(missing_cols) or "none")
return 1
print("ClientFlow v4.5 validation OK")
print(f"communications={comm_count} timeline_events={timeline_count}")
return 0
if __name__ == "__main__":
sys.exit(main())