"""Métricas textuais, consistência de citações e checklist de conformidade. Regras ABNT/NBR usadas: - NBR 10520: citações no texto devem existir na referências e vice-versa; - NBR 6022: elementos obrigatórios do artigo (título, resumo, palavras-chave, seções, referências); - NBR 14724: elementos pré/textuais/pós-textuais. """ from __future__ import annotations import re from collections import Counter from dataclasses import dataclass, field from .bib import Reference from .texparse import Article, Section, count_words # --------------------------------------------------------------------------- # Estruturas de resultado # --------------------------------------------------------------------------- @dataclass class Check: code: str label: str status: str # ok | warn | fail detail: str = "" @property def icon(self) -> str: return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status] @dataclass class Analysis: article: Article refs: dict[str, Reference] checks: list[Check] = field(default_factory=list) metrics: dict = field(default_factory=dict) issues: list[str] = field(default_factory=list) # --------------------------------------------------------------------------- # Métricas de prosa # --------------------------------------------------------------------------- VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]" def syllables(word: str) -> int: """Aproximação de sílabas: 1 sílaba por grupo de vogais. Contar *grupos* (em vez de caracteres) evita inflar ditongos como "ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto). """ w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower()) if not w: return 1 n = len(re.findall(VOWELS + "+", w)) return max(1, n) def sentences(text: str) -> list[str]: """Divide em frases respeitando as quebras de linha do .tex. Nos artigos processados cada parágrafo/item de lista ocupa uma linha própria (convenção verificada nas fontes), então linha vira fronteira segura de frase; dentro da linha, pontuação [.!?] segue valendo. """ out: list[str] = [] for line in text.splitlines(): line = line.strip() if not line: continue parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line) for p in parts: p = p.strip() # ignora marcadores de lista ("-", "1.") e ruído trivial if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p): continue out.append(p) return out def flesch(text: str) -> float | None: """Índice de Flesch adaptado para o português (0-100; mais alto = mais legível). Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com sílabas estimadas por grupos de vogais (aproximação, sem acentuação completa). Resultado clamped em [0, 100]. """ words = count_words(text) sents = sentences(text) if not words or not sents: return None syl = sum(syllables(w) for w in words) mpf = len(words) / len(sents) # média de palavras por frase vmp = syl / len(words) # vocabulário médio (sílabas/palavra) score = 206.835 - 1.015 * mpf - 84.6 * vmp return round(min(100.0, max(0.0, score)), 1) def flesch_band(score: float | None) -> str: if score is None: return "n/d" if score >= 80: return "Muito fácil" if score >= 60: return "Fácil" if score >= 40: return "Médio" if score >= 20: return "Difícil" return "Muito difícil" # --------------------------------------------------------------------------- # Heurísticas de qualidade # --------------------------------------------------------------------------- REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I) REDUNDANT = re.compile( r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I ) LONG_SENTENCE = 60 # palavras LONG_PARAGRAPH = 160 # palavras def find_grammar_issues(text: str) -> list[str]: issues: list[str] = [] for m in REPEATED_WORD.finditer(text): # "se se" é legítimo em português pronominal ("avaliou-se se cada # estudo..."); filtrado para não gerar falso-positivo if m.group(1).lower() == "se": continue issues.append( f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — " f"...{context_around(text, m.start())}..." ) for m in REDUNDANT.finditer(text): issues.append( f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — " f"...{context_around(text, m.start())}..." ) return issues def context_around(text: str, pos: int, span: int = 60) -> str: start = max(0, pos - span) end = min(len(text), pos + span) snippet = text[start:end].replace("\n", " ") return ("…" + snippet) if start > 0 else snippet def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]: out = [] for s in sentences(text): n = len(count_words(s)) if n > limit: out.append((s[:120] + ("…" if len(s) > 120 else ""), n)) return sorted(out, key=lambda x: -x[1])[:10] def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]: # Convenção das fontes analisadas: um parágrafo por linha no .tex, # portanto cada linha (depois da limpeza) é candidata a parágrafo. out = [] for p in (line.strip() for line in text.splitlines()): if not p: continue n = len(count_words(p)) if n > limit: out.append((p[:120] + "…", n)) return sorted(out, key=lambda x: -x[1])[:10] def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]: words = [w.lower() for w in count_words(text)] stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os", "as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são", "and", "or"} # operadores de busca do estilo PRISMA (AND/OR) words = [w for w in words if w not in stop and len(w) > 2] grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)] return Counter(grams).most_common(top) def citation_stats(article: Article, refs: dict[str, Reference]) -> dict: cited = Counter(article.citations) defined = set(refs) used = set(cited) return { "total_citations": len(article.citations), "unique_cited": len(used), "defined_refs": len(defined), "orphan_citations": sorted(used - defined), "unused_refs": sorted(defined - used), "most_cited": [ (k, n, refs.get(k).label() if k in refs else "?") for k, n in cited.most_common(8) ], "year_distribution": dict( Counter( (refs[k].year or "?") if k in refs else "?" for k in used ) ), } # --------------------------------------------------------------------------- # Análise completa # --------------------------------------------------------------------------- RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"] EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura", "estado da arte", "revisão de literatura", "fundamentação", "referencial"] def analyze(article: Article, refs: dict[str, Reference]) -> Analysis: checks: list[Check] = [] issues: list[str] = [] # --- estrutura / pré-textuais ---------------------------------------- checks.append(Check( "ABNT-01", "Título presente", "ok" if article.title else "fail", article.title or "não encontrado no \\titulo{}", )) checks.append(Check( "ABNT-02", "Autores informados", "ok" if article.authors else "fail", ", ".join(article.authors) or "não encontrado", )) checks.append(Check( "ABNT-03", "Local e data", "ok" if article.local and article.data else "warn", f"{article.local}, {article.data}".strip(",") or "incompleto", )) abs_words = len(count_words(article.abstract)) checks.append(Check( "ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)", "ok" if 150 <= abs_words <= 500 else "warn", f"{abs_words} palavras" if abs_words else "não encontrado", )) checks.append(Check( "ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)", "ok" if 3 <= len(article.keywords) <= 8 else "warn", ", ".join(article.keywords) or "não encontradas", )) all_titles = [s.title.lower() for s in article.sections] for expected in RECOGNIZED_SECTIONS: hit = next((t for t in all_titles if expected in t), None) checks.append(Check( "ABNT-06", f"Seção '{expected.capitalize()}'", "ok" if hit else "warn", hit or "não identificada", )) lit_hit = next((t for t in all_titles if any(e in t for e in EXPECTED_LITERATURE)), None) checks.append(Check( "ABNT-07", "Seção de revisão de literatura", "ok" if lit_hit else "fail", lit_hit or "faltando", )) # citações / referências ------------------------------------------------ stats = citation_stats(article, refs) checks.append(Check( "ABNT-08", "Citações no corpo do texto", "ok" if stats["total_citations"] >= 5 else "warn", f"{stats['total_citations']} citações, " f"{stats['unique_cited']} referências citadas", )) checks.append(Check( "ABNT-09", "Citações sem entrada no .bib (NBR 10520)", "ok" if not stats["orphan_citations"] else "fail", ", ".join(stats["orphan_citations"]) or "nenhuma", )) checks.append(Check( "ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)", "ok" if not stats["unused_refs"] else "warn", ", ".join(stats["unused_refs"]) or "nenhuma", )) # figuras ------------------------------------------------------------ checks.append(Check( "ABNT-11", "Figuras/quadros identificados", "ok" if (article.figures or article.tables) else "warn", f"{len(article.figures)} figuras, {len(article.tables)} quadros", )) # prosa --------------------------------------------------------------- body_text = "\n\n".join(s.text for s in article.sections) flesch_score = flesch(body_text) top_phrases = frequent_phrases(body_text) repeated = find_grammar_issues(article.abstract + " " + body_text) l_sentences = long_sentences(body_text) l_paragraphs = long_paragraphs(body_text) for i in repeated: issues.append(i) for s, n in l_sentences: issues.append(f"Frase longa ({n} palavras): \"{s}\"") for p, n in l_paragraphs: issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"") metrics = { "palavras_corpo": len(count_words(body_text)), "palavras_totais": article.text_words, "sentencas": len(sentences(body_text)), "media_palavras_por_sentenca": round( len(count_words(body_text)) / max(1, len(sentences(body_text))), 1 ), "vocabulario_unico": len( set(w.lower() for w in count_words(body_text)) ), "flesch": flesch_score, "flesch_faixa": flesch_band(flesch_score), "secoes": [ { "titulo": s.title, "nivel": s.level, "palavras": s.words, "percentual": round( 100 * s.words / max(1, article.text_words), 1 ), } for s in article.sections ], "top_frases": top_phrases, "citacoes": stats, } summary = { status: sum(1 for c in checks if c.status == status) for status in ("ok", "warn", "fail") } metrics["resumo_checks"] = summary return Analysis( article=article, refs=refs, checks=checks, metrics=metrics, issues=issues, )