- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
347 lines
12 KiB
Python
347 lines
12 KiB
Python
"""Métricas textuais, consistência de citações e checklist de conformidade.
|
||
|
||
Regras ABNT/NBR usadas:
|
||
- NBR 10520: citações no texto devem existir na referências e vice-versa;
|
||
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
|
||
palavras-chave, seções, referências);
|
||
- NBR 14724: elementos pré/textuais/pós-textuais.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from collections import Counter
|
||
from dataclasses import dataclass, field
|
||
|
||
from .bib import Reference
|
||
from .texparse import Article, Section, count_words
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Estruturas de resultado
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
@dataclass
|
||
class Check:
|
||
code: str
|
||
label: str
|
||
status: str # ok | warn | fail
|
||
detail: str = ""
|
||
|
||
@property
|
||
def icon(self) -> str:
|
||
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
|
||
|
||
|
||
@dataclass
|
||
class Analysis:
|
||
article: Article
|
||
refs: dict[str, Reference]
|
||
checks: list[Check] = field(default_factory=list)
|
||
metrics: dict = field(default_factory=dict)
|
||
issues: list[str] = field(default_factory=list)
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Métricas de prosa
|
||
# ---------------------------------------------------------------------------
|
||
|
||
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
|
||
|
||
|
||
def syllables(word: str) -> int:
|
||
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
|
||
|
||
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
|
||
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
|
||
"""
|
||
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
|
||
if not w:
|
||
return 1
|
||
n = len(re.findall(VOWELS + "+", w))
|
||
return max(1, n)
|
||
|
||
|
||
def sentences(text: str) -> list[str]:
|
||
"""Divide em frases respeitando as quebras de linha do .tex.
|
||
|
||
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
|
||
própria (convenção verificada nas fontes), então linha vira fronteira
|
||
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
|
||
"""
|
||
out: list[str] = []
|
||
for line in text.splitlines():
|
||
line = line.strip()
|
||
if not line:
|
||
continue
|
||
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
|
||
for p in parts:
|
||
p = p.strip()
|
||
# ignora marcadores de lista ("-", "1.") e ruído trivial
|
||
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
|
||
continue
|
||
out.append(p)
|
||
return out
|
||
|
||
|
||
def flesch(text: str) -> float | None:
|
||
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
|
||
|
||
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
|
||
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
|
||
completa). Resultado clamped em [0, 100].
|
||
"""
|
||
words = count_words(text)
|
||
sents = sentences(text)
|
||
if not words or not sents:
|
||
return None
|
||
syl = sum(syllables(w) for w in words)
|
||
mpf = len(words) / len(sents) # média de palavras por frase
|
||
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
|
||
score = 206.835 - 1.015 * mpf - 84.6 * vmp
|
||
return round(min(100.0, max(0.0, score)), 1)
|
||
|
||
|
||
def flesch_band(score: float | None) -> str:
|
||
if score is None:
|
||
return "n/d"
|
||
if score >= 80:
|
||
return "Muito fácil"
|
||
if score >= 60:
|
||
return "Fácil"
|
||
if score >= 40:
|
||
return "Médio"
|
||
if score >= 20:
|
||
return "Difícil"
|
||
return "Muito difícil"
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Heurísticas de qualidade
|
||
# ---------------------------------------------------------------------------
|
||
|
||
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
|
||
REDUNDANT = re.compile(
|
||
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
|
||
)
|
||
LONG_SENTENCE = 60 # palavras
|
||
LONG_PARAGRAPH = 160 # palavras
|
||
|
||
|
||
def find_grammar_issues(text: str) -> list[str]:
|
||
issues: list[str] = []
|
||
for m in REPEATED_WORD.finditer(text):
|
||
# "se se" é legítimo em português pronominal ("avaliou-se se cada
|
||
# estudo..."); filtrado para não gerar falso-positivo
|
||
if m.group(1).lower() == "se":
|
||
continue
|
||
issues.append(
|
||
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
|
||
f"...{context_around(text, m.start())}..."
|
||
)
|
||
for m in REDUNDANT.finditer(text):
|
||
issues.append(
|
||
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
|
||
f"...{context_around(text, m.start())}..."
|
||
)
|
||
return issues
|
||
|
||
|
||
def context_around(text: str, pos: int, span: int = 60) -> str:
|
||
start = max(0, pos - span)
|
||
end = min(len(text), pos + span)
|
||
snippet = text[start:end].replace("\n", " ")
|
||
return ("…" + snippet) if start > 0 else snippet
|
||
|
||
|
||
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
|
||
out = []
|
||
for s in sentences(text):
|
||
n = len(count_words(s))
|
||
if n > limit:
|
||
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
|
||
return sorted(out, key=lambda x: -x[1])[:10]
|
||
|
||
|
||
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
|
||
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
|
||
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
|
||
out = []
|
||
for p in (line.strip() for line in text.splitlines()):
|
||
if not p:
|
||
continue
|
||
n = len(count_words(p))
|
||
if n > limit:
|
||
out.append((p[:120] + "…", n))
|
||
return sorted(out, key=lambda x: -x[1])[:10]
|
||
|
||
|
||
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
|
||
words = [w.lower() for w in count_words(text)]
|
||
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
|
||
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
|
||
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
|
||
words = [w for w in words if w not in stop and len(w) > 2]
|
||
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
|
||
return Counter(grams).most_common(top)
|
||
|
||
|
||
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
|
||
cited = Counter(article.citations)
|
||
defined = set(refs)
|
||
used = set(cited)
|
||
return {
|
||
"total_citations": len(article.citations),
|
||
"unique_cited": len(used),
|
||
"defined_refs": len(defined),
|
||
"orphan_citations": sorted(used - defined),
|
||
"unused_refs": sorted(defined - used),
|
||
"most_cited": [
|
||
(k, n, refs.get(k).label() if k in refs else "?")
|
||
for k, n in cited.most_common(8)
|
||
],
|
||
"year_distribution": dict(
|
||
Counter(
|
||
(refs[k].year or "?") if k in refs else "?" for k in used
|
||
)
|
||
),
|
||
}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Análise completa
|
||
# ---------------------------------------------------------------------------
|
||
|
||
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
|
||
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
|
||
"estado da arte", "revisão de literatura",
|
||
"fundamentação", "referencial"]
|
||
|
||
|
||
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
|
||
checks: list[Check] = []
|
||
issues: list[str] = []
|
||
|
||
# --- estrutura / pré-textuais ----------------------------------------
|
||
checks.append(Check(
|
||
"ABNT-01", "Título presente",
|
||
"ok" if article.title else "fail",
|
||
article.title or "não encontrado no \\titulo{}",
|
||
))
|
||
checks.append(Check(
|
||
"ABNT-02", "Autores informados",
|
||
"ok" if article.authors else "fail",
|
||
", ".join(article.authors) or "não encontrado",
|
||
))
|
||
checks.append(Check(
|
||
"ABNT-03", "Local e data",
|
||
"ok" if article.local and article.data else "warn",
|
||
f"{article.local}, {article.data}".strip(",") or "incompleto",
|
||
))
|
||
|
||
abs_words = len(count_words(article.abstract))
|
||
checks.append(Check(
|
||
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
|
||
"ok" if 150 <= abs_words <= 500 else "warn",
|
||
f"{abs_words} palavras" if abs_words else "não encontrado",
|
||
))
|
||
checks.append(Check(
|
||
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
|
||
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
|
||
", ".join(article.keywords) or "não encontradas",
|
||
))
|
||
|
||
all_titles = [s.title.lower() for s in article.sections]
|
||
for expected in RECOGNIZED_SECTIONS:
|
||
hit = next((t for t in all_titles if expected in t), None)
|
||
checks.append(Check(
|
||
"ABNT-06", f"Seção '{expected.capitalize()}'",
|
||
"ok" if hit else "warn",
|
||
hit or "não identificada",
|
||
))
|
||
lit_hit = next((t for t in all_titles
|
||
if any(e in t for e in EXPECTED_LITERATURE)), None)
|
||
checks.append(Check(
|
||
"ABNT-07", "Seção de revisão de literatura",
|
||
"ok" if lit_hit else "fail",
|
||
lit_hit or "faltando",
|
||
))
|
||
|
||
# citações / referências ------------------------------------------------
|
||
stats = citation_stats(article, refs)
|
||
checks.append(Check(
|
||
"ABNT-08", "Citações no corpo do texto",
|
||
"ok" if stats["total_citations"] >= 5 else "warn",
|
||
f"{stats['total_citations']} citações, "
|
||
f"{stats['unique_cited']} referências citadas",
|
||
))
|
||
checks.append(Check(
|
||
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
|
||
"ok" if not stats["orphan_citations"] else "fail",
|
||
", ".join(stats["orphan_citations"]) or "nenhuma",
|
||
))
|
||
checks.append(Check(
|
||
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
|
||
"ok" if not stats["unused_refs"] else "warn",
|
||
", ".join(stats["unused_refs"]) or "nenhuma",
|
||
))
|
||
|
||
# figuras ------------------------------------------------------------
|
||
checks.append(Check(
|
||
"ABNT-11", "Figuras/quadros identificados",
|
||
"ok" if (article.figures or article.tables) else "warn",
|
||
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
|
||
))
|
||
|
||
# prosa ---------------------------------------------------------------
|
||
body_text = "\n\n".join(s.text for s in article.sections)
|
||
flesch_score = flesch(body_text)
|
||
top_phrases = frequent_phrases(body_text)
|
||
repeated = find_grammar_issues(article.abstract + " " + body_text)
|
||
l_sentences = long_sentences(body_text)
|
||
l_paragraphs = long_paragraphs(body_text)
|
||
for i in repeated:
|
||
issues.append(i)
|
||
for s, n in l_sentences:
|
||
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
|
||
for p, n in l_paragraphs:
|
||
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
|
||
|
||
metrics = {
|
||
"palavras_corpo": len(count_words(body_text)),
|
||
"palavras_totais": article.text_words,
|
||
"sentencas": len(sentences(body_text)),
|
||
"media_palavras_por_sentenca": round(
|
||
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
|
||
),
|
||
"vocabulario_unico": len(
|
||
set(w.lower() for w in count_words(body_text))
|
||
),
|
||
"flesch": flesch_score,
|
||
"flesch_faixa": flesch_band(flesch_score),
|
||
"secoes": [
|
||
{
|
||
"titulo": s.title,
|
||
"nivel": s.level,
|
||
"palavras": s.words,
|
||
"percentual": round(
|
||
100 * s.words / max(1, article.text_words), 1
|
||
),
|
||
}
|
||
for s in article.sections
|
||
],
|
||
"top_frases": top_phrases,
|
||
"citacoes": stats,
|
||
}
|
||
|
||
summary = {
|
||
status: sum(1 for c in checks if c.status == status)
|
||
for status in ("ok", "warn", "fail")
|
||
}
|
||
metrics["resumo_checks"] = summary
|
||
|
||
return Analysis(
|
||
article=article, refs=refs, checks=checks, metrics=metrics,
|
||
issues=issues,
|
||
)
|